diff --git a/.cargo/config.toml b/.cargo/config.toml index 92a480263..30f238b3a 100644 --- a/.cargo/config.toml +++ b/.cargo/config.toml @@ -6,5 +6,6 @@ docall = ["doc", "--release", "--workspace", "--no-deps"] [build] rustdocflags = ["-D", "warnings"] -[target.'cfg(all())'] +# Not for the RISC-V guests (`guests/`), which inherit this file: no RISC-V target accepts the host's CPU. +[target.'cfg(not(target_arch = "riscv64"))'] rustflags = ["-C", "target-cpu=native"] diff --git a/.github/workflows/doc.yml b/.github/workflows/doc.yml index 300b36174..1da1f73ed 100644 --- a/.github/workflows/doc.yml +++ b/.github/workflows/doc.yml @@ -5,15 +5,11 @@ on: branches: [ "main" ] paths: - 'doc/leanvm/**' - - 'doc/xmss/**' - - 'doc/sphincs/**' - 'doc/images/**' - '.github/workflows/doc.yml' pull_request: paths: - 'doc/leanvm/**' - - 'doc/xmss/**' - - 'doc/sphincs/**' - 'doc/images/**' - '.github/workflows/doc.yml' workflow_dispatch: @@ -38,22 +34,6 @@ jobs: - name: Fail on an undefined reference or citation run: | ! grep -qE 'Reference .* undefined|Citation .* undefined|multiply defined' doc/leanvm/.build/main.log - - name: Compile XMSS specification - uses: xu-cheng/latex-action@v3 - with: - working_directory: doc/xmss - root_file: main.tex - - name: Fail on an undefined XMSS reference or citation - run: | - ! grep -qE 'Reference .* undefined|Citation .* undefined|multiply defined' doc/xmss/.build/main.log - - name: Compile SPHINCS specification - uses: xu-cheng/latex-action@v3 - with: - working_directory: doc/sphincs - root_file: main.tex - - name: Fail on an undefined SPHINCS reference or citation - run: | - ! grep -qE 'Reference .* undefined|Citation .* undefined|multiply defined' doc/sphincs/.build/main.log build-pdf: if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main' @@ -67,21 +47,9 @@ jobs: with: working_directory: doc/leanvm root_file: main.tex - - name: Compile XMSS specification - uses: xu-cheng/latex-action@v3 - with: - working_directory: doc/xmss - root_file: main.tex - - name: Compile SPHINCS specification - uses: xu-cheng/latex-action@v3 - with: - working_directory: doc/sphincs - root_file: main.tex - name: Name the artifacts run: | cp doc/leanvm/.build/main.pdf leanVM.pdf - cp doc/xmss/.build/main.pdf XMSS.pdf - cp doc/sphincs/.build/main.pdf SPHINCS.pdf - name: Publish PDFs as release assets uses: softprops/action-gh-release@v2 with: @@ -91,10 +59,6 @@ jobs: Auto-built on every push to `main`. `leanVM.pdf` contains the leanVM specification. - `XMSS.pdf` contains the XMSS specification. - `SPHINCS.pdf` contains the SPHINCS specification. make_latest: false files: | leanVM.pdf - XMSS.pdf - SPHINCS.pdf diff --git a/.github/workflows/lean.yml b/.github/workflows/lean.yml deleted file mode 100644 index a7a1e8556..000000000 --- a/.github/workflows/lean.yml +++ /dev/null @@ -1,73 +0,0 @@ -name: Lean - -on: - push: - branches: [ "main" ] - pull_request: - workflow_dispatch: - -permissions: - contents: read - -concurrency: - group: lean-${{ github.ref }} - cancel-in-progress: true - -jobs: - xmss-formalization: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - # Cheap first pass: the axiom guard in `XmssSecurity.lean` catches a `sorry` - # or a `native_decide` reaching the root theorem, this catches one parked - # anywhere in the project. - - name: Forbid proof escapes - run: | - ! grep -rnE '\b(sorry|sorryAx|admit|native_decide|unsafe|implemented_by)\b|#exit' \ - --include='*.lean' --exclude-dir=.lake formal/xmss | grep -vE ':[0-9]+: *(--|/-)' - # Installs the toolchain from `formal/xmss/lean-toolchain`, fetches the - # mathlib cache, and runs `lake build` on the default target, which - # elaborates the root module and with it the `#guard_msgs` check. The - # checked-in manifest is used as is: no `lake update`. - - uses: leanprover/lean-action@v1 - with: - lake-package-directory: formal/xmss - # The root module guards its own footprint with `#guard_msgs`, which an - # edit to the expected message would silence. This asks again from - # outside, against the list written here. - - name: Check the axiom footprint - working-directory: formal/xmss - run: | - printf 'import XmssSecurity\n#print axioms XmssSecurity.xmss_has_127_bits_of_classical_security\n' \ - > "$RUNNER_TEMP/axioms.lean" - lake env lean "$RUNNER_TEMP/axioms.lean" | tee "$RUNNER_TEMP/axioms.txt" - grep -qF "'XmssSecurity.xmss_has_127_bits_of_classical_security' depends on axioms: [propext, Classical.choice, Quot.sound]" \ - "$RUNNER_TEMP/axioms.txt" - - uses: actions/upload-artifact@v4 - with: - name: xmss-axioms - path: ${{ runner.temp }}/axioms.txt - - sphincs-formalization: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - name: Forbid proof escapes - run: | - ! grep -rnE '\b(sorry|sorryAx|admit|native_decide|unsafe|implemented_by)\b|#exit' \ - --include='*.lean' --exclude-dir=.lake formal/sphincs | grep -vE ':[0-9]+: *(--|/-)' - - uses: leanprover/lean-action@v1 - with: - lake-package-directory: formal/sphincs - - name: Check the axiom footprint - working-directory: formal/sphincs - run: | - printf 'import SphincsSecurity\n#print axioms SphincsSecurity.sphincs_has_127_bits_of_classical_security\n' \ - > "$RUNNER_TEMP/axioms.lean" - lake env lean "$RUNNER_TEMP/axioms.lean" | tee "$RUNNER_TEMP/axioms.txt" - grep -qF "'SphincsSecurity.sphincs_has_127_bits_of_classical_security' depends on axioms: [propext, Classical.choice, Quot.sound]" \ - "$RUNNER_TEMP/axioms.txt" - - uses: actions/upload-artifact@v4 - with: - name: sphincs-axioms - path: ${{ runner.temp }}/axioms.txt diff --git a/.github/workflows/riscv.yml b/.github/workflows/riscv.yml new file mode 100644 index 000000000..9c6a03a96 --- /dev/null +++ b/.github/workflows/riscv.yml @@ -0,0 +1,46 @@ +name: RISC-V + +# ACT4 (conformance/act4/): the tests are generated from their pinned sources, not checked +# in. Generation is deterministic, so its output is cached under a hash of every input +# and the Docker image is only built when one of them changes. + +on: + push: + # Pushes to the branches PRs target save the generated tests where those PRs can + # restore them: a PR's own cache is visible to it alone. + branches: [ "main", "riscv-exploration" ] + pull_request: + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: riscv-${{ github.ref }} + cancel-in-progress: true + +env: + CARGO_TERM_COLOR: always + RUST_BACKTRACE: 1 + # As in rust.yml: the SIMD backend under test must not depend on the runner. + RUSTFLAGS: -C target-cpu=haswell + +jobs: + act4: + name: ACT4 + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + - name: Restore the generated tests + id: elf + uses: actions/cache@v4 + with: + path: conformance/act4/elf + key: act4-elf-${{ hashFiles('conformance/act4/Dockerfile', 'conformance/act4/*.sh', 'conformance/act4/*.yaml', 'conformance/act4/*.json', 'conformance/act4/*.ld', 'conformance/act4/*.h') }} + - name: Generate the tests from the pinned sources + if: steps.elf.outputs.cache-hit != 'true' + run: conformance/act4/generate.sh + - uses: dtolnay/rust-toolchain@stable + - uses: Swatinem/rust-cache@v2 + - name: Run the tests + run: cargo test --release -p lean_vm --test verifiers -- --ignored act4 diff --git a/.gitignore b/.gitignore index 60e0034dd..69de9f663 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,5 @@ .build target __pycache__/ +formal/ +conformance/act4/elf/ diff --git a/AGENTS.md b/AGENTS.md index 4f0e540a5..de68894f2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -2,19 +2,10 @@ ## What this is -A minimal virtual machine and recursive SNARK for signature aggregation and blob encoding. Proofs are not zero knowledge. +A RISC-V (rv64im) virtual machine and the SNARK that proves its execution. Proofs are not zero knowledge. Every rv64im instruction is proven, plus one custom instruction, the BLAKE2s compression (`blake2s rs1, rs2`). A program is a guest's ELF file (`guests/`, Rust built for `riscv64im-unknown-none-elf`, loaded by `rv::Guest::from_elf`) or a text assembled by hand with `lean_vm::rv::asm`; a run is proven on a public input (four words, RAM's first) and an advice (a region of memory the prover fills), and its statement is the program's digest, the input and the output (`a0..a3` at `exit`). One run is one proof: a run whose witness exceeds one commitment (`pcs::MAX_MU`) is refused up front with `Trap::TooLong`, continuations being unimplemented. -- `doc/leanvm/` is the LaTeX project describing the machine ISA and the snark that proves it. Its root is `doc/leanvm/main.tex`; build it with `cd doc/leanvm && latexmk -pdf main.tex`, which writes to the gitignored `doc/leanvm/.build/`. Sections live in `doc/leanvm/body/`, numbered `01`..`10` plus the lettered annexes `a` (ring switching), `b` (the PCS), `c` (Flock), and `d` (novel basis and additive NTT), and every symbol is defined once in `doc/leanvm/preamble/macros.tex`. If latexmk fails oddly (a bibtex error, or a missing `main.log`) right after inputs are renamed or `refs.bib` is edited, remove `doc/leanvm/.build` and rerun; it has not reproduced on unchanged inputs. **Drafting one section:** each section file carries a `% !TeX root` comment pointing at its generated driver in `doc/leanvm/drafts/`, so the LaTeX build key (`F5`, or the extension's `cmd+alt+b`) compiles only that section, numbered as in the full document and with cross-references and citations resolved against `.build/main.aux`; in `main.tex` the same key builds everything. Run `doc/leanvm/make-drafts.sh` after adding, renaming or renumbering a section. -- `doc/xmss/` is the standalone XMSS specification; `crates/xmss` implements its hash inputs and signature verification. -- `doc/sphincs/` is the standalone specification of the concrete SPHINCS+ instance used where statelessness matters; its root is `doc/sphincs/main.tex`, built the same way as `doc/xmss`, and implemented by `crates/sphincs`. It uses the same BLAKE2s primitive and target-sum encoding shape as XMSS, with its own tweak layout, target sum, and signing search. -- `formal/xmss/` and `formal/sphincs/` are Lean 4 proofs (over VCVio) of the ideal schemes' classical random-oracle security, `xmss_has_127_bits_of_classical_security` and `sphincs_has_127_bits_of_classical_security`; `formal/sphincs/` also proves correctness and completeness, `sphincs_is_correct` and `sphincs_is_complete`, stated in `SphincsSecurity/Completeness.lean`. Each project's `Scheme.lean`, under `XmssSecurity/` or `SphincsSecurity/`, contains the concrete parameters, the byte layout of every hash input, and the three algorithms; `Statement.lean` imports it and defines the SUF-CMA game, hash-query budget, and security claim. `lake exe cache get` once, then `lake build`. -- The one hash function is BLAKE2s, in `primitives::hash`: scalar, streaming, and a lane-transposed batched form for the PCS Merkle tree. The VM proves one compression per opcode, and BLAKE2s takes the byte counter and final-block flag as ordinary compression inputs, so repeated opcodes hash arbitrary byte strings by carrying the chaining value and setting the counter and final flag for each block. -- `crates/lean_compiler/zkDSL.md` documents the (pythonic) zkDSL (that compiles to the ISA that our VM runs, and that our snark proves). - -Primary uses: - -- Aggregate XMSS claims grouped by epoch and message, SPHINCS claims carrying individual messages, and LeanDA blob encoding claims. -- Recursively aggregate child proofs, proving that every published signature claim and DA root is supported by a raw input or a verified child. +- `doc/leanvm/` is the LaTeX project describing the machine ISA and the snark that proves it. Its root is `doc/leanvm/main.tex`; build it with `cd doc/leanvm && latexmk -pdf main.tex`, which writes to the gitignored `doc/leanvm/.build/`. Sections live in `doc/leanvm/body/`, numbered `01`..`08` plus the lettered annexes `a` (ring switching), `b` (the PCS), `c` (Flock), and `d` (novel basis and additive NTT), and every symbol is defined once in `doc/leanvm/preamble/macros.tex`. If latexmk fails oddly (a bibtex error, or a missing `main.log`) right after inputs are renamed or `refs.bib` is edited, remove `doc/leanvm/.build` and rerun; it has not reproduced on unchanged inputs. **Drafting one section:** each section file carries a `% !TeX root` comment pointing at its generated driver in `doc/leanvm/drafts/`, so the LaTeX build key (`F5`, or the extension's `cmd+alt+b`) compiles only that section, numbered as in the full document and with cross-references and citations resolved against `.build/main.aux`; in `main.tex` the same key builds everything. Run `doc/leanvm/make-drafts.sh` after adding, renaming or renumbering a section. +- The one hash function is BLAKE2s, in `primitives::hash`: scalar, streaming, and a lane-transposed batched form for the PCS Merkle tree. The VM's compression instruction is the generic gate-list circuit `rv::circuits::blake2s`, proven like every other class; `flock::hash` is the hand-optimized circuit of the same function with its own witness kernels, kept as flock's throughput benchmark and used by nothing else. Moving the precompile onto it is the known speed-up if hashing ever dominates a workload. ## Layout @@ -27,21 +18,20 @@ Dependency order, leaves first: | `primitives` | field kernels (NEON/AVX), bit transposes, multilinear helpers, streaming stores, `bench` | | `fiat_shamir` | VM-native `FiatShamirState` + prover/verifier transcript | | `pcs` | additive NTT, Merkle, ring switch, stacked WHIR | -| `flock` | batched R1CS over GF(2) for BLAKE2s: zerocheck + lincheck | -| `lean_vm` | arithmetization: tables, bus, constraints, `cpu::prove`/`verify` | -| `lean_compiler` | zkDSL (Python subset) → ISA | -| `xmss` | XMSS over BLAKE2s; an independent leaf, consumed only by `rec_aggregation` | -| `sphincs` | the stateless SPHINCS+ instance of `doc/sphincs`; an independent leaf, consumed only by `rec_aggregation` | -| `lean_da` | additive Reed-Solomon blob encoding, commitments, and membership vectors | -| `rec_aggregation` | recursive signature and DA aggregation: the guest, public entry points, and benchmarks | +| `flock` | batched R1CS over GF(2): zerocheck + lincheck; gate-list circuits over word ports (`circuit`), the u64 adder and multiplier in them (`arith`), the BLAKE2s circuit (`hash`) | +| `lean_vm` | the RISC-V machine (`rv`: decoder, class semantics and circuits, interpreter, assembler, ELF loader) and its arithmetization: tables, bus, constraints, `cpu::prove`/`verify` | + +`src/lib.rs` is the public API and the only thing a user imports: every crate above is `publish = false`, so a new user-facing item is a re-export there. `src/main.rs` is the CLI (`fibonacci`, the benchmark, and `guest --input a,b,c,d --advice ...`), `tests/api.rs` the end-to-end use of the API. -`src/lib.rs` is the public API and the only thing a user imports: every crate above is `publish = false`, so a new user-facing item is a re-export there. `src/main.rs` is the benchmark CLI, `tests/api.rs` the end-to-end use of the API; guests are zkDSL under `crates/rec_aggregation/guests/`. +`guests/` is a separate cargo workspace (its own toolchain file, nightly with `rust-src`, and `.cargo/config.toml` targeting `riscv64im-unknown-none-elf` with `-Zbuild-std=core`): the runtime crate `rt` (`_start`, `input()`, `advice()`, `output()`, a `Blake2s` hasher over the custom instruction through `.insn r`, a panic that is `unimp`), the guests, `link.ld` fixing the memory map, and `build.sh`, which refreshes the checked-in ELF fixtures in `guests/elf/` that `lean_vm/tests/verifiers/guests.rs` loads. **Rerun `build.sh` after touching a guest or the runtime**: nothing rebuilds the fixtures, so CI would keep testing the old ELF and say nothing. The nightly channel floats, so the fixtures are not reproducible byte for byte across toolchain updates, which is why they are not diffed in CI. The root `.cargo/config.toml` scopes `target-cpu=native` to `cfg(not(target_arch = "riscv64"))` because the guests inherit it. Atomics are why the target is `im` and not `imac`: the builtin target has no compare-and-swap, so a dependency needing one does not compile. + +`conformance/act4/` holds the official RISC-V architectural tests ([ACT4](https://github.com/riscv/riscv-arch-test), riscv-arch-test 4.1.0 at `6e8a4512`), their `I` and `M` suites, 64 tests, for an `rv64im` profile with no misaligned access. ACT4 runs each test on the Sail reference model and builds it again with Sail's results inside, so that the ELF file checks itself. The configuration is `test_config.yaml`, `leanvm-rv64im.yaml` (for the unified database, UDB, which defines `MXLEN` through `Sm` and so needs `Zicsr` claimed too: `rvmodel_macros.h` leaves `STANDARD_SM_SUPPORTED` undefined, so no test touches a CSR), `sail.json` (Sail's regions: leanVM's text and RAM), `link.ld` (the code in the text, everything else in RAM past the input words, RAM the smallest power of two holding the image) and `rvmodel_macros.h` (a test ends in `exit` with the output zero on a pass, `a0 = 1` and `a1` the caller when the framework halts on a failure). A failing check's handler reads the check back from the text, which leanVM cannot read, so it traps there, and `lean_vm/tests/verifiers/act4.rs` reports the check's description, the value computed and Sail's. `generate.sh` builds the Docker image of `Dockerfile` (Ubuntu 24.04 by digest, riscv-gnu-toolchain 2025.08.08 with GCC 15.1, Sail 0.13.1 and mise 2026.9.17 by checksum, the Ruby and Python tools locked by riscv-arch-test itself) and generates `elf/I` and `elf/M` byte for byte: it strips the symbol naming the compiler's temporary object file, which is random, and clears the compressed-instructions flag that ACT4's alignment directives set though nothing is compressed. **The ELF files are not checked in** (`elf/` is ignored), so `act4.rs`'s two tests are `#[ignore]`d: run `conformance/act4/generate.sh` once (Docker), then `cargo test --release -p lean_vm --test verifiers -- --ignored act4`. CI (`.github/workflows/riscv.yml`) runs them on every PR, caching the generated files under a hash of every file in `conformance/act4/`, so the image is only built when one of them changes. Left out by design: `Misalign` (a misaligned access traps, and the configuration says so), `Zicsr` (no CSRs), `Zifencei` (no self-modifying code), and `Zmmul`, whose tests are M's multiplication tests again. `act4.rs` runs every test on the interpreter, proves each and checks the proof with the Rust verifier, and with the Python verifier the first test to reach each table. ## Building / Testing / Formatting - `.cargo/config.toml` pins `-C target-cpu=native` and `-D warnings` for rustdoc -- always run in `--release` mode any test or benchmark touching the VM (the zkDSL compiler stack-overflows in `debug` mode) -- **One test binary per crate, not one per file:** new `lean_compiler` integration tests go in `tests/suite/main.rs`, one linked executable instead of seventeen. Exception: a test opening an arena phase (`lean_vm::init_prover`) needs its own binary. Phases are process-global, so two in one process reclaim each other's `ArenaVec`s and the symptom is a proof that stops verifying, never a crash (`rec_aggregation/tests/arena_prove.rs`). +- always run in `--release` mode any test or benchmark touching the VM +- **One test binary per crate, not one per file** (`lean_vm/tests/verifiers/main.rs`). Exception: a test opening an arena phase (`lean_vm::init_prover`) needs its own binary. Phases are process-global, so two in one process reclaim each other's `ArenaVec`s and the symptom is a proof that stops verifying, never a crash (`tests/api.rs` is that binary, and `tests/no_arena.rs` the one that must never enable the arena). An x86-only arm never compiles on an Apple dev machine, so a typo in one ships. Type-check the other target before pushing anything `cfg`-gated: @@ -60,17 +50,54 @@ cargo fmt --all # max_width = 120 ruff format --line-length 150 python-verifier/verifier.py # and `ruff check` it ``` -Heavy benches and measurement harnesses are `#[ignore]`d; run by name with `-- --ignored --nocapture`: `hash_batch_prove_verify`, `pcs_throughput`, `aggregate_three_levels`, `aggregate_statement_binds`, `aggregate_hints_bind`, `aggregate_rejects_a_bad_signature`, `print_whir_query_counts`, `encoding_grinding_bits`. +Heavy benches and measurement harnesses are `#[ignore]`d; run by name with `-- --ignored --nocapture`: `hash_batch_prove_verify`, `add_wrapping_prove_verify`, `mul_wrapping_prove_verify`, `mul_widening_prove_verify`, `pcs_throughput`, `multithreaded_throughput`, `print_whir_query_counts`, `print_whir_query_table`. ## Benchmarking The benchmarks we care about: -- `cargo run --release -- aggregate --xmss 900 --log-inv-rate 1 --repeat 3` -- `cargo run --release -- aggregate --sphincs 220 --log-inv-rate 1 --repeat 3` -- `cargo run --release -- recursion --n 2 --xmss-per-leaf 900 --log-inv-rate 2 --repeat 3` +- `cargo run --release -- fibonacci --n 2000000 --log-inv-rate 1 --repeat 3` (Fibonacci mod 2^64, on registers) +- `cargo run --release -- guest guests/elf/hash.elf --input 50000 --repeat 3` (the precompile, from a Rust guest) +- `BENCH_REPEAT=3 FLOCK_N_LOG=18 cargo test --release -p flock --test batch_proving_hashes -- hash_batch_prove_verify --exact --nocapture --include-ignored` (flock alone, on its hand-optimized circuit) + +## Read-write arrays + +The registers, RAM and the advice are read-write, by timestamped offline memory checking (`doc/leanvm` §sec:memchan). What to keep in mind before touching it: + +- **The clock rides the state tuple**, `(pc, ts)`, and advances by the row's stride (`ClassSpec::stride`: 4, or 18 for a hash row): a row's access in slot `k` carries the timestamp `g^k·ts`, which is what lets one row touch the same cell twice (`add a0, a0, a0`). Slots are `rs1` at 0, `rs2` at 1, the RAM access at 2, `rd` at 3; a hash row has no `rd` write and its sixteen block words take slots 2 to 17. The run starts at cycle 1, because a seed is stamped `g^0` and an access must be strictly later than the one before. +- **Strictness is the soundness.** An access pulls `(addr, prev, old)` and pushes `(addr, g^k·ts, new)`, with `prev·lo = g^k·ts·hi`, where `lo` and `hi` are read off two uncommitted range arrays (`{g^(j+1)}` and `{g^(-2^16·j)}`, 2^16 entries each). The `+1` in the low array is the strict `<`: with a gap of zero a read pulls the tuple it pushes and returns anything. +- **Padding rows have clock zero**, and zero is no power of `g`: their state tuples close around a fill block (`0·g^s = 0`), their accesses are forced to `prev = 0` and cancel themselves, and nothing they flush can meet a tuple of the run. They are written out by `cpu::execute`, not executed, and touch no memory. Any new table has to keep this true: every memory tuple's timestamp must be `g^k·ts` or the committed `prev`, nothing else. **The verifier's notion of a padding row is `ts = 0` and nothing else**, so a prover may close its zero-clock rows around any cycle of the program's own control flow rather than a fill block; that is inert too, and the fill blocks exist to make the fill exact, not to make it safe. What makes `prev = 0` is the gap check against the range arrays, whose entries are all nonzero, so a range array that ever held a zero would break this. +- **Two committed columns per array**: what it holds after the run, and each cell's last timestamp. What it holds before is public for the registers (zero) and RAM (`Coord::Sparse`: the input, the image, zeros, evaluated in time proportional to the image), and a third committed column for the advice (`ADV_INIT`), which is the prover's. RAM and the advice share `SEP_MEM`; they never share an address, every region's base being a multiple of its largest size (`rv::TEXT_BASE`, `rv::ADVICE_BASE`, `rv::RAM_BASE`), so word `z` of a region sits at `base ^ (z << 3)` and the seed block's address is the free `Coord::IntIndex`. +- **A padding row rewrites what it writes**, its `old` column set to its `new` (the register write's `vd_old`, the hash's `out_old`), which is why a row can never update a cell in place: the hash reads `h` and writes `out` in different words, since no chaining value is a fixed point of the compression. +- **A gap is below 2^32**, and a cell's first access is measured from zero, so a run is capped near 2^30 cycles. The executor asserts it. +- `a_stale_read_unbalances_the_bus` is the soundness regression test (a forged run that serves an overwritten register), `a_forged_load_unbalances_the_bus` its RAM counterpart, and `leaf::unmatched_leaves` (test-only) names the tuples a forged run leaves unmatched, which says more than a failing proof. `lean_vm/tests/verifiers/programs.rs` holds the hand-assembled programs checked by both verifiers, `guests.rs` the Rust guests. + +## The RISC-V machine + +`lean_vm::rv` is the machine, `lean_vm::tables` and `lean_vm::cpu` prove it. What to keep in mind: -`aggregate` takes a count per scheme, both defaulting to zero, so either alone or a mix of the two is one command; `recursion --sphincs-per-leaf` likewise puts both schemes in one tree. One SPHINCS verification uses 531 compressions against XMSS's 144; use the benchmark output to compare complete VM cycle counts. `aggregate --blobs` adds LeanDA blobs, and `recursion --blobs-per-leaf` includes them in each child. +- **The program is public, so decoding is free.** `rv::decode` turns each word into an `Entry` once: an instruction class, a `flags` word selecting what the class's one function does, the three register cells the row touches, the immediate already sign-extended, the branch target, and the `link` and `jalr` selectors. `LUI`, `AUIPC` and `JAL` fold to constants, `ECALL` is a jump to the halt slot, and everything rv64im does not define (reserved shift encodings, `EBREAK`, CSRs) is an illegal entry. The bytecode lookup returns those fields; no table ever decomposes an instruction word. An entry whose class has no table has tag zero, which no row can read. +- **The statement is about the decoded table**, so the rules that make it RISC-V are checked where a table enters, on both sides (`rv::Entry::is_well_formed` in `Program::new`, `check_bytecode` in Python): two registers below 32 are read, the cell written is in `1..=32`, the successor is `pc + 4`, the flags are ones the class defines. The proof system itself is sound for any table. +- **`x0` is hardwired by the decoder.** Every row reads two registers and writes one. An instruction with fewer reads `x0`; one with no destination, or with `rd = x0`, writes `SINK` (cell 32), which nothing reads. So cell 0 is never written, and its seed is zero. +- **Registers are a read-write array of their own**, under their own separator `REG`, so that loads and stores cannot reach them: 64 cells, a public zero seed, committed final values and timestamps, the same clock, gap check and range arrays as memory. A register's number comes straight from the bytecode, so an access needs no address arithmetic. +- **No integer addition happens on the bus.** A load's or a store's address is its circuit's word, with the misalignment bits ORed back in (`semantics::bus_address`), so a misaligned or out-of-range access names no seeded cell and the bus does not balance; a hash row's block words are at `v1 ^ 8k`, which is `Coord::Sum(Col(v1), Const(8k))`, the XOR being the sum in `K`. The `EXP` lookup of the leanISA days is gone. +- **State is `(pc, ts)`**, `pc` the real byte address. Instruction `z` sits at `TEXT_BASE ^ (z << 2)` and the bytecode block's address coordinate is `Coord::IntIndex { base, shift }`, whose MLE is linear. Everything sits inside one 2 GiB window and below `0x7FFF_F800` for the code models' sake. A computed or misaligned jump target needs no check: the next row's bytecode read finds no entry. +- **A trap is the absence of a proof** (`rv::Trap`): an illegal or unmapped `pc`, a misaligned or unmapped access, an `ecall` that is not `exit`, the cycle cap, and `TooLong`, a run that would not fit one proof. An illegal word follows the text, so a run falling off it traps instead of sliding into the padding blocks or the halt slot. +- **The halt is an exit.** The run ends on the last slot of the padded text, which is never executed. The verifier claims `a7 = 93` and `a0..a3 = output` on the committed final registers at the Boolean points naming them; the output seeds the transcript with the program's digest, which covers the decoded table, the entry and halt `pc`, RAM's and the advice's sizes and the image. Nothing the program fixes is read from the prover; the Python verifier gets the same things through `public.bin`. +- **The precompile is a class like the others** (`Class::Hash`, table `HASH`): `blake2s rs1, rs2` (custom-0 opcode `0x0b`, `funct3 = 1` on the final block, `rd = funct7 = 0`) compresses the 128-byte block at `rs1` (`h` in words 0..4, the result written to 4..8, the message in 8..16, `rv::hash`), with the counter in `rs2` and the finalization word in the bytecode's flags; a base that is no word address traps, an unaligned one permutes the words deterministically (a guest bug, not a forgery). Its row has no `rd`, `imm` or `out` columns (constants `SINK` and 0 in its bytecode tuple), eighteen accesses and a stride of 18. +- **The interpreter is the reference** (`rv::Machine`), tested against an executor written from the specification on byte-addressed memory that shares no code with it, and against the RISC-V architectural tests' `I` and `M` suites (ACT4, `conformance/act4/`). `cpu::execute` is that interpreter plus the memory argument's bookkeeping. + +## Instruction classes and their circuits + +Addition with carries, comparisons, shifts, AND/OR, multiplication and division are Boolean relations no degree-2 identity over `K` expresses, so each instruction class is a Boolean circuit proven by flock (`doc/leanvm` Annex C), and a table only does plumbing: + +- **One generic table, specialized by a `tables::ClassSpec`**: the state step, the bytecode read, two register reads, one register write (unless `Ram::Block`), and the class's RAM accesses (`Ram::None`, one cell `Read` or `Write`n at the circuit's address, or the hash's `Block`), with `npc = pc4 + taken·dt + jalr·(out + pc4)` and `rd <- out + link·(out + pc4)` as degree-2 bus forms for a class with control flow. A new class is a spec, a circuit in `rv::circuits`, its reference function in `rv::semantics`, a no-op word in `cpu::filler`, its decoding, and the same in Python (`Table(...)` in `TABLES`, its gate list, its flags in `check_bytecode`). +- **Every word the circuit reads or writes is a virtual column** of the table, living in the class's packed witness (`Q_BASE + t`, one committed column per table, instance `j` being row `j`): `flags` and `imm` from the bytecode tuple, `v1` and `v2` from the register tuples, a load's `address` and `cell`, a hash's block words, `out`, `taken`. The bus is the whole binding. `class_flock::Prepared::build` asserts that what the circuit computed is what the interpreter did. A hint (`DIV`'s quotient and remainder) is a port in no column, and what a circuit asserts (`Word::Bad`) rides bytecode slot 13, where the program is zero. +- **Circuits are gate lists over word ports** (`flock::circuit`): inputs, outputs, the constant, then products in the order they are made. A port bit with no gate is an empty row, hence zero, which is what makes a one-bit output such as `taken` a 0 or 1 field element: give such an output a word to itself and never put a free wire on its spare bits. What Rust and Python must agree on is the port layout and the ORDER PRODUCTS ARE MADE IN; XOR order is free. `alu_is_its_reference` pins a circuit to its reference function, and the end-to-end tests pin the Python mirror. +- **One reduction per class, one opening for all**: zerocheck plus lincheck per table, in table order after the exit claims, each leaving a claim on its own witness; `pcs::stack_open` takes one ring-switched region per witness, all under one map challenge. +- **Batch floors.** Flock needs eight instances and a zerocheck cube of `2^13` bits (`class_flock::n_blocks_log`); padding rows supply them, as honest instances on zero registers. +- **Witness generation is the generic walk of the gate list, bit by bit**, and is the prover's largest single stage. A word-arithmetic or bit-sliced generator per circuit is a known follow-up, the hash's `flock::hash` kernels being the model. +- **Circuit sizes are structure, not measurements**: `k_log` per class is pinned in its `ClassSpec` and asserted against the built circuit; product counts are in `doc/leanvm` Annex C. ## The proving arena (`zk_alloc`) @@ -90,20 +117,12 @@ No rayon. Every parallel site is "N independent items, each writing its own disj `LEANVM_NUM_THREADS` sets the **performance**-worker count, leaving E-workers in place. `1` = strictly sequential. -## Three verifiers, one protocol +## Two verifiers, one protocol -The same verification algorithm is written out three times, in three languages. Any change to snark protocol has to land in all three. +The same verification algorithm is written out twice, in two languages. Any change to the snark protocol has to land in both. 1. **Rust**, `lean_vm::cpu::verify`. The native verifier. -2. **Python**, `python-verifier/verifier.py` (no dependencies), for readability and simplicity. Pinned by `lean_vm/tests/verifiers/python_verifier.rs`. -3. **Recursive verifier**, `crates/rec_aggregation/guests/lean_ethereum.py`. Its zkDSL compiles to the ISA; proving its execution gives a proof of child proofs. - -Understand the third before changing the verifier. `guests/lean_ethereum.py` is zkDSL, not runnable Python. `lean_compiler` lowers it to the six-opcode, write-once-memory VM, so the prover proves every verifier step. Its size and instruction mix are what the recursion benchmark reports first. It verifies raw signatures of both schemes: a node's coverage table has one contiguous region per XMSS `(epoch, message)` group and separate regions for SPHINCS and DA roots, so the one range check a write already needs also keeps a signature off another group's declared keys, of either scheme, and the statement's signer lists say which scheme verified which key against which `(epoch, message)`. The XMSS signers are grouped by `(epoch, message)`, so one epoch signed at under several messages is one group per message, a runtime number of groups (at most `MAX_EPOCHS`) bound through the signer-set digest, which is plain BLAKE2s of a byte string (each list's own digest folded into it, likewise plain BLAKE2s): a run-time length rides the byte counter because the counter is a memory operand, split as `doc/leanvm` §sec:prog-byte-counter describes. A child's groups need not equal its parent's, a hinted map tying each child group to a parent group with the same epoch and message. A SPHINCS signer's message rides its own four-cell slot, so that list is `(key, message)` pairs; both lists count claims rather than distinct signers, an XMSS key claiming once per `(epoch, message)` it signed. Both schemes' tweaks are built in-circuit: XMSS's from the epochs the statement carries, derived once per group that verifies raw XMSS signatures and skipped by one that verifies none, SPHINCS's per signature from the index its message digest picks. Two consequences: - -- The guest is **self-referential**: it verifies proofs of itself, so `unified_guest` compiles it to a fixed point on its own log size. The digest needs no fixed point, riding the statement instead of the code, which is also what lets one bytecode serve any inner size and PCS rate. -- It does not verify *quite* everything in-circuit. Three claims on fixed polynomials (stacked bytecode, flock's A0/B0) are deferred. Each node batches its children's carried claims with the fresh ones its verifications raise, `2n` per polynomial down to one; only the root's are discharged natively, by `EthereumProof::verify` (explained in `doc/leanvm/`). - -`aggregate_two_to_one` is the fast end-to-end check; `aggregate_statement_binds` and `aggregate_hints_bind` are the adversarial ones, tampering the wire object and the witness respectively. A child must commit at least `2^MU_MIN` or the guest has no opening arm for it, so `aggregate` sets `Program::min_log_committed` and a smaller run grows its `SET` table through the fill blocks until it clears the floor. +2. **Python**, `python-verifier/verifier.py` (no dependencies), for readability and simplicity. Pinned by `lean_vm/tests/verifiers/python_verifier.rs`, which feeds it the raw proof `cpu::verify_to_raw` returns. ## Conventions that bite @@ -118,33 +137,26 @@ Understand the third before changing the verifier. `guests/lean_ethereum.py` is - Simpler is better. - **Fiat-Shamir:** `add_scalar`/`next_scalar` bind into the Fiat-Shamir state as a side effect. The public statement seeds the transcript at construction; the transport exposes no separate observe operation. Never re-observe data that rode the stream, which silently desynchronizes the two sides. - **Prover and verifier derive the layout identically** from announced sizes. Changes to `placements_of` or the schema land on both sides. `col_kappas` is derived from `col_kappa_sources` rather than written out twice, so the two can no longer drift; keep it that way. -- **The L0 lane fold binds the committed witness's TOP `INITIAL_FOLDING_FACTOR` variables**, because lane `l` of the interleaved commitment is the stack block `q[l·2^(μ-k) ..)`. That makes the witness's zero tail whole lanes, so `whir::commit` encodes only `StackShape::n_lanes` of them, and the opening's dense weight, its first `k` sumcheck rounds and the stack allocation shrink with it. **A leaf image is still `2^k` words**, the absent lanes contributing their codeword's zeros, but those zeros LEAD it (codeword lane `t` is stack block `n_lanes-1-t`): their whole 64-byte blocks are then one shared chaining value (`hash::zero_prefix_state`) the committer hashes once rather than per leaf, and only the image's tail rides the proof, so `PrunedMerklePaths` stores `n_lanes` words per L0 row while `RawMerklePath` (what the guest and the Python verifier read) carries the full image. The Rust and Python verifiers therefore derive `n_lanes` from the announced layout to read a row; the guest never needs it, its hints being full images. Since `mu = log2_ceil(placed)`, `n_lanes` is always in `[2^(k-1)+1, 2^k]`: the encode saving caps near half, the hashing saving is quantized to whole blocks of 8 lanes, and both are ~0 just above a power of two. The cost is that fold challenges arrive in round order while every transparent weight is written in witness coordinates, so all three verifiers rotate the terminal point left by `k` before evaluating it (`whir.rs` before `eval_b_at`, `verifier.py` before `evaluate_basis`, `open_stacked` in the guest). Anything else that reads the opening's point (per-level induced weights, the residual) stays in round order. -- **A failed guest `assert` surfaces as a write-once memory conflict**, not an assertion message, but it names the function and source line: `write-once conflict at cell 34: had ..., new ... at pc ... (in verify_sub (line 2906))`, as does every other `ExecError` (a failed range check, a wild `DEREF`). A conflict that `had 0:0:0` can instead be an ordering bug: an instruction read the cell, unwritten, before this write. Parse and lowering errors carry a line too. Reach for `DBG_DISASM` only when the line is not enough, or when the pc lands in a fill block, which has no source line by construction. -- Guests are single-file; the compiler skips `from snark_lib import *`, which exists only so editors accept the file as Python. -- **One symbol, one meaning, across the whole leanVM document.** All notation is defined in `doc/leanvm/preamble/macros.tex`: define a new macro there rather than inline, and check the letter is free first. Annex B's "Symbols" table maps its letters back to WHIR/Ligerito/BCHKS25, so read it before renaming one. A sumcheck round challenge is `\fc` everywhere, which is what keeps `\rho` free for the rate; `r` is the point a claim is made at, not a challenge. **A rename in the document is a rename in the implementations**: the Rust prover and verifier, `python-verifier/verifier.py`, and `guests/lean_ethereum.py` name their variables after the document's symbols, so the four have to move together. +- **The L0 lane fold binds the committed witness's TOP `INITIAL_FOLDING_FACTOR` variables**, because lane `l` of the interleaved commitment is the stack block `q[l·2^(μ-k) ..)`. That makes the witness's zero tail whole lanes, so `whir::commit` encodes only `StackShape::n_lanes` of them, and the opening's dense weight, its first `k` sumcheck rounds and the stack allocation shrink with it. **A leaf image is still `2^k` words**, the absent lanes contributing their codeword's zeros, but those zeros LEAD it (codeword lane `t` is stack block `n_lanes-1-t`): their whole 64-byte blocks are then one shared chaining value (`hash::zero_prefix_state`) the committer hashes once rather than per leaf, and only the image's tail rides the proof, so `PrunedMerklePaths` stores `n_lanes` words per L0 row while `RawMerklePath` (what the Python verifier reads) carries the full image. Both verifiers therefore derive `n_lanes` from the announced layout to read a row. Since `mu = log2_ceil(placed)`, `n_lanes` is always in `[2^(k-1)+1, 2^k]`: the encode saving caps near half, the hashing saving is quantized to whole blocks of 8 lanes, and both are ~0 just above a power of two. The cost is that fold challenges arrive in round order while every transparent weight is written in witness coordinates, so both verifiers rotate the terminal point left by `k` before evaluating it (`whir.rs` before `eval_b_at`, `verifier.py` before `evaluate_basis`). Anything else that reads the opening's point (per-level induced weights, the residual) stays in round order. +- **One symbol, one meaning, across the whole leanVM document.** All notation is defined in `doc/leanvm/preamble/macros.tex`: define a new macro there rather than inline, and check the letter is free first. Annex B's "Symbols" table maps its letters back to WHIR/Ligerito/BCHKS25, so read it before renaming one. A sumcheck round challenge is `\fc` everywhere, which is what keeps `\rho` free for the rate; `r` is the point a claim is made at, not a challenge. **A rename in the document is a rename in the implementations**: the Rust prover and verifier and `python-verifier/verifier.py` name their variables after the document's symbols, so the three have to move together. - **Doc labels are an API.** `crates/pcs` cites `thm:rbr` and `thm:mca-johnson` by name and several crates cite `doc/leanvm/main.tex` sections, so renaming a label breaks those pointers with nothing to catch it. `doc/leanvm/body/NN-*.tex` prefixes match section numbers, so inserting a section renumbers the rest. - **No em-dashes or en-dashes in prose**, anywhere a human reads it: docs, LaTeX, comments, commit messages. Restructure with a comma, colon, parentheses, or two sentences. - **Never hard-wrap prose in Markdown or LaTeX.** One paragraph is one line; let the editor wrap it. Artificial line breaks make every later edit a reflow, so diffs show rewrapped lines instead of changed words. Applies to `.md` and `.tex` alike; code blocks, tables and list items keep their own line. -## Soundness - -- In the recursion program, the prover transmits advice to the verifier, called "hints". Hints are untrusted witness data and must be checked by the verifier; a malicious prover must not be able to prove an invalid witness. - ## Env knobs | var | effect | | ------------------------------------------------------------------------------------------------------- | ------------------------------------------------ | | `LEANVM_NUM_THREADS` | performance-worker count; `1` = sequential | | `LEANVM_PROFILE` | per-stage prover timings | +| `LEANVM_ACT4` | directory of generated ACT4 ELF files (`I/`, `M/`) to test instead of `conformance/act4/elf/` | | `ZK_ALLOC_STATS` | arena peak/phase, high water, overflow | | `ZK_ALLOC_POISON` | fill released arena blocks, to catch use-after-free | | `BENCH_REPEAT`, `BENCH_COOLDOWN` | `--repeat`/`--cooldown` for `#[ignore]`d benches | -| `LEANVM_XMSS_N`, `LEANVM_HASH_N`, `LEANVM_HASH_UNROLL` | workload sizes in tests | | `FLOCK_N_LOG`, `FLOCK_PROVE_TRACE`, `FLOCK_ZC_TIMING`, `LINCHECK_TRACE` | flock batch size, stage traces | -| `PCS_LOG_N`, `PCS_LOG_INV_RATE`, `PCS_MIN_MU`, `PCS_SAMPLES` | PCS throughput bench | +| `PCS_LOG_N`, `PCS_LOG_INV_RATE`, `PCS_SAMPLES` | PCS throughput bench | | `WHIR_TRACE`, `WHIR_NUM_VARS`, `WHIR_LOG_INV_RATE` | WHIR NTT/Merkle split | -| `DBG_PROF{,_DUMP}`, `DBG_LOOPS`, `DBG_DISASM`, `DBG_LOWER`, `DBG_PLACEHOLDERS` | compiler / guest-cycle attribution | ## Side notes -- Grinding chooses the smallest valid nonce, including in parallel. Randomized signature inputs can still make proofs differ between runs. +- Grinding chooses the smallest valid nonce, including in parallel. diff --git a/Cargo.lock b/Cargo.lock index 42d9dc829..f9803310c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -11,54 +11,6 @@ dependencies = [ "memchr", ] -[[package]] -name = "alloy-primitives" -version = "1.7.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce7b00f0cb42c66ec353076ded1dff1fbf818f6e0e26c40c8a8456c04483fca4" -dependencies = [ - "alloy-rlp", - "bytes", - "cfg-if", - "const-hex", - "derive_more", - "fixed-cache", - "foldhash", - "hashbrown 0.17.1", - "indexmap 2.14.1", - "itoa", - "k256", - "keccak-asm", - "paste", - "proptest", - "rand 0.9.4", - "rapidhash", - "ruint", - "rustc-hash", - "secp256k1", - "serde", - "sha3", -] - -[[package]] -name = "alloy-rlp" -version = "0.3.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "24671b1f62edcf0f9b62994c7bf72cd621a04a4b99f5020ece1a647b40e2f103" -dependencies = [ - "arrayvec", - "bytes", -] - -[[package]] -name = "android_system_properties" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" -dependencies = [ - "libc", -] - [[package]] name = "ansi_term" version = "0.12.1" @@ -119,2087 +71,353 @@ dependencies = [ ] [[package]] -name = "ark-ff" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6b3235cc41ee7a12aaaf2c575a2ad7b46713a8a50bda2fc3b003a04845c05dd6" -dependencies = [ - "ark-ff-asm 0.3.0", - "ark-ff-macros 0.3.0", - "ark-serialize 0.3.0", - "ark-std 0.3.0", - "derivative", - "num-bigint", - "num-traits", - "paste", - "rustc_version 0.3.3", - "zeroize", -] - -[[package]] -name = "ark-ff" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ec847af850f44ad29048935519032c33da8aa03340876d351dfab5660d2966ba" -dependencies = [ - "ark-ff-asm 0.4.2", - "ark-ff-macros 0.4.2", - "ark-serialize 0.4.2", - "ark-std 0.4.0", - "derivative", - "digest 0.10.7", - "itertools 0.10.5", - "num-bigint", - "num-traits", - "paste", - "rustc_version 0.4.1", - "zeroize", -] - -[[package]] -name = "ark-ff" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a177aba0ed1e0fbb62aa9f6d0502e9b46dad8c2eab04c14258a1212d2557ea70" -dependencies = [ - "ark-ff-asm 0.5.0", - "ark-ff-macros 0.5.0", - "ark-serialize 0.5.0", - "ark-std 0.5.0", - "arrayvec", - "digest 0.10.7", - "educe", - "itertools 0.13.0", - "num-bigint", - "num-traits", - "paste", - "zeroize", -] - -[[package]] -name = "ark-ff" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f7a806ac6c8307b929df4645776290a50ee2aac754ad09d8bdf73391309e43af" -dependencies = [ - "ark-ff-asm 0.6.0", - "ark-ff-macros 0.6.0", - "ark-serialize 0.6.0", - "ark-std 0.6.0", - "digest 0.10.7", - "educe", - "num-bigint", - "num-traits", - "zeroize", -] - -[[package]] -name = "ark-ff-asm" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db02d390bf6643fb404d3d22d31aee1c4bc4459600aef9113833d17e786c6e44" -dependencies = [ - "quote", - "syn 1.0.109", -] - -[[package]] -name = "ark-ff-asm" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ed4aa4fe255d0bc6d79373f7e31d2ea147bcf486cba1be5ba7ea85abdb92348" -dependencies = [ - "quote", - "syn 1.0.109", -] - -[[package]] -name = "ark-ff-asm" -version = "0.5.0" +name = "bincode" +version = "1.3.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "62945a2f7e6de02a31fe400aa489f0e0f5b2502e69f95f853adb82a96c7a6b60" +checksum = "b1f45e9417d87227c7a56d22e471c6206462cba514c7590c09aff4cf6d1ddcad" dependencies = [ - "quote", - "syn 2.0.118", + "serde", ] [[package]] -name = "ark-ff-asm" -version = "0.6.0" +name = "cfg-if" +version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1479009684adc073dff49a1025d3a7065b317a9ead25aaaca38cdc70058ba8a2" -dependencies = [ - "quote", - "syn 2.0.118", -] +checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] -name = "ark-ff-macros" -version = "0.3.0" +name = "clap" +version = "4.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db2fd794a08ccb318058009eefdf15bcaaaaf6f8161eb3345f907222bac38b20" +checksum = "1ddb117e43bbf7dacf0a4190fef4d345b9bad68dfc649cb349e7d17d28428e51" dependencies = [ - "num-bigint", - "num-traits", - "quote", - "syn 1.0.109", + "clap_builder", + "clap_derive", ] [[package]] -name = "ark-ff-macros" -version = "0.4.2" +name = "clap_builder" +version = "4.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7abe79b0e4288889c4574159ab790824d0033b9fdcb2a112a3182fac2e514565" +checksum = "714a53001bf66416adb0e2ef5ac857140e7dc3a0c48fb28b2f10762fc4b5069f" dependencies = [ - "num-bigint", - "num-traits", - "proc-macro2", - "quote", - "syn 1.0.109", + "anstream", + "anstyle", + "clap_lex", + "strsim", ] [[package]] -name = "ark-ff-macros" -version = "0.5.0" +name = "clap_derive" +version = "4.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09be120733ee33f7693ceaa202ca41accd5653b779563608f1234f78ae07c4b3" +checksum = "f2ce8604710f6733aa641a2b3731eaa1e8b3d9973d5e3565da11800813f997a9" dependencies = [ - "num-bigint", - "num-traits", + "heck", "proc-macro2", "quote", - "syn 2.0.118", + "syn", ] [[package]] -name = "ark-ff-macros" -version = "0.6.0" +name = "clap_lex" +version = "1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4a0691ed21ef00ef89c1e9bda832eba493dda3ec2f8d892fb25b705f73f06bb8" -dependencies = [ - "num-bigint", - "num-traits", - "proc-macro2", - "quote", - "syn 2.0.118", -] +checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9" [[package]] -name = "ark-serialize" -version = "0.3.0" +name = "colorchoice" +version = "1.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d6c2b318ee6e10f8c2853e73a83adc0ccb88995aa978d8a3408d492ab2ee671" -dependencies = [ - "ark-std 0.3.0", - "digest 0.9.0", -] +checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" [[package]] -name = "ark-serialize" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "adb7b85a02b83d2f22f89bd5cac66c9c89474240cb6207cb1efc16d098e822a5" +name = "fiat_shamir" +version = "0.1.0" dependencies = [ - "ark-std 0.4.0", - "digest 0.10.7", - "num-bigint", + "parallel", + "primitives", + "serde", ] [[package]] -name = "ark-serialize" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f4d068aaf107ebcd7dfb52bc748f8030e0fc930ac8e360146ca54c1203088f7" +name = "flock" +version = "0.1.0" dependencies = [ - "ark-std 0.5.0", - "arrayvec", - "digest 0.10.7", - "num-bigint", + "fiat_shamir", + "parallel", + "pcs", + "primitives", + "zk_alloc", ] [[package]] -name = "ark-serialize" -version = "0.6.0" +name = "getrandom" +version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a74dd304fd536fb95d0a328e72be759209cc496a9da094c5bc56e5fea4f9e86b" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" dependencies = [ - "ark-serialize-derive", - "ark-std 0.6.0", - "digest 0.10.7", - "num-bigint", - "serde_with", + "cfg-if", + "libc", + "r-efi", + "wasip2", ] [[package]] -name = "ark-serialize-derive" -version = "0.6.0" +name = "heck" +version = "0.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4f153690697a2b91e5e1251ff98411ee5371500a111a0fd317a70e588eb300f9" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.118", -] +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" [[package]] -name = "ark-std" -version = "0.3.0" +name = "is_terminal_polyfill" +version = "1.70.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1df2c09229cbc5a028b1d70e00fdb2acee28b1055dfb5ca73eea49c5a25c4e7c" -dependencies = [ - "num-traits", - "rand 0.8.8", -] +checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" [[package]] -name = "ark-std" -version = "0.4.0" +name = "lazy_static" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94893f1e0c6eeab764ade8dc4c0db24caf4fe7cbbaafc0eba0a9030f447b5185" -dependencies = [ - "num-traits", - "rand 0.8.8", -] +checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" [[package]] -name = "ark-std" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "246a225cc6131e9ee4f24619af0f19d67761fff15d7ccc22e42b80846e69449a" +name = "lean_vm" +version = "0.1.0" dependencies = [ - "num-traits", - "rand 0.8.8", + "bincode", + "fiat_shamir", + "flock", + "parallel", + "pcs", + "primitives", + "tracing", + "zk_alloc", ] [[package]] -name = "ark-std" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "367c9c827ed431bff6868b7aa926e05b16eb46603cc8b6e768e4a5553fa1d155" +name = "leanvm" +version = "0.1.0" dependencies = [ - "num-traits", - "rand 0.8.8", + "bincode", + "clap", + "lean_vm", + "primitives", + "tracing", + "zk_alloc", ] [[package]] -name = "arrayvec" -version = "0.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" - -[[package]] -name = "auto_impl" -version = "1.3.0" +name = "libc" +version = "0.2.186" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ffdcb70bdbc4d478427380519163274ac86e52916e10f0a8889adf0f96d3fee7" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.118", -] +checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" [[package]] -name = "autocfg" -version = "1.5.1" +name = "log" +version = "0.4.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" +checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" [[package]] -name = "base16ct" +name = "matchers" version = "0.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4c7f02d4ea65f2c1853089ffd8d2787bdbc63de2f0d29dedbcf8ccdfa0ccd4cf" - -[[package]] -name = "base64" -version = "0.22.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" - -[[package]] -name = "base64ct" -version = "1.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2af50177e190e07a26ab74f8b1efbfe2ef87da2116221318cb1c2e82baf7de06" - -[[package]] -name = "bincode" -version = "1.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1f45e9417d87227c7a56d22e471c6206462cba514c7590c09aff4cf6d1ddcad" -dependencies = [ - "serde", -] - -[[package]] -name = "bitcoin-consensus-encoding" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6712f9c6fd6785b3b270884e57c441c403dc5d7e19ca45368c97c7a1de3000ec" +checksum = "d1525a2a28c7f4fa0fc98bb91ae755d1e2d1505079e05539e35bc876b5d65ae9" dependencies = [ - "bitcoin-internals", - "hex-conservative 1.2.0", - "serde", + "regex-automata", ] [[package]] -name = "bitcoin-internals" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d573f4cf32996a8dce612e4348cece65a241f1882ed594047c9ba348e8869fa5" - -[[package]] -name = "bitcoin-io" -version = "0.1.101" +name = "memchr" +version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb5de036369d1ac59d3c1819ebc4d850f89466f5401c571a285b6ed564a4cb78" -dependencies = [ - "bitcoin-consensus-encoding", -] +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" [[package]] -name = "bitcoin_hashes" -version = "0.14.101" +name = "nu-ansi-term" +version = "0.50.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bca4c7abb40c8817d77403c880988cfd484f23ab2365726afb2f798363e2c4a2" +checksum = "7957b9740744892f114936ab4a57b3f487491bbeafaf8083688b16841a4240e5" dependencies = [ - "bitcoin-io", - "hex-conservative 0.2.3", + "windows-sys", ] [[package]] -name = "bitflags" -version = "1.3.2" +name = "once_cell" +version = "1.21.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" [[package]] -name = "bitflags" -version = "2.13.1" +name = "once_cell_polyfill" +version = "1.70.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" +checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" [[package]] -name = "bitvec" -version = "1.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddcec3d12c579d40898fe0a9a358a803c23e9c52ca3c425707f81c9436211837" +name = "parallel" +version = "0.1.0" dependencies = [ - "funty", - "radium", - "tap", - "wyz", + "libc", ] [[package]] -name = "block-buffer" -version = "0.10.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +name = "pcs" +version = "0.1.0" dependencies = [ - "generic-array", + "bincode", + "fiat_shamir", + "parallel", + "primitives", + "tracing", + "zk_alloc", ] [[package]] -name = "block-buffer" -version = "0.12.1" +name = "pin-project-lite" +version = "0.2.17" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa" -dependencies = [ - "hybrid-array", -] +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" [[package]] -name = "bs58" -version = "0.5.1" +name = "ppv-lite86" +version = "0.2.21" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bf88ba1141d185c399bee5288d850d63b8369520c1eafc32a0430b5b6c287bf4" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" dependencies = [ - "tinyvec", + "zerocopy", ] [[package]] -name = "bumpalo" -version = "3.20.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" - -[[package]] -name = "byte-slice-cast" -version = "1.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7575182f7272186991736b70173b0ea045398f984bf5ebbb3804736ce1330c9d" - -[[package]] -name = "byteorder" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" - -[[package]] -name = "bytes" -version = "1.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" +name = "primitives" +version = "0.1.0" dependencies = [ + "bincode", + "libc", + "parallel", + "primitives", + "rand", "serde", + "tracing-forest", + "tracing-subscriber", + "zk_alloc", ] [[package]] -name = "cc" -version = "1.4.4" +name = "proc-macro2" +version = "1.0.106" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ad534f4357a5264cce5019c989cf66a4f0dc4e0d1b1d15f8aacec0ff7360273" -dependencies = [ - "find-msvc-tools", - "shlex", -] - -[[package]] -name = "cfg-if" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" - -[[package]] -name = "chrono" -version = "0.4.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" -dependencies = [ - "iana-time-zone", - "num-traits", - "serde", - "windows-link", -] - -[[package]] -name = "clap" -version = "4.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ddb117e43bbf7dacf0a4190fef4d345b9bad68dfc649cb349e7d17d28428e51" -dependencies = [ - "clap_builder", - "clap_derive", -] - -[[package]] -name = "clap_builder" -version = "4.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714a53001bf66416adb0e2ef5ac857140e7dc3a0c48fb28b2f10762fc4b5069f" -dependencies = [ - "anstream", - "anstyle", - "clap_lex", - "strsim", -] - -[[package]] -name = "clap_derive" -version = "4.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ce8604710f6733aa641a2b3731eaa1e8b3d9973d5e3565da11800813f997a9" -dependencies = [ - "heck", - "proc-macro2", - "quote", - "syn 2.0.118", -] - -[[package]] -name = "clap_lex" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8d4a3bb8b1e0c1050499d1815f5ab16d04f0959b233085fb31653fbfc9d98f9" - -[[package]] -name = "colorchoice" -version = "1.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" - -[[package]] -name = "const-hex" -version = "1.19.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "33e2a781ebdf4467d1428dc4593067825fb646f6871475098d8577421af73558" -dependencies = [ - "cfg-if", - "cpufeatures 0.2.17", - "proptest", - "serde_core", -] - -[[package]] -name = "const-oid" -version = "0.9.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c2459377285ad874054d797f3ccebf984978aa39129f6eafde5cdc8315b612f8" - -[[package]] -name = "const_format" -version = "0.2.36" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4481a617ad9a412be3b97c5d403fef8ed023103368908b9c50af598ff467cc1e" -dependencies = [ - "const_format_proc_macros", - "konst", -] - -[[package]] -name = "const_format_proc_macros" -version = "0.2.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d57c2eccfb16dbac1f4e61e206105db5820c9d26c3c472bc17c774259ef7744" -dependencies = [ - "proc-macro2", - "quote", - "unicode-xid", -] - -[[package]] -name = "convert_case" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "633458d4ef8c78b72454de2d54fd6ab2e60f9e02be22f3c6104cdc8a4e0fceb9" -dependencies = [ - "unicode-segmentation", -] - -[[package]] -name = "core-foundation-sys" -version = "0.8.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" - -[[package]] -name = "cpufeatures" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" -dependencies = [ - "libc", -] - -[[package]] -name = "cpufeatures" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" -dependencies = [ - "libc", -] - -[[package]] -name = "crunchy" -version = "0.2.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5" - -[[package]] -name = "crypto-bigint" -version = "0.5.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0dc92fb57ca44df6db8059111ab3af99a63d5d0f8375d9972e319a379c6bab76" -dependencies = [ - "generic-array", - "rand_core 0.6.4", - "subtle", - "zeroize", -] - -[[package]] -name = "crypto-common" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1bfb12502f3fc46cca1bb51ac28df9d618d813cdc3d2f25b9fe775a34af26bb3" -dependencies = [ - "generic-array", - "typenum", -] - -[[package]] -name = "crypto-common" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce6e4c961d6cd6c9a86db418387425e8bdeaf05b3c8bc1411e6dca4c252f1453" -dependencies = [ - "hybrid-array", -] - -[[package]] -name = "defmt" -version = "1.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e2953bfe4f93bbd20cc71198842756f77d161884c99ebbabc41d80231ded88d1" -dependencies = [ - "bitflags 1.3.2", - "defmt-macros", -] - -[[package]] -name = "defmt-macros" -version = "1.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bad9c72e7ca2137e0dc3813245a0d282fd6daad32fd800af018306a9169b5fe8" -dependencies = [ - "defmt-parser", - "proc-macro2", - "quote", - "syn 2.0.118", -] - -[[package]] -name = "defmt-parser" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" -dependencies = [ - "thiserror", -] - -[[package]] -name = "der" -version = "0.7.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7c1832837b905bbfb5101e07cc24c8deddf52f93225eee6ead5f4d63d53ddcb" -dependencies = [ - "const-oid", - "zeroize", -] - -[[package]] -name = "deranged" -version = "0.5.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" -dependencies = [ - "serde_core", -] - -[[package]] -name = "derivative" -version = "2.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fcc3dd5e9e9c0b295d6e1e4d811fb6f157d5ffd784b8d202fc62eac8035a770b" -dependencies = [ - "proc-macro2", - "quote", - "syn 1.0.109", -] - -[[package]] -name = "derive_more" -version = "2.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d751e9e49156b02b44f9c1815bcb94b984cdcc4396ecc32521c739452808b134" -dependencies = [ - "derive_more-impl", -] - -[[package]] -name = "derive_more-impl" -version = "2.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "799a97264921d8623a957f6c3b9011f3b5492f557bbb7a5a19b7fa6d06ba8dcb" -dependencies = [ - "convert_case", - "proc-macro2", - "quote", - "rustc_version 0.4.1", - "syn 2.0.118", - "unicode-xid", -] - -[[package]] -name = "digest" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3dd60d1080a57a05ab032377049e0591415d2b31afd7028356dbf3cc6dcb066" -dependencies = [ - "generic-array", -] - -[[package]] -name = "digest" -version = "0.10.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" -dependencies = [ - "block-buffer 0.10.4", - "const-oid", - "crypto-common 0.1.6", - "subtle", -] - -[[package]] -name = "digest" -version = "0.11.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2" -dependencies = [ - "block-buffer 0.12.1", - "crypto-common 0.2.2", -] - -[[package]] -name = "dyn-clone" -version = "1.0.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555" - -[[package]] -name = "ecdsa" -version = "0.16.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ee27f32b5c5292967d2d4a9d7f1e0b0aed2c15daded5a60300e4abb9d8020bca" -dependencies = [ - "der", - "digest 0.10.7", - "elliptic-curve", - "rfc6979", - "signature", - "spki", -] - -[[package]] -name = "educe" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d7bc049e1bd8cdeb31b68bbd586a9464ecf9f3944af3958a7a9d0f8b9799417" -dependencies = [ - "enum-ordinalize", - "proc-macro2", - "quote", - "syn 2.0.118", -] - -[[package]] -name = "either" -version = "1.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" - -[[package]] -name = "elliptic-curve" -version = "0.13.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b5e6043086bf7973472e0c7dff2142ea0b680d30e18d9cc40f267efbf222bd47" -dependencies = [ - "base16ct", - "crypto-bigint", - "digest 0.10.7", - "ff", - "generic-array", - "group", - "pkcs8", - "rand_core 0.6.4", - "sec1", - "subtle", - "zeroize", -] - -[[package]] -name = "enum-ordinalize" -version = "4.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "89dd01549b09589510cf0647475075d12071456586d70f5c75c98ae2a5537677" -dependencies = [ - "enum-ordinalize-derive", -] - -[[package]] -name = "enum-ordinalize-derive" -version = "4.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a65863d15a4ce2888bd2f0f543cc963d3879c3a022c8ee43f6141d479a3ac815" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.4", -] - -[[package]] -name = "equivalent" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" - -[[package]] -name = "ethereum_serde_utils" -version = "0.8.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "38df44a7a271ab43835678f9215b53cc2523e4714a215da6643d83dc110245da" -dependencies = [ - "alloy-primitives", - "hex", - "serde", - "serde_derive", - "serde_json", -] - -[[package]] -name = "ethereum_ssz" -version = "0.10.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e462875ad8693755ea8913d6e905715c76ea4836e2254e18c9cf0f7a8f8c2a13" -dependencies = [ - "alloy-primitives", - "ethereum_serde_utils", - "itertools 0.14.0", - "serde", - "serde_derive", - "smallvec", - "typenum", -] - -[[package]] -name = "fastrlp" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "139834ddba373bbdd213dffe02c8d110508dcf1726c2be27e8d1f7d7e1856418" -dependencies = [ - "arrayvec", - "auto_impl", - "bytes", -] - -[[package]] -name = "fastrlp" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce8dba4714ef14b8274c371879b175aa55b16b30f269663f19d576f380018dc4" -dependencies = [ - "arrayvec", - "auto_impl", - "bytes", -] - -[[package]] -name = "ff" -version = "0.13.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c0b50bfb653653f9ca9095b427bed08ab8d75a137839d9ad64eb11810d5b6393" -dependencies = [ - "rand_core 0.6.4", - "subtle", -] - -[[package]] -name = "fiat_shamir" -version = "0.1.0" -dependencies = [ - "parallel", - "primitives", - "serde", -] - -[[package]] -name = "find-msvc-tools" -version = "0.1.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890" - -[[package]] -name = "fixed-cache" -version = "0.1.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2fe63500644ef0269fe6b744e7e5dc5c20b5eebf3d881bc2be53f194636f6583" -dependencies = [ - "equivalent", - "rapidhash", -] - -[[package]] -name = "fixed-hash" -version = "0.8.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "835c052cb0c08c1acf6ffd71c022172e18723949c8282f2b9f27efbc51e64534" -dependencies = [ - "byteorder", - "rand 0.8.8", - "rustc-hex", - "static_assertions", -] - -[[package]] -name = "flock" -version = "0.1.0" -dependencies = [ - "fiat_shamir", - "parallel", - "pcs", - "primitives", - "zk_alloc", -] - -[[package]] -name = "foldhash" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77ce24cb58228fbb8aa041425bb1050850ac19177686ea6e0f41a70416f56fdb" - -[[package]] -name = "funty" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6d5a32815ae3f33302d95fdcb2ce17862f8c65363dcfd29360480ba1001fc9c" - -[[package]] -name = "futures-core" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" - -[[package]] -name = "futures-task" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" - -[[package]] -name = "futures-util" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" -dependencies = [ - "futures-core", - "futures-task", - "pin-project-lite", - "slab", -] - -[[package]] -name = "generic-array" -version = "0.14.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4bb6743198531e02858aeaea5398fcc883e71851fcbcb5a2f773e2fb6cb1edf2" -dependencies = [ - "typenum", - "version_check", - "zeroize", -] - -[[package]] -name = "getrandom" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" -dependencies = [ - "cfg-if", - "libc", - "wasi", -] - -[[package]] -name = "getrandom" -version = "0.3.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" -dependencies = [ - "cfg-if", - "libc", - "r-efi", - "wasip2", -] - -[[package]] -name = "group" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0f9ef7462f7c099f518d754361858f86d8a07af53ba9af0fe635bbccb151a63" -dependencies = [ - "ff", - "rand_core 0.6.4", - "subtle", -] - -[[package]] -name = "hashbrown" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a9ee70c43aaf417c914396645a0fa852624801b24ebb7ae78fe8272889ac888" - -[[package]] -name = "hashbrown" -version = "0.17.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" -dependencies = [ - "foldhash", - "serde", - "serde_core", -] - -[[package]] -name = "heck" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" - -[[package]] -name = "hex" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" - -[[package]] -name = "hex-conservative" -version = "0.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db3fef046dca3ca91ee1408a8c1b80ab777e80a4d308d1bf4e7adb3fcb047e08" -dependencies = [ - "arrayvec", -] - -[[package]] -name = "hex-conservative" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35431185f361ccf3ffc58254628af5f1f5d5f28531da2e02e5d6c82bbc282a10" -dependencies = [ - "arrayvec", -] - -[[package]] -name = "hmac" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6c49c37c09c17a53d937dfbb742eb3a961d65a994e6bcdcf37e7399d0cc8ab5e" -dependencies = [ - "digest 0.10.7", -] - -[[package]] -name = "hybrid-array" -version = "0.4.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "707114b52a152fa7bdb290cd7cd5912d9467273b6d74e21b8d81aca1f8533f6b" -dependencies = [ - "typenum", -] - -[[package]] -name = "iana-time-zone" -version = "0.1.65" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" -dependencies = [ - "android_system_properties", - "core-foundation-sys", - "iana-time-zone-haiku", - "js-sys", - "log", - "wasm-bindgen", - "windows-core", -] - -[[package]] -name = "iana-time-zone-haiku" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" -dependencies = [ - "cc", -] - -[[package]] -name = "impl-codec" -version = "0.6.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba6a270039626615617f3f36d15fc827041df3b78c439da2cadfa47455a77f2f" -dependencies = [ - "parity-scale-codec", -] - -[[package]] -name = "impl-trait-for-tuples" -version = "0.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0eb5a3343abf848c0984fe4604b2b105da9539376e24fc0a3b0007411ae4fd9" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.118", -] - -[[package]] -name = "indexmap" -version = "1.9.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bd070e393353796e801d209ad339e89596eb4c8d430d18ede6a1cced8fafbd99" -dependencies = [ - "autocfg", - "hashbrown 0.12.3", - "serde", -] - -[[package]] -name = "indexmap" -version = "2.14.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07aa2048142242915a31d35844fb311e0e53fcca590c3a0a40dcf1b841fa09eb" -dependencies = [ - "equivalent", - "hashbrown 0.17.1", - "serde", - "serde_core", -] - -[[package]] -name = "is_terminal_polyfill" -version = "1.70.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" - -[[package]] -name = "itertools" -version = "0.10.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b0fd2260e829bddf4cb6ea802289de2f86d6a7a690192fbe91b3f46e0f2c8473" -dependencies = [ - "either", -] - -[[package]] -name = "itertools" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "413ee7dfc52ee1a4949ceeb7dbc8a33f2d6c088194d9f922fb8318faf1f01186" -dependencies = [ - "either", -] - -[[package]] -name = "itertools" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2b192c782037fadd9cfa75548310488aabdbf3d2da73885b31bd0abd03351285" -dependencies = [ - "either", -] - -[[package]] -name = "itoa" -version = "1.0.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" - -[[package]] -name = "jiff" -version = "0.2.35" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "668b7183bd07af9a4885f5c35b0cc5c83c4607a913c16b7e17291832910d2dcc" -dependencies = [ - "defmt", - "jiff-core", - "jiff-static", - "jiff-tzdb-platform", - "log", - "portable-atomic", - "portable-atomic-util", - "serde_core", - "windows-link", -] - -[[package]] -name = "jiff-core" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7feca88439efe53da3754500c1851dedf3cb36c524dd5cf8225cc0794de95d09" -dependencies = [ - "defmt", -] - -[[package]] -name = "jiff-static" -version = "0.2.35" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a69dcb3a21cfb32ce1cd056169337ca284af0766dd766e7878819b251a49204" -dependencies = [ - "jiff-core", - "proc-macro2", - "quote", - "syn 2.0.118", -] - -[[package]] -name = "jiff-tzdb" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e" - -[[package]] -name = "jiff-tzdb-platform" -version = "0.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "875a5a69ac2bab1a891711cf5eccbec1ce0341ea805560dcd90b7a2e925132e8" -dependencies = [ - "jiff-tzdb", -] - -[[package]] -name = "js-sys" -version = "0.3.104" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" -dependencies = [ - "cfg-if", - "futures-util", - "wasm-bindgen", -] - -[[package]] -name = "k256" -version = "0.13.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6e3919bbaa2945715f0bb6d3934a173d1e9a59ac23767fbaaef277265a7411b" -dependencies = [ - "cfg-if", - "ecdsa", - "elliptic-curve", - "once_cell", - "sha2", -] - -[[package]] -name = "keccak" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d8f198d1db720e4940b5a493201d199d9f24f568f8f746bd13706243a2f71598" -dependencies = [ - "cfg-if", - "cpufeatures 0.3.1", -] - -[[package]] -name = "keccak-asm" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dd5dc2c0d691cbf7595cde551ced329cca99c2387c2cbc97754c5d0cd045d3ee" -dependencies = [ - "digest 0.10.7", - "sha3-asm", -] - -[[package]] -name = "konst" -version = "0.2.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "128133ed7824fcd73d6e7b17957c5eb7bacb885649bd8c69708b2331a10bcefb" -dependencies = [ - "konst_macro_rules", -] - -[[package]] -name = "konst_macro_rules" -version = "0.2.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4933f3f57a8e9d9da04db23fb153356ecaf00cbd14aee46279c33dc80925c37" - -[[package]] -name = "lazy_static" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" - -[[package]] -name = "lean_compiler" -version = "0.1.0" -dependencies = [ - "bincode", - "lean_vm", - "primitives", - "rand 0.9.4", -] - -[[package]] -name = "lean_da" -version = "0.1.0" -dependencies = [ - "fiat_shamir", - "parallel", - "pcs", - "primitives", - "rand 0.9.4", - "serde", - "tracing", -] - -[[package]] -name = "lean_vm" -version = "0.1.0" -dependencies = [ - "bincode", - "fiat_shamir", - "flock", - "lean_compiler", - "parallel", - "pcs", - "primitives", - "tracing", - "zk_alloc", -] - -[[package]] -name = "leanvm" -version = "0.1.0" -dependencies = [ - "clap", - "lean_da", - "lean_vm", - "primitives", - "rand 0.9.4", - "rec_aggregation", - "sphincs", - "xmss", - "zk_alloc", -] - -[[package]] -name = "libc" -version = "0.2.186" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" - -[[package]] -name = "libm" -version = "0.2.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" - -[[package]] -name = "log" -version = "0.4.33" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" - -[[package]] -name = "matchers" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d1525a2a28c7f4fa0fc98bb91ae755d1e2d1505079e05539e35bc876b5d65ae9" -dependencies = [ - "regex-automata", -] - -[[package]] -name = "memchr" -version = "2.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" - -[[package]] -name = "nu-ansi-term" -version = "0.50.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7957b9740744892f114936ab4a57b3f487491bbeafaf8083688b16841a4240e5" -dependencies = [ - "windows-sys", -] - -[[package]] -name = "num-bigint" -version = "0.4.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c89e69e7e0f03bea5ef08013795c25018e101932225a656383bd384495ecc367" -dependencies = [ - "num-integer", - "num-traits", -] - -[[package]] -name = "num-conv" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "521739c6d2bac4aa25192232afe6841231376b2b26d4d9fae5ecf8ca5772e441" - -[[package]] -name = "num-integer" -version = "0.1.47" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ce2d95d4b3734dc35aa2f45e1aa22cd416814592a4f9d9205e11affd5b8e10b" -dependencies = [ - "num-traits", -] - -[[package]] -name = "num-traits" -version = "0.2.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" -dependencies = [ - "autocfg", - "libm", -] - -[[package]] -name = "once_cell" -version = "1.21.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" - -[[package]] -name = "once_cell_polyfill" -version = "1.70.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" - -[[package]] -name = "parallel" -version = "0.1.0" -dependencies = [ - "libc", -] - -[[package]] -name = "parity-scale-codec" -version = "3.7.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "799781ae679d79a948e13d4824a40970bfa500058d245760dd857301059810fa" -dependencies = [ - "arrayvec", - "bitvec", - "byte-slice-cast", - "const_format", - "impl-trait-for-tuples", - "parity-scale-codec-derive", - "rustversion", - "serde", -] - -[[package]] -name = "parity-scale-codec-derive" -version = "3.7.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34b4653168b563151153c9e4c08ebed57fb8262bebfa79711552fa983c623e7a" -dependencies = [ - "proc-macro-crate", - "proc-macro2", - "quote", - "syn 2.0.118", -] - -[[package]] -name = "paste" -version = "1.0.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57c0d7b74b563b49d38dae00a0c37d4d6de9b432382b2892f0574ddcae73fd0a" - -[[package]] -name = "pcs" -version = "0.1.0" -dependencies = [ - "bincode", - "fiat_shamir", - "parallel", - "primitives", - "tracing", - "zk_alloc", -] - -[[package]] -name = "pest" -version = "2.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a07a60cc7a4d00c91f95c685609d1d2f79050e6804b70ebedd7650f0b839bcf" -dependencies = [ - "memchr", - "ucd-trie", -] - -[[package]] -name = "pin-project-lite" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" - -[[package]] -name = "pkcs8" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f950b2377845cebe5cf8b5165cb3cc1a5e0fa5cfa3e1f7f55707d8fd82e0a7b7" -dependencies = [ - "der", - "spki", -] - -[[package]] -name = "portable-atomic" -version = "1.15.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" - -[[package]] -name = "portable-atomic-util" -version = "0.2.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c2a106d1259c23fac8e543272398ae0e3c0b8d33c88ed73d0cc71b0f1d902618" -dependencies = [ - "portable-atomic", -] - -[[package]] -name = "powerfmt" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "439ee305def115ba05938db6eb1644ff94165c5ab5e9420d1c1bcedbba909391" - -[[package]] -name = "ppv-lite86" -version = "0.2.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" -dependencies = [ - "zerocopy", -] - -[[package]] -name = "primitive-types" -version = "0.12.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b34d9fd68ae0b74a41b21c03c2f62847aa0ffea044eee893b4c140b37e244e2" -dependencies = [ - "fixed-hash", - "impl-codec", - "uint", -] - -[[package]] -name = "primitives" -version = "0.1.0" -dependencies = [ - "bincode", - "libc", - "parallel", - "primitives", - "rand 0.9.4", - "serde", - "tracing-forest", - "tracing-subscriber", - "zk_alloc", -] - -[[package]] -name = "proc-macro-crate" -version = "3.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e67ba7e9b2b56446f1d419b1d807906278ffa1a658a8a5d8a39dcb1f5a78614f" -dependencies = [ - "toml_edit", -] - -[[package]] -name = "proc-macro2" -version = "1.0.106" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" dependencies = [ "unicode-ident", ] [[package]] -name = "proptest" -version = "1.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b45fcc2344c680f5025fe57779faef368840d0bd1f42f216291f0dc4ace4744" -dependencies = [ - "bitflags 2.13.1", - "num-traits", - "rand 0.9.4", - "rand_chacha 0.9.0", - "rand_xorshift", - "regex-syntax", - "unarray", -] - -[[package]] -name = "quote" -version = "1.0.46" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dfbc457d0c7a0759a614551b11a6409e5951f6c7537be1f1b7682b9ae9230368" -dependencies = [ - "proc-macro2", -] - -[[package]] -name = "r-efi" -version = "5.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" - -[[package]] -name = "radium" -version = "0.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09" - -[[package]] -name = "rand" -version = "0.8.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e058c7de0b26af77780c769414d6257830bb240f3c38477dbc2c16e5f54d6d4c" -dependencies = [ - "libc", - "rand_chacha 0.3.1", - "rand_core 0.6.4", -] - -[[package]] -name = "rand" -version = "0.9.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "44c5af06bb1b7d3216d91932aed5265164bf384dc89cd6ba05cf59a35f5f76ea" -dependencies = [ - "rand_chacha 0.9.0", - "rand_core 0.9.5", - "serde", -] - -[[package]] -name = "rand_chacha" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6c10a63a0fa32252be49d21e7709d4d4baf8d231c2dbce1eaa8141b9b127d88" -dependencies = [ - "ppv-lite86", - "rand_core 0.6.4", -] - -[[package]] -name = "rand_chacha" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" -dependencies = [ - "ppv-lite86", - "rand_core 0.9.5", -] - -[[package]] -name = "rand_core" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c" -dependencies = [ - "getrandom 0.2.17", -] - -[[package]] -name = "rand_core" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" -dependencies = [ - "getrandom 0.3.4", - "serde", -] - -[[package]] -name = "rand_xorshift" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "513962919efc330f829edb2535844d1b912b0fbe2ca165d613e4e8788bb05a5a" -dependencies = [ - "rand_core 0.9.5", -] - -[[package]] -name = "rapidhash" -version = "4.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5da7e78a036ce858e8d55b7e7dc8ba3a88b78350fd2155d3591bbd966b58589e" -dependencies = [ - "rustversion", -] - -[[package]] -name = "rec_aggregation" -version = "0.1.0" -dependencies = [ - "bincode", - "flock", - "lean_compiler", - "lean_da", - "lean_vm", - "parallel", - "pcs", - "primitives", - "rand 0.9.4", - "serde", - "sphincs", - "tracing", - "xmss", - "zk_alloc", -] - -[[package]] -name = "ref-cast" -version = "1.0.27" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e440fb4e4b4147295338efb76001ab9e4efc0e5839df2c47fc5ac2381d365c3" -dependencies = [ - "ref-cast-impl", -] - -[[package]] -name = "ref-cast-impl" -version = "1.0.27" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92ecd8964f8453721699a1ed72037b0db49ce2f5a5138486ee89bed6f67cdf3a" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.4", -] - -[[package]] -name = "regex-automata" -version = "0.4.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad" -dependencies = [ - "aho-corasick", - "memchr", - "regex-syntax", -] - -[[package]] -name = "regex-syntax" -version = "0.8.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" - -[[package]] -name = "rfc6979" -version = "0.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8dd2a808d456c4a54e300a23e9f5a67e122c3024119acbfd73e3bf664491cb2" -dependencies = [ - "hmac", - "subtle", -] - -[[package]] -name = "rlp" -version = "0.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb919243f34364b6bd2fc10ef797edbfa75f33c252e7998527479c6d6b47e1ec" -dependencies = [ - "bytes", - "rustc-hex", -] - -[[package]] -name = "ruint" -version = "1.20.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f5e99bff0393163bb25029a6af25d3d8d202ba5b5438a74d1bd8789f5c822970" -dependencies = [ - "alloy-rlp", - "ark-ff 0.3.0", - "ark-ff 0.4.2", - "ark-ff 0.5.0", - "ark-ff 0.6.0", - "bytes", - "fastrlp 0.3.1", - "fastrlp 0.4.0", - "num-bigint", - "num-integer", - "num-traits", - "parity-scale-codec", - "primitive-types", - "proptest", - "rand 0.8.8", - "rand 0.9.4", - "rlp", - "ruint-macro", - "serde_core", - "valuable", - "zeroize", -] - -[[package]] -name = "ruint-macro" -version = "1.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "48fd7bd8a6377e15ad9d42a8ec25371b94ddc67abe7c8b9127bec79bebaaae18" - -[[package]] -name = "rustc-hash" -version = "2.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" - -[[package]] -name = "rustc-hex" -version = "2.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e75f6a532d0fd9f7f13144f392b6ad56a32696bfcd9c78f797f16bbb6f072d6" - -[[package]] -name = "rustc_version" -version = "0.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0dfe2087c51c460008730de8b57e6a320782fbfb312e1f4d520e6c6fae155ee" -dependencies = [ - "semver 0.11.0", -] - -[[package]] -name = "rustc_version" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfcb3a22ef46e85b45de6ee7e79d063319ebb6594faafcf1c225ea92ab6e9b92" -dependencies = [ - "semver 1.0.28", -] - -[[package]] -name = "rustversion" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" - -[[package]] -name = "schemars" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4cd191f9397d57d581cddd31014772520aa448f65ef991055d7f61582c65165f" -dependencies = [ - "dyn-clone", - "ref-cast", - "serde", - "serde_json", -] - -[[package]] -name = "schemars" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "687274d293b6cdc6e73e0fee520bf2049650090d7164f87672d212a3c530cf4a" -dependencies = [ - "dyn-clone", - "ref-cast", - "serde", - "serde_json", -] - -[[package]] -name = "sec1" -version = "0.7.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3e97a565f76233a6003f9f5c54be1d9c5bdfa3eccfb189469f11ec4901c47dc" -dependencies = [ - "base16ct", - "der", - "generic-array", - "pkcs8", - "subtle", - "zeroize", -] - -[[package]] -name = "secp256k1" -version = "0.31.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2c3c81b43dc2d8877c216a3fccf76677ee1ebccd429566d3e67447290d0c42b2" -dependencies = [ - "bitcoin_hashes", - "rand 0.9.4", - "secp256k1-sys", -] - -[[package]] -name = "secp256k1-sys" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dcb913707158fadaf0d8702c2db0e857de66eb003ccfdda5924b5f5ac98efb38" -dependencies = [ - "cc", -] - -[[package]] -name = "semver" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f301af10236f6df4160f7c3f04eec6dbc70ace82d23326abad5edee88801c6b6" -dependencies = [ - "semver-parser", -] - -[[package]] -name = "semver" -version = "1.0.28" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" - -[[package]] -name = "semver-parser" -version = "0.10.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9900206b54a3527fdc7b8a938bffd94a568bac4f4aa8113b209df75a09c0dec2" -dependencies = [ - "pest", -] - -[[package]] -name = "serde" -version = "1.0.228" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" -dependencies = [ - "serde_core", - "serde_derive", -] - -[[package]] -name = "serde_core" -version = "1.0.228" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" -dependencies = [ - "serde_derive", -] - -[[package]] -name = "serde_derive" -version = "1.0.228" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.118", -] - -[[package]] -name = "serde_json" -version = "1.0.151" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" -dependencies = [ - "itoa", - "memchr", - "serde", - "serde_core", - "zmij", -] - -[[package]] -name = "serde_with" -version = "3.22.0" +name = "quote" +version = "1.0.46" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ee78f1fbe43ac4a0e47aadb3dbd357b69eb0d3793e948624cd03dd2750ab1c0a" +checksum = "dfbc457d0c7a0759a614551b11a6409e5951f6c7537be1f1b7682b9ae9230368" dependencies = [ - "base64", - "bs58", - "chrono", - "hex", - "indexmap 1.9.3", - "indexmap 2.14.1", - "jiff", - "schemars 0.9.0", - "schemars 1.2.2", - "serde_core", - "serde_json", - "time", + "proc-macro2", ] [[package]] -name = "sha2" -version = "0.10.9" +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + +[[package]] +name = "rand" +version = "0.9.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +checksum = "44c5af06bb1b7d3216d91932aed5265164bf384dc89cd6ba05cf59a35f5f76ea" dependencies = [ - "cfg-if", - "cpufeatures 0.2.17", - "digest 0.10.7", + "rand_chacha", + "rand_core", ] [[package]] -name = "sha3" -version = "0.11.0" +name = "rand_chacha" +version = "0.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "be176f1a57ce4e3d31c1a166222d9768de5954f811601fb7ca06fc8203905ce1" +checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" dependencies = [ - "digest 0.11.3", - "keccak", + "ppv-lite86", + "rand_core", ] [[package]] -name = "sha3-asm" -version = "0.1.8" +name = "rand_core" +version = "0.9.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a6287fd675f713484342a89cbf0a386abef5f15919cfad607e5e1f19e1e15331" +checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" dependencies = [ - "cc", - "cfg-if", + "getrandom", ] [[package]] -name = "sharded-slab" -version = "0.1.7" +name = "regex-automata" +version = "0.4.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f40ca3c46823713e0d4209592e8d6e826aa57e928f09752619fc696c499637f6" +checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad" dependencies = [ - "lazy_static", + "aho-corasick", + "memchr", + "regex-syntax", ] [[package]] -name = "shlex" -version = "2.0.1" +name = "regex-syntax" +version = "0.8.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" [[package]] -name = "signature" -version = "2.2.0" +name = "serde" +version = "1.0.228" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77549399552de45a898a580c1b41d445bf730df867cc44e6c0233bbc4b8329de" +checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" dependencies = [ - "digest 0.10.7", - "rand_core 0.6.4", + "serde_core", + "serde_derive", ] [[package]] -name = "slab" -version = "0.4.12" +name = "serde_core" +version = "1.0.228" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" +checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +dependencies = [ + "serde_derive", +] [[package]] -name = "smallvec" -version = "1.15.2" +name = "serde_derive" +version = "1.0.228" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" - -[[package]] -name = "sphincs" -version = "0.1.0" +checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" dependencies = [ - "parallel", - "primitives", - "rand 0.9.4", - "serde", + "proc-macro2", + "quote", + "syn", ] [[package]] -name = "spki" -version = "0.7.3" +name = "sharded-slab" +version = "0.1.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d91ed6c858b01f942cd56b37a94b3e0a1798290327d1236e4d9cf4eaca44d29d" +checksum = "f40ca3c46823713e0d4209592e8d6e826aa57e928f09752619fc696c499637f6" dependencies = [ - "base64ct", - "der", + "lazy_static", ] [[package]] -name = "static_assertions" -version = "1.1.0" +name = "smallvec" +version = "1.15.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2eb9349b6444b326872e140eb1cf5e7c522154d69e7a0ffb0fb81c06b37543f" +checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" [[package]] name = "strsim" @@ -2207,23 +425,6 @@ version = "0.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" -[[package]] -name = "subtle" -version = "2.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" - -[[package]] -name = "syn" -version = "1.0.109" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72b64191b275b66ffe2469e8af2c1cfe3bafa67b529ead792a6d0160888b4237" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - [[package]] name = "syn" version = "2.0.118" @@ -2235,23 +436,6 @@ dependencies = [ "unicode-ident", ] -[[package]] -name = "syn" -version = "3.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "tap" -version = "1.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "55937e1799185b12863d447f42597ed69d9928686b8d88a1df17376a097d8369" - [[package]] name = "thiserror" version = "2.0.18" @@ -2269,7 +453,7 @@ checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn", ] [[package]] @@ -2281,81 +465,6 @@ dependencies = [ "cfg-if", ] -[[package]] -name = "time" -version = "0.3.55" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" -dependencies = [ - "deranged", - "num-conv", - "powerfmt", - "serde_core", - "time-core", - "time-macros", -] - -[[package]] -name = "time-core" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9e1c906769ad99c88eaa54e728060edef082f8e358ff32030cb7c7d315e81109" - -[[package]] -name = "time-macros" -version = "0.2.32" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e689342a48d2ea927c87ea50cabf8594854bf940e9310208848d680d668ed85" -dependencies = [ - "num-conv", - "time-core", -] - -[[package]] -name = "tinyvec" -version = "1.12.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" -dependencies = [ - "tinyvec_macros", -] - -[[package]] -name = "tinyvec_macros" -version = "0.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" - -[[package]] -name = "toml_datetime" -version = "1.1.1+spec-1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3165f65f62e28e0115a00b2ebdd37eb6f3b641855f9d636d3cd4103767159ad7" -dependencies = [ - "serde_core", -] - -[[package]] -name = "toml_edit" -version = "0.25.13+spec-1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6975367e4d2ef766d86af01ffad14b622fecc8d4357a998fbc4deb6e9bacaf9b" -dependencies = [ - "indexmap 2.14.1", - "toml_datetime", - "toml_parser", - "winnow", -] - -[[package]] -name = "toml_parser" -version = "1.1.3+spec-1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d38ac1cf9b95face32296c0a3ede1fdc270627c9d9c02a7274dd6d960dc4d56" -dependencies = [ - "winnow", -] - [[package]] name = "tracing" version = "0.1.44" @@ -2375,7 +484,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", + "syn", ] [[package]] @@ -2430,54 +539,12 @@ dependencies = [ "tracing-log", ] -[[package]] -name = "typenum" -version = "1.20.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" - -[[package]] -name = "ucd-trie" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2896d95c02a80c6d6a5d6e953d479f5ddf2dfdb6a244441010e373ac0fb88971" - -[[package]] -name = "uint" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76f64bba2c53b04fcab63c01a7d7427eadc821e3bc48c34dc9ba29c501164b52" -dependencies = [ - "byteorder", - "crunchy", - "hex", - "static_assertions", -] - -[[package]] -name = "unarray" -version = "0.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eaea85b334db583fe3274d12b4cd1880032beab409c0d774be044d4480ab9a94" - [[package]] name = "unicode-ident" version = "1.0.24" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" -[[package]] -name = "unicode-segmentation" -version = "1.13.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6f5d3c3b1bf09027a88a6bc961fc00497d651009560b5463668dc81b0fa87a8" - -[[package]] -name = "unicode-xid" -version = "0.2.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853" - [[package]] name = "utf8parse" version = "0.2.2" @@ -2490,18 +557,6 @@ version = "0.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ba73ea9cf16a25df0c8caa16c51acb937d5712a8429db78a3ee29d5dcacd3a65" -[[package]] -name = "version_check" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" - -[[package]] -name = "wasi" -version = "0.11.1+wasi-snapshot-preview1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" - [[package]] name = "wasip2" version = "1.0.4+wasi-0.2.12" @@ -2511,51 +566,6 @@ dependencies = [ "wit-bindgen", ] -[[package]] -name = "wasm-bindgen" -version = "0.2.127" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" -dependencies = [ - "cfg-if", - "once_cell", - "rustversion", - "wasm-bindgen-macro", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-macro" -version = "0.2.127" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" -dependencies = [ - "quote", - "wasm-bindgen-macro-support", -] - -[[package]] -name = "wasm-bindgen-macro-support" -version = "0.2.127" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" -dependencies = [ - "bumpalo", - "proc-macro2", - "quote", - "syn 2.0.118", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-shared" -version = "0.2.127" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" -dependencies = [ - "unicode-ident", -] - [[package]] name = "winapi" version = "0.3.9" @@ -2578,65 +588,12 @@ version = "0.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f" -[[package]] -name = "windows-core" -version = "0.62.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" -dependencies = [ - "windows-implement", - "windows-interface", - "windows-link", - "windows-result", - "windows-strings", -] - -[[package]] -name = "windows-implement" -version = "0.60.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.118", -] - -[[package]] -name = "windows-interface" -version = "0.59.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.118", -] - [[package]] name = "windows-link" version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" -[[package]] -name = "windows-result" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-strings" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" -dependencies = [ - "windows-link", -] - [[package]] name = "windows-sys" version = "0.61.2" @@ -2646,42 +603,12 @@ dependencies = [ "windows-link", ] -[[package]] -name = "winnow" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" -dependencies = [ - "memchr", -] - [[package]] name = "wit-bindgen" version = "0.57.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" -[[package]] -name = "wyz" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05f360fc0b24296329c78fda852a1e9ae82de9cf7b27dae4b7f62f118f77b9ed" -dependencies = [ - "tap", -] - -[[package]] -name = "xmss" -version = "0.1.0" -dependencies = [ - "bincode", - "ethereum_ssz", - "parallel", - "primitives", - "rand 0.9.4", - "serde", -] - [[package]] name = "zerocopy" version = "0.8.52" @@ -2699,27 +626,7 @@ checksum = "1ae7f38b72ec2a254e2b87ef277cf2cd4fb97cbebf944faa6f33354da0867930" dependencies = [ "proc-macro2", "quote", - "syn 2.0.118", -] - -[[package]] -name = "zeroize" -version = "1.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" -dependencies = [ - "zeroize_derive", -] - -[[package]] -name = "zeroize_derive" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c50655cbb0fe3fc43170059e702f1ce5e19b84cec58dc87b037a09935c2f328" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.118", + "syn", ] [[package]] @@ -2728,9 +635,3 @@ version = "0.1.0" dependencies = [ "libc", ] - -[[package]] -name = "zmij" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/Cargo.toml b/Cargo.toml index cb60b7e46..e2d84e385 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,13 +12,10 @@ workspace = true [dependencies] primitives.workspace = true -rec_aggregation.workspace = true lean_vm.workspace = true -lean_da.workspace = true -xmss.workspace = true -sphincs.workspace = true zk_alloc.workspace = true -rand.workspace = true +bincode.workspace = true +tracing.workspace = true clap = { version = "4", features = ["derive"] } [workspace.package] @@ -39,17 +36,11 @@ fiat_shamir = { path = "crates/fiat_shamir" } pcs = { path = "crates/pcs" } flock = { path = "crates/flock" } lean_vm = { path = "crates/lean_vm" } -lean_compiler = { path = "crates/lean_compiler" } -lean_da = { path = "crates/lean_da" } -rec_aggregation = { path = "crates/rec_aggregation" } -xmss = { path = "crates/xmss" } -sphincs = { path = "crates/sphincs" } zk_alloc = { path = "crates/zk_alloc" } parallel = { path = "crates/parallel" } libc = "0.2" serde = { version = "1", features = ["derive"] } bincode = "1" -ethereum_ssz = "0.10" rand = "0.9" tracing = "0.1.26" tracing-forest = { version = "0.3.0", features = ["ansi", "smallvec"] } diff --git a/README.md b/README.md index a6070c30b..d424508f1 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ leanVM

-

minimal hash-based zkVM, for post-quantum Ethereum

+

minimal hash-based zkVM

Documentation @@ -12,31 +12,13 @@

- - - - - - - - - - - - -
leanXMSS aggregation1.65K/s
leanSPHINCS aggregation295/s
leanDA commitment2 MiB/s
- - - - - - - + +
2-to-1 recursion0.27s
hash compressions 480K/s
cheap cycles6,2M/sRISC-V cycles1.6M/s
@@ -56,82 +38,76 @@ leanVM is designed for security: Expect leanVM to change significantly: * **hash**: BLAKE2s is a placeholder. SHA2, SHA3, BLAKE3 are actively considered. -* **ISA**: A migration from leanISA to RISC-V (rv64im) is planned. +* **ISA**: leanVM proves RISC-V (rv64im) plus one custom instruction, the BLAKE2s compression. A run is one proof; continuations, for runs whose witness exceeds one commitment, are planned. * **zk**: Support for zero-knowledge is planned. **note**: Prior to binary fields leanVM used [KoalaBear](https://crates.io/crates/p3-koala-bear) and [Poseidon](https://eprint.iacr.org/2019/458). The historical design is in [this branch](https://github.com/leanEthereum/leanVM/tree/koalabear). +## guests + +A guest is a `no_std` Rust program built for `riscv64im-unknown-none-elf` against the runtime crate in [`guests/rt`](./guests/rt/src/lib.rs), which gives it its public input (four words), its advice (a region of memory the prover fills, which the statement says nothing about), its output (four words) and a BLAKE2s hasher over the custom instruction. The linker script fixes the memory map. Build them with `guests/build.sh` (a nightly toolchain, for `-Zbuild-std`), then prove and verify a run: + +```bash +cargo run --release -- guest guests/elf/preimage.elf --advice 5,0x6f6c6c6568 +``` + +The statement a proof makes is the program (an ELF file), the four input words and the four output words; everything a guest reads from its advice it has to check itself, which is what makes a proof a proof of knowledge (`preimage` outputs the digest of a message only the prover has). + ## benchmarks **machine**: M4 Max MacBook Pro (12 performance cores, 4 efficiency cores, 48GB RAM) **note**: The Metal GPU was not used. -### XMSS aggregation - -The XMSS parameters are specified in [XMSS.pdf](https://github.com/leanEthereum/leanVM/releases/download/doc-latest/XMSS.pdf), with a [(ROM) security proof in Lean 4](https://github.com/leanEthereum/leanMultisig/blob/main/formal/xmss/XmssSecurity/Statement.lean). +### Fibonacci ```bash -cargo run --release -- aggregate --xmss 900 --log-inv-rate 1 --repeat 3 +cargo run --release -- fibonacci --n 2000000 --log-inv-rate 1 --repeat 3 ``` ``` -aggregation, 900 XMSS signatures - cycles (VM steps) : 995,578 = 2^19.925 - details : DEREF 2^17.878 (24.2%) MUL 2^17.689 (21.2%) SET 2^17.425 (17.7%) BLAKE2S 2^16.989 (13.1%) XOR 2^16.979 (13.0%) JUMP 2^16.722 (10.9%) MEMORY 2^20.674 BYTECODE 2^17.737 TOTAL_COMMITTED 2^25.801 - proof size : 317.4 KiB - proving time : 0.545 s ± 1.5% peak memory 7.385 GiB - per signature : 1,651.687 signatures/s - verifying : 3.977 ms +Fibonacci (modulo 2^64), N = 2,000,000 + cycles (VM steps) : 2,097,208 + details : ALU 2^20.934 (100.0%) TOTAL_COMMITTED 2^26.395 + proof size : 337.5 KiB + proving : 1.281 s ± 1.2% 1,636,564 cycles/s peak memory 11.6 GiB + verifying : 6.186 ms ``` -### SPHINCS aggregation +### BLAKE2s in plain Rust -The SPHINCS parameters are specified in [SPHINCS.pdf](https://github.com/leanEthereum/leanVM/releases/download/doc-latest/SPHINCS.pdf), with a [(ROM) security proof in Lean 4](https://github.com/leanEthereum/leanMultisig/blob/main/formal/sphincs/SphincsSecurity/Statement.lean). +The `blake2s` guest is the hash function written in ordinary Rust, compiled by `rustc` for `riscv64im-unknown-none-elf` (`guests/blake2s`): 10,000 bytes, 157 compressions, a mix of arithmetic, shifts, loads and stores. ```bash -cargo run --release -- aggregate --sphincs 245 --log-inv-rate 1 --repeat 3 +cargo run --release -- guest guests/elf/blake2s.elf --input 10000 --repeat 3 --cooldown 2 ``` ``` -aggregation, 245 SPHINCS signatures - cycles (VM steps) : 2,131,611 = 2^21.024 - details : XOR 2^18.951 (23.8%) MUL 2^18.93 (23.4%) SET 2^18.845 (22.1%) DEREF 2^18.711 (20.1%) BLAKE2S 2^16.992 (6.1%) JUMP 2^16.543 (4.5%) MEMORY 2^21.46 BYTECODE 2^17.737 TOTAL_COMMITTED 2^26.301 - proof size : 299.9 KiB - proving time : 0.831 s ± 3.6% peak memory 9.275 GiB - per signature : 294.783 signatures/s - verifying : 3.644 ms +guests/elf/blake2s.elf + input : [2710, 0, 0, 0] + output : [8f9fc3d71d84c0cc, 515c979fa65679e8, 9ffc0e1e022efcc7, cef54d0c06836e56] + cycles (VM steps) : 1,015,824 + details : ALU 2^18.47 (55.2%) SHIFT 2^17.238 (23.5%) LOAD 2^16.356 (12.8%) STORE 2^15.139 (5.5%) MUL 2^13.288 (1.5%) MULH 2^13.288 (1.5%) TOTAL_COMMITTED 2^25.435 + proof size : 328.6 KiB + proving : 0.698 s ± 1.2% 1,454,472 cycles/s peak memory 5.18 GiB + verifying : 7.185 ms ``` -### data availability +### BLAKE2s through the precompile -```bash -cargo run --release -- aggregate --blobs 16 --log-inv-rate 1 --repeat 3 -``` - -``` -aggregation, 16 blobs - cycles (VM steps) : 2,989,506 = 2^21.511 - details : MUL 2^19.899 (32.7%) XOR 2^19.809 (30.7%) DEREF 2^19.138 (19.3%) JUMP 2^18.299 (10.8%) SET 2^16.787 (3.8%) BLAKE2S 2^16.295 (2.7%) MEMORY 2^21.695 BYTECODE 2^17.737 TOTAL_COMMITTED 2^26.695 - proof size : 324.0 KiB - proving time : 0.986 s ± 11.1% peak memory 12.534 GiB - blob throughput : 16.235 blobs/s, 2.029 MiB/s - verifying : 6.573 ms -``` - -### recursion +The `hash` guest hashes 50,000 bytes through the compression instruction, 782 compressions; most of its cycles generate the message. ```bash -cargo run --release -- recursion --n 2 --xmss-per-leaf 900 --log-inv-rate 2 --repeat 3 +cargo run --release -- guest guests/elf/hash.elf --input 50000 --repeat 3 ``` ``` -recursion 2→1, over leaves of 900 XMSS signatures - cycles (VM steps) : 562,737 = 2^19.102 - details : MUL 2^17.823 (41.2%) DEREF 2^16.964 (22.7%) XOR 2^16.728 (19.3%) SET 2^15.77 (9.9%) JUMP 2^14.486 (4.1%) BLAKE2S 2^13.932 (2.8%) MEMORY 2^19.481 BYTECODE 2^17.737 TOTAL_COMMITTED 2^24.086 - proof size : 190.2 KiB - proving time : 0.272 s ± 5.3% peak memory 8.388 GiB - verifying : 3.457 ms +guests/elf/hash.elf + cycles (VM steps) : 869,384 + details : ALU 2^18.641 (59.8%) SHIFT 2^16.61 (14.6%) STORE 2^15.915 (9.0%) MULH 2^15.61 (7.3%) MUL 2^15.61 (7.3%) LOAD 2^13.618 (1.8%) HASH 2^9.611 (0.1%) TOTAL_COMMITTED 2^25.49 + proof size : 331.0 KiB + proving : 0.816 s ± 0.9% 1,065,522 cycles/s peak memory 5.238 GiB + verifying : 7.796 ms ``` ### hashing @@ -143,32 +119,16 @@ BENCH_REPEAT=3 BENCH_COOLDOWN=2 FLOCK_N_LOG=18 cargo test --release --package fl ``` Flock BLAKE2s batch proving, 262,144 compressions (2^18 slots) setup (preprocessing, excluded) : 0.0 ms - witness-gen : 34.8 ms ± 26.8% 6.1% - commit : 102.2 ms ± 1.3% 17.8% - zerocheck : 244.9 ms ± 7.5% 42.6% - lincheck : 20.2 ms ± 3.4% 3.5% - pcs opening : 172.3 ms ± 1.3% 30.0% + witness-gen : 64.6 ms ± 7.8% 10.6% + commit : 101.2 ms ± 0.4% 16.6% + zerocheck : 238.3 ms ± 3.9% 39.0% + lincheck : 20.3 ms ± 12.2% 3.3% + pcs opening : 186.0 ms ± 2.9% 30.5% other : 0.0 ms 0.0% ------------------------------------------ - prove TOTAL (witness excluded) : 539.7 ms ± 3.3% 93.9% - verify : 2.0 ms - throughput : 485,765 compressions/s ± 3.3% - (~3327.2 XMSS/s equivalent at 146 compressions/signature) -``` - -### Fibonacci - -```bash -cargo run --release -- fibonacci --n 2000000 --log-inv-rate 1 --repeat 3 -``` - -``` -Fibonacci (in the exponent, i.e. modulo 2^64 - 1), N = 2,000,000 - cycles (VM steps) : 2,127,880 - details : MUL 2^20.944 (98.9%) SET 2^13.288 (0.5%) DEREF 2^12.967 (0.4%) JUMP 2^10.968 (0.1%) XOR 2^10.966 (0.1%) MEMORY 2^20.96 BYTECODE 2^11.352 TOTAL_COMMITTED 2^25.26 - proof size : 286.0 KiB - proving : 0.344 s ± 4.6% 6,181,087 cycles/s peak memory 5.199 GiB - verifying : 2.218 ms + prove TOTAL (witness excluded) : 545.8 ms ± 1.1% 89.4% + verify : 1.9 ms + throughput : 480,319 compressions/s ± 1.1% ``` ## SNARK machinery diff --git a/conformance/act4/Dockerfile b/conformance/act4/Dockerfile new file mode 100644 index 000000000..130f9069b --- /dev/null +++ b/conformance/act4/Dockerfile @@ -0,0 +1,54 @@ +# The environment that generates the ACT4 ELF files (generate.sh builds and runs it): +# riscv-arch-test, its RISC-V GCC, the Sail reference model and the tools of its +# framework, every one pinned by version and checksum. The configuration is mounted +# at run time, so a change to it does not rebuild the image. +FROM ubuntu:24.04@sha256:008173c23f95b170204355c12626cb5a965d779a7e1283b09e9cffbb1bf33ca3 + +ARG DEBIAN_FRONTEND=noninteractive +# build-essential builds the native gems of the unified database (UDB). +RUN apt-get update && apt-get install -y --no-install-recommends \ + build-essential ca-certificates curl git xz-utils \ + && rm -rf /var/lib/apt/lists/* + +# GCC 15.1 and binutils 2.45: ACT4 4.1.0 needs GCC 15 or later. +ARG TOOLCHAIN=2025.08.08 +ARG TOOLCHAIN_SHA256=2aaa09d5eb768d4874b85cba152b994f49605b13035fc8d6a71f14c51db5276e +RUN curl -fsSL -o /tmp/toolchain.tar.xz \ + "https://github.com/riscv-collab/riscv-gnu-toolchain/releases/download/${TOOLCHAIN}/riscv64-elf-ubuntu-24.04-gcc-nightly-${TOOLCHAIN}-nightly.tar.xz" \ + && echo "${TOOLCHAIN_SHA256} /tmp/toolchain.tar.xz" | sha256sum -c - \ + && tar -xJf /tmp/toolchain.tar.xz -C /opt \ + && rm /tmp/toolchain.tar.xz + +# ACT4 4.1.0 requires exactly Sail 0.13.1. +ARG SAIL=0.13.1 +ARG SAIL_SHA256=ee052f64494a2f5f071afd9c2cb4aa5eaae4ba84753e4f77e442b4f83f2e9469 +RUN curl -fsSL -o /tmp/sail.tar.gz \ + "https://github.com/riscv/sail-riscv/releases/download/${SAIL}/sail-riscv-Linux-x86_64.tar.gz" \ + && echo "${SAIL_SHA256} /tmp/sail.tar.gz" | sha256sum -c - \ + && mkdir /opt/sail \ + && tar -xzf /tmp/sail.tar.gz -C /opt/sail --strip-components=1 \ + && rm /tmp/sail.tar.gz + +# mise installs the uv, Ruby and Bundler versions riscv-arch-test's .mise.toml pins. +ARG MISE=2026.9.17 +ARG MISE_SHA256=63049bc35fb9065e8dc35ac8b25fdae53e9bd6f1885a843aedeba398e046a1ee +RUN curl -fsSL -o /usr/local/bin/mise \ + "https://github.com/jdx/mise/releases/download/v${MISE}/mise-v${MISE}-linux-x64" \ + && echo "${MISE_SHA256} /usr/local/bin/mise" | sha256sum -c - \ + && chmod +x /usr/local/bin/mise + +ENV PATH="/opt/riscv/bin:/opt/sail/bin:/root/.local/share/mise/shims:${PATH}" \ + MISE_YES=1 + +# riscv-arch-test 4.1.0, and its Ruby and Python dependencies (locked by its +# Gemfile.lock and uv.lock), so that generating needs no network. +ARG ARCH_TEST=6e8a45123f14cebfb3df151a0e7b849b4389b33b +WORKDIR /act4 +RUN git init -q . \ + && git fetch -q --depth 1 https://github.com/riscv/riscv-arch-test.git "${ARCH_TEST}" \ + && git checkout -q FETCH_HEAD \ + && mise trust \ + && mise install ruby gem:bundler uv \ + && BUNDLE_GEMFILE=/act4/framework/src/act/data/Gemfile bundle install \ + && uv sync --frozen \ + && riscv64-unknown-elf-gcc --version && sail_riscv_sim --version diff --git a/conformance/act4/generate.sh b/conformance/act4/generate.sh new file mode 100755 index 000000000..7bdb2e449 --- /dev/null +++ b/conformance/act4/generate.sh @@ -0,0 +1,18 @@ +#!/bin/sh +# Generate ACT4's self-checking I and M tests for leanVM into elf/ (elf/I, elf/M; not +# checked in), which lean_vm/tests/verifiers/act4.rs loads. `generate.sh DIR` writes them to DIR +# instead. Needs Docker, and network access to build the image (Dockerfile) that +# pins every tool; ACT4_IMAGE names an image already built from it instead. +set -eu +here=$(cd "$(dirname "$0")" && pwd) +out=${1:-$here/elf} +image=${ACT4_IMAGE:-} +if [ -z "$image" ]; then + image=leanvm-act4 + docker build -t "$image" - <"$here/Dockerfile" +fi + +mkdir -p "$out" +out=$(cd "$out" && pwd) +rm -rf "$out/I" "$out/M" +docker run --rm -v "$here:/config:ro" -v "$out:/out" -e OWNER="$(id -u):$(id -g)" "$image" sh /config/in-docker.sh diff --git a/conformance/act4/in-docker.sh b/conformance/act4/in-docker.sh new file mode 100755 index 000000000..29061192c --- /dev/null +++ b/conformance/act4/in-docker.sh @@ -0,0 +1,18 @@ +#!/bin/sh +# generate.sh's work inside the image: ACT4 builds the tests (Sail running each one +# to record the expected values the self-checking build embeds) into /out. +set -eu +# Of the other tests ACT4 would select for this configuration, Zicsr's need CSRs, which +# leanVM does not have (the configuration claims Zicsr only for UDB to define MXLEN), +# and Zmmul's are M's multiplication tests again. +/act4/.venv/bin/act /config/test_config.yaml --workdir /act4/work --test-dir tests --extensions I,M --fast +cp -r /act4/work/leanvm-rv64im/elfs/rv64i/I /act4/work/leanvm-rv64im/elfs/rv64i/M /out/ +for elf in /out/I/*.elf /out/M/*.elf; do + # The symbol naming the compiler's temporary object file, whose name is random. + riscv64-unknown-elf-objcopy --wildcard --strip-symbol='cc*.o' "$elf" + # ACT4 switches compressed instructions on around its alignment padding, which marks + # the file as using them (e_flags = EF_RISCV_RVC) though no instruction is compressed. + test "$(od -An -tu4 -j48 -N4 "$elf" | tr -d ' ')" = 1 + printf '\000' | dd of="$elf" bs=1 seek=48 conv=notrunc status=none +done +chown -R "$OWNER" /out/I /out/M diff --git a/conformance/act4/leanvm-rv64im.yaml b/conformance/act4/leanvm-rv64im.yaml new file mode 100644 index 000000000..439696369 --- /dev/null +++ b/conformance/act4/leanvm-rv64im.yaml @@ -0,0 +1,55 @@ +# yaml-language-server: $schema=https://raw.githubusercontent.com/riscv/riscv-unified-db/main/spec/schemas/config_schema.json +--- +$schema: config_schema.json# +kind: architecture configuration +type: fully configured +name: leanvm-rv64im +description: leanVM, RV64IM with no misaligned accesses + +implemented_extensions: + - { name: I, version: "= 2.1" } + - { name: M, version: "= 2.0" } + - { name: Zmmul, version: "= 1.0.0" } + # UDB defines MXLEN through Sm, which needs Zicsr. leanVM has neither: rvmodel_macros.h + # leaves STANDARD_SM_SUPPORTED undefined, so no test code touches a CSR. + - { name: Zicsr, version: "= 2.0" } + - { name: Sm, version: "= 1.12.0" } + +params: + MUTABLE_MISA_M: false + MXLEN: 64 + PRECISE_SYNCHRONOUS_EXCEPTIONS: true + TRAP_ON_ECALL_FROM_M: true + TRAP_ON_EBREAK: true + MARCHID_IMPLEMENTED: false + MIMPID_IMPLEMENTED: false + VENDOR_ID_BANK: 0x0 + VENDOR_ID_OFFSET: 0x0 + MISALIGNED_LDST: false + MISALIGNED_LDST_EXCEPTION_PRIORITY: high + TRAP_ON_ILLEGAL_WLRL: false + TRAP_ON_UNIMPLEMENTED_INSTRUCTION: true + TRAP_ON_RESERVED_INSTRUCTION: true + TRAP_ON_UNIMPLEMENTED_CSR: true + REPORT_VA_IN_MTVAL_ON_BREAKPOINT: false + REPORT_VA_IN_MTVAL_ON_LOAD_MISALIGNED: false + REPORT_VA_IN_MTVAL_ON_STORE_AMO_MISALIGNED: false + REPORT_VA_IN_MTVAL_ON_INSTRUCTION_MISALIGNED: false + REPORT_VA_IN_MTVAL_ON_LOAD_ACCESS_FAULT: false + REPORT_VA_IN_MTVAL_ON_STORE_AMO_ACCESS_FAULT: false + REPORT_VA_IN_MTVAL_ON_INSTRUCTION_ACCESS_FAULT: false + REPORT_ENCODING_IN_MTVAL_ON_ILLEGAL_INSTRUCTION: false + MTVAL_WIDTH: 64 + CONFIG_PTR_ADDRESS: 0 + PMA_GRANULARITY: 3 + PHYS_ADDR_WIDTH: 64 + M_MODE_ENDIANNESS: little + MISA_CSR_IMPLEMENTED: false + MTVEC_ACCESS: rw + MTVEC_MODES: [0] + MTVEC_BASE_ALIGNMENT_DIRECT: 4 + MTVEC_ILLEGAL_WRITE_BEHAVIOR: retain + NUM_PMP_ENTRIES: 0 + MCOUNTINHIBIT_IMPLEMENTED: false + HPM_COUNTER_EN: [false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false] + MCOUNTENABLE_EN: [false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false, false] diff --git a/conformance/act4/link.ld b/conformance/act4/link.ld new file mode 100644 index 000000000..0061687f6 --- /dev/null +++ b/conformance/act4/link.ld @@ -0,0 +1,46 @@ +/* ACT4's test layout on leanVM's memory map (guests/link.ld): the code in the text, + every other section in RAM past the input words, and one word of advice. Sail runs + the same file on the regions sail.json describes, so the two agree on every address. + RAM is the smallest power of two holding the image, the loader reading its size off + __stack_top: the tests use no stack. */ +OUTPUT_ARCH("riscv") +ENTRY(rvtest_entry_point) + +MEMORY { + TEXT (rx) : ORIGIN = 0x10000000, LENGTH = 256M + ADVICE (rw) : ORIGIN = 0x20000000, LENGTH = 8 + RAM (rw) : ORIGIN = 0x40000000, LENGTH = 1M +} + +PROVIDE(__stack_size = 0); +PROVIDE(__num_harts = 1); + +SECTIONS { + /* Separate output sections, so that a test's alignment does not move the entry. The + model's code comes last: it differs between the self-checking build and Sail's. */ + .text.init ORIGIN(TEXT) : { *(.text.init) } > TEXT + .text.rvtest : { *(.text.rvtest) *(.text.rvtest.*) } > TEXT + .text.rvmodel : { *(.text.rvmodel) *(.text.rvmodel.*) *(.text) *(.text.*) } > TEXT + + /* RAM's first four words are the run's public input: nothing is loaded there, and a + section reserving them would make GNU ld write zeros there into the data segment. */ + .data ORIGIN(RAM) + 32 : { + *(.rodata) *(.rodata.*) *(.srodata) *(.srodata.*) + *(.data) *(.data.*) *(.sdata) *(.sdata.*) + } > RAM + /* Sail's tohost, which the self-checking build does not have. */ + .tohost : { *(.tohost) } > RAM + .bss (NOLOAD) : { + __bss_start = .; + *(.sbss) *(.sbss.*) *(.bss) *(.bss.*) *(COMMON) + __bss_end = .; + } > RAM + + __stack_bottom = .; + _end = .; + __stack_top = ORIGIN(RAM) + (1 << LOG2CEIL(_end - ORIGIN(RAM))); + __advice_top = ORIGIN(ADVICE) + LENGTH(ADVICE); + ASSERT(__stack_top <= ORIGIN(RAM) + LENGTH(RAM), "the test does not fit RAM: grow RAM here and in sail.json") + + /DISCARD/ : { *(.comment) } +} diff --git a/conformance/act4/rvmodel_macros.h b/conformance/act4/rvmodel_macros.h new file mode 100644 index 000000000..51cfecd01 --- /dev/null +++ b/conformance/act4/rvmodel_macros.h @@ -0,0 +1,47 @@ +// ACT4's model interface for leanVM. The machine has no CSRs, no traps, no console +// and no tohost: a test ends in the one ecall it knows, exit (a7 = 93), with the +// output a0..a3. A pass is all zeros. A failure that reaches the model's halt is +// a0 = 1 and a1 = the return address of the call to it, which names the check. +// STANDARD_SM_SUPPORTED stays undefined, so ACT4 emits no CSR code. +// SPDX-License-Identifier: BSD-3-Clause + +#ifndef _RVMODEL_MACROS_H +#define _RVMODEL_MACROS_H + +// Only Sail's build has a tohost, which sail_macros.h defines. +#define RVMODEL_DATA_SECTION + +// Nothing traps, so the trap signature region, 15000 entries by default, holds nothing. +#define TRAP_SIGUPD_COUNT 0 + +#define RVMODEL_HALT_EXIT \ + li a2, 0 ;\ + li a3, 0 ;\ + li a7, 93 ;\ + ecall ; + +#define RVMODEL_HALT_PASS \ + li a0, 0 ;\ + li a1, 0 ;\ + RVMODEL_HALT_EXIT + +#define RVMODEL_HALT_FAIL \ + li a0, 1 ;\ + mv a1, ra ;\ + RVMODEL_HALT_EXIT + +#define RVMODEL_IO_WRITE_STR(_R1, _R2, _R3, _STR_PTR) + +// Required by check_defines.h, expanded only with STANDARD_SM_SUPPORTED. +#define RVMODEL_INTERRUPT_LATENCY 10 +#define RVMODEL_TIMER_INT_SOON_DELAY 100 +#define RVMODEL_SET_MEXT_INT(_R1, _R2) +#define RVMODEL_CLR_MEXT_INT(_R1, _R2) +#define RVMODEL_SET_MSW_INT(_R1, _R2) +#define RVMODEL_CLR_MSW_INT(_R1, _R2) +#define RVMODEL_SET_SEXT_INT(_R1, _R2) +#define RVMODEL_CLR_SEXT_INT(_R1, _R2) +#define RVMODEL_SET_SSW_INT(_R1, _R2) +#define RVMODEL_CLR_SSW_INT(_R1, _R2) + +#endif // _RVMODEL_MACROS_H diff --git a/conformance/act4/sail.json b/conformance/act4/sail.json new file mode 100644 index 000000000..2b94d9ddf --- /dev/null +++ b/conformance/act4/sail.json @@ -0,0 +1,595 @@ +{ + "$schema": "/opt/sail/share/sail-riscv/sail_riscv_config_schema.json", + "base": { + "xlen": 64, + "E": false, + "writable_misa": false, + "writable_fiom": false, + "writable_hpm_counters": { + "len": 32, + "value": "0x0" + }, + "scounteren_writable_bits": { + "len": 32, + "value": "0x0" + }, + "mcounteren_writable_bits": { + "len": 32, + "value": "0x0" + }, + "mtvec": { + "direct": { + "supported": true, + "base_alignment": 2 + }, + "vectored": { + "supported": false, + "base_alignment": 2 + } + }, + "stvec": { + "direct": { + "supported": false + }, + "vectored": { + "supported": false, + "base_alignment": 2 + } + }, + "medeleg": { + "delegatable_bits": { + "len": 64, + "value": "0x0000_0000_000c_b3FF" + } + }, + "mideleg": { + "delegatable_bits": { + "len": "xlen", + "value": "0x0000_0000_0000_2222" + } + }, + "xtval_nonzero": { + "illegal_instruction": false, + "software_breakpoint": false, + "hardware_breakpoint": false, + "load_address_misaligned": false, + "load_access_fault": false, + "load_page_fault": false, + "samo_address_misaligned": false, + "samo_access_fault": false, + "samo_page_fault": false, + "fetch_address_misaligned": false, + "fetch_access_fault": false, + "fetch_page_fault": false, + "software_check": false, + "reserved_exceptions": false + }, + "reserved_behavior": { + "amocas_odd_register": "AMOCAS_Illegal", + "fcsr_rm": "Fcsr_RM_Illegal", + "pmpcfg_write_only": "PMP_ClearPermissions", + "xenvcfg_cbie": "Xenvcfg_ClearPermissions", + "xtvec_mode": "Xtvec_Ignore", + "rv32zdinx_odd_register": "Zdinx_Illegal" + }, + "mstatus": { + "fs_legal_states": "ExtContext_Off", + "vs_legal_states": "ExtContext_Off" + }, + "privileged_isa_version": "Privileged_ISA_1_12" + }, + "memory": { + "physaddr_bits": 64, + "pmp": { + "grain": 0, + "count": 0, + "usable_count": 0, + "tor_supported": false, + "na4_supported": false, + "napot_supported": false + }, + "misaligned": { + "exceptions": { + "load_store": { + "Some": "AlignmentException" + }, + "vector": { + "None": null + }, + "amo": { + "Some": "AccessFault" + }, + "lrsc": "AccessFault" + }, + "order_decreasing": false, + "default_allowed_within_exp": 0, + "byte_by_byte": true + }, + "dtb_address": { + "len": 64, + "value": "0x0" + }, + "regions": [ + { + "base": { + "len": 64, + "value": "0x2000000" + }, + "size": { + "len": 64, + "value": "0x100000" + }, + "attributes": { + "mem_type": "IOMemory", + "cacheable": false, + "coherent": true, + "executable": false, + "readable": true, + "writable": true, + "read_idempotent": false, + "write_idempotent": false, + "misaligned_exceptions": { + "load_store": { + "Some": "AlignmentException" + }, + "vector": { + "None": null + }, + "amo": "AccessFault" + }, + "atomic_support": "AMONone", + "misaligned_atomicity_granule_size_exp": 0, + "vector_misaligned_atomicity_granule_size_exp": 0, + "reservability": "RsrvNone", + "supports_cbo_zero": false, + "supports_pte_read": false, + "supports_pte_write": false + }, + "include_in_device_tree": false + }, + { + "base": { + "len": 64, + "value": "0x10000000" + }, + "size": { + "len": 64, + "value": "0x10000000" + }, + "attributes": { + "mem_type": "MainMemory", + "cacheable": true, + "coherent": true, + "executable": true, + "readable": true, + "writable": true, + "read_idempotent": true, + "write_idempotent": true, + "misaligned_exceptions": { + "load_store": { + "Some": "AlignmentException" + }, + "vector": { + "None": null + }, + "amo": "AccessFault" + }, + "atomic_support": "AMONone", + "misaligned_atomicity_granule_size_exp": 0, + "vector_misaligned_atomicity_granule_size_exp": 0, + "reservability": "RsrvNone", + "supports_cbo_zero": false, + "supports_pte_read": false, + "supports_pte_write": false + }, + "include_in_device_tree": false + }, + { + "base": { + "len": 64, + "value": "0x40000000" + }, + "size": { + "len": 64, + "value": "0x100000" + }, + "attributes": { + "mem_type": "MainMemory", + "cacheable": true, + "coherent": true, + "executable": false, + "readable": true, + "writable": true, + "read_idempotent": true, + "write_idempotent": true, + "misaligned_exceptions": { + "load_store": { + "Some": "AlignmentException" + }, + "vector": { + "None": null + }, + "amo": "AccessFault" + }, + "atomic_support": "AMONone", + "misaligned_atomicity_granule_size_exp": 0, + "vector_misaligned_atomicity_granule_size_exp": 0, + "reservability": "RsrvNone", + "supports_cbo_zero": false, + "supports_pte_read": false, + "supports_pte_write": false + }, + "include_in_device_tree": false + } + ] + }, + "platform": { + "vendorid": 0, + "archid": 0, + "impid": 0, + "hartid": 0, + "cache_block_size_exp": 6, + "reservation": { + "reservation_set_size_exp": 3, + "require_exact_reservation_addr": false, + "invalidate_on_same_hart_store": false + }, + "clint": { + "supported": true, + "base": 33554432, + "size": 786432 + }, + "simple_interrupt_generator": { + "supported": true, + "base": 34340864 + }, + "clock_frequency": 1000000000, + "instructions_per_tick": 2, + "wfi_is_nop": true, + "wfi_available_to_user_mode": false, + "max_time_to_wait": 200 + }, + "extensions": { + "M": { + "supported": true + }, + "A": { + "supported": false + }, + "F": { + "supported": false, + "fflags_dirty_policy": "Fflags_Dirty_Precise" + }, + "D": { + "supported": false + }, + "V": { + "support_level": "Disabled", + "vlen_exp": 8, + "elen_exp": 6, + "reserved_behavior": { + "illegal_vtype": "IllegalVtype_SetVill", + "vstart_out_of_bounds": "Vstart_Illegal" + }, + "vl_use_ceil": false, + "max_index_eew_exp": 6, + "vstart": { + "zero_required": { + "arith": true, + "scalar_move": true + } + } + }, + "B": { + "supported": false + }, + "S": { + "supported": false + }, + "U": { + "supported": false + }, + "Zibi": { + "supported": false + }, + "Zic64b": { + "supported": false + }, + "Zicbom": { + "supported": false + }, + "Zicbop": { + "supported": false + }, + "Zicboz": { + "supported": false + }, + "Ziccamoa": { + "supported": false + }, + "Ziccamoc": { + "supported": false + }, + "Ziccif": { + "supported": false + }, + "Zicclsm": { + "supported": false + }, + "Ziccrse": { + "supported": false + }, + "Zicfilp": { + "supported": false + }, + "Zicfiss": { + "supported": false + }, + "Zicond": { + "supported": false + }, + "Zicntr": { + "supported": false + }, + "Zicsr": { + "supported": true + }, + "Zifencei": { + "supported": false + }, + "Zihintntl": { + "supported": false + }, + "Zihintpause": { + "supported": false + }, + "Zihpm": { + "supported": false + }, + "Zimop": { + "supported": false + }, + "Zmmul": { + "supported": true + }, + "Zaamo": { + "supported": false + }, + "Zabha": { + "supported": false + }, + "Zacas": { + "supported": false + }, + "Zalrsc": { + "supported": false + }, + "Zama16b": { + "supported": false + }, + "Zawrs": { + "supported": false, + "nto": { + "is_nop": false + }, + "sto": { + "is_nop": false + } + }, + "Zfa": { + "supported": false + }, + "Zfbfmin": { + "supported": false + }, + "Zfh": { + "supported": false + }, + "Zfhmin": { + "supported": false + }, + "Zfinx": { + "supported": false + }, + "Zdinx": { + "supported": false + }, + "Zca": { + "supported": false + }, + "Zcf": { + "supported": false + }, + "Zcd": { + "supported": false + }, + "Zcb": { + "supported": false + }, + "Zcmop": { + "supported": false + }, + "Zba": { + "supported": false + }, + "Zbb": { + "supported": false + }, + "Zbs": { + "supported": false + }, + "Zbc": { + "supported": false + }, + "Zbkb": { + "supported": false + }, + "Zbkc": { + "supported": false + }, + "Zbkx": { + "supported": false + }, + "Zknd": { + "supported": false + }, + "Zkne": { + "supported": false + }, + "Zknh": { + "supported": false + }, + "Zkr": { + "supported": false, + "sseed_reset_value": false, + "useed_reset_value": false, + "sseed_read_only_zero": false, + "useed_read_only_zero": false + }, + "Zksed": { + "supported": false + }, + "Zksh": { + "supported": false + }, + "Zkt": { + "supported": false + }, + "Zhinx": { + "supported": false + }, + "Zhinxmin": { + "supported": false + }, + "Zvabd": { + "supported": false + }, + "Zvfbfmin": { + "supported": false + }, + "Zvfbfwma": { + "supported": false + }, + "Zvfh": { + "supported": false + }, + "Zvfhmin": { + "supported": false + }, + "Zvbb": { + "supported": false + }, + "Zvbc": { + "supported": false + }, + "Zvkb": { + "supported": false + }, + "Zvkg": { + "supported": false + }, + "Zvkned": { + "supported": false + }, + "Zvknha": { + "supported": false + }, + "Zvknhb": { + "supported": false + }, + "Zvksed": { + "supported": false + }, + "Zvksh": { + "supported": false + }, + "Zvkt": { + "supported": false + }, + "Ssccptr": { + "supported": false + }, + "Sscofpmf": { + "supported": false + }, + "Sscounterenw": { + "supported": false + }, + "Sstc": { + "supported": false + }, + "Sstvala": { + "supported": false + }, + "Svade": { + "supported": false + }, + "Svadu": { + "supported": false + }, + "Svinval": { + "supported": false + }, + "Svrsw60t59b": { + "supported": false + }, + "Svnapot": { + "supported": false + }, + "Ssnpm": { + "supported": false, + "supported_pmlen_7": true, + "supported_pmlen_16": true + }, + "Smnpm": { + "supported": false, + "supported_pmlen_7": true, + "supported_pmlen_16": true + }, + "Smmpm": { + "supported": false, + "supported_pmlen_7": true, + "supported_pmlen_16": true + }, + "Smcntrpmf": { + "supported": false + }, + "Svbare": { + "supported": false, + "sfence_vma_illegal_if_svbare_only": true + }, + "Sv32": { + "supported": false + }, + "Sv39": { + "supported": false + }, + "Sv48": { + "supported": false + }, + "Sv57": { + "supported": false + }, + "Stateen": { + "Smstateen": { + "supported": false + }, + "Ssstateen": { + "supported": false + }, + "C_readonly_zero": true, + "SE0_readonly_zero": false + }, + "Ssqosid": { + "supported": false, + "rcid_length": 12, + "mcid_length": 12 + }, + "Svpbmt": { + "supported": false + }, + "Svvptc": { + "supported": false + } + } +} diff --git a/conformance/act4/test_config.yaml b/conformance/act4/test_config.yaml new file mode 100644 index 000000000..ad86091e4 --- /dev/null +++ b/conformance/act4/test_config.yaml @@ -0,0 +1,8 @@ +name: leanvm-rv64im +compiler_exe: riscv64-unknown-elf-gcc +objdump_exe: riscv64-unknown-elf-objdump +ref_model_exe: sail_riscv_sim +udb_config: leanvm-rv64im.yaml +linker_script: link.ld +dut_include_dir: . +include_priv_tests: false diff --git a/crates/fiat_shamir/src/lib.rs b/crates/fiat_shamir/src/lib.rs index 32a31618b..08ef8d963 100644 --- a/crates/fiat_shamir/src/lib.rs +++ b/crates/fiat_shamir/src/lib.rs @@ -9,8 +9,7 @@ use primitives::field::{F64, F192}; /// `f(a, b) = BLAKE2s(a‖b)` on two 256-bit halves laid out little-endian into /// 64 bytes, *exactly* the VM's `Blake2s` opcode: 64 input bytes → 32-byte /// digest, split back into four field words. THE primitive; the chain is a -/// chain of these, so a zkDSL program replays it with one `blake2s(...)` per -/// step. +/// chain of these, so a VM program replays it with one `BLAKE2S` row per step. /// /// A 64-byte input is one compression, so this is `compress(PARAM_IV, m, /// t = 64, last = true)` and nothing about the byte-level padding rules can @@ -40,9 +39,7 @@ const DS_POW_NONCE: F64 = F64(4); /// `compress(base, (nonce.c0, nonce.c1, nonce.c2, DS_POW_NONCE))` has its low `bits` /// bits zero: the grinding predicate over the VM compression. A CONTIGUOUS -/// low-bit window (rather than byte-wise leading zeros) so a recursive verifier -/// re-checks it with a single loop over the bit decomposition of the digest word -/// (`grind_check` in `guests/lean_ethereum.py`). `bits` is always `< 64`. +/// low-bit window rather than byte-wise leading zeros. `bits` is always `< 64`. #[inline] fn pow_bits_ok(base: [F64; 4], nonce: F192, bits: u32) -> bool { debug_assert!(bits < 64, "grinding deficit fits the digest's low word"); diff --git a/crates/fiat_shamir/src/merkle.rs b/crates/fiat_shamir/src/merkle.rs index bdab343e7..bdd9a111e 100644 --- a/crates/fiat_shamir/src/merkle.rs +++ b/crates/fiat_shamir/src/merkle.rs @@ -8,8 +8,8 @@ pub type Hash = [u8; 32]; /// Encode a Merkle hash as the two field words transcripts carry it in: two /// 128-bit halves, each a K pair with a spare top lane. Every digest in the -/// protocol uses this one split (the commitment root, the public input, the -/// guest's MD state), so the VM sees one shape everywhere. +/// protocol uses this one split (the commitment root, the public input), so the +/// VM sees one shape everywhere. #[inline] pub fn hash_to_scalars(hash: &Hash) -> [F192; 2] { let word_at = |offset: usize| u64::from_le_bytes(hash[offset..offset + 8].try_into().unwrap()); @@ -241,12 +241,12 @@ impl PrunedMerklePaths { /// /// The redundant form. Several queries of one phase repeat whatever siblings /// they share, which is exactly what makes it simple to consume: recomputing -/// the root is a walk up one path, with no dedup bookkeeping. Recursive witness -/// construction and the Python verifier consume this; the wire format -/// ([`PrunedMerklePaths`]) sends each shared sibling once. +/// the root is a walk up one path, with no dedup bookkeeping. The Python +/// verifier consumes this; the wire format ([`PrunedMerklePaths`]) sends each +/// shared sibling once. #[derive(Clone, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] pub struct RawMerklePath { - /// Transcript-derived position, retained for recursive witness construction. + /// Transcript-derived position. pub leaf_index: usize, pub leaf_data: Vec, pub path: Vec, diff --git a/crates/fiat_shamir/src/transcript.rs b/crates/fiat_shamir/src/transcript.rs index fefba8478..8fc900c5c 100644 --- a/crates/fiat_shamir/src/transcript.rs +++ b/crates/fiat_shamir/src/transcript.rs @@ -11,11 +11,10 @@ pub struct Proof { pub merkle: Vec, } -/// The proof the recursion guest and the Python verifier consume: [`Proof`] with -/// every query's Merkle path written out, which is the one thing they would -/// otherwise have to reconstruct. A verifier run yields it as a by-product -/// ([`VerifierState::into_raw_proof`]), so that expansion is written once, in -/// Rust, instead of three times in three languages. +/// The proof the Python verifier consumes: [`Proof`] with every query's Merkle +/// path written out, which is the one thing it would otherwise have to +/// reconstruct. A verifier run yields it as a by-product +/// ([`VerifierState::into_raw_proof`]), so that expansion is written once, in Rust. pub type RawProof = Proof; #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -44,8 +43,7 @@ pub trait Transmitter: Challenger { fn add_scalars(&mut self, xs: &[F192]); fn grind(&mut self, bits: u32); - /// Transmit a root as its two scalars, not as a byte string, so the recursion guest replays - /// one shape for every digest. Sending it is what binds it: no verifier absorbs a root + /// Transmit a root as its two scalars, not as a byte string. Sending it is what binds it: no verifier absorbs a root /// separately (see [`Receiver::next_root`], its mirror). fn add_root(&mut self, root: &Hash) { self.add_scalars(&hash_to_scalars(root)); @@ -200,13 +198,6 @@ impl<'a> VerifierState<'a> { } } - /// How many scalars have been read so far: the cursor into the stream a - /// caller needs to locate a sub-protocol's scalars without counting back - /// from the tail. - pub fn stream_offset(&self) -> usize { - self.offset - } - /// Assert the whole proof was consumed (no trailing/extra data). pub fn finish(&self) -> Result<(), Error> { if self.offset == self.stream.len() && self.phase == self.merkle.len() { diff --git a/crates/flock/src/arith.rs b/crates/flock/src/arith.rs new file mode 100644 index 000000000..aa3fbac92 --- /dev/null +++ b/crates/flock/src/arith.rs @@ -0,0 +1,270 @@ +//! u64 arithmetic as Flock R1CS circuits, one operation per block: wrapping +//! addition ([`add`]), and multiplication ([`mul`]) wrapping or widening. +//! +//! Each is a [`crate::circuit`] gate list over the ports `a`, `b` and the result, in +//! that order ([`A_BASE`], [`B_BASE`], [`OUT_BASE`]), built from [`add::Adder`] and +//! [`mul::Multiplier`], which take wires and return wires and so compose into +//! larger circuits. Here the witness is not the generic walk of the gate list but +//! word arithmetic on the structure the list is built from, one instance at a time. + +pub mod add; +pub mod mul; + +use crate::circuit::{Builder, Circuit}; +use crate::reduction::Block; +use zk_alloc::ArenaVec; + +pub const A_BASE: usize = 0; +pub const B_BASE: usize = 64; +pub const OUT_BASE: usize = 128; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum U64Op { + /// `a + b mod 2^64`. + WrappingAdd, + /// `a·b mod 2^64`. + WrappingMul, + /// `a·b` as a u128. + WideningMul, +} + +impl U64Op { + /// Bits of the committed result. + pub const fn out_bits(self) -> usize { + match self { + Self::WrappingAdd | Self::WrappingMul => 64, + Self::WideningMul => 128, + } + } +} + +/// One instance's words of `z`, `A·z` and `B·z`. +struct Instance<'a> { + z: &'a mut [u64], + az: &'a mut [u64], + bz: &'a mut [u64], +} + +impl Instance<'_> { + /// `width` rows from `slot` whose B side is the constant, with `A·z = z = v`. + fn unit_rows(&mut self, slot: usize, v: u128, width: usize) { + or_bits(self.z, slot, v); + or_bits(self.az, slot, v); + or_bits(self.bz, slot, u128::MAX >> (128 - width)); + } + + /// Product rows from `slot`, one per position of `mask`, with `A·z = left`, + /// `B·z = right` and `z = left·right`. + fn products(&mut self, slot: usize, mask: u128, left: u128, right: u128) { + if mask != 0 { + let shift = mask.trailing_zeros(); + or_bits(self.z, slot, (left & right & mask) >> shift); + or_bits(self.az, slot, (left & mask) >> shift); + or_bits(self.bz, slot, (right & mask) >> shift); + } + } +} + +/// OR `v` into `buf` from bit `at`. +#[inline(always)] +fn or_bits(buf: &mut [u64], at: usize, v: u128) { + let s = at % 64; + let words = [ + (v << s) as u64, + ((v >> 1) >> (63 - s)) as u64, + ((v >> 1) >> (127 - s)) as u64, + ]; + for (w, x) in buf[at / 64..].iter_mut().zip(words) { + *w |= x; + } +} + +/// What an operation's witness is computed from, besides its inputs. +enum Plan { + Add(add::Adder), + Mul(mul::Multiplier), +} + +pub struct U64Circuit { + op: U64Op, + circuit: Circuit, + plan: Plan, +} + +impl U64Circuit { + pub fn new(op: U64Op) -> Self { + let n = op.out_bits(); + let mut c = Builder::new(&[64, 64], &[n]); + let (a, b) = (c.input(0), c.input(1)); + let (out, plan) = match op { + U64Op::WrappingAdd => { + let (out, adder) = add::Adder::build(&mut c, &a, &b); + (out, Plan::Add(adder)) + } + U64Op::WrappingMul | U64Op::WideningMul => { + let (out, multiplier) = mul::Multiplier::build(&mut c, &a, &b, n); + (out, Plan::Mul(multiplier)) + } + }; + for (i, wire) in out.into_iter().enumerate() { + c.output(0, i, wire); + } + let circuit = c.finish(); + assert_eq!(circuit.const_pos(), OUT_BASE + n); + Self { op, circuit, plan } + } + + pub fn circuit(&self) -> &Circuit { + &self.circuit + } + + pub fn k_log(&self) -> usize { + self.circuit.k_log() + } + + pub fn useful_bits(&self) -> usize { + self.circuit.useful_bits() + } + + pub fn block(&self) -> Block<'_> { + self.circuit.block() + } + + /// `(z, a, b, z_lincheck)` for `pairs` padded with `(0, 0)` to + /// `2^n_blocks_log` instances: the bit-packed `z`, `A·z` and `B·z` + /// (`2^k_log / 64` words per instance), and lincheck's byte stripes. + pub fn generate_witness( + &self, + pairs: &[(u64, u64)], + n_blocks_log: usize, + ) -> (ArenaVec, ArenaVec, ArenaVec, ArenaVec) { + let n = self.op.out_bits(); + self.circuit + .generate_witness_with(pairs, &(0, 0), n_blocks_log, |&(a, b), z, az, bz| { + let mut witness = Instance { z, az, bz }; + let out = match &self.plan { + Plan::Add(adder) => adder.witness(a, b, &mut witness), + Plan::Mul(multiplier) => multiplier.witness(a, b, &mut witness), + }; + witness.unit_rows(A_BASE, a as u128, 64); + witness.unit_rows(B_BASE, b as u128, 64); + witness.unit_rows(OUT_BASE, out, n); + witness.unit_rows(self.circuit.const_pos(), 1, 1); + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::lincheck::LincheckCircuit; + use fiat_shamir::transcript::{ProverState, VerifierState}; + use primitives::field::F192; + use primitives::test_rng::Rng; + + const OPS: [U64Op; 3] = [U64Op::WrappingAdd, U64Op::WrappingMul, U64Op::WideningMul]; + + fn native(op: U64Op, a: u64, b: u64) -> u128 { + let (a, b) = (a as u128, b as u128); + match op { + U64Op::WrappingAdd => (a + b) as u64 as u128, + U64Op::WrappingMul => (a * b) as u64 as u128, + U64Op::WideningMul => a * b, + } + } + + /// Every pairing of the carry-heavy edge values, then random pairs. + fn pairs(n: usize, seed: u64) -> Vec<(u64, u64)> { + const EDGES: [u64; 6] = [0, 1, 2, 1 << 63, u64::MAX - 1, u64::MAX]; + let mut rng = Rng::new(seed); + EDGES + .iter() + .flat_map(|&x| EDGES.iter().map(move |&y| (x, y))) + .chain(std::iter::repeat_with(|| (rng.next_u64(), rng.next_u64()))) + .take(n) + .collect() + } + + /// The committed result is the native one and every row holds, which ties + /// the word-level witness to the gate list the walks read. + #[test] + fn witness_is_the_result_and_satisfies_r1cs() { + let n_log = 6; + for op in OPS { + let circuit = U64Circuit::new(op); + let k = circuit.circuit.n_cols(); + let pairs = pairs(1 << n_log, 0x3A11); + let (z, _, _, _) = circuit.generate_witness(&pairs, n_log); + for (t, &(x, y)) in pairs.iter().enumerate() { + let word = |w: usize| z[t * (k / 64) + w]; + let out = (word(2) as u128 | (word(3) as u128) << 64) & (u128::MAX >> (128 - op.out_bits())); + assert_eq!((word(0), word(1), out), (x, y, native(op, x, y)), "{op:?}"); + let block: Vec = (0..k) + .map(|i| { + if (z[(t * k + i) / 64] >> (i % 64)) & 1 == 1 { + F192::ONE + } else { + F192::ZERO + } + }) + .collect(); + let (ra, rb) = circuit.circuit.row_values(&block); + assert!((0..k).all(|i| ra[i] * rb[i] == block[i]), "{op:?} ({x}, {y})"); + } + } + } + + /// The generic walk of the gate list writes the very tables the word arithmetic does. + #[test] + fn generic_witness_is_the_word_arithmetic() { + let n_log = 4; + for op in OPS { + let circuit = U64Circuit::new(op); + let pairs = pairs(1 << n_log, 0x3A13); + let rows: Vec<[u64; 2]> = pairs.iter().map(|&(a, b)| [a, b]).collect(); + let fast = circuit.generate_witness(&pairs, n_log); + let generic = circuit.circuit.generate_witness(&rows, n_log); + assert!(fast.0[..] == generic.0[..], "{op:?}: z"); + assert!(fast.1[..] == generic.1[..], "{op:?}: A·z"); + assert!(fast.2[..] == generic.2[..], "{op:?}: B·z"); + assert!(fast.3[..] == generic.3[..], "{op:?}: stripes"); + } + } + + /// The reduction verifies an honest batch, which is also what ties the + /// prover's backward walk and the `A·z`, `B·z` tables to the verifier's + /// forward walk, and rejects one flipped witness bit. + #[test] + fn reduction_roundtrip_rejects_tampering() { + const LABEL: &[u8] = b"flock-arith-reduction-test"; + // The zerocheck needs a cube of at least 2^13 bits. + let n_log = 5; + for op in OPS { + let circuit = U64Circuit::new(op); + let block = circuit.block(); + let pairs = pairs(1 << n_log, 0x3A12); + let run = |tamper: Option| { + let (mut z, a, b, mut z_lincheck) = circuit.generate_witness(&pairs, n_log); + if let Some(bit) = tamper { + z[bit / 64] ^= 1 << (bit % 64); + z_lincheck[bit] ^= 1; + } + let mut ps = ProverState::from_label(LABEL); + let stage = block.prove_zerocheck(n_log, &z, &a, &b, &mut ps); + let claim = block.prove_lincheck(n_log, stage, &z_lincheck, &mut ps); + let proof = ps.into_proof(); + let mut vs = VerifierState::from_label(LABEL, &proof); + block.verify(n_log, &mut vs).is_ok_and(|r| r.claim == claim) && vs.finish().is_ok() + }; + assert!(run(None), "{op:?}"); + for bit in [ + A_BASE + 3, + OUT_BASE + 5, + circuit.circuit.const_pos(), + circuit.useful_bits() - 1, + ] { + assert!(!run(Some(bit)), "{op:?}: flipping bit {bit} must reject"); + } + } + } +} diff --git a/crates/flock/src/arith/add.rs b/crates/flock/src/arith/add.rs new file mode 100644 index 000000000..1744d2bc8 --- /dev/null +++ b/crates/flock/src/arith/add.rs @@ -0,0 +1,43 @@ +//! Wrapping addition, as a ripple-carry adder. The carry into position `i + 1` +//! is `maj(a_i, b_i, c_i) = (a_i ⊕ c_i)(b_i ⊕ c_i) ⊕ c_i`, one product, and every +//! other wire is affine, so the circuit pays 63 products: the carry out of bit +//! 63 falls off the modulus. The witness is the native sum, whose carries are +//! `(a + b) ⊕ a ⊕ b`. + +use super::Instance; +use crate::circuit::{Builder, Wire}; + +/// The carries into bits 1 to 63. +const CARRIES: u128 = (u64::MAX >> 1) as u128; + +pub struct Adder { + /// The first of the carries' 63 product slots. + slot: usize, +} + +impl Adder { + /// `a + b mod 2^64`, as wires. + pub fn build(c: &mut Builder, a: &[Wire], b: &[Wire]) -> (Vec, Self) { + let slot = c.next_slot(); + let mut carry = None; + let mut sum = Vec::with_capacity(64); + for i in 0..64 { + let ac = c.xor(a[i], carry); + let bc = c.xor(b[i], carry); + sum.push(c.xor(ac, b[i])); + if i < 63 { + let maj = c.and(ac, bc); + carry = c.xor(maj, carry); + } + } + (sum, Self { slot }) + } + + /// Writes the carries' rows and returns the result. + pub(super) fn witness(&self, a: u64, b: u64, witness: &mut Instance) -> u128 { + let sum = a.wrapping_add(b); + let carry_in = sum ^ a ^ b; + witness.products(self.slot, CARRIES, (a ^ carry_in) as u128, (b ^ carry_in) as u128); + sum as u128 + } +} diff --git a/crates/flock/src/arith/mul.rs b/crates/flock/src/arith/mul.rs new file mode 100644 index 000000000..4d126c7f8 --- /dev/null +++ b/crates/flock/src/arith/mul.rs @@ -0,0 +1,241 @@ +//! Multiplication. +//! +//! ## Partial products are free +//! +//! A schoolbook multiplier pays one product per partial product `a_i·b_j`, then +//! one per carry to sum them. Over GF(2) the first half is avoidable: with +//! `e_ij = ¬(a_i ⊕ b_j)`, `2·a_i·b_j = a_i + b_j − 1 + e_ij`. Summed, with +//! `M = 2^64 − 1` and `¬a = M − a` the 64-bit complement, +//! +//! ```text +//! 2ab = Σ_i (a_i ? b : ¬b)·2^i + ¬a + ¬b + (a + b)·2^64 + 1 − 2^128 +//! ``` +//! +//! Every row there is affine in the inputs. Its column 0, +//! `¬(a_0 ⊕ b_0) + ¬a_0 + ¬b_0 + 1`, is `2 + 2g` with `g = ¬a_0·¬b_0`, so after +//! that one product the identity halves: `a·b mod 2^N` is `1 + g` plus the other +//! columns shifted down a place. That is 66 rows of affine bits, with `1` and +//! `g` in the empty low bits of two of them. +//! +//! ## Compression +//! +//! A carry-save step turns three rows into their XOR and their majority shifted +//! up a place. The majority `(x ⊕ z)(y ⊕ z) ⊕ z` is one product at each position +//! where at least two rows have a bit, except where exactly two do and the carry +//! row is still free there: one of the two bits moves into it instead. Taking +//! the three rows that end lowest each time, the 64 steps cost as few products +//! as summing column by column, and a ripple-carry addition finishes the last +//! two rows. The top position's majority would carry out of the modulus, so it +//! is never a product. +//! +//! Every step is word arithmetic on `u128` rows and its products are one run of +//! slots, so an instance's witness is a few shifts and masks per step. + +use super::Instance; +use crate::circuit::{Builder, Wire}; + +/// The rows `(a_i ? b : ¬b)` for `i < 64`, then the `a` and `b` rows. +const N_ROWS: usize = 66; +const A_ROW: usize = 64; +const B_ROW: usize = 65; + +/// One carry-save step: rows `x`, `y`, `z` become the sum row, stored in `x`, +/// and the carry row, stored in `y`. +#[derive(Clone, Copy)] +struct Csa { + x: usize, + y: usize, + z: usize, + /// Positions whose majority is a product, one run of slots from `slot`. + products: u128, + /// Positions where the pair's `y` (or `z`) bit moves to the carry row. + move_y: u128, + move_z: u128, + slot: usize, +} + +/// A row's `(highest, lowest)` position. +fn ends(row: u128) -> (u32, u32) { + (127 - row.leading_zeros(), row.trailing_zeros()) +} + +fn is_run(mask: u128) -> bool { + let run = mask.checked_shr(mask.trailing_zeros()).unwrap_or(0); + run & run.wrapping_add(1) == 0 +} + +pub struct Multiplier { + g_slot: usize, + steps: Vec, + /// The two rows the steps leave, and where adding them makes a product. + last: (usize, usize), + carries: u128, + carry_slot: usize, + /// Positions below `N`. + width: u128, +} + +impl Multiplier { + /// The low `n` bits of `a·b`, as wires. + pub fn build(c: &mut Builder, a: &[Wire], b: &[Wire], n: usize) -> (Vec, Self) { + let width = u128::MAX >> (128 - n); + let one = c.one(); + let not_a: [Wire; 64] = std::array::from_fn(|i| c.xor(a[i], one)); + let not_b: [Wire; 64] = std::array::from_fn(|i| c.xor(b[i], one)); + let g_slot = c.next_slot(); + let g = c.and(not_a[0], not_b[0]); + + // Each row's wire per position, all shifted down a place: row 0's bit 0 + // is what `g` and the constant 1 replace. + let mut rows = vec![vec![None; n]; N_ROWS]; + for i in 0..64usize { + for j in 0..64 { + if let Some(p) = (i + j).checked_sub(1).filter(|&p| p < n) { + rows[i][p] = c.xor(b[j], not_a[i]); + } + } + } + // `(¬a ≫ 1) + a·2^63`, and the same for `b`. + for (row, low, high) in [(A_ROW, ¬_a, a), (B_ROW, ¬_b, b)] { + let len = 64.min(n - 63); + rows[row][..63].copy_from_slice(&low[1..]); + rows[row][63..63 + len].copy_from_slice(&high[..len]); + } + rows[2][0] = one; + rows[3][0] = g; + // Half of the constant `2^128`, which survives only mod `2^128`. + if n == 128 { + rows[A_ROW][127] = one; + } + let mut present: Vec = rows + .iter() + .map(|row| (0..n).filter(|&p| row[p].is_some()).fold(0, |m, p| m | (1 << p))) + .collect(); + + let mut live: Vec = (0..N_ROWS).collect(); + let mut steps = Vec::new(); + while live.len() > 2 { + let mut order: Vec = (0..live.len()).collect(); + order.sort_by_key(|&t| ends(present[live[t]])); + let [x, y, z] = [live[order[0]], live[order[1]], live[order[2]]]; + live.retain(|r| ![x, y, z].contains(r)); + live.extend([x, y]); + + let (px, py, pz) = (present[x], present[y], present[z]); + let pairs = ((px & py) | (px & pz) | (py & pz)) & (width >> 1); + let triples = px & py & pz; + let (mut products, mut moves) = (0u128, 0u128); + for p in (0..n - 1).filter(|&p| (pairs >> p) & 1 == 1) { + // The carry row is free at `p` unless `p − 1` has a product. + if (triples >> p) & 1 == 0 && (products << 1) >> p & 1 == 0 { + moves |= 1 << p; + } else { + products |= 1 << p; + } + } + assert!(is_run(products), "a step's products must be one run of slots"); + let step = Csa { + x, + y, + z, + products, + move_y: moves & !pz, + move_z: moves & pz, + slot: c.next_slot(), + }; + + let (mut sum, mut carry) = (vec![None; n], vec![None; n]); + for p in 0..n { + let (wx, wy, wz) = (rows[x][p], rows[y][p], rows[z][p]); + if (products >> p) & 1 == 1 { + let xz = c.xor(wx, wz); + let yz = c.xor(wy, wz); + let maj = c.and(xz, yz); + carry[p + 1] = c.xor(maj, wz); + sum[p] = c.xor(xz, wy); + } else if (step.move_z >> p) & 1 == 1 { + carry[p] = wz; + sum[p] = c.xor(wx, wy); + } else if (step.move_y >> p) & 1 == 1 { + carry[p] = wy; + sum[p] = wx; + } else { + let xz = c.xor(wx, wz); + sum[p] = c.xor(xz, wy); + } + } + rows[x] = sum; + rows[y] = carry; + present[x] = px | py | pz; + present[y] = (products << 1) | moves; + steps.push(step); + } + + let &[x, y] = live.as_slice() else { + unreachable!("the steps stop at two rows") + }; + let carry_slot = c.next_slot(); + let mut carries = 0u128; + let mut carry = None; + let mut product = Vec::with_capacity(n); + for p in 0..n { + let (wx, wy) = (rows[x][p], rows[y][p]); + let out = if p + 1 < n && [wx, wy, carry].iter().flatten().count() >= 2 { + carries |= 1 << p; + let xc = c.xor(wx, carry); + let yc = c.xor(wy, carry); + let maj = c.and(xc, yc); + carry = c.xor(maj, carry); + c.xor(xc, wy) + } else { + let xy = c.xor(wx, wy); + c.xor(xy, carry.take()) + }; + product.push(out); + } + assert!(is_run(carries), "the final carries must be one run of slots"); + + let multiplier = Self { + g_slot, + steps, + last: (x, y), + carries, + carry_slot, + width, + }; + (product, multiplier) + } + + /// Writes `g`'s row and every step's products, and returns the result. + pub(super) fn witness(&self, a: u64, b: u64, witness: &mut Instance) -> u128 { + let (na, nb) = (!a, !b); + let mut rows = [0u128; N_ROWS]; + for (i, row) in rows[..64].iter_mut().enumerate() { + let v = (b ^ ((a >> i) & 1).wrapping_sub(1)) as u128; + *row = if i == 0 { v >> 1 } else { v << (i - 1) }; + } + rows[A_ROW] = ((na >> 1) as u128) | ((a as u128) << 63) | (1 << 127); + rows[B_ROW] = ((nb >> 1) as u128) | ((b as u128) << 63); + rows[2] |= 1; + rows[3] |= (na & nb & 1) as u128; + for row in &mut rows { + *row &= self.width; + } + witness.products(self.g_slot, 1, na as u128, nb as u128); + + for s in &self.steps { + let (rx, ry, rz) = (rows[s.x], rows[s.y], rows[s.z]); + let (xz, yz) = (rx ^ rz, ry ^ rz); + let moved = (ry & s.move_y) | (rz & s.move_z); + rows[s.x] = rx ^ ry ^ rz ^ moved; + rows[s.y] = ((((xz & yz) ^ rz) & s.products) << 1) | moved; + witness.products(s.slot, s.products, xz, yz); + } + + let (rx, ry) = (rows[self.last.0], rows[self.last.1]); + let sum = rx.wrapping_add(ry) & self.width; + let carry_in = sum ^ rx ^ ry; + witness.products(self.carry_slot, self.carries, rx ^ carry_in, ry ^ carry_in); + sum + } +} diff --git a/crates/flock/src/circuit.rs b/crates/flock/src/circuit.rs new file mode 100644 index 000000000..bdc71744a --- /dev/null +++ b/crates/flock/src/circuit.rs @@ -0,0 +1,365 @@ +//! Boolean circuits as gate lists over word ports: what [`crate::arith`] and the +//! VM's instruction classes are written in. +//! +//! ## Witness layout per block +//! +//! ```text +//! z[0 .. 64·P) = the ports, a whole number of 64-bit words each: +//! inputs (free), then outputs (committed copies) +//! z[64·P] = 1 (constant wire) +//! z[64·P + 1 .. useful) = the circuit's products, in the order they are made +//! z[useful .. 2^k_log) = padding (forced to 0 by empty rows) +//! ``` +//! +//! A port bit with no gate is an empty row too, hence zero: an output narrower than +//! its words, a single bit say, is that value as a 64-bit word. A caller that binds +//! ports to something outside (memory words, in the VM) relies on exactly this. +//! +//! A circuit is one gate list, where a wire is the gate driving it and each +//! committed wire is a row. As in [`crate::hash`], no matrix is ever built: the +//! verifier walks the list forwards and the prover backwards (doc/leanvm, Annex +//! C "Evaluating the matrices"). Across implementations what has to agree is the +//! port layout and the order products are made in, which fixes their slots; the +//! order of the free XORs is nobody's business. + +use crate::lincheck::LincheckCircuit; +use crate::reduction::Block; +use crate::witness::drive_witness_packed_and_lincheck; +use primitives::field::F192; +use zk_alloc::ArenaVec; + +/// The gate driving a wire, or `None` for a structural zero. +pub type Wire = Option; + +#[derive(Clone, Copy)] +enum Gate { + /// The committed free wire at `slot`: an input bit, or the constant. Row `z[slot]·1 = z[slot]`. + Free(u32), + /// `w_x ⊕ w_y`, uncommitted. + Xor(u32, u32), + /// Row `w_x · w_y = z[slot]`. + And(u32, u32, u32), + /// Row `w_x · 1 = z[slot]`: an affine wire committed, which is how a result leaves the circuit. + Copy(u32, u32), +} + +/// A gate list under construction. +pub struct Builder { + gates: Vec, + next_slot: usize, + one: Wire, + inputs: Vec>, + /// Each output port's first bit and width. + outputs: Vec<(usize, usize)>, + n_input_words: usize, +} + +impl Builder { + /// Ports of the given widths in bits, each rounded up to whole words: the inputs, + /// whose bits are free wires, then the outputs. + pub fn new(input_bits: &[usize], output_bits: &[usize]) -> Self { + let words = |bits: &[usize]| bits.iter().map(|b| b.div_ceil(64)).sum::(); + let n_input_words = words(input_bits); + let const_pos = 64 * (n_input_words + words(output_bits)); + let mut c = Self { + gates: Vec::new(), + next_slot: const_pos + 1, + one: None, + inputs: Vec::new(), + outputs: Vec::new(), + n_input_words, + }; + c.one = Some(c.push(Gate::Free(const_pos as u32))); + let mut base = 0; + for &bits in input_bits { + let wires = (0..bits).map(|i| Some(c.push(Gate::Free((base + i) as u32)))).collect(); + c.inputs.push(wires); + base += 64 * bits.div_ceil(64); + } + for &bits in output_bits { + c.outputs.push((base, bits)); + base += 64 * bits.div_ceil(64); + } + c + } + + fn push(&mut self, gate: Gate) -> u32 { + self.gates.push(gate); + (self.gates.len() - 1) as u32 + } + + /// The constant 1. + pub fn one(&self) -> Wire { + self.one + } + + /// Input port `port`'s bits, low first. + pub fn input(&self, port: usize) -> Vec { + self.inputs[port].clone() + } + + /// The slot the next product takes. + pub fn next_slot(&self) -> usize { + self.next_slot + } + + pub fn xor(&mut self, x: Wire, y: Wire) -> Wire { + match (x, y) { + (Some(x), Some(y)) => Some(self.push(Gate::Xor(x, y))), + _ => x.or(y), + } + } + + pub fn not(&mut self, x: Wire) -> Wire { + self.xor(x, self.one) + } + + /// One product, and one slot, unless an operand is a structural zero. + pub fn and(&mut self, x: Wire, y: Wire) -> Wire { + let (x, y) = (x?, y?); + let slot = self.next_slot as u32; + self.next_slot += 1; + Some(self.push(Gate::And(x, y, slot))) + } + + pub fn or(&mut self, x: Wire, y: Wire) -> Wire { + let both = self.and(x, y); + let either = self.xor(x, y); + self.xor(either, both) + } + + /// `if s { x } else { y }`, one product. + pub fn mux(&mut self, s: Wire, x: Wire, y: Wire) -> Wire { + let d = self.xor(x, y); + let picked = self.and(s, d); + self.xor(picked, y) + } + + /// Commits `wire` as bit `bit` of output port `port`. A structural zero needs no + /// gate: the empty row is what forces the bit to zero. + pub fn output(&mut self, port: usize, bit: usize, wire: Wire) { + let (base, bits) = self.outputs[port]; + assert!(bit < bits, "output port {port} has {bits} bits"); + if let Some(wire) = wire { + self.push(Gate::Copy(wire, (base + bit) as u32)); + } + } + + pub fn finish(self) -> Circuit { + let useful_bits = self.next_slot; + Circuit { + const_pos: 64 * (self.n_input_words + self.outputs.iter().map(|o| o.1.div_ceil(64)).sum::()), + k_log: useful_bits.next_power_of_two().trailing_zeros() as usize, + useful_bits, + n_input_words: self.n_input_words, + gates: self.gates, + } + } +} + +pub struct Circuit { + gates: Vec, + const_pos: usize, + k_log: usize, + useful_bits: usize, + n_input_words: usize, +} + +impl Circuit { + /// `log2` of the bits one instance occupies. + pub fn k_log(&self) -> usize { + self.k_log + } + + pub fn useful_bits(&self) -> usize { + self.useful_bits + } + + /// The constant wire's position. + pub fn const_pos(&self) -> usize { + self.const_pos + } + + /// Products, which is what an instance pays beyond its ports. + pub fn n_products(&self) -> usize { + self.useful_bits - self.const_pos - 1 + } + + pub fn n_input_words(&self) -> usize { + self.n_input_words + } + + pub fn block(&self) -> Block<'_> { + Block { + k_log: self.k_log, + useful_bits: self.useful_bits, + circuit: self, + } + } + + /// One instance's `z`, `A·z` and `B·z`, by one walk of the gate list on bits. + /// `inputs` are the input ports' words; the buffers come zeroed. + pub fn witness_instance(&self, inputs: &[u64], z: &mut [u64], az: &mut [u64], bz: &mut [u64]) { + assert_eq!(inputs.len(), self.n_input_words); + let set = |buf: &mut [u64], slot: u32, v: bool| buf[slot as usize / 64] |= (v as u64) << (slot % 64); + let mut wires: Vec = Vec::with_capacity(self.gates.len()); + for &gate in &self.gates { + let v = match gate { + Gate::Free(s) => { + let v = s as usize == self.const_pos || (inputs[s as usize / 64] >> (s % 64)) & 1 == 1; + set(z, s, v); + set(az, s, v); + set(bz, s, true); + v + } + Gate::Xor(x, y) => wires[x as usize] ^ wires[y as usize], + Gate::And(x, y, s) => { + let (x, y) = (wires[x as usize], wires[y as usize]); + set(z, s, x & y); + set(az, s, x); + set(bz, s, y); + x & y + } + Gate::Copy(x, s) => { + let v = wires[x as usize]; + set(z, s, v); + set(az, s, v); + set(bz, s, true); + v + } + }; + wires.push(v); + } + } + + /// `(z, a, b, z_lincheck)` for `rows` of input words, padded with all-zero inputs + /// to `2^n_blocks_log` instances: the bit-packed `z`, `A·z` and `B·z` + /// (`2^k_log / 64` words per instance), and lincheck's byte stripes. + pub fn generate_witness( + &self, + rows: &[[u64; N]], + n_blocks_log: usize, + ) -> (ArenaVec, ArenaVec, ArenaVec, ArenaVec) { + assert_eq!(N, self.n_input_words); + self.generate_witness_with(rows, &[0; N], n_blocks_log, |row, z, az, bz| { + self.witness_instance(row, z, az, bz) + }) + } + + /// [`Self::generate_witness`] over the caller's own rows, `inputs` writing a row's + /// input words. `rows` fill the batch, or `padding` does. + pub fn generate_witness_by( + &self, + rows: &[S], + padding: &S, + n_blocks_log: usize, + inputs: impl Fn(&S, &mut [u64]) + Sync, + ) -> (ArenaVec, ArenaVec, ArenaVec, ArenaVec) { + const MAX_INPUT_WORDS: usize = 16; + assert!(self.n_input_words <= MAX_INPUT_WORDS); + self.generate_witness_with(rows, padding, n_blocks_log, |row, z, az, bz| { + let mut words = [0u64; MAX_INPUT_WORDS]; + inputs(row, &mut words[..self.n_input_words]); + self.witness_instance(&words[..self.n_input_words], z, az, bz) + }) + } + + /// [`Self::generate_witness`] with the caller's own rows, padding row and way to + /// fill an instance, for a circuit whose witness is cheaper as word arithmetic. + pub(crate) fn generate_witness_with( + &self, + rows: &[S], + padding: &S, + n_blocks_log: usize, + instance: impl Fn(&S, &mut [u64], &mut [u64], &mut [u64]) + Sync, + ) -> (ArenaVec, ArenaVec, ArenaVec, ArenaVec) { + drive_witness_packed_and_lincheck(rows, Some(padding), n_blocks_log, self.k_log, instance) + } + + /// The matrix-vector products `(A_0 w, B_0 w)`, by one forward walk. + pub(crate) fn row_values(&self, w: &[F192]) -> (Vec, Vec) { + let k = self.n_cols(); + assert_eq!(w.len(), k); + let wc = w[self.const_pos]; + let mut ra = vec![F192::ZERO; k]; + let mut rb = vec![F192::ZERO; k]; + let mut wires: Vec = Vec::with_capacity(self.gates.len()); + for &gate in &self.gates { + let v = match gate { + Gate::Free(s) => { + let s = s as usize; + (ra[s], rb[s]) = (w[s], wc); + w[s] + } + Gate::Xor(x, y) => wires[x as usize] + wires[y as usize], + Gate::And(x, y, s) => { + let s = s as usize; + (ra[s], rb[s]) = (wires[x as usize], wires[y as usize]); + w[s] + } + Gate::Copy(x, s) => { + let s = s as usize; + (ra[s], rb[s]) = (wires[x as usize], wc); + w[s] + } + }; + wires.push(v); + } + (ra, rb) + } +} + +impl LincheckCircuit for Circuit { + fn n_cols(&self) -> usize { + 1 << self.k_log + } + + fn const_pin_col(&self) -> usize { + self.const_pos + } + + /// `(A_0 + α B_0)ᵀ u`, by one backward walk: every gate, in reverse, hands + /// its wire's adjoint to its operands or deposits it on its slot. + fn fold_alpha_batched(&self, alpha: F192, u: &[F192]) -> Vec { + assert_eq!(u.len(), self.n_cols()); + let c = self.const_pos; + let mut m = vec![F192::ZERO; u.len()]; + let mut adj = vec![F192::ZERO; self.gates.len()]; + for (i, &gate) in self.gates.iter().enumerate().rev() { + let g = adj[i]; + match gate { + Gate::Free(s) => { + let s = s as usize; + m[s] += g + u[s]; + m[c] += alpha * u[s]; + } + Gate::Xor(x, y) => { + adj[x as usize] += g; + adj[y as usize] += g; + } + Gate::And(x, y, s) => { + let s = s as usize; + m[s] += g; + adj[x as usize] += u[s]; + adj[y as usize] += alpha * u[s]; + } + Gate::Copy(x, s) => { + let s = s as usize; + m[s] += g; + adj[x as usize] += u[s]; + m[c] += alpha * u[s]; + } + } + } + m + } + + fn bilinear_form(&self, alpha: F192, u: &[F192], w: &[F192]) -> Option { + let (ra, rb) = self.row_values(w); + Some( + u.iter() + .zip(ra.iter().zip(&rb)) + .fold(F192::ZERO, |acc, (&u, (&a, &b))| acc + u * (a + alpha * b)), + ) + } +} diff --git a/crates/flock/src/hash.rs b/crates/flock/src/hash.rs index 3a1bea461..67189e50b 100644 --- a/crates/flock/src/hash.rs +++ b/crates/flock/src/hash.rs @@ -86,17 +86,19 @@ use crate::gf2::{ ADD3_BITS, CARRY_BITS_PER_ADD, MatrixSide, WireWord, back_add, back_add3_fused, walk_add, walk_add3_fused, wire_from_const, wire_from_slot_base, wire_rotl, wire_rotr, wire_xor, }; +use crate::reduction::{self, Block}; use crate::verifier; -use crate::witness::packed_bytes; use crate::witness::{ BitRecord, add_carry_parts, add3_fused_parts, drive_witness_packed_and_lincheck, or_bit_at, write_lin_word_ab_packed, }; -use pcs::pack::{LOG_PACKING, PACKING_WIDTH}; -use pcs::stack_open::{RingSwitchClaim, RingSwitchOpen, RingSwitchVerify, RingSwitchVerifyClaim}; +use pcs::pack::LOG_PACKING; +use pcs::stack_open::{RingSwitchOpen, RingSwitchVerify}; use primitives::field::F192; use zk_alloc::ArenaVec; +pub use crate::reduction::{ReductionReplay, SliceClaim, ZerocheckStage, min_n_blocks_log}; + // --------------------------------------------------------------------------- // Public constants // --------------------------------------------------------------------------- @@ -108,13 +110,6 @@ pub const K: usize = 1 << K_LOG; /// Univariate-skip dim, must match [`crate::zerocheck::K_SKIP`]. pub const K_SKIP: usize = 6; -// A claim's `2^K_SKIP` slices are a ring-switch claim on `q_flock` only if that -// matches the packing width; otherwise `ring_claim` fails at run time. -const _: () = assert!( - K_SKIP == LOG_PACKING, - "the univariate skip must match the PCS packing width" -); - /// Number of BLAKE2s rounds. pub const N_ROUNDS: usize = primitives::hash::ROUNDS; /// Number of G calls per round (4 column + 4 diagonal). @@ -271,20 +266,12 @@ pub fn padding_block() -> Compression { /// commit that still had `build_matrices` and run `r1cs_digest_matches_baked`. /// /// The value is mirrored in `python-verifier/verifier.py`, which never could -/// rebuild the matrices, and in the recursion guest, so a deliberate circuit -/// change means bumping all three by hand. +/// rebuild the matrices, so a deliberate circuit change means bumping both by hand. pub const R1CS_DIGEST: [u8; 32] = [ 0x53, 0x7a, 0xd2, 0x07, 0x90, 0x30, 0x8f, 0x8e, 0xb8, 0xc0, 0xe8, 0xbd, 0x3e, 0x6c, 0x58, 0xee, 0x64, 0x57, 0x33, 0x71, 0xe3, 0xd5, 0x3c, 0x30, 0x61, 0x3d, 0xd0, 0x4d, 0x87, 0xc0, 0xb7, 0xea, ]; -/// Minimum `n_blocks_log` needed to prove `n_blocks` compressions, subject to -/// the lincheck floor of `n_blocks_log ≥ 3` (`n_outer ≥ 8`). -pub fn min_n_blocks_log(n_blocks: usize) -> usize { - assert!(n_blocks >= 1, "n_blocks must be ≥ 1"); - n_blocks.max(8).next_power_of_two().trailing_zeros() as usize -} - // --------------------------------------------------------------------------- // Circuit-walk evaluation: `(uᵀ A_0 w, uᵀ B_0 w)` in O(circuit) field ops, // over matrices that are never materialized. The row assignment these walks @@ -719,19 +706,12 @@ impl Blake2sSetup { // The zerocheck, lincheck, and ring-switch scalars use the shared transcript; // the caller carries the WHIR opening. -/// The one claim on the committed witness `q_flock` left by the Flock BLAKE2s -/// zerocheck + lincheck reduction, for the PCS to discharge: the `2^k_skip` -/// bit-slice values of `z` at `suffix_point`, transmitted and pinned inside the -/// reduction by lincheck's terminal identity (which batches A, B, the -/// constant-wire pin and C), so the PCS only has to bind them to the -/// commitment. -/// -/// This is the clean seam between Flock's reduction and the PCS. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct SliceClaim { - pub suffix_point: Vec, - pub s_hat_v: Vec, -} +/// The BLAKE2s circuit as the reduction sees it. +const BLOCK: Block<'static> = Block { + k_log: K_LOG, + useful_bits: USEFUL_BITS, + circuit: &WalkLincheckCircuit, +}; /// The variable count (`log2` length) of the committed `q_flock` column for /// `n_blocks` executed compressions: `K_LOG + min_n_blocks_log − LOG_PACKING`. @@ -741,103 +721,15 @@ pub fn qflock_kappa(n_blocks: usize) -> usize { K_LOG + min_n_blocks_log(n_blocks.max(1)) - LOG_PACKING } -/// One reduction claim as a tower [`RingSwitchClaim`]: the `2^k_skip` slices and -/// the suffix point they live at, which is the WHOLE multilinear tail of the -/// quirky point (`q_flock` has `2^qflock_vars` words, and the packing prefix is -/// exactly the skipped coordinates, so nothing is split off into it). -fn ring_claim(claim: &SliceClaim, qflock_vars: usize) -> RingSwitchClaim { - assert_eq!( - claim.suffix_point.len(), - qflock_vars, - "ring-switch suffix must span the q_flock cube" - ); - assert_eq!(claim.s_hat_v.len(), PACKING_WIDTH); - RingSwitchClaim { - suffix_point: claim.suffix_point.clone(), - s_hat_v: Some(claim.s_hat_v.clone()), - } -} - -/// Package the prover's reduction claim as a [`RingSwitchOpen`], so the PCS -/// discharges flock's validity in the same opening as the embedder's own point -/// claims. `offset` is `q_flock`'s slot in the committed stack; the opener -/// slices `q_flock` from there. +/// [`reduction::ring_switch_open`] for `n_blocks` compressions, `offset` being +/// `q_flock`'s slot in the committed stack. pub fn ring_switch_open(n_blocks: usize, offset: usize, reduced: &SliceClaim) -> RingSwitchOpen { - let qflock_vars = qflock_kappa(n_blocks); - RingSwitchOpen { - offset, - qflock_vars, - claims: vec![ring_claim(reduced, qflock_vars)], - } + reduction::ring_switch_open(qflock_kappa(n_blocks), offset, reduced) } -/// Verifier counterpart of [`ring_switch_open`]: package the recovered claim as -/// a [`RingSwitchVerify`], the same statement data. The transmitted opening -/// travels separately. +/// [`reduction::ring_switch_verify`] for `n_blocks` compressions. pub fn ring_switch_verify(n_blocks: usize, offset: usize, claim: &SliceClaim) -> RingSwitchVerify<'_> { - let qflock_vars = qflock_kappa(n_blocks); - assert_eq!( - claim.suffix_point.len(), - qflock_vars, - "ring-switch suffix must span the q_flock cube" - ); - RingSwitchVerify { - offset, - qflock_vars, - claims: vec![RingSwitchVerifyClaim { - suffix_point: &claim.suffix_point, - s_hat_v: claim.s_hat_v.as_slice().try_into().expect("ring-switch has 64 slices"), - }], - } -} - -/// Everything [`Blake2sSetup::verify_reduction`] recovers: the z-claim for the -/// PCS and the zerocheck / lincheck claims. -#[derive(Clone, Debug)] -pub struct ReductionReplay { - pub claim: SliceClaim, - pub zc_claim: crate::zerocheck::ZerocheckClaim, - pub lc_claim: crate::lincheck::LincheckClaim, -} - -/// The lincheck input point carried over from the zerocheck claim: the -/// univariate-skip coordinate, then the multilinear challenges split at -/// `inner_rest_len` into the inner-rest and outer halves. -fn x_ab_of(zc: &crate::zerocheck::ZerocheckClaim, inner_rest_len: usize) -> crate::lincheck::QuirkyPoint { - crate::lincheck::QuirkyPoint { - z_skip: zc.z, - x_inner_rest: zc.mlv_challenges[..inner_rest_len].to_vec(), - x_outer: zc.mlv_challenges[inner_rest_len..].to_vec(), - } -} - -/// The claim the reduction leaves for the PCS: lincheck's output point, whose -/// 64 slice values are `lc.s_hat_v`. Prover and verifier must derive it -/// identically, so they share this one derivation. -fn reduction_claim(lc: &crate::lincheck::LincheckClaim, x_outer: &[F192]) -> SliceClaim { - let mut suffix_point = lc.r_inner_rest.clone(); - suffix_point.extend_from_slice(x_outer); - SliceClaim { - suffix_point, - s_hat_v: lc.s_hat_v.clone(), - } -} - -/// What the zerocheck stage hands the lincheck stage: the zerocheck claim and -/// the quirky point lincheck runs at. Opaque; the two stages of -/// [`Blake2sSetup::prove_reduction_precomputed`] are split only so a caller can -/// time or profile them apart. -#[derive(Clone, Debug)] -pub struct ZerocheckStage { - x_ab: crate::lincheck::QuirkyPoint, -} - -/// One `FLOCK_PROVE_TRACE` line. `label` carries its own colon so the stages -/// line up. -fn trace_stage(label: &str, t: std::time::Instant) { - if std::env::var_os("FLOCK_PROVE_TRACE").is_some() { - eprintln!("[flock prove] {label:<11}{:8.2} ms", t.elapsed().as_secs_f64() * 1e3); - } + reduction::ring_switch_verify(qflock_kappa(n_blocks), offset, claim) } impl Blake2sSetup { @@ -877,35 +769,7 @@ impl Blake2sSetup { b_packed_words: &[u64], ps: &mut fiat_shamir::transcript::ProverState, ) -> ZerocheckStage { - let t_zerocheck = std::time::Instant::now(); - - // The fused generator packs 64 Boolean coordinates per word. - let packed_len = 1usize << (self.m() - 6); - assert_eq!(z_packed.len(), packed_len, "wrong packed witness length"); - assert_eq!(a_packed_words.len(), packed_len, "wrong packed A·z length"); - assert_eq!(b_packed_words.len(), packed_len, "wrong packed B·z length"); - - // No bind_statement here: the embedding protocol (leanVM) seeds its - // transcript with the R1CS digest and binds the instance - // count and commitment root before any challenge, so the statement is - // already fully transcript-bound. - - let padding = crate::zerocheck::PaddingSpec { - k_log: K_LOG, - useful_bits_per_block: USEFUL_BITS, - }; - let zc_claim = crate::zerocheck::prove_packed_padded( - packed_bytes(a_packed_words), - packed_bytes(b_packed_words), - packed_bytes(z_packed), // C = I, so c == z - self.m(), - &padding, - ps, - ); - - let x_ab = x_ab_of(&zc_claim, K_LOG - K_SKIP); - trace_stage("zerocheck:", t_zerocheck); - ZerocheckStage { x_ab } + BLOCK.prove_zerocheck(self.n_blocks_log, z_packed, a_packed_words, b_packed_words, ps) } /// **Flock reduction, second stage (prover): the lincheck.** Reduces the @@ -917,25 +781,7 @@ impl Blake2sSetup { z_packed_lincheck: &[u8], ps: &mut fiat_shamir::transcript::ProverState, ) -> SliceClaim { - let t_lincheck = std::time::Instant::now(); - let packed_len = 1usize << (self.m() - 6); - assert_eq!(z_packed_lincheck.len(), packed_len * 8, "wrong lincheck stripe length"); - - let ZerocheckStage { x_ab } = stage; - let lc_claim = crate::lincheck::prove_padded_capture_s_hat_v( - z_packed_lincheck, - self.m(), - K_LOG, - K_SKIP, - USEFUL_BITS, - &WalkLincheckCircuit, - &x_ab, - ps, - ); - - let claim = reduction_claim(&lc_claim, &x_ab.x_outer); - trace_stage("lincheck:", t_lincheck); - claim + BLOCK.prove_lincheck(self.n_blocks_log, stage, z_packed_lincheck, ps) } /// **Flock reduction (verifier).** Replay the BLAKE2s zerocheck and @@ -946,35 +792,7 @@ impl Blake2sSetup { &self, vs: &mut fiat_shamir::transcript::VerifierState<'_>, ) -> Result { - // Mirror of prove_reduction_precomputed: the statement is bound by the embedding - // protocol's seed (R1CS digest) + announced count + commitment root. - - let zc_claim = crate::zerocheck::verify(self.m(), vs).map_err(verifier::VerifyError::Zerocheck)?; - - let inner_rest_len = K_LOG - K_SKIP; - let x_ab = x_ab_of(&zc_claim, inner_rest_len); - // Walk-capable circuit: the verifier's lincheck consistency check is - // one circuit walk (O(circuit) field ops) instead of the ∝ NNZ CSC - // marginal fold. Same transcript, same accept/reject. - let lc_claim = crate::lincheck::verify( - self.m(), - K_LOG, - K_SKIP, - &WalkLincheckCircuit, - &x_ab, - zc_claim.a_eval, - zc_claim.b_eval, - zc_claim.c_eval, - vs, - ) - .map_err(verifier::VerifyError::Lincheck)?; - - let claim = reduction_claim(&lc_claim, &x_ab.x_outer); - Ok(ReductionReplay { - claim, - zc_claim, - lc_claim, - }) + BLOCK.verify(self.n_blocks_log, vs) } } diff --git a/crates/flock/src/lib.rs b/crates/flock/src/lib.rs index c5b111036..15e8c615a 100644 --- a/crates/flock/src/lib.rs +++ b/crates/flock/src/lib.rs @@ -12,12 +12,15 @@ //! 4. The PCS binds that family of slices ([`hash::SliceClaim`]) to the //! commitment. //! -//! [`hash`] is the one circuit: the BLAKE2s compression as a per-block R1CS, -//! plus its witness generation and the leanVM-facing reduction entry points -//! (`Blake2sSetup::{prove_reduction_precomputed, verify_reduction, …}`). Steps 2 to 4 above -//! are circuit-agnostic: they take the block shape as plain numbers and reach -//! the matrices only through [`lincheck::LincheckCircuit`], whose one live impl -//! walks the circuit rather than reading any matrix. +//! [`hash`] is the protocol's circuit: the BLAKE2s compression as a per-block +//! R1CS, plus its witness generation and the leanVM-facing reduction entry +//! points (`Blake2sSetup::{prove_reduction_precomputed, verify_reduction, …}`). [`circuit`] +//! is the gate-list vocabulary every other circuit is written in, and [`arith`] +//! holds u64 addition and multiplication in it. +//! Steps 2 to 4 above are +//! circuit-agnostic ([`reduction`]): they take the block shape as plain numbers +//! and reach the matrices only through [`lincheck::LincheckCircuit`], whose +//! impls walk the circuit rather than reading any matrix. //! //! BLAKE2s is a 32-bit ARX round whose XORs and rotations are free over GF(2), //! so its only nonlinear constraints are the product bits of the modular ADDs. @@ -25,9 +28,12 @@ //! gadgets, forwards and transposed, kept separate because the fused //! three-operand adder's bit boundaries are the subtlest thing here. +pub mod arith; +pub mod circuit; mod gf2; pub mod hash; pub mod lincheck; +pub mod reduction; /// The circuit driven through the whole reduction. A `src` module rather than /// its own test binary so it shares the process, and so the slow /// [`hash::matrices`] build, with the unit tests. diff --git a/crates/flock/src/lincheck.rs b/crates/flock/src/lincheck.rs index be77eef39..10052f110 100644 --- a/crates/flock/src/lincheck.rs +++ b/crates/flock/src/lincheck.rs @@ -1320,8 +1320,7 @@ pub fn verify( // The c term's `⟨eq_inner, w_col⟩`, by the tensor structure of both sides: // `eq_inner = eq(x_inner_rest) ⊗ λ(z_skip)` and `w_col = eq(r_inner_rest) ⊗ // z_partial`, so it is 8 eq factors times a 64-term Lagrange combination - // instead of a length-k inner product. That is the form the recursive - // verifier can afford. + // instead of a length-k inner product. let lambda_skip = lagrange_weights_naive(k_skip, x_ab.z_skip); let c_slice_value = lambda_skip .iter() diff --git a/crates/flock/src/reduction.rs b/crates/flock/src/reduction.rs new file mode 100644 index 000000000..3bda3cc47 --- /dev/null +++ b/crates/flock/src/reduction.rs @@ -0,0 +1,253 @@ +//! The circuit-agnostic half of Flock: zerocheck then lincheck over a batch of +//! `2^k_log`-bit blocks, reducing R1CS validity to ONE claim on the packed +//! witness, packaged for ring switching. A circuit supplies only its [`Block`]: +//! the shape, and the walks behind its [`LincheckCircuit`]. + +use crate::lincheck::{self, LincheckCircuit, LincheckClaim, QuirkyPoint}; +use crate::verifier::VerifyError; +use crate::witness::packed_bytes; +use crate::zerocheck::{self, K_SKIP, PaddingSpec, ZerocheckClaim}; +use fiat_shamir::transcript::{ProverState, VerifierState}; +use pcs::pack::{LOG_PACKING, PACKING_WIDTH}; +use pcs::stack_open::{RingSwitchClaim, RingSwitchOpen, RingSwitchVerify, RingSwitchVerifyClaim}; +use primitives::field::F192; + +// A claim's `2^K_SKIP` slices are a ring-switch claim on `q_flock` only if that +// matches the packing width; otherwise `ring_claim` fails at run time. +const _: () = assert!( + K_SKIP == LOG_PACKING, + "the univariate skip must match the PCS packing width" +); + +/// Minimum `n_blocks_log` needed to prove `n_blocks` instances, subject to the +/// lincheck floor of `n_blocks_log ≥ 3` (`n_outer ≥ 8`). +pub fn min_n_blocks_log(n_blocks: usize) -> usize { + assert!(n_blocks >= 1, "n_blocks must be ≥ 1"); + n_blocks.max(8).next_power_of_two().trailing_zeros() as usize +} + +/// A circuit as the reduction sees it: `2^k_log` witness bits per instance, of +/// which `[useful_bits, 2^k_log)` are zero padding the prover skips. +#[derive(Clone, Copy)] +pub struct Block<'a> { + pub k_log: usize, + pub useful_bits: usize, + pub circuit: &'a dyn LincheckCircuit, +} + +/// The one claim on the committed witness `q_flock` left by the zerocheck + +/// lincheck reduction, for the PCS to discharge: the `2^k_skip` bit-slice +/// values of `z` at `suffix_point`, transmitted and pinned inside the reduction +/// by lincheck's terminal identity (which batches A, B, the constant-wire pin +/// and C), so the PCS only has to bind them to the commitment. +/// +/// This is the clean seam between Flock's reduction and the PCS. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct SliceClaim { + pub suffix_point: Vec, + pub s_hat_v: Vec, +} + +/// Everything [`Block::verify`] recovers: the z-claim for the PCS and the +/// zerocheck / lincheck claims. +#[derive(Clone, Debug)] +pub struct ReductionReplay { + pub claim: SliceClaim, + pub zc_claim: ZerocheckClaim, + pub lc_claim: LincheckClaim, +} + +/// What the zerocheck stage hands the lincheck stage: the quirky point lincheck +/// runs at. Opaque; the two stages are split only so a caller can time or +/// profile them apart. +#[derive(Clone, Debug)] +pub struct ZerocheckStage { + x_ab: QuirkyPoint, +} + +/// One `FLOCK_PROVE_TRACE` line. `label` carries its own colon so the stages +/// line up. +pub(crate) fn trace_stage(label: &str, t: std::time::Instant) { + if std::env::var_os("FLOCK_PROVE_TRACE").is_some() { + eprintln!("[flock prove] {label:<11}{:8.2} ms", t.elapsed().as_secs_f64() * 1e3); + } +} + +/// The lincheck input point carried over from the zerocheck claim: the +/// univariate-skip coordinate, then the multilinear challenges split at +/// `inner_rest_len` into the inner-rest and outer halves. +fn x_ab_of(zc: &ZerocheckClaim, inner_rest_len: usize) -> QuirkyPoint { + QuirkyPoint { + z_skip: zc.z, + x_inner_rest: zc.mlv_challenges[..inner_rest_len].to_vec(), + x_outer: zc.mlv_challenges[inner_rest_len..].to_vec(), + } +} + +/// The claim the reduction leaves for the PCS: lincheck's output point, whose +/// 64 slice values are `lc.s_hat_v`. Prover and verifier must derive it +/// identically, so they share this one derivation. +fn reduction_claim(lc: &LincheckClaim, x_outer: &[F192]) -> SliceClaim { + let mut suffix_point = lc.r_inner_rest.clone(); + suffix_point.extend_from_slice(x_outer); + SliceClaim { + suffix_point, + s_hat_v: lc.s_hat_v.clone(), + } +} + +impl Block<'_> { + /// **First stage (prover): the zerocheck.** Reduces `a·b ⊕ c = 0` over the + /// cube of `2^n_blocks_log` blocks to evaluation claims on `(â, b̂, ĉ)`, all + /// three at one point. + pub fn prove_zerocheck( + &self, + n_blocks_log: usize, + z_packed: &[u64], + a_packed_words: &[u64], + b_packed_words: &[u64], + ps: &mut ProverState, + ) -> ZerocheckStage { + let t_zerocheck = std::time::Instant::now(); + let m = self.k_log + n_blocks_log; + + // The fused generator packs 64 Boolean coordinates per word. + let packed_len = 1usize << (m - 6); + assert_eq!(z_packed.len(), packed_len, "wrong packed witness length"); + assert_eq!(a_packed_words.len(), packed_len, "wrong packed A·z length"); + assert_eq!(b_packed_words.len(), packed_len, "wrong packed B·z length"); + + // No bind_statement here: the embedding protocol binds the circuit, the + // instance count and the commitment root before any challenge, so the + // statement is already fully transcript-bound. + + let padding = PaddingSpec { + k_log: self.k_log, + useful_bits_per_block: self.useful_bits, + }; + let zc_claim = zerocheck::prove_packed_padded( + packed_bytes(a_packed_words), + packed_bytes(b_packed_words), + packed_bytes(z_packed), // C = I, so c == z + m, + &padding, + ps, + ); + + let x_ab = x_ab_of(&zc_claim, self.k_log - K_SKIP); + trace_stage("zerocheck:", t_zerocheck); + ZerocheckStage { x_ab } + } + + /// **Second stage (prover): the lincheck.** Reduces the zerocheck's + /// `(â, b̂, ĉ)` claims to the `2^k_skip` bit slices of `z` at one point, + /// against the per-block matrices. + pub fn prove_lincheck( + &self, + n_blocks_log: usize, + stage: ZerocheckStage, + z_packed_lincheck: &[u8], + ps: &mut ProverState, + ) -> SliceClaim { + let t_lincheck = std::time::Instant::now(); + let m = self.k_log + n_blocks_log; + assert_eq!( + z_packed_lincheck.len(), + (1usize << m) / 8, + "wrong lincheck stripe length" + ); + + let ZerocheckStage { x_ab } = stage; + let lc_claim = lincheck::prove_padded_capture_s_hat_v( + z_packed_lincheck, + m, + self.k_log, + K_SKIP, + self.useful_bits, + self.circuit, + &x_ab, + ps, + ); + + let claim = reduction_claim(&lc_claim, &x_ab.x_outer); + trace_stage("lincheck:", t_lincheck); + claim + } + + /// **Verifier.** Replay the zerocheck and lincheck straight off the shared + /// transcript stream, recovering the one evaluation claim on the committed + /// witness `q_flock`. The PCS then discharges the returned claim. + pub fn verify(&self, n_blocks_log: usize, vs: &mut VerifierState<'_>) -> Result { + let m = self.k_log + n_blocks_log; + let zc_claim = zerocheck::verify(m, vs).map_err(VerifyError::Zerocheck)?; + + let x_ab = x_ab_of(&zc_claim, self.k_log - K_SKIP); + let lc_claim = lincheck::verify( + m, + self.k_log, + K_SKIP, + self.circuit, + &x_ab, + zc_claim.a_eval, + zc_claim.b_eval, + zc_claim.c_eval, + vs, + ) + .map_err(VerifyError::Lincheck)?; + + let claim = reduction_claim(&lc_claim, &x_ab.x_outer); + Ok(ReductionReplay { + claim, + zc_claim, + lc_claim, + }) + } +} + +/// One reduction claim as a tower [`RingSwitchClaim`]: the `2^k_skip` slices and +/// the suffix point they live at, which is the WHOLE multilinear tail of the +/// quirky point (`q_flock` has `2^qflock_vars` words, and the packing prefix is +/// exactly the skipped coordinates, so nothing is split off into it). +fn ring_claim(claim: &SliceClaim, qflock_vars: usize) -> RingSwitchClaim { + assert_eq!( + claim.suffix_point.len(), + qflock_vars, + "ring-switch suffix must span the q_flock cube" + ); + assert_eq!(claim.s_hat_v.len(), PACKING_WIDTH); + RingSwitchClaim { + suffix_point: claim.suffix_point.clone(), + s_hat_v: Some(claim.s_hat_v.clone()), + } +} + +/// Package the prover's reduction claim as a [`RingSwitchOpen`], so the PCS +/// discharges flock's validity in the same opening as the embedder's own point +/// claims. `q_flock` is the `2^qflock_vars`-word slice of the committed stack at +/// `offset`. +pub fn ring_switch_open(qflock_vars: usize, offset: usize, reduced: &SliceClaim) -> RingSwitchOpen { + RingSwitchOpen { + offset, + qflock_vars, + claims: vec![ring_claim(reduced, qflock_vars)], + } +} + +/// Verifier counterpart of [`ring_switch_open`]: package the recovered claim as +/// a [`RingSwitchVerify`], the same statement data. The transmitted opening +/// travels separately. +pub fn ring_switch_verify(qflock_vars: usize, offset: usize, claim: &SliceClaim) -> RingSwitchVerify<'_> { + assert_eq!( + claim.suffix_point.len(), + qflock_vars, + "ring-switch suffix must span the q_flock cube" + ); + RingSwitchVerify { + offset, + qflock_vars, + claims: vec![RingSwitchVerifyClaim { + suffix_point: &claim.suffix_point, + s_hat_v: claim.s_hat_v.as_slice().try_into().expect("ring-switch has 64 slices"), + }], + } +} diff --git a/crates/flock/tests/batch_proving_arithmetic.rs b/crates/flock/tests/batch_proving_arithmetic.rs new file mode 100644 index 000000000..66f85ef2b --- /dev/null +++ b/crates/flock/tests/batch_proving_arithmetic.rs @@ -0,0 +1,204 @@ +//! Standalone batch u64 arithmetic proving, isolated from the VM: wrapping +//! addition, and multiplication wrapping (a `u64` result) or widening (a `u128`). +//! +//! ```text +//! BENCH_REPEAT=3 BENCH_COOLDOWN=2 FLOCK_N_LOG=20 cargo test --release --package flock --test batch_proving_arithmetic -- mul_wrapping_prove_verify --exact --nocapture --include-ignored +//! ``` + +use std::sync::Mutex; +use std::time::Instant; + +use fiat_shamir::transcript::{ProverState, Receiver, Transmitter, VerifierState}; +use flock::arith::{U64Circuit, U64Op}; +use flock::reduction::{min_n_blocks_log, ring_switch_open, ring_switch_verify}; +use pcs::pack::LOG_PACKING; +use pcs::stack_open::{open_batch_mixed_whir_stacked, verify_opening_batch_mixed_whir_stacked}; +use pcs::whir::{INITIAL_FOLDING_FACTOR, LOG_INV_RATE_0}; +use pcs::whir::{commit, config_for_rate}; +use primitives::bench::{Plan, Timing}; +use primitives::{field::F64, pretty_integer, test_rng::Rng}; + +/// Arena phases are process-global, so the benchmarks must not overlap. +static ONE_AT_A_TIME: Mutex<()> = Mutex::new(()); + +#[test] +#[ignore = "manual release benchmark; needs substantial memory"] +fn add_wrapping_prove_verify() { + bench(U64Op::WrappingAdd); +} + +#[test] +#[ignore = "manual release benchmark; needs substantial memory"] +fn mul_wrapping_prove_verify() { + bench(U64Op::WrappingMul); +} + +#[test] +#[ignore = "manual release benchmark; needs substantial memory"] +fn mul_widening_prove_verify() { + bench(U64Op::WideningMul); +} + +fn bench(op: U64Op) { + let _serial = ONE_AT_A_TIME.lock().unwrap_or_else(|poisoned| poisoned.into_inner()); + let (title, unit) = match op { + U64Op::WrappingAdd => ("Wrapping u64 addition", "sums"), + U64Op::WrappingMul => ("Wrapping u64 multiplication", "products"), + U64Op::WideningMul => ("Widening u64 multiplication", "products"), + }; + let requested_n_log: usize = std::env::var("FLOCK_N_LOG") + .ok() + .map(|s| s.parse().expect("FLOCK_N_LOG must be an integer")) + .unwrap_or(16); + let n = 1usize + .checked_shl(requested_n_log as u32) + .expect("FLOCK_N_LOG exceeds the platform usize width"); + let n_log = min_n_blocks_log(n); + + let t = Instant::now(); + let circuit = U64Circuit::new(op); + let setup_ms = t.elapsed().as_secs_f64() * 1e3; + let block = circuit.block(); + let mu = circuit.k_log() + n_log - LOG_PACKING; + assert!( + mu >= 15, + "FLOCK_N_LOG too small: need a committed witness with mu >= 15" + ); + + let mut rng = Rng::new(0x9E37_79B9_7F4A_7C15 ^ n as u64); + let pairs: Vec<(u64, u64)> = (0..n).map(|_| (rng.next_u64(), rng.next_u64())).collect(); + let config = config_for_rate(mu, LOG_INV_RATE_0).expect("WHIR configuration"); + let label = format!("flock-{op:?}-batch").into_bytes(); + + // One full prove pass from the raw pairs, one arena phase, as in + // `batch_proving_hashes`. + zk_alloc::enable_arena(); + let prove_pass = || { + let _phase = zk_alloc::enter_phase(); + let t_pass = Instant::now(); + let t = Instant::now(); + let (z_packed, a_packed, b_packed, z_lincheck) = circuit.generate_witness(&pairs, n_log); + // SAFETY: `F64` is `repr(transparent)` over `u64`. + let q_flock: &[F64] = unsafe { std::slice::from_raw_parts(z_packed.as_ptr().cast(), z_packed.len()) }; + let witness_s = t.elapsed().as_secs_f64(); + assert_eq!(q_flock.len(), 1 << mu); + + let mut ps = ProverState::from_label(&label); + + let t = Instant::now(); + let (commitment, prover_data) = commit(q_flock, mu, INITIAL_FOLDING_FACTOR, LOG_INV_RATE_0); + ps.add_root(&commitment.root); + let commit_s = t.elapsed().as_secs_f64(); + + let t = Instant::now(); + let stage = block.prove_zerocheck(n_log, &z_packed, &a_packed, &b_packed, &mut ps); + let zerocheck_s = t.elapsed().as_secs_f64(); + + let t = Instant::now(); + let reduced = block.prove_lincheck(n_log, stage, &z_lincheck, &mut ps); + let lincheck_s = t.elapsed().as_secs_f64(); + drop((a_packed, b_packed, z_lincheck)); + + let t = Instant::now(); + let ring = ring_switch_open(mu, 0, &reduced); + open_batch_mixed_whir_stacked( + &mut ps, + mu, + q_flock, + &prover_data, + &config, + &[], + std::slice::from_ref(&ring), + ); + let open_s = t.elapsed().as_secs_f64(); + + let proof = ps.into_proof(); + let pass_s = t_pass.elapsed().as_secs_f64(); + (proof, [witness_s, commit_s, zerocheck_s, lincheck_s, open_s, pass_s]) + }; + + let env = |key: &str, default: usize| { + std::env::var(key).map_or(default, |s| { + s.parse().unwrap_or_else(|_| panic!("{key} must be an integer")) + }) + }; + let plan = Plan::new(env("BENCH_REPEAT", 1), env("BENCH_COOLDOWN", 2) as u64); + let mut stages: [Timing; 6] = std::array::from_fn(|_| Timing::default()); + let (transcript, _) = plan.warm_then_measure(|_final_pass| { + let (out, secs) = prove_pass(); + for (timing, s) in stages.iter_mut().zip(secs) { + timing.push(s); + } + out + }); + // The warmup pass also pushed a sample; drop the leading one per stage. + let [witness, commit_stage, zerocheck, lincheck, open, pass] = stages.map(|t| { + let mut kept = Timing::default(); + for &s in &t.samples()[1..] { + kept.push(s); + } + kept + }); + + let (_, verify_time) = Plan::new(plan.repeat, 0).measure_quiet(|_final_pass| { + let mut vs = VerifierState::from_label(&label, &transcript); + let root = vs.next_root().expect("commitment root"); + let replay = block.verify(n_log, &mut vs).expect("Flock reduction verifies"); + let ring = ring_switch_verify(mu, 0, &replay.claim); + assert!( + verify_opening_batch_mixed_whir_stacked( + &mut vs, + &config, + mu, + 1 << INITIAL_FOLDING_FACTOR, + &root, + &[], + std::slice::from_ref(&ring) + ) + .is_ok(), + "stacked PCS opening verifies" + ); + vs.finish().expect("transcript fully consumed"); + }); + + let pass_s = pass.mean(); + let share = |s: f64| format!("{:>5.1}%", 100.0 * s / pass_s); + let ms = |t: &Timing| format!("{:>8.1} ms{:<9}{}", t.mean() * 1e3, t.spread(), share(t.mean())); + let named = witness.mean() + commit_stage.mean() + zerocheck.mean() + lincheck.mean() + open.mean(); + println!( + "\nFlock {title} batch proving, {} {unit} (2^{n_log} slots)", + pretty_integer(n) + ); + println!( + " block : 2^{} bits, {} constrained", + circuit.k_log(), + pretty_integer(circuit.useful_bits()) + ); + println!(" setup (circuit, excluded) : {setup_ms:>8.1} ms"); + println!(" witness-gen : {}", ms(&witness)); + println!(" commit : {}", ms(&commit_stage)); + println!(" zerocheck : {}", ms(&zerocheck)); + println!(" lincheck : {}", ms(&lincheck)); + println!(" pcs opening : {}", ms(&open)); + println!( + " other : {:>8.1} ms{:<9}{}", + (pass_s - named) * 1e3, + "", + share(pass_s - named) + ); + println!(" ------------------------------------------"); + println!( + " prove TOTAL (witness included) : {:>8.1} ms{}", + pass_s * 1e3, + pass.spread() + ); + println!( + " verify : {:>8.1} ms", + verify_time.mean() * 1e3 + ); + println!( + " throughput : {:>14} {unit}/s{}", + pretty_integer((n as f64 / pass_s).round() as u64), + pass.spread() + ); +} diff --git a/crates/flock/tests/batch_proving_hashes.rs b/crates/flock/tests/batch_proving_hashes.rs index 8d77d26af..e183dfdaa 100644 --- a/crates/flock/tests/batch_proving_hashes.rs +++ b/crates/flock/tests/batch_proving_hashes.rs @@ -21,7 +21,6 @@ use primitives::{field::F64, pretty_integer, test_rng::Rng}; #[test] #[ignore = "manual release benchmark; needs a large-stack worker and substantial memory"] fn hash_batch_prove_verify() { - // The XMSS n=820 workload executes about 2^17 BLAKE2s compressions. let requested_n_log: usize = std::env::var("FLOCK_N_LOG") .ok() .map(|s| s.parse().expect("FLOCK_N_LOG must be an integer")) @@ -86,7 +85,15 @@ fn hash_batch_prove_verify() { let t = Instant::now(); let ring = ring_switch_open(n, 0, &reduced); - open_batch_mixed_whir_stacked(&mut ps, mu, q_flock, &prover_data, &config, &[], &ring); + open_batch_mixed_whir_stacked( + &mut ps, + mu, + q_flock, + &prover_data, + &config, + &[], + std::slice::from_ref(&ring), + ); let open_s = t.elapsed().as_secs_f64(); let prove_s = t_prove.elapsed().as_secs_f64(); @@ -138,7 +145,7 @@ fn hash_batch_prove_verify() { 1 << INITIAL_FOLDING_FACTOR, &root, &[], - &ring + std::slice::from_ref(&ring) ) .is_ok(), "stacked PCS opening verifies" @@ -181,8 +188,4 @@ fn hash_batch_prove_verify() { pretty_integer(compressions_per_second), prove.spread() ); - println!( - " (~{:.1} XMSS/s equivalent at 146 compressions/signature)", - n as f64 / prove_s / 146.0 - ); } diff --git a/crates/lean_compiler/Cargo.toml b/crates/lean_compiler/Cargo.toml deleted file mode 100644 index 0de4a791b..000000000 --- a/crates/lean_compiler/Cargo.toml +++ /dev/null @@ -1,16 +0,0 @@ -[package] -name = "lean_compiler" -version.workspace = true -edition.workspace = true -publish = false - -[lints] -workspace = true - -[dependencies] -primitives.workspace = true -lean_vm.workspace = true - -[dev-dependencies] -rand.workspace = true -bincode.workspace = true diff --git a/crates/lean_compiler/snark_lib.py b/crates/lean_compiler/snark_lib.py deleted file mode 100644 index 9715f64a1..000000000 --- a/crates/lean_compiler/snark_lib.py +++ /dev/null @@ -1,236 +0,0 @@ -# Import this in zkDSL .py files (`from snark_lib import *`) so editors and -# linters resolve the builtins. The compiler skips the import; it does not -# include other source files (single-file programs only). Resolved through -# `extraPaths` in the root pyrightconfig.json; see zkDSL.md. A guest is not a -# runnable Python file: its `*_PLACEHOLDER` names are filled in at compile time, -# so `import`ing one raises NameError. - -from typing import Any, Optional - -Const = Any -"""Parameter annotation: `def f(k: Const, x):`, where `k` is a compile-time -argument; the compiler specializes the function per distinct constant.""" - - -class _Elt: - """A 192-bit machine word in E = GF(2^192), represented as a cubic tower - over K = GF(2^64). Indices and addresses are K-valued powers of GEN, i.e. - "in the exponent": `GEN ** k` is the k-th index and `x * GEN` its successor. - A heap pointer is K-valued too; `buf[i]` is the write-once cell at `buf * i`.""" - - def __add__(self, other): # field addition = XOR - _ = other - return _Elt() - - __radd__ = __add__ - - def __mul__(self, other): # tower-field product - _ = other - return _Elt() - - __rmul__ = __mul__ - - def __truediv__(self, other): # field division a / b = a · b⁻¹ (single slash) - _ = other - return _Elt() - - __rtruediv__ = __truediv__ - - def __pow__(self, k: int): - _ = k - return _Elt() - - def __getitem__(self, idx): # heap read m[self · idx] - _ = idx - return _Elt() - - def __setitem__(self, idx, value): # heap store m[self · idx] (write-once) - _ = idx, value - - -def f192(c0: int, c1: int, c2: int) -> _Elt: - """Construct a field constant from its three little-endian GF(2^64) limbs.""" - _ = c0, c1, c2 - return _Elt() - - -GEN = _Elt() -"""The fixed generator g = x of K^× = GF(2^64)^× (order 2^64 - 1).""" - - -def hint_decompose_bits(bits, value, nbits: int) -> None: - """Computed advice: the prover writes the `nbits` bits of `value` into the - `bits` buffer. UNCONSTRAINED: the caller must check booleanity and that the - bits reconstruct `value` (a range check that `value < 2^nbits`).""" - _ = bits, value, nbits - - -def hint_decompose_bits_exponent(bits, x, nbits: int) -> None: - """Computed advice: the prover writes the `nbits` bits of n, where x = g^n - (recovered by a bounded discrete log at witness generation), into `bits`. - UNCONSTRAINED: the caller checks booleanity and Π g^(bit_j 2^j) == x.""" - _ = bits, x, nbits - - -def hint_log2_ceil(bits, nbits: int, floor: int) -> _Elt: - """Computed advice: returns `g^max(log2_ceil(v), floor)`, where `v` is the - integer the `nbits`-cell `bits` buffer decodes to. The prover fills it at - witness-generation; it is UNCONSTRAINED, so the caller must verify it (see the - log2_ceil_in_the_exponent wrapper in the recursion guest). log2 = base-2 log of the integer, NOT the - discrete log base g that `log(...)` means.""" - _ = bits, nbits, floor - return _Elt() - - -def const(e): - """`if const(a == b):` asks for the branch to be decided while compiling, and - `const(e)` in a value position asks for `e` itself to be read that way. - - Two things follow. The condition must be decidable then (both sides - compile-time integers), and it is read with INTEGER arithmetic, which is the - regime a compile-time constant lives in. Without the wrapper, a condition - whose integer and field readings disagree is rejected rather than silently - decided one way, since `+` is XOR in a value. A folded branch is - straight-line code, so unlike a runtime branch its bindings outlive it.""" - return e - - -def log(x) -> int: - """The discrete log base GEN: `x = GEN ** log(x)`. Only meaningful inside - a range-check assert (`assert log(x) < log(GEN ** k)`, equivalently - `assert log(x) < k`, proves `x ∈ {GEN**0, …, GEN**(k-1)}` in 3 cycles), - or as the scrutinee of `match`.""" - _ = x - return 0 - -# @inline decorator (does nothing in Python execution) -def inline(fn): - return fn - - -def match(value: int, *args): - """A `match` with generated arms: `match(log(x), range(a, b), - lambda i: …, …)` expands to one arm per integer of the contiguous ranges - (which must start at 0), the lambda applied to the concrete value; the - results bind to the assignment targets. In Python execution, finds the - matching range and calls its lambda.""" - for i in range(0, len(args), 2): - rng, fn = args[i], args[i + 1] - if value in rng: - return fn(value) - raise AssertionError(f"value {value} not in any range") - - -def mul_range(start, stop) -> list: - """The loop counter walked in the exponent: from element `start` to `stop` - (exclusive), ×GEN each iteration. `start` is a compile-time power of GEN - (`1`, `GEN`, or `GEN ** k`); `stop` is one too, or else a runtime g-power - element the walk must be able to reach.""" - _ = start, stop - return [] - - -def unroll(a: int, b: int) -> range: - """Compile-time unrolling: the body is replicated for i = a, …, b-1, the - counter substituted as an integer literal (usable as a stack index, slice - bound, or `Const` argument). Bounds are compile-time integers, including - `Const` parameters.""" - return range(a, b) - - -def HeapBuf(n) -> _Elt: - """Allocate a fresh, disjoint heap buffer; evaluates to its pointer (a - fresh g-power). `n` is either an integer literal (compile-time size), or a - runtime value carrying the cell count *in the exponent*: `g^k` allocates - `k` cells, so a size derived from a g-power count is plain field arithmetic - (`HeapBuf(cnt * cnt)` is `2·log(cnt)` cells). Allocation is a prover - convenience, so an under-size only trips write-once.""" - _ = n - return _Elt() - - -def StackBuf(n: int) -> _Elt: - """Allocate `n` consecutive frame (stack) cells. A size-2 StackBuf holds a - 256-bit value and is a valid `blake2s` operand.""" - _ = n - return _Elt() - - -def addr(buf) -> _Elt: - """The g-address of a StackBuf's first cell, so a frame run can be pointed - at: `p = addr(sb)` binds a pointer whose `p[i]` / `p * GEN ** k` behave like - a HeapBuf's, while `sb[k]` itself stays a direct frame cell. Costs one - materialization of `fp` per function (2 DEREFs, amortized; free in `main`). - Only valid as the whole right-hand side of an assignment.""" - _ = buf - return _Elt() - - -def hint_witness(dest, name: Optional[str] = None) -> Any: - """Take the next ENTRY (a slice of values) of the named prover witness - stream. - - Two forms. As a statement, `hint_witness(dest, "name")` fills `dest` (a - StackBuf, or a StackBuf/HeapBuf slice of any length). As an expression, - `x = hint_witness("name")` binds ONE value and needs no destination, the - entry then having to hold exactly one value. - - The same symbol may be hinted many times, each call popping the next entry - (`Program::set_witness`; test programs declare one `# witness name: v1, …` - line per entry). Zero cycles either way, and the values are completely - UNCONSTRAINED: the program must constrain them itself (asserts, range - checks, hashes).""" - _ = dest, name - return _Elt() - - -def assert_in_k(a, b) -> None: - """Prove that both machine words are GF(2^64)-valued. This is the sole - packing-related intrinsic and lowers to one untaken JUMP.""" - _ = a, b - - -def hint_f192_limbs(dest, value) -> None: - """Computed advice: write the first `len(dest)` GF(2^64) coordinate limbs - of `value` into a 1-to-3-cell StackBuf. UNCONSTRAINED; callers bind the - result with `assert_in_k` and field reconstruction.""" - _ = dest, value - - -def blake2s( - a, - b, - out, - *, - cv=None, - counter: Optional[int] = None, - final: Optional[int] = None, - last_node: int = 0, - md=None, -) -> None: - """One standard BLAKE2s compression of the two 256-bit message operands - `a`, `b`, written into the 2-cell run `out` (write-once: if `out` was - already written, this asserts it equals the chaining value). - - With no keywords this hashes exactly 64 bytes: the parameterized BLAKE2s-256 - initial chaining value, byte counter 64, final-block flag set. That is - `blake2s(a || b)`, the form every Fiat-Shamir step and Merkle node uses. - - For a longer message, drive the blocks yourself. `counter` is the CUMULATIVE - byte count through this block (`64 * whole_blocks_before + bytes_in_this_block`) - and `final=1` marks the last block; `cv` carries the previous block's output - and requires one of `counter`, `final`, `last_node` or `md`. Setting any of - `counter`, `final` or `last_node` makes `final` default - to 0, so a single short block needs `counter=, final=1`. Bytes past the - block's real length must be zero-filled by the program. `last_node` is - BLAKE2s's tree-mode `f1` and is 0 everywhere here. - - `md` is the whole 128-bit metadata word as a value the program computed, for - a hash whose block count is only known at run time. It replaces `counter`, - `final` and `last_node` (giving both is an error), and it must not name a - cell of `out`. - - Message, chaining-value, and output operands are size-2 StackBufs or - 2-cell slices `buf[lo:hi]` of larger StackBufs or HeapBufs (heap inputs are - bridged through the stack, one DEREF per cell).""" - _ = a, b, out, cv, counter, final, last_node, md diff --git a/crates/lean_compiler/src/ast.rs b/crates/lean_compiler/src/ast.rs deleted file mode 100644 index 73e75f3e2..000000000 --- a/crates/lean_compiler/src/ast.rs +++ /dev/null @@ -1,464 +0,0 @@ -//! The surface AST produced by the parser: expressions, statements, functions. - -use primitives::field::F192; - -/// An expression. Arithmetic is the field's own: `+` is `XOR`, `*` is `MUL`. -#[derive(Clone, Debug, PartialEq)] -pub enum Expr { - /// Integer / field literal: the source syntax provides a raw 128-bit value, - /// embedded into the low two limbs of the 192-bit tower element (`c2 = 0`). - Lit(u128), - /// The generator `g`, written `GEN`. A logical index `i` rides the exponent - /// as `gⁱ`, so `GEN` is the unit step. - Gen, - /// The field constant `g^k`. The exponent is a `u128`, so an index can be a - /// large logical value. - GPow(u128), - /// `GEN ** e` for a compile-time integer *expression* `e` (an `unroll` - /// variable, a constant, index arithmetic of those), resolved to a concrete - /// `g^k` at lowering. Lets `buf[GEN ** i]` name cell `i` with no cursor. - GenPow(Box), - /// `base ** e` with a non-`GEN` base and a compile-time exponent, by - /// square-and-multiply at lowering: integer arithmetic in an index or bound - /// position, field arithmetic in a value. The base may be runtime. - Pow(Box, Box), - /// A variable in scope. - Var(String), - Add(Box, Box), - Mul(Box, Box), - /// Integer subtraction, **compile-time only**: field subtraction is `+` - /// (XOR), so `-` means something only in index space. Using one as a runtime - /// field value is an error. - Sub(Box, Box), - /// Integer floor-division `a // b` and remainder `a % b`, **compile-time - /// only** (the field has no integer division). Valid where an index / - /// slice bound / `Const` argument is expected, or as a folded `if` - /// condition; using one as a runtime field value is an error. - Div(Box, Box), - Mod(Box, Box), - /// Field division `a / b` (single slash): a **runtime** field operation, - /// `a · b⁻¹`. Lowered to one `MUL` whose quotient operand is unset, so the - /// write-once back-solve fills it with `a · b⁻¹` and the `MUL` constraint - /// pins `quotient · b == a` (§range-check trick). No hint: the inverse is - /// nondeterministic but the constraint binds it. `b == 0` is rejected, - /// including `0 / 0`; `1 / b` therefore also enforces `b != 0`. Distinct - /// from the compile-time `//` ([`Expr::Div`]). - FieldDiv(Box, Box), - /// Single-return function call in expression position. - Call(String, Vec), - /// `HeapBuf(n)`: allocate a heap buffer of `n` cells; evaluates to its pointer. - HeapBuf(u64), - /// `HeapBuf(size)` with a *runtime* size carried **in the exponent**: the - /// buffer holds `k` cells where `size = g^k` (so a size derived from a - /// g-power count `n` is plain field arithmetic: `HeapBuf(n * n * GEN**2)` - /// is `2·log(n) + 2` cells). The allocation is a prover convenience (like - /// every base pointer), so an under-size only hurts the prover: - /// overlapping regions trip write-once. Evaluates to the pointer. - HeapBufDyn(Box), - /// `StackBuf(n)`: allocate `n` *consecutive* frame (stack) cells, bound as a - /// stack value. Its cells `sa[0..n]` are written/read directly (no heap deref), - /// and a size-2 `StackBuf` is a valid `blake2s` operand (the four 64-bit hash - /// words live as two lanes in each of two consecutive 128-bit cells). - StackBuf(u64), - /// `arr[idx]`: read a cell. For a heap `arr` (a pointer), `m[arr·idx]` (idx a - /// g-power). For a [`Expr::StackBuf`]: the frame cell `base + idx` (idx a - /// compile-time integer), read directly. - Index(Box, Box), - /// `buf[lo:hi]`: a run of cells of a [`Expr::StackBuf`] (frame cells - /// `base+lo..base+hi`) or of a [`Expr::HeapBuf`] (heap cells - /// `ptr·g^lo..ptr·g^hi`), with compile-time integer bounds (`hi` - /// exclusive). Only meaningful as a `blake2s` operand, where it must span - /// exactly 2 cells (one 256-bit value). - Slice(Box, Box, Box), - /// `[a, b, …]`: an initialized [`Expr::StackBuf`], so `x = [a, b]` allocates - /// a StackBuf of the element count and writes each element in place, sugar - /// for the alloc-then-store idiom. Only meaningful as the RHS of a plain - /// assignment (inside a function; a *top-level* `NAME = […]` is a constant - /// array, see [`Ast::const_arrays`]). - ListLit(Vec), -} - -/// A statement, with the source line it came from. The line is what every -/// lowering diagnostic names and what the pc-to-line table is built from: the -/// AST is the only place that still knows it, since lowering works in frame -/// cells and program counters. -#[derive(Clone, Debug)] -pub struct Stmt { - pub line: u32, - pub kind: StmtKind, -} - -impl Stmt { - pub fn new(line: u32, kind: StmtKind) -> Self { - Self { line, kind } - } - - /// A statement the compiler synthesized (a loop helper's body, a desugared - /// tail call): it inherits the line of whatever it was generated for. - pub fn at(&self, kind: StmtKind) -> Self { - Self { line: self.line, kind } - } -} - -/// What a statement does. -#[derive(Clone, Debug)] -pub enum StmtKind { - /// `x = expr` (immutable binding). - Let(String, Expr), - /// `x, y, … = f(args)`: call with multiple returns. - LetTuple(Vec, String, Vec), - /// `assert a == b`: a proof-enforced equality. - AssertEq(Expr, Expr), - /// `assert a != b`. Lowers to `x = a + b`, a hinted `inv = x⁻¹`, `p = x·inv` - /// and `SET p = 1`: `x = 0` forces `p = 0` whatever the hint, so the - /// write-once conflict is the assertion. See `FnLower::lower_assert_ne`. - AssertNe(Expr, Expr), - /// `assert log X < log Y`, or `assert log X < k`: a range check in the - /// exponent, proving `x < k` for `X = g^x`. See `FnLower::lower_assert_lt`. - AssertLt(Expr, LtBound), - /// `f(args)` as a statement (returns discarded). - Call(String, Vec), - /// `hint_witness(dest, "name")`: fill `dest` from the next entry of the named - /// witness stream (`Program::set_witness`), whose length must match. The same - /// name may be hinted many times, each call popping the next entry. Zero - /// cycles and completely unconstrained, so the program must constrain the - /// values itself. - HintWitness { dest: Expr, name: String }, - /// `x = hint_witness("stream")`: one hinted value bound to a name, as - /// unconstrained as any other hint. The run form above needs a destination - /// that already exists, so a lone scalar otherwise costs a one-cell - /// `StackBuf`, a slice of it, and a read back out. - LetHintWitness { name: String, stream: String }, - /// `print("label", expr)`: a prover-side debug print, witness generation only. - Print { label: String, value: Expr }, - /// `if lhs == rhs:` (`eq`) / `if lhs != rhs:`, with an optional `else` (an - /// `elif` parses as an `else` holding a nested `if`). One conditional `JUMP` - /// on the XOR of the sides. Bindings inside a branch are local to it, and the - /// branches communicate through write-once memory, only one of them running. - /// See `FnLower::lower_if`. - If { - eq: bool, - lhs: Expr, - rhs: Expr, - then: Vec, - els: Vec, - /// Written `if const(a == b):`. The author asks for the branch to be - /// decided while compiling, so a condition that cannot be decided then is - /// an error rather than a runtime test, and one whose integer and field - /// readings disagree is an error rather than a silent choice. - force_const: bool, - }, - /// `targets = match(log(x), range(a, b), lambda i: expr, …)`: the one dispatch - /// construct. Arm `j` is the lambda body with the parameter replaced by the - /// literal `j`, expanded at parse time, and `x = g^j` runs arm `j` through a - /// trampoline table in the bytecode. Every arm writes the same cells, exactly - /// one of them running, so a target may be a name bound at the join or a - /// `StackBuf` element written in place. See `FnLower::lower_match`. - Match { - targets: Vec, - x: Expr, - arms: Vec, - }, - /// `arr[idx] = value`: store into a heap cell (write-once). - Store(Expr, Expr, Expr), - /// `for i in mul_range(GEN ** lo, stop)`: the counter rides the exponent as - /// `gⁱ`, advancing by `×g` from `g^lo` until it reaches `stop`, which is not - /// itself executed. There is no step knob, the bounds being field elements - /// (`mul_range(1, GEN ** 10)` runs ten times). A runtime `stop` must be known - /// reachable: range-check its log, or the walk never terminates. - For { - var: String, - lo: u64, - hi: ForBound, - body: Vec, - }, - /// `for i in unroll(a, b)`: the body emitted `b − a` times with `i` - /// substituted by each literal, so `i` is usable wherever a literal is. No - /// call, no frame, no counter, at the price of code size. The bounds are - /// compile-time integer expressions evaluated at lowering, after - /// `Const`-parameter specialization, so `unroll(0, n)` with `n: Const` works. - Unroll { - var: String, - lo: Expr, - hi: Expr, - body: Vec, - }, - /// `return e, …` (a bare `return` is the empty vector). - Return(Vec), - /// Internal, from loop lowering: `if lhs != rhs: callee(args)` in tail - /// position, dispatched by `JUMP`'s nonzero test. - CallIfNe(Expr, Expr, String, Vec), -} - -/// A `mul_range` stop bound: a compile-time `GEN ** k`, or a runtime g-power -/// element (evaluated once in the enclosing scope and threaded through the -/// loop helper as a parameter). -#[derive(Clone, Debug)] -pub enum ForBound { - Const(u64), - Runtime(Expr), -} - -/// A range-check bound (`assert log X < …`): a compile-time exponent, or a -/// runtime `g^n`. Same gadget either way, only the cell holding `g^{k-1}` -/// differing. -/// -/// A runtime bound loses the `k ≤ 2^MIN_LOG_MEM` cap the compiler would check, -/// so the PROGRAM owes it: range-check the bound itself first, or `log X < log n` -/// bounds `X` only by the prover-announced memory size. -#[derive(Clone, Debug)] -pub enum LtBound { - Const(u64), - Runtime(Expr), -} - -/// Compile-time representation of a runtime parameter or return value. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum Shape { - /// One ordinary field element or address cell. Heap buffers use this shape: - /// allocation happens in the callee and only their pointer crosses. - Scalar, - /// A compile-time-sized run of consecutive frame cells. - StackBuf(u32), -} - -impl Shape { - /// Number of physical call-frame cells occupied by this value. - pub(crate) fn cells(self) -> u32 { - match self { - Self::StackBuf(n) => n, - Self::Scalar => 1, - } - } -} - -/// A function parameter and its calling convention. -#[derive(Clone, Debug)] -pub struct Param { - pub name: String, - pub kind: ParamKind, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ParamKind { - /// Substituted at the call site, with no runtime argument cell. - Const, - Runtime(Shape), -} - -impl Param { - pub(crate) fn shape(&self) -> Shape { - match self.kind { - ParamKind::Const => Shape::Scalar, - ParamKind::Runtime(shape) => shape, - } - } -} - -/// A function definition. `main` is the entry point. -#[derive(Clone, Debug)] -pub struct Func { - pub name: String, - /// A function with a `Const` parameter is a template, specialized per - /// distinct constant tuple before lowering. - pub params: Vec, - /// Compile-time shape of each source-level return value. Stack buffers use - /// multiple physical ABI cells; everything else uses one cell. - pub return_shapes: Vec, - pub body: Vec, - /// `@inline`: expand at each call site instead of emitting a call, so the - /// frame and the argument and return plumbing vanish. The body must be a - /// single tail `return`, and is never lowered standalone. - pub inline: bool, -} - -impl Func { - pub(crate) fn has_const_params(&self) -> bool { - self.params.iter().any(|p| p.kind == ParamKind::Const) - } - - pub(crate) fn param_shapes(&self) -> impl Iterator + '_ { - self.params.iter().map(Param::shape) - } -} - -/// A whole program: a set of functions including `main`. -#[derive(Clone, Debug)] -pub struct Ast { - pub funcs: Vec, - /// Top-level constant arrays `NAME = [a, b, c]`, in declaration order. - /// Indexed `NAME[i]` and measured `len(NAME)` at compile time only, `i` being - /// a literal, a constant or an `unroll` variable. Unlike a scalar constant - /// these are not textually substituted, but resolved at lowering. - pub const_arrays: Vec<(String, Vec)>, -} - -// Free-variable analysis: pure AST, no lowering state. Its one consumer is the -// `for` desugaring, which needs the names a loop body reads from outside itself. - -use std::collections::HashSet; - -/// Collect variable references in `e` into `refs` (in source order). -fn free_vars_expr<'a>(e: &'a Expr, refs: &mut Vec<&'a str>) { - match e { - Expr::Var(v) => refs.push(v.as_str()), - Expr::Add(a, b) - | Expr::Mul(a, b) - | Expr::Sub(a, b) - | Expr::Div(a, b) - | Expr::FieldDiv(a, b) - | Expr::Mod(a, b) - | Expr::Index(a, b) - | Expr::Pow(a, b) => { - free_vars_expr(a, refs); - free_vars_expr(b, refs); - } - Expr::Slice(a, lo, hi) => { - free_vars_expr(a, refs); - free_vars_expr(lo, refs); - free_vars_expr(hi, refs); - } - Expr::Call(_, args) | Expr::ListLit(args) => args.iter().for_each(|a| free_vars_expr(a, refs)), - Expr::HeapBufDyn(sz) | Expr::GenPow(sz) => free_vars_expr(sz, refs), - Expr::Lit(_) | Expr::Gen | Expr::GPow(_) | Expr::HeapBuf(_) | Expr::StackBuf(_) => {} - } -} - -/// Collect references in `s` into `refs` and names it binds into `bound`. -/// Every name the block binds, ignoring scope. [`free_vars_stmt`] deliberately -/// does not answer this: its `bound` set is scoped, so an arm-local binding is -/// discarded with the arm. -pub(crate) fn binds_anywhere<'a>(body: &'a [Stmt], out: &mut HashSet<&'a str>) { - for s in body { - match &s.kind { - StmtKind::Let(n, _) | StmtKind::LetHintWitness { name: n, .. } => { - out.insert(n.as_str()); - } - StmtKind::LetTuple(ns, ..) => ns.iter().for_each(|n| { - out.insert(n.as_str()); - }), - StmtKind::Match { targets, .. } => targets.iter().for_each(|t| { - if let Expr::Var(n) = t { - out.insert(n.as_str()); - } - }), - StmtKind::If { then, els, .. } => { - binds_anywhere(then, out); - binds_anywhere(els, out); - } - StmtKind::For { var, body, .. } | StmtKind::Unroll { var, body, .. } => { - out.insert(var.as_str()); - binds_anywhere(body, out); - } - _ => {} - } - } -} - -/// Collect references from a block whose bindings do not escape. -fn scoped_vars<'a>(body: &'a [Stmt], refs: &mut Vec<&'a str>) { - let mut inner = HashSet::new(); - for s in body { - free_vars_stmt(s, refs, &mut inner); - } -} - -pub(crate) fn free_vars_stmt<'a>(s: &'a Stmt, refs: &mut Vec<&'a str>, bound: &mut HashSet<&'a str>) { - match &s.kind { - StmtKind::Let(n, e) => { - free_vars_expr(e, refs); - bound.insert(n.as_str()); - } - StmtKind::LetTuple(ns, _, args) => { - args.iter().for_each(|a| free_vars_expr(a, refs)); - ns.iter().for_each(|n| { - bound.insert(n.as_str()); - }); - } - StmtKind::AssertEq(a, b) | StmtKind::AssertNe(a, b) => { - free_vars_expr(a, refs); - free_vars_expr(b, refs); - } - StmtKind::AssertLt(e, bound) => { - free_vars_expr(e, refs); - if let LtBound::Runtime(b) = bound { - free_vars_expr(b, refs); - } - } - StmtKind::HintWitness { dest, .. } => free_vars_expr(dest, refs), - StmtKind::LetHintWitness { name, .. } => { - bound.insert(name.as_str()); - } - StmtKind::Print { value, .. } => free_vars_expr(value, refs), - StmtKind::If { - lhs, - rhs, - then, - els, - force_const, - .. - } => { - free_vars_expr(lhs, refs); - free_vars_expr(rhs, refs); - // An arm's bindings are local to it (`zkDSL.md` §Control flow), so each - // gets its own scope. Sharing one made a name rebound in ONE arm count - // as loop-local everywhere, so the OUTER binding the other arm reads - // was never captured and a legal program failed with `unbound - // variable`. Over-collecting into `refs` is harmless: a capture naming - // nothing in the enclosing scope is dropped. - // `lower_if` FOLDS a compile-time condition and runs the taken branch - // without `scoped`, so its bindings persist exactly like an `unroll` - // body's. Modelling that as scoped over-captured, and the loop's own - // self-call then read a name the folded arm had rebound to a - // `StackBuf`. Both sides literal is the syntactic half of that test. - if *force_const || matches!((lhs, rhs), (Expr::Lit(_), Expr::Lit(_))) { - then.iter().for_each(|s| free_vars_stmt(s, refs, bound)); - els.iter().for_each(|s| free_vars_stmt(s, refs, bound)); - } else { - scoped_vars(then, refs); - scoped_vars(els, refs); - } - } - StmtKind::Match { targets, x, arms } => { - free_vars_expr(x, refs); - arms.iter().for_each(|a| free_vars_expr(a, refs)); - for t in targets { - match t { - Expr::Var(n) => { - bound.insert(n.as_str()); - } - // A store target is READ, not bound: `sb[i], e = …` needs `sb`. - other => free_vars_expr(other, refs), - } - } - } - StmtKind::CallIfNe(a, b, _, args) => { - free_vars_expr(a, refs); - free_vars_expr(b, refs); - args.iter().for_each(|e| free_vars_expr(e, refs)); - } - StmtKind::Call(_, args) => args.iter().for_each(|a| free_vars_expr(a, refs)), - StmtKind::Store(arr, idx, val) => { - free_vars_expr(arr, refs); - free_vars_expr(idx, refs); - free_vars_expr(val, refs); - } - StmtKind::Return(es) => es.iter().for_each(|e| free_vars_expr(e, refs)), - StmtKind::For { hi, body, .. } => { - if let ForBound::Runtime(b) = hi { - free_vars_expr(b, refs); - } - // A nested loop's body becomes its own function, so neither its - // counter nor its bindings exist out here. `unroll` below is the - // opposite: it replicates straight-line code into THIS scope, so its - // bindings really do persist and it keeps the shared set. - scoped_vars(body, refs); - } - StmtKind::Unroll { var, lo, hi, body } => { - free_vars_expr(lo, refs); - free_vars_expr(hi, refs); - bound.insert(var.as_str()); - body.iter().for_each(|s| free_vars_stmt(s, refs, bound)); - } - } -} diff --git a/crates/lean_compiler/src/filler.rs b/crates/lean_compiler/src/filler.rs deleted file mode 100644 index 5c4f1c66f..000000000 --- a/crates/lean_compiler/src/filler.rs +++ /dev/null @@ -1,58 +0,0 @@ -//! Fill blocks: extra rows so every table's height is a power of two. -//! -//! A table is proven over a power-of-two number of rows, so a table whose execution -//! needs fewer has to make up the difference. The alternative to this module is padding -//! rows, which are not real rows: they put default tuples on the bus that nothing -//! matches, so the verifier has to be told each table's real row count and divide those -//! tuples back out. That correction was the most delicate part of the bus argument, and -//! it existed only for the instruction tables; unread memory cells and unexecuted -//! program entries already need nothing, their seed and finalize tuples cancelling. -//! -//! So run the difference off instead. Every program carries, past `main`'s halt, one -//! block per table per size in `lean_vm::cpu::filler::SIZES`: that many dummy -//! instructions of the table's opcode, then a `JUMP` back to the block's own first -//! instruction. Each block is therefore a cycle, and no program code enters one: on the -//! bus its state tuples cancel against each other rather than against the program's -//! chain, so it can be traversed any number of times, and the interpreter walks the -//! blocks itself once the program has halted -//! (`lean_vm::cpu::Program::execute`). -//! -//! Nothing in a block counts, tests, or allocates: a traversal of the size-`s` block -//! costs exactly `s + 1` rows, `s` of its table and one `JUMP`. That is where the sizes -//! earn their keep: powers of two make any fill reachable exactly, while the bulk rides -//! the largest block at one jump per 128 rows. A table already on a power of two is -//! never entered. - -/// What a table's dummy instruction is: the cheapest instruction of that opcode that can -/// be executed any number of times in one frame, given write-once memory. All but -/// `Blake2s` name a single scratch cell as every operand, so the value they write there is -/// the value already there (`FnLower::lower_filler_blocks` fixes the frame offsets). -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum FillerOp { - /// `XOR s, s -> s`: pins the scratch cell to `m[s] + m[s] = 0`. - Xor, - /// `MUL s, s -> s`: pins it to `m[s]^2`, so to `0` given the above. - Mul, - /// `SET s = 0`. - Set, - /// `DEREF` through the frame's pointer cell, which the interpreter sets to `g^0`, so - /// the address is memory cell `0` and the value read is the public input's. - Deref, - /// `JUMP` on a cell nothing ever writes: a zero condition falls through, so the row - /// fills the table without closing the block's cycle. - Jump, - /// One compression of message and chaining-value cells nothing ever writes, its - /// digest placed clear of them, so every traversal compresses the same input. - Blake2s, -} - -/// The tables, in `lean_vm::cpu::Stats::TABLES` order, which is how the solver indexes -/// them. -pub const TABLES: [(u8, FillerOp); 6] = [ - (0, FillerOp::Xor), - (1, FillerOp::Mul), - (2, FillerOp::Set), - (3, FillerOp::Deref), - (4, FillerOp::Jump), - (5, FillerOp::Blake2s), -]; diff --git a/crates/lean_compiler/src/ir.rs b/crates/lean_compiler/src/ir.rs deleted file mode 100644 index 29d5a8ec3..000000000 --- a/crates/lean_compiler/src/ir.rs +++ /dev/null @@ -1,139 +0,0 @@ -//! Lowered intermediate instructions and hints, between the AST and final assembly. - -use super::*; - -pub(crate) type Off = u32; - -/// A `SET` immediate: a field constant, or a function entry address resolved -/// once entry program counters are fixed. -#[derive(Clone, Debug)] -pub(crate) enum KVal { - /// The stride to the next frame in a reserved loop run. - FrameSize, - /// A 192-bit machine-word constant. Source literals fill only c0/c1, while - /// compiler-generated constants may use the full field. - Const(F192), - Entry(String), - /// The halt sentinel pc `g^{B-1}` (last bytecode slot), fixed once the - /// padded bytecode size `B` is known. `main` jumps here to terminate. - EndSentinel, - /// An intra-function jump target: the `i`-th instruction of the function - /// this `SET` belongs to, resolved to `g^{entry + i}` once entry pcs are - /// fixed. Emitted with a placeholder by the `if`/`else` lowering and - /// backpatched ([`FnLower::patch_local`]). - Local(u32), -} - -#[derive(Clone, Debug)] -pub(crate) struct LInstr { - pub(crate) op: LOp, - /// Source line of the statement that emitted this instruction. Becomes - /// `Program::src_lines`. - pub(crate) line: u32, - /// Prover hints applied (in order) *before* this instruction during witness - /// generation. - pub(crate) hints: Vec, -} - -#[derive(Clone, Debug)] -pub(crate) enum LOp { - /// A copy into the next loop frame, outside this frame's own allocation. - MulNextFrame { - a: Off, - b: Off, - c: Off, - }, - Set { - o: Off, - k: KVal, - }, - Xor { - a: Off, - b: Off, - c: Off, - }, - Mul { - a: Off, - b: Off, - c: Off, - }, - Deref { - o1: Off, - o2: Off, - o3: Off, - mode: DerefMode, - }, - Jump { - oc: Off, - od: Off, - of: Off, - }, - /// `BLAKE2s`: the four 128-bit input chunks `ins` are addressed independently, - /// one frame cell each. The 32-byte output occupies the two consecutive - /// 128-bit cells `c, c+1`; `md` is the cell holding the byte counter and the - /// two flags. - Blake2s { - ins: [Off; 4], - cv: Off, - c: Off, - md: Off, - }, -} - -/// A prover hint attached to an instruction. Most are already the runtime's own -/// [`RHint`] and pass straight through; the allocation ones are compiler-side, -/// since their size is only known once every function's frame is laid out. -#[derive(Clone, Debug)] -pub(crate) enum Hint { - /// Index the next frame address without allocating it again. - NextFrameAddress, - /// Reserve the loop run, including the final untaken call's argument frame. - AllocLoopFrames { ptr: Off, callee: String, count: LoopCount }, - /// `m[fp·g^ptr] = g^{fresh base}`: a fresh, disjoint frame for `callee`. - AllocFrame { ptr: Off, callee: String }, - /// `AllocFrame` sized to the **largest** of several callees, a shared frame - /// for a dispatched call (all `callees` share the arg/return layout; only - /// their local count, hence frame size, differs). See [`FnLower::lower_dispatched_call`]. - AllocFrameMax { ptr: Off, callees: Vec }, - /// `m[fp·g^ptr] = g^{fresh base}`: a fresh, disjoint heap region of `size` - /// cells (a `HeapBuf(size)`), addressed by g-power offsets from the pointer. - AllocBuffer { ptr: Off, size: u32 }, - /// `AllocBuffer` with a *runtime* size in the exponent: the cell count is - /// the g-power exponent of `m[fp·g^size]` (a `HeapBuf(size_expr)`). - AllocBufferDyn { ptr: Off, size: Off }, - /// A hint that needs nothing from the layout: witness fills, computed - /// advice, debug prints. - Resolved(RHint), -} - -pub(crate) struct Lowered { - pub(crate) name: String, - pub(crate) code: Vec, - pub(crate) frame_size: u32, - /// The fill blocks this function carries, with `code`-relative pcs; only `main` has - /// any ([`crate::lower::FnLower::lower_filler_blocks`]). - pub(crate) filler: Vec, -} - -/// A resolved run of consecutive cells ([`crate::lower::FnLower::cell_run`]): a -/// frame (stack) run, used in place, or a heap slice (the buffer pointer's cell -/// plus the first g-power offset), which a `blake2s` operand must bridge through -/// the stack since `BLAKE2s` addresses only frame cells. -pub(crate) enum CellRun { - Stack { base: Off, len: u32 }, - Heap { ptr: Off, lo: u32, len: u32 }, -} - -impl CellRun { - pub(crate) fn cells(&self) -> u32 { - match *self { - CellRun::Stack { len, .. } | CellRun::Heap { len, .. } => len, - } - } -} - -#[derive(Clone, Debug)] -pub(crate) enum LoopCount { - Constant(u64), - Bound { cell: Off, start: u64 }, -} diff --git a/crates/lean_compiler/src/lib.rs b/crates/lean_compiler/src/lib.rs deleted file mode 100644 index f00e3ab4b..000000000 --- a/crates/lean_compiler/src/lib.rs +++ /dev/null @@ -1,318 +0,0 @@ -//! A compiler from a Python-like zkDSL (see `zkDSL.md`) to the ISA (`cpu::Op`). -//! Produces a [`lean_vm::cpu::Program`]: bytecode plus the prover's allocation hints. -//! -//! ## Calling convention -//! -//! A frame is fp-relative; operand `gᵏ` names cell `m[fp·gᵏ]`. Layout: -//! -//! | offset | contents | -//! |--------|----------| -//! | 0 | `retpc`, the return program counter | -//! | 1 | `retfp`, the caller frame pointer | -//! | 2 .. 2+nargs | arguments | -//! | 2+nargs .. 2+nargs+nretcells | flattened return cells | -//! | rest | locals / temporaries / frame-pointer hints | -//! -//! A scalar or `HeapBuf` pointer occupies one return cell. A returned -//! `StackBuf(n)` occupies `n` consecutive cells and is copied into a consecutive -//! run in the caller; source-level tuple arity therefore differs from physical -//! return-cell count when a tuple contains a stack buffer. -//! -//! A **call** is `DEREF`-then-`JUMP`: the callee frame pointer is a fresh -//! prover-hinted cell; the args and `retfp` are stored with `DEREF`(`Cell`/`Fp`), -//! then `DEREF`(`Pc`) stores the return address `g²·pc` (the resume point after the -//! call `JUMP`). The callee returns with one `JUMP[one, 0, 1]`. A **`mul_range` -//! loop** lowers to a recursive helper that tests `i == g^hi` and, while not done, -//! runs the body and recurses on `i·g`. - -use std::collections::HashMap; -use std::fmt::Write; - -use lean_vm::cpu::hints::{BitsDest, RHint}; -use lean_vm::cpu::{DerefMode, Op, Program}; -use primitives::{ - field::{F64, F192, g_pow}, - pretty_integer, -}; - -mod ast; -pub mod filler; -mod ir; -mod lower; -mod parser; -pub use ast::*; -pub(crate) use ir::*; -use lower::lower_func; -pub(crate) use parser::subst_stmts; -pub use parser::{parse, parse_const, parse_with_replacements}; - -/// Compile an [`Ast`] to a provable [`Program`]. Panics on a malformed program -/// (unbound variable, missing `main`, address overflow). -pub fn compile(ast: &Ast) -> Program { - // Every program carries, past `main`'s halt, the fill blocks that bring each table's - // row count up to a power of two, so that no table needs padding rows (see `filler`). - compile_inner(ast, true) -} - -/// [`compile`] without the fill blocks, so the program's own instruction mix is what -/// runs. Used by tests of instruction selection and execution. -pub fn compile_without_filler(ast: &Ast) -> Program { - compile_inner(ast, false) -} - -fn compile_inner(ast: &Ast, with_filler: bool) -> Program { - // Lower main first (entry pc 0), then the rest, expanding loop helpers. - let mut queue: Vec = Vec::new(); - let main = ast - .funcs - .iter() - .find(|f| f.name == "main") - .expect("program needs a `main`"); - assert!(!main.has_const_params(), "main cannot take Const parameters"); - assert!(!main.inline, "main cannot be `@inline`"); - queue.push(main.clone()); - for f in &ast.funcs { - if f.name != "main" { - queue.push(f.clone()); - } - } - // Definitions by name, for Const-parameter specialization at call sites. - let defs: HashMap<&str, &Func> = ast.funcs.iter().map(|f| (f.name.as_str(), f)).collect(); - // Constant arrays by name, resolved at lowering (`NAME[i]`, `len(NAME)`). - let const_arrays: HashMap<&str, &[F192]> = ast - .const_arrays - .iter() - .map(|(name, values)| (name.as_str(), values.as_slice())) - .collect(); - let dbg_lower = std::env::var("DBG_LOWER").is_ok(); - - let mut loop_bounds = HashMap::new(); - let mut loop_ctr = 0usize; - let mut lowered: Vec = Vec::new(); - let mut i = 0; - while i < queue.len() { - let f = &queue[i]; - i += 1; - // A function with Const parameters is a template (only its call-site - // specializations are lowered); an `@inline` function is expanded at - // each call site ([`FnLower::try_inline`]), never lowered standalone. - if f.has_const_params() || f.inline { - continue; - } - let f = f.clone(); - let low = lower_func( - &f, - &mut queue, - &mut loop_ctr, - &defs, - &const_arrays, - with_filler, - &mut loop_bounds, - ); - if dbg_lower { - eprintln!("== fn {} (frame {}) ==", low.name, pretty_integer(low.frame_size)); - for (i, ins) in low.code.iter().enumerate() { - let index = pretty_integer(i); - eprintln!(" {index:>5}: {:?}", ins.op); - } - } - lowered.push(low); - } - - // Assign entry program counters and frame sizes. - let mut entry = HashMap::new(); - let mut frame_size = HashMap::new(); - let mut pc = 0u32; - for l in &lowered { - entry.insert(l.name.clone(), pc); - frame_size.insert(l.name.clone(), l.frame_size); - pc += l.code.len() as u32; - } - // The padded bytecode size `B` is fixed by the lowered length, so the halt - // sentinel pc `g^{B-1}` (last slot) is known before resolving: `main`'s - // `EndSentinel` jump dest resolves to it, and the program halts there. - let total: usize = lowered.iter().map(|l| l.code.len()).sum(); - // The sentinel needs a slot of its own PAST all real code: pad from - // total + 1, so a program of exactly 2^k instructions doesn't collide - // its last instruction with the halt pc. - let bytecode_size = (total + 1).next_power_of_two(); - let sentinel = (bytecode_size - 1) as u32; - - // Resolve to bytecode + a hint map keyed by global pc. - let mut prog: Vec = Vec::new(); - // Source line per pc, so a run-time failure can name a line instead of a pc. - let mut src_lines: Vec = Vec::new(); - let mut hints: HashMap> = HashMap::new(); - for l in &mut lowered { - let base = entry[&l.name]; - for ins in &mut l.code { - let here = prog.len() as u32; - if !ins.hints.is_empty() { - let rhs = ins - .hints - .drain(..) - .map(|h| match h { - Hint::NextFrameAddress => RHint::FrameAddress { offset: l.frame_size }, - Hint::AllocLoopFrames { ptr, callee, count } => { - let size = frame_size[&callee]; - match count { - LoopCount::Constant(n) => RHint::Alloc { - ptr, - size: size - .checked_mul(u32::try_from(n).expect("loop range too large")) - .expect("loop frames overflow"), - }, - LoopCount::Bound { cell, start } => RHint::AllocFrames { - ptr, - size, - end: cell, - start_inverse: g_pow_u128(u128::from(start)).inv(), - }, - } - } - Hint::AllocFrame { ptr, callee } => RHint::Alloc { - ptr, - size: frame_size[&callee], - }, - Hint::AllocFrameMax { ptr, callees } => RHint::Alloc { - ptr, - size: callees.iter().map(|c| frame_size[c]).max().unwrap(), - }, - Hint::AllocBuffer { ptr, size } => RHint::Alloc { ptr, size }, - Hint::AllocBufferDyn { ptr, size } => RHint::AllocDyn { ptr, size }, - Hint::Resolved(r) => r, - }) - .collect(); - hints.insert(here, rhs); - } - src_lines.push(ins.line); - prog.push(resolve(&ins.op, &entry, sentinel, base, l.frame_size)); - } - } - - // Pad the bytecode to `B` (the sentinel slot g^{B-1} must exist for execution). - prog.resize(bytecode_size, Op::Set { o: 0, k: F192::ZERO }); - let mut program = Program::assemble(prog, hints, frame_size["main"]); - program.src_lines = src_lines; - program.fn_ranges = lowered - .iter() - .map(|l| (l.name.clone(), entry[&l.name], l.code.len() as u32)) - .collect(); - // The blocks are `main`'s, and `main` is lowered first, so its entry pc is 0 and the - // block pcs are already the global ones. - program.filler = std::mem::take(&mut lowered[0].filler); - program -} - -/// Render compiled bytecode as a human-readable disassembly. `fp[k]` is the cell -/// `m[fp·gᵏ]` (frame offset `k`); `*(p·gᵝ)` is the dereferenced cell. `SET` -/// constants that are small g-powers (code addresses, indices) show as `gʲ`. -pub fn disassemble(prog: &[Op]) -> String { - // Reverse index for small g-powers, to pretty-print code addresses/indices. - let mut gmap: HashMap = HashMap::new(); - let mut acc = F64::ONE; - for j in 0..(prog.len() + 512) { - gmap.entry(acc).or_insert(j); - acc *= primitives::field::G; - } - // A machine word is 192-bit; K-valued immediates (both high limbs zero) may be small - // g-powers (code addresses, indices), shown as `gʲ`. - let kfmt = |k: F192| match (k.c1 == 0 && k.c2 == 0).then(|| gmap.get(&F64(k.c0))).flatten() { - Some(j) => format!("g^{j}"), - None if k.c1 == 0 && k.c2 == 0 => format!("0x{:016x}", k.c0), - None => format!("0x{:016x}{:016x}{:016x}", k.c2, k.c1, k.c0), - }; - - let mut out = String::new(); - for (pc, op) in prog.iter().enumerate() { - let line = match op { - Op::Set { o, k } => format!("SET fp[{o}] = {}", kfmt(*k)), - Op::Xor { a, b, c } => format!("XOR fp[{c}] = fp[{a}] ^ fp[{b}]"), - Op::Mul { a, b, c } => format!("MUL fp[{c}] = fp[{a}] * fp[{b}]"), - Op::Deref { o1, o2, o3, mode } => { - let src = match mode { - DerefMode::Cell => format!("fp[{o3}]"), - DerefMode::Pc => "g²·pc".to_string(), - DerefMode::Fp => "fp".to_string(), - }; - format!("DEREF *(fp[{o1}]·g^{o2}) = {src} [{mode:?}]") - } - Op::Jump { oc, od, of } => { - format!("JUMP if fp[{oc}]≠0: pc=fp[{od}], fp=fp[{of}]") - } - Op::Blake2s { ins, cv, out, md } => { - format!( - "BLAKE2S fp[{out}..]= compress(cv=fp[{cv}..], m=fp[{}],fp[{}],fp[{}],fp[{}], meta=fp[{md}])", - ins[0], ins[1], ins[2], ins[3] - ) - } - }; - writeln!(out, "{:>6} {line}", pretty_integer(pc)).unwrap(); - } - out -} - -/// Embed a `u128` source literal into the low 128 bits of a 192-bit machine word. -pub(crate) fn lit_field(n: u128) -> F192 { - F192::new(n as u64, (n >> 64) as u64, 0) -} - -/// `g^e` for a `u128` exponent (square-and-multiply). `field::g_pow` only takes -/// a `usize`; an index carried in the exponent (a Fibonacci number, say) can -/// exceed 64 bits (`ord(g) = 2^64 − 1`, so the exponent wraps mod that). -fn g_pow_u128(mut e: u128) -> F64 { - let mut result = F64::ONE; - let mut base = primitives::field::G; - while e > 0 { - if e & 1 == 1 { - result *= base; - } - base = base * base; - e >>= 1; - } - result -} - -fn resolve(op: &LOp, entry: &HashMap, sentinel: u32, base: u32, frame_size: u32) -> Op { - let resolve_kval = |kv: &KVal| -> F192 { - match kv { - KVal::FrameSize => g_pow(frame_size as usize).into(), - KVal::Const(c) => *c, - // Address / entry / sentinel constants are K-valued g-powers; - // embed them canonically as (c0, 0, 0). - KVal::Entry(name) => g_pow(entry[name] as usize).into(), - KVal::EndSentinel => g_pow(sentinel as usize).into(), - KVal::Local(i) => g_pow((base + i) as usize).into(), - } - }; - match op { - LOp::MulNextFrame { a, b, c } => Op::Mul { - a: *a, - b: *b, - c: frame_size.checked_add(*c).expect("frame offset overflow"), - }, - LOp::Set { o, k: kv } => Op::Set { - o: *o, - k: resolve_kval(kv), - }, - LOp::Xor { a, b, c } => Op::Xor { a: *a, b: *b, c: *c }, - LOp::Mul { a, b, c } => Op::Mul { a: *a, b: *b, c: *c }, - LOp::Deref { o1, o2, o3, mode } => Op::Deref { - o1: *o1, - o2: *o2, - o3: *o3, - mode: *mode, - }, - LOp::Jump { oc, od, of } => Op::Jump { - oc: *oc, - od: *od, - of: *of, - }, - LOp::Blake2s { ins, cv, c, md } => Op::Blake2s { - ins: *ins, - cv: *cv, - out: *c, - md: *md, - }, - } -} diff --git a/crates/lean_compiler/src/lower.rs b/crates/lean_compiler/src/lower.rs deleted file mode 100644 index 2149e4c47..000000000 --- a/crates/lean_compiler/src/lower.rs +++ /dev/null @@ -1,1671 +0,0 @@ -//! Lowering: each function AST is compiled to a sequence of intermediate -//! [`LOp`] instructions (fp-relative offsets, backpatched jump targets). -//! -//! This file holds the walk itself, control flow, instruction emission, and -//! [`Scope`]. The rest is split by the question it answers, because each of the -//! four has an invariant worth stating once rather than rediscovering: -//! -//! - [`mod@eval`] asks what an expression is worth before anything runs. Every -//! function there takes `&self` and emits nothing, which is what lets a caller -//! ask without paying for the answer. -//! - [`mod@mem`] asks which cell a name means, and what writing to it costs. -//! Every index is bounds-checked, in every position, and every store emits: the -//! machine's write-once memory is what separates an assertion from a definition. -//! - [`mod@call`] is the call boundary. Caller and callee must agree on the -//! arity, because they place the return area from their own idea of it. -//! - [`mod@builtins`] is the precompile and the hints: the two places a value -//! arrives without an instruction computing it. -//! -//! [`Scope`] is what a name means HERE, and it reverts at a branch join, so a -//! cell whose `SET` sits inside a branch is never trusted outside it. - -use super::*; -use crate::filler::FillerOp; -use lean_vm::cpu::filler::Block; - -mod builtins; -mod call; -mod eval; -mod mem; -use call::ret_binding; -use eval::field_pow; - -/// A value equal to `pointer(base)·g^exp`, or the pure constant `g^exp` when -/// `base` is `None`. Heap-address arithmetic (`ptr·gᵏ`, and constant g-power -/// cursors such as a tweak-table index) is tracked symbolically so a later -/// access folds the whole offset into `DEREF`'s `β` immediate rather than -/// emitting a `SET`+`MUL` per step. A cursor read only as an index thus costs -/// nothing; one used as a value is materialized on demand ([`FnLower::materialize`]). -#[derive(Clone, Copy)] -struct GAddr { - base: Option, - exp: u128, - /// For a pointer minted by `addr(sb)`, the frame run it names, as - /// `(first cell, length)`. `base` is then the shared `fp` cell and `exp` the - /// absolute frame offset, so the run cannot be recovered from those two - /// alone: carrying it here is what lets [`FnLower::check_heap_bound`] hold a - /// frame pointer to the same bound a `HeapBuf` pointer gets. `None` for a - /// heap pointer (bounded through `heap_sizes`) and for a pure g-power. - run: Option<(Off, u32)>, -} - -/// The fixed prefix of every frame: the caller's return pc and frame pointer, -/// then the arguments, then the flattened return area (a `StackBuf(n)` return -/// occupies `n` consecutive cells). One place derives every offset in it, so a -/// caller writing into a callee's frame and the callee reading its own cannot -/// drift apart, and `2 + n_args + n_ret_cells` is not spelled out at each site. -struct Abi; - -impl Abi { - /// Where the caller leaves the return pc and the return frame pointer. - const RET_PC: Off = 0; - const RET_FP: Off = 1; - /// Argument `i`, after the two return slots and the preceding arguments. - fn arg(shapes: impl Iterator, i: usize) -> Off { - 2 + shapes.take(i).map(Shape::cells).sum::() - } - /// Total width of the argument area. - fn arg_cells(shapes: impl Iterator) -> u32 { - shapes.map(Shape::cells).sum() - } - /// Return cell `i` of a callee whose arguments occupy `arg_cells` cells. - fn ret(arg_cells: u32, i: u32) -> Off { - 2 + arg_cells + i - } - /// One past the last cell the CALLER touches, so the first local cell. - fn end(arg_cells: u32, n_ret_cells: u32) -> Off { - Self::ret(arg_cells, n_ret_cells) - } -} - -/// Cap on a `β`-folded exponent, inclusive: the operand g-power table is sized to -/// the largest immediate, so beyond this a huge constant index falls back to a -/// materialized pointer instead of inflating that table. The one cap: every site -/// that folds an exponent into `β` measures it against this. -const FOLD_MAX: u128 = 1 << lean_vm::cpu::MIN_LOG_MEM; - -/// The two pure operations worth interning. Both are commutative, so operands -/// are stored sorted. -#[derive(Clone, Copy, PartialEq, Eq, Hash)] -enum PureOp { - Xor, - Mul, -} - -/// How an inlined `@inline` tail-return value binds into the caller -/// ([`FnLower::inline_stack_ret`]): a `StackBuf` hands over its cell run and a -/// folded g-address hands over its symbolic pointer, both aliased at zero -/// copies (so `cvb = obs(cvb, x)` and a fused `fs, x, cur = fs_next(fs, cur)` -/// stay free); a scalar was already copied into its dst cell. -#[derive(Clone, Copy)] -enum RetBind { - Stack(Off, u32), - Gaddr(GAddr), - Scalar, -} - -/// What a name is bound to. The kinds are mutually exclusive: a name has exactly -/// one of them, which is why they are one enum and not four maps. -#[derive(Clone, Copy)] -enum Binding { - Scalar(Off), - Stack(Off, u32), - Gaddr(GAddr), - FConst(F192), -} - -/// What one name means: its value binding, plus an OPTIONAL compile-time integer -/// reading of the same expression. The two genuinely coexist (`x = 2` names the -/// field element 2 AND the index 2, while `n = len(A) - 1` has no g-power reading -/// at all), so `int` rides beside `val` rather than being another variant. One -/// entry per name is what makes a rebind atomic. -#[derive(Clone, Copy)] -struct Bound { - val: Binding, - int: Option, -} - -/// Everything a runtime branch may not have executed: the name bindings, plus the -/// lazily materialized cells whose `SET` sits wherever it was first needed. -/// [`FnLower::scoped`] restores one of these at the join, so a cell written on one -/// path is never trusted on another. -#[derive(Clone, Default)] -struct Scope { - names: HashMap, - /// The cell holding this function's own `fp`, materialized lazily - /// ([`FnLower::self_fp`]): local (`if`/`else`) jumps reload the frame - /// pointer on the taken branch. - self_fp_off: Option, - /// Results of pure operations: `(op, sorted operands)` → the cell holding it. - /// - /// Reverting at a branch join (with `const_cells`, below) is the whole - /// invalidation: a cached cell must dominate every later use, and nothing - /// clears this at a label target. That is sound only because every backward - /// edge crosses a function boundary (a loop body is its own `Func` with a - /// fresh `Scope`) and every `patch_local` target is a forward jump, so a - /// cached cell's defining instruction always precedes its reuse. A new - /// backward edge, or a `patch_local` that jumps backwards, would need this - /// cleared at the target. - pure_cells: HashMap<(PureOp, Off, Off), Off>, - /// Dominating memory equalities. Reads reuse the frame cell; stores still - /// emit their equality checks. Branch joins restore this with the scope. - load_cells: HashMap<(Off, u32), Off>, - /// Every lazily-`SET` constant cell: field value (as bits) → the frame cell - /// holding it. Cells are write-once and read-many, so one `SET` serves every - /// use in scope. A `SET` first emitted inside a branch must not be named from - /// outside it, where the other path leaves the cell unwritten and therefore - /// prover-chosen, which is why this reverts at a join with the bindings. - const_cells: HashMap<[u64; 3], Off>, - /// Two consecutive frame cells holding the standard BLAKE2s IV, emitted - /// lazily at the first dominating default-IV compression in this - /// control-flow scope. - blake2s_iv: Option, -} - -impl Scope { - fn bound(&self, n: &str) -> Option { - self.names.get(n).copied() - } - fn stack(&self, n: &str) -> Option<(Off, u32)> { - match self.bound(n)?.val { - Binding::Stack(base, size) => Some((base, size)), - _ => None, - } - } - /// The compile-time integer reading of `n`, when it has one. - fn int(&self, n: &str) -> Option { - self.bound(n)?.int - } - /// Attach the integer reading to the binding just made for `n`. - fn set_int(&mut self, n: &str, k: u128) { - if let Some(b) = self.names.get_mut(n) { - b.int = Some(k); - } - } -} - -struct FnLower<'a> { - scope: Scope, - next: Off, - /// Physical width of this function's argument area, which is not its - /// parameter COUNT once a parameter can be a run of cells. - arg_cells: u32, - /// Source-level return shapes for this function. Their physical cell widths - /// determine the reserved return area immediately after the arguments. - return_shapes: &'a [Shape], - is_main: bool, - code: Vec, - /// Declared size of each `HeapBuf`, keyed by its pointer cell. Shifted - /// aliases resolve to the same base cell through their gaddr, so a - /// compile-time index checks against the ORIGINAL buffer's bound. - heap_sizes: HashMap, - /// While inlining an `@inline` call ([`Self::try_inline`]), the destination - /// cells its tail `return` binds into instead of emitting a return jump. - /// `None` outside an inlined body. - inline_ret: Option>, - /// Set by an inlined tail `return`, one [`RetBind`] per returned value, - /// telling the caller's `let`/tuple how to bind each (alias a `StackBuf` run - /// or a folded g-address, or take the scalar dst cell). `None` outside an - /// inlined return. - inline_stack_ret: Option>, - /// Source line of the statement being lowered, for diagnostics and for the - /// pc-to-line table. Zero for a synthesized statement with no source. - cur_line: u32, - /// Hints queued to attach to the next emitted instruction. - pending: Vec, - /// Active `@inline` expansion stack. Nested inline helpers are allowed, - /// but direct or indirect recursion would otherwise recurse forever in - /// the compiler. - inline_calls: Vec, - /// Set by [`lower_func`] just before lowering a statement that sits in tail - /// position; consumed by the next [`Self::stmt`] call, so nested lowering - /// never inherits it. - tail_call: bool, - /// Generated loops that can reserve their complete run of frames. - loop_bounds: &'a mut HashMap)>, - queue: &'a mut Vec, - loop_ctr: &'a mut usize, - /// Function name for diagnostics and generated-loop handling. - fn_name: &'a str, - /// The program's function definitions by name, for `Const`-parameter - /// specialization at call sites ([`Self::specialize`]). - defs: &'a HashMap<&'a str, &'a Func>, - /// Top-level constant arrays, resolved at compile time: `NAME[i]` yields the - /// element (a field value or an index), `len(NAME)` its length. - const_arrays: &'a HashMap<&'a str, &'a [F192]>, -} - -impl FnLower<'_> { - /// Abort with a diagnostic naming the source line being lowered. Every - /// deliberate user-facing error goes through here; a bare `assert!` left in - /// the file is an internal invariant, i.e. a compiler bug rather than a - /// program one, and deliberately does NOT get a line. - fn fail(&self, msg: impl std::fmt::Display) -> ! { - // The line alone is not the site when the function being lowered is one - // the author never wrote: `hash_pair__L1` exists because of a `Const` - // call site, `__loop3` because of a `for` header, and an `@inline` body - // is lowered through the CALLER, so its statements report the caller's - // line. Naming the function, and the inline chain when there is one, is - // what turns "line 19" back into somewhere to look. - let site = match (self.cur_line, self.fn_name) { - (0, "main") => String::new(), - (0, f) => format!("in {f}: "), - (n, "main") => format!("line {n}: "), - (n, f) => format!("line {n} in {f}: "), - }; - match self.inline_calls.as_slice() { - [] => panic!("{site}{msg}"), - chain => panic!("{site}{msg} (inlined through {})", chain.join(" -> ")), - } - } - - fn fresh(&mut self) -> Off { - let o = self.next; - self.next += 1; - o - } - - fn emit(&mut self, op: LOp) { - let hints = std::mem::take(&mut self.pending); - self.code.push(LInstr { - op, - line: self.cur_line, - hints, - }); - } - - fn set(&mut self, o: Off, k: KVal) { - self.emit(LOp::Set { o, k }); - } - - fn set_const(&mut self, o: Off, v: F192) { - self.set(o, KVal::Const(v)); - } - - fn deref(&mut self, o1: Off, o2: u32, o3: Off, mode: DerefMode) { - self.emit(LOp::Deref { o1, o2, o3, mode }); - if mode == DerefMode::Cell { - self.scope.load_cells.entry((o1, o2)).or_insert(o3); - } - } - - /// A no-op instruction to hang a pending hint on, so it fires exactly here - /// instead of drifting onto whatever is emitted next (which may sit past a - /// branch join, or on a path this hint does not belong to). - fn anchor(&mut self) { - let o = self.fresh(); - self.set_const(o, F192::ZERO); - } - - /// A top-level constant name is reserved (`zkDSL.md` §Global constants). A - /// scalar one enforces that by construction, its value being substituted - /// textually so a shadowing binding becomes a literal and fails loudly. A - /// constant ARRAY is carried to lowering instead and resolved without - /// consulting the scope, so a colliding local would have its - /// compile-time-indexed reads folded to baked literals, including reads of a - /// `hint_witness` destination whose asserts would then run on the constant. - /// Reject the collision rather than pick a winner. - fn check_not_reserved(&self, name: &str) { - if self.const_arrays.contains_key(name) { - self.fail(format!( - "`{name}` is a top-level constant array, so the name is reserved: rename the local \ - or parameter (zkDSL.md §Global constants)" - )) - }; - } - - /// Replace the binding, clearing its previous compile-time integer reading. - fn rebind(&mut self, name: &str, b: Binding) { - self.check_not_reserved(name); - self.scope.names.insert(name.to_string(), Bound { val: b, int: None }); - } - - /// Bind each of `names` to the join cell holding its value, after a `match` - /// dispatch: whichever arm ran wrote them, so they are plain scalars now. - fn bind_targets(&mut self, binds: &[(&str, Off)]) { - for (name, cell) in binds { - self.rebind(name, Binding::Scalar(*cell)); - } - } - - /// The cell holding `a op b`, computed only if this scope has not already. - /// - /// **Route an operation here only when no later write to its result cell - /// could be the assertion.** A fresh cell is not sufficient: `assert a != b` - /// mints one for `x·inv` and writes it again with `SET p = 1`, and THAT write - /// is the assertion, so sharing the cell would let the next `assert a != b` - /// skip its `MUL` and assert nothing. A second writer is fine where the value - /// is already pinned, as in `q = x ** k / w`, whose division writes into the - /// cell the squaring chain already determined. - /// - /// So these stay out: the zero cell an `assert a == b` XORs into, the - /// `g^{k-1}` a range check multiplies into, `assert a != b`'s product, a - /// division's back-solve, and `expr_into`'s caller-chosen destination. - fn pure(&mut self, op: PureOp, a: Off, b: Off) -> Off { - let key = (op, a.min(b), a.max(b)); - if let Some(&o) = self.scope.pure_cells.get(&key) { - return o; - } - let o = self.fresh(); - match op { - PureOp::Xor => self.emit(LOp::Xor { a, b, c: o }), - PureOp::Mul => self.emit(LOp::Mul { a, b, c: o }), - } - self.scope.pure_cells.insert(key, o); - o - } - - /// The frame cell `arr[idx]` names, bounds-checked, or `None` when `arr` is - /// not a `StackBuf` and the caller should take its heap path. - /// - /// Emits nothing, so each caller keeps its own evaluation order. In - /// `hb[sb[0]] = f(sb, …)`, the address may read a cell the value writes. - /// The heap bound lives where the address is formed ([`Self::heap_addr`]). - fn frame_cell(&mut self, arr: &Expr, idx: &Expr) -> Option { - let (base, size) = self.stack_of(arr)?; - let k = self.const_index(idx); - if k >= size { - self.fail(format!("index {k} out of bounds (StackBuf size {size})")) - }; - Some(base + k) - } - - /// A frame cell holding `1` (always-taken `JUMP` condition). - fn one(&mut self) -> Off { - self.const_cell(F192::ONE) - } - - /// A frame cell holding `v`, shared by every dominated use in the current - /// scope: the one cache for every lazily-`SET` constant. - /// - /// The `SET` is emitted where the cell is allocated, before anything can name - /// it, which is what makes every later write a write-once equality against a - /// bytecode constant rather than a chance to choose the value. It reverts at - /// a branch join with the rest of [`Scope`], since a `SET` first emitted - /// inside a branch must not be named outside it, where the other path leaves - /// the cell unwritten and so prover-chosen. Several call sites hoist - /// [`Self::one`] above a branch on purpose; the revert is what makes that an - /// optimization rather than the thing holding the invariant up. - fn const_cell(&mut self, v: F192) -> Off { - let key = [v.c0, v.c1, v.c2]; - if let Some(&o) = self.scope.const_cells.get(&key) { - return o; - } - let o = self.fresh(); - self.set_const(o, v); - self.scope.const_cells.insert(key, o); - o - } - - /// A frame cell holding `0`, set lazily once: the source for forwarded zero - /// words (a `BLAKE2s` padding half), and the destination every `assert a == b` - /// in this scope XORs into. - fn zero(&mut self) -> Off { - self.const_cell(F192::ZERO) - } - - /// Terminate `main`: jump to the halt sentinel `g^{B-1}` with `fp = g^0`. - /// The cell holding `1` doubles as the (nonzero) jump condition and the new - /// frame pointer `g^0`; the dest cell holds `g^{B-1}` (doc §sec:e2e, final state). - fn halt(&mut self) { - let one = self.one(); - let dest = self.fresh(); - self.set(dest, KVal::EndSentinel); - self.emit(LOp::Jump { - oc: one, - od: dest, - of: one, - }); - } - - /// Emit the fill blocks: per table and per size in `lean_vm::cpu::filler::SIZES`, - /// that many dummy instructions of the table's opcode, then a `JUMP` back to the - /// block's own first instruction, in the same frame. - /// - /// A block is a cycle and nothing jumps into one: they sit past `main`'s halt - /// and the interpreter enters them itself once the program has stopped. The - /// state tuples a traversal pushes are the ones it pulls, so the cycle - /// balances for any number of traversals (`lean_vm::cpu::filler`). - /// - /// The closing jump is always taken (its destination is a g-power, so - /// nonzero) and reads its destination and frame from cells the interpreter - /// writes, so a traversal costs the block's rows plus that jump. A dummy uses - /// one scratch cell as each operand, writing the value already there, so a - /// block costs one cell whatever its size. - fn lower_filler_blocks(&mut self) -> Vec { - use lean_vm::cpu::filler::{SIZES, frame as fr}; - - // No statement wrote these, so they get the "unknown" line rather than - // whatever `main` happened to end on. - self.cur_line = 0; - - // A block runs in a frame the interpreter carves out, so the cells it reads are at - // fixed offsets in *that* frame rather than allocated from this function's - // counter, and nothing here touches `main`'s frame at all. - let mut blocks = Vec::new(); - for (table, op) in crate::filler::TABLES { - for size in SIZES { - blocks.push(Block { - pc: self.code.len() as u32, - size: size as u32, - table, - }); - for _ in 0..size { - self.emit(match op { - FillerOp::Xor => LOp::Xor { - a: fr::SCRATCH, - b: fr::SCRATCH, - c: fr::SCRATCH, - }, - FillerOp::Mul => LOp::Mul { - a: fr::SCRATCH, - b: fr::SCRATCH, - c: fr::SCRATCH, - }, - FillerOp::Set => LOp::Set { - o: fr::SCRATCH, - k: KVal::Const(F192::ZERO), - }, - FillerOp::Deref => LOp::Deref { - o1: fr::PTR, - o2: 0, - o3: fr::SCRATCH, - mode: DerefMode::Cell, - }, - // Its condition is a cell nothing ever writes, so it reads as - // zero: the dummy is not taken and falls through to the next - // instruction of the block instead of closing the cycle early. - FillerOp::Jump => LOp::Jump { - oc: fr::ZERO, - od: fr::ZERO, - of: fr::ZERO, - }, - // Its metadata cell is one no instruction writes, like its - // message cells: the interpreter leaves those zero, and a - // prover choosing otherwise only picks which compression the - // dummy proves, which nothing reads (`lean_vm::cpu::filler`). - FillerOp::Blake2s => LOp::Blake2s { - ins: [fr::DIGEST + 2, fr::DIGEST + 3, fr::DIGEST + 4, fr::DIGEST + 5], - cv: fr::SCRATCH, - c: fr::DIGEST, - md: fr::ZERO, - }, - }); - } - // Back to the top, closing the cycle. For the `JUMP` table this is one - // more row of its own, which is why the solver decomposes that table over - // `size + 1`. - self.emit(LOp::Jump { - oc: fr::DEST, - od: fr::DEST, - of: fr::NEXT_FP, - }); - } - } - blocks - } - - /// `dst = src` (no MOV: multiply by `1`). - fn copy(&mut self, src: Off, dst: Off) { - let one = self.one(); - self.emit(LOp::Mul { a: src, b: one, c: dst }); - } - - /// A frame cell holding this function's own `fp` (the g-power element), - /// materialized lazily once: a taken `JUMP` reloads the frame pointer - /// from a cell, so local (`if`/`else`) jumps must name it. The ISA has no - /// fp-read, so bounce it through a fresh 1-cell heap buffer: a - /// `DEREF`-fp writes it there and a `DEREF`-cell copies it back. In `main`, `fp = g^0 = 1`, which is - /// the [`Self::one`] cell. - fn self_fp(&mut self) -> Off { - if self.is_main { - return self.one(); - } - if let Some(o) = self.scope.self_fp_off { - return o; - } - let q = self.fresh(); - self.pending.push(Hint::AllocBuffer { ptr: q, size: 1 }); - self.deref(q, 0, 0, DerefMode::Fp); // m[q] := fp - let o = self.fresh(); - self.deref(q, 0, o, DerefMode::Cell); // m[fp·g^o] := m[q] - self.scope.self_fp_off = Some(o); - o - } - - /// Backpatch a [`KVal::Local`] `SET` (emitted with a placeholder) to name - /// the instruction at index `target` of this function's code. - fn patch_local(&mut self, set_idx: usize, target: usize) { - match &mut self.code[set_idx].op { - LOp::Set { k: KVal::Local(t), .. } => *t = target as u32, - other => unreachable!("patch_local on {other:?}"), - } - } - - /// Run `f` with branch-local scope: bindings AND the lazily cached cells - /// (`one`, `self_fp`, range-check bounds, default BLAKE2s IV) revert - /// afterwards, since a cell whose `SET` sits inside a conditionally-executed - /// region must not be trusted outside it. - fn scoped(&mut self, f: impl FnOnce(&mut Self)) { - let saved_scope = self.scope.clone(); - f(self); - // A hint pending at the end of a branch (e.g. a trailing - // `hint_witness`) must not attach to whatever instruction follows the - // join, which would fire it unconditionally. - if !self.pending.is_empty() { - self.anchor(); - } - self.scope = saved_scope; - } - - /// Lower a branch body with branch-local scope ([`Self::scoped`]). - fn branch(&mut self, body: &[Stmt]) { - self.scoped(|s| { - for st in body { - s.stmt(st); - } - }); - } - - /// The cell each multi-return target names, and the names still to bind. - /// - /// A plain name takes a fresh cell. A `StackBuf` element uses its existing - /// cell, so [`Self::call_into`] returns directly into the destination. - fn ret_targets<'a>(&mut self, targets: &'a [Expr]) -> (Vec, Vec<(&'a str, Off)>) { - let mut cells = Vec::with_capacity(targets.len()); - let mut binds = Vec::new(); - for t in targets { - match t { - // Only a frame cell can be a return slot. Rejecting here rather - // than after forming the address keeps a pointer `MUL` out of the - // program and reports the actual problem: routing a heap target - // through the heap path lectured about g-powers instead, and told a - // scalar it was a HeapBuf. - Expr::Index(arr, idx) => match self.frame_cell(arr, idx) { - Some(c) => cells.push(c), - None => self.fail(format!( - "a multi-value target must be a name or a StackBuf element, got `{t:?}`" - )), - }, - Expr::Var(n) => { - let c = self.fresh(); - cells.push(c); - binds.push((n.as_str(), c)); - } - other => self.fail(format!( - "a multi-value target must be a name or a StackBuf element, got `{other:?}`" - )), - } - } - (cells, binds) - } - - /// `targets = match(log(x), …)`: dispatch through the trampoline table - /// ([`Self::lower_match_dispatch`]), arm `j` evaluating the lambda body at - /// `i = j` into cells every arm shares. Write-once makes that sound, exactly - /// one arm running, and [`Self::ret_targets`] says which cells those are. - fn lower_match(&mut self, targets: &[Expr], x: &Expr, arms: &[Expr]) { - for arm in arms { - if let Expr::Call(f, _) = arm - && self - .defs - .get(f.as_str()) - .is_some_and(|d| !d.inline && d.return_shapes.iter().any(|s| matches!(s, Shape::StackBuf(_)))) - { - self.fail("a normal function's StackBuf return cannot cross a match join; bind it with `let`"); - } - } - // Calls with identical runtime args share one callee frame and a - // two-instruction trampoline per arm. Const args select specializations; - // see `lower_dispatched_call` for the shared argument/return layout checks. - // `@inline` arms instead expand into this frame, below. - let inline_arm = - |s: &Self, a: &Expr| matches!(a, Expr::Call(f, _) if s.defs.get(f.as_str()).is_some_and(|d| d.inline)); - if arms.iter().all(|a| matches!(a, Expr::Call(..)) && !inline_arm(self, a)) { - let specialized: Vec<(String, Vec<&Expr>)> = arms - .iter() - .map(|a| { - let Expr::Call(f, cargs) = a else { unreachable!() }; - self.specialize(f, cargs) - }) - .collect(); - let rt0 = &specialized[0].1; - if specialized.iter().all(|(_, rt)| rt == rt0) { - let callees: Vec = specialized.iter().map(|(c, _)| c.clone()).collect(); - self.lower_dispatched_call(targets, x, &callees, rt0); - return; - } - // Not uniform: fall through (the specializations queued above are - // re-requested idempotently by `call_into`). - } - let xo = self.expr(x); - let (rcells, binds) = self.ret_targets(targets); - self.lower_match_dispatch(xo, arms.len(), |s, j| { - s.scoped(|s| { - if let [rcell] = rcells.as_slice() { - s.expr_into(&arms[j], *rcell); - } else { - let Expr::Call(f, cargs) = &arms[j] else { - s.fail(format!( - "a multi-target match arm must be a function call, got `{:?}`", - arms[j] - )); - }; - s.inline_stack_ret = None; - s.call_into(f, cargs, &rcells); - // An @inline arm's aliased returns materialize into the - // shared join cells (a real call wrote them directly). - if let Some(binds) = s.inline_stack_ret.take() { - for (b, &rc) in binds.iter().zip(&rcells) { - match *b { - RetBind::Gaddr(ga) => { - let c = s.materialize(ga); - s.copy(c, rc); - } - RetBind::Stack(base, size) => { - if size != 1 { - s.fail("a multi-cell StackBuf return cannot cross a match join") - } - // `copy` reads the run's first cell, which is where - // the arm's single returned value sits. - let src = base; - s.copy(src, rc); - } - RetBind::Scalar => {} - } - } - } - } - }); - }); - self.bind_targets(&binds); - } - - /// The two-jump dispatch itself: `d = g^T · x²` names slot `x` of the - /// two-instruction trampoline table at bytecode base `T`. Returns the index - /// of the `SET` holding `T`, for the caller to patch once the table's - /// position is known. - fn emit_dispatch(&mut self, xo: Off, one: Off, of: Off) -> usize { - let kcell = self.fresh(); - let kset = self.code.len(); - self.set(kcell, KVal::Local(0)); // patched: table base T - let x2 = self.pure(PureOp::Mul, xo, xo); - let d = self.fresh(); - self.emit(LOp::Mul { a: kcell, b: x2, c: d }); - self.emit(LOp::Jump { oc: one, od: d, of }); - kset - } - - /// The `n` trampoline slots themselves, each `SET c = k(j); JUMP c`, sharing - /// `c` since one slot runs. Returns the table's start; slot `j` has its - /// `SET` at `start + 2*j`. - fn emit_slots(&mut self, n: usize, one: Off, of: Off, k: impl Fn(usize) -> KVal) -> usize { - let start = self.code.len(); - let c = self.fresh(); - for j in 0..n { - self.set(c, k(j)); - self.emit(LOp::Jump { oc: one, od: c, of }); - } - start - } - - /// The trampoline dispatch every `match` lowers through: jump to - /// `d = g^T · x²` (slot `j` of the two-instruction table at bytecode base - /// `T`), then to `body(j)`'s code; every non-final body exits to the - /// join. `body` lowers arm `j`, with its own branch-local scope. - /// - /// One arm runs, so the arms allocate their locals over the same cells. - fn lower_match_dispatch(&mut self, xo: Off, n: usize, mut body: impl FnMut(&mut Self, usize)) { - // Hoisted on purpose: these SETs must dominate the join. - let sfp = self.self_fp(); - let one = self.one(); - let join = self.fresh(); - let jset = self.code.len(); - self.set(join, KVal::Local(0)); // patched: the join - // Slot j (two instructions) sits at T + 2j. - let kset = self.emit_dispatch(xo, one, sfp); - // The trampoline table. - self.patch_local(kset, self.code.len()); - let start = self.emit_slots(n, one, sfp, |_| KVal::Local(0)); - // The arm blocks, each exiting to the join (the last falls through). - let arms_base = self.next; - let mut arms_end = arms_base; - for j in 0..n { - self.next = arms_base; - self.patch_local(start + 2 * j, self.code.len()); - body(self, j); - if j + 1 != n { - self.emit(LOp::Jump { - oc: one, - od: join, - of: sfp, - }); - } - arms_end = arms_end.max(self.next); - // The next arm may reuse these cells for something other than a `HeapBuf`. - self.heap_sizes.retain(|&o, _| o < arms_base); - } - self.next = arms_end; - self.patch_local(jset, self.code.len()); - } - - /// Lower `if` / `else`, arranging the blocks so the nonzero `XOR` result jumps to the correct arm. - fn lower_if(&mut self, eq: bool, lhs: &Expr, rhs: &Expr, then: &[Stmt], els: &[Stmt], force_const: bool) { - // A folded condition emits no test and no jump, so the taken arm is - // straight-line code and its bindings persist, unlike a runtime branch's. - // - // The fold reads INTEGERS while the runtime lowering below tests a field - // XOR, and the two disagree whenever a side's readings do. Neither can - // simply win, so an ambiguous condition is REJECTED and `const(...)` is - // how the author names the regime (`zkDSL.md`, "`if const(...)`"). - let (a, b) = (self.eval(lhs), self.eval(rhs)); - if let (Some(ai), Some(bi)) = ( - a.int.and_then(|n| u32::try_from(n).ok()), - b.int.and_then(|n| u32::try_from(n).ok()), - ) { - // Checked per SIDE, not by comparing the two verdicts. If each side's - // own readings agree then integer equality and field equality say the - // same thing, so a side that disagrees with ITSELF is the whole of the - // ambiguity. Comparing verdicts instead needs a field reading for both - // sides, and `try_field_const` has no arm for `-`, `//` or `%`, so - // `n == 3 - 1` slipped through and folded on the integer reading while - // `n == 2`, the same condition, was rejected. - if !force_const { - for (e, known) in [(lhs, a), (rhs, b)] { - if let Some((n, f)) = known.diverging_readings() { - self.fail(format!( - "`{e:?}` reads as the integer {n} where a condition folds, and as the field \ - element {:#x}:{:#x} where a value is wanted, so this branch would be decided \ - by one reading and its body run under the other. Write `if const(...)` to \ - decide it with integer arithmetic, or spell the operand so the two agree.", - f.c1, f.c0 - )) - } - } - } - for st in if (ai == bi) == eq { then } else { els } { - self.stmt(st); - } - return; - } - // `const(...)` also decides a condition only the field can read (`GEN ** 3`, - // or anything past `u32`), which has no integer reading to be ambiguous - // against. A plain `if` must NOT: folding it would rescope the arm, whose - // bindings then outlive it. - if force_const { - if let (Some(fa), Some(fb)) = (a.field, b.field) { - for st in if (fa == fb) == eq { then } else { els } { - self.stmt(st); - } - return; - } - self.fail("`if const(...)` asks for a compile-time decision, but this condition is not one: both sides must be compile-time constants") - } - // `x != 0` needs no XOR: the cell itself is the JUMP's nonzero test. - let x = if self.try_lit(rhs) == Some(0) { - self.expr(lhs) - } else if self.try_lit(lhs) == Some(0) { - self.expr(rhs) - } else { - let (la, lb) = (self.expr(lhs), self.expr(rhs)); - // x = lhs + rhs: nonzero ⇔ != - self.pure(PureOp::Xor, la, lb) - }; - // Hoisted on purpose: these SETs must dominate the join. - let sfp = self.self_fp(); - let one = self.one(); - let (a_block, b_block) = if eq { (then, els) } else { (els, then) }; - let bdest = self.fresh(); - let bset = self.code.len(); - self.set(bdest, KVal::Local(0)); // patched: start of B - self.emit(LOp::Jump { - oc: x, - od: bdest, - of: sfp, - }); - self.branch(a_block); - if b_block.is_empty() { - self.patch_local(bset, self.code.len()); - } else { - let edest = self.fresh(); - let eset = self.code.len(); - self.set(edest, KVal::Local(0)); // patched: the join - self.emit(LOp::Jump { - oc: one, - od: edest, - of: sfp, - }); - self.patch_local(bset, self.code.len()); - self.branch(b_block); - self.patch_local(eset, self.code.len()); - } - } - - /// `assert a != b`: `XOR` for `x = a + b`, a hinted `inv = x⁻¹`, then - /// `MUL p = x·inv` and `SET p = 1`, the write-once conflict being the - /// assertion (as for `assert a == b`). Sound because `x = 0` forces `p = 0` - /// whatever the hint, and `p` cannot then be `1`. - /// A compile-time-equal pair is a hard compile error. - fn lower_assert_ne(&mut self, a: &Expr, b: &Expr) { - // Compile-time literals (e.g. after `Const`-arg substitution): a - // trivially-true pair emits nothing, an equal pair is a hard error. - // Restricted to plain literals so a field value is never confused with a - // g-power index (unlike stack-index folding). - if let (Expr::Lit(x), Expr::Lit(y)) = (a, b) { - if x == y { - self.fail(format!("assert a != b: sides are the compile-time-equal literal {x}")) - }; - return; - } - let (la, lb) = (self.expr(a), self.expr(b)); - // x = a + b: nonzero ⇔ a != b - let x = self.pure(PureOp::Xor, la, lb); - let inv = self.fresh(); - self.pending.push(Hint::Resolved(RHint::Inverse { value: x, dst: inv })); - let p = self.fresh(); - self.emit(LOp::Mul { a: x, b: inv, c: p }); - self.set_const(p, F192::ONE); - } - - /// The frame cell holding `g^{k-1}`, the range-check product target, shared - /// by every check of that bound. - /// - /// An ordinary [`Self::const_cell`], so it is shared with any plain use of - /// the same constant: at `k = 1` the target is `g^0 = 1`, the very cell - /// [`Self::one`] hands out, which in `main` is also `self_fp`. Sound, since - /// the `SET` precedes every use and each later write is the write-once - /// equality, but a second WRITER on any of those paths would land on all. - fn bound_cell(&mut self, k: u64) -> Off { - self.const_cell(g_pow_u128((k - 1) as u128).into()) - } - - /// `assert log x < log GEN ** k`: the 3-cycle range check in the exponent - /// (`doc/leanvm/body/09-isa-programming.tex` §sec:prog-range-checks). With - /// `x = g^e`: - /// - /// 1. `DEREF` through `x`, so the bus proves `x = g^e` with `e < 2^h`; - /// 2. `MUL x·y` into the write-once cell holding `g^{k-1}`. The complement - /// `y = g^{k-1-e}` needs no hint, the result cell being already written, - /// so the runner back-solves the one unknown operand; - /// 3. `DEREF` through `y`, proving `y = g^f` with `f < 2^h`. - /// - /// Then `e + f ≡ k-1 (mod 2^64-1)` with `e, f < 2^h`, and a negative `k-1-e` - /// wraps to `≈ 2^64 ≫ 2^h`, so `e ≤ k-1` for ANY announced memory size, - /// provided `k ≤ 2^MIN_LOG_MEM`. Both `DEREF` destinations are unconstrained - /// touches, back-filled at the end of execution unless a cached read needs - /// their value sooner. Only the ADDRESS matters to the range check itself. - /// - /// A [`LtBound::Runtime`] bound reaches the same gadget through one extra - /// `MUL` for `g^{k-1} = Y·g^{-1}`, still back-solved rather than hinted, and - /// the `k ≤ 2^MIN_LOG_MEM` obligation moves to the program. - fn lower_assert_lt(&mut self, e: &Expr, bound: &LtBound) { - let kcell = match bound { - LtBound::Const(k) => { - if *k < 1 { - self.fail("range-check bound GEN ** 0 names the empty set") - }; - if *k > 1 << lean_vm::cpu::MIN_LOG_MEM { - self.fail(format!( - "range-check bound GEN ** {k} exceeds 2^{} (the minimum memory size)", - lean_vm::cpu::MIN_LOG_MEM - )) - }; - self.bound_cell(*k) - } - LtBound::Runtime(b) => { - // A bound that folds only after substitution (`GEN ** i` inside an - // `unroll`) reaches here rather than the arm above, and would then - // skip the `k <= 2^MIN_LOG_MEM` cap entirely. Reject it: the author - // wrote a compile-time bound and should get the compile-time check. - if self.try_field_const(b).is_some() { - self.fail(format!( - "a compile-time range-check bound must be written as `log GEN ** k` or an \ - integer, so that the 2^{} cap applies", - lean_vm::cpu::MIN_LOG_MEM - )) - }; - let bcell = self.expr(b); - let inv = self.const_cell(F192::new(primitives::field::G.inv().0, 0, 0)); - let c = self.fresh(); - self.emit(LOp::Mul { a: bcell, b: inv, c }); - c - } - }; - let x = self.expr(e); - let y = self.fresh(); // the complement g^{k-1-e}, back-solved by the MUL - let t1 = self.fresh(); // DEREF targets: unconstrained touch cells - let t2 = self.fresh(); - self.deref(x, 0, t1, DerefMode::Cell); - self.emit(LOp::Mul { a: x, b: y, c: kcell }); - self.deref(y, 0, t2, DerefMode::Cell); - } - - fn expr(&mut self, e: &Expr) -> Off { - // A wholly compile-time expression, whatever its shape, is one pooled - // `SET` ([`Self::const_cell`]): folded here once rather than arm by arm. - if let Some(v) = self.try_field_const(e) { - return self.const_cell(v); - } - match e { - Expr::Lit(_) | Expr::Gen | Expr::GPow(_) => unreachable!("a literal folds above"), - // Not folded above, so its exponent is not a compile-time integer, - // which `gpow_exp` reports. - Expr::GenPow(e) => { - let k = self.gpow_exp(e); - self.const_cell(g_pow_u128(k).into()) - } - Expr::Pow(b, e) => self.pow_expr(b, e), - Expr::Var(v) => match self.scope.bound(v).map(|b| b.val) { - Some(Binding::Stack(..)) => { - self.fail(format!("StackBuf `{v}` used as a scalar; index it (`{v}[k]`) or pass it to blake2s")); - } - Some(Binding::Gaddr(ga)) => self.materialize(ga), - Some(Binding::Scalar(o)) => o, - _ => { - // A `for` body that ASSIGNS to an enclosing name reads it before - // it binds it, and the capture set drops every name the body - // binds, so the read arrives here with nothing behind it. That is - // the loop-carry limitation rather than a typo, and it deserves - // the same courtesy the `StackBuf` case already gets: the - // tail-recursive helper threads its captures IN, never out, so an - // accumulator cannot come back. - if self.fn_name.starts_with("__loop") { - self.fail(format!( - "unbound variable `{v}` in a `for` loop body. If `{v}` names a value from \ - outside the loop that this body also assigns to, the loop cannot carry it: \ - the helper threads its captures in, not out. Assign to a new name inside \ - the body, or carry state through a `HeapBuf`." - )) - } - self.fail(format!("unbound variable `{v}`")) - } - }, - Expr::Add(a, b) => { - if let Some(x) = self.add_identity(a, b) { - return self.expr(x); - } - let (la, lb) = (self.expr(a), self.expr(b)); - self.pure(PureOp::Xor, la, lb) - } - Expr::Mul(a, b) => { - if let Some(x) = self.mul_identity(a, b) { - return self.expr(x); - } - let (la, lb) = (self.expr(a), self.expr(b)); - self.pure(PureOp::Mul, la, lb) - } - Expr::FieldDiv(a, b) => { - // q = a / b via the MUL write-once back-solve: emit `a = q * b` - // with the quotient `q` the unset operand. Witness-gen fills - // q = a·b⁻¹, and the MUL constraint pins q·b == a (so b == 0 is - // rejected unless a == 0). One MUL, no hint. The dividend cell - // `a` must already be written, which `self.expr(a)` guarantees. - let (la, lb) = (self.expr(a), self.expr(b)); - let q = self.fresh(); - self.emit(LOp::Mul { a: q, b: lb, c: la }); - q - } - // A well-formed one folds above, so this is a malformed call. - Expr::Call(f, _) if f == "f192" => self.fail("f192 needs three literal u64 limbs"), - // Folded above when it is what it claims to be, so reaching here means - // it is not: name that, rather than reporting an unknown function. - Expr::Call(f, args) if f == "const" => { - if args.len() != 1 { - self.fail(format!("const(...) takes one expression, got {}", args.len())) - }; - self.fail(format!( - "const(...) asks for a compile-time integer, and `{:?}` is not one", - args[0] - )) - } - Expr::Call(f, args) if f == "addr" => { - let ga = self.stack_addr(args); - self.materialize(ga) - } - Expr::Call(f, args) if f == "hint_log2_ceil" => { - // Computed advice: the prover fills g^log2_ceil (base-2 ceil-log) of the value in - // `bits` (a `nbits`-bit buffer), floored at `floor`. Returned - // UNCONSTRAINED, so the caller (log2_ceil) re-verifies it. Same - // "prover computes, circuit checks" pattern as `/`. - if args.len() != 3 { - self.fail(format!( - "hint_log2_ceil takes three arguments, `(bits, nbits, floor)`, got {}", - args.len() - )) -}; - let nbits = self.const_index(&args[1]); - let floor = self.const_index(&args[2]); - let bits = self.bits_dest(&args[0], nbits, "hint_log2_ceil"); - let dst = self.fresh(); - self.pending.push(Hint::Resolved(RHint::Log2Ceil { - bits, - dst, - nbits, - floor, - })); - dst - } - Expr::Call(f, args) => { - let d = self.call(f, args, 1)[0]; - self.take_inline_ret_cell(d) - } - Expr::HeapBuf(n) => { - let arr = self.fresh(); - self.heap_sizes.insert(arr, *n as u128); - // Allocate before the next instruction reads the pointer. - self.pending.push(Hint::AllocBuffer { - ptr: arr, - size: *n as u32, - }); - arr - } - Expr::HeapBufDyn(e) => { - // Evaluate the size first (its cell must be written when the - // alloc hint fires), then allocate before the pointer is read. - let size = self.expr(e); - let arr = self.fresh(); - self.pending.push(Hint::AllocBufferDyn { ptr: arr, size }); - arr - } - Expr::StackBuf(_) => { - self.fail("StackBuf(n) must be bound to a name: `x = StackBuf(n)`") - } - // A frame cell IS the answer; a heap cell needs a `DEREF` to read. - Expr::Index(arr, idx) => { - if let Some(c) = self.frame_cell(arr, idx) { - return c; - } - let (ptr, o2) = self.heap_addr(arr, idx); - if let Some(&cell) = self.scope.load_cells.get(&(ptr, o2)) { - return cell; - } - let dst = self.fresh(); - self.deref(ptr, o2, dst, DerefMode::Cell); - dst - } - Expr::Sub(..) | Expr::Div(..) | Expr::Mod(..) => { - self.fail(format!( - "`-`, `//`, `%` are compile-time only (field subtraction is `+`); use them in an index, a bound, or a `Const` argument, got `{e:?}`" - )) - } - Expr::Slice(..) => self.fail("a slice is not a scalar; it is only a blake2s operand"), - Expr::ListLit(..) => self.fail("a list literal must be bound to a name: `x = [a, b]`"), - } - } - - /// `base ** e` (non-`GEN` base, compile-time exponent `e`): a fully-constant - /// base folds to one `SET`; a runtime base is raised by square-and-multiply. - fn pow_expr(&mut self, b: &Expr, e: &Expr) -> Off { - let k = self - .try_const_index(e) - .unwrap_or_else(|| self.fail(format!("`**` exponent must be a compile-time integer, got `{e:?}`"))); - // Fully constant: evaluate in the field and emit a single `SET`. - if let Some(bc) = self.try_field_const(b) { - return self.const_cell(field_pow(bc, k)); - } - if k == 0 { - let o = self.fresh(); - self.set_const(o, F192::ONE); - return o; - } - // Runtime base: square-and-multiply over the compile-time exponent bits. - let base = self.expr(b); - let hi = 31 - k.leading_zeros(); // top set bit (k >= 1) - let mut acc = base; - for bit in (0..hi).rev() { - acc = self.pure(PureOp::Mul, acc, acc); - if (k >> bit) & 1 == 1 { - acc = self.pure(PureOp::Mul, acc, base); - } - } - acc - } - - /// Evaluate `e` into `dst`, writing constants, arithmetic, heap reads and call results directly. - /// Writing an existing cell enforces equality under write-once memory. - fn expr_into(&mut self, e: &Expr, dst: Off) { - // A wholly compile-time expression (a literal, a constant-array element, - // constant arithmetic) is one `SET` into `dst`, not a heap read. - if let Some(v) = self.try_field_const(e) { - self.set_const(dst, v); - return; - } - match e { - // Heap read straight into dst (a stack read falls through to the copy). - // Through the same resolver as every other index, so the frame/heap - // split is written once: a heap read `DEREF`s straight into `dst`, and a - // frame read is the cell, copied. - Expr::Index(arr, idx) => match self.frame_cell(arr, idx) { - Some(c) => self.copy(c, dst), - None => { - let (ptr, o2) = self.heap_addr(arr, idx); - // A DEREF links two unwritten sides; a MUL copy would read the - // source, writing zero to it. - self.deref(ptr, o2, dst, DerefMode::Cell); - } - }, - Expr::Pow(b, e) => { - let v = self.pow_expr(b, e); - self.copy(v, dst); - } - Expr::Add(a, b) => { - if let Some(x) = self.add_identity(a, b) { - self.expr_into(x, dst); - } else { - // Not `pure`: `dst` is the caller's, so it may be written - // again and the second write be the assertion. See its doc. - let (la, lb) = (self.expr(a), self.expr(b)); - self.emit(LOp::Xor { a: la, b: lb, c: dst }); - } - } - Expr::Mul(a, b) => { - if let Some(x) = self.mul_identity(a, b) { - self.expr_into(x, dst); - } else { - let (la, lb) = (self.expr(a), self.expr(b)); - self.emit(LOp::Mul { a: la, b: lb, c: dst }); - } - } - // A call writes its single return value straight into `dst` (an - // aliased inline return materializes, then copies into `dst`). - Expr::Call(f, args) => { - self.inline_stack_ret = None; - self.call_into(f, args, &[dst]); - let v = self.take_inline_ret_cell(dst); - if v != dst { - self.copy(v, dst); - } - } - _ => { - let v = self.expr(e); - self.copy(v, dst); - } - } - } - - fn stmt(&mut self, s: &Stmt) { - let tail = std::mem::take(&mut self.tail_call); - // Every diagnostic raised while lowering this statement, and every - // instruction it emits, is attributed to this line. - self.cur_line = s.line; - match &s.kind { - StmtKind::Let(name, e) => match e { - // `x = StackBuf(n)`: bind a run of `n` consecutive frame cells. - Expr::StackBuf(n) => { - let base = self.alloc_stack(*n as u32); - self.rebind(name, Binding::Stack(base, *n as u32)); - } - // `x = [a, b, …]`: an initialized StackBuf. Allocate the run and - // write each element in place, through the ordinary stack-store - // path. Elements are lowered before `name` rebinds, so they may - // read its old binding (`fs = [fs[1], fs[0]]`). - Expr::ListLit(es) => { - let base = self.alloc_stack(es.len() as u32); - for (k, el) in es.iter().enumerate() { - self.expr_into(el, base + k as u32); - } - self.rebind(name, Binding::Stack(base, es.len() as u32)); - } - // `p = addr(sb)` binds the address itself, so the offset folds - // into every later access; in any other position `expr` has to - // materialize it into a cell instead. - Expr::Call(f, cargs) if f == "addr" => { - let ga = self.stack_addr(cargs); - self.rebind(name, Binding::Gaddr(ga)); - } - // `x = other_stackbuf`: a compile-time alias of the same cell - // run (zero instructions), the chaining-state idiom `st = sn` - // of an MD loop. - Expr::Var(v) if self.scope.stack(v).is_some() => { - let (base, size) = self.scope.stack(v).expect("guarded above"); - self.rebind(name, Binding::Stack(base, size)); - } - _ => { - // NOTE: `name`'s old binding stays visible while the RHS is - // lowered (the MD-chain idiom `cvb = obs(cvb, x)` reads it); - // each terminal path below unbinds/rebinds afterwards. - // A compile-time integer binding (a literal, or an expression - // that folds: `FOLDBASE[lvl] + j`, `n // 2`, `len(A) - 1`) is - // usable as a compile-time index / bound / exponent, and that - // role survives the value binding chosen below. - let known = self.eval(e); - // A symbolic g-address (a constant g-power or a shifted - // pointer) or a compile-time field constant stays virtual: - // no instruction here, folded / materialized only on demand. - if let Some(ga) = known.addr { - self.rebind(name, Binding::Gaddr(ga)); - } else if let Some(c) = known.field { - self.rebind(name, Binding::FConst(c)); - } else if let Some(k) = known.int { - // Integer-only fold (`//`, `-`, `%` of constants): a - // compile-time value too, and as a scalar it is the field - // element with those 128 bits, materialized on demand. - self.rebind(name, Binding::FConst(lit_field(k))); - } else if let Expr::Call(cf, cargs) = e - && self.defs.contains_key(cf.as_str()) - { - // A bare `name = call(...)` of a user function: bind per - // the inlined return's RetBind, aliasing its StackBuf run - // or folded g-address at zero copies (the `cvb = obs(...)` - // / advanced-cursor idiom), else (a plain scalar, or a - // real call) bind the dst cell. Embedded calls do NOT - // take this path: `expr` materializes theirs - // ([`Self::take_inline_ret_cell`]). - let o = self.call(cf, cargs, 1)[0]; - let b = ret_binding(self.inline_stack_ret.take().and_then(|b| b.into_iter().next()), o); - self.rebind(name, b); - } else { - let o = self.expr(e); - self.rebind(name, Binding::Scalar(o)); - } - // The integer reading of the SAME expression, if it has one, - // rides alongside whichever value binding was chosen above. - // It is attached after, because a rebind clears it, and the - // RHS above still had to see `name`'s old reading. - if let Some(k) = known.int { - self.scope.set_int(name, k); - } - } - }, - StmtKind::LetTuple(names, f, args) => { - let dsts = self.call(f, args, names.len()); - // Each returned value binds per its RetBind (alias a StackBuf run - // or folded g-address, else take the scalar dst cell); a real call - // leaves the field None, so every name binds its scalar dst. - let binds = self.inline_stack_ret.take(); - for (i, (n, d)) in names.iter().zip(&dsts).enumerate() { - let b = ret_binding(binds.as_ref().and_then(|b| b.get(i).copied()), *d); - self.rebind(n, b); - } - } - // `a + b` into the frame's zero cell: the double write IS the - // assertion, so the `SET .. = 0` a fresh destination needed is gone - // and no cell is burned. Not through `pure`, for the same reason: - // sharing this cell would drop the assertion. - StmtKind::AssertEq(a, b) => { - let (la, lb) = (self.expr(a), self.expr(b)); - let z = self.zero(); - self.emit(LOp::Xor { a: la, b: lb, c: z }); - } - StmtKind::AssertNe(a, b) => self.lower_assert_ne(a, b), - StmtKind::AssertLt(e, bound) => self.lower_assert_lt(e, bound), - StmtKind::HintWitness { dest, name } => self.lower_hint_witness(dest, name), - // One hinted value into one fresh cell, bound to `name`. The run form - // names its destination's physical cells, and a scalar's cell is never - // a store target at all, being reachable only through a name. - StmtKind::LetHintWitness { name, stream } => { - let dst = self.fresh(); - self.pending.push(Hint::Resolved(RHint::WitnessStack { - name: stream.clone(), - base: dst, - len: 1, - })); - self.rebind(name, Binding::Scalar(dst)); - } - StmtKind::Print { label, value } => { - // Prover-side debug print: evaluate the value into a cell, hang - // a Print hint on a no-op anchor so it fires exactly here (and - // only on this path), at witness generation. No constraints. - let cell = self.expr(value); - self.pending.push(Hint::Resolved(RHint::Print { - label: label.clone(), - cell, - })); - self.anchor(); - } - StmtKind::If { - eq, - lhs, - rhs, - then, - els, - force_const, - } => self.lower_if(*eq, lhs, rhs, then, els, *force_const), - StmtKind::Match { targets, x, arms } => self.lower_match(targets, x, arms), - StmtKind::Call(f, args) => { - if !self.lower_builtin(f, args) { - self.call(f, args, 0); - } - } - StmtKind::Store(arr, idx, val) => { - // A frame write places the value in the cell; a heap write is the - // `DEREF` asserting `m[arr·idx] == val` (write-once). The VALUE is - // lowered first on the heap path, since the address may read a cell - // the value writes (`hb[sb[0]] = f(sb, …)`), and Python's own - // evaluation order for `a[i] = v` is the same. - if let Some(c) = self.frame_cell(arr, idx) { - self.expr_into(val, c); - } else { - let v = self.expr(val); - let (ptr, o2) = self.heap_addr(arr, idx); - self.deref(ptr, o2, v, DerefMode::Cell); - } - } - StmtKind::Return(es) => self.lower_return(es), - StmtKind::CallIfNe(lhs, rhs, callee, args) => { - // A conditional call: the frame setup runs either way, and the - // `JUMP`'s nonzero test decides whether the callee is entered, - // so the not-taken path continues straight after it. In tail - // position the callee inherits THIS frame's `retpc`/`retfp`, so - // a `mul_range` loop builds no unwind chain: only the final - // iteration returns, straight to the loop's original caller. - let (la, lb) = (self.expr(lhs), self.expr(rhs)); - // x = lhs + rhs; x != 0 ⇔ lhs != rhs - let x = self.pure(PureOp::Xor, la, lb); - self.lower_call(callee, args, Some(x), &[], tail); - } - StmtKind::For { var, lo, hi, body } => self.lower_for(var, *lo, hi, body), - // Compile-time unrolling: emit the body per integer, the counter - // substituted as its literal. Every copy executes (this is - // straight-line code, not a branch), so bindings simply rebind (a - // fresh binding per iteration) and lazy caches persist. - StmtKind::Unroll { var, lo, hi, body } => { - let bound = |s: &Self, e: &Expr| { - s.try_const_index(e).unwrap_or_else(|| { - self.fail(format!("unroll bounds must be compile-time integers, got `{e:?}`")) - }) - }; - let (lo, hi) = (bound(self, lo), bound(self, hi)); - if lo > hi { - self.fail(format!("unroll(a, b) needs a <= b, got ({lo}, {hi})")) - }; - for j in lo..hi { - for s in subst_stmts(body, var, &Expr::Lit(j as u128)) { - self.stmt(&s); - } - } - } - } - } - - /// `for i in mul_range(GEN**lo, GEN**hi)` → a single tail-recursive helper, with the - /// exit test folded into the recursion's condition (no separate branch, no - /// is-zero gadget): - /// ```text - /// loop(i): - /// - /// j = i·g - /// if j != g^hi: loop(j) // JUMP's nonzero test on (j − g^hi) - /// return - /// caller: if lo != hi: loop(g^lo) // resolved at compile time - /// ``` - /// Free variables of the body that are bound in the enclosing scope are - /// captured by value as extra helper parameters (e.g. a `HeapBuf` pointer - /// threaded through the loop). - fn lower_for(&mut self, var: &str, lo: u64, hi: &ForBound, body: &[Stmt]) { - let id = *self.loop_ctr; - *self.loop_ctr += 1; - let loop_name = format!("__loop{id}"); - let mut shadowed = std::collections::HashSet::new(); - binds_anywhere(body, &mut shadowed); - // Returns and counter rebinding can change the number of iterations. - if !contains_return(body) && !shadowed.contains(var) { - self.loop_bounds.insert( - loop_name.clone(), - ( - lo, - match hi { - ForBound::Const(end) => Some(*end), - ForBound::Runtime(_) => None, - }, - ), - ); - } - if std::env::var("DBG_LOOPS").is_ok() { - let bound = match hi { - ForBound::Const(h) => format!("g^{lo}..g^{h}"), - ForBound::Runtime(e) => format!("g^{lo}..{e:?}"), - }; - eprintln!("DBG_LOOPS {loop_name} in {} for {var} in {bound}", self.fn_name); - } - // A runtime stop bound is evaluated once here and threaded through the - // helper as an extra leading parameter (the exit test compares the - // advanced counter against it each iteration). - let bound_var = format!("__bound{id}"); - let (exit, entry_bound): (Expr, Expr) = match hi { - ForBound::Const(hi) => (Expr::GPow(*hi as u128), Expr::GPow(*hi as u128)), - ForBound::Runtime(e) => (Expr::Var(bound_var.clone()), e.clone()), - }; - - // Determine captures: referenced − locally-bound − the counter, kept if - // they exist in the enclosing scope (deterministic order). - let mut referenced = Vec::new(); - let mut bound = std::collections::HashSet::new(); - bound.insert(var); - for s in body { - free_vars_stmt(s, &mut referenced, &mut bound); - } - // Everything the body binds ANYWHERE, branch-local or not. `bound` above - // is scoped, which is what makes the capture set right; this flat one - // only answers "does the body have an `r` of its own?", which is what the - // StackBuf rejection below needs: a body that merely SHADOWS an enclosing - // `StackBuf` never touches it, so rejecting it names a capture that is - // not happening. - let mut captures = Vec::new(); - let mut seen = std::collections::HashSet::new(); - for r in &referenced { - if bound.contains(r) { - continue; - } - // A StackBuf is a run of cells, not a single scalar arg, and the - // tail-recursive loop helper can't thread one across iterations, so a - // StackBuf from the enclosing scope can't be captured. Reject with a - // clear error (not the misleading "unbound variable" the capture drop - // would otherwise trigger). Keep it inside the loop body, or carry - // state through a `HeapBuf`. - if self.scope.stack(r).is_some() && !shadowed.contains(r) { - self.fail(format!( - "StackBuf `{r}` cannot be captured into a `for` loop; \ - define it inside the loop body or carry state via a `HeapBuf`" - )); - } - // A compile-time field constant is capturable too: the body becomes - // its own function, so the constant is not in scope there, and the - // helper takes it as a parameter that the call site materializes - // with one `SET`. Dropping it made `c = 5` followed by a loop that - // reads `c` fail as "unbound variable", which named neither the - // cause nor a fix. - if matches!( - self.scope.bound(r).map(|b| b.val), - Some(Binding::Scalar(_) | Binding::Gaddr(_) | Binding::FConst(_)) - ) && seen.insert(*r) - { - captures.push((*r).to_string()); - } - } - - // The helper takes the counter, the runtime bound (if any), then the - // captures. `cap_args` builds an argument list (a leading expression, - // the bound, then the captures by name). - let runtime = matches!(hi, ForBound::Runtime(_)); - let mut params = vec![var.to_string()]; - if runtime { - params.push(bound_var.clone()); - } - params.extend(captures.iter().cloned()); - let cap_args = |first: Expr, bound: Expr| { - let mut a = vec![first]; - if runtime { - a.push(bound); - } - a.extend(captures.iter().map(|c| Expr::Var(c.clone()))); - a - }; - - // loop(i, [bound,] caps): run the body, advance to j = i·g, and - // tail-recurse while j != stop. The exit test is the recursive call's - // own condition (`JUMP`'s nonzero check on j − stop): no is-zero - // gadget, no inverse hint, and no extra call beyond the one a loop - // iteration already makes. - let next_var = format!("__next{id}"); - let next = Expr::Mul(Box::new(Expr::Var(var.to_string())), Box::new(Expr::Gen)); - let mut loop_body: Vec = body.to_vec(); - // The counter advance and the self-call belong to the `for` header. - let at = |kind| Stmt::new(self.cur_line, kind); - loop_body.push(at(StmtKind::Let(next_var.clone(), next))); - loop_body.push(at(StmtKind::CallIfNe( - Expr::Var(next_var.clone()), - exit, - loop_name.clone(), - cap_args(Expr::Var(next_var), Expr::Var(bound_var.clone())), - ))); - loop_body.push(at(StmtKind::Return(vec![]))); - self.queue.push(Func { - name: loop_name.clone(), - params: params - .into_iter() - .map(|name| Param { - name, - kind: ParamKind::Runtime(Shape::Scalar), - }) - .collect(), - return_shapes: vec![], - body: loop_body, - inline: false, - }); - - // Enter the loop iff it runs at least once: compile-time for constant - // bounds (an empty range compiles to nothing), a conditional call on - // `g^lo != stop` for runtime ones. - match hi { - ForBound::Const(hi) => { - if lo != *hi { - self.call( - &loop_name, - &cap_args(Expr::GPow(lo as u128), Expr::GPow(*hi as u128)), - 0, - ); - } - } - ForBound::Runtime(_) => { - let stmt = Stmt::new( - self.cur_line, - StmtKind::CallIfNe( - Expr::GPow(lo as u128), - entry_bound.clone(), - loop_name, - cap_args(Expr::GPow(lo as u128), entry_bound), - ), - ); - self.stmt(&stmt); - } - } - } -} - -fn contains_return(body: &[Stmt]) -> bool { - body.iter().any(|stmt| match &stmt.kind { - StmtKind::Return(_) => true, - StmtKind::If { then, els, .. } => contains_return(then) || contains_return(els), - StmtKind::For { body, .. } | StmtKind::Unroll { body, .. } => contains_return(body), - _ => false, - }) -} - -/// The literal `k` when `hi` is syntactically `lo + k` (either operand order): -/// the shape of a runtime slice, whose bounds cannot be evaluated at compile -/// time. -fn plus_k(lo: &Expr, hi: &Expr) -> Option { - match hi { - Expr::Add(a, b) => match (a.as_ref(), b.as_ref()) { - (Expr::Lit(k), other) | (other, Expr::Lit(k)) if other == lo => Some(*k), - _ => None, - }, - _ => None, - } -} - -/// Lower one function to its instruction list and frame size. -pub(crate) fn lower_func( - f: &Func, - queue: &mut Vec, - loop_ctr: &mut usize, - defs: &HashMap<&str, &Func>, - const_arrays: &HashMap<&str, &[F192]>, - with_filler: bool, - loop_bounds: &mut HashMap)>, -) -> Lowered { - let mut names: HashMap = HashMap::new(); - for (i, p) in f.params.iter().enumerate() { - assert!( - !const_arrays.contains_key(p.name.as_str()), - "`{}`: parameter `{}` collides with a top-level constant array, whose name is \ - reserved (zkDSL.md §Global constants)", - f.name, - p.name - ); - // A `StackBuf(n)` parameter binds the run the caller wrote, exactly as a - // local `StackBuf(n)` binds one it allocated. - let off = Abi::arg(f.param_shapes(), i); - let val = match p.shape() { - Shape::StackBuf(n) => Binding::Stack(off, n), - Shape::Scalar => Binding::Scalar(off), - }; - names.insert(p.name.clone(), Bound { val, int: None }); - } - // Reserve [0,1] retpc/retfp, params, then the flattened return area, then - // locals. A StackBuf(n) return occupies n consecutive physical slots. - let n_ret_cells: u32 = f.return_shapes.iter().map(|s| s.cells()).sum(); - let arg_cells = Abi::arg_cells(f.param_shapes()); - let abi_end = Abi::end(arg_cells, n_ret_cells); - // Loop callers write the callee's own fp just past its arguments. That - // equality is tied to the JUMP target, so the loop can use it directly. - let loop_frame = loop_bounds.contains_key(&f.name); - let mut lowerer = FnLower { - scope: Scope { - self_fp_off: loop_frame.then_some(abi_end), - names, - ..Default::default() - }, - next: abi_end + u32::from(loop_frame), - arg_cells, - return_shapes: &f.return_shapes, - is_main: f.name == "main", - fn_name: &f.name, - tail_call: false, - code: Vec::new(), - cur_line: 0, - heap_sizes: HashMap::new(), - inline_ret: None, - inline_stack_ret: None, - - pending: Vec::new(), - inline_calls: Vec::new(), - loop_bounds, - queue, - loop_ctr, - defs, - const_arrays, - }; - for (i, s) in f.body.iter().enumerate() { - // Tail position: a conditional call whose only successor is a bare - // `return`, in a function that returns nothing. The `mul_range` helper - // ends exactly like this, so its self-call stops building an unwind - // chain. - lowerer.tail_call = !lowerer.is_main - && f.return_shapes.is_empty() - && matches!(s.kind, StmtKind::CallIfNe(..)) - && matches!(f.body.get(i + 1).map(|n| &n.kind), Some(StmtKind::Return(r)) if r.is_empty()); - lowerer.stmt(s); - } - let mut filler = Vec::new(); - if lowerer.is_main { - lowerer.halt(); // main terminates at the sentinel pc, not by falling off - if with_filler { - // Past the halt, so no program code reaches them: the fill blocks are cycles - // the interpreter enters on its own ([`FnLower::lower_filler_blocks`]). - filler = lowerer.lower_filler_blocks(); - } - } else if !matches!(f.body.last().map(|s| &s.kind), Some(StmtKind::Return(_))) { - // A function must never fall off its end into whatever code the - // layout placed next: append the implicit bare return. - let last = f.body.last().map_or(0, |s| s.line); - lowerer.stmt(&Stmt::new(last, StmtKind::Return(vec![]))); - } - Lowered { - name: f.name.clone(), - code: lowerer.code, - frame_size: lowerer.next, - filler, - } -} diff --git a/crates/lean_compiler/src/lower/builtins.rs b/crates/lean_compiler/src/lower/builtins.rs deleted file mode 100644 index 68e39b635..000000000 --- a/crates/lean_compiler/src/lower/builtins.rs +++ /dev/null @@ -1,328 +0,0 @@ -//! The precompile and the hints: the two places a value arrives without an -//! instruction computing it. -//! -//! `blake2s` is a STATEMENT, not an expression: it writes its digest into a -//! two-cell run the caller names, so a pre-written destination checks the digest -//! instead of computing it, by the same write-once rule as any store. -//! -//! A hint writes values the prover chose and the circuit did not, so **the -//! program must constrain them**. A hint names its destination's PHYSICAL cells, -//! and since every store emits there is nothing for the compiler to prepare: a -//! later `s[k] = ` is a second write of that cell, which is the -//! assertion that pins the hinted value. - -use super::*; - -impl FnLower<'_> { - /// `blake2s(a, b, out)`: the digest of the two 256-bit operands lands in the - /// existing 2-cell run `out` (write-once: if `out` was already written, this - /// asserts the digest equals it). A heap `out` slice takes the digest via a - /// fresh stack pair and two `DEREF`s after the hash, the store direction - /// being the same instruction as the load (write-once fills the unset side). - /// Keyword arguments set the metadata: `counter=` / `final=` / `last_node=` - /// build it at compile time, `md=` takes the whole word from a value the - /// program computed. - fn lower_blake2s(&mut self, args: &[Expr]) { - let first_kw = args - .iter() - .position(|a| matches!(a, Expr::Call(name, _) if name.starts_with("__kw_"))) - .unwrap_or(args.len()); - if first_kw != 3 { - self.fail("blake2s takes three positional arguments: (a, b, out)") - }; - if !(args[first_kw..] - .iter() - .all(|a| matches!(a, Expr::Call(name, v) if name.starts_with("__kw_") && v.len() == 1))) - { - self.fail("keyword arguments must follow the three positional blake2s arguments") - }; - let mut kwargs: HashMap<&str, &Expr> = HashMap::new(); - for kw in &args[first_kw..] { - let Expr::Call(name, value) = kw else { unreachable!() }; - let key = name.strip_prefix("__kw_").unwrap(); - if kwargs.insert(key, &value[0]).is_some() { - self.fail(format!("duplicate blake2s keyword `{key}`")) - }; - } - let allowed = ["cv", "counter", "final", "last_node", "md"]; - if !(kwargs.keys().all(|k| allowed.contains(k))) { - // Sorted: a `HashMap`'s order would make the same mistake report - // differently between builds. - let mut bad: Vec<&&str> = kwargs.keys().filter(|k| !allowed.contains(k)).collect(); - bad.sort_unstable(); - self.fail(format!("unknown blake2s keyword {bad:?}; the keywords are {allowed:?}")) - }; - let customized = kwargs.keys().any(|k| matches!(*k, "counter" | "final" | "last_node")); - // `md=` hands over the whole metadata word as a runtime value, so it - // replaces the three keywords that would otherwise build it. - let runtime_md = kwargs.get("md").copied(); - if runtime_md.is_some() && customized { - self.fail("blake2s md= is the whole metadata word, so counter=, final= and last_node= cannot come with it") - }; - if kwargs.contains_key("cv") && !customized && runtime_md.is_none() { - self.fail( - "blake2s with cv= requires one of counter=, final=, last_node= or md=, since a chained \ - block is not the default one-block hash", - ) - }; - - let a = self.blake2s_input(&args[0]); - let b = self.blake2s_input(&args[1]); - let (c, heap_out) = match self.blake2s_operand(&args[2]) { - CellRun::Stack { base, .. } => (base, None), - CellRun::Heap { ptr, lo, .. } => (self.alloc_stack(2), Some((ptr, lo))), - }; - let cv = if let Some(value) = kwargs.get("cv") { - self.blake2s_cv(value) - } else { - self.default_blake2s_cv() - }; - let md = match runtime_md { - // A metadata word the program computes, which is what lets a hash whose - // block count is only known at run time carry the byte counter the - // standard asks for (doc §sec:prog-byte-counter). It owes the same - // canonical embedding as every other operand: the memory interaction - // carries a literal zero above its two low limbs. - // - // Aliasing the digest destination is the one case write-once does not - // catch: the runner reads the metadata before storing the digest, while - // the witness reads the finished memory image, so the two disagree and - // the proof fails its opening rather than saying why. - Some(expr) => { - let md = self.expr(expr); - if md == c || md == c + 1 { - self.fail("blake2s md= must not name a cell of the digest destination") - }; - md - } - None => { - let const_kw = |this: &Self, name: &str, default: u128| -> u128 { - kwargs - .get(name) - .map(|e| { - this.try_const_int(e).unwrap_or_else(|| { - self.fail(format!( - "BLAKE2s `{name}` must be a compile-time integer, got `{e:?}`; \ - a metadata word computed at run time goes through md=" - )) - }) - }) - .unwrap_or(default) - }; - // BLAKE2s metadata is just the cumulative byte counter and two flags, so - // a multi-block hash is `counter = 64 * blocks_before + bytes_in_this_block` - // and `final = 1` on the last block. The default is the one-block hash of - // a full 64-byte input, which is what `vmhash::compress` and every Merkle - // node use. - let counter = const_kw(self, "counter", 64); - let counter = u64::try_from(counter) - .unwrap_or_else(|_| self.fail(format!("blake2s counter= {counter} does not fit in u64"))); - let f0 = if const_kw(self, "final", if customized { 0 } else { 1 }) != 0 { - lean_vm::hash_flock::FINAL_FLAG - } else { - 0 - }; - let f1 = if const_kw(self, "last_node", 0) != 0 { - u32::MAX - } else { - 0 - }; - // A compile-time metadata is a pooled `SET`: one per distinct value - // per frame, however many compressions read it. - self.const_cell(lean_vm::hash_flock::metadata(counter, f0, f1)) - } - }; - // Each operand is two 128-bit chunk cells; the flexible opcode addresses - // the four input cells independently (`blake2s_input` forwards the real - // chunk sources where it can). The digest occupies the two consecutive - // output cells `c, g·c`. - self.emit(LOp::Blake2s { - ins: [a[0], a[1], b[0], b[1]], - cv, - c, - md, - }); - if let Some((ptr, lo)) = heap_out { - for k in 0..2 { - self.deref(ptr, lo + k, c + k, DerefMode::Cell); - } - } - } - - /// The statement-position builtins, `true` if `f` was one of them (else the - /// caller emits an ordinary call). The `hint_*` ones queue prover-side - /// advice, re-checked in-circuit by their caller: `hint_decompose_bits` - /// writes a value's bits into a buffer, `hint_decompose_bits_exponent` the - /// bits of `n` where the value is `g^n` (a bounded dlog at witness - /// generation), `hint_f192_limbs` a value's coordinate limbs. - pub(super) fn lower_builtin(&mut self, f: &str, args: &[Expr]) -> bool { - match f { - "hint_decompose_bits" | "hint_decompose_bits_exponent" => { - if args.len() != 3 { - self.fail(format!( - "{f} takes three arguments, `(bits, value, nbits)`, got {}", - args.len() - )) - }; - let nbits = self.const_index(&args[2]); - let bits = self.bits_dest(&args[0], nbits, f); - let value = self.expr(&args[1]); - self.pending.push(Hint::Resolved(if f == "hint_decompose_bits" { - RHint::BitDecompose { value, bits, nbits } - } else { - RHint::BitDecomposeExp { value, bits, nbits } - })); - } - "blake2s" => self.lower_blake2s(args), - "assert_in_k" => { - if args.len() != 2 { - self.fail("assert_in_k(a, b) takes two scalar cells") - }; - let a = self.expr(&args[0]); - let b = self.expr(&args[1]); - let zero = self.zero(); - self.emit(LOp::Jump { oc: zero, od: a, of: b }); - } - "hint_f192_limbs" => { - if args.len() != 2 { - self.fail(format!( - "hint_f192_limbs takes two arguments, `(dest, value)`, got {}", - args.len() - )) - }; - let (base, len) = self.stack_of(&args[0]).unwrap_or_else(|| { - self.fail(format!( - "hint_f192_limbs writes 1..=3 frame cells, so its destination must be a \ - StackBuf, got `{:?}`", - args[0] - )) - }); - if !((1..=3).contains(&len)) { - self.fail("hint_f192_limbs destination must have 1..=3 cells") - }; - let value = self.expr(&args[1]); - // Names the physical cells, as the two consumers above do: whatever - // the program stores into them afterwards is a second write, and so - // the assertion that pins these limbs. - self.pending - .push(Hint::Resolved(RHint::FieldLimbs { value, base, len })); - } - _ => return false, - } - true - } - - /// Resolve a `blake2s` operand: a [`Self::cell_run`] pinned to exactly 2 - /// cells, a 256-bit value being two 128-bit cells. Stack operands are used - /// in place; heap operands must be bridged through the stack, since - /// `BLAKE2s` addresses only frame cells (see [`Self::blake2s_input`]). - fn blake2s_operand(&mut self, e: &Expr) -> CellRun { - let run = self.cell_run(e); - if run.cells() != 2 { - self.fail("a blake2s operand must span exactly 2 cells (two 128-bit words); slice a larger buffer: `buf[lo:lo + 2]`") - }; - run - } - - /// A `blake2s` *input* operand as its two independently-addressed 128-bit - /// chunk bases (each chunk is ONE 128-bit cell): stack runs in place; a heap - /// slice is pulled into a fresh stack pair first, one `DEREF` per cell - /// (`m[ptr·g^{lo+k}] == m[fp+t+k]`, the `β` immediate doing the pointer - /// offset). The heap cells must already be written. - /// - /// A LIST LITERAL names its two words directly and allocates nothing. The - /// opcode addresses its four input chunks independently, so an operand - /// assembled out of values living elsewhere never has to be gathered into a - /// consecutive run: `blake2s([a, b], …)` is the spelling that says so. - pub(super) fn blake2s_input(&mut self, e: &Expr) -> [Off; 2] { - if let Expr::ListLit(words) = e { - if words.len() != 2 { - self.fail(format!( - "a blake2s operand written as a list needs exactly 2 words, got {}", - words.len() - )) - }; - return [self.expr(&words[0]), self.expr(&words[1])]; - } - match self.blake2s_operand(e) { - CellRun::Stack { base, .. } => [base, base + 1], - CellRun::Heap { ptr, lo, .. } => { - let t = self.alloc_stack(2); - for k in 0..2 { - self.deref(ptr, lo + k, t + k, DerefMode::Cell); - } - [t, t + 1] - } - } - } - - fn default_blake2s_cv(&mut self) -> Off { - if let Some(o) = self.scope.blake2s_iv { - return o; - } - let o = self.alloc_stack(2); - for (k, value) in lean_vm::hash_flock::IV_CELLS.into_iter().enumerate() { - self.set_const(o + k as u32, value); - self.scope - .const_cells - .entry([value.c0, value.c1, value.c2]) - .or_insert(o + k as u32); - } - self.scope.blake2s_iv = Some(o); - o - } - - /// A computed-advice bit buffer's destination ([`BitsDest`]). Not - /// [`Self::cell_run`]: these builtins take a bare `HeapBuf` and carry the - /// length in `nbits`, where a cell run would demand a slice. - pub(super) fn bits_dest(&mut self, e: &Expr, nbits: u32, what: &str) -> BitsDest { - match self.stack_of(e) { - Some((base, len)) => { - if len < nbits { - self.fail(format!( - "{what} needs {nbits} cells, its StackBuf destination has {len}" - )) - }; - // The hint names the physical cells, as `hint_f192_limbs` does. - BitsDest::Stack(base) - } - None => { - // Bounds-checked like the `StackBuf` arm above, and like every - // other heap consumer. Without this a `HeapBuf` destination wrote - // `nbits` cells with nothing checking the buffer held them, so the - // bits ran on into the next buffer while the same call with a - // `StackBuf` destination was rejected. - self.check_heap_bound(e, 0, u128::from(nbits)); - BitsDest::Heap(self.expr(e)) - } - } - } - - /// `hint_witness(dest, "name")`: resolve `dest` to a run of cells and - /// queue the witness-fill hint (no instructions: the values are written - /// by the runner before the next instruction executes, unconstrained). - pub(super) fn lower_hint_witness(&mut self, dest: &Expr, name: &str) { - let name = name.to_string(); - let hint = match self.cell_run(dest) { - CellRun::Stack { base, len } => RHint::WitnessStack { name, base, len }, - CellRun::Heap { ptr, lo, len } => RHint::WitnessHeap { name, ptr, lo, len }, - }; - self.pending.push(Hint::Resolved(hint)); - } - /// A BLAKE2s chaining value must occupy two consecutive frame cells because - /// the opcode carries one base offset for both words. Preserve a genuine - /// consecutive pair, including a heap pair already bridged by - /// [`Self::blake2s_input`]. A `cv` written as a two-word LIST exposes two - /// sources that need not be adjacent, so those are copied into a fresh - /// consecutive pair. - fn blake2s_cv(&mut self, e: &Expr) -> Off { - let pair = self.blake2s_input(e); - if pair[1] == pair[0] + 1 { - return pair[0]; - } - let cv = self.alloc_stack(2); - self.copy(pair[0], cv); - self.copy(pair[1], cv + 1); - cv - } -} diff --git a/crates/lean_compiler/src/lower/call.rs b/crates/lean_compiler/src/lower/call.rs deleted file mode 100644 index 6ce9ff744..000000000 --- a/crates/lean_compiler/src/lower/call.rs +++ /dev/null @@ -1,629 +0,0 @@ -//! The call boundary: arguments in, return values out, and the two ways a -//! callee can disappear into its caller. -//! -//! Every frame is laid out by [`Abi`], and CALLER AND CALLEE MUST AGREE ON THE -//! ARITY: each places the return area from its own idea of the argument count, -//! so a missing argument leaves the callee's cell unwritten and therefore -//! prover-chosen, and a surplus one overwrites the callee's first return slot. -//! Two paths need the check, the ordinary one and the fused `match` -//! dispatch. -//! -//! A callee vanishes into its caller two ways. `Const` specialization -//! monomorphises it per constant tuple; `@inline` expands the body into the -//! caller's own frame, so the caller's `one`, `self_fp` and constant cells stay -//! valid across it and only the name bindings reset. - -use super::*; -use std::borrow::Cow; - -/// How an inlined tail return binds in the caller: a `StackBuf` run and a folded -/// g-address alias at zero copies, while anything else (a plain scalar, or a real -/// call, which records no [`RetBind`]) takes the destination cell it wrote. -pub(super) fn ret_binding(b: Option, dst: Off) -> Binding { - match b { - Some(RetBind::Stack(base, size)) => Binding::Stack(base, size), - Some(RetBind::Gaddr(ga)) => Binding::Gaddr(ga), - _ => Binding::Scalar(dst), - } -} - -fn substitute_body<'a>(body: &'a [Stmt], substs: &[(&str, Expr)]) -> Cow<'a, [Stmt]> { - substs.iter().fold(Cow::Borrowed(body), |body, (name, value)| { - Cow::Owned(subst_stmts(&body, name, value)) - }) -} - -/// A single tail return, preceded by statements that can be expanded in the caller's frame. -fn body_inlinable(body: &[Stmt]) -> bool { - matches!(body.split_last(), Some((last, rest)) if matches!(last.kind, StmtKind::Return(_)) - && rest.iter().all(stmt_inline_safe)) -} - -fn stmt_inline_safe(s: &Stmt) -> bool { - match &s.kind { - StmtKind::Let(..) - | StmtKind::Store(..) - | StmtKind::HintWitness { .. } - | StmtKind::LetHintWitness { .. } - | StmtKind::Print { .. } - | StmtKind::AssertEq(..) - | StmtKind::AssertNe(..) - | StmtKind::AssertLt(..) - // Ordinary calls allocate their own frames. - | StmtKind::Call(..) => true, - StmtKind::If { then, els, .. } => then.iter().all(stmt_inline_safe) && els.iter().all(stmt_inline_safe), - StmtKind::Unroll { body, .. } => body.iter().all(stmt_inline_safe), - _ => false, - } -} - -impl FnLower<'_> { - /// Lower a call into the caller's return cells. Distinct `match` arms may - /// share these write-once destinations. - pub(super) fn lower_call(&mut self, callee: &str, args: &[Expr], cond: Option, dsts: &[Off], tail: bool) { - // Every parameter must be supplied. A missing argument leaves the - // callee's argument cell unwritten, hence prover-chosen, so an `assert` - // reading it is vacuous; a surplus one lands on the callee's first - // return slot, because caller and callee place the return area from - // their own idea of the argument count. Only `specialize` checked this, - // and only for a callee declaring `Const` parameters. - match self.callee_def(callee).map(|f| f.params.len()) { - Some(want) if want != args.len() => { - let plural = if want == 1 { "argument" } else { "arguments" }; - self.fail(format!("`{callee}` takes {want} {plural}, got {}", args.len())) - } - // Nothing by that name is going to be lowered, so the entry pc it - // needs will not exist. Caught here, where there is a line: a typo, a - // statement-only builtin used as a value (`x = assert_in_k(a, b)`), - // or an `@inline` callee reached where inlining did not happen, all - // used to die later in `resolve` as a bare `no entry found for key`. - None => self.fail(format!( - "no function named `{callee}`. A builtin that writes into a destination \ - (`blake2s`, `assert_in_k`, a `hint_*`) is a statement and returns nothing, so it \ - cannot be called for a value" - )), - _ => {} - } - let (callee, args) = self.specialize(callee, args); - let (callee, args) = (callee.as_str(), args.as_slice()); - // Each argument goes where its SHAPE puts it: a `StackBuf(n)` parameter - // takes n consecutive cells, exactly as a `StackBuf(n)` return value - // does. Resolved before the frame pointer is allocated, as before. - let shapes = self - .callee_def(callee) - .map(|f| f.param_shapes().collect::>()) - .unwrap_or_else(|| vec![Shape::Scalar; args.len()]); - let mut arg_offs: Vec<(Off, Off)> = Vec::new(); - for (i, a) in args.iter().enumerate() { - let base = Abi::arg(shapes.iter().copied(), i); - match shapes.get(i).copied().unwrap_or(Shape::Scalar) { - Shape::StackBuf(n) => { - let (src, len) = self.stack_of(a).unwrap_or_else(|| { - self.fail(format!( - "`{callee}` parameter {i} is a StackBuf({n}); pass one, got `{a:?}`" - )) - }); - if len != n { - self.fail(format!( - "`{callee}` parameter {i} is a StackBuf({n}), got a StackBuf({len})" - )) - } - for k in 0..n { - let cell = src + k; - arg_offs.push((base + k, cell)); - } - } - Shape::Scalar => { - let cell = self.expr(a); - arg_offs.push((base, cell)); - } - } - } - let callee_arg_cells = Abi::arg_cells(shapes.iter().copied()); - if tail && callee == self.fn_name && self.loop_bounds.contains_key(callee) { - let own_fp = self.self_fp(); - let scale = self.fresh(); - self.set(scale, KVal::FrameSize); - self.pending.push(Hint::NextFrameAddress); - let nfp = self.pure(PureOp::Mul, own_fp, scale); - let entry = self.fresh(); - let oc = cond.unwrap_or_else(|| self.one()); - let one = self.one(); - self.set(entry, KVal::Entry(callee.to_string())); - for &(off, ao) in &arg_offs { - self.emit(LOp::MulNextFrame { a: ao, b: one, c: off }); - } - self.emit(LOp::MulNextFrame { - a: nfp, - b: one, - c: Abi::end(callee_arg_cells, 0), - }); - self.emit(LOp::MulNextFrame { - a: 1, - b: one, - c: Abi::RET_FP, - }); - self.emit(LOp::MulNextFrame { - a: 0, - b: one, - c: Abi::RET_PC, - }); - self.emit(LOp::Jump { oc, od: entry, of: nfp }); - return; - } - let nfp = self.fresh(); - let entry = self.fresh(); - // Resolve the jump condition up front: `self.one()` may emit a `SET`, and - // nothing may sit between the retpc `DEREF` and the `JUMP` (the `g²·pc` - // return target assumes the `JUMP` is exactly one instruction later). - let oc = cond.unwrap_or_else(|| self.one()); - self.set(entry, KVal::Entry(callee.to_string())); - - // The frame-pointer hint fires before the first DEREF that reads `nfp`. - if let Some(&(start, end)) = self.loop_bounds.get(callee) { - let count = match end { - Some(end) => LoopCount::Constant(end - start + 1), - None => LoopCount::Bound { - cell: arg_offs[1].1, - start, - }, - }; - self.pending.push(Hint::AllocLoopFrames { - ptr: nfp, - callee: callee.to_string(), - count, - }); - } else { - self.pending.push(Hint::AllocFrame { - ptr: nfp, - callee: callee.to_string(), - }); - } - for &(off, ao) in &arg_offs { - self.deref(nfp, off, ao, DerefMode::Cell); - } - if self.loop_bounds.contains_key(callee) { - self.deref(nfp, Abi::end(callee_arg_cells, 0), nfp, DerefMode::Cell); - } - if tail { - // Tail call: hand the callee OUR return target, so it returns to our - // caller and we are never resumed. Cells 0/1 of this frame already - // hold that target (written by whoever called us). - self.deref(nfp, 1, 1, DerefMode::Cell); // retfp := our retfp - self.deref(nfp, 0, 0, DerefMode::Cell); // retpc := our retpc - } else { - self.deref(nfp, 1, 0, DerefMode::Fp); // retfp - self.deref(nfp, 0, 0, DerefMode::Pc); // retpc = g²·pc - } - self.emit(LOp::Jump { oc, od: entry, of: nfp }); - - for (i, &d) in dsts.iter().enumerate() { - self.deref(nfp, Abi::ret(callee_arg_cells, i as u32), d, DerefMode::Cell); - } - } - - /// `names = match(log(x), …, lambda k: f(args, k))` fused: the arms all - /// call one of `callees` (specializations sharing the arg/return layout) with - /// the same runtime `args`, so build the callee frame **once** and let the - /// dispatch jump straight into the selected entry, which returns to the join. - /// Each taken arm is then just the trampoline's `SET entry; JUMP`: no - /// per-arm frame setup, call, or return jump. - pub(super) fn lower_dispatched_call(&mut self, targets: &[Expr], x: &Expr, callees: &[String], rt_args: &[&Expr]) { - // The arms share ONE frame, so they must share one argument layout too: - // a `StackBuf` parameter in one callee and a scalar in another at the - // same position would put the return area in two places. The arity check - // below is the count; this is the widths. - if let Some(shared) = callees.iter().find_map(|c| self.callee_def(c)) { - for c in callees { - if let Some(def) = self.callee_def(c) - && !def.param_shapes().eq(shared.param_shapes()) - { - self.fail(format!( - "`{c}` does not take the same parameter shapes as the other arms of this dispatch" - )) - } - } - // Fused dispatch writes one cell per argument; only scalar parameters fit. - if let Some(i) = shared.param_shapes().position(|s| s != Shape::Scalar) { - self.fail(format!( - "a `match` arm cannot pass a `StackBuf` parameter (parameter {i} of `{}`): the \ - fused dispatch writes one cell per argument. Give the arms `Const` arguments so each \ - specializes into its own call instead of fusing", - callees.first().map(String::as_str).unwrap_or("?") - )) - } - } - let n_args = rt_args.len() as u32; - // The join below reads one return cell per bound name, so every callee has - // to declare exactly that many. Unchecked, a name past a callee's arity - // `DEREF`s a frame offset nothing on that path writes, and since the shared - // frame is sized to the LARGEST callee the offset exists: the surplus name - // binds a prover-chosen word. The non-fused path enforces this - // ([`Self::call_into`]), so leaving it out here means one source is rejected - // by one lowering of `match` and silently miscompiled by the other. - for callee in callees { - // Arguments for the same reason as returns below: the shared frame - // is sized to the largest callee, so a callee expecting more than - // the arms supply reads a cell that exists and nothing writes. - let Some(def) = self.callee_def(callee) else { - continue; - }; - let want = def.params.len(); - if want != rt_args.len() { - let plural = if want == 1 { "argument" } else { "arguments" }; - self.fail(format!( - "`{callee}` takes {want} {plural}, dispatched call passes {}", - rt_args.len() - )) - } - let shapes = &def.return_shapes; - if shapes.len() != targets.len() { - self.fail(format!( - "`{callee}` returns {} values, dispatched call binds {}", - shapes.len(), - targets.len() - )) - }; - if shapes.iter().any(|s| *s != Shape::Scalar) { - self.fail(format!( - "`{callee}`: a multi-cell StackBuf return cannot cross a dispatched join" - )) - }; - } - let (rcells, binds) = self.ret_targets(targets); - - // Shared callee frame: args, retfp, and retpc = the join (so the callee - // returns straight past the dispatch). Evaluated once. - let arg_offs: Vec = rt_args.iter().map(|a| self.expr(a)).collect(); - let xo = self.expr(x); - let one = self.one(); - let sfp = self.self_fp(); - - let nfp = self.fresh(); - self.pending.push(Hint::AllocFrameMax { - ptr: nfp, - callees: callees.to_vec(), - }); - for (i, &ao) in arg_offs.iter().enumerate() { - self.deref( - nfp, - Abi::arg(std::iter::repeat_n(Shape::Scalar, rt_args.len()), i), - ao, - DerefMode::Cell, - ); - } - self.deref(nfp, Abi::RET_FP, 0, DerefMode::Fp); - let join_cell = self.fresh(); - let join_set = self.code.len(); - self.set(join_cell, KVal::Local(0)); // patched: the join pc - self.deref(nfp, Abi::RET_PC, join_cell, DerefMode::Cell); // retpc = join - - let kset = self.emit_dispatch(xo, one, sfp); - - // Trampoline: slot j enters `callees[j]` with fp = nfp; the callee's own - // `return` jumps to retpc (the join) in the caller frame. - self.patch_local(kset, self.code.len()); - self.emit_slots(callees.len(), one, nfp, |j| KVal::Entry(callees[j].clone())); - - // Join: read the return values (written by whichever callee ran). - self.patch_local(join_set, self.code.len()); - for (i, &r) in rcells.iter().enumerate() { - self.deref(nfp, Abi::ret(n_args, i as u32), r, DerefMode::Cell); - } - - self.bind_targets(&binds); - } - - /// Inline an `@inline` `callee(args)` into the current frame, binding its - /// return values straight into `dsts`: no frame setup, no argument/return - /// plumbing, no call/return jumps. Returns `false` for a non-`@inline` - /// callee (the caller emits a real call). Panics if an `@inline` function - /// isn't inlinable ([`body_inlinable`]) or its `Const` args don't resolve. - pub(super) fn try_inline(&mut self, callee: &str, args: &[Expr], dsts: &[Off]) -> bool { - let Some(def) = self.defs.get(callee).copied().filter(|d| d.inline) else { - return false; - }; - let substs = self - .const_substs(def, args) - .unwrap_or_else(|_| self.fail(format!("`@inline {callee}`: bad arity or unresolved Const argument"))); - let body = substitute_body(&def.body, &substs); - let n_ret = def.return_shapes.len(); - if n_ret != dsts.len() { - self.fail(format!( - "`@inline {callee}` returns {n_ret} values, call binds {}", - dsts.len() - )) - }; - if !body_inlinable(&body) { - self.fail(format!("`@inline {callee}` requires one tail `return`, with no nested returns, tuple assignments, mul_range loops, or match")) - }; - if self.inline_calls.iter().any(|f| f == callee) { - self.fail(format!( - "recursive @inline expansion is not supported: {} -> {callee}", - self.inline_calls.join(" -> ") - )) - }; - // Bind the params from the caller-scope arguments (symbolically where we - // can, so a shifted-pointer arg keeps folding into `β`; a `StackBuf` arg - // aliases its cell run), then lower the body in a fresh variable - // environment, since a function sees only its params. The frame, `one`, - // `self_fp`, and range-check bounds stay the caller's: the inlined code - // runs in the caller's frame, so they fit. - let mut binds: Vec<(String, Binding)> = Vec::new(); - for (p, a) in def.params.iter().zip(args) { - if p.kind == ParamKind::Const { - continue; - } - let b = if let Some((base, size)) = self.stack_of(a) { - Binding::Stack(base, size) - } else if let Some(ga) = self.gaddr_of(a) { - Binding::Gaddr(ga) - } else { - Binding::Scalar(self.expr(a)) - }; - binds.push((p.name.clone(), b)); - } - // Only the name bindings reset: the inlined body runs in the caller's - // frame, so the caller's `one`, `self_fp`, constant and bound cells all - // still name valid cells and stay live. - let saved = std::mem::take(&mut self.scope.names); - for (p, b) in binds { - self.check_not_reserved(&p); - self.scope.names.insert(p, Bound { val: b, int: None }); - } - let saved_ret = self.inline_ret.replace(dsts.to_vec()); - // The body lowers through the CALLER's `FnLower`, so its statements move - // `cur_line` into the callee. Restoring it is what keeps the rest of the - // caller's expression attributed to the call site rather than to whatever - // line the callee happened to end on. - let saved_line = self.cur_line; - self.inline_calls.push(callee.to_string()); - for s in body.iter() { - self.stmt(s); - } - let popped = self.inline_calls.pop(); - debug_assert_eq!(popped.as_deref(), Some(callee)); - self.inline_ret = saved_ret; - self.cur_line = saved_line; - self.scope.names = saved; - true - } - - pub(super) fn lower_return(&mut self, exprs: &[Expr]) { - // Inlined (`@inline`): bind the return values into the caller's cells - // and fall through: this is the body's tail return, so no jump is needed. - if let Some(dsts) = self.inline_ret.take() { - // Each returned value is bound into the caller independently, exactly - // as a `let name = ` would: a `StackBuf` or a folded - // g-address hands over its run/pointer (alias, not copies: allocated - // in the caller's frame, so it outlives the inline scope), a scalar is - // copied into its dst cell. The per-slot record lets the caller's - // `let`/tuple pick the right binding, so a fused - // `fs, x, cur = fs_next(fs, cur)` returns a StackBuf, a scalar, and an - // advanced cursor together. - let mut binds = Vec::with_capacity(dsts.len()); - for (e, &d) in exprs.iter().zip(&dsts) { - binds.push(if let Some((base, size)) = self.stack_of(e) { - RetBind::Stack(base, size) - } else if let Some(ga) = self.gaddr_of(e) { - RetBind::Gaddr(ga) - } else { - self.expr_into(e, d); - RetBind::Scalar - }); - } - self.inline_stack_ret = Some(binds); - return; - } - if self.is_main { - return; // a `return` in main is a no-op; main halts via the trailing sentinel jump (lower_func). - } - let ret_base = Abi::ret(self.arg_cells, 0); - if exprs.len() != self.return_shapes.len() { - self.fail(format!( - "function returns {} values here, but its ABI declares {}", - exprs.len(), - self.return_shapes.len() - )) - }; - // Each logical value lands straight in its flattened return area. A - // StackBuf is copied cell-by-cell because its callee-frame offsets are - // not meaningful after control returns to the caller. - let mut ret = ret_base; - for (e, &shape) in exprs.iter().zip(self.return_shapes) { - match shape { - Shape::Scalar => self.expr_into(e, ret), - Shape::StackBuf(size) => { - let (base, actual) = self - .stack_of(e) - .unwrap_or_else(|| self.fail(format!("expected a StackBuf({size}) return, got `{e:?}`"))); - if actual != size { - self.fail(format!("returned StackBuf has size {actual}, expected {size}")) - }; - for k in 0..size { - let src = base + k; - self.copy(src, ret + k); - } - } - } - ret += shape.cells(); - } - let one = self.one(); - self.emit(LOp::Jump { oc: one, od: 0, of: 1 }); - } - - /// If `callee` declares `Const` parameters, monomorphize: the constant - /// arguments (literals, `GEN ** k`, or literal-bound names) substitute into a - /// copy of the callee, queued once per distinct constant tuple and named - /// `callee__L5_G3`-style, and only the runtime arguments remain. - pub(super) fn specialize<'a>(&mut self, callee: &str, args: &'a [Expr]) -> (String, Vec<&'a Expr>) { - let Some(def) = self.defs.get(callee).copied() else { - return (callee.to_string(), args.iter().collect()); // loop helpers, unknown names - }; - if !def.has_const_params() { - return (callee.to_string(), args.iter().collect()); - } - let substs = self.const_substs(def, args).unwrap_or_else(|e| self.fail(e)); - let mut name = format!("{callee}_"); - for (_, c) in &substs { - match c { - Expr::Lit(n) => write!(name, "_L{n}"), - Expr::GPow(k) => write!(name, "_G{k}"), - _ => unreachable!(), - } - .unwrap(); - } - let runtime = def.params.iter().zip(args).filter(|(p, _)| p.kind != ParamKind::Const); - if !self.queue.iter().any(|f| f.name == name) { - if self.queue.len() >= 10_000 { - self.fail("Const specialization explosion (recursive constants?)") - }; - self.queue.push(Func { - name: name.clone(), - params: runtime.clone().map(|(p, _)| p.clone()).collect(), - return_shapes: def.return_shapes.clone(), - body: substitute_body(&def.body, &substs).into_owned(), - inline: false, - }); - } - (name, runtime.map(|(_, a)| a).collect()) - } - - /// Lower a call; returns one caller offset per source-level return value. - /// A real-call StackBuf return is flattened into consecutive ABI cells and - /// copied into a fresh consecutive run in the caller. `inline_stack_ret` - /// describes those logical bindings to the surrounding let/tuple lowering. - pub(super) fn call(&mut self, callee: &str, args: &[Expr], n_ret: usize) -> Vec { - if callee == "blake2s" { - self.fail("blake2s is a statement: `blake2s(a, b, out)` writes the digest into the 2-cell stack run `out`") - }; - self.inline_stack_ret = None; - if self.defs.get(callee).is_some_and(|d| d.inline) { - let dsts: Vec = (0..n_ret).map(|_| self.fresh()).collect(); - self.call_into(callee, args, &dsts); - return dsts; - } - - let def = self.defs.get(callee).copied(); - if let Some(def) = def - && def.return_shapes.len() != n_ret - { - self.fail(format!( - "`{callee}` returns {} values, call binds {n_ret}", - def.return_shapes.len() - )) - }; - let mut logical = Vec::with_capacity(n_ret); - let mut physical = Vec::new(); - let mut binds = Vec::with_capacity(n_ret); - for i in 0..n_ret { - match def.map_or(Shape::Scalar, |d| d.return_shapes[i]) { - Shape::Scalar => { - let dst = self.fresh(); - logical.push(dst); - physical.push(dst); - binds.push(RetBind::Scalar); - } - Shape::StackBuf(size) => { - if size == 0 { - self.fail("a returned StackBuf must not be empty") - }; - let base = self.alloc_stack(size); - logical.push(base); - physical.extend(base..base + size); - binds.push(RetBind::Stack(base, size)); - } - } - } - self.lower_call(callee, args, None, &physical, false); - self.inline_stack_ret = Some(binds); - logical - } - - /// Evaluate `callee(args)` into `dsts`, inlining the callee when it is - /// `@inline` ([`Self::try_inline`]), else a real call. - pub(super) fn call_into(&mut self, callee: &str, args: &[Expr], dsts: &[Off]) { - if callee == "blake2s" { - self.fail("blake2s is a statement, not a value-returning call") - }; - if !self.try_inline(callee, args, dsts) { - if let Some(def) = self.defs.get(callee) { - if def.return_shapes.len() != dsts.len() { - self.fail(format!( - "`{callee}` returns {} values, call binds {}", - def.return_shapes.len(), - dsts.len() - )) - }; - if def.return_shapes.iter().any(|s| *s != Shape::Scalar) { - self.fail("a normal function's multi-cell StackBuf return needs a `let` binding") - }; - } - self.lower_call(callee, args, None, dsts, false); - } - } - - fn const_substs<'a>(&self, def: &'a Func, args: &[Expr]) -> Result, String> { - if args.len() != def.params.len() { - return Err(format!("call to `{}`: wrong arity", def.name)); - } - let mut substs = Vec::new(); - for (p, a) in def.params.iter().zip(args) { - if p.kind == ParamKind::Const { - let c = self.const_arg(a).ok_or_else(|| { - format!( - "argument for Const parameter `{}` of `{}` must be a compile-time \ - constant, got `{a:?}`", - p.name, def.name - ) - })?; - substs.push((p.name.as_str(), c)); - } - } - Ok(substs) - } - - /// Original definitions take precedence over generated functions in the queue. - fn callee_def(&self, callee: &str) -> Option<&Func> { - self.defs - .get(callee) - .copied() - .or_else(|| self.queue.iter().find(|f| f.name == callee)) - } - - /// Consume the [`RetBind`] a single-value inlined tail return recorded, - /// for a call in EXPRESSION position (embedded in arithmetic, a store - /// RHS, a single-target match arm): there is no name to alias-bind, so an - /// aliased return materializes into a plain cell (free for a var / an - /// exp-0 g-address; one `MUL` for a shifted pointer). `dst` is the call's - /// destination cell, already written by a real call or a plain-scalar - /// return, so it is the fallback. - pub(super) fn take_inline_ret_cell(&mut self, dst: Off) -> Off { - match self.inline_stack_ret.take().and_then(|b| b.into_iter().next()) { - Some(RetBind::Gaddr(ga)) => self.materialize(ga), - Some(RetBind::Stack(base, size)) => { - if size != 1 { - self.fail("a multi-cell StackBuf return needs a `let` binding, not an expression use") - }; - // The run's first cell: a single returned value sits there, and a - // `let` consumer reaches the same one by binding the run - // ([`ret_binding`]). - base - } - _ => dst, - } - } - /// The value a `Const` parameter takes, as the literal that substitutes for - /// it: a `GEN ** k` argument stays a g-power, everything else must fold to a - /// compile-time integer (a bound name, a const-array element `DEPTH[lvl]`, - /// `len(...)`, index arithmetic over other `Const` params). `None` when it - /// does not fold. - fn const_arg(&self, a: &Expr) -> Option { - Some(match a { - Expr::Lit(n) => Expr::Lit(*n), - Expr::Gen => Expr::GPow(1), - Expr::GPow(k) => Expr::GPow(*k), - other => Expr::Lit(self.try_const_index(other)? as u128), - }) - } -} diff --git a/crates/lean_compiler/src/lower/eval.rs b/crates/lean_compiler/src/lower/eval.rs deleted file mode 100644 index bd9cd6222..000000000 --- a/crates/lean_compiler/src/lower/eval.rs +++ /dev/null @@ -1,395 +0,0 @@ -//! Compile-time evaluation: what an expression is worth before anything runs. -//! -//! Queries emit no instructions and leave compiler state unchanged. -//! -//! There are two answers and the POSITION of a use picks one: -//! [`FnLower::try_const_int`] for a size, an index, a bound or an exponent, -//! [`FnLower::try_field_const`] for a value, where `+` is XOR. They disagree on -//! a sum of overlapping integers, and `const(...)` is how an author says which -//! was meant. - -use super::*; - -/// The readings of one expression: as many as its shape has. Produced by -/// [`FnLower::eval`], which is the only walk that computes them. -#[derive(Clone, Copy, Default)] -pub(super) struct Known { - /// The compile-time INTEGER, wanted by a size, an index, a bound, an exponent. - pub(super) int: Option, - /// The FIELD element a value position sees, where `+` is XOR. - pub(super) field: Option, - /// The ADDRESS the compiler tracks: a base cell times `g^exp`. - pub(super) addr: Option, -} - -impl Known { - /// The integer and field readings when both exist and disagree. - pub(super) fn diverging_readings(self) -> Option<(u128, F192)> { - let (n, f) = (self.int?, self.field?); - (f != lit_field(n)).then_some((n, f)) - } -} - -/// `a·b` in the [`GAddr`] representation: exponents add, and at most one factor -/// may carry a runtime base (two pointers can't be multiplied symbolically). -fn gmul(a: GAddr, b: GAddr) -> Option { - let base = match (a.base, b.base) { - (None, x) | (x, None) => x, - (Some(_), Some(_)) => return None, - }; - Some(GAddr { - base, - exp: a.exp.checked_add(b.exp)?, - // Only a based address carries a run, and at most one side is based, so - // the shift keeps its origin: `addr(sb) * GEN ** k` stays bounded by `sb`. - run: a.run.or(b.run), - }) -} - -/// `b^k` by square-and-multiply, in logarithmically many field operations. -pub(super) fn field_pow(b: F192, mut k: u32) -> F192 { - let (mut acc, mut sq) = (F192::ONE, b); - while k > 0 { - if k & 1 == 1 { - acc *= sq; - } - k >>= 1; - if k > 0 { - sq *= sq; - } - } - acc -} - -impl FnLower<'_> { - /// Everything `e` is worth before anything runs, from ONE walk. - /// - /// An expression genuinely has more than one reading, and which is wanted - /// depends on the POSITION of the use: `x = 2` names the integer 2, the field - /// element 2, and the address `g^1`, all three at once. So the evaluator - /// computes every reading a shape has and the caller takes the one its - /// position means. - /// - /// Keeping the readings together lets each use reject an ambiguous value - /// by comparing them, without evaluating the expression again. - pub(super) fn eval(&self, e: &Expr) -> Known { - // Deliberately NO address: only a LITERAL reads as one, and only under the - // guard below. Attaching it here gave `const(2^k)` and `len(A)` an address - // that `Expr::Lit` alone used to have, which slipped them past - // `heap_addr`'s ambiguity guard: `buf[const(8)]` on a `HeapBuf(4)` - // compiled and aliased cell 3, while the bare `buf[8]` it means was - // rejected. One spelling naming two different cells is exactly what that - // guard exists to stop. - let int = |n: u128| Known { - int: Some(n), - field: Some(lit_field(n)), - addr: None, - }; - let gpow = |exp: u128| Known { - int: None, - field: Some(g_pow_u128(exp).into()), - addr: Some(GAddr { - base: None, - exp, - run: None, - }), - }; - match e { - // `g = x`, so the literal `2^k` IS `g^k`, but ONLY while `k < 64`: at - // and above it the modulus folds the monomial back into the low limb - // while the literal's bit `k` lands in the next limb, the tower - // coefficient of `y`. Without the guard the guest's own `Y_TOWER = - // 2^64` would read as `g^64` in a pointer position. - Expr::Lit(n) => Known { - addr: (n.is_power_of_two() && *n < (1 << 64)).then(|| GAddr { - base: None, - exp: n.trailing_zeros() as u128, - run: None, - }), - ..int(*n) - }, - Expr::Gen => gpow(1), - Expr::GPow(k) => gpow(*k), - Expr::GenPow(x) => match self.eval(x).int.and_then(|n| u32::try_from(n).ok()) { - Some(k) => gpow(u128::from(k)), - None => Known::default(), - }, - Expr::Var(v) => { - let Some(b) = self.scope.bound(v) else { - return Known::default(); - }; - let int = b.int; - match b.val { - Binding::FConst(c) => Known { - int, - field: Some(c), - addr: None, - }, - Binding::Gaddr(ga) => Known { - int, - // A constant g-power also reads as that field element. - field: (ga.base.is_none()).then(|| g_pow_u128(ga.exp).into()), - addr: Some(ga), - }, - // A plain scalar is its own base, unshifted. - Binding::Scalar(c) => Known { - int, - field: None, - addr: Some(GAddr { - base: Some(c), - exp: 0, - run: None, - }), - }, - Binding::Stack(..) => Known { - int, - ..Known::default() - }, - } - } - // Each operand is evaluated ONCE: a reading per arm would re-walk the - // subtree, which is exponential in the nesting depth. - // - // `+` is integer addition in an index and XOR in a value, so it has - // both; `-`, `//` and `%` have no field meaning, so only the integer. - Expr::Add(a, b) | Expr::Sub(a, b) | Expr::Mul(a, b) | Expr::Div(a, b) | Expr::Mod(a, b) => { - let (x, y) = (self.eval(a), self.eval(b)); - if matches!(e, Expr::Div(..) | Expr::Mod(..)) && y.int == Some(0) { - self.fail(match e { - Expr::Div(..) => "compile-time division by zero", - _ => "compile-time modulo by zero", - }) - }; - let int = || { - let (n, m) = (x.int?, y.int?); - match e { - Expr::Add(..) => n.checked_add(m), - Expr::Sub(..) => n.checked_sub(m), - Expr::Mul(..) => n.checked_mul(m), - Expr::Div(..) => Some(n / m), - _ => Some(n % m), - } - }; - Known { - int: int(), - field: match e { - Expr::Add(..) => x.field.and_then(|f| Some(f + y.field?)), - Expr::Mul(..) => x.field.and_then(|f| Some(f * y.field?)), - _ => None, - }, - addr: match e { - Expr::Mul(..) => x.addr.and_then(|p| gmul(p, y.addr?)), - _ => None, - }, - } - } - Expr::Pow(b, x) => { - let (base, exp) = (self.eval(b), self.eval(x).int.and_then(|n| u32::try_from(n).ok())); - Known { - int: base.int.and_then(|n| n.checked_pow(exp?)), - field: base.field.and_then(|f| Some(field_pow(f, exp?))), - addr: None, - } - } - // A constant-array element, as a field value or as the integer those - // bits spell. - Expr::Index(..) => match self.const_array_elem(e) { - Some(v) => Known { - int: (v.c2 == 0).then_some(v.c0 as u128 | ((v.c1 as u128) << 64)), - field: Some(v), - addr: None, - }, - None => Known::default(), - }, - Expr::Call(f, args) if f == "f192" && args.len() == 3 => { - let limb = |i: usize| match &args[i] { - Expr::Lit(n) => u64::try_from(*n).ok(), - _ => None, - }; - Known { - field: (|| Some(F192::new(limb(0)?, limb(1)?, limb(2)?)))(), - ..Known::default() - } - } - // `const(e)`: the one construct that asks for the INTEGER reading in a - // position that would otherwise take the field one. It reinterprets the - // OPERATORS, so its leaves must mean the same thing either way. - Expr::Call(f, args) if f == "const" && args.len() == 1 => { - self.check_const_leaves(&args[0]); - match self.eval(&args[0]).int { - Some(n) => int(n), - None => Known::default(), - } - } - Expr::Call(..) => match self.const_len(e) { - Some(n) => int(n as u128), - None => Known::default(), - }, - _ => Known::default(), - } - } - - /// A compile-time integer. `None` means a runtime value, or arithmetic outside - /// the source language's `u128` literal domain. - pub(super) fn try_const_int(&self, e: &Expr) -> Option { - self.eval(e).int - } - - /// The address the compiler tracks for `e`: a base cell times `g^exp`. - pub(super) fn gaddr_of(&self, e: &Expr) -> Option { - self.eval(e).addr - } - - /// `e` as a compile-time FIELD constant, where `+` is XOR. `None` for a - /// runtime value or for arithmetic the field has no meaning for (`-`, `//`, - /// `%`). - pub(super) fn try_field_const(&self, e: &Expr) -> Option { - self.eval(e).field - } - - /// A compile-time integer index. The general integer evaluator is narrowed - /// here so stack offsets, bounds, and immediate exponents remain `u32`. - pub(super) fn try_const_index(&self, idx: &Expr) -> Option { - u32::try_from(self.try_const_int(idx)?).ok() - } - - /// A stack index or compile-time slice bound: [`Self::try_const_index`], - /// required to succeed. - pub(super) fn const_index(&self, idx: &Expr) -> u32 { - let k = self.try_const_int(idx).unwrap_or_else(|| { - self.fail(format!( - "a StackBuf index must be a compile-time integer, got `{idx:?}`" - )) - }); - u32::try_from(k).unwrap_or_else(|_| self.fail(format!("stack index {k} does not fit in u32"))) - } - - /// The exponent of `GEN ** e`: a compile-time integer, required to succeed. - pub(super) fn gpow_exp(&self, e: &Expr) -> u128 { - self.try_const_index(e) - .unwrap_or_else(|| self.fail(format!("`GEN ** e` needs a compile-time integer exponent, got `{e:?}`"))) - as u128 - } - - /// If `e` is `NAME[i]` for a top-level constant array `NAME` with a - /// compile-time index `i`, its element (a raw `u128`). - fn const_array_elem(&self, e: &Expr) -> Option { - if let Expr::Index(arr, idx) = e - && let Expr::Var(v) = arr.as_ref() - && let Some(a) = self.const_arrays.get(v.as_str()) - { - let i = self.try_const_index(idx)? as usize; - return Some( - *a.get(i).unwrap_or_else(|| { - self.fail(format!("const array `{v}` index {i} out of bounds (len {})", a.len())) - }), - ); - } - None - } - - /// If `e` is `len(NAME)` for a top-level constant array `NAME`, its length. - fn const_len(&self, e: &Expr) -> Option { - if let Expr::Call(f, args) = e - && f == "len" - && args.len() == 1 - && let Expr::Var(v) = &args[0] - { - return self.const_arrays.get(v.as_str()).map(|a| a.len()); - } - None - } - - /// The surviving operand of `a + b` when the other is a compile-time zero, - /// which contributes nothing and (being a constant) has no side effect to - /// preserve. So `x + 0` lowers to just `x`: no cell, no XOR. Kills the - /// `acc = 0; acc = acc + t` accumulator seed and similar. - pub(super) fn add_identity<'e>(&self, a: &'e Expr, b: &'e Expr) -> Option<&'e Expr> { - if self.try_field_const(a) == Some(F192::ZERO) { - return Some(b); - } - (self.try_field_const(b) == Some(F192::ZERO)).then_some(a) - } - - /// The surviving operand of `a * b` when the other is a compile-time one, a - /// no-op multiply. Kills the `acc = GEN ** 0` (= 1) accumulator seed's first - /// `1 * f` in every product loop. - pub(super) fn mul_identity<'e>(&self, a: &'e Expr, b: &'e Expr) -> Option<&'e Expr> { - if self.try_field_const(a) == Some(F192::ONE) { - return Some(b); - } - (self.try_field_const(b) == Some(F192::ONE)).then_some(a) - } - - /// The field value of `e` when it is a trivial compile-time constant (a - /// literal, a literal-bound name, or `GEN ** 0`), for the `x*1`/`x+0` - /// arithmetic identities and the `== 0` test of [`Self::lower_if`]. - pub(super) fn try_lit(&self, e: &Expr) -> Option { - match e { - Expr::Lit(n) => u64::try_from(*n).ok(), - Expr::Var(v) => self.scope.int(v).and_then(|n| u64::try_from(n).ok()), - Expr::GPow(0) => Some(1), - _ => None, - } - } - - /// Check the LEAVES of a `const(...)`. The wrapper reinterprets the - /// OPERATORS as integer arithmetic, which is its whole purpose, so their two - /// readings are expected to diverge (`3 + 1` is the integer 4 and the value - /// 2). A LEAF is different: `const(...)` cannot change what a name already - /// stands for, so a leaf whose own two readings disagree would have the - /// wrapper hand back a value that leaf never had. - /// - /// `n = 2 + 3` is the case. The cell holds `2 XOR 3` = 1 while the name's - /// integer reading is 5, so `assert n == 1` and `assert const(n) == 5` both - /// passed, in one program. This is the ambiguity `if const(...)` already - /// rejects per side, one level up, and the rule is the same one. - fn check_const_leaves(&self, e: &Expr) { - match e { - Expr::Add(a, b) - | Expr::Sub(a, b) - | Expr::Mul(a, b) - | Expr::Div(a, b) - | Expr::Mod(a, b) - | Expr::Pow(a, b) => { - self.check_const_leaves(a); - self.check_const_leaves(b); - } - Expr::Call(f, args) if f == "const" && args.len() == 1 => self.check_const_leaves(&args[0]), - leaf => { - if let Some((n, f)) = self.eval(leaf).diverging_readings() { - self.fail(format!( - "const(...) reads its operators as integer arithmetic, but it cannot reinterpret \ - `{leaf:?}`, which is the integer {n} and the value {:#x}:{:#x}: two different \ - numbers. Bind it in one regime and name that one", - f.c1, f.c0 - )) - } - } - } - } - - /// The exponent of `e` when it is a *constant* g-power small enough to ride a - /// `DEREF` `β` immediate, for the constant factor of a product index. - /// - /// Folds `e` only where the readings agree: either the compiler tracks it as an - /// address, or `e` is the integer `2^j` AND its field value is `g^j`, in - /// which case both readings name cell `j` and folding decides nothing. - pub(super) fn const_gpow(&self, e: &Expr) -> Option { - let k = self.eval(e); - if let Some(GAddr { base: None, exp, .. }) = k.addr - && exp <= FOLD_MAX - { - return Some(exp as u32); - } - // `g = x`, so the integer `2^j` reads as `g^j`, but ONLY for `j < 64`: - // at and above that the modulus folds the monomial back into the low - // limb while the literal's bit lands in the tower coefficient of `y`. - let n = k.int?; - if !n.is_power_of_two() || n >= (1 << 64) { - return None; - } - let j = n.trailing_zeros(); - (u128::from(j) <= FOLD_MAX && k.field? == g_pow_u128(u128::from(j)).into()).then_some(j) - } -} diff --git a/crates/lean_compiler/src/lower/mem.rs b/crates/lean_compiler/src/lower/mem.rs deleted file mode 100644 index a2f8875f3..000000000 --- a/crates/lean_compiler/src/lower/mem.rs +++ /dev/null @@ -1,294 +0,0 @@ -//! Addressing, and the store path: which cell a name means, and what a write to -//! it costs. Two rules, both of which have been broken here before. -//! -//! **Every index is bounds-checked, in every position.** A store's right-hand -//! side is an index as much as an expression is, and a slice checks its whole -//! SPAN rather than its first cell. -//! -//! **A store always emits, and the machine decides what it means.** If the cell -//! already holds a value the store is the write-once equality ASSERTION, which is -//! what makes `s[k] = ` pin a hint; if it does not, the store is -//! what gives the cell its value. The compiler tracks nothing to tell those -//! apart, so there is no state here that could disagree with the machine. - -use super::*; - -impl FnLower<'_> { - /// Resolve an expression naming a run of consecutive cells: a whole - /// `StackBuf`, a `StackBuf` slice, a `HeapBuf` slice with compile-time - /// bounds, or a runtime-start heap slice `buf[i:i + k]` (whose length is - /// the only thing its bounds reveal). Heap runs fold the buffer's symbolic - /// shift and the slice start into the pointer offset. - pub(super) fn cell_run(&mut self, e: &Expr) -> CellRun { - match e { - Expr::Var(_) => { - let (base, len) = self.stack_of(e).unwrap_or_else(|| { - self.fail(format!( - "only a StackBuf names a run of cells unsliced, got `{e:?}`; slice a \ - HeapBuf instead: `buf[lo:lo + k]`" - )) - }); - CellRun::Stack { base, len } - } - Expr::Slice(arr, lo, hi) => match (self.try_const_index(lo), self.try_const_index(hi)) { - // Compile-time bounds: integer cell indexes `lo..hi` (frame - // offsets for a stack, g-power exponents for the heap). - (Some(lo), Some(hi)) => { - if lo >= hi { - self.fail(format!("empty slice {lo}:{hi}")) - }; - let len = hi - lo; - if let Some((base, size)) = self.stack_of(arr) { - if hi > size { - self.fail(format!("slice {lo}:{hi} out of bounds (StackBuf size {size})")) - }; - CellRun::Stack { base: base + lo, len } - } else { - self.check_heap_bound(arr, lo as u128, len as u128); - let (ptr, lo) = self.heap_base(arr, lo as u128); - CellRun::Heap { ptr, lo, len } - } - } - // Runtime start (heap only): `buf[i:i + k]` with a runtime - // g-power index `i` names the cells `buf·i·g^j`, j < k. The - // `hi` bound cannot be evaluated, only shape-checked against - // `lo`. One MUL folds `i` into the pointer. - _ => { - if self.stack_of(arr).is_some() { - self.fail( - "a StackBuf slice needs compile-time bounds (frame offsets are baked into the bytecode)", - ) - }; - let k = plus_k(lo, hi).unwrap_or_else(|| { - self.fail(format!("a runtime slice must be `buf[i:i + k]`, got `{lo:?}:{hi:?}`")) - }); - let len = - u32::try_from(k).unwrap_or_else(|_| self.fail(format!("slice length {k} does not fit in u32"))); - if len == 0 { - self.fail(format!( - "a runtime slice `{lo:?}:{hi:?}` has length 0, so it names no cell" - )) - }; - // `heap_addr` bounds-checks ONE cell. A start that folds - // (`GEN ** k`, or a name bound to one) reaches this arm because - // it is not an INTEGER, yet its offset IS known, so the run's - // length has to be checked here or it never is. - if let Some(GAddr { base: None, exp, .. }) = self.gaddr_of(lo) { - self.check_heap_bound(arr, exp, u128::from(len)); - } - let (ptr, lo) = self.heap_addr(arr, lo); - CellRun::Heap { ptr, lo, len } - } - }, - other => self.fail(format!( - "expected a StackBuf, a StackBuf slice, or a HeapBuf slice, got `{other:?}`" - )), - } - } - - /// Address `arr[idx]` as `(base_cell, β)`. A constant g-power `idx` folds - /// into `β` ([`Self::heap_base`]); a runtime index materializes the pointer. - pub(super) fn heap_addr(&mut self, arr: &Expr, idx: &Expr) -> (Off, u32) { - // A compile-time index that is a plain field constant but NOT a - // g-power (`buf[0]`, `buf[2]`, an integer unroll var) can never name - // a heap cell (cell k lives at `buf · g^k`) and would deref a wild - // address at proving time. Reject it here, where the source is known. - // A BARE literal index is rejected even now that `gaddr_of` reads `2^k` - // as `g^k` in a pointer: `hb[2]` would silently mean cell 1, while slice - // bounds stayed integer (`hb[2:4]` starts at cell 2), so one spelling - // would name two different cells. An index built from `GEN` is fine, and - // `1` is `g^0` either way. - let bare_int = matches!(self.try_lit(idx), Some(n) if n != 1); - let known = self.eval(idx); - if (bare_int || known.addr.is_none()) - && let Some(c) = known.field - { - // Two reasons reach here and they read differently. A bare integer - // index may well BE a g-power (4 is g²), and is rejected for being - // ambiguous against slice syntax rather than for naming nothing. - let why = match self.const_gpow(idx) { - Some(k) => format!( - "is a plain integer naming cell {k}, while the slice `buf[n:n + 1]` reads the \ - same number as cell n. Write `buf[GEN ** {k}]` and say which you mean" - ), - None => format!( - "folds to the field constant {:#x}:{:#x}, which is not a g-power, so it names \ - no heap cell (did an integer index leak in from a StackBuf conversion?)", - c.c1, c.c0 - ), - }; - self.fail(format!("heap index {why}")); - } - match known.addr { - Some(GAddr { base: None, exp, .. }) => return self.heap_base(arr, exp), - // A runtime-base index carrying a constant g-power shift - // (`buf[cursor * GEN ** k]`): fold the whole constant part (the - // index's shift plus `arr`'s own symbolic shift) into `β`, and - // emit ONE pointer multiply instead of materializing g^k. - Some(GAddr { - base: Some(ib), exp, .. - }) => { - if let Some(ga) = self.gaddr_of(arr) - && let (Some(ab), Some(total)) = (ga.base, ga.exp.checked_add(exp)) - && total <= FOLD_MAX - { - let ptr = self.pure(PureOp::Mul, ab, ib); - return (ptr, total as u32); - } - } - None => {} - } - // Fall back to the constant-g-power-factor fold (a runtime index still - // materializes the pointer `MUL`, with any constant factor in `β`). - self.array_ptr(arr, idx) - } - - /// Compile-time bounds check: when `arr` resolves to a sized `HeapBuf` - /// (directly or through shifted aliases) and the whole index is the - /// compile-time exponent `exp`, reject `exp + span > size`. Runtime - /// indices are not checked (their value is not known here). - pub(super) fn check_heap_bound(&self, arr: &Expr, extra: u128, span: u128) { - let Some(ga) = self.gaddr_of(arr) else { return }; - let (Some(base), Some(exp)) = (ga.base, ga.exp.checked_add(extra)) else { - return; - }; - // A frame pointer from `addr(sb)`: `exp` is an absolute frame offset and - // `base` is the shared `fp` cell, so the run comes from the address's own - // provenance rather than from `heap_sizes`. - if let Some((start, len)) = ga.run { - let (start, len) = (start as u128, len as u128); - if exp < start || exp + span > start + len { - let off = exp.saturating_sub(start); // `exp < start` is unreachable: gmul only adds - let what = if span == 1 { - format!("index {off}") - } else { - format!("slice {off}:{}", off + span) - }; - self.fail(format!( - "frame {what} out of bounds for the StackBuf({len}) named by `addr`" - )); - } - return; - } - let Some(&size) = self.heap_sizes.get(&base) else { - return; - }; - if exp + span > size { - // Several names can share a cell, so pick the first alphabetically - // rather than the first the map happens to yield: the same program - // must blame the same name on every build. - let name = self - .scope - .names - .iter() - .filter(|(_, b)| matches!(b.val, Binding::Scalar(c) if c == base)) - .map(|(n, _)| n.as_str()) - .min() - .unwrap_or("?"); - if span == 1 { - self.fail(format!( - "heap index {exp} out of bounds for `{name}` (HeapBuf size {size})" - )); - } - self.fail(format!( - "heap slice {exp}:{} out of bounds for `{name}` (HeapBuf size {size})", - exp + span - )); - } - } - - /// `addr(sb)`: the g-address `fp·g^base` of a `StackBuf`'s first cell, so a - /// frame run can be pointed at (indexed at runtime, or handed to a callee) - /// while its own accesses stay direct frame cells. Materializing `fp` is the - /// ISA's one cost here, amortized per function ([`Self::self_fp`]); the - /// address is a folded [`GAddr`], so `addr(sb) * GEN ** k` stays virtual. - pub(super) fn stack_addr(&mut self, args: &[Expr]) -> GAddr { - if args.len() != 1 { - self.fail("addr(buf) takes one StackBuf") - }; - let (base, len) = self.stack_of(&args[0]).unwrap_or_else(|| { - self.fail(format!( - "addr() names a frame run, so it takes a StackBuf, got `{:?}`", - args[0] - )) - }); - GAddr { - base: Some(self.self_fp()), - exp: base as u128, - run: Some((base, len)), - } - } - - /// Address `arr·g^extra` as `(base_cell, β)`, folding `arr`'s symbolic shift - /// and the constant `extra` into `β`. Falls back to a materialized pointer - /// (`β = 0`) when there is no runtime base or the offset exceeds [`FOLD_MAX`]. - fn heap_base(&mut self, arr: &Expr, extra: u128) -> (Off, u32) { - self.check_heap_bound(arr, extra, 1); - if let Some(ga) = self.gaddr_of(arr) - && let (Some(base), Some(exp)) = (ga.base, ga.exp.checked_add(extra)) - && exp <= FOLD_MAX - { - return (base, exp as u32); - } - let a = self.expr(arr); - if extra == 0 { - return (a, 0); - } - let k = self.const_cell(g_pow_u128(extra).into()); - (self.pure(PureOp::Mul, a, k), 0) - } - - /// Resolve a heap access `arr[idx]` to a `DEREF`-ready pair: a cell - /// holding a pointer `p` and a compile-time exponent `o2`, the accessed - /// cell being `m[p·g^o2]` (heap addressing in the exponent: cell `g^k` - /// of the buffer sits at `arr·g^k`). The fallback of [`Self::heap_addr`], - /// which has already folded away a wholly constant index: here a constant - /// g-power *factor* still goes into the `o2` immediate, so only the - /// runtime factor costs a pointer `MUL`. - fn array_ptr(&mut self, arr: &Expr, idx: &Expr) -> (Off, u32) { - // `buf[r * GEN ** k]` (either factor order): o2 takes the constant, - // the pointer MUL takes only the runtime factor `r`. - if let Expr::Mul(a, b) = idx { - for (c, r) in [(a, b), (b, a)] { - if let Some(k) = self.const_gpow(c) { - let (la, lr) = (self.expr(arr), self.expr(r)); - return (self.pure(PureOp::Mul, la, lr), k); - } - } - } - let (la, li) = (self.expr(arr), self.expr(idx)); - (self.pure(PureOp::Mul, la, li), 0) - } - - /// Realize a [`GAddr`] into a frame cell holding its value: a constant is one - /// `SET`; a base with no shift is already that cell; a shifted base is a - /// `SET`+`MUL`. - pub(super) fn materialize(&mut self, ga: GAddr) -> Off { - match ga { - GAddr { - base: Some(c), exp: 0, .. - } => c, - GAddr { base, exp, .. } => { - let k = self.const_cell(g_pow_u128(exp).into()); - let Some(c) = base else { return k }; - self.pure(PureOp::Mul, c, k) - } - } - } - - /// If `e` names a `StackBuf` variable, its `(base, size)`. - pub(super) fn stack_of(&self, e: &Expr) -> Option<(Off, u32)> { - match e { - Expr::Var(v) => self.scope.stack(v), - _ => None, - } - } - - /// Allocate `n` *consecutive* fresh frame cells (a stack run), returning the - /// base. Nothing else may `fresh()` between them, so they stay adjacent. - pub(super) fn alloc_stack(&mut self, n: u32) -> Off { - let base = self.next; - self.next += n; - base - } -} diff --git a/crates/lean_compiler/src/parser.rs b/crates/lean_compiler/src/parser.rs deleted file mode 100644 index 892498938..000000000 --- a/crates/lean_compiler/src/parser.rs +++ /dev/null @@ -1,907 +0,0 @@ -//! Parser: a minimal indentation-based Python-like surface syntax → [`Ast`]. -//! -//! This file holds the line-oriented part: the indentation structure, the -//! statement forms, and the top level's global constants. The rest is split by -//! the question it answers: -//! -//! - [`mod@expr`] reads structure out of a line, under one rule: a string -//! literal is one opaque token, and every scan goes through `depth0`. -//! - [`mod@consts`] evaluates a constant at parse time, which five syntactic -//! positions demand before lowering ever runs. It reads a constant as an -//! INTEGER, deliberately, which is what makes a derived size right. -//! - [`mod@subst`] substitutes an expression for a name through a statement -//! tree, which is how `unroll` and `Const` bind their variable. - -use super::*; -use std::collections::BTreeMap; - -mod consts; -mod expr; -mod subst; -pub use consts::parse_const; -use consts::{apply_replacements, const_int_expr, eval_const_int, gpow_bound, parse_f192_const, parse_gpow_bound}; -use expr::{ - binding_name, call_args, is_ident, parse_expr, split_assign, split_aug, split_once_top, split_top, string_lit, - strip_comment, strip_const_wrapper, top_level_cmp, -}; -pub(crate) use subst::subst_stmts; -use subst::subst_var; -/// Parse zkDSL source (the Python-shaped surface syntax of `zkDSL.md`) into an -/// [`Ast`]. The `snark_lib` import is skipped; any other import is an error, -/// since a program is a single file. -pub fn parse(src: &str) -> Result { - parse_with_replacements(src, &BTreeMap::new()) -} - -/// Like [`parse`], but first substitutes compile-time **placeholders**: an -/// identifier that is a key of `replacements` becomes its value everywhere it -/// appears, which is how a host bakes sizes and flags into a program without -/// editing it. The top level then peels off the **global constants**, each -/// evaluated as a compile-time integer (or an `f192` literal / constant array) -/// and substituted into the `def`s below. See the "Placeholders" and "Global -/// constants" sections of `zkDSL.md`. -pub fn parse_with_replacements(src: &str, replacements: &BTreeMap) -> Result { - // Substitution runs over the RAW source, before lines are split, so a - // replacement carrying a newline would shift every line after it and make - // every diagnostic below name the wrong one. It would also inject statements - // at whatever indentation it landed on. Reject it instead. - // A `#` truncates the rest of the line just as silently, and changes the - // compiled program with no diagnostic at all. - // A `"` reshapes the line just as a `#` does, now that a string literal is one - // opaque token: an odd number of them swallows the rest of the line. - if let Some((k, _)) = replacements - .iter() - .find(|(_, v)| v.contains('\n') || v.contains('#') || v.contains('"')) - { - return Err(format!( - "placeholder `{k}` contains a newline, `#` or `\"`, which would reshape the line" - )); - } - let src = apply_replacements(src, replacements); - // (source line, indent, content) for each significant line. The source line - // is what every diagnostic below names; blanks, comments and imports are - // skipped, so the index into this vector is NOT it. - let mut lines: Vec = Vec::new(); - for (src_line, raw) in src.lines().enumerate() { - let no_comment = strip_comment(raw); - if no_comment.trim().is_empty() { - continue; - } - let t = no_comment.trim(); - if let Some(rest) = t.strip_prefix("import ").or_else(|| t.strip_prefix("from ")) { - let module = rest.split_whitespace().next().unwrap_or(""); - if module != "snark_lib" { - return Err(locate( - src_line + 1, - format!("file imports are not supported (only the `snark_lib` stub): `{t}`"), - )); - } - continue; // the stub is for Python tooling; the compiler skips it - } - let indent = no_comment.len() - no_comment.trim_start().len(); - lines.push(Line { - src: src_line + 1, - indent, - text: no_comment.trim().to_string(), - }); - } - // Peel off the leading top-level constant declarations (before any `def`), - // each evaluated to a field value and rendered as a single decimal literal. - // Building a `name → literal` map lets later constants and the functions - // reference them by plain text substitution, so a constant works even in - // positions that demand a parse-time literal (`StackBuf`, `**`, `assert log - // _ < _`). - let mut consts: BTreeMap = BTreeMap::new(); - let mut const_arrays: Vec<(String, Vec)> = Vec::new(); - let mut start = 0; - while start < lines.len() { - let Line { - src, - indent, - text: line, - } = &lines[start]; - let at = |e: String| locate(*src, e); - if *indent == 0 && (line.starts_with("def ") || line.starts_with('@')) { - break; - } - if *indent != 0 { - return Err(at(format!("unexpected indentation at top level: `{line}`"))); - } - let (lhs, rhs) = split_assign(line).ok_or_else(|| { - at(format!( - "top level: expected `def`, a global constant `NAME = value`, or the `snark_lib` import, got `{line}`" - )) - })?; - let name = lhs.trim().to_string(); - if !is_ident(&name) { - return Err(at(format!( - "global constant name must be a plain identifier: `{}`", - lhs.trim() - ))); - } - if consts.contains_key(&name) || const_arrays.iter().any(|(n, _)| n == &name) { - return Err(at(format!("global constant `{name}` is declared twice"))); - } - // A scalar constant is substituted textually, so one named after a builtin - // rewrites the builtin's own call sites: `match = 4` turned - // `v = match(log(x), …)` into `4(log(x), …)`, whose diagnostic names - // neither the constant nor `match`. - if BUILTINS.contains(&name.as_str()) { - return Err(at(format!( - "`{name}` is a builtin, so a global constant of that name would be substituted \ - into its own call sites. Rename it" - ))); - } - // Resolve earlier scalar constants inside the value first. - let rhs = apply_replacements(rhs.trim(), &consts); - let rhs = rhs.trim(); - if let Some(inner) = rhs.strip_prefix('[').and_then(|s| s.strip_suffix(']')) { - // A constant array `NAME = [a, b, c]`: each element a compile-time - // integer / field value. Not textually substituted, but carried to - // lowering, indexed/measured there. - let mut elems = Vec::new(); - for part in split_top(inner, ',') { - let p = part.trim(); - if p.is_empty() { - continue; // tolerate a trailing comma - } - let elem = if let Some(v) = parse_f192_const(p) { - v.map_err(|e| at(format!("global constant array `{name}`: {e}")))? - } else { - let n = eval_const_int(p).map_err(|e| at(format!("global constant array `{name}`: {e}")))?; - F192::new(n as u64, (n >> 64) as u64, 0) - }; - elems.push(elem); - } - const_arrays.push((name, elems)); - } else { - // A scalar constant: an `f192` literal, else a compile-time integer, - // else a field-valued expression. - if let Some(value) = parse_f192_const(rhs) { - let v = value.map_err(|e| at(format!("global constant `{name}`: {e}")))?; - consts.insert(name, format!("f192({},{},{})", v.c0, v.c1, v.c2)); - } else if let Ok(value) = eval_const_int(rhs) { - consts.insert(name, value.to_string()); - } else { - // `GEN ** 2` and friends. The ISA is written in g-powers, so this - // is the natural spelling for a constant one, and it is not an - // integer expression. Rendered as a decimal wherever the value - // fits the low two limbs, so the constant still works in the - // positions that demand a literal rather than only as a value. - let v = parse_const(rhs).map_err(|e| at(format!("global constant `{name}`: {e}")))?; - consts.insert( - name, - if v.c2 == 0 { - (v.c0 as u128 | ((v.c1 as u128) << 64)).to_string() - } else { - format!("f192({},{},{})", v.c0, v.c1, v.c2) - }, - ); - } - } - start += 1; - } - // Substitute the constants into every remaining (function) line, then parse. - let func_lines: Vec = lines[start..] - .iter() - .map(|l| Line { - src: l.src, - indent: l.indent, - text: apply_replacements(&l.text, &consts), - }) - .collect(); - let mut p = Parser { - lines: &func_lines, - i: 0, - }; - let mut funcs = Vec::new(); - while p.i < p.lines.len() { - funcs.push(p.func()?); - } - // Two names that used to be accepted and then silently picked a winner. - // A repeated `def` lowered both bodies and kept the last, and a name - // beginning with `__` can collide with a compiler-generated one: a loop - // helper is `__loopN`, and a `Const` specialization of `f` is `f__L1`, so a - // user function called `f__L1` took the specialization's place and the call - // ran the wrong body. - for (i, f) in funcs.iter().enumerate() { - if funcs[..i].iter().any(|g| g.name == f.name) { - return Err(format!("function `{}` is defined twice", f.name)); - } - if f.name.contains("__") { - return Err(format!( - "function name `{}` may not contain `__`, which is reserved for the names the \ - compiler generates (a loop helper, a `Const` specialization)", - f.name - )); - } - // A builtin wins at the call site, so a function with a builtin's name is - // never called and its body, constraints included, silently disappears. - // `def const(x): assert x == 99` was skipped outright by `v = const(4)`, - // and by whether the ARGUMENT folded, so one call site had two meanings. - if BUILTINS.contains(&f.name.as_str()) { - return Err(format!( - "`{}` is a builtin, so a function of that name could never be called: \ - the builtin takes every call site. Rename it", - f.name - )); - } - } - infer_return_shapes(&mut funcs)?; - Ok(Ast { funcs, const_arrays }) -} - -/// Every name the lowerer resolves before it looks for a user function. A `def` -/// may not take one of these, since the builtin would win and the body would be -/// dead code that still looked live. -const BUILTINS: &[&str] = &[ - "addr", - "assert_in_k", - "blake2s", - "const", - "f192", - "hint_decompose_bits", - "hint_decompose_bits_exponent", - "hint_f192_limbs", - "hint_log2_ceil", - "hint_witness", - "len", - "match", - "HeapBuf", - "StackBuf", -]; - -/// Infer the compile-time representation of each tail-return value: a `StackBuf` -/// carries its static size, a `HeapBuf` stays a one-cell pointer (its allocation -/// hint ran in the creating function). Iterated to a fixed point, so a wrapper -/// may return a buffer produced by a function declared later. -fn infer_return_shapes(funcs: &mut [Func]) -> Result<(), String> { - fn expr_shape( - e: &Expr, - locals: &HashMap<&str, Shape>, - known: &HashMap>, - ) -> Result { - let fits = |n: u64| u32::try_from(n).map_err(|_| format!("StackBuf size {n} does not fit in u32")); - Ok(match e { - Expr::Var(v) => locals.get(v.as_str()).copied().unwrap_or(Shape::Scalar), - Expr::StackBuf(n) => Shape::StackBuf(fits(*n)?), - Expr::ListLit(es) => Shape::StackBuf(fits(es.len() as u64)?), - Expr::Call(f, _) => known - .get(f) - .filter(|r| r.len() == 1) - .and_then(|r| r.first()) - .copied() - .unwrap_or(Shape::Scalar), - _ => Shape::Scalar, - }) - } - - fn scan( - body: &[Stmt], - params: &[Param], - known: &HashMap>, - n_ret: usize, - ) -> Result, String> { - // Seeded from the DECLARED shapes: a `s: StackBuf(n)` parameter is a run - // here as much as a local one is, so `return s` returns the run rather - // than reporting it used as a scalar. - let mut locals: HashMap<&str, Shape> = params.iter().map(|p| (p.name.as_str(), p.shape())).collect(); - let mut returns = vec![Shape::Scalar; n_ret]; - for stmt in body { - match &stmt.kind { - StmtKind::Let(name, e) => { - let shape = expr_shape(e, &locals, known)?; - locals.insert(name.as_str(), shape); - } - StmtKind::LetTuple(names, f, _) => { - let shapes = known.get(f); - for (i, name) in names.iter().enumerate() { - let shape = shapes.and_then(|s| s.get(i)).copied().unwrap_or(Shape::Scalar); - locals.insert(name.as_str(), shape); - } - } - // `unroll` is straight-line expansion, so a binding in its last - // copy remains visible afterward. One symbolic scan is enough - // for representation shapes (the iteration value is scalar). - StmtKind::Unroll { var, body, .. } => { - locals.insert(var.as_str(), Shape::Scalar); - for inner in body { - if let StmtKind::Let(name, e) = &inner.kind { - let shape = expr_shape(e, &locals, known)?; - locals.insert(name.as_str(), shape); - } - } - } - StmtKind::LetHintWitness { name, .. } => { - locals.insert(name.as_str(), Shape::Scalar); - } - StmtKind::Return(es) => { - returns = es - .iter() - .map(|e| expr_shape(e, &locals, known)) - .collect::>()?; - } - _ => {} - } - } - Ok(returns) - } - - let mut known: HashMap> = funcs - .iter() - .map(|f| (f.name.clone(), f.return_shapes.clone())) - .collect(); - // A shape can only move from Scalar to one of the finite constructor - // shapes (or acquire one through a call), so `funcs.len() + 1` rounds are - // sufficient for the longest acyclic wrapper chain. - for _ in 0..=funcs.len() { - let next: HashMap> = funcs - .iter() - .map(|f| Ok((f.name.clone(), scan(&f.body, &f.params, &known, f.return_shapes.len())?))) - .collect::>()?; - if next == known { - break; - } - known = next; - } - for f in funcs { - f.return_shapes = known - .remove(&f.name) - .expect("every function has inferred return shapes"); - } - Ok(()) -} - -/// One significant source line: its 1-based position in the ORIGINAL file, its -/// indentation, and its text with comments stripped and constants substituted. -/// Prefix a diagnostic with the source line it came from, unless an inner frame -/// already named a more specific one. -/// Does the leading call span the WHOLE of `s`? -/// -/// `call_args` strips the first `(` and the LAST `)`, so it also matches a line -/// where the call is only the first factor: `hint_witness("a") * f("b")` came -/// back with the stream name `a") * f("b`, and the rest of the line vanished. -/// A `)` that closes below depth zero means the call ended before the line did. -fn whole_call(s: &str) -> bool { - let Some(inner) = s.find('(').map(|i| &s[i + 1..]) else { - return false; - }; - let (mut depth, mut in_str) = (0i32, false); - for (i, c) in inner.bytes().enumerate() { - if in_str { - in_str = c != b'"'; - continue; - } - match c { - b'"' => in_str = true, - b'(' | b'[' => depth += 1, - b')' | b']' => { - depth -= 1; - // The close that matches the call's own `(`. It has to be the - // last thing on the line, or something followed the call. - if depth < 0 { - return i + 1 == inner.len(); - } - } - _ => {} - } - } - false -} - -fn locate(line: usize, e: String) -> String { - if e.starts_with("line ") { - e - } else { - format!("line {line}: {e}") - } -} - -struct Line { - src: usize, - indent: usize, - text: String, -} - -struct Parser<'a> { - lines: &'a [Line], - i: usize, -} - -impl Parser<'_> { - /// The source line the cursor is on, for a diagnostic raised where no - /// `func`/`stmt` frame is open: those two stamp the line they were ENTERED - /// on, which is an enclosing header, not the line that is actually wrong. - fn here(&self) -> usize { - self.lines - .get(self.i) - .or_else(|| self.lines.last()) - .map_or(0, |l| l.src) - } - - fn func(&mut self) -> Result { - let here = self.lines.get(self.i).map_or(0, |l| l.src); - self.func_inner().map_err(|e| locate(here, e)) - } - - fn func_inner(&mut self) -> Result { - let mut line = &self.lines[self.i]; - // Optional `@inline` decorator on its own line before `def`. - let inline = if let Some(dec) = line.text.strip_prefix('@') { - if dec.trim() != "inline" { - return Err(format!("unknown decorator `@{}` (only `@inline`)", dec.trim())); - } - self.i += 1; - line = self.lines.get(self.i).ok_or("`@inline` must precede a `def`")?; - true - } else { - false - }; - // Re-stamped here: the decorator path advanced past its own line, so - // `func`'s frame would name the `@inline` while quoting the `def`. - let at_def = self.here(); - self.func_header(inline, line.indent, &line.text) - .map_err(|e| locate(at_def, e)) - } - - fn func_header(&mut self, inline: bool, indent: usize, line: &str) -> Result { - let header = line - .strip_prefix("def ") - .ok_or_else(|| format!("expected `def`, got `{line}`"))?; - let header = header.strip_suffix(':').ok_or("function header needs `:`")?; - let open = header.find('(').ok_or("function header needs `(`")?; - let name = header[..open].trim().to_string(); - let params_str = header[open + 1..header.rfind(')').ok_or("missing `)`")?].trim(); - let mut params = Vec::new(); - if !params_str.is_empty() { - for part in params_str.split(',') { - if part.trim().is_empty() { - return Err(format!("`def {name}`: empty parameter (a trailing comma?)")); - } - // `x`, `x: Const` (compile-time, specialized), or - // `x: StackBuf(n)` (a run of n cells, passed whole). - let (param_name, annotation) = part.split_once(':').map_or((part, None), |(n, a)| (n, Some(a.trim()))); - let kind = match annotation { - None => ParamKind::Runtime(Shape::Scalar), - Some("Const") => ParamKind::Const, - Some(ann) => { - let size = ann.strip_prefix("StackBuf(").and_then(|r| r.strip_suffix(')')).ok_or_else(|| { - format!("unsupported parameter annotation `{ann}` (`Const`, or `StackBuf(n)` to pass a run of cells)") - })?; - let k = - eval_const_int(size).map_err(|e| format!("`def {name}`: StackBuf parameter size: {e}"))?; - let k = u32::try_from(k).map_err(|_| format!("`def {name}`: StackBuf({k}) is too large"))?; - if k == 0 { - return Err(format!("`def {name}`: a StackBuf parameter needs at least one cell")); - } - ParamKind::Runtime(Shape::StackBuf(k)) - } - }; - params.push(Param { - name: binding_name(param_name, "parameter name")?, - kind, - }); - } - } - if let Some(dup) = params - .iter() - .enumerate() - .find_map(|(i, p)| params[..i].iter().any(|q| q.name == p.name).then_some(&p.name)) - { - return Err(format!("parameter `{dup}` is declared twice")); - } - self.i += 1; - let body = self.block(indent)?; - let n_ret = body - .iter() - .filter_map(|s| { - if let StmtKind::Return(es) = &s.kind { - Some(es.len()) - } else { - None - } - }) - .max() - .unwrap_or(0); - Ok(Func { - name, - params, - return_shapes: vec![Shape::Scalar; n_ret], - body, - inline, - }) - } - - /// Parse a block: all statements indented strictly more than `parent`. - fn block(&mut self, parent: usize) -> Result, String> { - let mut stmts = Vec::new(); - let block_indent = match self.lines.get(self.i) { - Some(Line { indent: ind, .. }) if *ind > parent => *ind, - _ => return Err(locate(self.here(), "expected an indented block".into())), - }; - while let Some(Line { indent: ind, .. }) = self.lines.get(self.i) { - if *ind != block_indent { - if *ind > parent && *ind > block_indent { - return Err(locate(self.here(), "inconsistent indentation".into())); - } - break; - } - stmts.push(self.stmt(block_indent)?); - } - Ok(stmts) - } - - fn stmt(&mut self, indent: usize) -> Result { - let here = self.lines.get(self.i).map_or(0, |l| l.src); - self.stmt_inner(indent) - .map(|kind| Stmt::new(here as u32, kind)) - .map_err(|e| locate(here, e)) - } - - fn stmt_inner(&mut self, indent: usize) -> Result { - let line = self.lines[self.i].text.as_str(); - if let Some(rest) = line.strip_prefix("for ") { - // for VAR in mul_range(START, STOP): the counter walks gᵏ from START - // to STOP, ×g each iteration (STOP is exclusive). Bounds are field - // elements (powers of GEN), so the multiplicative walk is explicit. - let rest = rest.strip_suffix(':').ok_or("`for` needs `:`")?; - let (var, iter) = rest.split_once(" in ").ok_or("`for` needs `in`")?; - // `for i in unroll(a, b):` is compile-time replication; the bounds - // are integer expressions, evaluated at lowering (so a `Const` - // parameter works as a bound). - if let Some(parts) = call_args(iter, "unroll") { - if parts.len() != 2 { - return Err("unroll needs `a, b` (compile-time integers)".into()); - } - let (lo, hi) = (parse_expr(parts[0])?, parse_expr(parts[1])?); - self.i += 1; - let body = self.block(indent)?; - return Ok(StmtKind::Unroll { - var: binding_name(var, "loop counter")?, - lo, - hi, - body, - }); - } - let parts = call_args(iter, "mul_range").ok_or("`for` needs `mul_range(start, stop)` or `unroll(a, b)`")?; - if parts.len() != 2 { - return Err("mul_range needs `start, stop`".into()); - } - let lo = parse_gpow_bound(parts[0])?; - // The stop bound: a compile-time power of GEN, or any expression, - // a runtime g-power element the walk must be able to reach. - let hi = match parse_gpow_bound(parts[1]) { - Ok(hi) => { - if lo > hi { - return Err(format!("mul_range: start GEN**{lo} must not exceed stop GEN**{hi}")); - } - ForBound::Const(hi) - } - Err(_) => { - let stop = parse_expr(parts[1])?; - // A compile-time value that is not a power of GEN can never be - // REACHED: the counter walks by multiplication and exits on - // equality, so the loop runs forever at witness generation with - // no diagnostic. The `lo` side has always been checked, which - // made `mul_range(0, GEN ** 3)` a clean parse error while - // `mul_range(1, 10)` was a hang. - // A power of two reached `gpow_bound` above, so a literal here - // is one the multiplicative walk can never hit. This catches a - // bare `10` and a constant that substitutes to one; a value - // built by arithmetic (`5 * 2`) still slips through to the - // runtime path and still hangs, which wants a field-level - // constant folder the parser does not have. - if let Expr::Lit(n) = stop { - return Err(format!( - "mul_range stop bound `{n}` is not a power of GEN, so the multiplicative walk \ - never reaches it: write `GEN ** k`" - )); - } - ForBound::Runtime(stop) - } - }; - self.i += 1; - let body = self.block(indent)?; - return Ok(StmtKind::For { - var: binding_name(var, "loop counter")?, - lo, - hi, - body, - }); - } - if let Some(rest) = line.strip_prefix("if ") { - return self.if_stmt(rest, indent); - } - self.i += 1; - if line == "return" { - return Ok(StmtKind::Return(vec![])); - } - if let Some(rest) = line.strip_prefix("return ") { - return Ok(StmtKind::Return( - split_top(rest, ',') - .iter() - .map(|e| parse_expr(e)) - .collect::>()?, - )); - } - // `print(expr)` / `print("label", expr)`: prover-side debug print; - // the label defaults to the argument's source text. - if let Some(parts) = call_args(line, "print") { - let (label, value) = match parts.as_slice() { - [l, v] if l.trim().starts_with('"') => { - let l = string_lit(l).ok_or("print's label is a string literal: print(\"label\", expr)")?; - (l.to_string(), v.trim()) - } - [v] => (v.trim().to_string(), v.trim()), - _ => return Err("print takes `print(expr)` or `print(\"label\", expr)`".into()), - }; - return Ok(StmtKind::Print { - label, - value: parse_expr(value)?, - }); - } - // `hint_witness(dest, "name")`: the string literal is not an - // expression; parsed here. `whole_call` for the same reason as the - // scalar form: the string arg is what lets a trailing `* f("b")` - // vanish into the stream name instead of failing to parse. - if let Some(parts) = call_args(line, "hint_witness").filter(|_| whole_call(line)) { - let [dest, name] = parts.as_slice() else { - return Err("hint_witness(dest, \"name\") takes two arguments".into()); - }; - let name = string_lit(name).ok_or("hint_witness's second argument is a string literal: \"name\"")?; - return Ok(StmtKind::HintWitness { - dest: parse_expr(dest)?, - name: name.to_string(), - }); - } - if let Some(rest) = line.strip_prefix("assert ") { - if let Some((a, b)) = split_once_top(rest, "==") { - return Ok(StmtKind::AssertEq(parse_expr(a)?, parse_expr(b)?)); - } - if let Some((a, b)) = split_once_top(rest, "!=") { - return Ok(StmtKind::AssertNe(parse_expr(a)?, parse_expr(b)?)); - } - // `assert log X < log Y` (`Y` a compile-time g-power, or any runtime - // g-power) or `assert log X < k` (`k` an integer exponent) is a - // range check in the exponent: proves `log_g(X) < k`. - if let Some((a, b)) = split_once_top(rest, "<") { - let x = - strip_log(a).ok_or("a `<` assert compares logs: `assert log X < log Y` or `assert log X < k`")?; - let bound = match strip_log(b) { - // `log GEN ** k = k` when the bound folds to a power of GEN; - // otherwise it is a runtime g-power and the gadget derives - // `g^{k-1}` from it. A bound that folds to something else is a - // mistake rather than a runtime bound: `log 8` names the field - // element 8, whose g-log is nothing in particular, and taking it - // for a bound would fail only at witness generation. - Some(y) => { - let y = parse_expr(y)?; - match gpow_bound(&y) { - Ok(k) => LtBound::Const(k), - Err(e) if const_int_expr(&y).is_some() => return Err(e), - Err(_) => LtBound::Runtime(y), - } - } - // An integer bound folds like any parse-time size (`CAP + 1`). - None => match const_int_expr(&parse_expr(b)?) { - Some(k) => { - LtBound::Const(u64::try_from(k).map_err(|_| format!("log bound {k} does not fit in u64"))?) - } - None => { - return Err(format!( - "a log bound must be `log _` or a parse-time integer, got `{b}`" - )); - } - }, - }; - return Ok(StmtKind::AssertLt(parse_expr(x)?, bound)); - } - return Err("`assert` needs `==`, `!=`, or `log _ < _`".into()); - } - // Augmented assignment `x OP= rhs` (Python `*=`, `+=`, `//=`, `%=`, - // `-=`) desugars to `x = x OP (rhs)`. - let expanded = split_aug(line)?.map(|(lhs, op, rhs)| format!("{lhs} = {lhs} {op} ({rhs})")); - let line = expanded.as_deref().unwrap_or(line); - // Assignment or bare call. - if let Some((lhs, rhs)) = split_assign(line) { - // `names = match(…)` carries lambdas, which `parse_expr` - // does not speak, so it gets its own parser. - if rhs.trim_start().starts_with("match(") { - return parse_match(lhs, rhs); - } - // `x = hint_witness("stream")`: one hinted value, no buffer. The - // string is not an expression, so like the run form it is parsed - // here rather than by `parse_expr`. - if let Some(parts) = call_args(rhs.trim(), "hint_witness").filter(|_| whole_call(rhs.trim())) { - let [stream] = parts.as_slice() else { - return Err( - "`x = hint_witness(\"stream\")` takes one argument; to fill a run, write \ - `hint_witness(dest, \"stream\")` as a statement" - .into(), - ); - }; - let stream = string_lit(stream).ok_or("hint_witness's argument is a string literal: \"stream\"")?; - return Ok(StmtKind::LetHintWitness { - name: binding_name(lhs, "binding name")?, - stream: stream.to_string(), - }); - } - let rhs_expr = parse_expr(rhs)?; - // Indexed LHS `arr[idx] = value` is a heap store. - if lhs.trim_end().ends_with(']') { - let lhs = lhs.trim(); - let open = lhs.find('[').ok_or("malformed store target")?; - let arr = parse_expr(&lhs[..open])?; - let idx = parse_expr(&lhs[open + 1..lhs.len() - 1])?; - return Ok(StmtKind::Store(arr, idx, rhs_expr)); - } - let targets = split_top(lhs, ','); - if targets.len() == 1 { - return Ok(StmtKind::Let(binding_name(targets[0], "binding name")?, rhs_expr)); - } - // Tuple assignment: RHS must be a call. - if let Expr::Call(f, args) = rhs_expr { - let names = targets - .iter() - .map(|t| binding_name(t, "binding name")) - .collect::, _>>()?; - return Ok(StmtKind::LetTuple(names, f, args)); - } - return Err("tuple assignment requires a call on the right".into()); - } - // A bare comparison is not a statement, and `split_assign` used to split - // one on its `=` into a binding named `x !`. In a verifier that is an - // `assert` compiled to nothing, so name the fix rather than letting the - // expression parser fail on an operator it does not have. - // `while`/`elif`/`else` reach here as unknown keywords, not comparisons, so - // suggesting `assert while x < y:` would be worse than the generic error. - let keyword = ["while ", "elif ", "else", "for ", "def "] - .iter() - .any(|k| line.starts_with(k)); - if let Some(op) = top_level_cmp(line).filter(|_| !keyword) { - let fix = if matches!(op, "==" | "!=") { - format!("write `assert {line}`") - } else { - format!("`{op}` is not a predicate: order facts come from `assert log x < k`") - }; - return Err(format!("`{line}` is a comparison, not a statement: {fix}")); - } - // Bare call statement. - if let Expr::Call(f, args) = parse_expr(line)? { - return Ok(StmtKind::Call(f, args)); - } - Err(format!("statement has no effect: `{line}`")) - } - - /// `if a == b:` / `if a != b:` (the current line, its `if `/`elif ` - /// prefix already stripped into `header`), with an optional `elif`/`else` - /// tail at the same indent (an `elif` is sugar for an `else` holding a - /// nested `if`). - fn if_stmt(&mut self, header: &str, indent: usize) -> Result { - let cond = header.strip_suffix(':').ok_or("`if` needs `:`")?; - let (cond, force_const) = match strip_const_wrapper(cond) { - Some(inner) => (inner, true), - // A near miss (`const (a == b)`, an unbalanced one) otherwise falls - // through to an ordinary parse error that never says the word, so name - // it here. Two things are NOT near misses: `const == 4`, a variable - // that happens to be called `const`, since nothing follows the name but - // the comparison; and a condition with a comparison of its own, since - // `const(...)` is a value expression too, so `if const(k) == n:` is an - // ordinary runtime test of a folded literal and rejecting it made the - // two operand orders behave differently. - None if top_level_cmp(cond).is_none() - && cond - .trim_start() - .strip_prefix("const") - .is_some_and(|r| r.trim_start().starts_with('(')) => - { - return Err( - "`const(...)` must wrap the WHOLE condition and balance its brackets: `if const(a == b):`".into(), - ); - } - None => (cond, false), - }; - let (eq, l, r) = if let Some((l, r)) = split_once_top(cond, "==") { - (true, l, r) - } else if let Some((l, r)) = split_once_top(cond, "!=") { - (false, l, r) - } else { - return Err("an `if` condition must be `a == b` or `a != b`".into()); - }; - let (lhs, rhs) = (parse_expr(l)?, parse_expr(r)?); - self.i += 1; - let then = self.block(indent)?; - let mut els = Vec::new(); - if let Some(Line { - indent: ind, - text: line, - .. - }) = self.lines.get(self.i) - && *ind == indent - { - if line == "else:" { - self.i += 1; - els = self.block(indent)?; - } else if let Some(rest) = line.strip_prefix("elif ") { - // `if_stmt` recurses into ITSELF for an `elif`, so no `stmt` - // frame opens and the whole chain would report the first `if`. - let at_elif = self.here(); - els = vec![Stmt::new( - at_elif as u32, - self.if_stmt(rest, indent).map_err(|e| locate(at_elif, e))?, - )]; - } - } - Ok(StmtKind::If { - eq, - lhs, - rhs, - then, - els, - force_const, - }) - } -} - -type Aug<'a> = Option<(&'a str, &'static str, &'a str)>; - -/// Strip a leading `log` token (`log x`, `log(x)`), if present. The token must -/// end at a boundary, so a variable named `logx` is not a log of `x`. -fn strip_log(s: &str) -> Option<&str> { - let r = s.trim_start().strip_prefix("log")?; - r.starts_with([' ', '(']).then_some(r) -} - -/// `names = match(log(x), range(a, b), lambda i: expr, …)`: leanVM's -/// `match`, expanded at parse time: one arm per integer of the -/// contiguous `(range, lambda)` pairs, arm `j` being the lambda body with the -/// parameter substituted by the literal `j`. The union of the ranges must be -/// gapless and start at 0 (this compiler's `match` rule). Everything sits on -/// one line, since there is no line continuation. -fn parse_match(lhs: &str, rhs: &str) -> Result { - // A target is a name, or a `StackBuf` element, which the arms then write - // into directly: the ABI already returns into cells the caller picks, so - // `sb[i], e = match(…)` costs no copy where a name plus a store did. - let targets = split_top(lhs, ',') - .iter() - .map(|t| match parse_expr(t)? { - e @ (Expr::Var(_) | Expr::Index(..)) => Ok(e), - other => Err(format!( - "a `match` target must be a name or a StackBuf element, got `{other:?}`" - )), - }) - .collect::, _>>()?; - let chunks = call_args(rhs, "match").ok_or("malformed `match(…)`")?; - let (first, pairs) = chunks.split_first().ok_or("match needs arguments")?; - let x = strip_log(first).ok_or("`match` matches logs: `match(log(x), …)`")?; - let x = parse_expr(x)?; - if pairs.is_empty() || !pairs.len().is_multiple_of(2) { - return Err("match needs `range(a, b), lambda i: …` pairs after the scrutinee".into()); - } - let mut arms = Vec::new(); - for pair in pairs.chunks(2) { - let (lo, hi) = match parse_expr(pair[0])? { - Expr::Call(f, args) if f == "range" => match args.as_slice() { - [Expr::Lit(a), Expr::Lit(b)] if a < b => (*a, *b), - _ => return Err("match needs `range(a, b)` with integer literals, a < b".into()), - }, - other => return Err(format!("expected `range(a, b)`, got `{other:?}`")), - }; - if lo != arms.len() as u128 { - return Err(format!( - "match ranges must be contiguous from 0: expected a range starting at {}, got {lo}", - arms.len() - )); - } - let lam = pair[1] - .trim() - .strip_prefix("lambda ") - .ok_or("expected `lambda i: …` after each range")?; - let (param, body) = split_once_top(lam, ":").ok_or("`lambda` needs `:`")?; - let body = parse_expr(body)?; - for j in lo..hi { - arms.push(subst_var(&body, param.trim(), &Expr::Lit(j))); - } - } - Ok(StmtKind::Match { targets, x, arms }) -} diff --git a/crates/lean_compiler/src/parser/consts.rs b/crates/lean_compiler/src/parser/consts.rs deleted file mode 100644 index ebaa3d046..000000000 --- a/crates/lean_compiler/src/parser/consts.rs +++ /dev/null @@ -1,148 +0,0 @@ -//! Evaluating a constant at parse time, and the one substitution that still -//! happens on text. -//! -//! Five syntactic positions demand a literal before lowering ever runs: a buffer -//! size, a `mul_range` bound, a range-check bound, a `match` range, and a -//! `case`. That is why a global constant is substituted rather than bound, and -//! it is the whole reason this module exists. -//! -//! A constant is read as a compile-time INTEGER, which is deliberate and -//! load-bearing: it is what makes a derived size come out right. So the same -//! text means different things here and in a value position, where it would fold -//! in the field with `+` as XOR. Neither reading is wrong, and `const(...)` is -//! how an author says which was meant (`zkDSL.md`, "`const(...)` in a value -//! position"). It is transparent here, a constant having only this reading. - -use super::*; - -/// Evaluate a compile-time **integer** constant expression: decimal literals -/// combined with `+ - * / // % **` and parentheses. This is ordinary integer -/// arithmetic (a global constant is a count, a size, an exponent), deliberately -/// distinct from runtime field arithmetic, so a derived size like `2 + (W - 1) * -/// V + LOG_LIFETIME` comes out right; a single `/` divides like `//` here. -/// References to earlier constants are already substituted to their decimal -/// values, so the input is pure arithmetic. Overflow, division by zero, and a -/// negative intermediate are errors. -pub(super) fn eval_const_int(s: &str) -> Result { - parse_expr(s) - .ok() - .and_then(|e| const_int_expr(&e)) - .ok_or_else(|| format!("not a compile-time integer constant expression: `{}`", s.trim())) -} - -/// Fold a compile-time INTEGER expression (literals combined with the usual -/// operators) to its value; `None` if any leaf is not a literal. Placeholders -/// are substituted before parsing, so `GEN ** (K_SKIP + 1)`-style exponents -/// fold here. -pub(super) fn const_int_expr(e: &Expr) -> Option { - match e { - Expr::Lit(k) => Some(*k), - // `const(e)` asks for the integer reading, which is the only reading a - // parse-time position has, so it is transparent rather than redundant. It - // used to be a parse error in a `StackBuf` size and a `log` bound while - // being accepted in a `HeapBuf` size, an `unroll` count and a `GEN **` - // exponent, which is one construct with two meanings depending on where it - // stood. - Expr::Call(f, args) if f == "const" && args.len() == 1 => const_int_expr(&args[0]), - Expr::Add(a, b) => const_int_expr(a)?.checked_add(const_int_expr(b)?), - Expr::Sub(a, b) => const_int_expr(a)?.checked_sub(const_int_expr(b)?), - Expr::Mul(a, b) => const_int_expr(a)?.checked_mul(const_int_expr(b)?), - // A single `/` between compile-time integers is integer division: in a - // constant (a count, a size), there is no field to divide in. - Expr::Div(a, b) | Expr::FieldDiv(a, b) => match const_int_expr(b)? { - 0 => None, - d => Some(const_int_expr(a)? / d), - }, - Expr::Mod(a, b) => match const_int_expr(b)? { - 0 => None, - d => Some(const_int_expr(a)? % d), - }, - Expr::Pow(a, b) => const_int_expr(a)?.checked_pow(u32::try_from(const_int_expr(b)?).ok()?), - _ => None, - } -} - -/// Evaluate a compile-time constant expression (integer literals, `GEN`, -/// `GEN ** k`, and `+`/`*` combinations of those) to its field element. -/// Used for the `# public_input: , ` annotation of `.py` test -/// programs (see `tests/py_source.rs`). -pub fn parse_const(s: &str) -> Result { - fn eval(e: &Expr) -> Result { - match e { - // An integer literal is the raw 128-bit bit pattern of a machine word. - Expr::Lit(n) => Ok(F192::new(*n as u64, (*n >> 64) as u64, 0)), - Expr::Gen => Ok(g_pow(1).into()), - Expr::GPow(k) => Ok(g_pow_u128(*k).into()), - Expr::Add(a, b) => Ok(eval(a)? + eval(b)?), - Expr::Mul(a, b) => Ok(eval(a)? * eval(b)?), - other => Err(format!("not a constant expression: `{other:?}`")), - } - } - eval(&parse_expr(s)?) -} - -pub(super) fn parse_f192_const(s: &str) -> Option> { - let inner = s.trim().strip_prefix("f192(")?.strip_suffix(')')?; - let parts = split_top(inner, ','); - Some((|| { - if parts.len() != 3 { - return Err("f192 needs exactly three limbs".into()); - } - let mut limbs = [0u64; 3]; - for (i, p) in parts.iter().enumerate() { - limbs[i] = - u64::try_from(eval_const_int(p.trim())?).map_err(|_| "an f192 limb does not fit in u64".to_string())?; - } - Ok(F192::new(limbs[0], limbs[1], limbs[2])) - })()) -} - -/// A range bound (`mul_range` bounds and `assert log _ < log _` bounds): a -/// compile-time power of the generator (`1` = `g^0`, `GEN` = `g^1`, or -/// `GEN ** k`), returning the exponent `k`. Both uses walk/compare exponents, -/// so the bound must name `g^k` explicitly (an element that is not a known -/// power of `g` has no usable exponent). -pub(super) fn gpow_bound(e: &Expr) -> Result { - match e { - // `g` is `x`, so the literal `2^k` IS `g^k`, and `1` is `g^0`. Rejecting - // these used to make `mul_range(1, 8)` an error although it runs exactly - // like `mul_range(1, GEN ** 3)`, and `mul_range(2, GEN ** 5)` an error - // although `2 == GEN`. It also contradicted `gaddr_of`, which reads the - // same literal as the same element. - Expr::Lit(n) if (n.is_power_of_two() && *n < (1 << 64)) || *n == 1 => Ok(n.trailing_zeros() as u64), - Expr::Gen => Ok(1), - Expr::GPow(k) => u64::try_from(*k).map_err(|_| format!("bound exponent {k} does not fit in u64")), - other => Err(format!( - "a range bound must be a power of GEN (`1`, `GEN`, or `GEN ** k`), got `{other:?}`" - )), - } -} - -pub(super) fn parse_gpow_bound(s: &str) -> Result { - gpow_bound(&parse_expr(s)?) -} - -/// Apply identifier-level **placeholder** replacements to source text before -/// parsing: each maximal run of alphanumeric characters and underscores that -/// equals a key of `replacements` is replaced by its value; other text, -/// including substrings of longer identifiers, is untouched. An empty map -/// returns the source unchanged. -pub(super) fn apply_replacements(src: &str, replacements: &BTreeMap) -> String { - if replacements.is_empty() { - return src.to_string(); - } - let mut out = String::with_capacity(src.len()); - let mut start = 0; - let flush = |out: &mut String, word: &str| { - out.push_str(replacements.get(word).map_or(word, String::as_str)); - }; - for (i, c) in src.char_indices() { - if !(c.is_alphanumeric() || c == '_') { - flush(&mut out, &src[start..i]); - out.push(c); - start = i + c.len_utf8(); - } - } - flush(&mut out, &src[start..]); - out -} diff --git a/crates/lean_compiler/src/parser/expr.rs b/crates/lean_compiler/src/parser/expr.rs deleted file mode 100644 index d46edce86..000000000 --- a/crates/lean_compiler/src/parser/expr.rs +++ /dev/null @@ -1,447 +0,0 @@ -//! Reading structure out of a line: where a line ends, where an operator sits, -//! and what an expression means. -//! -//! One rule governs all of it. **A string literal is one opaque token**, and -//! everything here that scans a line goes through [`depth0`], which skips both -//! bracketed and quoted text. Before it did, a `,` or a `]` spelled inside a -//! hint name moved an argument boundary, and [`strip_comment`]'s `#` truncated -//! the line from inside a string, in both cases producing a DIFFERENT program -//! that still parsed. -//! -//! Precedence is the usual one and is correct as it stands: `+` and `-` split -//! first, so they bind loosest; then `*`, `/`, `//` and `%`; then `**`, which -//! splits at its FIRST occurrence and recurses right, so it is -//! right-associative as in Python. There is no unary minus, since field -//! subtraction is `+`. - -use super::*; - -/// Parse an expression with `+` (lowest) then `*`, atoms being integer literals, -/// variables, calls `f(args)`, and parenthesised sub-expressions. -pub(super) fn parse_expr(s: &str) -> Result { - let s = s.trim(); - // `+` / `-` at top level (lowest precedence), left-associative. `-` is - // compile-time integer subtraction (field subtraction is `+` = XOR). - let (segs, ops) = split_add(s); - if !ops.is_empty() { - return fold_ops(&segs, &ops, |op, l, r| match op { - b'+' => Expr::Add(l, r), - _ => Expr::Sub(l, r), - }); - } - // `*`, `/`, `//`, `%` (bind tighter than `+`), skipping the two-char `**`. - let (segs, ops) = split_mul(s); - if !ops.is_empty() { - return fold_ops(&segs, &ops, |op, l, r| match op { - b'*' => Expr::Mul(l, r), - b'/' => Expr::Div(l, r), - b'd' => Expr::FieldDiv(l, r), - _ => Expr::Mod(l, r), - }); - } - // `**` (compile-time power), tightest binding: `base ** k` with `k` an - // integer literal (possibly large), or a parenthesised compile-time - // integer expression like `GEN ** (2 * s + 1)`, evaluated at lowering, - // so it can reference `unroll` counters and constants. - if let Some((base, exp)) = split_once_top(s, "**") { - let base = parse_expr(base)?; - let exp_e = parse_expr(exp)?; - return match base { - // `GEN ** k`: a compile-time integer exponent (a literal or a - // constant expression like `K_SKIP + 1`) folds to `g^k`; a runtime - // expression (e.g. an `unroll` var) becomes `GenPow`, resolved at - // lowering. - Expr::Gen => match const_int_expr(&exp_e) { - Some(k) => Ok(Expr::GPow(k)), - None => Ok(Expr::GenPow(Box::new(exp_e))), - }, - // Any other base with a compile-time exponent: square-and-multiply. - _ => Ok(Expr::Pow(Box::new(base), Box::new(exp_e))), - }; - } - // Atom. - if s.starts_with('(') && s.ends_with(')') { - return parse_expr(&s[1..s.len() - 1]); - } - if s == "GEN" { - return Ok(Expr::Gen); - } - if let Ok(n) = s.parse::() { - return Ok(Expr::Lit(n)); - } - // List literal `[a, b, …]`: an initialized StackBuf (only meaningful as - // the RHS of an assignment; top-level constant arrays are parsed earlier). - if s.starts_with('[') && s.ends_with(']') { - let inner = s[1..s.len() - 1].trim(); - if inner.is_empty() { - return Err("a list literal needs at least one element".into()); - } - return Ok(Expr::ListLit( - split_top(inner, ',') - .iter() - .map(|e| parse_expr(e)) - .collect::>()?, - )); - } - // Index `base[idx]` or slice `base[lo:hi]` (binds tightest, like a call). - if s.ends_with(']') { - let open = s.find('[').ok_or_else(|| format!("unbalanced `]` in `{s}`"))?; - let base = parse_expr(&s[..open])?; - let inner = &s[open + 1..s.len() - 1]; - if let Some((lo, hi)) = split_once_top(inner, ":") { - return Ok(Expr::Slice( - Box::new(base), - Box::new(parse_expr(lo)?), - Box::new(parse_expr(hi)?), - )); - } - let idx = parse_expr(inner)?; - return Ok(Expr::Index(Box::new(base), Box::new(idx))); - } - if let Some(open) = s.find('(') - && s.ends_with(')') - { - let name = s[..open].trim().to_string(); - let args_str = s[open + 1..s.len() - 1].trim(); - let mut args = if args_str.is_empty() { - vec![] - } else { - split_top(args_str, ',') - .iter() - .map(|a| { - if let Some((key, value)) = split_once_top(a, "=") { - let key = key.trim(); - if key.is_empty() || !key.chars().all(|c| c.is_ascii_alphanumeric() || c == '_') { - return Err(format!("invalid keyword argument `{key}`")); - } - Ok(Expr::Call(format!("__kw_{key}"), vec![parse_expr(value)?])) - } else { - parse_expr(a) - } - }) - .collect::>()? - }; - // `HeapBuf(n)` / `StackBuf(n)` are allocations, not ordinary calls. A size - // that folds as parse-time integer arithmetic (`MAXQ + 1`, constants - // already substituted) is a static size like a bare literal. - // A cell count is a frame/heap size: reject one that does not fit rather - // than wrapping it into a plausible small buffer. - let cells = |n: u128| u64::try_from(n).map_err(|_| format!("{name} size {n} does not fit in u64")); - if name == "HeapBuf" || name == "StackBuf" { - let size = match args.as_slice() { - [arg] => const_int_expr(arg), - _ => None, - }; - return match (name.as_str(), size) { - ("HeapBuf", Some(n)) => Ok(Expr::HeapBuf(cells(n)?)), - ("StackBuf", Some(n)) => Ok(Expr::StackBuf(cells(n)?)), - ("HeapBuf", None) if args.len() == 1 => Ok(Expr::HeapBufDyn(Box::new(args.pop().unwrap()))), - ("HeapBuf", None) => Err("HeapBuf(size) takes one argument".into()), - _ => Err("StackBuf(n) needs a parse-time integer size".into()), - }; - } - return Ok(Expr::Call(name, args)); - } - if s.chars().all(|c| c.is_alphanumeric() || c == '_') && !s.is_empty() { - return Ok(Expr::Var(s.to_string())); - } - if s.is_empty() { - return Err("expected an expression".into()); - } - Err(format!("cannot parse expression `{s}`")) -} - -/// Combine one tier's operands left-associatively: `node` builds the AST node -/// for each operator (as [`split_add`] / [`split_mul`] tag it). -fn fold_ops(segs: &[&str], ops: &[u8], node: impl Fn(u8, Box, Box) -> Expr) -> Result { - // An empty operand is an operator missing a side: a leading `-`, a trailing - // operator, or two in a row. Naming it beats letting `parse_expr("")` report - // an empty backtick, which is what every one of these used to say. - if let Some(i) = segs.iter().position(|seg| seg.trim().is_empty()) { - let (op, side) = if i == 0 { - (ops[0], "left") - } else { - (ops[i - 1], "right") - }; - // `split_mul` encodes the two divisions, so spell them back out. - let shown = match op { - b'/' => "//".to_string(), - b'd' => "/".to_string(), - c => (c as char).to_string(), - }; - let hint = if op == b'-' && i == 0 { - ": there is no unary minus, and field subtraction is `+`" - } else { - "" - }; - return Err(format!("`{shown}` has no {side} operand{hint}")); - } - let mut acc = parse_expr(segs[0])?; - for (&op, seg) in ops.iter().zip(&segs[1..]) { - let rhs = Box::new(parse_expr(seg)?); - acc = node(op, Box::new(acc), rhs); - } - Ok(acc) -} - -/// The bytes of `s` that sit outside every `(…)` / `[…]` group, with their -/// index: the one scanner behind all the top-level splits below. Brackets -/// themselves are never yielded, so a separator that is a bracket never splits. -pub(super) fn depth0(s: &str) -> impl Iterator + '_ { - let mut depth = 0i32; - let mut in_str = false; - s.as_bytes().iter().enumerate().filter_map(move |(i, &c)| { - // A string literal is one opaque token. Without this every splitter - // below reads the brackets and operators SPELLED INSIDE a stream name as - // structure: `hint_witness(b, "a,b")` split into three arguments. - if in_str { - in_str = c != b'"'; - return None; - } - match c { - b'"' => { - in_str = true; - None - } - b'(' | b'[' => { - depth += 1; - None - } - b')' | b']' => { - depth -= 1; - None - } - _ => (depth == 0).then_some((i, c)), - } - }) -} - -/// Split `s` at the top-level additive tier: operands and the `+` / `-` -/// operators between them. Left-associative; parenthesised/bracketed sub-terms -/// are left intact. -fn split_add(s: &str) -> (Vec<&str>, Vec) { - let (mut segs, mut ops) = (Vec::new(), Vec::new()); - let mut start = 0usize; - for (i, c) in depth0(s) { - if c == b'+' || c == b'-' { - segs.push(&s[start..i]); - ops.push(c); - start = i + 1; - } - } - segs.push(&s[start..]); - (segs, ops) -} - -/// Split `s` at the top-level multiplicative tier: the operands and the -/// operators between them (`*`, `//` for floor-division, `/` for runtime field -/// division, `%` for remainder). A `**` power is left intact (bound tighter). -/// Left-associative. -fn split_mul(s: &str) -> (Vec<&str>, Vec) { - let b = s.as_bytes(); - let (mut segs, mut ops) = (Vec::new(), Vec::new()); - let (mut start, mut next) = (0usize, 0usize); - for (i, c) in depth0(s) { - if i < next { - continue; // the second `/` of a `//`, already consumed - } - let (op, len) = match c { - b'*' if b.get(i + 1) == Some(&b'*') || (i > 0 && b[i - 1] == b'*') => continue, // `**` - b'*' => (b'*', 1), - b'/' if b.get(i + 1) == Some(&b'/') => (b'/', 2), // `//` compile-time floor-division - b'/' => (b'd', 1), // `/` runtime field division - b'%' => (b'%', 1), - _ => continue, - }; - segs.push(&s[start..i]); - ops.push(op); - next = i + len; - start = next; - } - segs.push(&s[start..]); - (segs, ops) -} - -/// Split `s` once on a top-level multi-char operator `op`. -pub(super) fn split_once_top<'a>(s: &'a str, op: &str) -> Option<(&'a str, &'a str)> { - let b = s.as_bytes(); - for (i, _) in depth0(s) { - if b[i..].starts_with(op.as_bytes()) { - return Some((&s[..i], &s[i + op.len()..])); - } - } - None -} - -/// Split `s` on every top-level occurrence of the ASCII char `sep`. -pub(super) fn split_top(s: &str, sep: char) -> Vec<&str> { - let sep = sep as u8; - let mut parts = Vec::new(); - let mut start = 0; - for (i, c) in depth0(s) { - if c == sep { - parts.push(&s[start..i]); - start = i + 1; - } - } - parts.push(&s[start..]); - parts -} - -/// Split on a top-level BARE `=`: not part of `==`, not the tail of a -/// comparison (`!=`, `<=`, `>=`), and not the tail of a compound assignment -/// ([`split_aug`] owns those, and rejects the ones this language lacks). -/// Without the last two exclusions a bare `x != y` split into a binding named -/// `x !`, which in a verifier is an `assert` that compiled to nothing. -pub(super) fn split_assign(s: &str) -> Option<(&str, &str)> { - let b = s.as_bytes(); - for (i, c) in depth0(s) { - if c != b'=' || b.get(i + 1) == Some(&b'=') { - continue; - } - if i > 0 && matches!(b[i - 1], b'=' | b'<' | b'>' | b'!' | b'+' | b'-' | b'*' | b'/' | b'%') { - continue; - } - return Some((&s[..i], &s[i + 1..])); - } - None -} - -/// A top-level augmented assignment `lhs OP= rhs` -> `(lhs, "OP", rhs)`, for -/// OP in `+ - * // %`. `Ok(None)` for a plain `=` or a comparison (`==`, `!=`, -/// `<=`, `>=`); an `Err` for a compound spelling this language does not have. -/// The operator's `=` must sit at depth 0 and be immediately preceded by -/// exactly the operator characters. -/// -/// The unsupported spellings must be REJECTED rather than declined: falling -/// through left [`split_assign`] to split the bare `=`, so `x /= 2` became a -/// binding named `x /` and the program silently kept the old `x`. -pub(super) fn split_aug(s: &str) -> Result, String> { - let b = s.as_bytes(); - for (i, c) in depth0(s) { - if c != b'=' { - continue; - } - // Nothing to the left is no assignment at all; `split_assign` and the - // binding-name check below give that its error. - if i == 0 { - return Ok(None); - } - // not `==` and not a comparison tail (`<=`, `>=`, `!=`) - if b.get(i + 1) == Some(&b'=') || matches!(b[i - 1], b'=' | b'<' | b'>' | b'!') { - return Ok(None); - } - let prev2 = b.get(i.wrapping_sub(2)).copied(); - let (op, plen): (&str, usize) = match b[i - 1] { - b'+' => ("+", 1), - b'-' => ("-", 1), - b'%' => ("%", 1), - b'*' if prev2 == Some(b'*') => { - return Err("`**=` is not supported; write `x = x ** k`".into()); - } - b'*' => ("*", 1), - b'/' if prev2 == Some(b'/') => ("//", 2), - b'/' => { - return Err( - "`/=` is not supported; `/` is runtime field division, so write `x = x / y` (or `//=` for the \ - compile-time floor division)" - .into(), - ); - } - _ => return Ok(None), - }; - return Ok(Some((s[..i - plen].trim(), op, s[i + 1..].trim()))); - } - Ok(None) -} - -/// The top-level arguments of a `name(a, b, …)` call, or `None` when `line` is -/// not one. Zero arguments come back as one empty string, as [`split_top`] -/// gives them. -pub(super) fn call_args<'a>(line: &'a str, name: &str) -> Option> { - let inner = line.trim().strip_prefix(name)?.strip_prefix('(')?.strip_suffix(')')?; - Some(split_top(inner, ',')) -} - -/// The contents of a `"…"` string literal. -pub(super) fn string_lit(s: &str) -> Option<&str> { - s.trim().strip_prefix('"')?.strip_suffix('"') -} - -/// `raw` without its trailing comment. A `#` inside a string literal is part of -/// the string: truncating there dropped the rest of the line, and the shortened -/// line usually still parsed. -pub(super) fn strip_comment(raw: &str) -> &str { - let mut in_str = false; - for (i, c) in raw.char_indices() { - match c { - '"' => in_str = !in_str, - '#' if !in_str => return &raw[..i], - _ => {} - } - } - raw -} - -/// The first top-level comparison operator in `s`. Used only on a line that -/// reached the bare-call fallback, so an `assert` or an assignment never gets -/// here and a comparison nested in a call sits at depth > 0. -pub(super) fn top_level_cmp(s: &str) -> Option<&'static str> { - let b = s.as_bytes(); - for (i, c) in depth0(s) { - let eq = b.get(i + 1) == Some(&b'='); - match c { - b'=' if eq => return Some("=="), - b'!' if eq => return Some("!="), - b'<' => return Some(if eq { "<=" } else { "<" }), - b'>' => return Some(if eq { ">=" } else { ">" }), - _ => {} - } - } - None -} - -/// A plain identifier: non-empty, starts with a letter or `_`, all -/// `[A-Za-z0-9_]` (no operators, brackets, or commas). -pub(super) fn is_ident(s: &str) -> bool { - let mut cs = s.chars(); - matches!(cs.next(), Some(c) if c.is_alphabetic() || c == '_') && s.chars().all(|c| c.is_alphanumeric() || c == '_') -} - -/// A binding or parameter name, validated. Anything else here is a mis-split -/// (`x /`, `x !`) or a top-level constant substituted into a binding position: -/// constants are replaced textually, so `V = 8` with `def scale(V)` arrives as a -/// parameter literally named `8` while the body's `V` reads the constant. -/// `zkDSL.md` §Global constants reserves the name; this enforces it. -pub(super) fn binding_name(raw: &str, what: &str) -> Result { - let n = raw.trim(); - if is_ident(n) { - return Ok(n.to_string()); - } - Err(format!( - "`{n}` is not a valid {what}: a name must be a plain identifier. A top-level constant's name is \ - reserved, and is substituted before parsing, so a parameter or local may not reuse one." - )) -} - -/// The inside of a `const(...)` wrapping the WHOLE of `s`, or `None`. -/// -/// `const(a) == b` is not one: its first `)` closes before the end, so the -/// wrapper is a subterm and the condition as a whole is an ordinary one. -pub(super) fn strip_const_wrapper(s: &str) -> Option<&str> { - let inner = s.trim().strip_prefix("const(")?.strip_suffix(')')?; - let mut depth = 0i32; - for c in inner.bytes() { - match c { - b'(' | b'[' => depth += 1, - b')' | b']' => { - depth -= 1; - if depth < 0 { - return None; - } - } - _ => {} - } - } - Some(inner) -} diff --git a/crates/lean_compiler/src/parser/subst.rs b/crates/lean_compiler/src/parser/subst.rs deleted file mode 100644 index 0affd3ed9..000000000 --- a/crates/lean_compiler/src/parser/subst.rs +++ /dev/null @@ -1,181 +0,0 @@ -//! Substituting an expression for a name, throughout a statement tree. -//! -//! This is how the two compile-time replications bind their variable: an -//! `unroll` body is re-emitted per integer with the counter substituted as a -//! literal, and a `Const` parameter is substituted into the monomorphised copy. -//! Both happen BEFORE lowering, so the result is ordinary source that no longer -//! mentions the name. -//! -//! Every arm has to be exhaustive over [`StmtKind`] and carry its fields -//! through: a field silently dropped here is a construct that loses its meaning -//! only inside an unrolled loop or a specialisation, which is the hardest place -//! to notice it. - -use super::*; - -/// Substitute `Var(name)` → `to` through a statement list, stopping at a -/// statement that rebinds `name` (later uses refer to the new binding). -/// Nested blocks recurse independently, since their bindings are branch-local, -/// matching the lowering's scoping. Used by `Const`-parameter specialization. -pub(crate) fn subst_stmts(stmts: &[Stmt], name: &str, to: &Expr) -> Vec { - let mut out = Vec::with_capacity(stmts.len()); - let mut active = true; - for s in stmts { - if !active { - out.push(s.clone()); - continue; - } - let (kind, rebinds) = subst_kind(&s.kind, name, to); - out.push(s.at(kind)); - active = !rebinds; - } - out -} - -/// Substitute one statement kind; the flag says whether it rebinds `name`. -fn subst_kind(s: &StmtKind, name: &str, to: &Expr) -> (StmtKind, bool) { - let e = |x: &Expr| subst_var(x, name, to); - match s { - StmtKind::Let(n, x) => (StmtKind::Let(n.clone(), e(x)), n == name), - StmtKind::LetTuple(ns, f, args) => ( - StmtKind::LetTuple(ns.clone(), f.clone(), args.iter().map(e).collect()), - ns.iter().any(|n| n == name), - ), - StmtKind::AssertEq(a, b) => (StmtKind::AssertEq(e(a), e(b)), false), - StmtKind::AssertNe(a, b) => (StmtKind::AssertNe(e(a), e(b)), false), - StmtKind::AssertLt(a, bound) => ( - StmtKind::AssertLt( - e(a), - match bound { - LtBound::Const(k) => LtBound::Const(*k), - LtBound::Runtime(b) => LtBound::Runtime(e(b)), - }, - ), - false, - ), - StmtKind::Call(f, args) => (StmtKind::Call(f.clone(), args.iter().map(e).collect()), false), - StmtKind::Print { label, value } => ( - StmtKind::Print { - label: label.clone(), - value: e(value), - }, - false, - ), - // Binds `name` and mentions no expression, so a substitution stops at it - // exactly as it does at any other binder. - StmtKind::LetHintWitness { name: n, stream } => ( - StmtKind::LetHintWitness { - name: n.clone(), - stream: stream.clone(), - }, - n == name, - ), - StmtKind::HintWitness { dest, name: n } => ( - StmtKind::HintWitness { - dest: e(dest), - name: n.clone(), - }, - false, - ), - StmtKind::Store(a, i, v) => (StmtKind::Store(e(a), e(i), e(v)), false), - StmtKind::Return(es) => (StmtKind::Return(es.iter().map(e).collect()), false), - StmtKind::CallIfNe(a, b, f, args) => ( - StmtKind::CallIfNe(e(a), e(b), f.clone(), args.iter().map(e).collect()), - false, - ), - StmtKind::For { var, lo, hi, body } => { - let hi = match hi { - ForBound::Const(k) => ForBound::Const(*k), - ForBound::Runtime(b) => ForBound::Runtime(e(b)), - }; - // The counter shadows `name` inside the body only. - let body = if var == name { - body.clone() - } else { - subst_stmts(body, name, to) - }; - ( - StmtKind::For { - var: var.clone(), - lo: *lo, - hi, - body, - }, - false, - ) - } - StmtKind::Unroll { var, lo, hi, body } => { - let body = if var == name { - body.clone() - } else { - subst_stmts(body, name, to) - }; - ( - StmtKind::Unroll { - var: var.clone(), - lo: e(lo), - hi: e(hi), - body, - }, - false, - ) - } - StmtKind::If { - eq, - lhs, - rhs, - then, - els, - force_const, - } => ( - StmtKind::If { - eq: *eq, - lhs: e(lhs), - rhs: e(rhs), - then: subst_stmts(then, name, to), - els: subst_stmts(els, name, to), - force_const: *force_const, - }, - false, - ), - StmtKind::Match { targets, x, arms } => ( - StmtKind::Match { - // A name target is a BINDER, so it is never substituted: the - // `shadow` flag below already stops substitution past this - // statement, and rewriting the binder itself turned `k, e = …` - // under a `Const k` into a literal target. An INDEX target is a - // use (`sb[k]` needs `k`), so it is substituted. - targets: targets - .iter() - .map(|t| if matches!(t, Expr::Var(_)) { t.clone() } else { e(t) }) - .collect(), - x: e(x), - arms: arms.iter().map(e).collect(), - }, - targets.iter().any(|t| matches!(t, Expr::Var(n) if n == name)), - ), - } -} - -/// `e` with every `Var(name)` replaced by `to`: the `match` arm -/// expansion, where the lambda parameter becomes the arm's integer literal. -pub(super) fn subst_var(e: &Expr, name: &str, to: &Expr) -> Expr { - let s = |b: &Expr| Box::new(subst_var(b, name, to)); - match e { - Expr::Var(v) if v == name => to.clone(), - Expr::Add(a, b) => Expr::Add(s(a), s(b)), - Expr::Mul(a, b) => Expr::Mul(s(a), s(b)), - Expr::Sub(a, b) => Expr::Sub(s(a), s(b)), - Expr::Div(a, b) => Expr::Div(s(a), s(b)), - Expr::FieldDiv(a, b) => Expr::FieldDiv(s(a), s(b)), - Expr::Mod(a, b) => Expr::Mod(s(a), s(b)), - Expr::Index(a, b) => Expr::Index(s(a), s(b)), - Expr::Slice(a, lo, hi) => Expr::Slice(s(a), s(lo), s(hi)), - Expr::GenPow(e) => Expr::GenPow(s(e)), - Expr::Pow(a, b) => Expr::Pow(s(a), s(b)), - Expr::HeapBufDyn(sz) => Expr::HeapBufDyn(s(sz)), - Expr::ListLit(es) => Expr::ListLit(es.iter().map(|a| subst_var(a, name, to)).collect()), - Expr::Call(f, args) => Expr::Call(f.clone(), args.iter().map(|a| subst_var(a, name, to)).collect()), - other => other.clone(), - } -} diff --git a/crates/lean_compiler/tests/programs/conditionals.py b/crates/lean_compiler/tests/programs/conditionals.py deleted file mode 100644 index c7394440b..000000000 --- a/crates/lean_compiler/tests/programs/conditionals.py +++ /dev/null @@ -1,29 +0,0 @@ -# `if` / `elif` / `else` on field equality (`==` / `!=`): one XOR and one -# conditional JUMP. Bindings made inside a branch are branch-local; branches -# communicate through write-once memory: only one branch executes, so both -# may write the same cell. The loop body's `if` runs in a helper frame (its -# own fp cell). Published: (5, 13 + 17) = (5, 28): `+` is XOR. -# public_input: 5, 28 -from snark_lib import * - - -def main(): - r = HeapBuf(4) - x = GEN ** 3 - if x == GEN ** 3: - r[1] = 5 - else: - r[1] = 7 - if x == GEN: - r[GEN] = 11 - elif x == GEN ** 3: - r[GEN] = 13 - else: - r[GEN] = 15 - for i in mul_range(1, GEN ** 4): - if i == GEN ** 2: - r[GEN ** 2] = 17 - p = GEN ** 0 - p[1] = r[1] - p[GEN] = r[GEN] + r[GEN ** 2] - return diff --git a/crates/lean_compiler/tests/programs/const_params.py b/crates/lean_compiler/tests/programs/const_params.py deleted file mode 100644 index 81a1e833c..000000000 --- a/crates/lean_compiler/tests/programs/const_params.py +++ /dev/null @@ -1,30 +0,0 @@ -# `Const` parameters: `def hash_pair(buf, k: Const)` is a template: each call -# site passes a compile-time constant and gets a monomorphized copy with `k` -# substituted as the integer literal, usable in compile-time positions (the -# slice bounds below). The direct call and match arm 0 share the k=0 -# specialization. A 256-bit BLAKE2s value occupies two canonical cells. -# Published: the two 128-bit digest cells of H(quad0, quad0) XOR H(quad1, quad1) -#: the direct k=0 digest XORed with the arm the runtime x = GEN selects (k=1). -# public_input: 252517949230448393340326710819579834691, 263897057969456650752475895236275386743 -from snark_lib import * - - -def main(): - buf = HeapBuf(4) - buf[1] = 5 - buf[GEN] = 7 - buf[GEN ** 2] = 11 - buf[GEN ** 3] = 13 - a0, a1 = hash_pair(buf, 0) - x = GEN - b0, b1 = match(log(x), range(0, 2), lambda i: hash_pair(buf, i)) - p = GEN ** 0 - p[1] = a0 + b0 - p[GEN] = a1 + b1 - return - - -def hash_pair(buf, k: Const): - h = StackBuf(2) - blake2s(buf[k * 2:k * 2 + 2], buf[k * 2:k * 2 + 2], h) - return h[0], h[1] diff --git a/crates/lean_compiler/tests/programs/fibonacci.py b/crates/lean_compiler/tests/programs/fibonacci.py deleted file mode 100644 index b672e93d3..000000000 --- a/crates/lean_compiler/tests/programs/fibonacci.py +++ /dev/null @@ -1,20 +0,0 @@ -# Fibonacci in the exponent: cell fib[g^k] holds GEN ** F_k, and the field -# product adds exponents: one MUL per Fibonacci step. The evolving state is -# carried through a HeapBuf (a mul_range body cannot capture a StackBuf). -# public_input: GEN ** 89, GEN ** 89 -from snark_lib import * - - -def main(): - fib = HeapBuf(12) - fib[1] = GEN ** 0 # F_0 = 0 - fib[GEN] = GEN # F_1 = 1 - for i in mul_range(1, GEN ** 10): - fib[i * GEN * GEN] = fib[i] * fib[i * GEN] - out = fib[GEN ** 11] - assert out == GEN ** 89 # F_11 = 89 - assert log(out) < log(GEN ** 128) - p = GEN ** 0 - p[1] = out - p[GEN] = out - return diff --git a/crates/lean_compiler/tests/programs/hash_heap_chain.py b/crates/lean_compiler/tests/programs/hash_heap_chain.py deleted file mode 100644 index 6a6950b16..000000000 --- a/crates/lean_compiler/tests/programs/hash_heap_chain.py +++ /dev/null @@ -1,21 +0,0 @@ -# Runtime slices: `buf[i:i + 2]` with a runtime g-power index `i` names the -# heap cells `buf·i·g^k`, k < 2 (one MUL folds `i` into the pointer). A BLAKE2s -# chain over heap pairs (256-bit BLAKE2s value = two canonical cells), -# addressed by the loop counter: value k sits at cells g^{2k}..g^{2k+1}, and -# value k+1 = H(value k, value k). Published: the two 128-bit digest cells of -# H^3(5, 7). -# public_input: 64347157528245356000384183465036755063, 163839818445703091465558660402169004232 -from snark_lib import * - - -def main(): - buf = HeapBuf(8) - buf[1] = 5 - buf[GEN] = 7 - for i in mul_range(1, GEN ** 3): - b = i * i # value k at cells g^{2k}..g^{2k+1} - blake2s(buf[b:b + 2], buf[b:b + 2], buf[b * GEN ** 2:b * GEN ** 2 + 2]) - p = GEN ** 0 - p[1] = buf[GEN ** 6] - p[GEN] = buf[GEN ** 7] - return diff --git a/crates/lean_compiler/tests/programs/hash_slices.py b/crates/lean_compiler/tests/programs/hash_slices.py deleted file mode 100644 index 5fdfd1184..000000000 --- a/crates/lean_compiler/tests/programs/hash_slices.py +++ /dev/null @@ -1,27 +0,0 @@ -# BLAKE2s over slices: `buf[lo:hi]` (2 cells) is a 256-bit operand under 128-bit -# machine words, with compile-time bounds: literals, literal-bound names, and -# their integer arithmetic (`x:x + 2`). Slices work on a large StackBuf (in -# place) and on a HeapBuf (bridged through the stack, one DEREF per cell), as -# inputs and as the output. Published: the two 128-bit digest cells of -# H(H(a[0:2], hb[0:2]), a[0:2]) read back from the heap. -# public_input: 249862442812096632729038560305983163980, 150628675827268462743577983046613573776 -from snark_lib import * - - -def main(): - a = StackBuf(4) - a[0] = 5 - a[1] = 7 - a[2] = 0 - a[3] = 0 - hb = HeapBuf(4) - hb[1] = 11 # heap cell g^0 - hb[GEN] = 13 # heap cell g^1 - x = 0 - h = StackBuf(2) - blake2s(a[x:x + 2], hb[0:2], h) # stack slice + heap input slice - blake2s(h, a[0:2], hb[2:4]) # digest lands in heap cells g^2, g^3 - p = GEN ** 0 - p[1] = hb[GEN ** 2] - p[GEN] = hb[GEN ** 3] - return diff --git a/crates/lean_compiler/tests/programs/heapbuf_dyn.py b/crates/lean_compiler/tests/programs/heapbuf_dyn.py deleted file mode 100644 index ad6dc5a4e..000000000 --- a/crates/lean_compiler/tests/programs/heapbuf_dyn.py +++ /dev/null @@ -1,22 +0,0 @@ -# Runtime-sized HeapBuf: the cell count is carried *in the exponent*: the -# buffer holds k cells where the size value is g^k. So a size derived from a -# runtime g-power is plain field arithmetic. Here a hinted count m = g^2 gives -# a buffer of m·m = g^4 = 4 cells; the four cells are filled from a witness -# stream and XOR-summed (`+` is XOR): 1^2^4^8 = 15. Published: (15, GEN ** 3). -# public_input: 15, GEN ** 3 -# witness m: GEN ** 2 -# witness vals: 1, 2, 4, 8 -from snark_lib import * - - -def main(): - mb = StackBuf(1) - hint_witness(mb[0:1], "m") - m = mb[0] - buf = HeapBuf(m * m) # runtime size in the exponent: g^2 · g^2 = g^4 = 4 cells - hint_witness(buf[0:4], "vals") - s = buf[1] + buf[GEN] + buf[GEN ** 2] + buf[GEN ** 3] - p = GEN ** 0 - p[1] = s - p[GEN] = GEN ** 3 - return diff --git a/crates/lean_compiler/tests/programs/hint.py b/crates/lean_compiler/tests/programs/hint.py deleted file mode 100644 index ae5d22c9e..000000000 --- a/crates/lean_compiler/tests/programs/hint.py +++ /dev/null @@ -1,28 +0,0 @@ -# `hint_witness(dest, "name")` pops the next *entry* (a slice of values) of a -# named prover stream into a StackBuf or a StackBuf/HeapBuf slice: zero -# cycles, and completely unconstrained: every hinted value below is pinned -# down by the program itself (a range check, equality asserts, an XOR -# relation). The same symbol may be hinted many times: each `# witness` line -# is one entry, and the two pops of "r" consume its two entries in order. -# Published: (GEN ** 5, 6). -# public_input: GEN ** 5, 6 -# witness r: GEN ** 5, 12 -# witness r: 9 -# witness h: 3, 5, 6 -from snark_lib import * - - -def main(): - sb = StackBuf(2) - hint_witness(sb, "r") # first "r" entry: (GEN ** 5, 12) - assert log(sb[0]) < 8 # constrain the hinted g-power - assert sb[1] == 12 - hb = HeapBuf(4) - hint_witness(hb[0:3], "h") # heap slice: the (3, 5, 6) entry - assert hb[1] + hb[GEN] == hb[GEN ** 2] # constrain: 3 + 5 = 6 (XOR) - hint_witness(hb[3:4], "r") # second "r" entry: (9) - assert hb[GEN ** 3] == 9 - p = GEN ** 0 - p[1] = sb[0] - p[GEN] = hb[GEN ** 2] - return diff --git a/crates/lean_compiler/tests/programs/identities.py b/crates/lean_compiler/tests/programs/identities.py deleted file mode 100644 index b77563651..000000000 --- a/crates/lean_compiler/tests/programs/identities.py +++ /dev/null @@ -1,13 +0,0 @@ -# Field identities, checked entirely in-program: no `# public_input:` -# annotation, so the harness runs it with the empty public input (two zero -# field elements) and nothing is published. -from snark_lib import * - - -def main(): - x = GEN * GEN - assert x == GEN ** 2 - assert log(x) < 3 - y = x + x # + is XOR: anything plus itself vanishes - assert y == 0 - return diff --git a/crates/lean_compiler/tests/programs/match.py b/crates/lean_compiler/tests/programs/match.py deleted file mode 100644 index 1aec4271f..000000000 --- a/crates/lean_compiler/tests/programs/match.py +++ /dev/null @@ -1,24 +0,0 @@ -# `match(log(x), range(a, b), lambda j: …)`: x = GEN ** j runs arm j. Dispatch is -# two jumps through a trampoline table in the bytecode, landing on the j-th -# two-instruction slot (SET the block address; JUMP to it). Arms must cover -# consecutive integers from 0, and a hinted scrutinee must be range-checked -# first. Published: (21, 5 + 7 + 9) = (21, 11): `+` is XOR. -# public_input: 21, 11 -from snark_lib import * - -FIRST = [11, 17, 21, 27, 31, 37] -SECOND = [5, 7, 9] - - -def main(): - r = HeapBuf(4) - x = GEN ** 2 - v = match(log(x), range(0, 6), lambda j: FIRST[j]) - r[1] = v - for i in mul_range(1, GEN ** 3): - w = match(log(i), range(0, 3), lambda j: SECOND[j]) - r[i * GEN] = w - p = GEN ** 0 - p[1] = r[1] - p[GEN] = r[GEN] + r[GEN ** 2] + r[GEN ** 3] - return diff --git a/crates/lean_compiler/tests/programs/match_arms.py b/crates/lean_compiler/tests/programs/match_arms.py deleted file mode 100644 index c4205f3cf..000000000 --- a/crates/lean_compiler/tests/programs/match_arms.py +++ /dev/null @@ -1,27 +0,0 @@ -# `match(log(x), range(a, b), lambda i: …, …)`: a match with generated -# arms: arm j is the lambda body with i replaced by the integer literal j, and -# every arm writes its results into the same fresh cells (write-once: exactly -# one arm executes), bound to the assignment targets. Ranges are contiguous -# from 0. With x = GEN ** 3: shift(3) = 3·g = 6, and the second pair's -# two(3) = (3, 3·g) = (3, 6), so a + b = 3 + 6 = 5 (`+` is XOR). -# public_input: 6, 5 -from snark_lib import * - - -def main(): - x = GEN ** 3 - r = match(log(x), range(0, 6), lambda i: shift(i)) - assert r == 6 - a, b = match(log(x), range(0, 2), lambda i: two(1), range(2, 6), lambda i: two(i)) - p = GEN ** 0 - p[1] = r - p[GEN] = a + b - return - - -def shift(v): - return v * GEN - - -def two(v): - return v, v * GEN diff --git a/crates/lean_compiler/tests/programs/nested.py b/crates/lean_compiler/tests/programs/nested.py deleted file mode 100644 index 0f7fe153a..000000000 --- a/crates/lean_compiler/tests/programs/nested.py +++ /dev/null @@ -1,37 +0,0 @@ -# Deep nesting across frames: a mul_range loop whose helper body range-checks the -# counter, dispatches on it, calls a recursive function from one arm (five frames -# deep, with its base-case `return` inside an `if` branch), and carries a runtime -# branch in the SAME frame as the dispatch, so self_fp and the hoisted caches are -# shared between the two. The branch is TAKEN, so the published values come from -# its fall-through path and a lowering that ignored the condition would show up -# here; `conditionals.py` covers the other polarity. -# geom(1) = 1 + g + g² + g³ + g⁴ = 31. Published: (31 + 5, 9) = (26, 9). -# public_input: 26, 9 -from snark_lib import * - -TAIL = [0, 5, 9] - - -def main(): - acc = HeapBuf(6) - for i in mul_range(1, GEN ** 3): - assert log(i) < 3 - v = match(log(i), range(0, 1), lambda j: geom(1), range(1, 3), lambda j: TAIL[j]) - # TAKEN on every iteration, so the published values ride the fall-through - # path: an `if` whose condition is never true rides the jump instead and - # stops detecting a lowering that ignores the condition. - if i == i * GEN ** 0: - acc[i] = v - else: - acc[i] = 0 - p = GEN ** 0 - p[1] = acc[1] + acc[GEN] - p[GEN] = acc[GEN ** 2] - return - - -def geom(x): - if x == GEN ** 4: - return x # early return from inside the branch - y = geom(x * GEN) - return x + y diff --git a/crates/lean_compiler/tests/programs/runtime_loop.py b/crates/lean_compiler/tests/programs/runtime_loop.py deleted file mode 100644 index de70d5811..000000000 --- a/crates/lean_compiler/tests/programs/runtime_loop.py +++ /dev/null @@ -1,32 +0,0 @@ -# A runtime mul_range stop bound: the loop walks ×GEN from the start element -# until it reaches a *runtime* g-power: here a hinted count, range-checked -# first (an unreachable bound would never terminate, so bounding the log is -# the program's duty). Repeated squaring: buf[g^k] holds g^{2^k}, so after -# n = 5 iterations buf[n] = g^32. The second loop's hinted bound equals its -# start: zero iterations, its impossible assert never runs. -# Published: (g^32, g^5). -# public_input: GEN ** 32, GEN ** 5 -# witness n: GEN ** 5 -# witness m: 1 -from snark_lib import * - - -def main(): - nb = StackBuf(1) - hint_witness(nb[0:1], "n") - n = nb[0] - assert log(n) < 16 - buf = HeapBuf(40) - buf[1] = GEN - for i in mul_range(1, n): - buf[i * GEN] = buf[i] * buf[i] - mb = StackBuf(1) - hint_witness(mb[0:1], "m") - m = mb[0] - assert log(m) < 16 - for j in mul_range(1, m): - assert 1 == 0 # empty runtime range: never entered - p = GEN ** 0 - p[1] = buf[n] - p[GEN] = n - return diff --git a/crates/lean_compiler/tests/programs/scoping.py b/crates/lean_compiler/tests/programs/scoping.py deleted file mode 100644 index 053a82e98..000000000 --- a/crates/lean_compiler/tests/programs/scoping.py +++ /dev/null @@ -1,33 +0,0 @@ -# Pins the scoping semantics: bindings (and compile-time index constants) -# made inside a branch are local to it, and the lazily-cached range-check -# constant cells revert at the join. The not-taken branch below materializes -# a bound-16 cell that must NOT leak to the check after the join: a leak -# would read an unwritten cell and fail witness generation loudly. -# Published: (9, 6). -# public_input: 9, 6 -from snark_lib import * - - -def main(): - x = GEN ** 3 - assert log(x) < 8 # bound-8 cell cached in main - v = 5 - k = 2 - if x == GEN ** 3: - v = 7 # branch-local rebinding - assert v == 7 - assert log(x) < 8 # reuses the pre-branch bound-8 cell - assert v == 5 # the outer binding is untouched at the join - if x != GEN ** 3: - k = x # (not taken) kills k's const-ness: locally only - assert log(x) < 16 # (not taken) caches bound-16 inside the branch - sb = StackBuf(4) - sb[k] = 9 # k is still the compile-time 2 - assert log(x) < 16 # must re-materialize its bound cell after the join - y = 3 - y = y * GEN # rebinding reads the old binding: 3·g = 6 - assert y == 6 - p = GEN ** 0 - p[1] = sb[2] - p[GEN] = y - return diff --git a/crates/lean_compiler/tests/programs/unroll.py b/crates/lean_compiler/tests/programs/unroll.py deleted file mode 100644 index 84c622964..000000000 --- a/crates/lean_compiler/tests/programs/unroll.py +++ /dev/null @@ -1,31 +0,0 @@ -# `for i in unroll(a, b)` replicates the body at compile time, i substituted -# as the integer literal of each iteration: zero loop overhead (no call, no -# frame, no counter). Bounds are compile-time integers, including Const -# parameters: `chain(buf, 3)` specializes and unrolls three BLAKE2s steps over -# heap slices indexed by `i` (a 256-bit BLAKE2s value is two canonical cells). -# Published: the two 128-bit digest cells of H^3(5, 7): same chain as -# hash_heap_chain.py, unrolled instead of looped. -# public_input: 64347157528245356000384183465036755063, 163839818445703091465558660402169004232 -from snark_lib import * - - -def main(): - sb = StackBuf(8) - sb[0] = 1 - for i in unroll(0, 7): - sb[i + 1] = sb[i] * GEN # sb[k] = g^k - assert sb[7] == GEN ** 7 - buf = HeapBuf(8) - buf[1] = 5 - buf[GEN] = 7 - chain(buf, 3) - p = GEN ** 0 - p[1] = buf[GEN ** 6] - p[GEN] = buf[GEN ** 7] - return - - -def chain(buf, n: Const): - for i in unroll(0, n): - blake2s(buf[i * 2:i * 2 + 2], buf[i * 2:i * 2 + 2], buf[i * 2 + 2:i * 2 + 4]) - return diff --git a/crates/lean_compiler/tests/programs/wots_walk.py b/crates/lean_compiler/tests/programs/wots_walk.py deleted file mode 100644 index 5558b2aa7..000000000 --- a/crates/lean_compiler/tests/programs/wots_walk.py +++ /dev/null @@ -1,38 +0,0 @@ -# A miniature WOTS-style chain walk bundling the DSL's moving parts: a -# runtime digit is range-checked (dispatch soundness), then match -# dispatches it to a Const-specialized walker whose BLAKE2s chain is unrolled -# over heap slices (a 256-bit BLAKE2s value occupies two canonical cells); -# the walker also builds g^{2n} at runtime (unrolled MULs) to read its final -# pair back through g-power indexing. The recomputation at the end lands on an -# already-written StackBuf pair, so write-once turns the hash into a digest -# assertion; the dead `if` branch holds an impossible assert that must never -# execute. Published: the two 128-bit digest cells of H^2(5, 7). -# public_input: 218111983282286173876109193675516368367, 307986319416510097496844621979881208676 -from snark_lib import * - - -def main(): - buf = HeapBuf(16) - buf[1] = 5 - buf[GEN] = 7 - d = GEN ** 2 # the runtime digit - assert log(d) < 4 # bound the scrutinee before dispatching on it - t0, t1 = match(log(d), range(0, 4), lambda i: walk(buf, i)) - if d != GEN ** 2: - assert 1 == 0 # dead branch: never executes - v = StackBuf(2) - v[0] = t0 - v[1] = t1 - blake2s(buf[2:4], buf[2:4], v) # recompute H(value1, value1): asserts v[0:2] == (t0, t1) - p = GEN ** 0 - p[1] = t0 - p[GEN] = t1 - return - - -def walk(buf, n: Const): - p = 1 - for i in unroll(0, n): - blake2s(buf[i * 2:i * 2 + 2], buf[i * 2:i * 2 + 2], buf[i * 2 + 2:i * 2 + 4]) - p = p * GEN * GEN - return buf[p], buf[p * GEN] diff --git a/crates/lean_compiler/tests/suite/assert_ne.rs b/crates/lean_compiler/tests/suite/assert_ne.rs deleted file mode 100644 index 16a036b43..000000000 --- a/crates/lean_compiler/tests/suite/assert_ne.rs +++ /dev/null @@ -1,200 +0,0 @@ -//! `assert a != b`: a proof-enforced inequality. It lowers to `XOR x = a + b`, -//! a hinted `inv = x⁻¹`, `MUL p = x·inv` and `SET p = 1`, the write-once -//! conflict on `p` being the assertion. Sound whatever the hint: `x = 0` forces -//! `p = 0`, which cannot then be set to `1`. Three rows and no `JUMP`. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{Op, prove, verify}; -use primitives::field::{F64, F192, g_pow}; - -/// Honest inequality over runtime values: prove + verify pass, and corrupting -/// the public output is still caught (the assert does not disturb the trace). -#[test] -fn assert_ne_end_to_end() { - let src = "\ -def main(): - x = GEN ** 5 - y = GEN ** 7 - z = x * y - assert z != x - assert z != y - p = 1 - p[1] = z - p[GEN] = x - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(g_pow(12)), F192::from(g_pow(5))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("inequality program verifies"); - - let bad = [F192::from(g_pow(11)), F192::from(g_pow(5))]; - assert!( - verify(&program, &bad, &proof).is_err(), - "wrong public input must be rejected" - ); -} - -/// The adversarial case: two hinted cells the prover sets *equal*, asserted -/// unequal. Honest witness (distinct) verifies; the equal witness leaves -/// `p = 0·inv = 0`, so `SET p = 1` conflicts and no valid proof continues. No -/// inverse hint can rescue it, which is the whole soundness argument. -#[test] -fn assert_ne_runtime_equal_rejected() { - let src = "\ -def main(): - v = StackBuf(2) - hint_witness(v[0:2], \"vals\") - assert v[0] != v[1] - p = 1 - p[1] = v[0] - p[GEN] = v[1] - return -"; - let run = |a: F64, b: F64| -> bool { - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("vals", vec![vec![F192::from(a), F192::from(b)]]); - let pi = [F192::from(a), F192::from(b)]; - prove(&program, pi, lean_vm::pcs::TEST_LOG_INV_RATE) - .is_ok_and(|(proof, _)| verify(&program, &pi, &proof).is_ok()) - }; - assert!(run(g_pow(3), g_pow(5)), "distinct hints must verify"); - assert!(!run(g_pow(3), g_pow(3)), "equal hints must be rejected by `assert !=`"); -} - -/// `assert a != b` inside a `mul_range` body: the check is emitted once per -/// compiled body and runs on every iteration, each of which differs from the -/// fixed value, so the honest loop verifies. -#[test] -fn assert_ne_in_loop() { - let src = "\ -def main(): - c = GEN ** 9 - for i in mul_range(1, GEN ** 6): - assert i != c - p = 1 - p[1] = 5 - p[GEN] = 7 - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(F64(5)), F192::from(F64(7))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("loop inequality verifies"); -} - -/// A compile-time-equal literal pair (e.g. after `Const`-arg substitution) is a -/// hard compile error: the assertion could never hold, so it is caught early. -#[test] -#[should_panic(expected = "compile-time-equal")] -fn assert_ne_compile_time_equal_rejected() { - let src = "def main():\n assert 5 != 5\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// Opcode counts of a body with and without one `assert !=`. -fn opcode_delta(with: &str, without: &str) -> (i64, i64, i64) { - let count = |src: &str| { - let p = compile(&parse(src).expect("parse")); - let (mut xor, mut mul, mut jump) = (0i64, 0i64, 0i64); - for op in &p.prog { - match op { - Op::Xor { .. } => xor += 1, - Op::Mul { .. } => mul += 1, - Op::Jump { .. } => jump += 1, - _ => {} - } - } - (xor, mul, jump) - }; - let (a, b) = (count(with), count(without)); - (a.0 - b.0, a.1 - b.1, a.2 - b.2) -} - -/// The check costs one `XOR`, one `MUL` and, above all, no `JUMP`: a -/// branch-based lowering would put an `E`-valued condition back on the one table -/// that carries constraints. The `SET` that closes the check is not counted, -/// constant materialisation elsewhere moving with the frame layout. -#[test] -fn assert_ne_emits_no_jump() { - let body = |extra: &str| { - format!( - "\ -def main(): - x = GEN ** 5 - y = GEN ** 7 -{extra} p = 1 - p[1] = x - p[GEN] = y - return -" - ) - }; - let delta = opcode_delta(&body(" assert x != y\n"), &body("")); - assert_eq!(delta, (1, 1, 0), "one XOR, one MUL, no JUMP"); -} - -/// The check survives cell sharing. `SET p = 1` writes a constant another cell -/// may already hold, and `MUL p = x·inv` a product that could otherwise be -/// shared; both are kept because `p` is written twice, which is what makes the -/// write-once conflict the assertion. Dropping either would delete the check -/// silently, so this pins it: the same product exists elsewhere in the frame, -/// and a `1` is already live. -#[test] -fn assert_ne_survives_cell_sharing() { - let src = "\ -def main(): - one = 1 - v = StackBuf(2) - hint_witness(v[0:2], \"vals\") - d = v[0] + v[1] - spare = d * one - assert v[0] != v[1] - p = 1 - p[1] = spare - p[GEN] = one - return -"; - let run = |a: F64, b: F64| -> bool { - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("vals", vec![vec![F192::from(a), F192::from(b)]]); - let pi = [F192::from(a) + F192::from(b), F192::ONE]; - prove(&program, pi, lean_vm::pcs::TEST_LOG_INV_RATE) - .is_ok_and(|(proof, _)| verify(&program, &pi, &proof).is_ok()) - }; - assert!(run(g_pow(3), g_pow(5)), "distinct hints must verify"); - assert!( - !run(g_pow(4), g_pow(4)), - "equal hints must be rejected even with a live `1` and a shareable product" - ); -} - -/// The inverse is prover advice, so the guest-level idiom must reject a wrong -/// one. Written out by hand here, the way a guest would if it hinted its own -/// inverse: only `inv = (a+b)⁻¹` makes the product `1`. -#[test] -fn assert_ne_wrong_inverse_hint_rejected() { - let src = "\ -def main(): - v = StackBuf(3) - hint_witness(v[0:3], \"vals\") - d = v[0] + v[1] - prod = d * v[2] - assert prod == 1 - p = 1 - p[1] = v[0] - p[GEN] = v[1] - return -"; - let run = |a: F192, b: F192, inv: F192| -> bool { - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("vals", vec![vec![a, b, inv]]); - prove(&program, [a, b], lean_vm::pcs::TEST_LOG_INV_RATE) - .is_ok_and(|(proof, _)| verify(&program, &[a, b], &proof).is_ok()) - }; - let (a, b) = (F192::from(g_pow(3)), F192::from(g_pow(5))); - let d = a + b; - assert!(run(a, b, d.inv()), "the true inverse verifies"); - assert!(!run(a, b, d.inv() + F192::ONE), "a wrong inverse must be rejected"); - assert!(!run(a, a, F192::ONE), "equal sides admit no inverse at all"); -} diff --git a/crates/lean_compiler/tests/suite/common/mod.rs b/crates/lean_compiler/tests/suite/common/mod.rs deleted file mode 100644 index ee558f2b9..000000000 --- a/crates/lean_compiler/tests/suite/common/mod.rs +++ /dev/null @@ -1,32 +0,0 @@ -//! Helpers shared by the integration tests, reached as `crate::common::…`. -#![allow(dead_code)] - -use lean_compiler::{compile_without_filler, parse}; -use primitives::field::F192; - -/// The program's own instruction mix: a build without the fill blocks, executed but not -/// proven. Proving needs them, since a table's height has to be a power of two with no -/// padding rows, but their dummy rows would drown out exactly what these counts are -/// measuring. -pub fn mix(src: &str, pi: [F192; 2]) -> [usize; lean_vm::cpu::Stats::TABLES.len()] { - compile_without_filler(&parse(src).expect("parse")) - .execute(pi) - .unwrap() - .base_counts -} - -/// An AST's shape with source lines stripped. Two spellings of the same program -/// (a constant against its substituted value, a placeholder against the filled -/// text) are the same program but rarely occupy the same lines, so comparing -/// them by `Debug` has to ignore that field. -pub fn without_lines(ast: &lean_compiler::Ast) -> String { - let d = format!("{ast:?}"); - let (mut out, mut rest) = (String::with_capacity(d.len()), d.as_str()); - while let Some(i) = rest.find("line: ") { - out.push_str(&rest[..i]); - let after = &rest[i + "line: ".len()..]; - rest = &after[after.find(", ").expect("`line` is followed by another field") + 2..]; - } - out.push_str(rest); - out -} diff --git a/crates/lean_compiler/tests/suite/const_placeholder.rs b/crates/lean_compiler/tests/suite/const_placeholder.rs deleted file mode 100644 index df7e0c19c..000000000 --- a/crates/lean_compiler/tests/suite/const_placeholder.rs +++ /dev/null @@ -1,322 +0,0 @@ -//! Global constants and compile-time placeholders in the zkDSL. -//! -//! A top-level `NAME = ` is a **global constant**: it is evaluated -//! to its field value and substituted (as one literal) everywhere its name -//! appears below: so a constant is usable in every position a literal is, -//! including `StackBuf`/`HeapBuf` sizes, `**` exponents, and `assert log _ < _` -//! bounds. A **placeholder** is any identifier text-replaced before parsing via -//! [`parse_with_replacements`]; the idiom is a placeholder feeding a constant -//! (`V = V_PLACEHOLDER` with `"V_PLACEHOLDER" ↦ "128"`), as in leanVM. - -use std::collections::BTreeMap; - -use lean_compiler::{compile, parse, parse_with_replacements}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::g_pow; - -/// A global constant substitutes exactly like writing its value inline: even -/// in a `StackBuf` size, which demands a parse-time literal. The two programs -/// produce identical ASTs. -#[test] -fn const_inlines_like_literal() { - let with_const = "\ -N = 5 - -def main(): - a = StackBuf(N) - a[0] = N - a[1] = N + 2 - assert a[0] == 5 - return -"; - let inlined = "\ -def main(): - a = StackBuf(5) - a[0] = 5 - a[1] = 5 + 2 - assert a[0] == 5 - return -"; - let ac = parse(with_const).expect("const program parses"); - let ai = parse(inlined).expect("inlined program parses"); - assert_eq!( - crate::common::without_lines(&ac), - crate::common::without_lines(&ai), - "constant must inline to its value" - ); - let _ = compile(&ac); // and it lowers to a real program -} - -/// A constant may be used as a `**` exponent and an `assert log _ < _` bound - -/// positions that previously required a bare integer literal. -#[test] -fn const_in_literal_only_positions() { - let src = "\ -LEN = 3 -BOUND = 8 - -def main(): - x = GEN ** LEN - assert log x < BOUND - return -"; - let inlined = "\ -def main(): - x = GEN ** 3 - assert log x < 8 - return -"; - assert_eq!( - crate::common::without_lines(&parse(src).unwrap()), - crate::common::without_lines(&parse(inlined).unwrap()), - ); - let _ = compile(&parse(src).unwrap()); -} - -/// A global constant may be a g-power, which is how the ISA writes every -/// address and index. -/// -/// The scalar path tried an `f192` literal, then an integer expression, and -/// stopped, so `GEN ** 2` was rejected as "not a compile-time integer constant -/// expression" while `f192(4, 0, 0)` naming the same element was accepted. It -/// now falls back to the field evaluator and renders the value as a decimal -/// wherever it fits the low two limbs, so the constant still works in the -/// positions that demand a literal rather than only as a value. -#[test] -fn a_global_constant_may_be_a_g_power() { - for (decl, exp) in [("GEN ** 2", 2usize), ("GEN * GEN", 2), ("GEN ** 70", 70)] { - let src = format!( - "STEP = {decl} - -def main(): - p = GEN ** 0 - p[1] = STEP - p[GEN] = GEN ** 0 - return -" - ); - let program = compile(&parse(&src).unwrap_or_else(|e| panic!("`{decl}`: {e}"))); - let want = [g_pow(exp).into(), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).unwrap_or_else(|e| panic!("`{decl}` is not g^{exp}: {e:?}")); - } -} - -/// A constant may reference an earlier constant (chaining). `B = A` gives `B` -/// the value of `A`; both are usable as sizes. -#[test] -fn const_chains() { - let src = "\ -A = 4 -B = A - -def main(): - p = StackBuf(A) - q = StackBuf(B) - return -"; - let inlined = "\ -def main(): - p = StackBuf(4) - q = StackBuf(4) - return -"; - assert_eq!( - crate::common::without_lines(&parse(src).unwrap()), - crate::common::without_lines(&parse(inlined).unwrap()), - ); - let _ = compile(&parse(src).unwrap()); -} - -/// Constant expressions use **integer** arithmetic (`+ - * / **`), not runtime -/// field arithmetic, so derived sizes/counts come out right. Filled -/// via placeholders, the whole set of derivations resolves to plain literals. -#[test] -fn const_integer_arithmetic_derivations() { - let templated = "\ -V = V_PLACEHOLDER -W = W_PLACEHOLDER -LOG_LIFETIME = LOG_LIFETIME_PLACEHOLDER -CHAIN_STEPS = W - 1 -N_TWEAK_WORDS = 2 + CHAIN_STEPS * V + LOG_LIFETIME -N_TWEAK_BLOCKS = N_TWEAK_WORDS / 2 -FIXED_BLOCKS = 1 + N_TWEAK_BLOCKS + LOG_LIFETIME / 2 -FIXED_BYTES = FIXED_BLOCKS * 32 -N_SIGS_BOUND = 2 ** 16 - -def main(): - a = StackBuf(N_TWEAK_WORDS) - x = GEN ** FIXED_BYTES - assert log x < N_SIGS_BOUND - for i in unroll(0, N_TWEAK_BLOCKS): - assert a[0] == W - return -"; - // V = 42, W = 8, LOG_LIFETIME = 32 → the standard XMSS instance. - let mut repl = BTreeMap::new(); - repl.insert("V_PLACEHOLDER".to_string(), "42".to_string()); - repl.insert("W_PLACEHOLDER".to_string(), "8".to_string()); - repl.insert("LOG_LIFETIME_PLACEHOLDER".to_string(), "32".to_string()); - let filled = parse_with_replacements(templated, &repl).expect("derivations resolve"); - - // N_TWEAK_WORDS = 2 + 7*42 + 32 = 328, N_TWEAK_BLOCKS = 164, - // FIXED_BLOCKS = 1 + 164 + 16 = 181, FIXED_BYTES = 5792, N_SIGS_BOUND = 65536. - let concrete = "\ -def main(): - a = StackBuf(328) - x = GEN ** 5792 - assert log x < 65536 - for i in unroll(0, 164): - assert a[0] == 8 - return -"; - assert_eq!( - crate::common::without_lines(&filled), - crate::common::without_lines(&parse(concrete).unwrap()) - ); - let _ = compile(&filled); -} - -/// `const(...)` is TRANSPARENT in a parse-time position, and the test is that -/// the wrapped and bare spellings parse to the same AST. -/// -/// The wrapper means "read this with integer arithmetic". A size, a count, an -/// exponent, a bound and a stack index have no other reading, so it changes -/// nothing there. It was a parse error in a `StackBuf` size, a `log` bound and a -/// top-level constant while being accepted in a `HeapBuf` size, an `unroll` count -/// and a `GEN **` exponent, which made one construct mean two things depending on -/// where it stood. -#[test] -fn const_is_transparent_where_the_reading_is_already_integer() { - for (wrapped, bare) in [ - ("s = StackBuf(const(2 + 2))", "s = StackBuf(4)"), - ("h = HeapBuf(const(2 + 2))", "h = HeapBuf(4)"), - ("x = GEN ** const(1 + 1)", "x = GEN ** 2"), - ] { - let src = |b: &str| format!("def main():\n {b}\n return\n"); - assert_eq!( - crate::common::without_lines(&parse(&src(wrapped)).unwrap_or_else(|e| panic!("{wrapped}: {e}"))), - crate::common::without_lines(&parse(&src(bare)).expect("bare")), - "`{wrapped}` must parse as `{bare}`" - ); - } - // A `log` bound and an `unroll` count, which are their own parse paths. - let bound = |b: &str| format!("def main():\n v = GEN ** 2\n assert log v < {b}\n return\n"); - assert_eq!( - crate::common::without_lines(&parse(&bound("const(4 + 4)")).expect("wrapped bound")), - crate::common::without_lines(&parse(&bound("8")).expect("bare bound")), - ); - // An `unroll` count keeps its expression for the lowerer to fold, in either - // spelling, so the baseline is the unwrapped expression rather than a literal. - let count = |b: &str| format!("def main():\n for i in unroll(0, {b}):\n v = 1\n return\n"); - let unrolled = |b: &str| compile(&parse(&count(b)).unwrap_or_else(|e| panic!("{b}: {e}"))).code_len(); - assert_eq!( - unrolled("const(1 + 1)"), - unrolled("1 + 1"), - "the count must fold the same" - ); - assert_eq!(unrolled("const(1 + 1)"), unrolled("2")); - // And a global constant, where the whole declaration is already integer. - assert_eq!( - crate::common::without_lines( - &parse("N = const(3 + 1)\n\ndef main():\n s = StackBuf(N)\n return\n").expect("wrapped") - ), - crate::common::without_lines(&parse("N = 4\n\ndef main():\n s = StackBuf(N)\n return\n").expect("bare")), - ); - // Transparent means transparent: an illegal value is still illegal, so the - // wrapper is no route past a bound the bare spelling would fail. - for b in ["0", "const(0)"] { - let ast = parse(&bound(b)).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("`{b}` was accepted as a bound"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains("empty set"), "{b}: got `{msg}`"); - } -} - -/// A placeholder is text-replaced before parsing; feeding a constant is the -/// idiom. The filled program equals the one written with the value inline. -#[test] -fn placeholder_fills_constant() { - let templated = "\ -V = V_PLACEHOLDER - -def main(): - a = StackBuf(V) - a[0] = V - assert a[0] == 7 - return -"; - let mut repl = BTreeMap::new(); - repl.insert("V_PLACEHOLDER".to_string(), "7".to_string()); - let filled = parse_with_replacements(templated, &repl).expect("placeholder fills"); - - let concrete = "\ -def main(): - a = StackBuf(7) - a[0] = 7 - assert a[0] == 7 - return -"; - assert_eq!( - crate::common::without_lines(&filled), - crate::common::without_lines(&parse(concrete).unwrap()) - ); - let _ = compile(&filled); -} - -/// Replacement is identifier-bounded: a key does not match a substring of a -/// longer identifier. -#[test] -fn placeholder_is_identifier_bounded() { - let src = "\ -def main(): - FOOBAR = 1 - x = FOOBAR - assert x == 1 - return -"; - let mut repl = BTreeMap::new(); - repl.insert("FOO".to_string(), "999".to_string()); - // `FOO` must NOT rewrite the `FOO` inside `FOOBAR`. - assert_eq!( - crate::common::without_lines(&parse_with_replacements(src, &repl).unwrap()), - crate::common::without_lines(&parse(src).unwrap()), - ); -} - -/// An unfilled placeholder (or an undeclared constant) is a clear error, and a -/// constant may not be declared twice. -#[test] -fn errors() { - let unfilled = "\ -V = V_PLACEHOLDER - -def main(): - return -"; - let err = parse(unfilled).expect_err("an unfilled placeholder must fail"); - assert!( - err.contains("V_PLACEHOLDER"), - "error should name the placeholder: {err}" - ); - - let dup = "\ -N = 1 -N = 2 - -def main(): - return -"; - assert!(parse(dup).is_err(), "a constant declared twice must fail"); - - // A top-level line that is neither a `def` nor a `NAME = value` is rejected. - let junk = "\ -1 + 1 - -def main(): - return -"; - assert!(parse(junk).is_err(), "malformed top-level line must fail"); -} diff --git a/crates/lean_compiler/tests/suite/determinism.rs b/crates/lean_compiler/tests/suite/determinism.rs deleted file mode 100644 index 68ed44ef5..000000000 --- a/crates/lean_compiler/tests/suite/determinism.rs +++ /dev/null @@ -1,107 +0,0 @@ -//! One source compiles to one program, always, and the same program it compiled -//! to yesterday. -//! -//! The bytecode digest leads the Fiat--Shamir transcript, so two builds of one -//! source that disagree are two incompatible proof systems, and the symptom is a -//! proof that stops verifying rather than a crash. -//! -//! Two different properties, and only the first is about determinism: -//! -//! * *Within a process*, compiling twice is a real perturbation rather than a -//! repeat, since `RandomState` bumps its seed once per map, so the second -//! compilation hashes with different keys than the first. -//! * *Across commits*, `GOLDEN` is a SNAPSHOT of the compiler's output. It does -//! not prove determinism (nothing iterating a hash container reaches the -//! bytecode today, and deliberately reversing the branch-output order at a join -//! moves no digest). It earns its place a different way: a codegen change that -//! was not intended shows up here and nowhere else, and every entry that moved -//! this far was a change someone then had to justify. -//! -//! So a moved digest is a question, not a chore: update `GOLDEN` in the same -//! commit and say in the message which change moved it. - -use std::collections::BTreeMap; -use std::fs; - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::Program; - -/// `tests/programs/.py` against the digest of the bytecode it compiles to. -/// The list is closed: a new program must be added here, so one cannot be added -/// without a digest. -#[rustfmt::skip] -const GOLDEN: &[(&str, &str)] = &[ - ("conditionals", "e8df2b807d3eed366d83843e207f0a9441c1e2014ee0add958a399af562a761c"), - ("const_params", "a746b339afc434c52c4c04388e0af25054eef20405a9c99a6a6d59987a37d0de"), - ("fibonacci", "1419063250d0c54fa808499fdca691748dfb91f84f1007c6f0e939d5ef6779b6"), - ("hash_heap_chain", "9d8735c8393dd6cde7e07a71277a52435ca903186c6a984161b2b8821d6b552d"), - ("hash_slices", "99b3de0af47b797b51b8f38f37b5ccb7ff64d3f5acd9ba23bdd1ff4cfc47e9c3"), - ("heapbuf_dyn", "9e570732cf9258379eec6caf1efb2bd2d8ed484b5aa8c7b7da5360d2a79103ef"), - ("hint", "d9601df070184a3f1a5702d769e3897013c17a264889161df8ae7efe0daecc13"), - ("identities", "49a2bbd6bf785786f2ce8bf8f63a57a21bb6c547eae0c0f245f20a5222ac1c7a"), - ("match", "6a7265d2ed56e512c939f76024afea98e5cd707aee4776df09c2e4a863208354"), - ("match_arms", "c1b74b466538ca9387ae43db9b86c621decf70bf3ae92e73e6634b37b5afb616"), - ("nested", "2d5c743a2692704207e1c45ea2518a504da1f3a18743a204600bc6ed04bdc87d"), - ("runtime_loop", "f33c6b5b82ed8ade198fb8978ba443554fc8714082a98dc784dbabdcee92e0fe"), - ("scoping", "c466babcc1af1deba56dda628e730d815d2fdffe065d695f9118167e0bb8669f"), - ("unroll", "08ebe1f4b51d862c6335b90694cf60d2fd2841d3a9913f6400cb53322117d309"), - ("wots_walk", "82f5dc859eec827ec2862c423fc20a2f83c687081f7fe6e37a84a414bf3ae720"), -]; - -fn digest(p: &Program) -> String { - primitives::hash::hash(format!("{:?}", p.prog).as_bytes()) - .iter() - .map(|b| format!("{b:02x}")) - .collect() -} - -/// Every program in `tests/programs/`, compiled twice. -#[test] -fn bytecode_is_reproducible() { - let dir = concat!(env!("CARGO_MANIFEST_DIR"), "/tests/programs"); - let mut paths: Vec<_> = fs::read_dir(dir) - .expect("tests/programs") - .map(|e| e.expect("dir entry").path()) - .filter(|p| p.extension().is_some_and(|x| x == "py")) - .collect(); - paths.sort(); - assert!(!paths.is_empty(), "no .py programs found"); - - let mut actual: Vec<(String, String)> = Vec::new(); - for path in &paths { - let name = path.file_stem().expect("file stem").to_string_lossy().into_owned(); - let src = fs::read_to_string(path).unwrap_or_else(|e| panic!("{name}: read: {e}")); - let one = compile(&parse(&src).unwrap_or_else(|e| panic!("{name}: parse: {e}"))); - let two = compile(&parse(&src).unwrap_or_else(|e| panic!("{name}: parse: {e}"))); - assert_eq!( - digest(&one), - digest(&two), - "{name}: two compilations of one source produced different bytecode, \ - so the compiler is reading a hash seed" - ); - actual.push((name, digest(&one))); - } - - let want: BTreeMap<&str, &str> = GOLDEN.iter().copied().collect(); - let moved: Vec<&str> = actual - .iter() - .filter(|(n, d)| want.get(n.as_str()) != Some(&d.as_str())) - .map(|(n, _)| n.as_str()) - .collect(); - let dropped: Vec<&str> = want - .keys() - .copied() - .filter(|n| !actual.iter().any(|(a, _)| a == n)) - .collect(); - assert!( - dropped.is_empty(), - "GOLDEN names a program that no longer exists: {dropped:?}" - ); - if !moved.is_empty() { - let table: String = actual - .iter() - .map(|(n, d)| format!(" (\"{n}\", \"{d}\"),\n")) - .collect(); - panic!("bytecode changed for {moved:?}\n\nif that was intended, GOLDEN is now:\n{table}"); - } -} diff --git a/crates/lean_compiler/tests/suite/disassemble.rs b/crates/lean_compiler/tests/suite/disassemble.rs deleted file mode 100644 index 462f55490..000000000 --- a/crates/lean_compiler/tests/suite/disassemble.rs +++ /dev/null @@ -1,47 +0,0 @@ -//! `disassemble` must render every one of the six opcodes without panicking, -//! so it stays usable for the `DBG_DISASM` workflow (a failed guest `assert` -//! surfaces as a write-once conflict, and the pc is all you get). - -use lean_compiler::{compile, disassemble, parse}; -use primitives::pretty_integer; - -#[test] -fn disassemble_covers_every_opcode() { - let src = "\ -@inline -def pack64x2(a, b): - assert_in_k(a, b) - return a + f192(0, 1, 0) * b - -def main(): - buff = HeapBuf(6) - buff[1] = 1 - buff[GEN] = GEN - for i in mul_range(1, GEN ** 4): - buff[i * GEN ** 2] = buff[i] * buff[i * GEN] - h = StackBuf(2) - h[0] = 5 - h[1] = 7 - d = StackBuf(2) - blake2s(h, h, d) - packed = pack64x2(5, 7) - p = 1 - p[1] = buff[GEN ** 4] + packed - p[GEN] = d[0] - return -"; - - let program = compile(&parse(src).expect("parse")); - - println!("\n=== zkDSL source ===\n{src}"); - println!( - "=== compiled ISA ({} instructions) ===", - pretty_integer(program.prog.len()) - ); - let text = disassemble(&program.prog); - print!("{text}"); - - for mnemonic in ["SET", "XOR", "MUL", "DEREF", "JUMP", "BLAKE2S"] { - assert!(text.contains(mnemonic), "disassembly is missing {mnemonic}"); - } -} diff --git a/crates/lean_compiler/tests/suite/field_div.rs b/crates/lean_compiler/tests/suite/field_div.rs deleted file mode 100644 index f2166b8dd..000000000 --- a/crates/lean_compiler/tests/suite/field_div.rs +++ /dev/null @@ -1,85 +0,0 @@ -//! Field division `a / b` (single slash): a runtime `a · b⁻¹`, distinct from -//! the compile-time floor-division `//`. It lowers to a single `MUL` whose -//! quotient operand is left unset: the write-once back-solve fills it with -//! `a · b⁻¹` and the `MUL` constraint `quotient · b == a` binds it, with no -//! prover hint (the same back-solve the range-check gadget already uses). A -//! zero divisor is rejected (the back-solve cannot invert 0). - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::{F64, F192, g_pow}; - -/// `a / b` and `1 / b` over runtime values: the quotient satisfies `q·b == a`, -/// checked by publishing it and reproducing the dividend. -#[test] -fn field_div_end_to_end() { - let src = "\ -def main(): - a = GEN ** 20 - b = GEN ** 7 - q = a / b - r = 1 / b - p = 1 - p[1] = q * b - p[GEN] = r * b - return -"; - let program = compile(&parse(src).expect("parse")); - // q·b must reproduce a = g^20; r·b must be 1. - let want = [F192::from(g_pow(20)), F192::from(F64::ONE)]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("division program verifies"); - - let bad = [F192::from(g_pow(21)), F192::from(F64::ONE)]; - assert!( - verify(&program, &bad, &proof).is_err(), - "wrong quotient product rejected" - ); -} - -/// `//` stays compile-time floor division (an index), `/` is the runtime field -/// op: the two must not collide. Here `8 // 2 == 4` picks a stack slot while -/// `x / y` is a field quotient. -#[test] -fn field_div_vs_floordiv() { - let src = "\ -def main(): - x = GEN ** 6 - q = x / (GEN ** 2) - z = GEN ** (6 // 2) - p = 1 - p[1] = q - p[GEN] = z - return -"; - let program = compile(&parse(src).expect("parse")); - // q = g^6 / g^2 = g^4 (runtime `/`); z = g^(6//2) = g^3 (compile-time `//`). - let want = [F192::from(g_pow(4)), F192::from(g_pow(3))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("mixed //-and-/ program verifies"); -} - -/// A zero divisor: the prover hints `b = 0`, and `1 / b` cannot be back-solved -/// (`1 = q·0` has no solution), so witness generation / verification rejects. -#[test] -fn field_div_by_zero_rejected() { - let src = "\ -def main(): - v = StackBuf(1) - hint_witness(v[0:1], \"den\") - r = 1 / v[0] - p = 1 - p[1] = r * v[0] - p[GEN] = 1 - return -"; - let run = |den: F64| -> bool { - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("den", vec![vec![F192::from(den)]]); - let pi = [F192::from(F64::ONE), F192::from(F64::ONE)]; - prove(&program, pi, lean_vm::pcs::TEST_LOG_INV_RATE) - .is_ok_and(|(proof, _)| verify(&program, &pi, &proof).is_ok()) - }; - assert!(run(g_pow(4)), "nonzero divisor must verify"); - assert!(!run(F64::ZERO), "zero divisor must be rejected"); -} diff --git a/crates/lean_compiler/tests/suite/field_towers.rs b/crates/lean_compiler/tests/suite/field_towers.rs deleted file mode 100644 index 2988ed2e3..000000000 --- a/crates/lean_compiler/tests/suite/field_towers.rs +++ /dev/null @@ -1,35 +0,0 @@ -//! Random cross-check of F192 = GF((2^64)^3) against its portable reference, -//! through the same `primitives` re-export path the VM uses. `primitives` -//! itself only pins the dispatched `Mul` against `software::mul` on a handful -//! of fixed Python-generated vectors (plus 10k random inputs on aarch64), so on -//! a pclmulqdq x86 host this is the random-input check on that dispatch. - -use primitives::field::{F192, F192Unreduced}; -use rand::Rng; - -fn rand_f192(rng: &mut impl Rng) -> F192 { - F192::new(rng.random(), rng.random(), rng.random()) -} - -#[test] -fn f192_field_behaviour() { - let mut rng = rand::rng(); - for _ in 0..500 { - let (a, b, c) = (rand_f192(&mut rng), rand_f192(&mut rng), rand_f192(&mut rng)); - // ring axioms + agreement with the portable reference - assert_eq!(a * b, primitives::field::gf2_64x3::software::mul(a, b)); - assert_eq!(a * b, b * a); - assert_eq!((a * b) * c, a * (b * c)); - assert_eq!(a * (b + c), a * b + a * c); - assert_eq!(a.square(), a * a); - if !a.is_zero() { - assert_eq!(a * a.inv(), F192::ONE); - } - let mut acc = F192Unreduced::ZERO; - acc ^= a.mul_unreduced(b); - acc ^= a.mul_unreduced(c); - assert_eq!(acc.reduce(), a * b + a * c); - } - // y^3 = y + 1 (the defining relation) - assert_eq!(F192::Y * F192::Y * F192::Y, F192::Y + F192::ONE); -} diff --git a/crates/lean_compiler/tests/suite/filler.rs b/crates/lean_compiler/tests/suite/filler.rs deleted file mode 100644 index c67139c40..000000000 --- a/crates/lean_compiler/tests/suite/filler.rs +++ /dev/null @@ -1,58 +0,0 @@ -//! Fill blocks (`lean_compiler::filler`): extra real rows so a table's height needs no -//! padding. -//! -//! What has to hold, and is checked here end to end: -//! -//! - every table comes out an exact power of two, whatever the program's row mix, so -//! nothing is ever padded; -//! - the fill costs exactly what the solver says, a traversal being its block's rows -//! plus one jump and nothing else; -//! - the filled trace proves and verifies. -//! -//! The second is what lets the interpreter solve once the chain has halted, in one -//! interpretation of the program, so it is worth pinning rather than assuming. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::filler; -use lean_vm::cpu::{prove, verify}; -use primitives::field::F192; - -const PROGRAMS: [&str; 5] = [ - // Folds to nothing, so the fill is all there is. - "def main():\n x = GEN ** 5\n y = x * x\n return\n", - "def main():\n b = HeapBuf(4)\n b[1] = GEN\n y = b[1] * b[1]\n return\n", - "def main():\n for i in mul_range(1, GEN ** 20):\n z = i * i\n return\n", - // A compression, so BLAKE2s is non-empty too. - "def main():\n a = StackBuf(2)\n a[0] = 5\n a[1] = 7\n c = StackBuf(2)\n blake2s(a, a, c)\n return\n", - "def main():\n for i in mul_range(1, GEN ** 300):\n z = i + GEN\n return\n", -]; - -#[test] -fn every_table_lands_on_a_power_of_two() { - for src in PROGRAMS { - let program = compile(&parse(src).expect("parse")); - let pi = [F192::ZERO, F192::ZERO]; - let (proof, stats) = prove(&program, pi, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - assert!(filler::is_filled(stats.counts), "{:?} for {src:?}", stats.counts); - verify(&program, &pi, &proof).expect("a filled program verifies"); - } -} - -/// The solver's cost model is the whole reason one pass suffices, so check it against -/// what the machine did: from the program's own rows, the plan the interpreter solved -/// must predict the proven counts exactly. -#[test] -fn the_cost_model_is_exact() { - for src in PROGRAMS { - let program = compile(&parse(src).expect("parse")); - let stats = prove(&program, [F192::ZERO, F192::ZERO], lean_vm::pcs::TEST_LOG_INV_RATE) - .unwrap() - .1; - let plan = filler::solve(stats.base_counts, filler::NO_FLOORS).expect("solvable"); - assert_eq!( - filler::filled(stats.base_counts, &plan), - stats.counts, - "predicted against actual for {src:?}" - ); - } -} diff --git a/crates/lean_compiler/tests/suite/hint_log2_ceil.rs b/crates/lean_compiler/tests/suite/hint_log2_ceil.rs deleted file mode 100644 index 4d66a8c67..000000000 --- a/crates/lean_compiler/tests/suite/hint_log2_ceil.rs +++ /dev/null @@ -1,68 +0,0 @@ -//! `hint_log2_ceil(bits, nbits, floor)`: computed advice returning -//! `g^max(log2_ceil(v), floor)`, where `v` is the integer the `bits` buffer -//! decodes to. The prover fills it at witness-generation (the runtime has `v` -//! concretely); it is unconstrained on its own: `log2_ceil` re-verifies it - -//! so this test checks only that the advice computes the right value. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::{F64, F192, g_pow}; - -fn log2_ceil_of(v: u128) -> usize { - if v <= 1 { - 0 - } else { - (128 - (v - 1).leading_zeros()) as usize - } -} - -#[test] -fn log2_ceil_advice_computes_the_log() { - let src = "\ -def main(): - bits = HeapBuf(GEN ** 8) - hint_witness(bits[0:8], \"bits\") - g_mu = hint_log2_ceil(bits, 8, 0) - p = 1 - p[1] = g_mu - p[GEN] = 1 - return -"; - for v in [1u128, 2, 3, 4, 5, 7, 8, 200] { - let mut program = compile(&parse(src).expect("parse")); - let bits: Vec = (0..8).map(|j| F192::from(F64(((v >> j) & 1) as u64))).collect(); - program.set_witness("bits", vec![bits]); - let want = [F192::from(g_pow(log2_ceil_of(v))), F192::from(F64::ONE)]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).unwrap_or_else(|_| panic!("v={v}: log2_ceil advice must verify")); - let bad = [F192::from(g_pow(log2_ceil_of(v) + 1)), F192::from(F64::ONE)]; - assert!( - verify(&program, &bad, &proof).is_err(), - "v={v}: wrong g_mu must be rejected" - ); - } -} - -/// The `floor` argument: `max(log2_ceil(v), floor)`. With floor = 5, small -/// values are lifted to g^5. -#[test] -fn log2_ceil_advice_floor() { - let src = "\ -def main(): - bits = HeapBuf(GEN ** 8) - hint_witness(bits[0:8], \"bits\") - g_mu = hint_log2_ceil(bits, 8, 5) - p = 1 - p[1] = g_mu - p[GEN] = 1 - return -"; - for (v, mu) in [(2u128, 5usize), (4, 5), (64, 6), (200, 8)] { - let mut program = compile(&parse(src).expect("parse")); - let bits: Vec = (0..8).map(|j| F192::from(F64(((v >> j) & 1) as u64))).collect(); - program.set_witness("bits", vec![bits]); - let want = [F192::from(g_pow(mu)), F192::from(F64::ONE)]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).unwrap_or_else(|_| panic!("v={v}: floored log2_ceil must verify")); - } -} diff --git a/crates/lean_compiler/tests/suite/inline_expr.rs b/crates/lean_compiler/tests/suite/inline_expr.rs deleted file mode 100644 index 3a0bd768b..000000000 --- a/crates/lean_compiler/tests/suite/inline_expr.rs +++ /dev/null @@ -1,62 +0,0 @@ -//! `@inline` calls in EXPRESSION position: embedded in arithmetic, as a heap -//! store's RHS, or under further ops: must produce the same values as the -//! statement-position form. Regression test for the dropped-RetBind bug: an -//! inlined tail return of a plain var records a g-address alias, which only -//! `let`/tuple bindings used to consume; expression positions read the -//! never-written dst cell (zeros) and left the stale bind to corrupt the next -//! `let`. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::{F64, F192}; - -#[test] -fn inline_call_in_expression_positions() { - let src = "\ -@inline -def wprod(ch, n: Const, idx: Const): - # eq-tensor weight of compile-time idx over ch[0..n) - w = GEN ** 0 - for c in unroll(0, n): - cv = ch[GEN ** c] - if (idx // (2 ** c)) % 2 == 1: - w *= cv - else: - w *= (1 + cv) - return w - -def main(): - b = HeapBuf(4) - b[1] = 3 - b[GEN] = 5 - x = wprod(b, 2, 2) - y = 7 * wprod(b, 2, 1) - out = HeapBuf(2) - out[1] = wprod(b, 2, 3) - p = 1 - p[1] = x - p[GEN] = y + out[1] - return -"; - let program = compile(&parse(src).expect("parse")); - - let (f3, f5, f7) = (F64(3), F64(5), F64(7)); - let one = F64::ONE; - // statement position: idx 2 -> (1+3)·5 - let x = (one + f3) * f5; - // embedded in a product: idx 1 -> 7·(3·(1+5)) - let y = f7 * (f3 * (one + f5)); - // heap-store RHS: idx 3 -> 3·5 - let o = f3 * f5; - let want = [F192::from(x), F192::from(y + o)]; - - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("expression-position inline calls compute correctly"); - - let mut bad = want; - bad[0] += F192::ONE; - assert!( - verify(&program, &bad, &proof).is_err(), - "wrong published value must be rejected" - ); -} diff --git a/crates/lean_compiler/tests/suite/loop_frames.rs b/crates/lean_compiler/tests/suite/loop_frames.rs deleted file mode 100644 index f098094f0..000000000 --- a/crates/lean_compiler/tests/suite/loop_frames.rs +++ /dev/null @@ -1,130 +0,0 @@ -use lean_compiler::{compile, compile_without_filler, parse}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::{F64, F192, g_pow}; - -#[test] -fn loop_frames_preserve_escaped_cells_and_nested_allocations() { - lean_vm::init_prover_pool(); - let source = r#" -def make_heap(x): - h = HeapBuf(2) - h[1] = x - h[GEN] = x * x - return h - -def run(stop): - saved = HeapBuf(GEN ** 12) - heaps = HeapBuf(GEN ** 12) - nested = HeapBuf(GEN ** 36) - sums = HeapBuf(GEN ** 13) - sums[GEN ** 2] = 0 - for x in mul_range(GEN ** 2, stop): - pair = [x, x * x] - saved[x] = addr(pair) - heaps[x] = make_heap(x) - for y in mul_range(1, GEN ** 3): - local = [x, y] - nested[x ** 3 * y] = addr(local) - sums[x * GEN] = sums[x] + x - for x in mul_range(GEN ** 2, stop): - p = saved[x] - h = heaps[x] - assert p[1] == x - assert p[GEN] == x * x - assert h[1] == x - assert h[GEN] == x * x - for y in mul_range(1, GEN ** 3): - q = nested[x ** 3 * y] - assert q[1] == x - assert q[GEN] == y - return sums[stop] - -def main(): - public = 1 - result = run(public[GEN]) - assert result == public[1] - return -"#; - let program = compile(&parse(source).unwrap()); - for end in [2, 3, 9] { - let sum = (2..end).fold(F64::ZERO, |sum, i| sum + g_pow(i)); - let public = [F192::from(sum), F192::from(g_pow(end))]; - assert!(program.execute(public).unwrap().unconstrained_reads.is_empty()); - if end == 9 { - let (proof, _) = prove(&program, public, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &public, &proof).unwrap(); - } - } -} - -#[test] -fn early_return_does_not_reserve_the_unused_range() { - let source = r#" -def main(): - out = HeapBuf(1) - for x in mul_range(1, GEN ** 4294967296): - if x == 1: - out[1] = 7 - return - public = 1 - assert public[1] == out[1] - return -"#; - let program = compile_without_filler(&parse(source).unwrap()); - let execution = program.execute([F192::from(F64(7)), F192::ZERO]).unwrap(); - assert!(execution.unconstrained_reads.is_empty()); -} - -#[test] -fn runtime_frame_count_uses_the_distance_from_the_start() { - let source = r#" -def main(): - public = 1 - out = HeapBuf(2) - for x in mul_range(GEN ** 65535, public[GEN]): - if x == GEN ** 65535: - out[1] = x - else: - out[GEN] = x - assert out[GEN] == public[1] - return -"#; - let program = compile_without_filler(&parse(source).unwrap()); - assert!( - program - .execute([g_pow(65536).into(), g_pow(65537).into()]) - .unwrap() - .unconstrained_reads - .is_empty() - ); -} - -#[test] -fn rebound_counter_keeps_incremental_frames() { - lean_vm::init_prover_pool(); - let source = r#" -def bump(x): - if x == 1: - return GEN ** 5 - if x == GEN ** 6: - return 1 - return x - -def main(): - public = 1 - seen = HeapBuf(7) - for x in mul_range(1, STOP): - seen[x] = x - x = bump(x) - assert seen[GEN ** 6] == GEN ** 6 - assert seen[GEN] == GEN - return -"#; - let public = [F192::ZERO, g_pow(2).into()]; - for bound in ["GEN ** 2", "public[GEN]"] { - let program = compile(&parse(&source.replace("STOP", bound)).unwrap()); - assert!(program.execute(public).unwrap().unconstrained_reads.is_empty()); - let (proof, _) = prove(&program, public, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &public, &proof).unwrap(); - } -} diff --git a/crates/lean_compiler/tests/suite/main.rs b/crates/lean_compiler/tests/suite/main.rs deleted file mode 100644 index f7d070318..000000000 --- a/crates/lean_compiler/tests/suite/main.rs +++ /dev/null @@ -1,28 +0,0 @@ -//! Compiler integration tests share one binary and its initialization caches. -//! -//! These tests leave the proving arena disabled. A test that enables it needs -//! its own process (see `rec_aggregation`'s `arena_prove`). - -mod common; - -mod assert_ne; -mod const_placeholder; -mod determinism; -mod disassemble; -mod field_div; -mod field_towers; -mod filler; -mod hint_log2_ceil; -mod inline_expr; -mod loop_frames; -mod pack64x2; -mod print_debug; -mod py_source; -mod range_check; -mod sharing; -mod soundness; -mod stack_bits; -mod stack_buf; -mod statements; -mod transcript_helpers; -mod vm_proofs; diff --git a/crates/lean_compiler/tests/suite/pack64x2.rs b/crates/lean_compiler/tests/suite/pack64x2.rs deleted file mode 100644 index eb082c492..000000000 --- a/crates/lean_compiler/tests/suite/pack64x2.rs +++ /dev/null @@ -1,69 +0,0 @@ -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{Fault, prove, verify}; -use primitives::field::{F64, F192}; - -use crate::common::mix; - -#[test] -fn pack64x2_proves_and_verifies() { - let src = "\ -@inline -def pack64x2(a, b): - assert_in_k(a, b) - return a + f192(0, 1, 0) * b - -def main(): - a = 5 - b = 7 - packed = pack64x2(a, b) - p = 1 - p[1] = packed - p[GEN] = packed - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::new(5, 7, 0), F192::new(5, 7, 0)]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - let counts = mix(src, want); - assert_eq!( - (counts[0], counts[1], counts[4]), - (1, 2, 2), - "XOR, MUL and JUMP lowering" - ); - verify(&program, &want, &proof).expect("pack64x2 program verifies"); -} - -#[test] -fn pack64x2_rejects_extension_field_source() { - let src = "\ -@inline -def pack64x2(a, b): - assert_in_k(a, b) - return a + f192(0, 1, 0) * b - -def main(): - a = StackBuf(1) - hint_witness(a[0:1], \"a\") - packed = pack64x2(a[0], 7) - p = 1 - p[1] = packed - p[GEN] = packed - return -"; - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("a", vec![vec![F192::new(5, 1, 0)]]); - let err = program - .execute([F192::from(F64::ONE), F192::from(F64::ONE)]) - .err() - .expect("the run must fail"); - assert!( - matches!( - err.fault, - Fault::NotInK { - what: "JUMP target", - .. - } - ), - "{err}" - ); -} diff --git a/crates/lean_compiler/tests/suite/print_debug.rs b/crates/lean_compiler/tests/suite/print_debug.rs deleted file mode 100644 index 33b77eeba..000000000 --- a/crates/lean_compiler/tests/suite/print_debug.rs +++ /dev/null @@ -1,28 +0,0 @@ -//! `print(...)`: a prover-side debug print: must compile, execute during -//! witness generation, and leave proving/verification untouched. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::{F64, F192}; - -#[test] -fn print_is_constraint_free() { - let src = "\ -def main(): - x = 5 - y = x * GEN - print(y) - print(\"the product\", y * y) - b = HeapBuf(2) - b[1] = 3 - print(b[1]) - p = 1 - p[1] = y - p[GEN] = b[1] - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(F64(5) * primitives::field::g_pow(1)), F192::from(F64(3))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("prints must not disturb proving"); -} diff --git a/crates/lean_compiler/tests/suite/py_source.rs b/crates/lean_compiler/tests/suite/py_source.rs deleted file mode 100644 index d3f3e6b8d..000000000 --- a/crates/lean_compiler/tests/suite/py_source.rs +++ /dev/null @@ -1,98 +0,0 @@ -//! zkDSL sources as `.py` files (as in leanVM's `test_data`): the `snark_lib` -//! stub import makes them valid Python for editors/linters, and the compiler -//! skips it (single-file programs only: importing anything else is an error). -//! -//! The harness is generic: every `tests/programs/*.py` is parsed, compiled, -//! proven, and verified. A program declares the public input it expects with a -//! top-of-file annotation of two constant field elements, -//! -//! ```text -//! # public_input: GEN ** 89, 101229015297003380629709256178361811305 -//! ``` -//! -//! or omits it to run with the empty public input (two zeros). - -use std::fs; - -use lean_compiler::{compile, parse, parse_const}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::F192; - -/// The `# public_input: , ` annotation, or `[0, 0]` if absent. -fn public_input(src: &str) -> [F192; 2] { - for line in src.lines() { - if let Some(rest) = line.trim().strip_prefix("# public_input:") { - let parts: Vec<&str> = rest.split(',').collect(); - assert_eq!( - parts.len(), - 2, - "`# public_input:` needs two field elements, got `{rest}`" - ); - let elt = |s: &str| parse_const(s).unwrap_or_else(|e| panic!("bad public_input: {e}")); - return [elt(parts[0]), elt(parts[1])]; - } - } - [F192::ZERO; 2] -} - -/// The `# witness : , …` annotations: one line per *entry* -/// (repeated lines with the same name are the stream's successive entries, -/// popped by successive `hint_witness` calls). -fn witness(src: &str) -> std::collections::HashMap>> { - let mut streams: std::collections::HashMap>> = Default::default(); - for rest in src.lines().filter_map(|l| l.trim().strip_prefix("# witness ")) { - let (name, vals) = rest.split_once(':').expect("`# witness` needs `name: values`"); - let entry = vals - .split(',') - .map(|s| parse_const(s).unwrap_or_else(|e| panic!("bad witness value: {e}"))) - .collect(); - streams.entry(name.trim().to_string()).or_default().push(entry); - } - streams -} - -/// Every program in `tests/programs/`, end to end. -#[test] -fn all_py_programs() { - let dir = concat!(env!("CARGO_MANIFEST_DIR"), "/tests/programs"); - let mut paths: Vec<_> = fs::read_dir(dir) - .expect("tests/programs") - .map(|e| e.expect("dir entry").path()) - .filter(|p| p.extension().is_some_and(|x| x == "py")) - .collect(); - paths.sort(); - assert!(!paths.is_empty(), "no .py programs found"); - - for path in paths { - let name = path.file_name().unwrap().to_string_lossy().into_owned(); - let src = fs::read_to_string(&path).unwrap_or_else(|e| panic!("{name}: read: {e}")); - let want = public_input(&src); - let ast = parse(&src).unwrap_or_else(|e| panic!("{name}: parse: {e}")); - let mut program = compile(&ast); - for (stream, entries) in witness(&src) { - program.set_witness(stream, entries); - } - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).unwrap_or_else(|e| panic!("{name}: verify: {e:?}")); - println!("{name}: ok"); - } -} - -/// Both import spellings are tolerated (and skipped). -#[test] -fn snark_lib_import_forms() { - for import in ["import snark_lib", "from snark_lib import *"] { - let src = format!("{import}\ndef main():\n return\n"); - parse(&src).expect("snark_lib import is skipped"); - } -} - -/// Importing anything else is a parse error: no multi-file programs (yet). -#[test] -fn other_imports_rejected() { - for import in ["import math", "from utils import *"] { - let src = format!("{import}\ndef main():\n return\n"); - let err = parse(&src).expect_err("non-snark_lib import must be rejected"); - assert!(err.contains("file imports are not supported"), "{err}"); - } -} diff --git a/crates/lean_compiler/tests/suite/range_check.rs b/crates/lean_compiler/tests/suite/range_check.rs deleted file mode 100644 index 533adca00..000000000 --- a/crates/lean_compiler/tests/suite/range_check.rs +++ /dev/null @@ -1,223 +0,0 @@ -//! Range checks *in the exponent*: `assert log x < log GEN ** k` (or -//! `assert log x < k`) proves `log_g(x) < k`, i.e. `x ∈ {g^0, g^1, …, g^{k-1}}`, -//! in 3 cycles: `DEREF x` bounds `log(x)` by the memory size, a `MUL` into the -//! write-once constant cell `g^{k-1}` back-solves and binds the complement -//! `y = g^{k-1-log(x)}`, and `DEREF y` bounds the complement. leanVM's DEREF -//! range-check trick, transported to g-powers; the only nondeterminism is the -//! end-of-run resolution of the two touched cells. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{Fault, prove, verify}; -use primitives::field::{F64, F192, g_pow}; - -use crate::common::mix; - -/// Both bound forms (`log GEN ** k` and a plain integer exponent) with the -/// boundary elements (`g^{k-1}`, `1 = g^0`), end-to-end: prove + verify, and a -/// wrong public input is rejected. Also pins the gadget's cost: 2 DEREFs per -/// check. -#[test] -fn range_check_end_to_end() { - let src = "\ -def main(): - x = GEN ** 5 - assert log x < log GEN ** 8 - y = GEN ** 7 - assert log y < 8 - assert log 1 < log GEN ** 8 - z = x * y - assert log z < 13 - p = 1 - p[1] = z - p[GEN] = x - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(g_pow(12)), F192::from(g_pow(5))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - // 2 DEREFs per range check (4 checks) + 2 publishing stores. - assert_eq!(mix(src, want)[3], 10, "DEREF count"); - verify(&program, &want, &proof).expect("range-checked program verifies"); - - let bad = [F192::from(g_pow(12)), F192::from(g_pow(6))]; - assert!( - verify(&program, &bad, &proof).is_err(), - "wrong public input must be rejected" - ); -} - -/// A check whose two touched cells (`m[300]` and the complement's `m[99]`) are -/// never written by the program: their `DEREF`s only link them to fresh cells, -/// all of which stay ZERO, and the bus still balances. -#[test] -fn range_check_unwritten_cells() { - let src = "\ -def main(): - x = GEN ** 300 - assert log x < 400 - p = 1 - p[1] = x - p[GEN] = x - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(g_pow(300)), F192::from(g_pow(300))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("unwritten touches verify"); -} - -/// The largest allowed bound, `2^16` = the minimum prover memory, end to end: -/// the complement cell is `g^65535`, the last cell of that memory, so an -/// off-by-one in the bound check or in the memory size shows up here. -#[test] -fn range_check_max_bound() { - let src = "\ -def main(): - x = GEN ** 5 - assert log x < 65536 - p = 1 - p[1] = x - p[GEN] = x - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(g_pow(5)); 2]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("max-bound range check verifies"); -} - -/// Range checks inside a `mul_range` body: the check runs once per iteration in -/// a fresh helper frame (its own `g^{k-1}` constant cell each time), and the -/// touched low cells mix already-written ones (`m[0]`, `m[1]`: the public -/// input) with unwritten ones. -#[test] -fn range_check_in_loop() { - let src = "\ -def main(): - for i in mul_range(1, GEN ** 6): - assert log i < log GEN ** 6 - p = 1 - p[1] = 5 - p[GEN] = 7 - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(F64(5)), F192::from(F64(7))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - // 6 iterations × 2 range-check DEREFs, plus call/publish plumbing. - assert!(mix(src, want)[3] >= 12, "at least the 12 range-check DEREFs"); - verify(&program, &want, &proof).expect("loop range checks verify"); -} - -/// `log(g^8) < 8` is false: the complement back-solves to a huge-exponent -/// element, and its DEREF fails witness generation: the honest-execution -/// surface of a failing range check. -#[test] -fn range_check_at_bound_rejected() { - let src = "def main():\n x = GEN ** 8\n assert log x < 8\n return\n"; - let program = compile(&parse(src).expect("parse")); - let err = program - .execute([F192::ZERO, F192::ZERO]) - .err() - .expect("the run must fail"); - assert!(matches!(err.fault, Fault::WildPointer { .. }), "{err}"); -} - -/// A value that is no small g-power at all (5 = x^2 + 1) fails at the first -/// DEREF, the same way. -#[test] -fn range_check_non_g_power_rejected() { - let src = "def main():\n x = 5\n assert log x < 8\n return\n"; - let program = compile(&parse(src).expect("parse")); - let err = program - .execute([F192::ZERO, F192::ZERO]) - .err() - .expect("the run must fail"); - assert!(matches!(err.fault, Fault::WildPointer { .. }), "{err}"); -} - -/// Bound 0 names the empty set: rejected at compile time. -#[test] -#[should_panic(expected = "names the empty set")] -fn range_check_empty_bound_rejected() { - let src = "def main():\n x = 1\n assert log x < 0\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// Bounds beyond `2^16` (the minimum memory size) would not be sound for every -/// prover memory choice: rejected at compile time. -#[test] -#[should_panic(expected = "exceeds 2^16")] -fn range_check_bound_too_big_rejected() { - let src = "def main():\n x = 1\n assert log x < 65537\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// A `<` assert without `log` is rejected: field elements have no order, only -/// their logs do. -#[test] -#[should_panic(expected = "compares logs")] -fn range_check_without_log_rejected() { - let src = "def main():\n x = 1\n assert x < 8\n return\n"; - let _ = parse(src).map_err(|e| panic!("{e}")); -} - -/// A *runtime* bound, `assert log x < log n`: the same gadget with `g^{k-1}` -/// derived as `n·g^{-1}` instead of pooled from a constant. The bound rides a -/// hint here, as it does in the aggregation guest, where the signer count is -/// prover-announced. -#[test] -fn range_check_runtime_bound() { - let src = "\ -def main(): - nb = StackBuf(1) - hint_witness(nb[0:1], \"n\") - n = nb[0] - assert log n < 64 - x = GEN ** 5 - assert log x < log n - assert log 1 < log n - p = 1 - p[1] = x - p[GEN] = n - return -"; - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("n", vec![vec![F192::from(g_pow(6))]]); - let want = [F192::from(g_pow(5)), F192::from(g_pow(6))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("runtime-bound range check verifies"); -} - -/// The runtime bound binds: `log(g^5) < log(g^5)` is false, and the complement -/// back-solves to a huge-exponent element whose DEREF fails, exactly as for a -/// compile-time bound at its boundary. -#[test] -fn range_check_runtime_bound_at_bound_rejected() { - let src = "\ -def main(): - nb = StackBuf(1) - hint_witness(nb[0:1], \"n\") - x = GEN ** 5 - assert log x < log nb[0] - return -"; - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("n", vec![vec![F192::from(g_pow(5))]]); - let err = program - .execute([F192::ZERO, F192::ZERO]) - .err() - .expect("the run must fail"); - assert!(matches!(err.fault, Fault::WildPointer { .. }), "{err}"); -} - -/// A bound that folds at parse time but is not a power of `GEN` stays a parse -/// error rather than quietly becoming a runtime bound. `log 8` names the field -/// element 8, whose g-log is nothing in particular, so a program meaning `< 8` -/// must not compile into a check that can only fail at witness generation. -#[test] -#[should_panic(expected = "must be a power of GEN")] -fn range_check_folded_non_gpower_bound_rejected() { - let src = "def main():\n x = GEN ** 3\n assert log x < log 5\n return\n"; - let _ = parse(src).map_err(|e| panic!("{e}")); -} diff --git a/crates/lean_compiler/tests/suite/sharing.rs b/crates/lean_compiler/tests/suite/sharing.rs deleted file mode 100644 index 4626ae6ad..000000000 --- a/crates/lean_compiler/tests/suite/sharing.rs +++ /dev/null @@ -1,365 +0,0 @@ -//! The lowerer shares one cell between identical pure operations (`FnLower::pure`). -//! These programs pin the cases where a "duplicate" is NOT dead, so sharing it -//! would drop a constraint. They were written against a value-numbering pass that -//! ran after lowering, and they outlived it: the hazard belongs to the sharing, -//! not to where it happens. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{Fault, ProveError, prove, verify}; -use primitives::field::{F64, F192, g_pow}; - -/// A returned value that repeats a constant computed earlier in the same -/// function. The return slot lives in the callee frame and is read by the -/// CALLER, so eliminating that write leaves the caller reading an unwritten -/// (prover-chosen) cell: `walk` in the XMSS guest returned a flag exactly this -/// way, and folding it produced a proof whose caller-side assert failed. -#[test] -fn duplicate_constant_in_a_return_slot_survives() { - let src = "\ -def tag(x): - # `marker` is the same constant the flag below returns, and it is computed - # first, so the flag's `SET` is a textual duplicate of it. - marker = 7 - return x * marker, 7 - -def main(): - v, flag = tag(GEN ** 3) - p = 1 - p[1] = v - p[GEN] = flag - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(g_pow(3)) * F192::from(F64(7)), F192::from(F64(7))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("returned duplicate constant is preserved"); -} - -/// An argument slot written with a value that already exists in the caller: the -/// callee reads its arguments out of its own frame, so the store must stay. -#[test] -fn duplicate_argument_value_survives() { - let src = "\ -def add_both(a, b): - return a + b - -def main(): - k = GEN ** 5 - # Both arguments are the same expression, and the sum is computed here too, - # so every operand the call needs has a duplicate in this frame. - local = k + k - s = add_both(k, k) - p = 1 - p[1] = s + local - return -"; - let program = compile(&parse(src).expect("parse")); - // (k + k) + (k + k) == 0 in characteristic two. - let want = [F192::ZERO, F192::ZERO]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("duplicated call arguments are preserved"); -} - -/// A duplicate on one side of a branch must not be shared with the other side's -/// computation: the cache reverts at a join, so each arm recomputes what it -/// needs. Shared, the arm that runs would read a cell only the untaken arm -/// writes, leaving it unwritten and so prover-chosen. -/// -/// The condition is FALSE on purpose, so the arm that runs is the SECOND one -/// emitted. Written the other way the taken arm is the one that mints the cell, -/// any sharing can only redirect the untaken arm, and the test cannot fail: it -/// passed with both caches leaking past the join and with the scope revert -/// deleted outright. -#[test] -fn duplicates_are_not_folded_across_a_branch() { - let src = "\ -def main(): - x = GEN ** 3 - r = HeapBuf(2) - # The same constant in both arms, and the else arm is the one that runs. - if x == GEN ** 5: - r[1] = GEN ** 4 - else: - r[1] = GEN ** 4 - p = 1 - p[1] = r[1] - p[GEN] = x - return -"; - let run = |pi: [F192; 2]| -> bool { - let program = compile(&parse(src).expect("parse")); - prove(&program, pi, lean_vm::pcs::TEST_LOG_INV_RATE) - .is_ok_and(|(proof, _)| verify(&program, &pi, &proof).is_ok()) - }; - assert!( - run([F192::from(g_pow(4)), F192::from(g_pow(3))]), - "the arm that runs keeps its own constant" - ); - // The wrong value is the half that bites: a cell only the untaken arm writes - // is the prover's to choose, and the honest claim would verify anyway. - assert!( - !run([F192::from(g_pow(7)), F192::from(g_pow(3))]), - "the stored constant is pinned by the arm that ran" - ); -} - -/// The assert idiom is `XOR fp[t] = a ^ b` into the pooled zero cell, whose -/// second write IS the assertion, so that cell must never be shared and the -/// `XOR` must never be skipped in favour of one computed earlier. -/// -/// The operands are HEAP READS on purpose. Written `a = GEN ** 9`, both sides -/// fold and `diff = a + b` emits no `XOR` at all, so the duplicate the test -/// names does not exist and the test cannot fail: skipping the assert on a cache -/// hit then passed the whole suite. Read from a `HeapBuf` the two values are -/// runtime cells, `diff` really does emit `XOR fp[t] = a ^ b`, and the assert's -/// own `XOR` has a genuine duplicate to be folded into. -fn duplicated_comparison(second: u32) -> String { - format!( - "\ -def main(): - hb = HeapBuf(2) - hb[1] = GEN ** 9 - hb[GEN] = GEN ** {second} - a = hb[1] - b = hb[GEN] - # The same XOR the assert needs, as a live value. - diff = a + b - assert a == b - p = 1 - p[1] = diff - p[GEN] = a - return -" - ) -} - -#[test] -fn assert_survives_a_duplicated_comparison() { - let program = compile(&parse(&duplicated_comparison(9)).expect("parse")); - let want = [F192::ZERO, F192::from(g_pow(9))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("passing assert still verifies"); -} - -/// The same shape with the assert failing: it must still fail, which is what -/// says the assertion is really there. -#[test] -fn failing_assert_still_conflicts() { - let program = compile(&parse(&duplicated_comparison(10)).expect("parse")); - let want = [F192::ZERO, F192::from(g_pow(9))]; - let Err(ProveError::Execution(err)) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE) else { - panic!("a failing assert makes no proof") - }; - assert!(matches!(err.fault, Fault::Conflict { .. }), "{err}"); -} - -/// A hint at the end of a runtime branch is attached to a no-op anchor by the -/// lowerer. Even when that anchor repeats an earlier pure instruction, it has to -/// be emitted: moving the hint to the next textual instruction would move it to -/// the join and execute it when the branch is not taken. -#[test] -fn trailing_branch_hint_stays_in_its_branch() { - let src = "\ -def main(): - flag = StackBuf(1) - hint_witness(flag, \"flag\") - data = StackBuf(1) - if flag[0] == 1: - print(\"anchor\", flag[0]) - hint_witness(data, \"data\") - p = 1 - p[1] = flag[0] - p[GEN] = 0 - return -"; - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("flag", vec![vec![F192::ZERO]]); - let want = [F192::ZERO, F192::ZERO]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("untaken branch must not consume its witness"); -} - -/// A BLAKE2s chaining value names a CONSECUTIVE PAIR, so neither half may be -/// folded into a canonical elsewhere and the base may not be rewritten: a -/// substitution speaks for one cell, and redirecting the base silently redirects -/// the second word too. `rewrite_reads` used to map `cv` like any single-cell -/// read, so when the first of the two assembling copies duplicated an earlier -/// copy of the same source, the compression absorbed the OTHER pair's second -/// word. Silent, and a soundness break in a transcript. -/// -/// The two compressions here differ in nothing but their chaining value, and -/// their two `cv` pairs share a first word, which is what made the first copy a -/// duplicate. If either pair is rewritten or dropped, the digests coincide and -/// the inequality fails at witness generation. -#[test] -fn a_chaining_value_pair_is_neither_rewritten_nor_dropped() { - let src = "\ -def main(): - hb = HeapBuf(4) - hb[1] = GEN ** 11 - hb[GEN] = GEN ** 22 - hb[GEN ** 2] = GEN ** 33 - hb[GEN ** 3] = GEN ** 44 - x = hb[1] - y = hb[GEN] - z = hb[GEN ** 2] - w = hb[GEN ** 3] - msg = StackBuf(4) - msg[0] = y - msg[1] = y - msg[2] = y - msg[3] = y - t = StackBuf(2) - t[0] = x - t[1] = z - o1 = StackBuf(2) - blake2s(msg[0:2], msg[2:4], o1, cv=t, counter=64, final=1) - s = StackBuf(2) - s[0] = x - s[1] = w - o2 = StackBuf(2) - blake2s(msg[0:2], msg[2:4], o2, cv=s, counter=64, final=1) - assert o1[0] != o2[0] - p = 1 - p[1] = x - p[GEN] = y - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(g_pow(11)), F192::from(g_pow(22))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("each compression absorbs its own chaining value"); -} - -#[test] -fn cached_loads_do_not_cross_branches_or_erase_stores() { - let source = r#" -def main(): - h = HeapBuf(3) - hint_witness(h[0:3], "values") - public = 1 - if public[1] == 0: - a = h[1] + h[GEN] - else: - b = h[1] + h[GEN] - h[GEN ** 2] = h[1] - assert h[GEN ** 2] == h[1] - h[1] = public[GEN] - return -"#; - let value = F192::new(17, 31, 43); - let mut program = compile(&parse(source).unwrap()); - program.set_witness("values", vec![vec![value, F192::ONE, value]]); - for branch in [F192::ZERO, F192::ONE] { - assert!(program.execute([branch, value]).unwrap().unconstrained_reads.is_empty()); - assert!(program.execute([branch, value + F192::ONE]).is_err()); - } -} - -#[test] -fn cached_loads_see_linked_equalities_before_use() { - lean_vm::init_prover_pool(); - let source = r#" -def fill(h): - value = hint_witness("value") - h[1] = value - return - -def main(): - h = HeapBuf(1) - TOUCH - FILL - out = StackBuf(2) - out[0] = h[1] * h[1] - out[1] = h[1] - public = 1 - assert public[1] == out[0] - assert public[GEN] == out[1] - return -"#; - let value = F192::from(F64(7)); - let public = [value * value, value]; - for touch in ["assert log(h) < 1024", "early = StackBuf(1)\n early[0] = h[1]"] { - for fill in ["hint_witness(h[0:1], \"value\")", "fill(h)"] { - let source = source.replace("TOUCH", touch).replace("FILL", fill); - let mut program = compile(&parse(&source).unwrap()); - program.set_witness("value", vec![vec![value]]); - assert!(program.execute(public).unwrap().unconstrained_reads.is_empty()); - let (proof, _) = prove(&program, public, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &public, &proof).unwrap(); - assert!(program.execute([public[0] + F192::ONE, value]).is_err()); - } - } -} - -/// A load that runs before its store takes the stored value, as write-once memory -/// says, even where the loaded name is used directly rather than loaded again. -#[test] -fn a_load_before_its_store_sees_the_store() { - lean_vm::init_prover_pool(); - let source = "\ -def main(): - h = HeapBuf(1) - x = h[1] - h[1] = GEN ** 3 - public = 1 - assert public[1] == x * GEN - return -"; - let program = compile(&parse(source).unwrap()); - let public = [F192::from(g_pow(4)), F192::ZERO]; - let (proof, _) = prove(&program, public, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &public, &proof).unwrap(); -} - -/// A heap cell nothing stores to is prover-chosen, so using a value loaded from it -/// is an unconstrained read, exactly as for a stack cell. -#[test] -fn a_load_of_an_unwritten_heap_cell_is_an_unconstrained_read() { - let source = "\ -def main(): - h = HeapBuf(2) - x = h[1] - public = 1 - public[1] = x * GEN - return -"; - let exec = compile(&parse(source).unwrap()).execute([F192::ZERO; 2]).unwrap(); - assert!(!exec.unconstrained_reads.is_empty()); -} - -#[test] -fn cached_copies_see_later_stores() { - lean_vm::init_prover_pool(); - let source = r#" -def square(h): - return h[GEN] * h[GEN] - -def main(): - h = HeapBuf(2) - early = StackBuf(1) - other = StackBuf(1) - early[0] = h[GEN] - value = hint_witness("value") - FILL_DEST - DEST[0] = h[GEN] - result = square(h) - public = 1 - assert public[1] == result - return -"#; - let value = F192::from(F64(7)); - let public = [value * value, F192::ZERO]; - for (dest, fill) in [ - ("early", "early[0] = value"), - ("other", "other[0] = value"), - ("other", "other[0] = h[GEN]\n h[GEN] = value"), - ] { - let source = source.replace("FILL_DEST", fill).replace("DEST", dest); - let mut program = compile(&parse(&source).unwrap()); - program.set_witness("value", vec![vec![value]]); - assert!(program.execute(public).unwrap().unconstrained_reads.is_empty()); - let (proof, _) = prove(&program, public, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &public, &proof).unwrap(); - } -} diff --git a/crates/lean_compiler/tests/suite/soundness/cases.rs b/crates/lean_compiler/tests/suite/soundness/cases.rs deleted file mode 100644 index c2a2f30fc..000000000 --- a/crates/lean_compiler/tests/suite/soundness/cases.rs +++ /dev/null @@ -1,558 +0,0 @@ -//! Layer 1: perturbation. Each case is one program with one valid trial and a -//! table of single-cell pokes that must break it. -//! -//! Coverage is by *lowering*, not by feature list: every case exercises a -//! construct whose lowering could plausibly drop the check it stands for, and -//! every poke names one constraint. A poke that is accepted says which one is -//! missing. -//! -//! The pokes lean on witness streams rather than the public input, because two -//! public words is all there is and because the streams are where a real guest's -//! untrusted data actually enters. - -use super::{Case, Trial, check_case, g, k, pi, wit}; -use primitives::field::F192; - -/// `XOR`/`MUL` relations, both assert forms, and the division back-solve. The -/// quotient cell is written by nothing but the back-solve, so this case also -/// pins the one legitimate way a cell may be read before any instruction writes -/// it. -#[test] -fn arithmetic_and_asserts() { - check_case(&Case { - name: "arithmetic_and_asserts", - src: "\ -def main(): - v = StackBuf(3) - hint_witness(v, \"w\") - assert v[0] * v[1] == v[2] - assert v[0] != v[1] - q = v[2] / v[0] - assert q == v[1] - p = GEN ** 0 - p[1] = v[2] - p[GEN] = v[0] + v[1] - return -", - valid: Trial::new([g(8), g(3) + g(5)]).stream("w", vec![vec![g(3), g(5), g(8)]]), - pokes: vec![ - // Each of the three hinted cells breaks the product relation. - wit("w", 0, g(4)), - wit("w", 1, g(6)), - wit("w", 2, g(9)), - // Equal operands: the product relation would still need v[2] = g^10, - // but this is the poke that `assert !=` exists for. - wit("w", 0, g(5)), - // Both published words. - pi(0, g(9)), - pi(1, g(3) + g(6)), - ], - }); -} - -/// The exponent range check and `match` dispatch. The dispatch is only -/// sound because the matched value was range-checked first (doc §Match -/// statements), so a poke past the bound must be caught by the check rather than -/// land at an attacker-chosen arm. -#[test] -fn range_check_and_dispatch() { - check_case(&Case { - name: "range_check_and_dispatch", - src: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - assert log(v[0]) < 8 - r = match(log(v[0]), range(0, 8), lambda i: sq(i)) - assert r == v[1] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = r - return - - -def sq(x): - return x * x -", - // Arm 3 runs: sq(3) = 3·3 in K = (x+1)^2 = x^2+1 = 5. - valid: Trial::new([g(3), k(5)]).stream("w", vec![vec![g(3), k(5)]]), - pokes: vec![ - // Past the bound: the range check's complement DEREF must catch it. - wit("w", 0, g(8)), - wit("w", 0, g(63)), - // A different arm runs, so the claimed square is wrong. - wit("w", 0, g(4)), - // The claimed square itself. - wit("w", 1, k(6)), - pi(0, g(4)), - pi(1, k(6)), - ], - }); -} - -/// An `@inline` arm runs in the dispatching frame, so it writes the caller's -/// `StackBuf` directly and returns its own `Const`. A poke that selects another -/// arm or changes the written value must be caught. -#[test] -fn inline_arms_write_the_callers_buffer() { - check_case(&Case { - name: "inline_arms_write_the_callers_buffer", - src: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - assert log(v[0]) < 4 - out = StackBuf(1) - e = match(log(v[0]), range(0, 4), lambda i: put(out, v[1], i)) - p = GEN ** 0 - p[1] = out[0] - p[GEN] = e - return - - -@inline -def put(out, x, n: Const): - out[0] = x * GEN ** n - return const(n + 1) -", - // Arm 2: out = g^5·g^2, e = 3. - valid: Trial::new([g(7), k(3)]).stream("w", vec![vec![g(2), g(5)]]), - pokes: vec![ - wit("w", 0, g(4)), - wit("w", 0, g(1)), - wit("w", 1, g(6)), - pi(0, g(8)), - pi(1, k(2)), - ], - }); -} - -/// `if`/`else` communicating through a write-once heap cell: only one arm runs, -/// so both may write it and the join reads it back. A lowering that lets the -/// join read anything other than the taken arm's value shows up as a poke that -/// selects the other arm and is still accepted. -#[test] -fn branch_join() { - check_case(&Case { - name: "branch_join", - src: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - assert log(v[0]) < 4 - r = HeapBuf(1) - if v[0] == GEN ** 2: - r[1] = v[1] * GEN - else: - r[1] = v[1] * GEN ** 3 - p = GEN ** 0 - p[1] = r[1] - p[GEN] = v[0] - return -", - valid: Trial::new([g(6), g(2)]).stream("w", vec![vec![g(2), g(5)]]), - pokes: vec![ - // Takes the else arm, which multiplies by g^3 instead of g. - wit("w", 0, g(1)), - wit("w", 0, g(3)), - // Past the bound. - wit("w", 0, g(4)), - // The value the taken arm shifts. - wit("w", 1, g(4)), - pi(0, g(7)), - pi(1, g(3)), - ], - }); -} - -/// A `mul_range` loop with a runtime bound and heap-carried state. The bound is -/// hinted, so the loop terminates only because its log was checked first; the -/// pokes cover both a bound that changes the trip count and one past the check. -#[test] -fn loop_with_runtime_bound() { - check_case(&Case { - name: "loop_with_runtime_bound", - src: "\ -def main(): - v = StackBuf(1) - hint_witness(v, \"n\") - assert log(v[0]) < 8 - acc = HeapBuf(16) - acc[1] = GEN ** 0 - for i in mul_range(1, v[0]): - acc[i * GEN] = acc[i] * GEN ** 2 - p = GEN ** 0 - p[1] = acc[v[0]] - p[GEN] = v[0] - return -", - // n = g^5: five iterations, acc[j] = g^{2j}, so acc[5] = g^10. - valid: Trial::new([g(10), g(5)]).stream("n", vec![vec![g(5)]]), - pokes: vec![ - // Fewer and more iterations: acc[n] is then g^8 and g^12. - wit("n", 0, g(4)), - wit("n", 0, g(6)), - // Past the bound. - wit("n", 0, g(8)), - pi(0, g(11)), - pi(1, g(4)), - ], - }); -} - -/// `pack64x2`'s range assertion: both sources must lie in K. Its untaken JUMP -/// puts them in the destination and frame slots, whose memory reads have -/// literal-zero upper limbs. -#[test] -fn pack64x2_range_assertion() { - check_case(&Case { - name: "pack64x2_range_assertion", - src: "\ -@inline -def pack64x2(a, b): - assert_in_k(a, b) - return a + f192(0, 1, 0) * b - -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - c = pack64x2(v[0], v[1]) - p = GEN ** 0 - p[1] = c - p[GEN] = v[0] - return -", - valid: Trial::new([F192::new(5, 7, 0), k(5)]).stream("w", vec![vec![k(5), k(7)]]), - pokes: vec![ - // Either source outside K. - wit("w", 0, F192::new(5, 1, 0)), - wit("w", 0, F192::new(5, 0, 1)), - wit("w", 1, F192::new(7, 1, 0)), - // In K, but not the packing that was published. - wit("w", 0, k(6)), - wit("w", 1, k(8)), - pi(0, F192::new(5, 8, 0)), - pi(1, k(6)), - ], - }); -} - -/// The digest-as-verification idiom: a hinted preimage, hashed, and the result -/// pinned against a hinted digest through a heap store. This is the shape a -/// signature verifier has, so it is the one that most needs a regression test. -/// -/// The digest constant comes from [`print_blake2s_digest`], not from a hand -/// computation: what the case tests is that a *wrong* digest is rejected, and -/// for that the honest value only has to be honest. -#[test] -fn digest_pins_its_preimage() { - check_case(&Case { - name: "digest_pins_its_preimage", - src: BLAKE2S_PIN_SRC, - valid: Trial::new([k(5), k(7)]) - .stream("msg", vec![vec![k(5), k(7), F192::ZERO, F192::ZERO]]) - .stream("dig", vec![vec![DIGEST_5_7[0], DIGEST_5_7[1]]]), - pokes: vec![ - // A different preimage hashes to something else. - wit("msg", 0, k(6)), - wit("msg", 1, k(8)), - wit("msg", 2, k(1)), - wit("msg", 3, k(1)), - // A wrong digest is what the write-once store has to catch. - wit("dig", 0, F192::ZERO), - wit("dig", 1, F192::ZERO), - wit("dig", 0, DIGEST_5_7[0] + F192::ONE), - wit("dig", 1, DIGEST_5_7[1] + F192::ONE), - // The published preimage words. - pi(0, k(6)), - pi(1, k(8)), - ], - }); -} - -const BLAKE2S_PIN_SRC: &str = "\ -def main(): - m = StackBuf(4) - hint_witness(m, \"msg\") - d = StackBuf(2) - blake2s(m[0:2], m[2:4], d) - e = HeapBuf(2) - hint_witness(e[0:2], \"dig\") - e[1] = d[0] - e[GEN] = d[1] - p = GEN ** 0 - p[1] = m[0] - p[GEN] = m[1] - return -"; - -/// BLAKE2s of the 64-byte block whose four canonical cells are `(5, 7, 0, 0)`. -pub const DIGEST_5_7: [F192; 2] = [ - F192::new(0xbbc8_c175_8cb7_7642, 0xf299_5d40_1fad_f4ff, 0), - F192::new(0x83ea_6ade_289a_53c8, 0x57e6_e523_12ec_734b, 0), -]; - -/// Regenerate [`DIGEST_5_7`]: `cargo test --release -p lean_compiler -/// print_blake2s_digest -- --ignored --nocapture`. Kept so the constant above is -/// reproducible rather than folklore. -#[test] -#[ignore = "prints a constant; not a check"] -fn print_blake2s_digest() { - let src = "\ -def main(): - m = StackBuf(4) - hint_witness(m, \"msg\") - d = StackBuf(2) - blake2s(m[0:2], m[2:4], d) - print(d[0]) - print(d[1]) - return -"; - let mut p = super::build(src); - p.set_witness("msg", vec![vec![k(5), k(7), F192::ZERO, F192::ZERO]]); - p.execute([F192::ZERO, F192::ZERO]).unwrap(); -} - -/// The fused `match` path must reject a call that binds more names than -/// the callee returns, exactly as the non-fused path does. Before this check the -/// surplus name `DEREF`ed a callee-frame offset nothing on the taken path wrote, -/// and since the shared frame is sized to the largest callee that offset exists, -/// so the name bound a prover-chosen word. -/// -/// Fusion needs every arm to be a call to the same function with identical -/// runtime arguments, so the two programs below are the fused shape: one over -/// mixed-arity callees, one over a single over-bound callee. -#[test] -#[should_panic(expected = "dispatched call binds")] -fn dispatched_call_rejects_a_mixed_arity_arm() { - super::build( - "\ -def main(): - x = GEN ** 2 - a, b, c = match(log(x), range(0, 2), lambda i: three(x, i), range(2, 4), lambda i: one(x, i)) - p = GEN ** 0 - p[1] = b - p[GEN] = c - return - - -def three(v, k: Const): - q = v * GEN ** k - return q, q * q, q * q * q - - -def one(v, k: Const): - return v * GEN ** k -", - ); -} - -#[test] -#[should_panic(expected = "dispatched call binds")] -fn dispatched_call_rejects_an_over_bound_callee() { - super::build( - "\ -def main(): - x = GEN ** 1 - a, b = match(log(x), range(0, 4), lambda i: one(x, i)) - p = GEN ** 0 - p[1] = a - p[GEN] = b - return - - -def one(v, k: Const): - return v * GEN ** k -", - ); -} - -/// A local whose name collides with a top-level constant array must be rejected. -/// `zkDSL.md` §Global constants reserves the name; a scalar constant enforces that -/// by construction (the parser substitutes its value, so the shadowing binding -/// becomes a literal and fails loudly), but a constant array was carried to -/// lowering, where `const_array_elem` resolved `NAME[i]` against it without -/// consulting the scope and `expr` folded it before the local could be seen. -/// -/// The consequence was the catastrophic direction for a hint: the range check -/// below ran against the baked constant `g^3` and passed, while the actual witness -/// `g^40` was never bounded and never read. -#[test] -#[should_panic(expected = "reserved")] -fn a_local_may_not_shadow_a_constant_array() { - super::build( - "\ -Q = [8, 32] - - -def main(): - Q = StackBuf(2) - hint_witness(Q, \"w\") - assert log(Q[0]) < 8 - p = GEN ** 0 - p[1] = Q[0] - p[GEN] = Q[1] - return -", - ); -} - -/// Same rule for a parameter, which is the other half of what the doc reserves. -#[test] -#[should_panic(expected = "reserved")] -fn a_parameter_may_not_shadow_a_constant_array() { - super::build( - "\ -Q = [8, 32] - - -def main(): - r = pick(GEN ** 2) - p = GEN ** 0 - p[1] = r - p[GEN] = r - return - - -def pick(Q): - return Q * GEN -", - ); -} - -/// A `StackBuf` target's index is bounds-checked. The arms write their return -/// straight into that cell, so an unchecked index puts a callee's return into -/// whatever buffer follows: with the check removed, a program that never assigns -/// `b[0]` publishes a value from the arms and PROVES IT, which is the shape -/// `copy_alias` had before its own bounds check went in. -#[test] -#[should_panic(expected = "out of bounds")] -fn a_stackbuf_target_index_is_bounds_checked() { - super::build( - "\ -def main(): - a = StackBuf(2) - b = StackBuf(2) - a[0] = 0 - a[1] = 0 - x = GEN ** 1 - a[2], e = match(log(x), range(0, 2), lambda i: two(x, i)) - p = GEN ** 0 - p[1] = b[0] - p[GEN] = e - return - - -def two(v, k: Const): - return v * GEN ** k, GEN ** k -", - ); -} - -/// A `blake2s` input operand written as a list is the same hash as gathering the -/// words into a buffer, so hashing one way and the other must agree. -/// -/// Self-comparing on purpose: an equivalence pair cannot check this, because a -/// trial that must be ACCEPTED has to name the digest and no test should carry a -/// hash constant. Asserting the two digests equal needs no constant, and the two -/// operands are DIFFERENT words so that reordering within a list is visible: with -/// both operands equal the swap would cancel out. -#[test] -fn a_blake2s_word_list_hashes_like_the_buffer_it_replaces() { - check_case(&Case { - name: "a_blake2s_word_list_hashes_like_the_buffer_it_replaces", - src: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - named = StackBuf(2) - blake2s([v[0], v[1]], [v[1], v[0]], named) - l = StackBuf(2) - l[0] = v[0] - l[1] = v[1] - r = StackBuf(2) - r[0] = v[1] - r[1] = v[0] - gathered = StackBuf(2) - blake2s(l, r, gathered) - assert named[0] == gathered[0] - assert named[1] == gathered[1] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return -", - valid: Trial::new([k(11), k(22)]).stream("w", vec![vec![k(11), k(22)]]), - pokes: vec![wit("w", 0, k(12)), wit("w", 1, k(23))], - }); -} - -/// A frame STORE's index is bounds-checked, like a read's and a target's. -/// -/// The three callers of `frame_cell` each need their own case: with the check -/// removed at the store site alone, all 148 tests still passed, and `a[2] = …` on -/// a `StackBuf(2)` wrote the next buffer's first cell, published it, and the proof -/// VERIFIED. That is the `copy_alias` bug reappearing at a different caller. -#[test] -#[should_panic(expected = "out of bounds")] -fn a_frame_store_index_is_bounds_checked() { - super::build( - "\ -def main(): - a = StackBuf(2) - b = StackBuf(2) - a[0] = 0 - a[1] = 0 - b[1] = 0 - a[2] = GEN ** 7 - p = GEN ** 0 - p[1] = b[0] - p[GEN] = b[1] - return -", - ); -} - -/// One spelling of a heap index must not name two different cells. -/// -/// A heap index is a g-power: `buf[GEN ** k]` is cell `k`, and a plain integer is -/// rejected because `buf[4]` reads as cell 2 (`4 = g^2`) while the slice -/// `buf[4:4+2]` reads as cells 4 and 5. Only a LITERAL carries that g-power -/// reading, and briefly `const(...)` and `len(...)` carried it too, so -/// `buf[const(8)]` on a `HeapBuf(4)` compiled and aliased cell 3 while the bare -/// `buf[8]` it means was rejected. The golden digests cannot see this: the guest's -/// only `const(...)` uses are blake2s operands, not indexes. -#[test] -fn an_integer_heap_index_is_rejected_however_it_is_spelled() { - for idx in ["8", "const(8)", "const(4 + 4)", "len(EIGHT)", "GEN * const(4)"] { - let src = format!( - "EIGHT = [0, 0, 0, 0, 0, 0, 0, 0]\n\ndef main():\n buf = HeapBuf(4)\n buf[{idx}] = 9\n p = GEN ** 0\n p[1] = GEN ** 0\n p[GEN] = GEN ** 0\n return\n" - ); - let Err(err) = std::panic::catch_unwind(|| super::build(&src)) else { - panic!("`buf[{idx}]` was accepted as a heap index"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!( - msg.contains("plain integer naming cell") || msg.contains("not a g-power"), - "`{idx}`: wanted the ambiguity guard, got `{msg}`" - ); - } -} - -/// A large `**` exponent costs its LOG, not its value. -/// -/// The field reading of `b ** k` is computed whether or not the caller wants it, -/// and `field_pow` multiplied `k` times, so this program (which compiles: the -/// index is `1`) took 39 seconds. Square-and-multiply makes it under a -/// millisecond, and a test that would otherwise hang is the way to keep it so. -#[test] -fn a_large_exponent_does_not_cost_its_value() { - let src = "def main():\n sa = StackBuf(2)\n sa[1 ** 4294967295] = 7\n p = GEN ** 0\n p[1] = sa[0]\n p[GEN] = GEN ** 0\n return\n"; - let started = std::time::Instant::now(); - let _ = super::build(src); - let took = started.elapsed(); - assert!( - took < std::time::Duration::from_secs(2), - "a u32::MAX exponent took {took:?}: field_pow is multiplying k times again" - ); -} diff --git a/crates/lean_compiler/tests/suite/soundness/mod.rs b/crates/lean_compiler/tests/suite/soundness/mod.rs deleted file mode 100644 index 48220a10f..000000000 --- a/crates/lean_compiler/tests/suite/soundness/mod.rs +++ /dev/null @@ -1,272 +0,0 @@ -//! Compiler-soundness harness: does the emitted bytecode still carry every -//! constraint the source asked for? -//! -//! A dropped constraint is invisible to ordinary tests. The happy path passes -//! either way, the compiler emits no diagnostic, and the symptom only appears as -//! a proof that accepts something it should not. So the three checks below all -//! attack the *absence* of a constraint rather than the presence of a value. -//! -//! 1. [`check_case`], **perturbation**: one valid trial that must run, and a -//! table of single-cell pokes at the public input or a witness stream, each of -//! which must make the run fail. A dropped assertion shows up as a poke that -//! is accepted. (This is the shape of `leanVM`'s own soundness suite.) -//! 2. [`check_pair`], **equivalence**: two spellings the language documents as -//! interchangeable must accept exactly the same trials. Every dropped-constraint -//! bug found so far is an *asymmetry*: an assertion that survives one spelling -//! and vanishes in the other, so comparing the two finds it without anyone -//! having to guess which side is wrong. -//! 3. [`Execution::unconstrained_reads`], **unconstrained reads**, asserted on -//! every accepting run of both layers above. A cell an instruction read that -//! nothing ever wrote is a live value from outside the constraint system. -//! -//! The three are complementary, and a fix should land with whichever one catches -//! it. Layer 3 sees a dropped store whose cell is then *read* (the value came from -//! nowhere); layer 2 sees a dropped store whose cell is then *ignored* (the value -//! came from the alias instead, and the physical write is orphaned), and layer 3 is -//! blind to that one, because nothing reads the orphan. Layer 1 needs a program -//! whose assertion the poke can violate, and in exchange it needs no second -//! spelling to compare against. - -#![allow(dead_code)] - -use lean_compiler::{compile_without_filler, parse}; -use lean_vm::cpu::Program; -use primitives::field::{F64, F192, g_pow}; - -mod cases; -mod pairs; - -/// `g^k` as a machine word, the way every index, address and counter is written. -pub fn g(k: usize) -> F192 { - F192::from(g_pow(k)) -} - -/// A K-valued literal in the low lane. -pub fn k(x: u64) -> F192 { - F192::from(F64(x)) -} - -/// One `hint_witness` stream: the name, then one entry per call naming it. -pub type Stream = (&'static str, Vec>); - -/// Everything a run consumes: the public statement and the prover's advice. -#[derive(Clone)] -pub struct Trial { - pub pi: [F192; 2], - pub streams: Vec, -} - -impl Trial { - pub fn new(pi: [F192; 2]) -> Self { - Self { - pi, - streams: Vec::new(), - } - } - - /// Add a stream whose every call takes one entry of `cells`. - pub fn stream(mut self, name: &'static str, entries: Vec>) -> Self { - self.streams.push((name, entries)); - self - } - - fn poke(&self, p: &Poke) -> Self { - let mut t = self.clone(); - match *p { - Poke::Pi { slot, to } => t.pi[slot] = to, - Poke::Wit { name, entry, cell, to } => { - let s = t - .streams - .iter_mut() - .find(|(n, _)| *n == name) - .unwrap_or_else(|| panic!("no stream `{name}` in this trial")); - s.1[entry][cell] = to; - } - } - t - } -} - -/// A single-cell mutation of a trial. One cell, so a poke that is accepted names -/// exactly the constraint that is missing. -#[derive(Clone, Copy)] -pub enum Poke { - /// Public-input word 0 or 1. - Pi { slot: usize, to: F192 }, - /// Cell `cell` of entry `entry` of witness stream `name`. - Wit { - name: &'static str, - entry: usize, - cell: usize, - to: F192, - }, -} - -impl Poke { - fn label(&self) -> String { - match self { - Poke::Pi { slot, to } => format!("pi[{slot}] := {:x}:{:x}:{:x}", to.c2, to.c1, to.c0), - Poke::Wit { name, entry, cell, to } => { - format!("{name}[{entry}][{cell}] := {:x}:{:x}:{:x}", to.c2, to.c1, to.c0) - } - } - } -} - -/// Poke a public-input word. -pub fn pi(slot: usize, to: F192) -> Poke { - Poke::Pi { slot, to } -} - -/// Poke cell `cell` of the first entry of stream `name`. -pub fn wit(name: &'static str, cell: usize, to: F192) -> Poke { - Poke::Wit { - name, - entry: 0, - cell, - to, - } -} - -/// Poke cell `cell` of entry `entry` of stream `name`. -pub fn wit_at(name: &'static str, entry: usize, cell: usize, to: F192) -> Poke { - Poke::Wit { name, entry, cell, to } -} - -/// What an honest run of the emitted bytecode did. -pub enum Ran { - /// It completed. Carries the cells it read that nothing ever wrote, which - /// must be empty for the program to mean what its source says. - Ok { unconstrained: Vec }, - /// It failed: a write-once conflict (which is how every `assert` fails), a - /// wild dereference, or any other [`lean_vm::cpu::Fault`]. - Rejected, -} - -impl Ran { - pub fn accepted(&self) -> bool { - matches!(self, Ran::Ok { .. }) - } - fn verb(&self) -> &'static str { - if self.accepted() { "ACCEPTED" } else { "rejected" } - } -} - -/// Compile once. Kept out of [`run`] so a compiler panic is a loud test failure -/// rather than a silent "rejected". -pub fn build(src: &str) -> Program { - compile_without_filler(&parse(src).expect("parse")) -} - -/// Run `program` on `t`. The fill blocks are irrelevant to what the program -/// asserts, so this executes the unfilled build. -pub fn run(program: &Program, t: &Trial) -> Ran { - let mut p = program.clone(); - for (name, entries) in &t.streams { - p.set_witness(*name, entries.clone()); - } - let pi = t.pi; - match p.execute(pi) { - Ok(exec) => Ran::Ok { - unconstrained: exec.unconstrained_reads, - }, - Err(_) => Ran::Rejected, - } -} - -// --------------------------------------------------------------------------- -// Layer 1: perturbation -// --------------------------------------------------------------------------- - -/// One program, one valid trial, and the pokes that must break it. -pub struct Case { - pub name: &'static str, - pub src: &'static str, - pub valid: Trial, - pub pokes: Vec, -} - -pub fn check_case(c: &Case) { - let program = build(c.src); - match run(&program, &c.valid) { - Ran::Ok { unconstrained } => assert!( - unconstrained.is_empty(), - "{}: the valid trial reads cells nothing writes: {unconstrained:?}. \ - A live value came from outside the constraint system, so the lowering \ - dropped a store the source asked for.", - c.name - ), - Ran::Rejected => panic!("{}: the valid trial must run, and did not", c.name), - } - assert!(!c.pokes.is_empty(), "{}: a case with no pokes checks nothing", c.name); - for p in &c.pokes { - assert!( - !run(&program, &c.valid.poke(p)).accepted(), - "{}: poke `{}` was ACCEPTED. The constraint that should have caught it \ - is not in the emitted bytecode.", - c.name, - p.label() - ); - } -} - -// --------------------------------------------------------------------------- -// Layer 2: equivalence -// --------------------------------------------------------------------------- - -/// Two spellings the language documents as interchangeable, and the trials that -/// have to agree. `why` names the guarantee, so a failure reads as a broken -/// promise rather than as a diff. -pub struct Pair<'a> { - pub name: &'static str, - pub why: &'static str, - pub a: &'a str, - pub b: &'a str, - pub trials: Vec, -} - -pub fn check_pair(p: &Pair<'_>) { - let (pa, pb) = (build(p.a), build(p.b)); - assert!(!p.trials.is_empty(), "{}: a pair with no trials checks nothing", p.name); - let (mut agreed_reject, mut agreed_accept) = (false, false); - for (i, t) in p.trials.iter().enumerate() { - let (ra, rb) = (run(&pa, t), run(&pb, t)); - assert_eq!( - ra.accepted(), - rb.accepted(), - "{}: trial {i}: spelling A {} but spelling B {}.\n {}\n\ - One of the two dropped a constraint; the more permissive side is the buggy one.", - p.name, - ra.verb(), - rb.verb(), - p.why - ); - for (which, r) in [("A", &ra), ("B", &rb)] { - if let Ran::Ok { unconstrained } = r { - assert!( - unconstrained.is_empty(), - "{}: trial {i}: spelling {which} reads cells nothing writes: {unconstrained:?}", - p.name - ); - } - } - agreed_reject |= !ra.accepted(); - agreed_accept |= ra.accepted(); - } - // A pair needs a trial of each verdict. All-accepted would pass if both - // spellings dropped everything; all-REJECTED would pass if both spellings - // were nonsense, which is the easier mistake to make, since an accepting - // trial has to name the right answer and a rejecting one does not. - assert!( - agreed_reject, - "{}: every trial was accepted by both spellings, so this pair would pass even \ - if both sides dropped the constraint. Add a trial that must be rejected.", - p.name - ); - assert!( - agreed_accept, - "{}: no trial was accepted, so this pair compares two programs that always fail \ - and would pass however either was broken. Add a trial that must be accepted.", - p.name - ); -} diff --git a/crates/lean_compiler/tests/suite/soundness/pairs.rs b/crates/lean_compiler/tests/suite/soundness/pairs.rs deleted file mode 100644 index 92841635b..000000000 --- a/crates/lean_compiler/tests/suite/soundness/pairs.rs +++ /dev/null @@ -1,540 +0,0 @@ -//! Layer 2: equivalence. Two spellings the language documents as interchangeable -//! must accept exactly the same trials. -//! -//! This is the layer that finds dropped constraints without anyone having to -//! guess where they went. A dropped store is invisible on its own: the program -//! still runs, and its one honest witness still passes. It becomes visible the -//! moment you have a second spelling of the same intent that *kept* the store, -//! because then one side rejects a witness the other accepts, and the more -//! permissive side is the buggy one. -//! -//! Every pair here is a promise `zkDSL.md` makes. When one fails, quote the -//! promise in the bug report; the `why` field is there to be quoted. - -use super::{Pair, Trial, check_pair, g, k}; -use primitives::field::F192; - -/// One hinted pair of cells, published so the trial's public input pins them. -fn two(a: F192, b: F192) -> Trial { - Trial::new([a, b]).stream("w", vec![vec![a, b]]) -} - -/// `@inline` is documented as a pure call-site expansion: "the body is inlined at -/// each call site" with the same semantics as the call. So a function's -/// observable behaviour cannot depend on whether it carries the decorator. -#[test] -fn inline_and_plain_calls_agree() { - let body = "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - assert shift(v[0]) == v[1] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return - - -@INLINE -def shift(x): - return x * GEN -"; - check_pair(&Pair { - name: "inline_and_plain_calls_agree", - why: "zkDSL.md §`@inline`: inlining is a call-site expansion, not a change of meaning.", - a: &body.replace("@INLINE\n", "@inline\n"), - b: &body.replace("@INLINE\n", ""), - trials: vec![ - two(g(3), g(4)), // shift(g^3) = g^4 - two(g(3), g(5)), // rejected by both - two(g(0), g(1)), - two(g(7), g(7)), - ], - }); -} - -/// Write-once memory is the assertion mechanism, so `assert a == b` and two -/// stores of `a` and `b` into one heap cell are the same statement. `zkDSL.md` -/// §Memory: "a second write of the same value is a no-op, of a different value a -/// proof failure. This turns stores into equality assertions". -#[test] -fn assert_eq_and_double_heap_store_agree() { - check_pair(&Pair { - name: "assert_eq_and_double_heap_store_agree", - why: "zkDSL.md §Memory: a store into an already-written cell IS an equality assertion.", - a: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - assert v[0] == v[1] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return -", - b: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - h = HeapBuf(1) - h[1] = v[0] - h[1] = v[1] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return -", - trials: vec![ - two(g(3), g(3)), - two(g(3), g(4)), - two(F192::ZERO, F192::ZERO), - two(F192::ZERO, k(1)), - ], - }); -} - -/// `zkDSL.md` §field: "`/` is runtime field division … the compiler leaves the -/// quotient cell unset and emits the checked relation `quotient · b == a`". So -/// dividing and then comparing must equal comparing the product, wherever the -/// divisor is nonzero (division by zero is documented undefined, so no trial -/// takes it there). -#[test] -fn division_and_checked_product_agree() { - check_pair(&Pair { - name: "division_and_checked_product_agree", - why: "zkDSL.md §field: `a / b` emits exactly the relation `quotient · b == a`.", - a: "\ -def main(): - v = StackBuf(3) - hint_witness(v, \"w\") - q = v[1] / v[0] - assert q == v[2] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[2] - return -", - b: "\ -def main(): - v = StackBuf(3) - hint_witness(v, \"w\") - assert v[2] * v[0] == v[1] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[2] - return -", - trials: vec![ - three(g(3), g(8), g(5)), // g^8 / g^3 = g^5 - three(g(3), g(8), g(6)), // rejected by both - three(k(1), g(9), g(9)), - three(g(2), g(2), g(0)), - ], - }); -} - -fn three(a: F192, b: F192, c: F192) -> Trial { - Trial::new([a, c]).stream("w", vec![vec![a, b, c]]) -} - -/// `unroll` is documented as compile-time unrolling, so a loop and its expansion -/// are the same program. A structural pair: it pins the loop machinery itself -/// rather than any one assertion, which is what catches a lowering that -/// mis-addresses one iteration. -#[test] -fn unroll_and_expansion_agree() { - check_pair(&Pair { - name: "unroll_and_expansion_agree", - why: "zkDSL.md §unroll: the loop is expanded at compile time, so it IS the expansion.", - a: "\ -def main(): - v = StackBuf(1) - hint_witness(v, \"w\") - a = HeapBuf(4) - a[1] = v[0] - for i in unroll(0, 3): - a[GEN ** (i + 1)] = a[GEN ** i] * GEN - p = GEN ** 0 - p[1] = a[GEN ** 3] - p[GEN] = v[0] - return -", - b: "\ -def main(): - v = StackBuf(1) - hint_witness(v, \"w\") - a = HeapBuf(4) - a[1] = v[0] - a[GEN] = a[1] * GEN - a[GEN ** 2] = a[GEN] * GEN - a[GEN ** 3] = a[GEN ** 2] * GEN - p = GEN ** 0 - p[1] = a[GEN ** 3] - p[GEN] = v[0] - return -", - trials: vec![ - one(g(3), g(0)), // a[3] = g^0·g^3 - one(g(4), g(0)), // rejected by both - one(g(8), g(5)), - one(g(5), g(5)), - ], - }); -} - -/// A published pair whose first word is the claim and whose second is the hint. -fn one(published: F192, hint: F192) -> Trial { - Trial::new([published, hint]).stream("w", vec![vec![hint]]) -} - -/// An `@inline` function returning a one-cell `StackBuf`, used in expression -/// position, must hold the value its body stored. `zkDSL.md` §`@inline` makes the -/// decorator a call-site expansion, and §StackBuf makes `s[0]` the cell the body -/// wrote, so binding the call with `let` and using it inline are the same program. -/// -/// Regression test: `take_inline_ret_cell` used to hand back the raw frame cell -/// rather than following the deferred-copy alias, so the caller read a cell no -/// instruction ever wrote. The `assert` then compared that cell instead of the -/// value, which made it vacuous, and the honest runner back-solved the cell to -/// whatever the public statement demanded. -#[test] -fn inline_stackbuf_return_in_expression_position() { - let body = "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - ASSERTION - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return - - -@inline -def pick(x): - s = StackBuf(1) - s[0] = x - return s -"; - check_pair(&Pair { - name: "inline_stackbuf_return_in_expression_position", - why: "zkDSL.md §`@inline` + §StackBuf: `pick(x)[0]` is the cell the body stored `x` into, \ - whether the caller binds the call or writes it inline.", - a: &body.replace("ASSERTION", "assert pick(v[0]) != v[1]"), - b: &body.replace("ASSERTION", "r = pick(v[0])\n assert r[0] != v[1]"), - trials: vec![ - two(g(3), g(5)), // distinct: accepted by both - two(g(3), g(3)), // equal and nonzero: the inequality must fail for both - two(k(1), k(1)), - two(g(7), g(2)), - ], - }); -} - -/// A store into a cell something already gave a value to is the write-once -/// equality assertion of `zkDSL.md` §Memory ("a second write ... of a different -/// value a proof failure. This turns stores into equality assertions"), whether -/// the cell is a `StackBuf` cell or a `HeapBuf` cell. -/// -/// Regression test: `stack_store` deferred a copy-or-constant RHS as an alias -/// unconditionally, so a `StackBuf` store never pinned a hint. `hint_witness` -/// named the raw cells while every read forwarded past them, and the check the -/// author wrote was applied to nothing. -#[test] -fn stack_store_pins_a_hint_like_a_heap_store() { - check_pair(&Pair { - name: "stack_store_pins_a_hint_like_a_heap_store", - why: "zkDSL.md §Memory: a store into an already-written cell IS an equality assertion, \ - and §Hints: `s[k] = ` is how a program pins prover advice.", - a: "\ -def main(): - s = StackBuf(2) - hint_witness(s, \"w\") - s[0] = GEN ** 3 - p = GEN ** 0 - p[1] = s[0] - p[GEN] = s[1] - return -", - b: "\ -def main(): - s = StackBuf(2) - hint_witness(s, \"w\") - h = HeapBuf(1) - h[1] = s[0] - h[1] = GEN ** 3 - p = GEN ** 0 - p[1] = s[0] - p[GEN] = s[1] - return -", - trials: vec![ - pinned(g(3), g(9)), // the hint agrees with the pin - pinned(g(4), g(9)), // it does not: both must reject - pinned(F192::ZERO, g(1)), - pinned(g(2), g(3)), - ], - }); -} - -/// A trial for the pinning pair above: the public input carries the PIN, not the -/// hint. Publishing the hint would hide a dropped pin, since the publication then -/// forwards through the very alias that dropped it and both spellings agree by -/// accident. Publishing the pin makes a dropped pin visible as a program that -/// accepts every hint. -fn pinned(hint0: F192, hint1: F192) -> Trial { - Trial::new([g(3), hint1]).stream("w", vec![vec![hint0, hint1]]) -} - -/// `zkDSL.md` §BLAKE2s: "If `out` was already written, the statement *asserts* -/// the digest equals it, write-once turning the hash into a verification, which -/// is exactly what a signature verifier wants." That has to hold for a `StackBuf` -/// `out` as much as for a `HeapBuf` one, since the doc recommends the idiom -/// without qualifying which. -/// -/// Regression test: the `BLAKE2s` output arm named the raw run, so a `StackBuf` -/// `out` whose cells had been pre-written by copies or constants had its digest -/// written where nothing read it. The "verification" checked nothing, and the -/// prover could put any message under the hash. -#[test] -fn prewritten_blake2s_out_asserts_the_digest() { - check_pair(&Pair { - name: "prewritten_blake2s_out_asserts_the_digest", - why: "zkDSL.md §BLAKE2s: a pre-written `out` turns the hash into a verification.", - a: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - m = StackBuf(4) - m[0] = 5 - m[1] = 7 - m[2] = 0 - m[3] = 0 - d = StackBuf(2) - d[0] = v[0] - d[1] = v[1] - blake2s(m[0:2], m[2:4], d) - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return -", - b: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - m = StackBuf(4) - m[0] = 5 - m[1] = 7 - m[2] = 0 - m[3] = 0 - d = HeapBuf(2) - d[1] = v[0] - d[GEN] = v[1] - blake2s(m[0:2], m[2:4], d[0:2]) - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return -", - trials: vec![ - // The real digest of the block whose cells are (5, 7, 0, 0). - two(super::cases::DIGEST_5_7[0], super::cases::DIGEST_5_7[1]), - // Anything else must be rejected by both spellings. - two(F192::ZERO, F192::ZERO), - two(super::cases::DIGEST_5_7[0], F192::ZERO), - two(g(3), g(5)), - ], - }); -} - -/// Two stores of different values into one cell is the write-once equality -/// assertion of `zkDSL.md` §Memory, on a `StackBuf` cell as much as on a `HeapBuf` -/// cell. The doc draws no distinction, and the whole "stores are assertions" -/// promise rests on there being none. -#[test] -fn two_stack_stores_to_one_cell_assert_equality() { - check_pair(&Pair { - name: "two_stack_stores_to_one_cell_assert_equality", - why: "zkDSL.md §Memory: a second write of a different value is a proof failure.", - a: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - s = StackBuf(1) - s[0] = v[0] - s[0] = v[1] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return -", - b: "\ -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - h = HeapBuf(1) - h[1] = v[0] - h[1] = v[1] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = v[1] - return -", - trials: vec![two(g(3), g(3)), two(g(3), g(4)), two(g(0), g(0)), two(g(5), g(9))], - }); -} - -/// A store made inside a runtime branch into a cell that already carried a value -/// from before the branch is the same assertion whether the cell is a `StackBuf` -/// cell or a `HeapBuf` cell. `zkDSL.md` §`if`: "branches communicate through -/// write-once cells: only one branch executes, so both may write the *same* cell", -/// and §Memory makes a second write of a different value a failure. -/// -/// Regression test: `scoped` materialized the branch's value into the cell and -/// then restored the pre-branch alias over it, so post-join reads forwarded to the -/// pre-branch source on every path and the materialized write was orphaned. The -/// published value was the pre-branch one whichever arm ran. -#[test] -fn store_inside_a_branch_asserts_against_the_pre_branch_value() { - check_pair(&Pair { - name: "store_inside_a_branch_asserts_against_the_pre_branch_value", - why: "zkDSL.md §`if` + §Memory: both arms may write one cell, and a second write \ - of a different value is a proof failure.", - a: "\ -def main(): - v = StackBuf(3) - hint_witness(v, \"w\") - s = StackBuf(1) - s[0] = v[0] - if v[1] == v[2]: - s[0] = v[1] - else: - s[0] = v[2] - p = GEN ** 0 - p[1] = s[0] - p[GEN] = v[0] - return -", - b: "\ -def main(): - v = StackBuf(3) - hint_witness(v, \"w\") - h = HeapBuf(1) - h[1] = v[0] - if v[1] == v[2]: - h[1] = v[1] - else: - h[1] = v[2] - p = GEN ** 0 - p[1] = h[1] - p[GEN] = v[0] - return -", - trials: vec![ - // else arm, and v[0] != v[2]: the assertion must fail for both. - branch3(g(1), g(2), g(3)), - // else arm, and v[0] == v[2]: accepted by both. - branch3(g(1), g(2), g(1)), - // then arm, and v[0] == v[1]: accepted by both. - branch3(g(1), g(1), g(1)), - // then arm, and v[0] != v[1] (v[1] == v[2] picks it): must fail. - branch3(g(1), g(4), g(4)), - ], - }); -} - -/// A trial for the branch pair: publishes `s[0]` and `v[0]`, which the assertion -/// makes equal on every accepting path. -fn branch3(a: F192, b: F192, c: F192) -> Trial { - Trial::new([a, a]).stream("w", vec![vec![a, b, c]]) -} - -/// A multi-value target may be a `StackBuf` element, and `zkDSL.md` §match -/// says the arms then write straight into it. That is only a saved copy, so it -/// must accept exactly what the name-plus-store spelling accepts. The bug this -/// guards is the arms writing PAST the buffer: the target cell is the callee's -/// return slot now, so an unchecked index puts a return into the next buffer, -/// which is the shape `copy_alias` had before its bounds check was added. -#[test] -fn a_stackbuf_target_and_a_name_plus_store_agree() { - let body = "\ -def pick(x, i: Const): - return x * GEN ** i, GEN ** i - -def main(): - v = StackBuf(2) - hint_witness(v, \"w\") - sb = StackBuf(2) - sb[1] = 0 - TARGET - p = GEN ** 0 - p[1] = sb[0] - p[GEN] = e - return -"; - check_pair(&Pair { - name: "a_stackbuf_target_and_a_name_plus_store_agree", - why: "zkDSL.md §match: a target may be a name or a StackBuf element; the element form \ - only saves the copy.", - a: &body.replace( - "TARGET", - "sb[0], e = match(log(v[0]), range(0, 2), lambda i: pick(v[1], i))", - ), - b: &body.replace( - "TARGET", - "t, e = match(log(v[0]), range(0, 2), lambda i: pick(v[1], i))\n sb[0] = t", - ), - trials: vec![ - Trial::new([g(4), g(0)]).stream("w", vec![vec![g(0), g(4)]]), - Trial::new([g(5), g(1)]).stream("w", vec![vec![g(1), g(4)]]), - Trial::new([g(4), g(1)]).stream("w", vec![vec![g(1), g(4)]]), - Trial::new([g(9), g(0)]).stream("w", vec![vec![g(0), g(4)]]), - ], - }); -} - -/// An `@inline` arm expands into the dispatching frame, its locals over cells the -/// other arms share, where a plain one fuses into a real call. Both must accept -/// the same trials. Arm `n` allocates `n` locals and the join allocates after -/// them, so a cell the arms or the join wrongly share shows up here. -#[test] -fn an_inline_match_arm_and_a_called_one_agree() { - let body = "\ -def main(): - v = StackBuf(3) - hint_witness(v, \"w\") - assert log(v[0]) < 4 - r, e = match(log(v[0]), range(0, 4), lambda i: pw(v[1], i)) - t = r * e - assert t == v[2] - p = GEN ** 0 - p[1] = v[0] - p[GEN] = t - return - - -@INLINE -def pw(x, n: Const): - y = x - for j in unroll(0, n): - y = y * x - return y, GEN ** n -"; - // Arm n publishes t = x^(n+1)·g^n. - let arm = |n: usize, x: usize, t: usize| Trial::new([g(n), g(t)]).stream("w", vec![vec![g(n), g(x), g(t)]]); - check_pair(&Pair { - name: "an_inline_match_arm_and_a_called_one_agree", - why: "zkDSL.md §`@inline`: an inlined arm is a call-site expansion, not a change of meaning.", - a: &body.replace("@INLINE\n", "@inline\n"), - b: &body.replace("@INLINE\n", ""), - trials: vec![ - arm(2, 1, 5), - arm(0, 3, 3), - arm(3, 2, 11), - arm(2, 1, 6), - arm(1, 1, 5), - arm(4, 1, 9), - ], - }); -} diff --git a/crates/lean_compiler/tests/suite/stack_bits.rs b/crates/lean_compiler/tests/suite/stack_bits.rs deleted file mode 100644 index 579784750..000000000 --- a/crates/lean_compiler/tests/suite/stack_bits.rs +++ /dev/null @@ -1,222 +0,0 @@ -//! Computed-advice bit buffers in the frame: `hint_decompose_bits` into a -//! `StackBuf`, and `addr()` naming the run so it can still be indexed through a -//! pointer (at a runtime index, or from a callee). - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{Fault, Stats, prove, verify}; -use primitives::field::{F64, F192}; - -const V: u64 = 0b1011_0110; - -fn deref_index() -> usize { - Stats::TABLES.iter().position(|&t| t == "DEREF").expect("a DEREF table") -} - -/// The same eight-bit decomposition, once through a `HeapBuf` and once through a -/// `StackBuf`. Every index is compile-time, so the frame run needs no `DEREF` at -/// all where the heap one needs two per bit (the read and the booleanity pin). -#[test] -fn a_frame_bit_buffer_costs_no_deref() { - let src = |decl: &str, idx: &str| { - format!( - "\ -def main(): - bits = {decl} - hint_decompose_bits(bits, {V}, 8) - acc = 0 - for i in unroll(0, 8): - b = bits[{idx}] - bits[{idx}] = b * b - acc += b * (2 ** i) - assert acc == {V} - p = 1 - p[1] = acc - p[GEN] = 1 - return -" - ) - }; - let want = [F192::from(F64(V)), F192::from(F64::ONE)]; - let deref = |s: &str| crate::common::mix(s, want)[deref_index()]; - assert_eq!( - deref(&src("HeapBuf(GEN ** 8)", "GEN ** i")) - deref(&src("StackBuf(8)", "i")), - 16, - "a frame bit buffer must drop both DEREFs per bit" - ); -} - -/// A stack bit run is addressed by CONTIGUITY, so no cell of one may be given -/// away to a duplicate elsewhere: `hint_log2_ceil` reads `fp+base+k` whatever the -/// lowerer decided, so a dropped store would leave it holding nothing. The -/// duplicate `MUL` here comes FIRST, which is the order that would make the store -/// the one dropped. -#[test] -fn a_stack_bit_run_survives_cell_sharing() { - let src = "\ -def main(): - src = StackBuf(4) - hint_witness(src, \"bits\") - bits = StackBuf(4) - for i in unroll(0, 4): - dup = src[i] * src[i] - bits[i] = src[i] * src[i] - assert dup == bits[i] - g_mu = hint_log2_ceil(bits, 4, 0) - p = 1 - p[1] = g_mu - p[GEN] = 1 - return -"; - // bits 1011 = 11, whose ceil-log is 4. Executing is enough: a folded-away - // store leaves its cell unwritten, and the advice then computed off the hole - // collides with the published public input. - let mut program = compile(&parse(src).expect("parse")); - let bits: Vec = [1u64, 1, 0, 1].iter().map(|&b| F192::from(F64(b))).collect(); - program.set_witness("bits", vec![bits]); - let want = [F192::from(primitives::field::g_pow(4)), F192::from(F64::ONE)]; - let exec = program.execute(want).unwrap(); - assert!( - exec.unconstrained_reads.is_empty(), - "every cell of the run must still be written" - ); -} - -/// `addr(sb)` in a non-`main` function, where naming `fp` costs the two-`DEREF` -/// bounce: the pointer reads the very cells the direct indices wrote, at a -/// compile-time offset and at a runtime one. -#[test] -fn addr_names_the_frame_run() { - let src = "\ -def main(): - w = StackBuf(1) - hint_witness(w, \"v\") - out = probe(w[0]) - p = 1 - p[1] = out - p[GEN] = 1 - return - -def probe(v): - bits = StackBuf(8) - hint_decompose_bits(bits, v, 8) - ptr = addr(bits) - acc = 0 - for i in unroll(0, 8): - b = bits[i] - bits[i] = b * b - acc += b * (2 ** i) - assert acc == v - total = 0 - for i in unroll(0, 8): - total += ptr[GEN ** i] * (2 ** i) - assert total == v - for x in mul_range(1, GEN ** 8): - chk = ptr[x] - assert chk * chk == chk - assert addr(bits) == ptr - return total -"; - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("v", vec![vec![F192::from(F64(V))]]); - let want = [F192::from(F64(V)), F192::from(F64::ONE)]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the pointer reads the frame run"); - let bad = [F192::from(F64(V + 1)), F192::from(F64::ONE)]; - assert!(verify(&program, &bad, &proof).is_err(), "a wrong value is rejected"); -} - -/// A store into a cell the program has ALREADY written through an `addr()` -/// pointer is the write-once equality assertion of `zkDSL.md` §Memory, not an -/// assembly copy. `addr()` hands out an ordinary `GAddr`, so a write through it -/// is a `DEREF` whose only `phys`-recorded cell is its source; the run's own -/// cells looked untouched, the store deferred as an alias and emitted nothing, -/// and the program went on to "prove" that one cell held two different values. -#[test] -fn a_store_after_a_write_through_addr_still_asserts() { - let src = "\ -def main(): - b = StackBuf(1) - p = addr(b) - v = GEN ** 7 - p[1] = v - b[0] = GEN ** 9 - return -"; - let err = compile(&parse(src).expect("parse")) - .execute([F192::ZERO; 2]) - .err() - .expect("the run must fail"); - assert!(matches!(err.fault, Fault::Conflict { .. }), "{err}"); -} - -/// The same hazard for a run declared AFTER the escape, which the test above -/// does not reach: `unsealed_runs` is already `None` by then, so `alloc_stack` -/// has to seal the run on the spot. Left transparent, `b[0] = GEN ** 9` defers -/// as a constant alias, the assert folds to `const == const`, and the program -/// proves that a cell holding `g^7` holds `g^9`. -#[test] -fn a_run_declared_after_the_escape_is_sealed_too() { - let src = "\ -def poke(q): - q[GEN ** 3] = GEN ** 7 - return 0 - -def main(): - a = StackBuf(2) - pa = addr(a) - b = StackBuf(1) - z = poke(pa) - b[0] = GEN ** 9 - assert b[0] == GEN ** 9 - return -"; - let err = compile(&parse(src).expect("parse")) - .execute([F192::ZERO; 2]) - .err() - .expect("the run must fail"); - assert!(matches!(err.fault, Fault::Conflict { .. }), "{err}"); -} - -/// A frame pointer carries the same compile-time bound a `HeapBuf` pointer gets. -/// `check_heap_bound` keys `heap_sizes` by the pointer's own cell, but every -/// `addr()` in a function shares the `fp` cell as its base, so the run has to -/// come from the address's own provenance instead. -#[test] -#[should_panic(expected = "out of bounds")] -fn addr_pointers_are_bounds_checked() { - let src = "\ -def main(): - b = StackBuf(2) - p = addr(b) - p[GEN ** 40] = GEN ** 3 - return -"; - compile(&parse(src).expect("parse")); -} - -/// The pointer `addr()` hands out addresses the WHOLE frame, not the run it was -/// taken from, so sealing only that run leaves the next `StackBuf` transparent: -/// one off-by-one in a callee writes it, and its later store then defers as an -/// alias and drops the write-once assertion, exactly as before the fix. An -/// escaped frame address therefore seals every run in the function. -#[test] -fn an_escaped_frame_address_seals_every_run() { - let src = "\ -def poke(q): - q[GEN ** 2] = GEN ** 7 - return 0 - -def main(): - a = StackBuf(2) - b = StackBuf(1) - z = poke(addr(a)) - b[0] = GEN ** 9 - assert b[0] == GEN ** 9 - return -"; - let err = compile(&parse(src).expect("parse")) - .execute([F192::ZERO; 2]) - .err() - .expect("the run must fail"); - assert!(matches!(err.fault, Fault::Conflict { .. }), "{err}"); -} diff --git a/crates/lean_compiler/tests/suite/stack_buf.rs b/crates/lean_compiler/tests/suite/stack_buf.rs deleted file mode 100644 index 3a2405e3a..000000000 --- a/crates/lean_compiler/tests/suite/stack_buf.rs +++ /dev/null @@ -1,813 +0,0 @@ -//! `StackBuf`: a run of consecutive frame (stack) cells in the zkDSL. Indexed -//! reads/writes go straight to `base+k` (no heap deref), and a size-2 `StackBuf` -//! is a `blake2s` operand: its two canonical 128-bit cells hold the 256-bit value, so -//! `blake2s(a, b, out)` reads them in place with no copies (a self-hash -//! `blake2s(h, h, out)` aliases one pair into both input operands) and writes -//! the digest into the pre-allocated pair `out`. -//! -//! Since these DSL scalars are K-embedded F192 cells, a `StackBuf(2)` written -//! cell-by-cell holds the flock words `[v0, 0, v1, 0]` -//!: the reference `compress` is fed that lane layout. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{Op, prove, verify}; -use lean_vm::hash_flock::{compression, digest, metadata, unpack_metadata}; -use lean_vm::vmhash::compress; -use primitives::field::{F64, F192, g_pow}; - -use crate::common::mix; - -/// The two 128-bit digest cells of `compress(a, b)` as `F192`s (lo = word 0/2, -/// hi = word 1/3): what a `blake2s(...)` output `StackBuf(2)` holds cell-by-cell. -fn digest_cells(a: [F64; 4], b: [F64; 4]) -> [F192; 2] { - let d = compress(a, b); - [F192::new(d[0].0, d[1].0, 0), F192::new(d[2].0, d[3].0, 0)] -} - -/// A size-2 `StackBuf` fed to `blake2s` as a self-hash `blake2s(h, h)`, then the -/// digest's two 128-bit cells published to `m[0], m[1]`. Proves and verifies, and -/// a wrong published digest is rejected: so the whole path (StackBuf load → -/// aliased blake2s → stack read → publish) is exercised end-to-end. -#[test] -fn stack_buf_blake2s_self_hash() { - let src = "\ -def main(): - a = StackBuf(2) - a[0] = 5 - a[1] = 7 - c = StackBuf(2) - blake2s(a, a, c) - p = 1 - p[1] = c[0] - p[GEN] = c[1] - return -"; - let program = compile(&parse(src).expect("parse")); - - // Each cell holds one scalar in its low lane, so the hashed words are [5,0,7,0]. - let h = [F64(5), F64(0), F64(7), F64(0)]; - let want = digest_cells(h, h); - - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - assert_eq!(mix(src, want)[5], 1, "one BLAKE2s instruction"); - verify(&program, &want, &proof).expect("StackBuf self-hash verifies"); - - let mut bad = want; - bad[0] += F192::ONE; - assert!(verify(&program, &bad, &proof).is_err(), "wrong digest must be rejected"); -} - -/// Optional BLAKE2s metadata and a memory-supplied chaining value reproduce a -/// standard two-block (80-byte) BLAKE2s hash. -#[test] -fn blake2s_keywords_standard_multiblock() { - let src = "\ -def main(): - block0 = [1, 2, 3, 4] - tail = [5, 0, 0, 0] - cv = StackBuf(2) - blake2s(block0[0:2], block0[2:4], cv, counter=64, final=0) - out = StackBuf(2) - blake2s(tail[0:2], tail[2:4], out, cv=cv, counter=80, final=1) - p = 1 - p[1] = out[0] - p[GEN] = out[1] - return -"; - let program = compile(&parse(src).expect("parse")); - let mut input = Vec::new(); - for value in 1u64..=5 { - input.extend_from_slice(&value.to_le_bytes()); - input.extend_from_slice(&0u64.to_le_bytes()); - } - let d = primitives::hash::hash(&input); - let word = |o: usize| u64::from_le_bytes(d[o..o + 8].try_into().unwrap()); - let want = [F192::new(word(0), word(8), 0), F192::new(word(16), word(24), 0)]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - assert_eq!(mix(src, want)[5], 2); - verify(&program, &want, &proof).expect("standard two-block BLAKE2s verifies"); -} - -/// The same 80-byte hash with its second block's metadata computed at run time, -/// the shape a hash of runtime length needs: the counter's high part is a word -/// the program produced and its low part a compile-time constant, and their set -/// bits are disjoint, so one `XOR` is their integer sum (doc -/// §sec:prog-byte-counter). The hint stands in for the high part a real absorb -/// loop derives from its own counter. -#[test] -fn blake2s_runtime_metadata_matches_the_standard_hash() { - let src = "\ -def main(): - block0 = [1, 2, 3, 4] - tail = [5, 0, 0, 0] - cv = StackBuf(2) - blake2s(block0[0:2], block0[2:4], cv, counter=64, final=0) - high = hint_witness(\"high\") - assert high == 64 - out = StackBuf(2) - blake2s(tail[0:2], tail[2:4], out, cv=cv, md=high + f192(16, 4294967295, 0)) - p = 1 - p[1] = out[0] - p[GEN] = out[1] - return -"; - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("high", vec![vec![F192::new(64, 0, 0)]]); - let mut input = Vec::new(); - for value in 1u64..=5 { - input.extend_from_slice(&value.to_le_bytes()); - input.extend_from_slice(&0u64.to_le_bytes()); - } - let d = primitives::hash::hash(&input); - let word = |o: usize| u64::from_le_bytes(d[o..o + 8].try_into().unwrap()); - let want = [F192::new(word(0), word(8), 0), F192::new(word(16), word(24), 0)]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("a runtime metadata word hashes to the standard digest"); -} - -#[test] -fn blake2s_counter_accepts_full_u64_range() { - let src = "\ -def main(): - block = [1, 2, 3, 4] - out = StackBuf(2) - counter = 18446744073709551615 // 1 - blake2s(block[0:2], block[2:4], out, counter=counter, final=1) - return -"; - let program = compile(&parse(src).expect("parse")); - // The metadata is a memory operand, so what carries the counter is the `SET` - // immediate that wrote the cell the instruction reads. - let md = program - .prog - .iter() - .find_map(|op| match op { - Op::Blake2s { md, .. } => Some(*md), - _ => None, - }) - .expect("BLAKE2s instruction"); - let metadata = program - .prog - .iter() - .find_map(|op| match op { - Op::Set { o, k } if *o == md => Some(*k), - _ => None, - }) - .expect("the metadata cell's SET"); - assert_eq!(unpack_metadata(metadata), (u64::MAX, u32::MAX, 0)); -} - -#[test] -#[should_panic(expected = "counter= 18446744073709551616 does not fit in u64")] -fn blake2s_counter_rejects_values_above_u64() { - let src = "\ -def main(): - block = [1, 2, 3, 4] - out = StackBuf(2) - blake2s(block[0:2], block[2:4], out, counter=18446744073709551616, final=1) - return -"; - compile(&parse(src).expect("parse")); -} - -/// A default IV first materialized in an untaken runtime branch must not leak -/// into the post-join lowering state. Both executions must initialize the IV -/// on the path that reaches the second hash. -#[test] -fn blake2s_default_iv_after_runtime_branch() { - let src = "\ -def main(): - flag = StackBuf(1) - hint_witness(flag, \"flag\") - a = [1, 2, 3, 4] - if flag[0] == 1: - ignored = StackBuf(2) - blake2s(a[0:2], a[2:4], ignored) - out = StackBuf(2) - blake2s(a[0:2], a[2:4], out) - p = 1 - p[1] = out[0] - p[GEN] = out[1] - return -"; - let want = digest_cells([F64(1), F64(0), F64(2), F64(0)], [F64(3), F64(0), F64(4), F64(0)]); - for flag in [0, 1] { - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("flag", vec![vec![F192::new(flag, 0, 0)]]); - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("post-join default IV is initialized on both paths"); - } -} - -/// Each mutually exclusive branch gets a path-local IV initialization when no -/// dominating default-IV hash exists before the branch. -#[test] -fn blake2s_default_iv_in_both_runtime_branches() { - let src = "\ -def main(): - flag = StackBuf(1) - hint_witness(flag, \"flag\") - a = [1, 2, 3, 4] - out = StackBuf(2) - if flag[0] == 1: - blake2s(a[0:2], a[2:4], out) - else: - blake2s(a[0:2], a[2:4], out) - p = 1 - p[1] = out[0] - p[GEN] = out[1] - return -"; - let want = digest_cells([F64(1), F64(0), F64(2), F64(0)], [F64(3), F64(0), F64(4), F64(0)]); - for flag in [0, 1] { - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("flag", vec![vec![F192::new(flag, 0, 0)]]); - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("each branch initializes its default IV"); - } -} - -/// Deferred aliases may expose non-adjacent source words for a syntactically -/// consecutive CV StackBuf. The compiler must materialize that pair because -/// the BLAKE2s opcode carries only one CV base offset. -#[test] -fn blake2s_materializes_aliased_cv_pair() { - let src = "\ -def main(): - msg = [1, 2, 3, 4] - sources = [5, 99, 6] - cv = [sources[0], sources[2]] - out = StackBuf(2) - blake2s(msg[0:2], msg[2:4], out, cv=cv, counter=128) - p = 1 - p[1] = out[0] - p[GEN] = out[1] - return -"; - let program = compile(&parse(src).expect("parse")); - let block = compression( - [F64(1), F64(0), F64(2), F64(0)], - [F64(3), F64(0), F64(4), F64(0)], - [F64(5), F64(0), F64(6), F64(0)], - metadata(128, 0, 0), - ); - let d = digest(&block); - let want = [F192::new(d[0].0, d[1].0, 0), F192::new(d[2].0, d[3].0, 0)]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("materialized custom CV verifies"); -} - -/// A custom CV with the default one-block metadata is not a chained block. -/// A metadata cell inside the digest destination would be read before the digest -/// is stored and re-read from the finished image by the witness, so the two would -/// disagree and the proof would fail its opening with nothing to point at. Every -/// other overlap is a write-once conflict, which does say where it happened. -#[test] -#[should_panic(expected = "md= must not name a cell of the digest destination")] -fn blake2s_metadata_inside_the_destination_is_rejected() { - let src = "\ -def main(): - msg = [1, 2, 3, 4] - out = StackBuf(2) - out[0] = 7 - blake2s(msg[0:2], msg[2:4], out, md=out[0]) - return -"; - let _ = compile(&parse(src).expect("parse")); -} - -/// Require the caller to state the byte counter explicitly. -#[test] -#[should_panic(expected = "blake2s with cv= requires")] -fn blake2s_cv_alone_is_rejected() { - let src = "\ -def main(): - msg = [1, 2, 3, 4] - cv = [5, 6] - out = StackBuf(2) - blake2s(msg[0:2], msg[2:4], out, cv=cv) - return -"; - let _ = compile(&parse(src).expect("parse")); -} - -/// A general (non-blake2s) `StackBuf(3)`: indexed writes, an indexed read feeding -/// an arithmetic write into another slot, then two slots published. Confirms the -/// stack cells are plain consecutive frame cells addressable by index. -#[test] -fn stack_buf_indexing() { - let src = "\ -def main(): - sa = StackBuf(3) - sa[0] = 3 - sa[1] = 4 - sa[2] = sa[0] + sa[1] - p = 1 - p[1] = sa[2] - p[GEN] = sa[1] - return -"; - let program = compile(&parse(src).expect("parse")); - // `+` is XOR: 3 ^ 4 = 7. Published: (sa[2], sa[1]) = (7, 4). - let want = [F192::from(F64(7)), F192::from(F64(4))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - assert_eq!(mix(src, want)[5], 0, "no BLAKE2s here"); - verify(&program, &want, &proof).expect("StackBuf indexing verifies"); -} - -/// A normal (non-`@inline`) function may return a StackBuf. Its cells cross the -/// call boundary through consecutive return slots and bind as a StackBuf in the -/// caller, including through another normal wrapper function. -#[test] -fn normal_function_returns_stackbuf() { - let src = "\ -def main(): - out = forward(5) - p = 1 - p[1] = out[0] + out[1] - p[GEN] = out[2] - return - -def forward(v): - out = make(v) - return out - -def make(v): - out = StackBuf(3) - out[0] = v - out[1] = v + 3 - out[2] = 11 - return out -"; - let program = compile(&parse(src).expect("parse")); - // Field addition is XOR: 5 ^ (5 ^ 3) == 3. - program.execute([F192::from(F64(3)), F192::from(F64(11))]).unwrap(); -} - -/// Tuple returns retain their source-level arity even though a StackBuf member -/// occupies several physical return cells. -#[test] -fn normal_function_returns_stackbuf_and_scalar() { - let src = "\ -def main(): - out, x = make(9) - p = 1 - p[1] = out[0] + out[1] - p[GEN] = x - return - -def make(v): - out = [v, 6] - return out, v + 1 -"; - let program = compile(&parse(src).expect("parse")); - program.execute([F192::from(F64(15)), F192::from(F64(8))]).unwrap(); -} - -/// HeapBuf already crosses a normal call as its one-cell pointer. Allocation -/// happened in the callee, so the caller needs no size metadata to dereference -/// and use the returned buffer. -#[test] -fn normal_function_returns_heapbuf_pointer() { - let src = "\ -def main(): - out = make() - p = 1 - p[1] = out[1] - p[GEN] = out[GEN] - return - -def make(): - out = HeapBuf(2) - out[1] = 17 - out[GEN] = 23 - return out -"; - let program = compile(&parse(src).expect("parse")); - program.execute([F192::from(F64(17)), F192::from(F64(23))]).unwrap(); -} - -/// A StackBuf index literal that does not fit `u32` is rejected at compile time, -/// not silently truncated modulo 2^32 (which would resolve `sa[2^32]` to `sa[0]`). -#[test] -#[should_panic(expected = "does not fit in u32")] -fn stack_buf_index_overflow_rejected() { - let src = "def main():\n sa = StackBuf(2)\n x = sa[4294967296]\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// Rebinding a StackBuf name to a scalar clears the stack binding, so the name -/// is a plain scalar afterward (the old bug left a stale `stacks` entry that made -/// `x` still look like a StackBuf, panicking on scalar use). -#[test] -fn stack_buf_rebind_to_scalar() { - let src = "def main():\n x = StackBuf(2)\n x = 5\n p = 1\n p[1] = x\n p[GEN] = x\n return\n"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(F64(5)), F192::from(F64(5))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("rebound-scalar program verifies"); -} - -/// A StackBuf from the enclosing scope referenced inside a `for` loop cannot be -/// captured; the compiler rejects it with a clear message (not a misleading -/// "unbound variable" from the capture being silently dropped). -#[test] -#[should_panic(expected = "cannot be captured into a `for` loop")] -fn stack_buf_loop_capture_rejected() { - let src = "def main():\n h = StackBuf(2)\n h[0] = 1\n h[1] = 2\n for i in mul_range(1, GEN ** 4):\n x = h[0]\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// An `@inline` may return a `StackBuf` *and* a scalar together (a tuple bind): -/// the `StackBuf` slot aliases its cell run into the caller (zero copies, usable -/// as a StackBuf downstream: here fed straight back into a second call, the -/// MD-chain idiom), while the scalar slot binds a value cell. This is the fused -/// `state, x = read_obs(state, cursor)` shape the recursion guest relies on. -#[test] -fn inline_returns_stackbuf_and_scalar() { - let src = "\ -def main(): - s = StackBuf(2) - s[0] = 5 - s[1] = 7 - s, x = step(s, 9) - s, y = step(s, x) - p = 1 - p[1] = s[0] - p[GEN] = s[1] - return - -@inline -def step(state, v): - tg = StackBuf(2) - tg[0] = v - tg[1] = 3 - nb = StackBuf(2) - blake2s(state, tg, nb) - return nb, v -"; - let program = compile(&parse(src).expect("parse")); - - // Each cell = one scalar in its low lane, so a StackBuf(2) hashes words - // [c0, 0, c1, 0]. x == v == 9 (the scalar return), so both steps use tag 9. - let tag = [F64(9), F64(0), F64(3), F64(0)]; - let s1 = compress([F64(5), F64(0), F64(7), F64(0)], tag); - let s2 = compress(s1, tag); // the returned StackBuf (holding s1's words) fed back in - let want = [F192::new(s2[0].0, s2[1].0, 0), F192::new(s2[2].0, s2[3].0, 0)]; - - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - assert_eq!(mix(src, want)[5], 2, "two BLAKE2s instructions (one per inlined step)"); - verify(&program, &want, &proof).expect("inline StackBuf+scalar tuple return verifies"); - - let mut bad = want; - bad[1] += F192::ONE; - assert!( - verify(&program, &bad, &proof).is_err(), - "wrong published state must be rejected" - ); -} - -/// Deferred stores made by a runtime branch must initialize buffers allocated -/// by the surrounding inline call, including when its tuple result is rebound -/// inside an unrolled loop. -#[test] -fn branch_writes_survive_unrolled_tuple_return() { - let src = "\ -def main(): - public = GEN ** 0 - flag = public[1] - a = [5, 7] - b = [13, 17] - for i in unroll(0, 1): - a, b = select_pair(flag, a, b) - assert a[0] == 13 - assert a[1] == 17 - assert b[0] == 5 - assert b[1] == 7 - return - -@inline -def select_pair(flag, a, b): - first = StackBuf(2) - second = StackBuf(2) - if flag == 0: - first[0] = a[0] - first[1] = a[1] - second[0] = b[0] - second[1] = b[1] - else: - first[0] = b[0] - first[1] = b[1] - second[0] = a[0] - second[1] = a[1] - return first, second -"; - let program = compile(&parse(src).expect("parse")); - program.execute([F192::ONE, F192::ZERO]).unwrap(); -} - -/// An `@inline` may also alias-return a folded **g-address** among its values: -/// `fs, x, cur = step(fs, cur)` hands back the Fiat-Shamir state (StackBuf), the -/// consumed word (scalar), and the ADVANCED cursor (`cursor * GEN`) as a -/// zero-cost folded pointer, so the caller keeps reading through it with no -/// manual `cur *= GEN`. This is the shape `fs_next` uses to walk the stream. -#[test] -fn inline_returns_advanced_cursor() { - let src = "\ -def main(): - hb = HeapBuf(4) - hb[1] = 10 - hb[GEN] = 20 - hb[GEN ** 2] = 30 - fs = StackBuf(2) - fs[0] = 1 - fs[1] = 2 - cur = hb - fs, a, cur = step(fs, cur) - fs, b, cur = step(fs, cur) - v = cur[GEN ** 0] - p = 1 - p[1] = a + b - p[GEN] = v - return - -@inline -def step(state, cursor): - x = cursor[GEN ** 0] - tg = StackBuf(2) - tg[0] = x - tg[1] = 3 - nb = StackBuf(2) - blake2s(state, tg, nb) - return nb, x, cursor * GEN -"; - let program = compile(&parse(src).expect("parse")); - // a = hb[0] = 10, b = hb[1] = 20, v = hb[2] = 30 read through the cursor - // returned twice-advanced. a + b is XOR: 10 ^ 20 = 30. - let want = [F192::from(F64(30)), F192::from(F64(30))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("inline advanced-cursor return verifies"); -} - -/// `x = [a, b, c, d]`: the list-literal StackBuf initializer: allocates the run -/// and writes the elements in place, sugar for alloc-then-store. The test mixes a -/// runtime value, a constant, and an expression; feeds the result to blake2s; and -/// swaps a buffer through itself (`s = [s[1], s[0], …]` reads the OLD binding, -/// per the let-rebind rule). -#[test] -fn stack_buf_list_literal() { - let src = "\ -def main(): - s = [5, 7] - s = [s[1], s[0]] - t = [s[0] + s[1], 3] - out = StackBuf(2) - blake2s(s, t, out) - p = 1 - p[1] = out[0] - p[GEN] = out[1] - return -"; - let program = compile(&parse(src).expect("parse")); - // s = [7, 5] after the swap → words [7,0,5,0]; t = [7 ^ 5, 3] = [2, 3] → [2,0,3,0]. - let want = digest_cells([F64(7), F64(0), F64(5), F64(0)], [F64(2), F64(0), F64(3), F64(0)]); - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - assert_eq!(mix(src, want)[5], 1, "one BLAKE2s instruction"); - verify(&program, &want, &proof).expect("list-literal StackBuf verifies"); -} - -/// A list literal anywhere but the RHS of an assignment is rejected with a -/// clear message, not lowered as a phantom scalar. -#[test] -#[should_panic(expected = "a list literal must be bound to a name")] -fn stack_buf_list_literal_as_value_rejected() { - let src = "def main():\n x = 1 + [2, 3]\n assert x == x\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// A compile-time heap index past the buffer's declared size is a compile -/// error, not a runtime wild deref. -#[test] -#[should_panic(expected = "heap index 8 out of bounds for `hb` (HeapBuf size 8)")] -fn heap_index_oob_rejected() { - let src = "def main():\n hb = HeapBuf(8)\n x = hb[GEN ** 8]\n assert x == x\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// The bound follows shifted aliases back to the original buffer: a pointer -/// alias `row = hb * GEN ** k` checks `row[GEN ** j]` against size − k. -#[test] -#[should_panic(expected = "heap index 9 out of bounds for `hb` (HeapBuf size 8)")] -fn heap_alias_index_oob_rejected() { - let src = "def main():\n hb = HeapBuf(8)\n row = hb * GEN ** 6\n x = row[GEN ** 3]\n assert x == x\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// A hint slice whose end exceeds the buffer is rejected at compile time. -#[test] -#[should_panic(expected = "heap slice 0:9 out of bounds for `hb` (HeapBuf size 8)")] -fn heap_hint_slice_oob_rejected() { - let src = "def main():\n hb = HeapBuf(8)\n hint_witness(hb[0:9], \"w\")\n x = hb[GEN ** 0]\n assert x == x\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// A blake2s heap slice straddling the buffer end is rejected. The 256-bit -/// operand `hb[7:9]` is two 128-bit cells, so the bound check trips at -/// `7 + 2 = 9 > 8`. -#[test] -#[should_panic(expected = "heap slice 7:9 out of bounds for `hb` (HeapBuf size 8)")] -fn heap_blake2s_slice_oob_rejected() { - let src = "def main():\n hb = HeapBuf(8)\n hb[GEN ** 7] = 5\n out = StackBuf(2)\n blake2s(hb[7:9], hb[7:9], out)\n return\n"; - let _ = compile(&parse(src).expect("parse")); -} - -/// A heap index that folds to a non-g-power field constant (an integer loop -/// var leaking in from a StackBuf conversion) can never name a heap cell (cell -/// k lives at `buf · g^k`) and used to survive to proving time as a -/// wild-pointer DEREF. It must be a compile-time error. -#[test] -#[should_panic(expected = "not a g-power")] -fn integer_heap_index_is_rejected() { - let src = "\ -def main(): - b = HeapBuf(4) - b[1] = 3 - b[GEN] = 5 - x = 0 - for k in unroll(0, 2): - p = 1 - p[GEN ** k] = b[k] - return -"; - compile(&parse(src).expect("parse")); -} - -/// The last in-bounds index still compiles and runs. -#[test] -fn heap_index_boundary_ok() { - let src = "def main():\n hb = HeapBuf(8)\n hb[GEN ** 7] = 5\n row = hb * GEN ** 4\n y = row[GEN ** 3]\n assert y == 5\n return\n"; - let program = compile(&parse(src).expect("parse")); - let pi = [F192::from(F64(3)), F192::from(F64(4))]; - let (proof, _) = prove(&program, pi, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &pi, &proof).expect("boundary access verifies"); -} - -/// A store into a run PARAMETER is the write-once assertion, not a fresh store. -/// -/// A run parameter's cells are already written, by the caller, before the -/// callee's first instruction; a local `StackBuf`'s are not. That is the whole -/// difference, and missing it dropped the assertion: `s[k] = ` inside a -/// callee recorded a deferred alias and emitted nothing, so the idiom that pins -/// an unconstrained hint pinned nothing and the prover kept its own values. -#[test] -fn a_store_into_a_run_parameter_asserts() { - let pin = "\ -def pin(s: StackBuf(2)): - s[0] = GEN ** 5 - s[1] = GEN ** 6 - return GEN ** 0 - -def main(): - b = StackBuf(2) - hint_witness(b, \"adv\") - z = pin(b) - p = GEN ** 0 - p[1] = b[0] - p[GEN] = b[1] - return -"; - let ast = parse(pin).expect("parse"); - // The honest prover hints what the callee asserts, and it verifies. - let mut program = compile(&ast); - program.set_witness("adv", vec![vec![g_pow(5).into(), g_pow(6).into()]]); - let want = [g_pow(5).into(), g_pow(6).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the honest hint matches the pin"); - - // A prover hinting anything else must be rejected: that is what the pin is. - let mut bad = compile(&ast); - bad.set_witness("adv", vec![vec![g_pow(13).into(), g_pow(14).into()]]); - let dishonest = [g_pow(13).into(), g_pow(14).into()]; - assert!( - bad.execute(dishonest).is_err(), - "the pin must reject a hint it does not match" - ); - - // The same rule with no hint involved: one cell cannot hold two values. - let two = "\ -def f(s: StackBuf(2)): - s[0] = s[1] - return s[0] - -def main(): - b = StackBuf(2) - b[0] = GEN ** 9 - b[1] = GEN ** 3 - r = f(b) - p = GEN ** 0 - p[1] = r - p[GEN] = GEN ** 0 - return -"; - let program = compile(&parse(two).expect("parse")); - let want = [g_pow(3).into(), g_pow(0).into()]; - assert!( - program.execute(want).is_err(), - "`s[0] = s[1]` asserts that they are equal" - ); -} - -/// A multi-cell value can cross a call in BOTH directions. -/// -/// It could always be returned as a run of cells and never passed as one, so a -/// two-cell digest went in through a pointer or an `@inline` expansion while -/// coming back out whole. A `s: StackBuf(n)` parameter takes the same n -/// consecutive cells a `StackBuf(n)` return value occupies, placed by the same -/// `Abi`, which is why the argument area is now a WIDTH rather than a count. -#[test] -fn a_stack_buf_can_be_passed_as_well_as_returned() { - let src = "\ -def swap(s: StackBuf(2)): - t = StackBuf(2) - t[0] = s[1] - t[1] = s[0] - return t - -def main(): - b = StackBuf(2) - b[0] = GEN ** 1 - b[1] = GEN ** 2 - r = swap(b) - p = GEN ** 0 - p[1] = r[0] - p[GEN] = r[1] - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [g_pow(2).into(), g_pow(1).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the run went in and the swapped run came back"); - - // The shape is checked at the call, in both directions of mismatch. - for (arg, want) in [ - ( - "b = StackBuf(3)\n b[0] = GEN ** 1\n r = f(b)", - "got a StackBuf(3)", - ), - ("r = f(GEN ** 1)", "pass one"), - ] { - let src = format!( - "def f(s: StackBuf(2)):\n return s[0]\n\ndef main():\n {arg}\n p = GEN ** 0\n p[1] = r\n p[GEN] = GEN ** 0\n return\n" - ); - let ast = parse(&src).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("accepted: {arg}"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains(want), "got `{msg}`"); - } -} - -/// `g` is `x`, so the literal `2^k` IS `g^k`. `try_gpow_index` always knew that -/// and `gaddr_of` did not, so one field element had three answers: `hb[GEN * 2]` -/// was rejected as "not a g-power" while `hb[GEN * GEN]` compiled, and -/// `hb[r * 2]` compiled again as soon as `r` was runtime. All three name cell 2. -/// A BARE literal index stays rejected: see `integer_heap_index_is_rejected`. -#[test] -fn a_literal_power_of_two_is_a_g_power() { - let src = "\ -def main(): - hb = HeapBuf(8) - hb[GEN * GEN] = GEN ** 5 - p = GEN ** 0 - p[1] = hb[GEN * 2] - p[GEN] = hb[GEN ** 2] - return -"; - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(g_pow(5)); 2]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("three spellings of cell 2 agree"); -} - -/// A `HeapBuf` sized at run time is bounded by the address space alone, not by -/// how far the g-power table happens to reach. -#[test] -fn heap_buf_runtime_size_beyond_the_g_power_table() { - let src = "\ -def main(): - public = 1 - n = public[1] - i = public[GEN] - buf = HeapBuf(n) - buf[i] = 7 - return -"; - let program = compile(&parse(src).expect("parse")); - let size = (1 << 20) + 1; - let exec = program - .execute([F192::from(g_pow(size)), F192::from(g_pow(size - 1))]) - .unwrap(); - assert!(exec.unconstrained_reads.is_empty()); - assert!(exec.mem_used > size); -} diff --git a/crates/lean_compiler/tests/suite/statements.rs b/crates/lean_compiler/tests/suite/statements.rs deleted file mode 100644 index be9f76073..000000000 --- a/crates/lean_compiler/tests/suite/statements.rs +++ /dev/null @@ -1,1058 +0,0 @@ -//! Statement forms that used to be accepted and mean something else. Each of -//! these compiled clean before, so the pin is that they are now REJECTED: a -//! diagnostic is the whole fix. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::{F192, g_pow}; - -/// A name is a plain identifier. Two different mistakes arrive here, and both -/// used to compile: a mis-split statement (`x /= 2` became a binding named -/// `x /`) and a top-level constant reused as a parameter or local. Constants -/// are substituted textually before parsing, so `V = 8` followed by -/// `def scale(V)` gave a parameter literally named `8` and a body reading the -/// constant: `scale(3)` returned 16. -#[test] -fn a_constant_may_not_be_reused_as_a_parameter() { - let src = "\ -V = 8 - -def scale(V): - return V * 2 - -def main(): - p = GEN ** 0 - p[1] = scale(3) - p[GEN] = scale(3) - return -"; - let err = parse(src).expect_err("a parameter may not reuse a constant's name"); - assert!(err.contains("not a valid parameter name"), "{err}"); -} - -/// `/=` and `**=` are not compound assignments here. `split_aug` declined them -/// and `split_assign` then split the bare `=`, leaving a dead binding and the -/// old value in place. `/` being a real runtime field operation is what made -/// `x /= y` look legal. -#[test] -fn an_unsupported_compound_assignment_is_rejected() { - let body = |stmt: &str| { - format!( - "\ -def main(): - x = 4 - {stmt} - p = GEN ** 0 - p[1] = x - p[GEN] = x - return -" - ) - }; - for (stmt, want) in [ - ("x /= 2", "`/=` is not supported"), - ("x **= 2", "`**=` is not supported"), - ] { - let err = parse(&body(stmt)).expect_err(stmt); - assert!(err.contains(want), "{stmt}: {err}"); - } - // The five that ARE supported still desugar. - let ok = body("x += 1\n x *= 2\n x //= 2\n x %= 3\n x -= 1"); - compile(&parse(&ok).expect("the supported compound assignments parse")); -} - -/// A bare comparison is not a statement. `split_assign` used to split `x != y` -/// on its `=` into a binding named `x !`, so an `assert` that lost its keyword -/// to an edit compiled to nothing at all: in a verifier, a deleted check with -/// no diagnostic and no cycle to notice. -#[test] -fn a_bare_comparison_is_rejected() { - let body = |stmt: &str| { - format!( - "\ -def main(): - x = 4 - y = 5 - {stmt} - p = GEN ** 0 - p[1] = x - p[GEN] = x - return -" - ) - }; - for stmt in ["x != y", "x == y", "x <= y", "x >= y", "x < y", "x > y"] { - let err = parse(&body(stmt)).expect_err(stmt); - assert!(err.contains("is a comparison, not a statement"), "{stmt}: {err}"); - } - // The equality forms name the fix; the order forms say they are not predicates. - assert!(parse(&body("x != y")).unwrap_err().contains("write `assert x != y`")); - assert!(parse(&body("x < y")).unwrap_err().contains("order facts come from")); -} - -/// A stack store whose value IS its own destination recorded `alias[dst] = dst`, -/// and `word_src` chased that forever: a compiler that never returns, with no -/// output at all. Two stores could close the same loop in two steps. Both are -/// now no-ops (write-once makes a second write of the same value one), and the -/// documented swap must keep working. -#[test] -fn a_self_referential_stack_store_terminates() { - let cases = [ - // one statement - " s[0] = s[0]\n", - // two, closing the cycle - " s[0] = s[1]\n s[1] = s[0]\n", - ]; - for tail in cases { - let src = format!( - "\ -def main(): - s = StackBuf(2) -{tail} p = 1 - p[1] = s[0] - p[GEN] = s[0] - return -" - ); - compile(&parse(&src).expect("parse")).execute([F192::ZERO; 2]).unwrap(); - } - // `s = [s[1], s[0]]` rebinds to a fresh run and must still swap. - let swap = "\ -def main(): - s = StackBuf(2) - s[0] = 5 - s[1] = 7 - s = [s[1], s[0]] - p = 1 - p[1] = s[0] - p[GEN] = s[1] - return -"; - let want = [ - F192::from(primitives::field::F64(7)), - F192::from(primitives::field::F64(5)), - ]; - compile(&parse(swap).expect("parse")).execute(want).unwrap(); -} - -/// A `mul_range` stop bound that is a compile-time value but not a power of GEN -/// can never be REACHED: the counter walks by multiplication and exits on -/// equality, so the loop ran forever at witness generation with no diagnostic. -/// The `lo` side was always checked, which made `mul_range(0, GEN ** 3)` a clean -/// parse error while `mul_range(1, 10)` was a hang. -#[test] -fn a_loop_bound_must_be_reachable() { - let src = "\ -def main(): - hb = HeapBuf(8) - hb[1] = 1 - for i in mul_range(1, 10): - hb[i * GEN] = hb[i] - return -"; - let err = parse(src).expect_err("an unreachable stop bound must be rejected"); - assert!(err.contains("is not a power of GEN"), "{err}"); - - // A power-of-two literal IS a power of GEN and the walk does reach it, so it - // must be accepted, and as a compile-time bound rather than a runtime one. - let ok = src.replace("mul_range(1, 10)", "mul_range(1, 16)"); - compile(&parse(&ok).expect("`16` is `g^4`")); -} - -/// A binding made inside one arm of an `if` is local to that arm, so the other -/// arm reads the OUTER binding and the loop must capture it. `free_vars_stmt` -/// threaded one flat set through both arms, so a name rebound anywhere in the -/// body counted as loop-local everywhere and the outer binding was never -/// captured: this legal program failed with `unbound variable`. -#[test] -fn a_branch_local_rebinding_does_not_hide_the_outer_binding() { - let src = "\ -def main(): - hb = HeapBuf(8) - w = GEN ** 3 - for i in mul_range(1, GEN ** 4): - if i == GEN: - w = GEN - hb[i * GEN] = w - else: - hb[i * GEN] = w - p = GEN ** 0 - p[1] = hb[GEN ** 2] - p[GEN] = hb[GEN ** 3] - return -"; - // Cell 2 is written on the taken arm (the branch-local `w = GEN`); cell 3 on - // the other, from the outer `w`. - let program = compile(&parse(src).expect("parse")); - let want = [F192::from(g_pow(1)), F192::from(g_pow(3))]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the else arm reads the outer binding"); -} - -/// `g` is `x`, so a literal `2^k` is `g^k` ONLY while `k < 64`: at and above -/// that the modulus folds the monomial back into the low limb while the -/// literal's bit `k` lands in the next one, the tower coefficient of `y`. The -/// guest's own `Y_TOWER` is exactly `2^64`, machine-generated by `dsl_u128`, so -/// a g-power recognizer without that guard gives one literal two values -/// depending on whether it went through a binding. -#[test] -fn a_power_of_two_literal_is_a_g_power_only_below_two_to_the_64() { - let src = "\ -Y_TOWER = 18446744073709551616 - -def main(): - yt = Y_TOWER - a = yt * GEN - b = Y_TOWER * GEN - assert a == b - p = GEN ** 0 - p[1] = a - p[GEN] = b - return -"; - let program = compile(&parse(src).expect("parse")); - // y·x, i.e. the tower coefficient shifted, NOT g^65. - let want = [F192::new(0, 2, 0); 2]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("2^64 is the tower element y, not g^64"); -} - -/// An operator missing a side says which operator and which side. -/// -/// Every one of these used to report `cannot parse expression \`\``: the empty -/// operand was handed to the expression parser, which had nothing to name. A -/// leading `-` is the common one, since the language has no unary minus. -#[test] -fn an_operator_missing_an_operand_says_so() { - for (src, want) in [ - ("-3 + 5", "`-` has no left operand"), - ("1 +", "`+` has no right operand"), - ("* 2", "`*` has no left operand"), - ("4 // ", "`//` has no right operand"), - ] { - let err = lean_compiler::parse_const(src).expect_err(src); - assert!(err.contains(want), "{src}: got `{err}`, wanted `{want}`"); - } -} - -/// One hinted value needs no buffer, and costs exactly what the buffer did. -/// -/// `hint_witness` fills a destination that already exists, so a single hinted -/// scalar cost a one-cell `StackBuf`, a slice of it, and a read back out. The -/// guest declared thirty such buffers, twenty-eight of them for nothing else. -#[test] -fn one_hinted_value_needs_no_buffer() { - let body = |bind: &str| { - format!( - "def main(): -{bind} assert log m < 8 - p = GEN ** 0 - p[1] = m - p[GEN] = GEN ** 0 - return -" - ) - }; - let old = body(" mb = StackBuf(1)\n hint_witness(mb[0:1], \"m\")\n m = mb[0]\n"); - let new = body(" m = hint_witness(\"m\")\n"); - let run = |src: &str| { - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("m", vec![vec![g_pow(5).into()]]); - let want = [g_pow(5).into(), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("verifies"); - program.execute(want).unwrap().base_counts.iter().sum::() - }; - assert_eq!(run(&old), run(&new), "the sugar must cost what it replaces"); - - // It binds inside a loop body too, where substitution walks the statement. - let looped = "\ -def main(): - hb = HeapBuf(4) - for i in mul_range(1, 8): - w = hint_witness(\"w\") - hb[i] = w - p = GEN ** 0 - p[1] = hb[GEN] - p[GEN] = GEN ** 0 - return -"; - let mut program = compile(&parse(looped).expect("parse")); - program.set_witness( - "w", - vec![vec![g_pow(1).into()], vec![g_pow(2).into()], vec![g_pow(3).into()]], - ); - let want = [g_pow(2).into(), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - // `mul_range(1, 8)` is g^0..g^3, so THREE iterations and all three entries - // are popped. With `mul_range(1, 4)` the third would sit unread and mask an - // off-by-one in that direction. - verify(&program, &want, &proof).expect("one hint per iteration"); -} - -/// A trailing factor after a `hint_witness` call may not vanish into the -/// stream name. `call_args` strips the line's LAST `)`, and the string -/// argument then absorbs anything between the call's own `)` and there: -/// `hint_witness(rb[0:1], "a") * f("b")` parsed as a hint for the stream -/// `a") * f("b`, and the rest of the line was gone. Both forms are guarded -/// by `whole_call`; the line then falls through to `parse_expr`, which -/// refuses the string literal, with the line. -#[test] -fn a_hint_call_spans_its_whole_line() { - let stmt = "\ -def main(): - rb = StackBuf(1) - hint_witness(rb[0:1], \"a\") * f(\"b\") - return -"; - let expr = "\ -def main(): - m = hint_witness(\"a\") * f(\"b\") - return -"; - for src in [stmt, expr] { - let err = parse(src).expect_err("a trailing factor must not parse"); - assert!(err.contains("cannot parse expression"), "{err}"); - } -} - -/// The scalar hint binds like any other statement that binds. -/// -/// Three `StmtKind` walkers have a catch-all arm, and each silently swallowed -/// the new variant: `@inline` rejected a body that is a single tail return, -/// the `for` capture check raised a `StackBuf` false positive, and return-shape -/// inference lost the binding. All three fail closed, so the cost was a -/// diagnostic naming the wrong cause rather than a miscompile, and all three -/// made the sugar not a drop-in for the idiom it replaces. -#[test] -fn the_scalar_hint_binds_like_any_other_binder() { - let tail = " p = GEN ** 0\n p[1] = r\n p[GEN] = GEN ** 0\n return\n"; - // `@inline` accepts it: the body IS a single tail return. - let inlined = format!( - "@inline\ndef pick():\n m = hint_witness(\"m\")\n return m\n\ndef main():\n r = pick()\n{tail}" - ); - let mut program = compile(&parse(&inlined).expect("an @inline body may bind a hint")); - program.set_witness("m", vec![vec![g_pow(5).into()]]); - let want = [g_pow(5).into(), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the inlined binding fires"); - - // A `for` body that shadows an enclosing StackBuf's name from inside an arm - // is not capturing it, which is what `binds_anywhere` exists to notice. - let shadowed = "\ -def main(): - sb = StackBuf(2) - sb[0] = GEN ** 1 - sb[1] = GEN ** 2 - hb = HeapBuf(4) - for i in mul_range(1, 4): - if i == i: - sb = hint_witness(\"w\") - hb[i] = sb - p = GEN ** 0 - p[1] = sb[0] - p[GEN] = GEN ** 0 - return -"; - let mut program = compile(&parse(shadowed).expect("shadowing is not capturing")); - program.set_witness("w", vec![vec![g_pow(7).into()], vec![g_pow(8).into()]]); - let want = [g_pow(1).into(), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the outer StackBuf is untouched"); - - // Return-shape inference, the third walker: rebinding a `StackBuf` name to a - // scalar hint has to REPLACE the shape, or the callee is inferred to return a - // run of cells and the caller dies on "StackBuf used as a scalar". Only a - // rebinding reaches it, which is why the other two cases above do not. - let rebound = format!( - "def pick():\n s = StackBuf(2)\n s[0] = GEN ** 1\n s[1] = GEN ** 2\n s = hint_witness(\"m\")\n return s\n\ndef main():\n r = pick()\n{tail}" - ); - let mut program = compile(&parse(&rebound).expect("a hint may rebind a StackBuf name")); - program.set_witness("m", vec![vec![g_pow(6).into()]]); - let want = [g_pow(6).into(), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the rebound name returns as a scalar"); -} - -/// A bit-decomposition hint checks its heap destination, like its stack one. -/// -/// `bits_dest`'s `StackBuf` arm rejected a destination too small for `nbits`; -/// its `HeapBuf` arm checked nothing, so the bits ran on into the next buffer. -/// The same shape as three earlier bugs: one omission beside a checked -/// counterpart. The guest uses the shifted-alias form, so that is checked here -/// too. -#[test] -fn a_bit_decomposition_cannot_overrun_its_heap_destination() { - let prog = |decl: &str, dest: &str| { - format!( - "def main(): - hb = HeapBuf({decl}) - v = GEN ** 5 - hint_decompose_bits_exponent({dest}, v, 8) - p = GEN ** 0 - p[1] = hb[1] - p[GEN] = GEN ** 0 - return -" - ) - }; - for (decl, dest, want) in [("2", "hb", "0:8"), ("8", "hb * GEN ** 4", "4:12")] { - let ast = parse(&prog(decl, dest)).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("HeapBuf({decl}) accepted 8 bits at `{dest}`"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains(want) && msg.contains("out of bounds"), "got `{msg}`"); - } - // Both forms still compile where the buffer really does hold the bits. - compile(&parse(&prog("8", "hb")).expect("parses")); - compile(&parse(&prog("16", "hb * GEN ** 4")).expect("parses")); -} - -/// Four names and calls that used to pick a winner or die without a line. -#[test] -fn a_program_names_what_it_means() { - let tail = " p = GEN ** 0\n p[1] = GEN ** 0\n p[GEN] = GEN ** 0\n return\n"; - // Both bodies used to be lowered, and the last definition won. - let dup = format!("def f(a):\n return a\n\ndef f(a):\n return a\n\ndef main():\n{tail}"); - assert!(parse(&dup).expect_err("duplicate def").contains("defined twice")); - // `f__L1` is what a `Const` specialization of `f` is called, and `__loopN` - // what a loop helper is called, so a user function of that name took its - // place and the call ran the wrong body. - let reserved = format!("def f__L1(a):\n return a\n\ndef main():\n{tail}"); - assert!(parse(&reserved).expect_err("reserved name").contains("reserved")); - - // A call to something that will never be lowered died in the assembler as a - // bare `no entry found for key`, with no line. - for callee in ["nosuchfn(1)", "assert_in_k(GEN ** 1, GEN ** 1)"] { - let src = format!("def main():\n x = {callee}\n{tail}"); - let ast = parse(&src).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("`{callee}` was accepted"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains("no function named"), "{callee}: got `{msg}`"); - } - - // A compile-time constant is capturable into a `for` body: the body becomes - // its own function, so the constant is not in scope there. Dropping it made - // this "unbound variable", which named neither the cause nor a fix. - let captured = "\ -def main(): - c = 5 - hb = HeapBuf(4) - for i in mul_range(1, 4): - hb[i] = c - p = GEN ** 0 - p[1] = hb[GEN] - p[GEN] = GEN ** 0 - return -"; - let program = compile(&parse(captured).expect("a constant may be captured")); - let want = [F192::new(5, 0, 0), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the captured constant reaches the body"); -} - -/// Three ways a program could read a frame or heap cell it does not own. -/// -/// Each compiled clean and left a cell that nothing writes, which the prover -/// then chooses, so an `assert` reading it proved nothing. The first also read -/// the NEXT buffer and its assert passed. -/// -/// They are one omission each, in three places that each had a checked -/// counterpart: the store path's `copy_alias` did not bounds-check its stack -/// index although the identical read in expression position did; `lower_call` -/// did not compare a call's argument count against the callee's parameters -/// although `try_inline` did; and a runtime-start heap slice bounds-checked one -/// cell rather than its length although the compile-time-bounds arm beside it -/// checked the whole span. -#[test] -fn a_program_cannot_reach_a_cell_it_does_not_own() { - let rejected = |src: &str, want: &str| { - let ast = parse(src).unwrap_or_else(|e| panic!("should parse: {e}")); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("accepted:\n{src}"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains(want), "got `{msg}`, wanted `{want}`"); - }; - let tail = " p = GEN ** 0\n p[1] = GEN ** 0\n p[GEN] = GEN ** 0\n return\n"; - - // A store's RHS, and a list literal element, each index one past the end. - rejected( - &format!("def main():\n a = StackBuf(2)\n b = StackBuf(2)\n c = StackBuf(2)\n c[0] = a[2]\n{tail}"), - "index 2 out of bounds (StackBuf size 2)", - ); - rejected( - &format!("def main():\n a = StackBuf(2)\n b = StackBuf(2)\n lst = [a[2], a[3]]\n{tail}"), - "index 2 out of bounds (StackBuf size 2)", - ); - // A call supplying too few arguments, and too many. - rejected( - &format!("def check(a, b):\n assert a == b\n return\n\ndef main():\n check(0)\n{tail}"), - "`check` takes 2 arguments, got 1", - ); - rejected( - &format!("def one(a):\n return a\n\ndef main():\n r = one(GEN ** 1, GEN ** 2)\n{tail}"), - "`one` takes 1 argument, got 2", - ); - // A runtime-start slice whose start folds: the SPAN leaves the buffer. - rejected( - &format!( - "def main():\n hb = HeapBuf(2)\n nxt = HeapBuf(2)\n hint_witness(hb[GEN ** 1:GEN ** 1 + 2], \"w\")\n{tail}" - ), - "heap slice 1:3 out of bounds", - ); - - // The in-bounds run of the same shape still compiles and proves. - let ok = "\ -def main(): - hb = HeapBuf(4) - hint_witness(hb[GEN ** 1:GEN ** 1 + 2], \"w\") - p = GEN ** 0 - p[1] = hb[GEN] - p[GEN] = GEN ** 0 - return -"; - let mut program = compile(&parse(ok).expect("parse")); - program.set_witness("w", vec![vec![F192::new(1234, 0, 0), F192::new(5678, 0, 0)]]); - let want = [F192::new(1234, 0, 0), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("an in-bounds runtime-start slice still works"); -} - -/// A `for` body that assigns to an enclosing name says why that cannot work. -/// -/// The capture set drops every name the body binds, so a body that ASSIGNS to -/// an enclosing name also loses the read that precedes the assignment, and the -/// read arrived at lowering as a bare "unbound variable". That is the loop-carry -/// limitation, not a typo: the tail-recursive helper threads its captures in and -/// never out, so an accumulator cannot come back. The `StackBuf` form of the -/// same limitation already said so; the scalar form did not. -#[test] -fn a_loop_that_cannot_carry_a_value_says_so() { - let accumulator = "\ -def main(): - s = GEN ** 0 - for i in mul_range(1, 8): - s = s * GEN - p = GEN ** 0 - p[1] = s - p[GEN] = GEN ** 0 - return -"; - let ast = parse(accumulator).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("a loop-carried accumulator was accepted"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains("the loop cannot carry it"), "got `{msg}`"); - - // A real typo, outside any loop, still gets the plain message. - let typo = "def main():\n p = GEN ** 0\n p[1] = nosuch\n p[GEN] = GEN ** 0\n return\n"; - let ast = parse(typo).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("a typo was accepted"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!( - msg.contains("unbound variable `nosuch`") && !msg.contains("loop"), - "got `{msg}`" - ); -} - -/// A `match` target is a BINDER, and every field of the statement is walked. -/// -/// Four facts, none of which was pinned by anything in the crate: a name target -/// is not substituted (a `Const k` used to rewrite the binder into a literal, and -/// the target was then rejected); an INDEX target is, since `sb[k]` needs `k`; the -/// scrutinee and the arms are; and the statement REBINDS its name targets, so -/// substitution stops after it. The last is the dangerous one, being the only -/// mutation of the four that produced a wrong published value rather than a -/// diagnostic. -#[test] -fn a_match_target_binds_and_every_field_is_walked() { - let two = "def two(i: Const):\n return GEN ** i, GEN ** i\n\n"; - let publish = " p = GEN ** 0\n p[1] = r\n p[GEN] = GEN ** 0\n return\n"; - // Arm 1 returns (g, g), so the two returns XOR to zero wherever they are read. - let want = [F192::ZERO, g_pow(0).into()]; - let verifies = |src: &str, why: &str| { - let program = compile(&parse(src).unwrap_or_else(|e| panic!("{why}: {e}"))); - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).unwrap_or_else(|e| panic!("{why}: {e:?}")); - }; - - // A `Const` parameter whose name is also a target: the binder survives, and - // what is read after the statement is the DISPATCH's output, not the constant. - verifies( - &format!( - "{two}def pick(x, k: Const):\n k, e = match(log(x), range(0, 2), lambda i: two(i))\n return k + e\n\ndef main():\n r = pick(GEN ** 1, 7)\n{publish}" - ), - "a Const name may also be a target", - ); - - // The scrutinee and the arms ARE substituted: both mention the constant, and a - // missed field shows up as a `Var` reaching a position that needs an integer. - verifies( - &format!( - "{two}def pick(k: Const):\n a, b = match(log(GEN ** k), range(0, 2), lambda i: two(i + k))\n return a + b\n\ndef main():\n r = pick(1)\n{publish}" - ), - "the scrutinee and the arms see the constant", - ); - - // An INDEX target is substituted, so a `Const` index names a cell. - verifies( - &format!( - "{two}def pick(x, k: Const):\n sb = StackBuf(2)\n sb[0] = 0\n sb[k], e = match(log(x), range(0, 2), lambda i: two(i))\n return sb[1] + e\n\ndef main():\n r = pick(GEN ** 1, 1)\n{publish}" - ), - "a Const index target resolves to its cell", - ); - - // An `unroll` counter as a target name obeys the same rule. - let counter = format!( - "{two}def main():\n for j in unroll(0, 2):\n j, e = match(log(GEN ** 1), range(0, 2), lambda i: two(i))\n r = j + e\n{publish}" - ); - let _ = compile(&parse(&counter).expect("an unroll counter may also be a target")); -} - -/// The same construct in a VALUE position, for the same reason. `+` there is -/// XOR, so `lvl + 1` with `lvl = 3` is 2 and not 4, silently: the SPHINCS guest -/// could not write a Merkle level into a tweak and carried a generated table of -/// one literal per level to get the integer reading instead. -/// -/// `const(e)` reads `e` with integer arithmetic and emits the literal, so one -/// construct means one thing in both positions. The two readings must really -/// differ here, or the test proves nothing. -#[test] -fn a_value_may_ask_for_the_integer_regime() { - // Parse and compile OUTSIDE the catch, so a negative arm can only fail on the - // assert. Inside, `!accepted` would also hold if the program stopped parsing. - let accepted = |expr: &str, want: u64| { - let src = format!( - "def main():\n for i in unroll(3, 4):\n v = {expr}\n assert v == {want}\n return\n" - ); - let program = compile(&parse(&src).expect("parse")); - program.execute([F192::ZERO, F192::ZERO]).is_ok() - }; - - // i = 3: the integer reading is 4, the field reading `3 XOR 1` is 2. - assert!(accepted("const(i + 1)", 4), "const(...) is the integer reading"); - assert!(!accepted("const(i + 1)", 2), "and not the field one"); - assert!(accepted("i + 1", 2), "undeclared, `+` in a value position stays XOR"); - assert!(!accepted("i + 1", 4)); - - // Subtraction has no field meaning at all, so it is only reachable this way. - assert!(accepted("const(i - 1)", 2)); - - let rejected = |src: &str, want: &str| { - let ast = parse(src).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("accepted: {src}"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains(want), "wanted `{want}`, got `{msg}`"); - }; - // A `const(...)` that is not a compile-time integer names itself. - rejected( - "def main():\n w = GEN ** 0\n v = const(w + 1)\n return\n", - "compile-time integer", - ); - // The wrapper reinterprets its OPERATORS, and cannot reinterpret a leaf, so a - // leaf whose own two readings diverge is rejected rather than silently read one - // way: `n = 2 + 3` holds `2 XOR 3` = 1 while its integer reading is 5, and - // `assert n == 1` and `assert const(n) == 5` both used to pass in one program. - rejected( - "def main():\n n = 2 + 3\n v = const(n)\n return\n", - "cannot reinterpret", - ); -} - -/// A `def` may not take a builtin's name. The builtin wins at the call site, so -/// the body is never reached and its constraints silently disappear: `def const(x)` -/// with an `assert` in it was skipped outright by `v = const(4)`, and skipped or -/// not depending on whether the ARGUMENT folded, so one call site had two -/// meanings. True of `f192` before `const` existed, so this is the class, not one -/// name. -#[test] -fn a_function_may_not_shadow_a_builtin() { - for name in ["const", "f192", "addr", "blake2s", "len", "hint_witness", "StackBuf"] { - let src = format!("def {name}(x):\n assert x == 99\n return x\n\ndef main():\n return\n"); - let err = parse(&src).expect_err(&format!("`def {name}` must be rejected")); - assert!(err.contains("is a builtin"), "got `{err}`"); - } - // A global CONSTANT of that name is rejected too, and for a sharper reason: a - // scalar constant is substituted textually, so `match = 4` rewrites - // `v = match(log(x), …)` into `4(log(x), …)`. - for name in ["match", "blake2s", "len"] { - let src = format!("{name} = 4\n\ndef main():\n return\n"); - let err = parse(&src).expect_err(&format!("`{name} = 4` must be rejected")); - assert!(err.contains("is a builtin"), "got `{err}`"); - } - // An ordinary name still works, including one that contains a builtin's. - for name in ["helper", "constant", "addr_of"] { - let src = format!("def {name}(x):\n return x\n\ndef main():\n return\n"); - parse(&src).unwrap_or_else(|e| panic!("`def {name}` must be accepted: {e}")); - } -} - -/// A compile-time branch is decided by a regime the author names. -/// -/// The fold decides on the integer reading while the runtime test of the same -/// condition compares field values, so the two contradict each other whenever a -/// side's readings do. `3 + 1` is the integer 4 and the field element -/// `3 XOR 1` = 2, and `if 3 + 1 == 4` used to fold into an arm whose own -/// condition is false as a value; `if K == 4: assert K == 4` compiled clean and -/// died at witness generation. -/// -/// Neither reading can win: deciding in the field breaks `if 1 + 1 == 2` and -/// every `if i + 1 == n` in an `unroll`, and cannot read `-`, `//` or `%` at -/// all. So an ambiguous condition is rejected, and `const(...)` is how the -/// author says the integer regime was meant. -#[test] -fn an_ambiguous_compile_time_branch_must_be_declared() { - let prog = |cond: &str| { - format!( - "def main(): - hb = HeapBuf(4) - if {cond}: - hb[GEN ** 0] = 5 - else: - hb[GEN ** 0] = 7 - p = GEN ** 0 - p[1] = hb[GEN ** 0] - p[GEN] = GEN ** 0 - return -" - ) - }; - let fold = |cond: &str| { - let program = compile(&parse(&prog(cond)).unwrap_or_else(|e| panic!("{cond}: {e}"))); - let want = [F192::new(5, 0, 0), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).unwrap_or_else(|e| panic!("`{cond}` did not take the then arm: {e:?}")); - }; - // Declared, so it folds with integer arithmetic and the arm runs. - fold("const(3 + 1 == 4)"); - fold("const(1 + 1 == 2)"); - fold("const(1 + 1 == 3 - 1)"); - // Only the field can read these, and `const(...)` decides them too. A plain - // `if` must not: folding one would rescope its arm. - fold("const(GEN ** 3 == GEN ** 3)"); - fold("const(2 ** 40 == 2 ** 40)"); - // Undeclared but unambiguous (6 either way), so it folds as it always did. - fold("2 * 3 == 6"); - - // Undeclared and ambiguous: rejected rather than silently decided. The - // last three are the same condition as the first with the OTHER side - // written using an operator the field cannot read, which is how the first - // version of this check let them through: it compared the two sides' - // verdicts, and `try_field_const` has no arm for `-`, `//` or `%`, so one - // missing reading disabled the whole guard. The check is per side now. - for cond in [ - "3 + 1 == 4", - "1 + 1 == 2", - "1 + 1 == 3 - 1", - "1 + 1 == 8 // 4", - "1 + 1 == 9 % 7", - ] { - let src = prog(cond); - let ast = parse(&src).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("`{cond}` was accepted"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains("where a value is wanted"), "{cond}: got `{msg}`"); - } - // A near miss of the wrapper says the word, where the ordinary parse error - // for a malformed condition never would. A condition carrying a comparison of - // its own is NOT a near miss, since `const(...)` is a value expression too. - for cond in ["const (a == b)", "const(a == b"] { - let err = parse(&prog(cond)).expect_err(cond); - assert!(err.contains("must wrap the WHOLE condition"), "{cond}: got `{err}`"); - } - // So one side of an ordinary comparison may be a `const(...)`, in either - // order. Rejecting these made the two operand orders behave differently. - fold("const(1 + 1) == 2"); - fold("2 == const(1 + 1)"); - // A variable that merely starts with `const` is not a near miss. - let plain = "def main():\n const = 4\n hb = HeapBuf(4)\n if const == 4:\n hb[GEN ** 0] = 5\n else:\n hb[GEN ** 0] = 7\n p = GEN ** 0\n p[1] = hb[GEN ** 0]\n p[GEN] = GEN ** 0\n return\n"; - let program = compile(&parse(plain).expect("a name beginning with `const` is an ordinary name")); - let want = [F192::new(5, 0, 0), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("`const == 4` is a comparison, not a wrapper"); - - // Declared, but not actually decidable while compiling. - let src = prog("const(hb == 4)"); - let ast = parse(&src).expect("parses"); - let Err(err) = std::panic::catch_unwind(|| compile(&ast)) else { - panic!("a runtime `if const(...)` was accepted"); - }; - let msg = err.downcast_ref::().map(String::as_str).unwrap_or(""); - assert!(msg.contains("asks for a compile-time decision"), "got `{msg}`"); -} - -/// A string literal is one opaque token. -/// -/// Two passes used to read structure out of the middle of one. Comments were -/// stripped with `raw.split('#')`, so a `#` in a stream name truncated the line, -/// and the shortened line often still parsed. Bracket depth was counted without -/// any notion of a string, so every top-level splitter (arguments, `+`/`-`, -/// `*`//`/`%`, `**`, augmented assignment, comparisons) read a `,` or a `]` -/// spelled inside the name as structure: this call split into three arguments. -#[test] -fn a_string_literal_is_not_scanned_for_syntax() { - let src = "\ -def main(): - rb = StackBuf(1) - hint_witness(rb[0:1], \"x,y#z]w\") - p = GEN ** 0 - p[1] = rb[0] - p[GEN] = GEN ** 0 - return -"; - let mut program = compile(&parse(src).expect("a `,`, `#` or `]` inside a string is part of the name")); - program.set_witness("x,y#z]w", vec![vec![F192::new(7, 0, 0)]]); - let want = [F192::new(7, 0, 0), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the stream name survived parsing intact"); -} - -/// Naming a constant may not change what it means. -/// -/// `K = 3 + 1` folds two ways at once. In the field, where a value expression's -/// constants live, it is `3 XOR 1` = 2 = `g^1`. As a compile-time integer, which -/// is what an index position wants, it is 4. A second g-power recognizer used to -/// match `Expr::Var` against the integer binding and read those bits as an -/// exponent, so `K` was `g^1` as a value and `g^2` as an index: one name, two -/// meanings, in one function. Spelling the constant inline was unaffected, since -/// the integer view only ever reached a *name*. -/// -/// The buffer holds a distinct value per cell, so the proof pins which cell the -/// index named rather than merely that it compiled. -#[test] -fn naming_a_constant_does_not_change_which_heap_cell_it_names() { - let src = "\ -def main(): - rb = StackBuf(1) - hint_witness(rb[0:1], \"r\") - r = rb[0] - hb = HeapBuf(16) - hint_witness(hb[0:4], \"vals\") - K = 3 + 1 - x = hb[(r * r) * K] - p = GEN ** 0 - p[1] = x - p[GEN] = GEN ** 0 - return -"; - let mut program = compile(&parse(src).expect("parse")); - program.set_witness("r", vec![vec![g_pow(0).into()]]); - program.set_witness("vals", vec![(10u64..14).map(|v| F192::new(v, 0, 0)).collect()]); - // `r` is 1, so the index is `K` itself: cell 1, holding 11. Reading the - // integer view instead would name cell 2, holding 12. - let want = [F192::new(11, 0, 0), g_pow(0).into()]; - let (proof, _) = prove(&program, want, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &want, &proof).expect("the index means what the value means"); -} - -/// A loop body that merely SHADOWS an enclosing `StackBuf` never touches it, so -/// rejecting the program names a capture that is not happening. Scoping the -/// arms took the arm-local binding out of the set the rejection consults, which -/// needs the flat "does the body bind this at all" answer instead. -#[test] -fn a_shadowed_stack_buf_is_not_a_capture() { - let src = "\ -def main(): - sa = StackBuf(2) - sa[0] = GEN - sa[1] = GEN ** 2 - hb = HeapBuf(8) - for i in mul_range(1, GEN ** 3): - if i == GEN: - sa = StackBuf(2) - sa[0] = GEN ** 5 - assert sa[0] == GEN ** 5 - else: - hb[i] = GEN ** 7 - assert sa[0] == GEN - return -"; - let exec = compile(&parse(src).expect("parse")).execute([F192::ZERO; 2]).unwrap(); - assert!(exec.unconstrained_reads.is_empty(), "no prover-chosen read"); -} - -/// `lower_if` FOLDS a compile-time condition and runs the taken branch without a -/// scope, so its bindings persist exactly like an `unroll` body's. Modelling it -/// as scoped over-captured, and the loop's own self-call then evaluated a name -/// the folded arm had rebound to a `StackBuf`. -#[test] -fn a_folded_branch_keeps_its_bindings() { - let src = "\ -def main(): - hb = HeapBuf(64) - w = HeapBuf(16) - for i in mul_range(1, GEN ** 3): - if 1 == 1: - w = StackBuf(2) - w[0] = i - w[1] = i * GEN - hb[i] = w[1] - assert hb[1] == GEN - return -"; - let exec = compile(&parse(src).expect("parse")).execute([F192::ZERO; 2]).unwrap(); - assert!(exec.unconstrained_reads.is_empty(), "no prover-chosen read"); -} - -/// Every diagnostic names its source line. The line vector used to carry the -/// INDENT, not the source index, and the collection loop skips blanks, comments -/// and imports without counting, so nothing could name one: against a -/// 3,274-line guest an error read `inconsistent indentation` and no more. -#[test] -fn a_parse_error_names_its_source_line() { - // Counting from 1: blank, comment, constant, blank, blank, def, let, error. - let src = "\n# a comment\nV = 8\n\n\ndef main():\n x = 4\n x != y\n return\n"; - let err = parse(src).unwrap_err(); - assert!(err.starts_with("line 8:"), "{err}"); - - // A constant declaration, above the `def`s and parsed by a different loop. - let konst = "\nN = 1\n\nM = 1 +\n\ndef main():\n return\n"; - assert!(parse(konst).unwrap_err().starts_with("line 4:"), "{konst}"); - - // Inside a block, past a blank line. - let block = - "\n\ndef main():\n a = StackBuf(2)\n\n for i in mul_range(1, 10):\n a[0] = 1\n return\n"; - assert!(parse(block).unwrap_err().starts_with("line 6:")); -} - -/// The four diagnostics raised where no `func`/`stmt` frame is open: those two -/// stamp the line they were ENTERED on, so an enclosing header would be named -/// instead of the line that is wrong. `inconsistent indentation` is the one the -/// whole change exists for, and stamping it from the enclosing frame put it -/// 1,074 lines away from the fault in the real guest. -#[test] -fn an_error_raised_between_frames_still_names_its_line() { - // 1 blank, 2 def, 3 let, 4 if, 5 body, 6 stray deeper indent. - let indent = "\ndef main():\n x = 4\n if x == 4:\n x = 5\n x = 6\n return\n"; - assert!(parse(indent).unwrap_err().starts_with("line 6:"), "{:?}", parse(indent)); - - // 1 blank, 2 @inline, 3 the broken def. - let deco = "\n@inline\nde f(x):\n return x\n"; - assert!(parse(deco).unwrap_err().starts_with("line 3:"), "{:?}", parse(deco)); - - // 1 blank, 2 def, 3 let, 4 if, 5 body, 6 the broken elif. - let elif = "\ndef main():\n x = 4\n if x == 4:\n x = 5\n elif x ~~ 5:\n x = 6\n return\n"; - assert!(parse(elif).unwrap_err().starts_with("line 6:"), "{:?}", parse(elif)); - - // 1 blank, 2 def, 3 let, 4 the match whose ranges are not contiguous. - let arms = "\ndef main():\n x = GEN ** 0\n v = match(log(x), range(0, 1), lambda j: 5, range(2, 3), lambda j: 6)\n return\n"; - assert!(parse(arms).unwrap_err().starts_with("line 4:"), "{:?}", parse(arms)); -} - -/// A replacement carrying a newline shifts every later line, so the numbers -/// above would be wrong and the injected text would land at whatever -/// indentation it fell on. A `#` is the same shape of hazard: it truncates the -/// rest of the line and changes the compiled program with no diagnostic. All -/// 134 of the guest's placeholders are clean; this keeps it that way. -#[test] -fn a_multi_line_placeholder_is_rejected() { - let mut reps = std::collections::BTreeMap::new(); - let src = "FOO = 1\n\ndef main():\n return\n"; - for bad in ["1\nz = 9", "1 #"] { - reps.insert("FOO".to_string(), bad.to_string()); - let err = lean_compiler::parse_with_replacements(src, &reps).unwrap_err(); - assert!(err.contains("would reshape the line"), "{bad}: {err}"); - } -} - -/// A lowering error names its line too, not just a parse error. `lower.rs` works -/// in frame cells and program counters, so the statement's line is the last -/// place that knows where the program said it: 30 diagnostics used to print a -/// Rust `Debug` dump of an AST node and nothing else. -#[test] -fn a_lowering_error_names_its_source_line() { - // 1 blank, 2 def, 3 let, 4 store, 5 blank, 6 let, 7 the unbound read. - let src = "\ndef main():\n hb = HeapBuf(4)\n hb[GEN] = 7\n\n p = GEN ** 0\n p[1] = nope\n return\n"; - let err = std::panic::catch_unwind(|| { - compile(&parse(src).expect("parse")); - }) - .expect_err("an unbound variable must abort"); - let msg = err - .downcast_ref::() - .cloned() - .unwrap_or_else(|| err.downcast_ref::<&str>().map(|s| s.to_string()).unwrap_or_default()); - assert!(msg.starts_with("line 7:"), "{msg}"); -} - -/// And a check that fails at witness generation names it, which is the one that -/// matters day to day: a failed guest `assert` surfaces as a write-once -/// conflict, and `AGENTS.md` used to say to disassemble around the reported pc. -#[test] -fn a_failed_assert_names_its_source_line() { - // 1 blank, 2 def, 3 let, 4 blank, 5 the assert that cannot hold. - let src = "\ndef main():\n x = GEN ** 3\n\n assert x == GEN ** 4\n return\n"; - let program = compile(&parse(src).expect("parse")); - let err = program.execute([F192::ZERO; 2]).err().expect("the assert cannot hold"); - assert!(err.site.contains("line 5"), "{err}"); -} - -/// An `@inline` body lowers through the CALLER's `FnLower`, so its statements -/// move `cur_line`. Without restoring it, every instruction the caller emitted -/// after the call was blamed on whatever line the callee ended on: a line that -/// cannot fail, reported with confidence. -#[test] -fn an_inline_call_does_not_steal_the_call_site_line() { - // 1 blank, 2 @inline, 3 def, 4 let, 5 return, 6-7 blank, 8 def main, ... 12 the assert. - let src = "\n@inline\ndef idf(x):\n y = x * x\n return y\n\n\ndef main():\n hb = HeapBuf(4)\n hb[GEN] = GEN ** 3\n a = hb[GEN]\n assert idf(a) == GEN\n return\n"; - let program = compile(&parse(src).expect("parse")); - let err = program.execute([F192::ZERO; 2]).err().expect("the assert cannot hold"); - assert!( - err.site.contains("line 12"), - "the call site, not the callee's line 5: {err}" - ); -} - -/// The fill blocks are not source code, so they carry the unknown line rather -/// than whatever `main` happened to end on. They are most of a small program's -/// instructions, so blaming them on a real line makes the table mostly wrong. -#[test] -fn fill_blocks_carry_no_source_line() { - let src = "\ndef main():\n hb = HeapBuf(4)\n hb[GEN] = 7\n return\n"; - let program = compile(&parse(src).expect("parse")); - let start = program.filler.first().map(|b| b.pc).expect("main carries fill blocks") as usize; - assert!( - program.src_lines[start..].iter().all(|&l| l == 0), - "a fill block must not be attributed to a source line" - ); - assert!( - program.src_lines[..start].iter().any(|&l| l != 0), - "real code keeps its lines" - ); -} - -/// A `match` join reads one cell per bound name, so a multi-cell `StackBuf` -/// return would bind only the run's FIRST cell and leave the rest where nothing -/// reads them. The guard must check every return, not all of them at once: -/// testing `all(is not scalar)` rather than `any(is not scalar)` fired only when -/// EVERY return was a buffer, so a `(scalar, StackBuf)` pair walked through. -#[test] -#[should_panic(expected = "StackBuf return cannot cross a match join")] -fn a_mixed_stack_buf_return_cannot_cross_a_match_join() { - let src = "\ -@inline -def f(k: Const): - s = StackBuf(2) - s[0] = GEN ** 7 - s[1] = GEN ** 9 - return GEN ** k, s - -def main(): - x = GEN - a, b = match(log(x), range(0, 2), lambda i: f(i)) - assert a == a - assert b == GEN ** 7 - return -"; - compile(&parse(src).expect("parse")); -} diff --git a/crates/lean_compiler/tests/suite/transcript_helpers.rs b/crates/lean_compiler/tests/suite/transcript_helpers.rs deleted file mode 100644 index a6f128acc..000000000 --- a/crates/lean_compiler/tests/suite/transcript_helpers.rs +++ /dev/null @@ -1,57 +0,0 @@ -use lean_compiler::{compile, parse}; -use primitives::field::F192; - -#[test] -fn transcript_helpers_are_ordinary_nested_inline_zkdsl() { - let src = r#" -from snark_lib import * - -Y = f192(0, 1, 0) - -@inline -def pack64x2(a, b): - assert_in_k(a, b) - return a + Y * b - -@inline -def challenge_from_state(state): - lo = StackBuf(2) - hi = StackBuf(2) - hint_f192_limbs(lo, state[0]) - hint_f192_limbs(hi, state[1]) - state[0] = pack64x2(lo[0], lo[1]) - state[1] = pack64x2(hi[0], hi[1]) - return lo[0] + Y * (lo[1] + Y * hi[0]) - -@inline -def fs_compress(state, scalar, tail, out): - limbs = StackBuf(3) - hint_f192_limbs(limbs, scalar) - block = StackBuf(2) - block[0] = pack64x2(limbs[0], limbs[1]) - block[1] = pack64x2(limbs[2], tail) - assert scalar == limbs[0] + Y * (limbs[1] + Y * limbs[2]) - blake2s(state, block, out) - return - -@inline -def observe(state, scalar): - out = StackBuf(2) - fs_compress(state, scalar, 13, out) - return out - -def main(): - state = StackBuf(2) - state[0] = f192(1, 2, 0) - state[1] = f192(3, 4, 0) - out = observe(state, f192(5, 6, 7)) - challenge = challenge_from_state(out) - assert challenge == challenge - return -"#; - // `assert challenge == challenge` is the zkDSL keep-alive idiom: it forces - // the value to be materialized. The assertion under test is that `execute` - // runs the lowered helpers without a write-once memory conflict. - let program = compile(&parse(src).expect("parse transcript helpers")); - program.execute([F192::ZERO; 2]).unwrap(); -} diff --git a/crates/lean_compiler/tests/suite/vm_proofs.rs b/crates/lean_compiler/tests/suite/vm_proofs.rs deleted file mode 100644 index bc2f85af7..000000000 --- a/crates/lean_compiler/tests/suite/vm_proofs.rs +++ /dev/null @@ -1,144 +0,0 @@ -//! Proof-level properties of the VM: what a proof is bound to, that its channels -//! carry everything, and that tampering with either channel is rejected. -//! -//! These live here rather than beside the prover because a provable program's tables -//! must all be powers of two, which is the compiler's fill blocks' job -//! (`lean_compiler::filler`). A hand-written bytecode program would have to fill itself, -//! duplicating their knowledge of what a dummy row looks like. - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{CpuError, Proof, ProveError, prove, verify}; -use lean_vm::vmhash::compress; -use primitives::field::{F64, F192}; - -/// A program that hashes one block and publishes the digest, so its proof carries -/// flock's sub-proof over a real compression. -const HASHING: &str = "\ -def main(): - a = StackBuf(2) - a[0] = 5 - a[1] = 7 - c = StackBuf(2) - blake2s(a, a, c) - p = 1 - p[1] = c[0] - p[GEN] = c[1] - return -"; - -/// The public input `HASHING` publishes. -fn hashing_pi() -> [F192; 2] { - let h = [F64(5), F64(0), F64(7), F64(0)]; - let d = compress(h, h); - [F192::new(d[0].0, d[1].0, 0), F192::new(d[2].0, d[3].0, 0)] -} - -fn hashing_proof() -> (lean_vm::cpu::Program, [F192; 2], Proof) { - let program = compile(&parse(HASHING).expect("parse")); - let pi = hashing_pi(); - let (proof, _) = prove(&program, pi, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &pi, &proof).expect("honest proof verifies"); - (program, pi, proof) -} - -/// The stacked WHIR opening rides the proof's Merkle phases. Nothing there is -/// bound by the transcript (only by the Merkle structure), so tampering an opened -/// row must still be rejected. -#[test] -fn a_tampered_opening_is_rejected() { - let (program, pi, mut proof) = hashing_proof(); - let phase = proof.merkle.first_mut().expect("stacked WHIR opening"); - phase.leaf_data[0][0].0 ^= 1; - assert!( - verify(&program, &pi, &proof).is_err(), - "a tampered validity proof must be rejected" - ); -} - -/// flock's reduction sub-proof (zerocheck, lincheck, ring switch) rides the scalar -/// stream as raw transport, but its values re-enter the transcript through the verifier's -/// replay, so a flipped transport word diverges the recovered flock claim. -#[test] -fn a_tampered_reduction_word_is_rejected() { - let (program, pi, proof) = hashing_proof(); - let mut tampered = proof; - let n = tampered.stream.len(); - // The second-to-last word is always meaningful bytes; only the final one may be - // zero-padded. - tampered.stream[n - 2] += F192::ONE; - assert!( - verify(&program, &pi, &tampered).is_err(), - "a tampered reduction transport word must be rejected" - ); -} - -/// A proof is bound to its exact program. The two programs here have the same shape, -/// so the same layout and announced sizes, and differ in one constant; the program -/// digest seeds the transcript, so it diverges at the first squeeze. This is -/// the adaptive-statement forgery that the bytecode bus's single-point check does not, -/// on its own, prevent. -#[test] -fn a_proof_does_not_verify_against_another_program() { - // The differing constant has to reach the bytecode: an unused one folds away at - // compile time and the two programs come out byte-identical. Hashing it does the - // job, and publishing nothing keeps the public input the same for both. - let src = |k: u32| { - format!( - "def main():\n a = StackBuf(2)\n a[0] = {k}\n a[1] = 7\n \ - c = StackBuf(2)\n blake2s(a, a, c)\n return\n" - ) - }; - let program = compile(&parse(&src(5)).expect("parse")); - let other = compile(&parse(&src(6)).expect("parse")); - let pi = [F192::ZERO, F192::ZERO]; - let (proof, _) = prove(&program, pi, lean_vm::pcs::TEST_LOG_INV_RATE).unwrap(); - verify(&program, &pi, &proof).expect("honest proof verifies"); - assert!( - verify(&other, &pi, &proof).is_err(), - "a proof must not verify against a different program" - ); -} - -/// Out-of-process verification: everything travels in the two channels, so a proof -/// serializes, crosses a process boundary and verifies, and a flipped announced size -/// is caught before any reduction runs. -#[test] -fn a_proof_roundtrips_through_bytes() { - let (program, pi, proof) = hashing_proof(); - let bytes = bincode::serialize(&proof).expect("proof serializes"); - let decoded: Proof = bincode::deserialize(&bytes).expect("proof deserializes"); - verify(&program, &pi, &decoded).expect("a deserialized proof verifies"); - - // The announced sizes lead the stream: memory log, the table log heights, then - // the PCS rate. - let mut bad_rate = decoded.clone(); - bad_rate.stream[1 + lean_vm::cpu::Stats::TABLES.len()] = F192::new(5, 0, 0); - assert!( - matches!(verify(&program, &pi, &bad_rate), Err(CpuError::PublicInput)), - "the announced PCS rate must be in 1..=4" - ); - - // A BLAKE2s height below flock's instance floor describes a layout the - // arithmetization cannot express, and all three verifiers reject it there. - let blake2s = lean_vm::cpu::Stats::TABLES - .iter() - .position(|&t| t == "BLAKE2S") - .unwrap(); - let mut sub_floor = decoded; - sub_floor.stream[1 + blake2s] = F192::new(2, 0, 0); - assert!( - matches!(verify(&program, &pi, &sub_floor), Err(CpuError::PublicInput)), - "the announced BLAKE2s height must reach flock's instance floor" - ); -} - -/// An unsupported rate is an error, not a panic. -#[test] -fn an_unsupported_rate_is_an_error() { - let program = compile(&parse(HASHING).expect("parse")); - let rate = lean_vm::pcs::MAX_LOG_INV_RATE + 1; - assert_eq!( - prove(&program, hashing_pi(), rate).err(), - Some(ProveError::InvalidRate { log_inv_rate: rate }) - ); -} diff --git a/crates/lean_compiler/zkDSL.md b/crates/lean_compiler/zkDSL.md deleted file mode 100644 index cc8b13199..000000000 --- a/crates/lean_compiler/zkDSL.md +++ /dev/null @@ -1,551 +0,0 @@ -# zkDSL Language Reference (leanVM) - -The zkDSL is a Python-syntax language that compiles to the leanVM ISA: six instructions (`XOR`, `MUL`, `SET`, `DEREF`, `JUMP`, `BLAKE2s`) over the binary field GF(2^192), with write-once memory and all indices carried "in the exponent" as powers of a fixed generator. For the underlying VM and proving system, see [`doc/leanvm/main.tex`](../../doc/leanvm/main.tex). - -Source files use the `.py` extension and are **Python-shaped**: they import the [`snark_lib`](snark_lib.py) stub, which defines `GEN`, `log`, `mul_range`, `HeapBuf`, `StackBuf`, `assert_in_k`, and `blake2s`, so editors and linters resolve the intrinsic names. The compiler skips the import. Ordinary helpers such as `pack64x2` are defined in the single-file guest. A program that uses placeholders is not a runnable Python file: its `*_PLACEHOLDER` identifiers are undefined until the host fills them in, so importing it raises `NameError`. - -Entry points: `lean_compiler::parse` / `parse_with_replacements` → `lean_compiler::compile` → `lean_vm::cpu::prove` / `verify`. - -## Dev experience - -The repo ships a root [`pyrightconfig.json`](../../pyrightconfig.json) with `"extraPaths": ["crates/lean_compiler"]`, so any `.py` program anywhere in the repo resolves `snark_lib` when the repo root is opened in the editor. A placeholder-free program also runs as plain Python (`PYTHONPATH=crates/lean_compiler python3 crates/lean_compiler/tests/programs/foo.py`); the stubs are no-ops, so this only checks that the file is well-formed. - -## The field, and indices in the exponent - -The fields are - -`K = GF(2)[x]/(x^64 + x^4 + x^3 + x + 1)` and `E = K[y]/(y^3 + y + 1) = GF(2^192)`. - -Machine **words** (the contents of a memory cell, an immediate, a hashed value, the `JUMP` condition) are elements of `E`. **Addresses**, the program counter, the frame pointer, read counters, operands, opcodes, and domain separators live in the 64-bit subfield `K = GF(2^64)`. There are no runtime integers. - -- `+` is field addition = bitwise **XOR** (192-bit on words, so `x + x == 0`), -- `*` is multiplication in `E`; for g-powers and addresses it stays within `K`, -- `/` is runtime field division, `a / b = a · b⁻¹`. It costs one `MUL`: the compiler leaves the quotient cell unset and emits the checked relation `quotient · b == a`, which witness generation back-solves. Division by zero is undefined. This is distinct from `//`, compile-time integer floor division in sizes and indices, -- an integer literal `n` supplies up to 128 raw bits and is embedded as `F192(c0, c1, 0)`. This is a source-syntax limit, not the machine-word width: words have three 64-bit limbs. Thus `5` is `1 + x^2`, not the integer five, and the literal `18446744073709551616` is the tower element `y`. Written `2 ** 64` in a value position it is not: `**` there is a field power, so it is `x^64` reduced. `const(2 ** 64)` is the literal. Full-width constants use `f192(c0, c1, c2)`, with each limb an unsigned 64-bit compile-time integer, -- `GEN` is the fixed generator `g = x` of the 64-bit subfield `K^×` (multiplicative order `2^64 − 1`), -- `GEN ** e` is the compile-time constant `g^e ∈ K` (`**` takes base `GEN` and a compile-time integer exponent: a literal, a constant, an `unroll` variable, `len(...)`, or index arithmetic of those). So `buf[GEN ** i]` names heap cell `i` directly inside an `unroll` loop, with no running-pointer cursor. -- constant arithmetic means different things in the two positions, and this is a silent trap: `a + b` on two constants is **integer** addition in an index, a bound or a keyword (`buf[GEN ** (i + 1)]`, `unroll(0, n + 1)`, `counter=64 * (q + 1)`), and **XOR** in a value, where `1 + 1` is `0`. So a literal built in a value position must not add overlapping integers: `tweak = base + (level + 1) * SHIFT` drops the whole term on odd levels. Products are safe (an integer times a power of two is that shift, as long as the top bit stays inside the limb); to add, name the regime with `const(...)`: `tweak = base + const((level + 1) * SHIFT)`. -- a **global constant** and a **`Const` parameter** are integer-arithmetic throughout, which is deliberate (it is what makes a derived size right) but means the same text means different things in the two places: `STEP = 3 + 1` is the integer `4` everywhere it appears, while the identical `x = 3 + 1` written inside a function is the field element `2`. Neither is wrong; they are two regimes, and a name crossing between them is where the trap bites. -- a **compile-time `if`** must say which regime it means when the two disagree. The fold decides on the integer reading while the runtime test of the same condition compares field values, so `if 3 + 1 == 4` is true one way and false the other. Such a condition is **rejected**; write `if const(3 + 1 == 4):` to decide it with integer arithmetic, or spell the operands so the two readings agree (a product or a shift rather than a sum of overlapping integers). A condition whose readings already agree needs no wrapper. -- `base ** e` with a **non-`GEN`** base and a compile-time exponent `e` is square-and-multiply: integer arithmetic in an index/bound position (`2 ** c`), or field arithmetic in a value position (`x ** k`, e.g. a loop counter `g^i` raised to a stride to reach cell `i·stride`). The base may be runtime. - -A logical **index** `i` is carried as `g^i` in the 64-bit subfield (order `2^64 − 1`): incrementing is one multiplication by `GEN`, and memory/bytecode addresses are g-powers. This is the design idiom of the whole VM: loops, heap addressing, and range checks below all live in the exponent, in `K`. - -## Program shape - -A program is a **single** `.py` file: - -```python -from snark_lib import * # for Python tooling; skipped by the compiler - - -def main(): # required entry point - ... - return - -def helper(a, b): # other functions - ... - return a * b -``` - -`import snark_lib` / `from snark_lib import *` are the only imports accepted; anything else is a compile error (no multi-file programs yet). Comments (`#`) and blank lines are free. Indentation is block structure, as in Python. - -Ordinary functions may return scalars, `HeapBuf` pointers, and `StackBuf` values, including mixtures in a tuple return. A returned `StackBuf(n)` has a compile-time-known size: its `n` cells are copied through `n` consecutive return slots and the caller binds the result as a new `StackBuf(n)`. A `HeapBuf` return is just its one-cell pointer; the allocation hint already ran where the buffer was created, so no size metadata needs to cross the call. - -## Public input - -Memory cells `m[0]` and `m[1]` hold the two public-input words, each an F192 machine word. A program *publishes* results by asserting them against those cells through the write-once heap store (the pointer `g^0` addresses absolute memory): - -```python -p = GEN ** 0 -p[1] = result_a # m[p·1] = m[0], an equality assert against the public input -p[GEN] = result_b # m[p·g] = m[1] -``` - -Test programs under `tests/programs/` declare the public input they expect with a top-of-file annotation of two constant elements (or omit it to run with two zeros); the generic harness `tests/py_source.rs` proves and verifies every program in the directory: - -```python -# public_input: GEN ** 89, 101229015297003380629709256178361811305 -``` - -## Global constants and placeholders - -Above the functions (after the optional `snark_lib` import) a program may declare **global constants**, top-level `NAME = `: - -```python -from snark_lib import * - -N = 8 # an integer size / value -STEP = GEN ** 2 # a g-power constant (index carried in the exponent) -WIDE = N + 1 # compile-time INTEGER arithmetic (`+ - * / **`); - # references to *earlier* constants are allowed - -def main(): - buf = StackBuf(N) # a constant is a plain literal: usable as a size, - x = GEN ** N # a `**` exponent, a stack/slice index, an operand, - assert log x < N # an `assert log _ < _` bound, or a `Const` argument - return -``` - -Each constant is **evaluated as a compile-time integer expression** (or an `f192` literal, or a field-valued one such as `GEN ** 2`) and substituted as a single literal everywhere its name appears below, so unlike a `Const` parameter it needs no call site and works in every literal position. Integer arithmetic is the point: it is what makes a derived size come out right, as in `N_TWEAK_WORDS = 2 + CHAIN_STEPS * V + LOG_LIFETIME`. Constants must precede the `def`s and are resolved *before* variables, so a constant name is **reserved**: do not reuse it as a parameter or local name. (Syntactically, `N = 8` is just a Python module global.) - -**Placeholders** let a host fill values at compile time without editing the source. Any identifier may be mapped to replacement text before parsing (`parse_with_replacements`, taking a `BTreeMap`); the replacement is identifier-bounded (`FOO` does not touch `FOOBAR`). The idiom is a placeholder feeding a constant: - -```python -V = V_PLACEHOLDER # with replacement "V_PLACEHOLDER" ↦ "128" -LOG_LIFETIME = LOG_LIFETIME_PLACEHOLDER - -def main(): - ... # V is the constant 128 throughout -``` - -so one source template compiles at many sizes. An unfilled placeholder (no replacement, no matching constant) is a compile error, not a silent variable. - -### Constant arrays - -A global constant may be a **list literal**, `NAME = [a, b, c]`, of compile-time values (integers or field values, each a ``). Unlike a scalar constant it is **not** textually substituted; it is carried to lowering and consumed at compile time: - -```python -QUERIES = [290, 177, 145] # or QUERIES = QUERIES_PLACEHOLDER, filled "[290, 177, 145]" -Z = [Z0_PLACEHOLDER, Z1_PLACEHOLDER, Z2_PLACEHOLDER] # arbitrary field values - -def main(): - for lvl in unroll(0, len(QUERIES)): # len(NAME) is a compile-time count - n = QUERIES[lvl] # NAME[i] with a compile-time index i - row = buf[GEN ** QUERIES[lvl]] # (i a literal / constant / unroll var) - ... -``` - -`NAME[i]` yields the element (as a field value in value position, or as an integer where an index / slice bound / `unroll` count / `**` exponent is expected), and `len(NAME)` its length. The index `i` must be compile-time (a literal, a constant, or an `unroll` variable). This is what lets one source file adapt to a per-level config vector (query counts, fold factors, sizes) without Rust-side code generation. Nested lists are not (yet) supported: flatten a 2-D table into one array plus an offsets array. - -## Functions - -```python -def f(a, b): - return a + b, a * b # multiple returns - -x, y = f(p, q) # tuple assignment -z = f(p, q) # expression position: first return -f(p, q) # statement: returns discarded -``` - -Functions may recurse. Each call gets a **fresh frame**: the frame pointer is prover-hinted (write-once memory makes an unconstrained cell prover-chosen), arguments and the return address/frame are stored with `DEREF`s, and control transfers with one `JUMP`. Cost: about `n_args + n_returns + 4` instructions per call. Every non-`main` function must end in an explicit `return`; in `main`, `return` is a no-op (main halts at a sentinel automatically). - -### `StackBuf` parameters - -```python -def compress(cv: StackBuf(2), block: StackBuf(2)): - out = StackBuf(2) - blake2s(cv, block, out) - return out -``` - -`s: StackBuf(n)` marks a parameter as a **run of n cells**, passed whole. The caller must pass a `StackBuf` of exactly that size (a whole named one: a slice is not yet accepted), and the run is copied into the callee's frame. - -Those cells arrive **already written**, unlike a local `StackBuf`'s, so a store into one is the write-once equality *assertion* rather than a fresh store. That is what makes a callee able to pin its caller's values: `s[k] = ` inside the callee asserts that the caller's cell already held it. It also means a run parameter initializes nothing, so passing a partly-written buffer passes its unwritten cells, which the prover then chooses, exactly as anywhere else. - -A fused `match` arm cannot pass one: the fused dispatch writes one cell per argument, so give such arms `Const` arguments and let each specialize instead, or make the callee `@inline`. This is otherwise the same mechanism a `StackBuf` **return** already used, in the other direction: the argument area is a width rather than a count, and a run occupies the cells its size asks for. Without it a two-cell value could come out of a function whole but only go in through a `HeapBuf` pointer or an `@inline` expansion, which grows the caller's frame at every call site. - -### `Const` parameters - -```python -def hash_pair(buf, k: Const): - h = StackBuf(2) - blake2s(buf[k * 2:k * 2 + 2], buf[k * 2:k * 2 + 2], h) - return h[0], h[1] -``` - -`k: Const` marks a **compile-time parameter**: the call site must pass a constant (an integer literal, `GEN ** k`, or a literal-bound name), and the compiler *specializes* the function per distinct constant tuple, a monomorphized copy (`hash_pair__L1`) with the parameter substituted as its literal, shared by every call with the same constants; only the runtime arguments are passed. Inside the body the parameter *is* the literal, so it works in compile-time positions: stack indexes, slice bounds. A function with a `Const` parameter is a template: it is never lowered itself. The idiomatic pairing dispatches a runtime index to a const-indexed helper: - -```python -r = match(log(x), range(0, 4), lambda i: hash_pair(buf, i)) -``` - -### `@inline`: inline a function at its call sites - -```python -@inline -def combine(a, b, k: Const): - s = StackBuf(2) - if k % 2 == 0: # a folded `if` (see below): baked per Const value - s[0] = a - else: - s[0] = b - s[1] = a + b - return s[k % 2] -``` - -An `@inline` function is **expanded at each call site** instead of emitting a real call: no frame, no argument/return `DEREF`s, no call/return `JUMP`s. Its body must end with one top-level `return`. Builtins, ordinary calls, nested inline calls, `if`, and `unroll` are allowed; `mul_range` loops, `match`, tuple assignments within the body, and nested/early returns are rejected. Ordinary calls retain their own frames; nested inline calls expand recursively, with direct or indirect recursive expansion rejected. - -An `@inline` function may also **return a `StackBuf`**: the caller's binding aliases the returned cell run (zero copies), and `StackBuf` arguments alias likewise. - -An `@inline` call may also sit in **expression position**: embedded in arithmetic, as a store's RHS, or as a single-target `match` arm. An aliased return (a folded g-address) then materializes into a plain cell (free for a var; one `MUL` for a shifted pointer); a multi-cell `StackBuf` return still needs a `let` binding, since only a name can alias a cell run. - -An `@inline` `match` arm expands into the dispatching frame like any other call site, specialized on its `Const` arguments, so it can take a `StackBuf` and write the caller's cells directly. The arms of one `match` share their local cells, one of them running. - -Because the body runs in the *caller's* frame, a `Const` parameter whose `if`s fold (below) bakes straight-line, per-case code, the idiom for a `match` arm that must specialize on the arm value. The trade-off is frame cells: each call site gets its own copy, so `@inline` pays off for small, hot callees; inlining a large body at many sites grows the committed witness (more data memory), so it is opt-in, not automatic. - -## Variables - -Bindings are **immutable**: `x = e` names a fresh cell. Re-binding a name is allowed (it's a new cell; the old value is unaffected), but there is no mutation. Compound assignment (`+=`, `-=`, `*=`, `//=`, `%=`) is sugar for a re-binding: `x += e` desugars to `x = x + e`. - -A name bound to an integer literal (`x = 2`) additionally acts as a **compile-time index constant**, usable in stack indexes and slice bounds (see below). Any other re-binding clears that role. - -Two families of binding are folded and carried **virtually**, costing no instruction until used as a value: - -- **g-powers and shifted pointers**: a cursor like `s = s * GEN` or a pointer view `p = buf * GEN ** k`. The offset folds into the `DEREF` address of each access; only a scalar use materializes it. -- **field constants**: a value built from literals / `GEN ** k` by field `+` and `*`, e.g. a running weight `w = w * CHAIN_LENGTH` in an unrolled loop. The arithmetic that advances it is compile-time (zero instructions); each use is one `SET` of the folded constant. -A store into a stack cell is NOT virtual: `sa[k] = other` always emits. If the cell already holds a value the store is the write-once equality *assertion* below, which is what makes `s[k] = ` pin a hint and a pre-written `blake2s` output verify a digest; if it does not, the store is what gives the cell its value. The compiler tracks nothing to tell those apart, the machine's write-once memory being what distinguishes them. - -## Debugging - -`print(expr)` / `print("label", expr)` displays a value at witness generation (prover side only, with no constraints and nothing entering the transcript). The label defaults to the argument's source text; output goes to stderr as `[print] label = ...`, showing the decimal reading for small integers, `g^k` when the value is a small g-power (both when they overlap: `8 (g^3)`), or `c2:c1:c0` hex otherwise, from the most significant limb to the least significant. Each print costs one anchor instruction, so the witness differs from a print-free build: strip prints before benchmarking. - -## Memory - -All memory is **write-once**: a cell is set once; a second write of the same value is a no-op, of a different value a proof failure. This turns stores into equality assertions and is used throughout (publishing, `blake2s` outputs). Reading a cell nobody ever writes yields an unconstrained value (zero in witness generation): don't. Nor use a value before the store that gives it has run: witness generation runs forward, so the use sees zero and the run is rejected. Loading it early is fine: `x = hb[i]` before `hb[i] = v` makes `x` equal to `v`, provided `x` is used after the store. - -### `HeapBuf(n)`: heap buffers, indexed in the exponent - -```python -buf = HeapBuf(4) # fresh, disjoint region; `buf` is its pointer (a g-power) -buf[1] = 5 # m[buf·1] is cell g^0 -buf[GEN] = 7 # m[buf·g] is cell g^1 -v = buf[i] # m[buf·i], i any runtime g-power (e.g. a loop counter) -buf[i * GEN] = v # the next cell along -``` - -The index is a field element; cell `k` of the buffer lives at address `buf · g^k`. A read or store is one `DEREF`. A **runtime** index costs one extra `MUL` for the `buf·i` pointer, but a **compile-time g-power** offset (`buf[1]`, `buf[GEN ** k]`, or a cursor advanced by `× GEN ** m`) folds into the `DEREF`'s address immediate for free: no `MUL`, no `SET`, and the cursor arithmetic itself vanishes (so a `× GEN` walk over consecutive cells is zero instructions). - -**Compile-time indices are bounds-checked.** When the whole index is a compile-time exponent and the pointer resolves to a declared `HeapBuf` (directly, or through shifted aliases like `row = buf * GEN ** k`), the compiler rejects `index >= size`, and the same for the spans of `hint_witness` and `blake2s` slices. **Runtime** indices are not checked (their value is unknown at compile time): there the buffer remains a region convention, and a stray access surfaces at proving time as a write-once conflict or wild deref. - -### `StackBuf(n)`: frame-cell runs, indexed by compile-time integers - -```python -sa = StackBuf(3) # n consecutive cells of the current frame -sa[0] = 3 # direct frame cell: no DEREF, but the store is an instruction -sa[2] = sa[0] + sa[1] -x = 1 -v = sa[x + 1] # indexes: literals, literal-bound names, and + * // % of those -tg = [v, 7] # list literal: an initialized StackBuf, one cell per element -``` - -A **list literal** `x = [a, b, …]` is an initialized `StackBuf`: it allocates one cell per element and writes each element in place, exactly the alloc-then-store idiom above, in one line. Elements are arbitrary runtime expressions; each write goes through the same stack-store path. It exists only as the RHS of a plain assignment inside a function; a *top-level* `NAME = [...]` is a constant array (see "Constant arrays"). The elements are lowered before the name rebinds, so `s = [s[1], s[0]]` swaps through the old binding. - -Stack indexes and slice bounds are **compile-time integers**, and index arithmetic (`+ * // %`) is *integer* arithmetic (`x + 1` above is 2, `k // 2` floor-divides, `k % 2` is a remainder: index space, not the field, where XOR is what `+` means and `//`/`%` have no meaning at all: using one as a runtime field value is a compile error). Bounds are checked at compile time. A `StackBuf` name is a run of cells, not a scalar: using it as one is an error, and it cannot be captured into a `for` loop body (carry state through a `HeapBuf` instead). - -`p = addr(sb)` names the run's first cell as a **pointer** (`GEN ** k` times the frame pointer), so `p[i]` reads the same cells at a runtime index, `p` can be passed to a callee or stored, and `sb[k]` stays a direct frame cell throughout. Only valid as a whole right-hand side. It costs one materialization of `fp` per function (2 `DEREF`s, amortized with `if`'s; free in `main` and in loops with reserved frames), which is the price of the ISA having no fp-read. This is what lets a bit buffer live in the frame and still be walked by a `mul_range` loop. - -A runtime index through such a pointer is unchecked, as on the heap, but it fails more quietly: every frame cell is a real cell, so `p[i]` with a hinted `i` reaches any of them and usually neither faults nor conflicts. The program owes the range check itself (`assert log i < n`) wherever `i` is not a loop counter the compiler produced. - -### Slices: `buf[lo:hi]` - -`buf[lo:hi]` names a run of cells (`hi` exclusive). BLAKE2s operands must span exactly two cells; `hint_witness` accepts any supported literal length. Two forms: - -- **compile-time bounds** (integers, as for stack indexes): frame cells `base+lo .. base+hi` of a `StackBuf`, or heap cells `ptr·g^lo .. ptr·g^hi` of a `HeapBuf`, so `hb[2:4]` is the pair `g^2, g^3`; -- **runtime start, heap only**: `buf[i:i + k]` with a runtime g-power index `i` (e.g. a loop counter) and literal length `k` names the cells `buf·i`, `buf·i·g`, and so on; one `MUL` folds `i` into the pointer. The `hi` bound cannot be evaluated, only shape-checked: it must be syntactically `lo + k` (`buf[b * GEN ** 2 : b * GEN ** 2 + 2]` is fine). A `StackBuf` slice cannot have a runtime start: frame offsets are baked into the bytecode operands. - -Note the two index spaces, consistent with plain indexing: compile-time bounds are integer exponents (`hb[2:4]` ≡ `hb[GEN ** 2 : GEN ** 2 + 2]`), runtime starts are g-power elements. - -## Control flow - -### `for i in mul_range(start, stop)`: loops in the exponent - -```python -for i in mul_range(1, GEN ** 10): # i = g^0, g^1, …, g^9 - buf[i * GEN * GEN] = buf[i] * buf[i * GEN] -``` - -The counter walks multiplicatively: it starts at `start`, advances by `×GEN` each iteration, and stops on reaching `stop` (exclusive). The start is a compile-time power of `GEN` (`1`, `GEN`, or `GEN ** k`); the stop is either compile-time too (an empty range compiles to nothing) or a **runtime** g-power element, e.g. a hinted count: - -```python -hint_witness(nb[0:1], "n_blocks") -n = nb[0] -assert log(n) < 16 # the walk terminates only by REACHING the bound: -for j in mul_range(1, n): # bound its log first, or it never does - ... -``` - -A runtime bound is evaluated once at entry and threaded through the loop as an extra parameter (+1 argument per iteration call); entry itself is the same `!=` test, so a bound equal to the start runs zero iterations. - -Lowering: the body becomes a tail-recursive helper function whose exit test is folded into the recursion's `JUMP` condition: one call per iteration, no separate is-zero gadget. Free variables of the body are captured **by value** as extra parameters; a `HeapBuf` pointer threads through fine, a `StackBuf` does not (compile error). - -The compiler reserves consecutive frames for a loop that has no early return and does not rebind its counter. Its back edge advances by the frame size and copies arguments directly into the next frame, while ordinary function calls and heap allocations remain disjoint from the reserved run. Each iteration still owns fresh write-once cells, so a pointer to an earlier iteration remains valid. Loops with an early return or counter rebinding keep incremental allocation. - -### `for i in unroll(a, b)`: compile-time unrolling - -```python -for i in unroll(0, 7): - sb[i + 1] = sb[i] * GEN # i is the integer literal of each copy - -def chain(buf, n: Const): - for i in unroll(0, n): # a Const parameter as a bound - blake2s(buf[i * 2:i * 2 + 2], buf[i * 2:i * 2 + 2], buf[i * 2 + 2:i * 2 + 4]) - return -``` - -The body is replicated `b − a` times with `i` substituted by each integer literal in turn, usable anywhere a literal is (stack indexes, slice bounds, `Const` arguments). Zero loop overhead: no call, no frame, no counter; the price is code size. Bounds are compile-time integer expressions, evaluated after `Const` specialization, so `unroll(0, n)` with `n: Const` works (unlike `mul_range`, whose bounds are parse-time literals). Every copy executes (this is straight-line code, not a branch), so bindings simply rebind, a fresh binding per iteration. - -### `if` / `elif` / `else` - -```python -if x == GEN ** 3: - r[1] = 5 -elif x != y: - r[1] = 7 -else: - r[1] = 9 -``` - -Conditions are field-equality tests: `a == b` or `a != b` (there are no other predicates: order facts come from range-check asserts). The lowering is one `XOR` plus one conditional `JUMP` on it; the taken jump goes to whichever block the test doesn't fall into, so no negation gadget is needed. An `elif` is sugar for an `else` holding a nested `if`. - -When **both sides are compile-time integers** (e.g. after a `Const` parameter is substituted, `if k % 2 == 0:`), the condition is known at compile time and the `if` **folds** to just the taken branch: no `XOR`, no `JUMP`, no `self-fp`. This is what lets an `@inline` function bake different straight-line code per `Const` value. A side whose integer reading and field reading disagree (`3 + 1` is the integer 4 and the field element 2) is **rejected** rather than folded either way, since the fold and a runtime test of the same condition would answer differently; write `if const(...)` below to decide it with integer arithmetic. Note that the rejection is per side, so it fires however the OTHER side is spelled. - -Two write-once-flavored rules: - -- **bindings made inside a branch are local to it**: the compile-time scope reverts at the join. Branches communicate through memory: only one branch executes, so both may write the *same* cell (`r[1]` above), and the join reads it. -- a cell nobody wrote (e.g. skipped-branch territory) stays unconstrained, the same rule as everywhere else in write-once memory. - -Local jumps must carry the frame pointer, which the ISA cannot read directly; each branching function materializes its own `fp` once (2 `DEREF`s through a 1-cell heap bounce; free in `main`, where `fp = g^0 = 1`). - -### `match` - -```python -r = match(log(x), range(0, 6), lambda j: f(j)) -a, b = match(log(x), range(0, 2), lambda j: g(1), range(2, 6), lambda j: g(j)) -``` - -The one dispatch construct. It matches the **log** of a g-power scrutinee against integer arms, which must cover consecutive integers from 0 (the dispatch table is dense; there is no default arm). Arm `j` is the lambda body with the parameter replaced by the **integer literal** `j`, usable as a field constant or a compile-time index, expanded at parse time over the contiguous `(range, lambda)` pairs. The whole call sits on one line, there being no line continuation. - -Arms produce VALUES: every arm writes its results into the same cells, which is sound under write-once because exactly one arm runs. A target may be a name, bound after the join, or a **`StackBuf` element**, which the arms write into directly and which costs one instruction less than a name plus a store. The ABI returns into cells the CALLER picks, the same reason `sb[i] = f(x)` never needed a temporary, so reach for the element form wherever a returned value's home is a buffer slot. A target index must be a compile-time integer inside the buffer, both errors naming the line; a `HeapBuf` element is not a target, its cells not being frame cells. Multiple targets take a multi-return call as the arm body. - -A branch body with statements in it goes in a function, and the arm calls it. A plain function fuses (below), so each taken arm still pays the shared frame's argument and return plumbing, and cannot take a `StackBuf`. An `@inline` one expands into the dispatching frame instead: no frame, no plumbing, and its `StackBuf` arguments alias, so an arm can hash or hint straight into the caller's buffers (the XMSS chain walk in the recursion guest, `lambda k: walk(chain_start, chain_tweaks, pp, md, tips, i, k)`). The price is code size, a copy of every arm at each `match`. - -**Lowering** is two jumps through a *trampoline table* in the bytecode: the dispatch jumps to `g^T · x²`, the j-th two-instruction slot (`SET` the arm's address, `JUMP` to it) of a table at base `T`, and the slot jumps to the arm, which can sit anywhere, unaligned and of any length. Cost is about 7 cycles, independent of the arm count. - -(Why not leanVM's single-jump `pc = a + b·x`: that affine address needs integer *scaling* by the common block size `b`, which in the exponent becomes `x^b`, log₂ b squarings, plus padding every block to the longest; the trampoline collapses the aligned region to 2-instruction slots, so the scaling is the single squaring `x²`. Other layouts exist, e.g. a memory-resident address table dispatched with a single jump, worthwhile for many repeated small matches, but only the trampoline is implemented.) - -**Soundness**: nothing in the dispatch bounds `x`, so a scrutinee outside `[0, n)` jumps to an arbitrary pc. A hinted value must be range-checked first (`assert log(x) < n`, 3 cycles), as in leanVM. - -**Dispatched-call fusion.** When *every* arm is a call to the same function with identical runtime arguments (the common `lambda k: f(a, b, k)`, where only a `Const` argument varies), the compiler builds the callee frame **once** and the dispatch jumps straight into the selected specialization's entry, which returns past the join. Each taken arm is then just the trampoline's two instructions (`SET entry; JUMP`) instead of a full call: no per-arm frame setup, call jump, or return jump. An `@inline` callee does not fuse: it expands, as above. - -Statements without effect are rejected. - -### `if const(...)`: a branch decided while compiling - -```python -if const(level + 1 == DEPTH): # decided now, with integer arithmetic - tail = 0 -``` - -Wrapping a condition in `const(...)` asks for the branch to be decided while compiling. Two things follow. The condition must be decidable then, so both sides must be compile-time integers, and a runtime one is an error rather than a silent fallback to a runtime test. And it is read with **integer** arithmetic, the regime a compile-time constant lives in, which is what makes `const(...)` the answer when a condition's two readings disagree (see "The field, and indices in the exponent"). - -A folded branch emits no test and no jump, and its body is straight-line code, so **its bindings outlive it** where a runtime branch's are branch-local. That is the other reason to reach for the wrapper: it states that the arm's bindings are meant to escape. - -A plain `if` still folds on its own when both sides are compile-time integers and neither side's two readings disagree, so the wrapper is needed only where one does, where the condition is decidable only in the field (`GEN ** 3 == GEN ** 3`, which a plain `if` lowers to a real runtime branch), or where you want the compiler to insist. - -### `const(...)` in a value position - -```python -tweak = TW_NODE + const((level + 1) * P_MUL) + tau # (level+1)*P_MUL as integers -``` - -The same wrapper, the same meaning: read this with **integer** arithmetic and emit the literal. It is needed because `+` in a value position is XOR, so `level + 1` with `level = 3` is 2 rather than 4, and silently: the value is well-formed, just not the one the arithmetic reads like. `-`, `//` and `%` have no field meaning at all, so `const(...)` is the only way to write them in a value position. - -The inner expression must be a compile-time integer (a literal, a global constant, a `Const` parameter, an `unroll` counter, a name bound to one, a constant-array element, and `+ - * // % **` of those), and one that is not says so rather than falling back to a runtime computation. The result is one pooled `SET`, so a repeat costs nothing. - -In a position that is ALREADY integer arithmetic (a size, a count, an exponent, a bound, a stack index, a global constant) the wrapper is transparent: it asks for the only reading there is, so it changes nothing and is allowed rather than redundant. Where it earns its keep is a value, a condition, and anywhere `-`, `//` or `%` has to appear. - -The wrapper reinterprets the **operators**, not the leaves, and that is the whole of its meaning. Two consequences. A leaf whose own two readings disagree is rejected rather than silently read one way, so `n = 2 + 3` (the cell holds `2 XOR 3` = 1, the name's integer reading is 5) may not appear inside one: bind it in one regime and name that one. And the arithmetic runs on a leaf's **bit pattern**, so an element of a field-valued constant array is read as the integer those bits spell, which is not what field arithmetic on it would give: `const(TABLE[i] * 2)` doubles the bit pattern where `TABLE[i] * 2` is a field product. - -## Assertions - -### `assert a == b` - -A proof-enforced equality: 1 cycle (`XOR` into the frame's zero cell, whose write-once double write is the assert). - -### `assert a != b` - -A proof-enforced inequality, in **3 instructions and no branch**: `XOR` for `x = a + b`, a prover-hinted `inv = x⁻¹`, then `MUL p = x·inv` and `SET p = 1`, where the write-once conflict is the assertion, exactly as for `assert a == b`. It is sound because `x = 0` forces `p = 0` whatever the prover hints, and `p` cannot then also be `1`; the hint needs no checking of its own, which is why an unconstrained value is safe here. Since there is no `JUMP` there is no self-frame or branch setup to amortize either. A compile-time assertion such as `assert 5 != 5` is rejected while compiling. - -### Range checks: `assert log x < log Y` and `assert log x < k` - -The *range check in the exponent*: proves `x ∈ {g^0, g^1, …, g^{k-1}}`, i.e. `log_g(x) < k`. A compile-time bound is either `log GEN ** k` or a plain integer exponent `k`, with `1 ≤ k ≤ 2^16` (the minimum memory size, which keeps the gadget provable at every memory size the prover may announce). `log x` and `log(x)` both parse; the parenthesized form is the valid-Python spelling. A bare `assert x < y` is rejected: field elements have no order, only their logs do. - -```python -assert log(x) < log(GEN ** 8) -assert log(x) < 8 # the same check -assert log(x) < log(n) # n = g^k runtime: same gadget, +1 cycle -``` - -A **runtime** bound costs one extra `MUL` for `g^{k-1} = n·g⁻¹` and is otherwise identical, except that the `k ≤ 2^16` cap becomes the program's to enforce: range-check the bound itself first, with `assert log n < 2^16`. That check is not optional: without it the gadget is unsound. - -Cost: **3 cycles** (leanVM's DEREF range-check trick, in the exponent) plus one amortized `SET` per distinct bound per frame: - -1. `DEREF` through `x`: the dereferenced address must be one of the memory's `2^h` g-power addresses, so the memory bus itself proves `x = g^e`, `e < 2^h`; -2. `MUL x·y` into the write-once cell holding `g^{k-1}`: the runner back-solves the complement `y = g^{k-1-e}` (the one unknown operand of a known product), and the double-write asserts `x·y = g^{k-1}`; -3. `DEREF` through `y`: bounds the complement; a "negative" `k-1-e` would wrap to `≈ 2^64`, far beyond any memory size, so together `e ≤ k-1`. - -The two `DEREF` target cells are unconstrained touches, back-filled at the end of execution. A failing check surfaces at witness generation as the complement's `DEREF` panic ("not a small g-power … a failed range check"). - -## K membership and packing - -```python -assert_in_k(lo, hi) -``` - -`assert_in_k(a, b)` is the sole packing-related compiler intrinsic. It proves that both source memory words are in the base field GF(2^64) with one untaken `JUMP`: its condition is a known zero, while its destination and frame operands are `a` and `b`. Although neither value affects the successor state, both memory reads carry literal-zero upper limbs, so a source outside GF(2^64) cannot balance the memory permutation. - -Packing is ordinary zkDSL built on that assertion: - -```python -@inline -def pack64x2(a, b): - assert_in_k(a, b) - return a + f192(0, 1, 0) * b -``` - -The inline helper takes three cycles, one `JUMP`, one `MUL` and one `XOR`, and returns the canonical 128-bit packing `(a.c0, b.c0, 0)`. Assignment-target lowering writes its return directly into the destination, including an already-written cell whose second write is an equality assertion. A caller needing only membership uses `assert_in_k` directly and pays no packing arithmetic. - -The recursion transcript uses `challenge_from_state(state)` to reinterpret the first three 64-bit lanes of a canonical two-cell BLAKE2s digest as one extension field challenge. For `state = [s0, s1]`, it lowers exactly as follows (the limb hints cost no cycles, but are not trusted): - -```python -d2 = StackBuf(1) -hint_f192_limbs(d2, state[1]) -d3 = (state[1] + d2[0]) * Y_INV -assert_in_k(d2[0], d3) -challenge = state[0] + d2[0] * f192(0, 0, 1) -``` - -Both state words are BLAKE2s outputs, so their top limbs are already zero. Only `d2` must be exposed separately; deriving `d3 = (s1+d2)/Y` and proving both values lie in GF(2^64) binds the one hinted limb by uniqueness of the tower representation. The challenge is `s0+d2·Y² = d0+d1·Y+d2·Y²`, while `d3` is checked but deliberately discarded. `challenge_from_state` is not a compiler intrinsic: this is the complete `@inline` helper used by the recursion guest. - -## BLAKE2s - -```python -h = StackBuf(2) -blake2s(a, b, h) # digest of (a, b) written into h -blake2s(t[0:2], t[x:x + 2], t[4:6]) # slices of one large StackBuf -blake2s(h, hb[0:2], hb[2:4]) # HeapBuf slices, input and output -blake2s(hb[i:i + 2], h, hb[j:j + 2]) # runtime-indexed heap slices (i, j g-powers) - -# A standard 80-byte hash as two blocks. Keyword values are compile-time. -block0 = [1, 2, 3, 4] # 64 bytes -tail = [5, 0, 0, 0] # 16 more, the rest of the block zero-filled -blake2s(block0[0:2], block0[2:4], cv, counter=64, final=0) -blake2s(tail[0:2], tail[2:4], out, cv=cv, counter=80, final=1) - -# The same, with the second block's metadata computed at run time. -blake2s(tail[0:2], tail[2:4], out, cv=cv, md=high + f192(16, 4294967295, 0)) -``` - -The three positional arguments form a **statement**: one standard BLAKE2s compression consumes the two 256-bit message operands `a`, `b` (64 bytes) and writes its 32-byte result into the 2-cell run `out`. With no keywords it computes the standard hash of exactly 64 bytes: the parameterized BLAKE2s-256 initial chaining value (digest length 32, unkeyed, fanout and depth 1), byte counter 64, final-block flag `f0` set. That is `blake2s(a || b)`, the form every Fiat-Shamir step and Merkle node uses. - -Every compression also has a 256-bit chaining value and a 128-bit metadata word. The optional keywords are: - -- `cv=`: a consecutive 2-cell chaining value, the previous block's output; omitting it selects the parameterized IV above. On each runtime path, a function emits two `SET`s at its first such hash only and reuses those cells thereafter. Supplying `cv=` also requires one of the four below, since a chained block is never the default one-block hash; -- `counter=`: BLAKE2s's byte counter `t`, **cumulative** through this block, so `64 * whole_blocks_before + bytes_in_this_block`. Defaults to 64; -- `final=<0|1>`: BLAKE2s's final-block flag `f0`. It defaults to 1 for the bare three-argument call, but to **0** as soon as `counter=` or `last_node=` appears, so a chained hash must set `final=1` on its last block and a single short block needs `counter=, final=1`. Any compile-time expression works, nonzero meaning set, which is what lets the guests write a predicate like `final=(q + 1) // BLOCKS_PER_HASH`; -- `last_node=<0|1>`: BLAKE2s's tree-mode flag `f1`. Defaults to 0, and nothing here uses tree mode; -- `md=`: the whole 128-bit metadata word, as a value the program computed, for a hash whose block count is only known at run time. It replaces the three keywords above (giving both is an error) and it owes the same canonical embedding as every other operand, its top limb being read as a literal zero. The cheap way to build one is the disjoint-bit split of `doc/leanvm` §Byte counters for a hash of runtime length: XOR a runtime high part against a compile-time `metadata(64·j, f0, f1)` constant, one instruction per block. - -The metadata is packed as `counter:u64 | f0:u32 | f1:u32`, little-endian, into one memory cell the instruction reads, like every other operand. With compile-time keywords that cell is one pooled `SET`: a frame emits it once per distinct metadata value, however many compressions read it, and the immediate that wrote it is public bytecode. There is no block-length field: the counter is what states how many of the 64 bytes are message, so only the last block may be partial and the program must zero-fill the bytes past its real length, which the compression circuit does not enforce. A multi-block hash therefore feeds each result back with `cv=`, advances `counter=` by the bytes actually absorbed, and sets `final=1` on the last block. - -Operands are size-2 `StackBuf`s or 2-cell slices: - -- an **input operand written as a list**, `blake2s([a, b], [c, d], out)`, names its two words directly and allocates nothing: the opcode addresses its four input chunks independently, so an operand whose words live in different places never has to be gathered into a consecutive run. This is the spelling to reach for instead of `p = StackBuf(2); p[0] = a; p[1] = b`; -- **stack operands** are read in place, at zero copies; a self-hash `blake2s(h, h, out)` names one 2-cell pair as both inputs; -- the instruction addresses its **four canonical 128-bit message chunks independently** (each is a full F192 memory cell constrained at this use to the BLAKE2s subspace `c2 = 0`), so an operand gathered into a buffer (`p = StackBuf(2); p[0] = t0; p[1] = t1; blake2s(p, …)`) costs one instruction per assembling store, which the list form above avoids entirely; -- the chaining value has only one opcode offset and therefore must be consecutive. If a 2-cell `cv` was assembled from non-adjacent copied cells, the compiler materializes those two cells into a fresh consecutive run; -- **heap slices** are still bridged through the stack for the *input pull* (the operand's words come from the heap): +1 `DEREF` per heap cell, and the output, if a heap slice, is stored after: write-once memory fills whichever side is unset. - -If `out` was already written, the statement *asserts* the digest equals it, write-once turning the hash into a verification, which is exactly what a signature verifier wants. - -The compression, including its chaining value and metadata, is proven by the flock-derived BLAKE2s R1CS (`crates/flock`, see `doc.pdf` §BLAKE2s); one instruction is one 64-byte-block compression. - -## Hints: `hint_witness(dest, "name")` - -```python -sb = StackBuf(2) -hint_witness(sb, "r") # fill the whole StackBuf -hint_witness(hb[0:3], "h") # or any StackBuf/HeapBuf slice (any length) -assert log(sb[0]) < 8 # hinted values are UNCONSTRAINED: pin them down -``` - -A single hinted value needs no destination at all: - -```python -m = hint_witness("m") # one value, bound to a name -assert log m < 8 # still unconstrained: pin it -``` - -which is the one-line form of allocating a `StackBuf(1)`, filling a slice of it, and reading the cell back out, and costs exactly the same (nothing). Everything below about a stream's entries applies to it: each such binding pops one entry, whose length must be 1. - -Prover-supplied data (leanVM's `hint_witness`): a stream is a sequence of **entries**, one slice of values per `hint_witness` call, and the same symbol may be hinted many times. Each call pops the stream's next entry (whose length must match the destination run) and writes it into `dest` through the hint mechanism, at **zero cycles**. The values are completely unconstrained; the program must constrain them itself (asserts, range checks, hashes): an unconstrained hint consumed by anything security-relevant is a critical vulnerability. Runtime-start heap slices (`buf[i:i + k]`, `k` a literal) work too. - -The prover supplies streams with `program.set_witness("name", entries)` (`Vec>`); test programs declare them as annotations, one line per entry, and repeated lines with the same name are its successive entries: - -```python -# witness r: GEN ** 5, 12 -# witness r: 9 -``` - -### Computed-advice hints - -Three builtins have the prover compute the values at witness generation instead of popping a stream entry. Like `hint_witness`, the results are completely unconstrained: the program must re-verify them in-circuit. - -- `hint_decompose_bits(bits, value, nbits)`: writes the low `nbits` bits of `value` into the buffer `bits`, one field element (`0`/`1`) per bit. -- `hint_decompose_bits_exponent(bits, x, nbits)`: writes the `nbits` bits of the exponent `n` where `x = GEN ** n` into `bits` (a bounded dlog at witness generation). -- `g = hint_log2_ceil(bits, nbits, floor)`: returns `GEN ** log2_ceil(v)` for the value `v` held bitwise in the `nbits`-bit buffer `bits`, floored at `floor`. - -`bits` is a `HeapBuf` or a `StackBuf` (of at least `nbits` cells). Prefer the `StackBuf`: a frame cell is addressed directly, so `bits[i]` at a compile-time index is free where a heap read is a `DEREF`, and the booleanity pin `bits[i] = b * b` is then one `MUL` rather than a `MUL` and a `DEREF`. Use `addr` below where the run must also be indexed at runtime or reached from elsewhere. - -## Cost cheat sheet - -| construct | instructions | -|---|---| -| `x = ` / `GEN ** k` | 1 `SET` | -| `a + b` | 1 `XOR` | -| `a * b` | 1 `MUL` | -| `a / b` | 1 `MUL` (write-once back-solve; division by zero is undefined) | -| heap read / store `buf[i]` | 1 `DEREF`; +1 `MUL` for a *runtime* index (a compile-time g-power offset folds into the `DEREF`, for free) | -| stack read `sa[k]` | 0 (direct cell addressing); a *store* is 1, like any other write | -| `assert a == b` | 1 (+ 1 `SET` amortized per frame for the zero cell) | -| `assert a != b` | 3 (`XOR`, `MUL`, `SET`), no branch, one hinted inverse | -| `assert log x < k` | 3 (+1 `SET` amortized per bound per frame; a runtime bound costs 1 `MUL` instead) | -| `if a == b: …` | 3 (+2 to skip a non-empty `else`; +2 amortized `self-fp` per branching function); **0 if the condition is compile-time** | -| `… = match(log(x), …)` | ≈ 7 for the dispatch + the arm; results written into the targets directly. Uniform-call arms (`lambda k: f(a, b, k)`) **fuse**: one shared frame + dispatch to entry, each arm just `SET`+`JUMP`; `@inline` arms run in place, with no frame | -| function call | ≈ `n_args + n_returns + 4` (0 when the callee is `@inline`) | -| `mul_range` iteration | body + ≈ 1 `MUL` + 1 `XOR` + call overhead | -| `unroll` iteration | body only (compile-time replication) | -| `blake2s(a, b, out, ...)` | 1; plus one `SET` once per frame per distinct metadata value (nothing with `md=`, which costs whatever building the word costs), and two more when `cv` is omitted; message/CV words are read in place, +1 `DEREF` per heap input or CV word, +1 `MUL` per runtime slice start | -| `hint_witness(dest, "name")` | 0 (+1 `MUL` for a runtime slice start) | - -Every cost above is the FIRST occurrence. Two identical pure operations in one function share one cell and the second is free, so `hb[i]` twice, or `row[i]` where `row = hb * GEN ** 2`, costs one pointer `MUL` between them. The sharing stops at a branch: a cell whose instruction sits inside an `if` is not reused after the join, because the other path leaves it unwritten and therefore prover-chosen. - -## Example - -Fibonacci in the exponent (`tests/programs/fibonacci.py`): `fib[g^k]` holds `GEN ** F_k`, so one field `MUL` is one Fibonacci step. - -```python -# public_input: GEN ** 89, GEN ** 89 -from snark_lib import * - - -def main(): - fib = HeapBuf(12) - fib[1] = GEN ** 0 # F_0 = 0 - fib[GEN] = GEN # F_1 = 1 - for i in mul_range(1, GEN ** 10): - fib[i * GEN * GEN] = fib[i] * fib[i * GEN] - out = fib[GEN ** 11] - assert out == GEN ** 89 # F_11 = 89 - assert log(out) < log(GEN ** 128) - p = GEN ** 0 - p[1] = out - p[GEN] = out - return -``` - -## Not (yet) supported - -Mutable variables; conditions other than field (in)equality; `match` default and non-contiguous arms; multi-file imports; `Const` parameters as `mul_range` or range-check bounds (a substituted literal is a bit-pattern element, not the g-power a bound needs); runtime slice starts on a `StackBuf`; precompiles beyond `BLAKE2s`. diff --git a/crates/lean_da/Cargo.toml b/crates/lean_da/Cargo.toml deleted file mode 100644 index 79c118ad7..000000000 --- a/crates/lean_da/Cargo.toml +++ /dev/null @@ -1,19 +0,0 @@ -[package] -name = "lean_da" -version.workspace = true -edition.workspace = true -publish = false - -[lints] -workspace = true - -[dependencies] -primitives.workspace = true -parallel.workspace = true -pcs.workspace = true -fiat_shamir.workspace = true -serde.workspace = true -tracing.workspace = true - -[dev-dependencies] -rand.workspace = true diff --git a/crates/lean_da/src/commit.rs b/crates/lean_da/src/commit.rs deleted file mode 100644 index 8e8d594f8..000000000 --- a/crates/lean_da/src/commit.rs +++ /dev/null @@ -1,223 +0,0 @@ -//! Two commitment branches over the same cell digests. -//! -//! A row digest hashes its first `k/c` cell digests, covering the systematic -//! payload. A Merkle tree over these digests gives `root_row`. -//! A second tree has all cell digests as leaves, in column-major order; its -//! intermediate column roots authenticate samples, and its root is `root_col`. -//! The final commitment is `H(root_row, root_col)`. - -use fiat_shamir::merkle::{Hash, hash_pair}; -use primitives::hash::{OUT_LEN, hash, hash_many_dyn}; - -use crate::{CELL_SYMBOLS, CELLS_PER_ROW, CODEWORD_SYMBOLS, PAYLOAD_CELLS, encode_rows, row_count}; - -/// What the builder publishes. -#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub struct DaCommitment { - /// `H(root_row, root_col)`, binding the matrix including its zero padding. - pub root: Hash, - pub root_row: Hash, - pub root_col: Hash, -} - -/// What the builder keeps, to answer samples and to feed the proof. -pub struct DaWitness { - /// `n_rows · m` symbols, row-major. Padding rows are zero and not stored. - pub codewords: Vec, - /// `n_rows · ℓ` cell digests, row-major. - pub cell_digests: Vec, - /// The flat tree over the column-major cell digests; see the module docs. - pub column_tree: Vec, - /// The flat tree over the `n_padded` row digests. - pub row_tree: Vec, -} - -impl DaWitness { - /// The `ℓ` column roots `C_j`, the level at height `log n_padded`. - pub fn column_roots(&self) -> &[Hash] { - let (n_pad, cells) = ( - row_count(self.codewords.len(), CODEWORD_SYMBOLS).next_power_of_two(), - CELLS_PER_ROW, - ); - let offset = 2 * n_pad * cells - 2 * cells; - &self.column_tree[offset..offset + cells] - } -} - -/// Commit to payload rows of little-endian 64-bit symbols and their systematic encoding. -#[tracing::instrument(name = "Commit", skip_all)] -pub fn commit(rows: &[u64]) -> (DaCommitment, DaWitness) { - let codewords = encode_rows(rows); - commit_codewords(codewords) -} - -/// [`commit`] over rows that are already encoded. -pub fn commit_codewords(codewords: Vec) -> (DaCommitment, DaWitness) { - let n_rows = row_count(codewords.len(), CODEWORD_SYMBOLS); - let (cells, n_pad, t) = (CELLS_PER_ROW, n_rows.next_power_of_two(), PAYLOAD_CELLS); - let cell_digests = hash_cells(&codewords); - let (padding_cell, padding_row) = padding_digests(); - - // Row branch: hash each payload's cell digests. - let mut prefixes = vec![Hash::default(); n_rows * t]; - parallel::chunks_mut(&mut prefixes, t, |i, prefix| { - prefix.copy_from_slice(&cell_digests[i * cells..i * cells + t]); - }); - let mut row_digests = vec![padding_row; n_pad]; - hash_many_dyn( - prefixes.as_flattened(), - t * OUT_LEN, - row_digests[..n_rows].as_flattened_mut(), - ); - drop(prefixes); - let row_tree = tree_from_leaves(row_digests); - - // Column branch: the same digests, column-major, one tree carrying both levels. - let mut column_major = vec![Hash::default(); cells * n_pad]; - parallel::chunks_mut(&mut column_major, n_pad, |j, column| { - for (i, slot) in column.iter_mut().enumerate() { - *slot = if i < n_rows { - cell_digests[i * cells + j] - } else { - padding_cell - }; - } - }); - let column_tree = tree_from_leaves(column_major); - - let (root_row, root_col) = (*row_tree.last().unwrap(), *column_tree.last().unwrap()); - let commitment = DaCommitment { - root: hash_pair(&root_row, &root_col), - root_row, - root_col, - }; - let witness = DaWitness { - codewords, - cell_digests, - column_tree, - row_tree, - }; - (commitment, witness) -} - -/// Shape-dependent padding digests: the zero cell and its repeated digest for a row. -pub fn padding_digests() -> (Hash, Hash) { - let cell = hash(&[0u8; CELL_SYMBOLS * size_of::()]); - (cell, hash([cell; PAYLOAD_CELLS].as_flattened())) -} - -/// `e_{i,j} = H(W_{i,j})` for every cell of every real row, row-major. -#[tracing::instrument(name = "Hashing cells", skip_all)] -fn hash_cells(codewords: &[u64]) -> Vec { - let (cells, m) = (CELLS_PER_ROW, CODEWORD_SYMBOLS); - let n_rows = codewords.len() / m; - let mut digests = vec![Hash::default(); n_rows * cells]; - parallel::chunks_mut(&mut digests, cells, |i, row| { - hash_many_dyn( - as_bytes(&codewords[i * m..(i + 1) * m]), - CELL_SYMBOLS * size_of::(), - row.as_flattened_mut(), - ); - }); - digests -} - -/// The flat Merkle tree over leaves that are already digests: `tree[..n]` is the -/// leaves, then each level in turn, the root last. Unlike [`pcs::merkle`] the -/// leaves are not re-hashed, since a cell digest is already the leaf. -fn tree_from_leaves(mut tree: Vec) -> Vec { - let n = tree.len(); - assert!(n.is_power_of_two(), "leaf count must be a power of two"); - tree.resize(2 * n - 1, Hash::default()); - - let (mut base, mut width) = (0, n); - while width > 1 { - let (read, write) = tree.split_at_mut(base + width); - hash_many_dyn( - read[base..].as_flattened(), - 2 * OUT_LEN, - write[..width / 2].as_flattened_mut(), - ); - base += width; - width /= 2; - } - tree -} - -fn as_bytes(data: &[u64]) -> &[u8] { - const { assert!(cfg!(target_endian = "little"), "digests use little-endian symbols") }; - // SAFETY: u64 has no padding; the endian check fixes the byte representation. - unsafe { core::slice::from_raw_parts(data.as_ptr().cast::(), size_of_val(data)) } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::BLOB_SYMBOLS; - use rand::{Rng, SeedableRng, rngs::StdRng}; - - fn payload(n_rows: usize, seed: u64) -> Vec { - let mut rng = StdRng::seed_from_u64(seed); - (0..n_rows * BLOB_SYMBOLS).map(|_| rng.random()).collect() - } - - /// `column_roots` indexes into the flat tree by an offset derived from the - /// shape, which is the easiest thing here to get wrong by a level. Each one has - /// to be the root of its own column's subtree, rebuilt independently. - #[test] - fn column_roots_are_their_own_subtrees() { - let n_rows = 5usize; - let (_, witness) = commit(&payload(n_rows, 1)); - let (n_pad, cells) = (n_rows.next_power_of_two(), CELLS_PER_ROW); - let padding_cell = hash(&[0u8; CELL_SYMBOLS * size_of::()]); - - for (j, &got) in witness.column_roots().iter().enumerate() { - let column: Vec = (0..n_pad) - .map(|i| { - if i < n_rows { - witness.cell_digests[i * cells + j] - } else { - padding_cell - } - }) - .collect(); - assert_eq!(got, *tree_from_leaves(column).last().unwrap(), "column {j}"); - } - } - - /// Both branches have to be binding, and independently so: a change inside the - /// first half must move `root_row`, and a change in the second half must - /// still move `root_col` even though no row digest covers it. - #[test] - fn both_branches_bind() { - let n_rows = 3usize; - let codewords = crate::encode_rows(&payload(n_rows, 2)); - let (base, _) = commit_codewords(codewords.clone()); - - for &position in &[0usize, BLOB_SYMBOLS - 1, BLOB_SYMBOLS, CODEWORD_SYMBOLS - 1] { - let mut corrupted = codewords.clone(); - corrupted[position] ^= 1; - let (moved, _) = commit_codewords(corrupted); - assert_ne!(moved.root, base.root, "root ignored a flip at {position}"); - assert_ne!(moved.root_col, base.root_col, "root_col ignored a flip at {position}"); - if position < BLOB_SYMBOLS { - assert_ne!(moved.root_row, base.root_row, "root_row ignored a payload flip"); - } - } - } - - /// Padding rows are zero codewords sharing one cell digest. A payload whose - /// real rows are unchanged must commit identically whether or not the row count - /// happens to be a power of two, up to the padding the shape declares. - #[test] - fn padding_rows_are_the_zero_codeword() { - let n_rows = 3usize; - let rows = payload(n_rows, 3); - let mut extended = rows.clone(); - extended.extend(std::iter::repeat_n(0, BLOB_SYMBOLS)); - - let (short, _) = commit(&rows); - let (long, _) = commit(&extended); - assert_eq!(short.root, long.root); - } -} diff --git a/crates/lean_da/src/encode.rs b/crates/lean_da/src/encode.rs deleted file mode 100644 index 1f3f27376..000000000 --- a/crates/lean_da/src/encode.rs +++ /dev/null @@ -1,84 +0,0 @@ -//! Systematic Reed-Solomon encoding at rate 1/2. -//! -//! An inverse additive NTT interpolates each payload row into novel-basis -//! coefficients. Encoding these on the full domain preserves the payload in -//! the first half of the codeword and adds redundancy in the second half. - -use pcs::ntt::AdditiveNttF64; -use primitives::field::F64; - -use crate::{BLOB_SYMBOLS, CODEWORD_SYMBOLS, DA_LOG_K, LOG_M, row_count}; - -/// Encode `n` rows of `k` symbols into `n` codewords of `m` symbols, row-major. -/// -/// `rows` is the payload, `n_rows · k` long. The returned buffer is `n_rows · m` -/// long, with each payload row unchanged in its codeword's first half. -/// Symbols are little-endian 64-bit words; every bit pattern is valid. -/// Padding rows are not materialized, being zero codewords. -#[tracing::instrument(name = "Encoding", skip_all)] -pub fn encode_rows(rows: &[u64]) -> Vec { - let n_rows = row_count(rows.len(), BLOB_SYMBOLS); - let (k, m) = (BLOB_SYMBOLS, CODEWORD_SYMBOLS); - let interpolation = AdditiveNttF64::standard(DA_LOG_K); - let ntt = AdditiveNttF64::standard(LOG_M); - - // One row at a time: the transform dispatches internally at any row worth - // encoding, and the pool panics on a nested dispatch, so the row loop must stay - // sequential and let the NTT own the parallelism. - let mut codewords = vec![0; n_rows * m]; - for (i, codeword) in codewords.chunks_exact_mut(m).enumerate() { - codeword[..k].copy_from_slice(&rows[i * k..(i + 1) * k]); - let codeword = as_field_mut(codeword); - interpolation.inverse_transform(&mut codeword[..k]); - ntt.encode_interleaved_in_place(codeword, 1, 1); - } - codewords -} - -/// Borrow native symbols as field elements without allocating or copying. -pub(crate) fn as_field_mut(data: &mut [u64]) -> &mut [F64] { - // SAFETY: F64 is repr(transparent) over u64, with every bit pattern valid. - unsafe { core::slice::from_raw_parts_mut(data.as_mut_ptr().cast::(), data.len()) } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn encoding_preserves_payload_and_degree_bound() { - let payload: Vec = (0..3 * BLOB_SYMBOLS) - .map(|i| 0x9E37_79B9_7F4A_7C15u64.wrapping_mul(i as u64 + 1)) - .collect(); - let mut codewords = encode_rows(&payload); - let ntt = AdditiveNttF64::standard(LOG_M); - for (row, codeword) in payload - .as_chunks::() - .0 - .iter() - .zip(codewords.as_chunks_mut::().0.iter_mut()) - { - assert_eq!(&codeword[..BLOB_SYMBOLS], row); - ntt.inverse_transform(as_field_mut(codeword)); - assert!(codeword[BLOB_SYMBOLS..].iter().all(|&x| x == 0)); - } - } - - /// Encoding commutes with XOR of payloads. - #[test] - fn encoding_is_linear() { - let n = 2 * BLOB_SYMBOLS; - let a: Vec = (0..n) - .map(|i| 0x9E37_79B9_7F4A_7C15u64.wrapping_mul(i as u64 + 1)) - .collect(); - let b: Vec = (0..n) - .map(|i| 0xBF58_476D_1CE4_E5B9u64.wrapping_mul(i as u64 + 3)) - .collect(); - let sum: Vec = a.iter().zip(&b).map(|(&x, &y)| x ^ y).collect(); - - let (ca, cb, cs) = (encode_rows(&a), encode_rows(&b), encode_rows(&sum)); - for ((&x, &y), &z) in ca.iter().zip(&cb).zip(&cs) { - assert_eq!(x ^ y, z); - } - } -} diff --git a/crates/lean_da/src/lib.rs b/crates/lean_da/src/lib.rs deleted file mode 100644 index 622d176ee..000000000 --- a/crates/lean_da/src/lib.rs +++ /dev/null @@ -1,62 +0,0 @@ -//! LeanDA over `K = GF(2^64)` with BLAKE2s, following -//! . -//! -//! Each payload row contains `k` symbols, systematically encoded to `m = 2k` -//! evaluations on an additive domain. The row branch commits to the first `k` -//! evaluations, which are the original payload; the column branch authenticates -//! sampled cells. A row belongs to the code iff it is orthogonal to the commitment's -//! [`membership_vector`], with high probability over the Fiat-Shamir challenges. -//! The aggregation guest proves this check together with both commitment branches. -//! -//! Row and cell widths are powers of two. Trees pad the row count with zero rows. -//! A root binds this padded matrix, not the original row count: a trailing zero -//! row is indistinguishable from padding within the same padded height. - -mod commit; -mod encode; -mod membership; - -pub use commit::{DaCommitment, DaWitness, commit, commit_codewords, padding_digests}; -pub use encode::encode_rows; -pub use membership::{dual_codeword, membership_challenges, membership_vector, vector_digest}; - -/// Payload symbols per blob, as a logarithm (128 KiB). -pub const DA_LOG_K: usize = 14; -/// Symbols per sampling cell, as a logarithm (2 KiB). -pub const DA_LOG_CELL: usize = 8; -/// Maximum blobs in one commitment. -pub const DA_MAX_ROWS: usize = 1024; - -/// 64-bit symbols in one payload blob. -pub const BLOB_SYMBOLS: usize = 1 << DA_LOG_K; -/// 64-bit symbols in one encoded blob. -pub const CODEWORD_SYMBOLS: usize = 2 * BLOB_SYMBOLS; -/// 64-bit symbols in one sampling cell. -pub const CELL_SYMBOLS: usize = 1 << DA_LOG_CELL; -/// Sampling cells in one encoded blob. -pub const CELLS_PER_ROW: usize = CODEWORD_SYMBOLS / CELL_SYMBOLS; -const PAYLOAD_CELLS: usize = BLOB_SYMBOLS / CELL_SYMBOLS; -const LOG_M: usize = DA_LOG_K + 1; - -fn row_count(symbols: usize, width: usize) -> usize { - assert!(symbols.is_multiple_of(width), "partial blob row"); - let rows = symbols / width; - assert!((1..=DA_MAX_ROWS).contains(&rows), "expected 1..=DA_MAX_ROWS blobs"); - rows -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn row_counts_reject_empty_partial_and_oversized_inputs() { - for width in [BLOB_SYMBOLS, CODEWORD_SYMBOLS] { - assert_eq!(row_count(width, width), 1); - assert_eq!(row_count(DA_MAX_ROWS * width, width), DA_MAX_ROWS); - for symbols in [0, width - 1, width + 1, (DA_MAX_ROWS + 1) * width, usize::MAX] { - assert!(std::panic::catch_unwind(|| row_count(symbols, width)).is_err()); - } - } - } -} diff --git a/crates/lean_da/src/membership.rs b/crates/lean_da/src/membership.rs deleted file mode 100644 index 0159ce8e1..000000000 --- a/crates/lean_da/src/membership.rs +++ /dev/null @@ -1,194 +0,0 @@ -//! Reed-Solomon membership on an additive domain. -//! -//! For an additive domain `D` of size `m`, `Σ_{x∈D} h(x) = 0` when -//! `deg(h) <= m-2`. Pairing codewords and comparing dimensions therefore gives -//! `RS_k(D)^⊥ = RS_{m-k}(D)`. At rate 1/2 the code is self-dual. -//! -//! The check uses `L(X) = ∏_{j < log2(k)} (1 + z_j · Vhat_j(X))`, where -//! `Vhat_j = X_{2^j}` is the normalized subspace polynomial of degree `2^j`. -//! Expanding this product gives every novel-basis coefficient below `k` with -//! tensor weights `⊗_j(1, z_j)`, so `L` is a codeword. -//! -//! For a fixed invalid row `w`, `⟨L,w⟩` is a nonzero multilinear polynomial in -//! `z`. Fresh independent uniform challenges accept it with probability at most -//! `log2(k)/2^192`. The external verifier derives the vector from the commitment. -//! The guest binds its hinted vector by hashing it and checks the inner products. - -use fiat_shamir::FiatShamirState; -use fiat_shamir::merkle::hash_to_scalars; -use pcs::ntt::AdditiveNttF64; -use primitives::field::{F64, F192}; - -use crate::{CODEWORD_SYMBOLS, DA_LOG_K, LOG_M}; - -/// Transcript label, so a membership challenge can never be replayed as any other -/// challenge in the stack. -const LABEL: &[u8] = b"leanDA/rs-membership/v1"; - -/// `z_0 … z_{log k - 1}`, bound to the commitment. -/// -/// The matrix root binds both commitment branches. The vector hash is not observed, -/// avoiding a circular dependency between the challenges and the vector. -pub fn membership_challenges(root: &[u8; 32]) -> Vec { - let mut fs = FiatShamirState::from_label(LABEL); - for scalar in hash_to_scalars(root) { - fs.observe(scalar); - } - fs.sample_vec(DA_LOG_K) -} - -/// The test vector deterministically derived from a matrix root. -pub fn membership_vector(root: &[u8; 32]) -> Vec { - dual_codeword(&membership_challenges(root)) -} - -/// BLAKE2s of the vector's three little-endian 64-bit limbs per entry, in domain order. -pub fn vector_digest(vector: &[F192]) -> [u8; 32] { - assert_eq!(vector.len(), CODEWORD_SYMBOLS); - let bytes: Vec = vector - .iter() - .flat_map(|v| [v.c0, v.c1, v.c2].into_iter().flat_map(u64::to_le_bytes)) - .collect(); - primitives::hash::hash(&bytes) -} - -/// `L` over the whole domain: the tensor `⊗_j (1, z_j)` encoded as a codeword. -/// -/// The additive NTT's twiddles are `K`-valued and the transform is `K`-linear, so -/// an `E`-valued transform is three `K`-transforms on the limbs. `F192` is -/// `repr(C)` over three `u64`, which is exactly the interleaved layout -/// [`AdditiveNttF64::encode_interleaved_in_place`] wants for `num_ntts = 3`. -pub fn dual_codeword(z: &[F192]) -> Vec { - assert_eq!(z.len(), DA_LOG_K, "one challenge per novel-basis variable"); - let m = CODEWORD_SYMBOLS; - - let mut buffer = vec![F192::ZERO; m]; - buffer[0] = F192::ONE; - for (j, &zj) in z.iter().enumerate() { - let (low, high) = buffer[..1 << (j + 1)].split_at_mut(1 << j); - for (h, &l) in high.iter_mut().zip(low.iter()) { - *h = l * zj; - } - } - - // SAFETY: `F192` is `repr(C)` over three `u64` and `F64` is `repr(transparent)` - // over `u64`, so the buffer is `3m` limb-interleaved `F64` words with no padding. - let limbs = unsafe { core::slice::from_raw_parts_mut(buffer.as_mut_ptr().cast::(), 3 * m) }; - AdditiveNttF64::standard(LOG_M).encode_interleaved_in_place(limbs, 3, 1); - buffer -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::encode_rows; - use crate::{BLOB_SYMBOLS, DaCommitment}; - use rand::{Rng, SeedableRng, rngs::StdRng}; - - /// Draw challenges from the commitment and test every row for membership. - /// The caller must separately bind `codewords` to the commitment. - fn check_membership(commitment: &DaCommitment, codewords: &[u64]) -> bool { - let dual = membership_vector(&commitment.root); - codewords.chunks(CODEWORD_SYMBOLS).all(|row| { - let residual = dual - .iter() - .zip(row) - .fold(F192::ZERO, |acc, (&l, &w)| acc + l.mul_base(F64(w))); - residual == F192::ZERO - }) - } - - fn random_rows(rng: &mut StdRng, n: usize) -> Vec { - (0..n).map(|_| rng.random()).collect() - } - - /// The claim the whole check rests on: over an F_2-subspace domain at rate - /// 1/2 the code is self-dual, so any two codewords are orthogonal. If this - /// fails, `check_membership` is testing the wrong relation. - #[test] - fn code_is_self_dual() { - let mut rng = StdRng::seed_from_u64(3); - for _ in 0..4 { - let a = encode_rows(&random_rows(&mut rng, BLOB_SYMBOLS)); - let b = encode_rows(&random_rows(&mut rng, BLOB_SYMBOLS)); - let dot = a.iter().zip(&b).fold(F64::ZERO, |acc, (&x, &y)| acc + F64(x) * F64(y)); - assert_eq!(dot, F64::ZERO); - } - } - - /// `dual_codeword` runs one interleaved transform over the three `F192` limbs - /// in place, which leans on `F192` being `repr(C)` over three `u64`. A single-lane - /// transform of each limb must give the same codeword. - #[test] - fn dual_codeword_matches_limbwise_transform() { - let mut rng = StdRng::seed_from_u64(5); - let z: Vec = (0..DA_LOG_K) - .map(|_| F192::new(rng.random(), rng.random(), rng.random())) - .collect(); - let dual = dual_codeword(&z); - - let mut tensor = vec![F192::ZERO; BLOB_SYMBOLS]; - tensor[0] = F192::ONE; - for (j, &zj) in z.iter().enumerate() { - let (low, high) = tensor[..1 << (j + 1)].split_at_mut(1 << j); - for (h, &l) in high.iter_mut().zip(low.iter()) { - *h = l * zj; - } - } - let ntt = AdditiveNttF64::standard(LOG_M); - for limb in 0..3 { - let mut encoded = vec![F64::ZERO; CODEWORD_SYMBOLS]; - for (out, c) in encoded.iter_mut().zip(&tensor) { - *out = F64([c.c0, c.c1, c.c2][limb]); - } - ntt.encode_interleaved_in_place(&mut encoded, 1, 0); - for (x, (&got, &want)) in dual.iter().zip(&encoded).enumerate() { - assert_eq!(F64([got.c0, got.c1, got.c2][limb]), want, "limb {limb} at {x}"); - } - } - } - - /// An honestly encoded payload passes, and flipping a single symbol of a single - /// row fails. That single flip is the case the check exists for: it is exactly - /// what a builder would do to make a row unrecoverable while still answering - /// every sample consistently. - #[test] - fn membership_accepts_codewords_and_rejects_a_flipped_symbol() { - let n_rows = 5; - let mut rng = StdRng::seed_from_u64(11); - let rows = random_rows(&mut rng, n_rows * BLOB_SYMBOLS); - let codewords = encode_rows(&rows); - let (commitment, _) = crate::commit_codewords(codewords.clone()); - - assert!(check_membership(&commitment, &codewords)); - - for &position in &[0usize, 1, BLOB_SYMBOLS - 1, BLOB_SYMBOLS, CODEWORD_SYMBOLS - 1] { - let mut corrupted = codewords.clone(); - let index = 3 * CODEWORD_SYMBOLS + position; - corrupted[index] ^= 1; - let (corrupted_commitment, _) = crate::commit_codewords(corrupted.clone()); - assert!( - !check_membership(&corrupted_commitment, &corrupted), - "a flip at {position} passed the membership check" - ); - } - } - - /// A row that is a codeword of the wrong degree (the full domain rather than - /// the degree bound) must fail: rate is what the check enforces, and a - /// rate-1 "codeword" is what an unrecoverable payload looks like. - #[test] - fn membership_rejects_a_full_degree_row() { - let n_rows = 2; - let mut rng = StdRng::seed_from_u64(13); - let rows = random_rows(&mut rng, n_rows * BLOB_SYMBOLS); - let mut codewords = encode_rows(&rows); - let ntt = AdditiveNttF64::standard(LOG_M); - let mut full = random_rows(&mut rng, CODEWORD_SYMBOLS); - ntt.encode_interleaved_in_place(crate::encode::as_field_mut(&mut full), 1, 0); - codewords[..CODEWORD_SYMBOLS].copy_from_slice(&full); - let (commitment, _) = crate::commit_codewords(codewords.clone()); - - assert!(!check_membership(&commitment, &codewords)); - } -} diff --git a/crates/lean_vm/Cargo.toml b/crates/lean_vm/Cargo.toml index e7d5326da..0d7c5abcd 100644 --- a/crates/lean_vm/Cargo.toml +++ b/crates/lean_vm/Cargo.toml @@ -17,5 +17,4 @@ tracing.workspace = true zk_alloc.workspace = true [dev-dependencies] -lean_compiler.workspace = true bincode.workspace = true diff --git a/crates/lean_vm/src/class_flock.rs b/crates/lean_vm/src/class_flock.rs new file mode 100644 index 000000000..463372885 --- /dev/null +++ b/crates/lean_vm/src/class_flock.rs @@ -0,0 +1,151 @@ +//! Bridge to flock for the instruction classes: each class's circuit +//! ([`crate::rv::circuits`]), proven over one packed witness of its own. +//! +//! That witness is one more committed column of the stacked witness: instance `j` of +//! the batch is row `j` of the class's table, and flock's R1CS validity is discharged +//! by the same stacked WHIR opening, through one ring-switched region per class. +//! +//! A register or bytecode word is 64 bits and the packing is 64 bits a word, so the +//! words a row puts on the bus ARE packed words of its instance, the circuit's ports +//! ([`ClassSpec::ports`]). The table's columns for them are therefore virtual, their +//! claims routed to those words. + +use crate::cpu::Row; +use crate::rv::Entry; +use crate::tables::{ClassSpec, N_TABLES, Word}; +use crate::transcript::{ProverState, VerifierState}; +use ::pcs::pack::LOG_PACKING; +use flock::circuit::Circuit; +use flock::reduction::{ReductionReplay, SliceClaim}; +use flock::verifier::VerifyError; +use primitives::field::F64; +use std::sync::OnceLock; +use zk_alloc::ArenaVec; + +/// The zerocheck's cube has at least this many variables (flock's univariate skip +/// plus its fixed-point dimensions), which floors the batch of a small circuit. +pub const MIN_CUBE_LOG: usize = 13; + +/// `log2` of an instance's packed words: the stride between consecutive instances' +/// same-port words. +pub const fn stride_log(spec: &ClassSpec) -> usize { + spec.k_log - LOG_PACKING +} + +/// Table `t`'s gate list, built once. +pub fn circuit(t: usize) -> &'static Circuit { + static CIRCUITS: [OnceLock; N_TABLES] = [const { OnceLock::new() }; N_TABLES]; + CIRCUITS[t].get_or_init(|| { + let spec = crate::tables::CLASSES[t]; + let circuit = (spec.circuit)(); + assert_eq!(circuit.k_log(), spec.k_log, "{}'s block size moved", spec.name); + assert_eq!( + circuit.n_input_words(), + spec.n_inputs, + "{}'s input ports moved", + spec.name + ); + circuit + }) +} + +/// `log2` of the batch proving `n_rows` instances: a power of two, at least flock's +/// stripe floor and at least what the zerocheck's cube needs. +pub const fn n_blocks_log(spec: &ClassSpec, n_rows: usize) -> usize { + let n = if n_rows > 8 { n_rows } else { 8 }; + let natural = n.next_power_of_two().trailing_zeros() as usize; + let floor = MIN_CUBE_LOG.saturating_sub(spec.k_log); + if natural > floor { natural } else { floor } +} + +/// A row's value of one circuit word. +fn word_of(word: Word, row: &Row, entry: &Entry) -> u64 { + match word { + Word::Flags => entry.flags, + Word::Imm => entry.imm, + Word::V1 => row.v1, + Word::V2 => row.v2, + Word::Out => row.out, + Word::Taken => row.taken as u64, + Word::Address => row.ram.address, + Word::Cell(k) => row.hash.as_ref().map_or(row.ram.old, |h| h.block[k as usize]), + Word::CellNew(k) => row.hash.as_ref().map_or(row.ram.new, |h| h.word_after(k as usize)), + Word::Bad => 0, + Word::HintQ => crate::rv::semantics::div_hints(row.v1, row.v2, entry.flags).0, + Word::HintR => crate::rv::semantics::div_hints(row.v1, row.v2, entry.flags).1, + } +} + +/// The flock-native tables of one class's batch, kept from the pass that wrote its +/// committed column so the reduction needs no second witness pass. +pub(crate) struct Prepared { + table: usize, + n_blocks_log: usize, + z: ArenaVec, + a: ArenaVec, + b: ArenaVec, + z_lincheck: ArenaVec, +} + +impl Prepared { + /// Build table `t`'s batch, one instance per row, and write its packed witness + /// into `window`, the class's committed column. + pub(crate) fn build(t: usize, rows: &[Row], entries: &[Entry], window: &mut [F64]) -> Self { + let spec = crate::tables::CLASSES[t]; + let n_blocks_log = n_blocks_log(spec, rows.len()); + assert_eq!( + rows.len(), + 1 << n_blocks_log, + "a table's rows fill its batch (cpu::filler)" + ); + let circuit = circuit(t); + let (z, a, b, z_lincheck) = circuit.generate_witness_by(rows, &rows[0], n_blocks_log, |row, words| { + for (word, &port) in words.iter_mut().zip(spec.ports) { + *word = word_of(port, row, &entries[row.index as usize]); + } + }); + assert_eq!(window.len(), z.len(), "the committed column is the wrong size"); + let stride = 1 << stride_log(spec); + // `F64` is `repr(transparent)` over `u64`, and the packing is bit `i` at + // position `i` on both sides. + const BATCH: usize = 1 << 10; + parallel::chunks_mut_zip(window, &z, stride * BATCH, |batch, dst, src| { + for (j, src) in src.chunks_exact(stride).enumerate() { + // What the circuit computed is what the interpreter did, or the bus + // would carry one and flock prove the other. + let row = &rows[batch * BATCH + j]; + for (k, &port) in spec.ports.iter().enumerate().skip(spec.n_inputs) { + let expected = word_of(port, row, &entries[row.index as usize]); + assert_eq!( + src[k], expected, + "{}'s circuit disagrees with the interpreter on {port:?}", + spec.name + ); + } + } + for (d, &s) in dst.iter_mut().zip(src) { + *d = F64(s); + } + }); + Self { + table: t, + n_blocks_log, + z, + a, + b, + z_lincheck, + } + } + + /// Flock's zerocheck then lincheck, leaving the one claim on the committed column. + pub(crate) fn prove(&self, ps: &mut ProverState) -> SliceClaim { + let block = circuit(self.table).block(); + let stage = block.prove_zerocheck(self.n_blocks_log, &self.z, &self.a, &self.b, ps); + block.prove_lincheck(self.n_blocks_log, stage, &self.z_lincheck, ps) + } +} + +/// The verifier's replay of table `t`'s reduction: zerocheck, then lincheck. +pub fn verify_reduction(t: usize, n_blocks_log: usize, vs: &mut VerifierState) -> Result { + circuit(t).block().verify(n_blocks_log, vs) +} diff --git a/crates/lean_vm/src/constraints.rs b/crates/lean_vm/src/constraints.rs index 53772b270..ab0f993f1 100644 --- a/crates/lean_vm/src/constraints.rs +++ b/crates/lean_vm/src/constraints.rs @@ -24,12 +24,11 @@ //! `eq(ζ_m, Y)·p(Y) + Y·u`. Its constant, quadratic and cubic coefficients are //! sent; the running claim fixes the linear coefficient. The verifier evaluates //! it at the challenge. Heights and `ζ` enter only the per-table `weights`, which -//! may be accumulated along the way or deferred to the end, as the recursion guest does. +//! may be accumulated along the way or deferred to the end. //! //! The eq point is the caller's, not a fresh one (the bus's GKR point `ζ`), which //! is what lets the forms' sums settle the bus. Batching derived in `doc/leanvm/main.tex` -//! §sec:air. Both sides take `n = max τ_t` from the announced heights; a recursive -//! verifier certifies that maximum with one hinted `g`-power (§recursion), so there +//! §sec:air. Both sides take `n = max τ_t` from the announced heights, so there //! are no rounds in which no table has joined. use crate::PAR_THRESHOLD; diff --git a/crates/lean_vm/src/cpu/execute.rs b/crates/lean_vm/src/cpu/execute.rs index 2e2811de2..3caacc30c 100644 --- a/crates/lean_vm/src/cpu/execute.rs +++ b/crates/lean_vm/src/cpu/execute.rs @@ -1,989 +1,260 @@ -//! The write-once execution interpreter: run the compiled program to produce -//! the final memory image and the per-opcode [`Trace`]. +//! Run the program on the reference interpreter ([`crate::rv::Machine`]) and record, +//! for every access, what the memory argument needs ([`Trace`]). -use std::collections::HashMap; - -use super::hints::MAX_CELLS; use super::*; -use primitives::{ - field::{F64, F192, mul_by_g}, - pretty_f64, pretty_integer, -}; +use crate::rv::{self, ADVICE_BASE, Class, LOG_REGS, Machine, RAM_BASE, Trap, hash, machine::compute}; +use crate::tables::{CLASSES, CLOCK_STRIDE, RAM_SLOT, RANGE_LOG, REG_SLOTS, block_slot}; +use primitives::field::{F64, mul_by_g}; pub struct Execution { - pub mem: Vec, // data memory after the run, write-once (size cells, power of two) - pub cycles: usize, // number of instructions the run executed (trace length) - pub mem_used: usize, // cells actually touched, before the power-of-two pad of `mem` - /// Rows per table before the fill blocks ran: the work the program itself does, as + /// The public output: `a0..a3` as the run left them. + pub output: [u64; 4], + pub cycles: usize, // number of rows proven, padding rows included + /// Rows per table before the padding rows: the work the program itself does, as /// against the power-of-two heights that get proven. Cost measurements want this one. pub base_counts: [usize; crate::tables::N_TABLES], - /// Cells an instruction read before anything wrote them, up to the halt, so the - /// value it read was ZERO here and prover-chosen in a proof: memory is a committed - /// array and the bus only forces accesses to one address to *agree*, never that - /// the address was written. `zkDSL.md` says don't; this is what says whether the - /// emitted code did. A non-empty list means a live value came from outside the - /// constraint system, so an `assert` on it is vacuous and a published value is - /// free: the source read a cell nothing stores, or the lowering dropped or - /// misplaced a store the source asked for. - /// - /// Legitimate unconstrained cells are absent by construction, not by - /// exemption: a range-check touch only links its two cells, reading neither, - /// and an arithmetic back-solve writes its operand before reading it. - pub unconstrained_reads: Vec, - pub(crate) trace: Trace, // rows + final access-count columns, emitted in the same walk + pub(crate) trace: Trace, // rows, final timestamps and counts, emitted in the same walk } -/// Why a run has no execution the interpreter can find: the program, on this public -/// input and advice, asks for something the machine cannot do. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ExecError { - /// The instruction the run failed at, or whose hints it failed in. - pub pc: u32, - /// Its function and source line ([`Program::site_at`]). - pub site: String, - pub fault: Fault, +/// Running read counts `g^{count}` of the two range arrays' entries, which every +/// read-write array's gap checks share. +struct Ranges { + lo: Vec, + hi: Vec, } -/// What the run asked for that the machine cannot do. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum Fault { - /// A second value for a written cell: a failed `assert`, two writes that disagree, - /// or a write to a cell an instruction already read, unwritten, as ZERO. `hint` - /// names the hint that wrote it, if one did. - Conflict { - cell: u32, - had: F192, - new: F192, - hint: Option<&'static str>, - }, - /// A `DEREF` through a word that is no small g-power: a wild pointer, or a failed - /// range check. - WildPointer { value: F192 }, - /// A word that has to be in `K` is not: a `JUMP` operand, or one a hint reads. - NotInK { what: &'static str, value: F192 }, - /// A word that has to be an address or a count is not a g-power in its range. - NotAGPower { what: &'static str, value: F64 }, - /// A `BLAKE2s` operand outside the 128-bit embedding. - NotCanonical { operand: &'static str, value: F192 }, - /// A `MUL` asked to solve `a·x = c` for `x` with `a = 0`. - BackSolveThroughZero, - /// The advice does not fit the program: a witness stream is missing or - /// exhausted, or an entry has the wrong length. - Witness(String), - /// The program allocates past the address space. - OutOfMemory, - /// The halt `pc` reached in a frame other than `main`'s. - HaltOutsideMain { fp: u32 }, - /// Too many instructions: runaway recursion, or a loop that never ends. - StepLimit, +impl Ranges { + /// One read of each range array, at the chunks of `gap`. + #[inline(always)] + fn read(&mut self, gap: u32) -> (F64, F64) { + let (lo, hi) = ((gap & ((1 << RANGE_LOG) - 1)) as usize, (gap >> RANGE_LOG) as usize); + let counts = (self.lo[lo], self.hi[hi]); + self.lo[lo] = mul_by_g(counts.0); + self.hi[hi] = mul_by_g(counts.1); + counts + } } -/// The longest run the interpreter executes. -const MAX_STEPS: usize = 100_000_000; +/// What the memory argument keeps per cell of one read-write array (§sec:memchan): +/// its last access, as the clock's exponent and as its g-power. +struct Cells { + last: Vec, + last_ts: Vec, +} -impl ExecError { - fn new(program: &Program, pc: u32, fault: Fault) -> Self { +impl Cells { + /// Every cell starts last accessed at the seed's `g^0`. + fn new(n: usize) -> Self { Self { - pc, - site: program.site_at(pc), - fault, + last: vec![0; n], + last_ts: vec![F64::ONE; n], } } -} - -impl std::fmt::Display for ExecError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "{} at pc {} (in {})", self.fault, self.pc, self.site) - } -} - -impl std::error::Error for ExecError {} - -/// A word as its limbs, most significant first. -fn word(w: F192) -> String { - format!("{:x}:{:x}:{:x}", w.c2, w.c1, w.c0) -} -impl std::fmt::Display for Fault { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::Conflict { cell, had, new, hint } => { - write!( - f, - "write-once conflict at cell {cell}: had {}, new {}", - word(*had), - word(*new) - )?; - match hint { - Some(hint) => write!(f, " (hint {hint})"), - None => Ok(()), - } - } - Self::WildPointer { value } => write!( - f, - "DEREF pointer is not a small g-power: a wild pointer, or a failed range check (value {})", - word(*value) - ), - Self::NotInK { what, value } => write!(f, "{what} is not a K-valued word ({})", word(*value)), - Self::NotAGPower { what, value } => write!(f, "{what} is not a g-power in range (0x{:016x})", value.0), - Self::NotCanonical { operand, value } => write!( - f, - "BLAKE2s {operand} cell is not a canonical 128-bit embedding: top limb 0x{:016x}", - value.c2 - ), - Self::BackSolveThroughZero => write!(f, "cannot back-solve MUL through a zero operand"), - Self::Witness(problem) => write!(f, "{problem}"), - Self::OutOfMemory => write!(f, "the program allocates past 2^{} cells", MAX_CELLS.ilog2()), - Self::HaltOutsideMain { fp } => write!(f, "the halt pc reached in frame {fp}, not main's"), - Self::StepLimit => write!(f, "step limit of {MAX_STEPS} exceeded (runaway recursion?)"), + /// Access `cell` at clock `y`, whose g-power is `ts`. + #[inline(always)] + fn access(&mut self, ranges: &mut Ranges, cell: usize, y: u32, ts: F64) -> Access { + let (x, x_ts) = (self.last[cell], self.last_ts[cell]); + // The clock only moves forward, and starts after the seed's zero. + let gap = y - x - 1; + let (count_lo, count_hi) = ranges.read(gap); + self.last[cell] = y; + self.last_ts[cell] = ts; + Access { + x: x_ts, + gap, + count_lo, + count_hi, } } } -/// A memory word interpreted as a K-valued address: valid only when both -/// extension limbs are zero (every g-power is a K-element). -fn as_addr(v: F192) -> Option { - (v.c1 == 0 && v.c2 == 0).then_some(F64(v.c0)) +/// A padding row's access: clock zero on both sides, so the identity holds with the +/// first entry of each range array. +fn padding_access(ranges: &mut Ranges) -> Access { + let (count_lo, count_hi) = ranges.read(0); + Access { + x: F64::ZERO, + gap: 0, + count_lo, + count_hi, + } } -/// [`as_addr`], or why the word is none. -fn in_k(what: &'static str, v: F192) -> Result { - as_addr(v).ok_or(Fault::NotInK { what, value: v }) +/// The access a row does not make: its columns do not exist, so nothing reads it. +fn padding_access_unread() -> Access { + Access { + x: F64::ZERO, + gap: 0, + count_lo: F64::ZERO, + count_hi: F64::ZERO, + } } -fn pop_witness<'a>( - witness: &'a HashMap>>, - positions: &mut HashMap<&'a str, usize>, - name: &'a str, - len: u32, -) -> Result<&'a [F192], Fault> { - let entries = witness - .get(name) - .ok_or_else(|| Fault::Witness(format!("no witness stream `{name}` (Program::set_witness)")))?; - let position = positions.entry(name).or_default(); - let entry = entries.get(*position).ok_or_else(|| { - Fault::Witness(format!( - "witness stream `{name}` exhausted (needs entry {}, has {})", - *position + 1, - entries.len() - )) - })?; - if entry.len() != len as usize { - return Err(Fault::Witness(format!( - "witness `{name}` entry {} holds {} values, the destination {len}", - *position, - entry.len() - ))); - } - *position += 1; - Ok(entry) +/// `ts·g^k`. +fn advance(ts: F64, k: u32) -> F64 { + (0..k).fold(ts, |t, _| mul_by_g(t)) } impl Program { - /// Run the program in write-once *fill* mode to produce its [`Execution`]: - /// the final memory image and the step count, or why it has none. The public - /// input seeds the first two memory cells `m[0], m[1]` (§sec:e2e-pi), and the - /// fill blocks bring every table to a power of two, growing one until the - /// witness reaches [`Program::min_log_committed`]. Compilation yields the - /// `Program`; executing it (here) and proving it are separate later phases. - pub fn execute(&self, public_input: [F192; 2]) -> Result { - // One interpretation of the program, then the fill. The blocks that bring every - // table to a power of two are cycles no program code enters (`cpu::filler`), so - // they run after the chain has halted, by which point the program's own row - // counts are final; a traversal costs exactly its block's size plus its closing - // jump, so the solve is exact and no second run is needed to correct it. - use super::hints::{BitsDest, GPow, RHint}; - - let ending_pc = (self.prog.len() - 1) as u32; // last bytecode slot, g^{B-1} - - // g^j and its reverse index g^j ↦ j, seeded for the pcs and return targets, - // the main frame and the range-check floor, then grown by each reservation. - let seed = (self.prog.len() + 2) - .max(self.main_frame as usize) - .max(1 << MIN_LOG_MEM); - let mut g = GPow::new(seed); - - // Dense write-once data memory (read path stays a vector for speed), the - // per-cell access count (g^{count}, default g^0 = 1), and each cell's state. - let n0 = self.main_frame.max(2) as usize; - let mut m = Mem { - cells: vec![F192::ZERO; n0], - state: vec![State::Unwritten; n0], - count: vec![F64::ONE; n0], - links: HashMap::new(), - unwritten_reads: Vec::new(), - program: self, - dbg_pc: 0, - dbg_hint: None, + /// Run the program on `input` and `advice`, recording every row, then write out the + /// padding rows that bring each table to a power of two ([`filler`]). A run that + /// traps has no proof. + pub fn execute(&self, input: [u64; rv::INPUT_WORDS], advice: &[u64]) -> Result { + let p = &self.rv; + let mut m = Machine::new(p, input, advice); + let adv_init: Vec = m.advice().iter().map(|&w| F64(w)).collect(); + let mut ranges = Ranges { + lo: vec![F64::ONE; 1 << RANGE_LOG], + hi: vec![F64::ONE; 1 << RANGE_LOG], }; - // Seed the public input into m[0], m[1] (addresses g^0, g^1, §sec:e2e-pi). - m.put(0, public_input[0])?; - m.put(1, public_input[1])?; - - // Per-pc bytecode execution count (g^{count}). - let mut bytecode_count: Vec = vec![F64::ONE; self.prog.len()]; - - let mut next_free = self.main_frame; - let (mut pc, mut fp) = (0u32, 0u32); - let mut steps = 0usize; - // Per-pc hint index. `self.hints` is keyed by pc, so probing it each step - // costs a hash of the counter for what is almost always a miss; the - // program has one bytecode slot per pc, so flatten the map into a dense - // table: 0 = no hints, otherwise the 1-based index into `hint_lists`. - let mut hint_at = vec![0u32; self.prog.len()]; - let mut hint_lists: Vec<&[RHint]> = Vec::with_capacity(self.hints.len()); - for (&hpc, hs) in &self.hints { - hint_lists.push(hs.as_slice()); - hint_at[hpc as usize] = hint_lists.len() as u32; - } - // `DBG_PROF=1`: per-pc step counts, printed as a per-function cycle - // profile after the run (needs `fn_ranges`, i.e. a compiled program). - let mut prof: Option> = std::env::var("DBG_PROF").is_ok().then(|| vec![0u64; self.prog.len()]); - - // Per-stream cursor into the named witness data (`hint_witness` pops - // sequentially). - let mut witness_positions: HashMap<&str, usize> = HashMap::new(); - // Baby-step table for `hint_decompose_bits_exponent`, built on first use. - let mut dlog_cache: Option<(GPow, F64)> = None; - - // Rows per table before the fill runs, and the cells the program read unwritten, - // both captured when the chain halts: the fill's rows exist to reach a - // power-of-two height and are soundness-neutral (doc §Filling the tables), so - // they read cells nobody writes as a matter of course. - let mut base_counts: Option<[usize; crate::tables::N_TABLES]> = None; - let mut unconstrained_reads: Vec = Vec::new(); - - // Per-opcode trace rows, accumulated during the walk and assembled into the - // `Trace` once the run finishes (alongside the final count columns). - let mut xor: Vec = Vec::new(); - let mut mul: Vec = Vec::new(); - let mut set: Vec = Vec::new(); - let mut deref: Vec = Vec::new(); - let mut jump: Vec = Vec::new(); - let mut blake2s: Vec = Vec::new(); - - // A cell that is not `Written` holds ZERO. A `DEREF` between two unwritten cells - // links them, and a write to either reaches the other at once. - #[derive(Clone, Copy, PartialEq, Eq)] - enum State { - Unwritten, - Linked, - Written, - } - - // The three dense per-cell vectors, kept in lockstep. Every method on the hot - // path is `#[inline(always)]`: they sit in the interpreter's opcode loop. - struct Mem<'a> { - cells: Vec, - state: Vec, - count: Vec, - /// The cells a `DEREF` linked each cell to. - links: HashMap>, - /// Cells an instruction read before anything wrote them. - unwritten_reads: Vec, - /// To name where a conflict happened. - program: &'a Program, - /// The pc of the currently executing instruction, and the name of the - /// computed-advice hint if the write comes from one, so a write-once - /// conflict can report where it happened. Plain fields rather than - /// thread-locals: this is written on every step, and a thread-local - /// costs a lazy-init check each time. - dbg_pc: u32, - dbg_hint: Option<&'static str>, - } - impl Mem<'_> { - // Grow the dense vectors so `idx` is in range. All accessed cells - // satisfy cell < next_free after their frame's allocation, so this - // only ever extends. - #[inline(always)] - fn ensure(&mut self, idx: usize) { - if idx >= self.cells.len() { - let n = idx + 1; - self.cells.resize(n, F192::ZERO); - self.state.resize(n, State::Unwritten); - self.count.resize(n, F64::ONE); - } - } - #[inline(always)] - fn written(&self, cell: u32) -> bool { - self.state.get(cell as usize) == Some(&State::Written) - } - // A hint's read, which writes nothing: what a hint computes is advice the - // instructions go on to check. - #[inline(always)] - fn get(&self, cell: u32) -> F192 { - self.cells.get(cell as usize).copied().unwrap_or(F192::ZERO) - } - // An instruction's read. An unwritten cell reads as ZERO, which is then written - // there: the row is filled from the final image, which has to agree. - #[inline(always)] - fn read(&mut self, cell: u32) -> F192 { - let c = cell as usize; - self.ensure(c); - if self.state[c] != State::Written { - self.state[c] = State::Written; - self.unwritten_reads.push(cell); - } - self.cells[c] - } - // Write-once store, carried to every cell linked to this one. - #[inline(always)] - fn put(&mut self, cell: u32, v: F192) -> Result<(), ExecError> { - if self.store(cell, v)? { - self.spread(cell, v)?; - } - Ok(()) + let mut regs = Cells::new(1 << LOG_REGS); + // RAM's cells, then the advice's, as the machine numbers them. + let mut ram = Cells::new((1 << p.log_ram) + (1 << p.log_advice)); + let cell_of = |address: u64| -> usize { + if address >= RAM_BASE { + ((address - RAM_BASE) / 8) as usize + } else { + (1 << p.log_ram) + ((address - ADVICE_BASE) / 8) as usize } - // `put` on one cell; whether it was linked. - #[inline(always)] - fn store(&mut self, cell: u32, v: F192) -> Result { - let c = cell as usize; - self.ensure(c); - let s = self.state[c]; - if s == State::Written && self.cells[c] != v { - return Err(self.conflict(cell, v)); - } - self.cells[c] = v; - self.state[c] = State::Written; - Ok(s == State::Linked) + }; + // Per-pc bytecode execution count (g^{count}). + let mut bytecode_count: Vec = vec![F64::ONE; p.entries.len()]; + let mut fetch = |index: usize| { + let v = bytecode_count[index]; + bytecode_count[index] = mul_by_g(v); + v + }; + let mut rows: [Vec; crate::tables::N_TABLES] = std::array::from_fn(|_| Vec::new()); + + // The clock, as an exponent and as its g-power: cycle 1, so that the first + // access comes strictly after the seeds. + let mut tick = CLOCK_STRIDE; + let mut ts = crate::tables::CLOCK_START; + while !m.halted() { + // A cell's first access is measured from the seed, so the whole run has + // to fit the range a gap can take (§sec:memchan). + if tick >= u32::MAX - block_slot(hash::WORDS) { + return Err(Trap::CycleCap); } - #[cold] - fn spread(&mut self, cell: u32, v: F192) -> Result<(), ExecError> { - let mut todo = vec![cell]; - while let Some(c) = todo.pop() { - for p in self.links.remove(&c).unwrap_or_default() { - if self.store(p, v)? { - todo.push(p); - } - } + let step = m.step()?; + let e = &p.entries[step.index]; + let table = crate::tables::table_of(e.class).expect("every class that runs has a table"); + let spec = CLASSES[table]; + // The register accesses, then the RAM access if the class has one. Their + // order here is the order of their columns, not of their clock slots. + let cells = [e.a1, e.a2, e.ad].map(|cell| cell as usize); + let n_regs = if spec.writes_register() { 3 } else { 2 }; + let mut acc: [Access; 4] = std::array::from_fn(|i| match cells.get(i) { + Some(&cell) if i < n_regs => { + regs.access(&mut ranges, cell, tick + REG_SLOTS[i], advance(ts, REG_SLOTS[i])) } - Ok(()) + _ => padding_access_unread(), + }); + if let Some(access) = step.ram { + let cell = cell_of(access.address); + acc[3] = ram.access(&mut ranges, cell, tick + RAM_SLOT, advance(ts, RAM_SLOT)); } - fn link(&mut self, a: u32, b: u32) { - for (x, y) in [(a, b), (b, a)] { - self.ensure(x as usize); - if self.state[x as usize] == State::Unwritten { - self.state[x as usize] = State::Linked; - } - self.links.entry(x).or_default().push(y); + // A hash row's block, word `k` at `v1 ^ 8k`, after its two register reads. + let hash = step.hash.map(|h| { + let mut all = [padding_access_unread(); 2 + hash::WORDS]; + all[..2].copy_from_slice(&acc[..2]); + for k in 0..hash::WORDS { + let cell = cell_of(step.v1 ^ (8 * k as u64)); + all[2 + k] = ram.access(&mut ranges, cell, tick + block_slot(k), advance(ts, block_slot(k))); } - } - #[cold] - fn conflict(&self, cell: u32, v: F192) -> ExecError { - let fault = Fault::Conflict { - cell, - had: self.cells[cell as usize], - new: v, - hint: self.dbg_hint, - }; - ExecError::new(self.program, self.dbg_pc, fault) - } - // Read the running access count and advance it by ×g (the free increment). - // ×g is ×x, i.e. `mul_by_g`, a shift+fold rather than a PMULL; this runs on every - // memory access (several million per run), so the cheap form matters. - #[inline(always)] - fn bump_access_count(&mut self, cell: u32) -> F64 { - self.ensure(cell as usize); - let cell_idx = cell as usize; - let count = self.count[cell_idx]; - self.count[cell_idx] = mul_by_g(count); - count - } - } - // Bounded discrete log for `hint_decompose_bits_exponent`: find n < 2^nbits - // with g^n = x, by baby-step giant-step (baby table g^j for j < 2^17, - // built once per run; giant step ×g^(-2^17)). Prover-side only: the - // guest re-verifies the hinted bits in-circuit. - fn bounded_dlog(cache: &mut Option<(GPow, F64)>, x: F64, nbits: u32) -> Option { - const LOG_BABY: u32 = 17; - let (baby, giant) = cache.get_or_insert_with(|| { - let baby = GPow::new((1usize << LOG_BABY) - 1); - // g^(2^17), one past the table; its inverse is the giant step. - let giant = mul_by_g(baby.pow((1usize << LOG_BABY) - 1)).inv(); - (baby, giant) + Box::new(HashRow { + block: h.block, + out: h.out, + acc: all, + }) }); - let mut y = x; - let max_giant = if nbits > LOG_BABY { - 1u64 << (nbits - LOG_BABY) - } else { - 1 - }; - for a in 0..max_giant { - if let Some(j) = baby.log(y) { - return Some((a as u128) << LOG_BABY | j as u128); - } - y *= *giant; - } - None - } - - // A runtime count (a loop's frame span, a `HeapBuf` size) as the exponent of - // a g-power: the address table when it covers it, else a bounded discrete log. - fn count(g: &mut GPow, dlog: &mut Option<(GPow, F64)>, x: F64) -> Option { - g.log(x) - .or_else(|| bounded_dlog(dlog, x, MAX_CELLS.ilog2()).map(|n| n as u32)) - } - - // The cell a heap run starts at: read the pointer back out of memory and - // invert it. Shared by every hint that writes through one. - fn heap_base(m: &Mem<'_>, g: &mut GPow, cell: u32, what: &'static str) -> Result { - let p = in_k(what, m.get(cell))?; - g.log(p).ok_or(Fault::NotAGPower { what, value: p }) + rows[table].push(Row { + index: step.index as u32, + ts, + v1: step.v1, + v2: step.v2, + out: step.out, + taken: step.taken, + vd_old: step.vd_old, + ram: step.ram.unwrap_or_default(), + acc, + hash, + bytecode_read: fetch(step.index), + }); + tick += spec.stride(); + ts = advance(ts, spec.stride()); } - - // Where a computed-advice bit buffer starts: a frame run needs no lookup - // at all, which is the point of having one. - fn bits_base(m: &Mem<'_>, g: &mut GPow, fp: u32, dest: BitsDest, what: &'static str) -> Result { - match dest { - BitsDest::Stack(base) => Ok(fp + base), - BitsDest::Heap(ptr) => heap_base(m, g, fp + ptr, what), - } + let syscall = m.regs[rv::SYSCALL_REG as usize]; + if syscall != rv::SYS_EXIT { + return Err(Trap::NotAnExit { syscall }); } - - // The program's own chain runs to the halt sentinel; the fill blocks then run, one - // cycle at a time. A cycle is entered at its block's first instruction, in a frame - // of its own, and traversed for the rows `filler::plan` asks of it, always a whole - // number of traversals, so the state tuples it pushes are exactly the ones it pulls - // (doc §Filling the tables). `left` is the rows still to run in the current cycle, - // and `None` while the chain runs. - let mut cycles: std::vec::IntoIter<(u32, u32, usize)> = Vec::new().into_iter(); - let mut left: Option = None; - loop { - // The chain reached the sentinel, or a cycle has run its rows. - let switch = match left { - None => pc == ending_pc, - Some(n) => n == 0, - }; - if switch { - if left.is_none() { - if fp != 0 { - return Err(ExecError::new(self, pc, Fault::HaltOutsideMain { fp })); - } - let counts = [xor.len(), mul.len(), set.len(), deref.len(), jump.len(), blake2s.len()]; - base_counts = Some(counts); - unconstrained_reads = std::mem::take(&mut m.unwritten_reads); - use super::filler::{self, frame as fr}; - let mut runs: Vec<(u32, u32, usize)> = Vec::new(); - // A hand-assembled program carries no blocks and has to land on - // powers of two by itself, which `Layout` checks. - if !self.filler.is_empty() { - // A whole frame per cycle, past every cell the program touched, so - // the memory the fill leaves is known before it runs and the plan - // reaching `min_log_committed` is solved for here, once. - let start = m.cells.len(); - let log_bytecode = crate::log2_strict_usize(self.prog.len()); - let plan = filler::plan(counts, self.min_log_committed, |plan| { - let cells = start + filler::frames(plan) * fr::CELLS as usize; - super::layout::committed_log( - crate::log2_strict_usize(cells.next_power_of_two().max(1 << MIN_LOG_MEM)), - log_bytecode, - filler::filled(counts, plan).map(crate::log2_strict_usize), - ) - }); - let mut frame = start as u32; - for (block_pc, size, n) in filler::cycles(&self.filler, &plan) { - g.grow_to(frame as usize); - g.note(frame as usize); - m.ensure((frame + fr::CELLS - 1) as usize); - // What the closing jump reads: back to the block's own first - // instruction, in this same frame. Then the pointer the `DEREF` - // dummy follows, memory cell `0`. - m.put(frame + fr::DEST, F192::from(g.pow(block_pc as usize)))?; - m.put(frame + fr::NEXT_FP, F192::from(g.pow(frame as usize)))?; - m.put(frame + fr::PTR, F192::ONE)?; - runs.push((block_pc, frame, n * (size as usize + 1))); - frame += fr::CELLS; - } - } - cycles = runs.into_iter(); - } - match cycles.next() { - Some((p, f, rows)) => { - pc = p; - fp = f; - left = Some(rows); - } - None => break, - } - } - if let Some(n) = &mut left { - *n -= 1; - } - if steps >= MAX_STEPS { - return Err(ExecError::new(self, pc, Fault::StepLimit)); - } - m.dbg_pc = pc; - let fail = move |fault| ExecError::new(self, pc, fault); - if let Some(p) = prof.as_mut() { - p[pc as usize] += 1; - } - - // Apply the hints scheduled before this instruction. - if hint_at[pc as usize] != 0 { - let hs = hint_lists[hint_at[pc as usize] as usize - 1]; - for h in hs { - m.dbg_hint = Some(match h { - RHint::FrameAddress { .. } => "FrameAddress", - RHint::AllocFrames { .. } => "AllocFrames", - RHint::Alloc { .. } => "Alloc", - RHint::AllocDyn { .. } => "AllocDyn", - RHint::WitnessStack { .. } => "WitnessStack", - RHint::WitnessHeap { .. } => "WitnessHeap", - RHint::Log2Ceil { .. } => "Log2Ceil", - RHint::BitDecompose { .. } => "BitDecompose", - RHint::BitDecomposeExp { .. } => "BitDecomposeExp", - RHint::FieldLimbs { .. } => "FieldLimbs", - RHint::Inverse { .. } => "Inverse", - RHint::Print { .. } => "Print", + let output = rv::OUTPUT_REGS.map(|r| m.regs[r as usize]); + let base_counts: [usize; crate::tables::N_TABLES] = std::array::from_fn(|t| rows[t].len()); + + // The padding rows, written out rather than executed: they sit at clock zero + // and touch nothing, every read holding zero and every write rewriting what it + // writes (`filler`). Their circuit instance is an honest one, on those zeros. + for (first, size, traversals) in super::filler::cycles(&self.filler, base_counts) { + for _ in 0..traversals { + for index in first..=first + size { + let e = &p.entries[index]; + let table = crate::tables::table_of(e.class).expect("a fill block's class has a table"); + let (out, taken, access) = compute(e, 0, 0, 0); + let n_accesses = CLASSES[table].n_accesses(); + // The row's accesses are all padding ones, in a hash row's own array. + let mut acc = [padding_access_unread(); 4]; + let mut hash = (e.class == Class::Hash).then(|| { + // The compression of a zero block, whose result the row rewrites. + let mut h = rv::machine::compute_hash([0; hash::WORDS], 0, e.flags); + h.block[hash::OUT as usize / 8..][..4].copy_from_slice(&h.out); + Box::new(HashRow { + block: h.block, + out: h.out, + acc: [padding_access_unread(); 2 + hash::WORDS], + }) }); - match h { - RHint::FrameAddress { offset } => { - g.note((fp + offset) as usize); - } - // A fresh region: write its base `g^{next_free}` into the - // pointer cell (once) and reserve `size` cells. `AllocDyn` - // reads the size from a cell at runtime. - RHint::Alloc { .. } | RHint::AllocDyn { .. } | RHint::AllocFrames { .. } => { - let (ptr, size) = match *h { - RHint::Alloc { ptr, size } => (ptr, size), - RHint::AllocFrames { - ptr, - size, - end, - start_inverse, - } => { - let span = in_k("loop bound", m.get(fp + end)).map_err(fail)? * start_inverse; - if span.is_zero() { - return Err(fail(Fault::NotAGPower { - what: "loop bound", - value: span, - })); - } - let n = count(&mut g, &mut dlog_cache, span) - .ok_or_else(|| fail(Fault::OutOfMemory))? - + 1; - (ptr, size.checked_mul(n).ok_or_else(|| fail(Fault::OutOfMemory))?) - } - // A runtime size is carried in the exponent: - // the cell holds g^k, allocate k cells. - RHint::AllocDyn { ptr, size } => { - let sz = in_k("HeapBuf size", m.get(fp + size)).map_err(fail)?; - let cells = count(&mut g, &mut dlog_cache, sz).ok_or_else(|| { - fail(Fault::NotAGPower { - what: "HeapBuf size", - value: sz, - }) - })?; - (ptr, cells) - } - _ => unreachable!(), - }; - let cell = fp + ptr; - if !m.written(cell) { - let base = next_free; - next_free = base - .checked_add(size) - .filter(|&end| end < MAX_CELLS) - .ok_or_else(|| fail(Fault::OutOfMemory))?; - g.grow_to((base + size) as usize); - // The base is about to become a pointer in memory. - g.note(base as usize); - m.ensure(next_free as usize); - m.put(cell, F192::from(g.pow(base as usize)))?; - } - } - RHint::Print { label, cell } => { - let c = fp + cell; - if m.written(c) { - let v = m.cells[c as usize]; - // Small integers and small g-powers overlap (8 = x^3 - // = g^3): show every reading that applies. Only a - // K-valued word (extension limbs 0) can be a g-power. - let k = as_addr(v).and_then(|lo| g.log(lo)); - let small = v.c2 == 0 && v.c1 == 0 && v.c0 < 1 << 32; - match (k, small) { - (Some(k), true) => eprintln!( - "[print] {label} = {} (g^{})", - pretty_integer(v.c0), - pretty_integer(k) - ), - (Some(k), false) => { - eprintln!("[print] {label} = g^{}", pretty_integer(k)) - } - (None, true) => { - eprintln!("[print] {label} = {}", pretty_integer(v.c0)) - } - (None, false) => { - eprintln!("[print] {label} = {:#x}:{:#x}:{:#x}", v.c2, v.c1, v.c0) - } - } - } else { - eprintln!("[print] {label} = "); - } - } - RHint::WitnessStack { name, base, len } => { - let values = - pop_witness(&self.witness, &mut witness_positions, name, *len).map_err(fail)?; - for (k, &value) in values.iter().enumerate() { - m.put(fp + base + k as u32, value)?; - } - } - RHint::WitnessHeap { name, ptr, lo, len } => { - let b = heap_base(&m, &mut g, fp + ptr, "hint_witness heap pointer").map_err(fail)?; - let values = - pop_witness(&self.witness, &mut witness_positions, name, *len).map_err(fail)?; - for (k, &value) in values.iter().enumerate() { - m.put(b + lo + k as u32, value)?; - } - } - RHint::Log2Ceil { - bits, - dst, - nbits, - floor, - } => { - let b = bits_base(&m, &mut g, fp, *bits, "log2_ceil pointer").map_err(fail)?; - let mut word: u128 = 0; - for j in 0..*nbits { - if !m.get(b + j).is_zero() { - word |= 1u128 << j; - } - } - let cl = if word <= 1 { - 0 - } else { - u128::BITS - (word - 1).leading_zeros() - }; - let mu = cl.max(*floor); - m.put(fp + dst, F192::from(primitives::field::g_pow(mu as usize)))?; - } - RHint::BitDecompose { value, bits, nbits } => { - assert!(*nbits <= 192, "a machine word has 192 bits"); - let v = m.get(fp + value); - let limbs = [v.c0, v.c1, v.c2]; - let bb = bits_base(&m, &mut g, fp, *bits, "decompose pointer").map_err(fail)?; - for j in 0..*nbits { - let bit = (limbs[j as usize / 64] >> (j % 64)) & 1; - m.put(bb + j, F192::new(bit, 0, 0))?; - } - } - RHint::BitDecomposeExp { value, bits, nbits } => { - const WHAT: &str = "hint_decompose_bits_exponent value"; - let x = in_k(WHAT, m.get(fp + value)).map_err(fail)?; - let n = bounded_dlog(&mut dlog_cache, x, *nbits) - .ok_or_else(|| fail(Fault::NotAGPower { what: WHAT, value: x }))?; - let bb = bits_base(&m, &mut g, fp, *bits, "hint_decompose_bits_exponent pointer") - .map_err(fail)?; - for j in 0..*nbits { - let bit = ((n >> j) & 1) as u64; - m.put(bb + j, F192::new(bit, 0, 0))?; - } - } - RHint::FieldLimbs { value, base, len } => { - assert!((1..=3).contains(len), "an F192 value has three K limbs"); - let v = m.get(fp + value); - let limbs = [v.c0, v.c1, v.c2]; - for j in 0..*len { - m.put(fp + base + j, F192::new(limbs[j as usize], 0, 0))?; - } - } - RHint::Inverse { value, dst } => { - let v = m.get(fp + value); - m.put(fp + dst, if v.is_zero() { F192::ZERO } else { v.inv() })?; - } + let all = hash.as_mut().map_or(&mut acc[..], |h| &mut h.acc[..]); + for a in &mut all[..n_accesses] { + *a = padding_access(&mut ranges); } - m.dbg_hint = None; - } - } - // Cover the g-powers this step may index (g²·pc return target, g^fp). - // Guarded so the steady state is a length compare, not a call. - let need = (pc as usize + 2).max(fp as usize); - if g.covered() <= need { - g.grow_to(need); - } - - let bytecode_read = { - let v = bytecode_count[pc as usize]; - bytecode_count[pc as usize] = mul_by_g(v); - v - }; - - // Loaded once: the shared Xor/Mul arm needs the discriminant again, - // and `Op` is wide enough that re-reading it costs a second load. - let op = self.prog[pc as usize]; - match op { - Op::Xor { a, b, c } | Op::Mul { a, b, c } => { - let is_xor = matches!(op, Op::Xor { .. }); - let (aa, ab, ac) = (fp + a, fp + b, fp + c); - // The row is the equality `m[c] = m[a] op m[b]` over write-once - // memory. Normally the operands are known and the result is - // computed forward; for a `MUL` whose result is already written - // and exactly one of whose operands is not, the runner - // back-solves that operand, which is what produces the - // range-check complement `y = g^{k-1}·x^{-1}` from - // `MUL x·y = g^{k-1}`, and the quotient of `a / b`, with no - // dedicated hint. - // - // `XOR` deliberately does NOT deduce. Nothing asks it to (both - // users are `MUL`), and an `XOR` into an already-written cell is - // how `assert a == b` is spelled, so deducing there would define - // the operand the assert exists to check instead of failing on it. - if !is_xor && m.written(ac) { - let (ha, hb) = (m.written(aa), m.written(ab)); - if ha ^ hb { - let vk = m.get(if ha { aa } else { ab }); - if vk.is_zero() { - return Err(fail(Fault::BackSolveThroughZero)); - } - m.put(if ha { ab } else { aa }, m.get(ac) * vk.inv())?; - } - } - let va = m.read(aa); - let vb = m.read(ab); - let vc = if is_xor { va + vb } else { va * vb }; - m.put(ac, vc)?; - let ra = m.bump_access_count(aa); - let rb = m.bump_access_count(ab); - let rc = m.bump_access_count(ac); - let row = Xrow { - pc, - fp, - ra, - rb, - rc, - bytecode_read, - }; - if is_xor { - xor.push(row); - } else { - mul.push(row); - } - pc += 1; - } - Op::Set { o, k } => { - let a = fp + o; - m.put(a, k)?; - let r = m.bump_access_count(a); - set.push(Srow { - pc, - fp, - r, - bytecode_read, + rows[table].push(Row { + index: index as u32, + ts: F64::ZERO, + v1: 0, + v2: 0, + out, + taken, + vd_old: if e.link { p.pc_of(index) + 4 } else { out }, + ram: access, + acc, + hash, + bytecode_read: fetch(index), }); - pc += 1; - } - Op::Deref { o1, o2, o3, mode } => { - let a1 = fp + o1; - let p = m.read(a1); - let p_addr = as_addr(p).ok_or_else(|| fail(Fault::WildPointer { value: p }))?; - let base = g.log(p_addr).ok_or_else(|| fail(Fault::WildPointer { value: p }))?; - let a2 = base + o2; - let a3 = fp + o3; - match mode { - // Equality m[a2] == m[a3]: carry a written side to the other - // (`put` checks it when both are), or link two unwritten sides - // until either is written. A range-check touch may leave both - // unwritten for good: only the validity of `a2` matters there. - DerefMode::Cell => { - if m.written(a2) { - m.put(a3, m.cells[a2 as usize])?; - } else if m.written(a3) { - m.put(a2, m.cells[a3 as usize])?; - } else { - m.link(a2, a3); - } - } - DerefMode::Pc => { - // The return target and the frame base are stored as - // addresses, and JUMP reads them back. - g.note(pc as usize + 2); - let v = F192::from(g.pow(pc as usize + 2)); - m.put(a2, v)?; - } - DerefMode::Fp => { - g.note(fp as usize); - let v = F192::from(g.pow(fp as usize)); - m.put(a2, v)?; - } - } - let r1 = m.bump_access_count(a1); - let r2 = m.bump_access_count(a2); - let r3 = m.bump_access_count(a3); - deref.push(Drow { - pc, - fp, - r1, - r2, - r3, - bytecode_read, - }); - pc += 1; - } - Op::Jump { oc, od, of } => { - let (ac, ad, af) = (fp + oc, fp + od, fp + of); - // All three cells are K-valued on EVERY row, taken or not: the - // table commits one lane each and their memory flushes carry - // literal zeros above it (§sec:tab-jump), which the bus would - // otherwise not balance. A guest branches on g-powers; the one - // idiom that once branched on a word, `assert a != b`, takes an - // inverse hint instead (§sec:prog-div-ne). - let c = in_k("JUMP condition", m.read(ac)).map_err(fail)?; - let d = in_k("JUMP target", m.read(ad)).map_err(fail)?; - let f = in_k("JUMP fp", m.read(af)).map_err(fail)?; - // The is-nonzero witness `w = c⁻¹` is never used for control - // flow, only recorded as a witness column, so it is not - // computed here at all: `JumpTable::fill` batch-inverts every - // row's condition at once (§the trace rows in `cpu::trace`). - let rc = m.bump_access_count(ac); - let rd = m.bump_access_count(ad); - let rf = m.bump_access_count(af); - let taken = !c.is_zero(); - jump.push(Jrow { - pc, - fp, - rc, - rd, - rf, - bytecode_read, - }); - if taken { - pc = g.log(d).filter(|&t| (t as usize) < self.prog.len()).ok_or_else(|| { - fail(Fault::NotAGPower { - what: "JUMP target", - value: d, - }) - })?; - fp = g.log(f).ok_or_else(|| { - fail(Fault::NotAGPower { - what: "JUMP fp", - value: f, - }) - })?; - } else { - pc += 1; - } } - Op::Blake2s { ins, cv, out, md } => { - // Four independently-addressed 128-bit message chunks, each a - // single cell; the chaining value and the output each span two - // consecutive cells; the metadata is one more cell. - let (aa0, aa1, ab0, ab1) = (fp + ins[0], fp + ins[1], fp + ins[2], fp + ins[3]); - let acv = fp + cv; - let ac = fp + out; - let amd = fp + md; - let words = [aa0, aa1, ab0, ab1, acv, acv + 1, amd].map(|a| m.read(a)); - // Naming the operand and the line matters most for the metadata, - // the one a guest builds with field arithmetic rather than reads. - if let Some((i, &value)) = words.iter().enumerate().find(|(_, w)| w.c2 != 0) { - const CELLS: [&str; 7] = ["m0", "m1", "m2", "m3", "cv0", "cv1", "md"]; - return Err(fail(Fault::NotCanonical { - operand: CELLS[i], - value, - })); - } - let va = [F64(words[0].c0), F64(words[0].c1), F64(words[1].c0), F64(words[1].c1)]; - let vb = [F64(words[2].c0), F64(words[2].c1), F64(words[3].c0), F64(words[3].c1)]; - let vcv = [F64(words[4].c0), F64(words[4].c1), F64(words[5].c0), F64(words[5].c1)]; - let metadata = words[6]; - // Compress the 64 message bytes to the 32-byte result, then - // write it to c's two cells. No table constraint covers the - // digest (the relation is proven by flock, §hash_flock); the - // interpreter still computes the definite digest so the output - // cells are consistent for any later read. - let vc = blake2s_compress(va, vb, vcv, metadata); - let outputs = [F192::new(vc[0].0, vc[1].0, 0), F192::new(vc[2].0, vc[3].0, 0)]; - m.put(ac, outputs[0])?; - m.put(ac + 1, outputs[1])?; - let ra = [m.bump_access_count(aa0), m.bump_access_count(aa1)]; - let rb = [m.bump_access_count(ab0), m.bump_access_count(ab1)]; - let rcv = [m.bump_access_count(acv), m.bump_access_count(acv + 1)]; - let rc = [m.bump_access_count(ac), m.bump_access_count(ac + 1)]; - // Last, matching the flush order, so an md cell aliasing another - // operand still pairs each read with its own count. - let rmd = m.bump_access_count(amd); - blake2s.push(Brow { - pc, - fp, - ra, - rb, - rcv, - rc, - rmd, - bytecode_read, - }); - pc += 1; - } - } - steps += 1; - } - - if let Some(p) = &prof { - let mut rows: Vec<(String, u64)> = self - .fn_ranges - .iter() - .map(|(name, entry, len)| { - let total: u64 = p[*entry as usize..(*entry + *len) as usize].iter().sum(); - (name.clone(), total) - }) - .collect(); - rows.sort_by_key(|(_, c)| std::cmp::Reverse(*c)); - // `DBG_PROF_DUMP=path`: also write the raw per-pc counts plus the - // function table, so an offline pass can attribute the straight-line - // cycles of one big function to its source regions (the call sites of - // the lowered `for` helpers are the landmarks). - if let Ok(path) = std::env::var("DBG_PROF_DUMP") { - let mut out = format!("# steps {steps}\n"); - for (name, entry, len) in &self.fn_ranges { - out += &format!("F {name} {entry} {len}\n"); - } - for (pc, c) in p.iter().enumerate() { - if *c > 0 { - out += &format!("{pc} {c}\n"); - } - } - let path = if std::path::Path::new(&path).exists() { - format!("{path}.{}", std::process::id()) - } else { - path - }; - std::fs::write(&path, out).expect("write DBG_PROF_DUMP"); - eprintln!("== DBG_PROF: per-pc counts written to {path}"); - } - eprintln!("== DBG_PROF: cycles by function ({} total) ==", pretty_integer(steps)); - for (name, c) in rows.iter().filter(|(_, c)| *c > 0) { - eprintln!( - " {:>13} {:>7}% {name}", - pretty_integer(c), - pretty_f64(100.0 * *c as f64 / steps as f64) - ); } } - // Pad memory to a power of two (the boundary tables read a dense image), - // at least 2^MIN_LOG_MEM cells (doc §Memory). - let mem_used = m.cells.len(); - let cells = m.cells.len().next_power_of_two().max(1 << MIN_LOG_MEM); - assert!(cells <= 1 << MAX_LOG_MEM, "data memory exceeds 2^{MAX_LOG_MEM} cells"); - m.cells.resize(cells, F192::ZERO); - m.count.resize(cells, F64::ONE); + let cycles = rows.iter().map(Vec::len).sum(); + let (ram_ts, adv_ts) = ram.last_ts.split_at(1 << p.log_ram); let trace = Trace { - xor, - mul, - set, - deref, - jump, - blake2s, - mem_count: m.count, + rows, + reg_fin: m.regs.iter().map(|&r| F64(r)).collect(), + reg_ts: regs.last_ts, + ram_fin: m.ram().iter().map(|&w| F64(w)).collect(), + ram_ts: ram_ts.to_vec(), + adv_init, + adv_fin: m.advice().iter().map(|&w| F64(w)).collect(), + adv_ts: adv_ts.to_vec(), bytecode_count, + range_lo_count: ranges.lo, + range_hi_count: ranges.hi, + ts_final: ts, }; Ok(Execution { - mem: m.cells, - cycles: steps, - mem_used, - // Taken when the chain halted, which every run does before it can leave the - // loop at all. - base_counts: base_counts.expect("the run halted, so its own counts were taken"), - unconstrained_reads, + output, + cycles, + base_counts, trace, }) } diff --git a/crates/lean_vm/src/cpu/filler.rs b/crates/lean_vm/src/cpu/filler.rs index cb35f664a..fcfb44ee3 100644 --- a/crates/lean_vm/src/cpu/filler.rs +++ b/crates/lean_vm/src/cpu/filler.rs @@ -1,105 +1,115 @@ -//! Filling every table to a power of two, so that no table has padding rows. +//! Filling every table to a power of two with padding rows at clock zero. //! //! A table is proven over a power-of-two number of rows, so a run whose counts are not -//! powers of two has to make up the difference, and it makes it up by executing more -//! instructions. The bytecode carries, per table and per size in [`SIZES`], a *block*: -//! that many dummy instructions of the table's opcode, then a `JUMP` back to the block's -//! own first instruction, in the same frame (`lean_compiler::filler`). +//! powers of two has to make up the difference. The text carries, per table and per +//! size in [`SIZES`], a *block*: that many no-ops of the table's class, then a `JAL` +//! back to the block's own first instruction ([`append_blocks`]). //! -//! So a block is a **cycle**, and no program code jumps into it. That is what makes this -//! work: the state channel's tuples are pushed and pulled around the cycle and cancel -//! among themselves, for any number of traversals, so the fill is a closed loop running -//! beside the program's chain rather than part of it (doc §Filling the tables). The -//! prover picks how many times each block is traversed, and nothing has to be counted, -//! tested, or entered. +//! So a block is a **cycle**, and no program code jumps into it. Its rows carry the +//! clock `ts = 0`, which is what makes it balance: `0·g^k = 0`, so the state tuples +//! pushed and pulled around the cycle cancel among themselves for any number of +//! traversals, where a real clock would have moved on. And zero is no power of `g`, so +//! nothing such a row puts on the bus can meet a tuple of the run itself: its register +//! accesses are forced to the previous timestamp `0` as well, and each cancels against +//! itself (doc §Filling the tables). The rows therefore touch nothing, and the prover +//! writes them out rather than executing anything. //! //! A traversal of the size-`s` block costs exactly `s + 1` rows: `s` of its own table and -//! one `JUMP`. Nothing else, and nothing on any other table. That is what makes the solve -//! here exact, with no calibrated cost model and no residual to correct, and it is why -//! one interpretation of the program suffices: the interpreter solves once the program -//! has halted, by which point its row counts are final. +//! one of `ALU`'s, the jump. Nothing else, and nothing on any other table. That is what +//! makes the solve here exact, with no calibrated cost model and no residual to correct. //! //! The sizes are powers of two so any fill is reachable exactly, while the bulk rides the -//! largest block at one `JUMP` per 128 rows. A table already sitting on a power of two is -//! never entered at all. +//! largest block at one jump per 128 rows. A table already sitting on a power of two is +//! never entered at all. `ALU`, where the jumps land, also has the block of size zero, a +//! jump to itself: its traversals cost `s + 1` rows each, and that one makes a gap of a +//! single row reachable. -use crate::tables::N_TABLES; - -/// No table asked to grow past its natural power of two. -pub const NO_FLOORS: [usize; N_TABLES] = [0; N_TABLES]; - -/// The table a committed-size floor grows ([`crate::cpu::Program::min_log_committed`]). -/// `SET` writes one cell from an immediate: no operand reads, no second memory -/// touch, and no precompile behind it, so its rows are the cheapest to prove. -pub const PAD_TABLE: usize = 2; +use crate::rv::Class; +use crate::rv::asm; +use crate::tables::{CLASSES, N_TABLES}; /// Block sizes, largest first: a fill of `f` rows takes `f / 128` traversals of the -/// largest block and then one per set bit of the remainder. -pub const SIZES: [usize; 8] = [128, 64, 32, 16, 8, 4, 2, 1]; +/// largest block and then one per set bit of the remainder. The last, a lone jump, +/// is `ALU`'s only. +pub const SIZES: [usize; 9] = [128, 64, 32, 16, 8, 4, 2, 1, 0]; -/// Least rows a table can be proven over. Only `BLAKE2s` has one above `1`: flock sizes -/// its argument to at least eight instances, so filling that table below the floor -/// would leave it padded up to it, which is the padding this exists to avoid. -pub const MIN_ROWS: [usize; N_TABLES] = [1, 1, 1, 1, 1, 8]; +/// Least rows a table can be proven over: flock sizes a batch to at least eight +/// instances and its zerocheck to a cube of at least `2^13` bits +/// ([`crate::class_flock::n_blocks_log`]). Filling a table below its floor would leave +/// it padded up to it, which is the padding this exists to avoid. +pub fn min_rows(t: usize) -> usize { + 1 << crate::class_flock::n_blocks_log(CLASSES[t], 1) +} -/// The `JUMP` table's index in [`crate::cpu::Stats::TABLES`]. Every traversal of every -/// block lands its closing jump here, so this table is solved last, absorbing the cost -/// of the whole fill. -pub const JUMP: usize = 4; +/// `ALU`'s index in [`CLASSES`]. Every traversal of every block lands its closing jump +/// here, so this table is solved last, absorbing the cost of the whole fill. +pub const JUMP: usize = 0; +const _: () = assert!(matches!(CLASSES[JUMP].class, Class::Alu)); -/// One block in the bytecode: `size` dummy rows of `table`'s opcode at `pc`, then the -/// jump back to `pc`. +/// One block in the text: `size` no-ops of `table`'s class from entry `index`, then the +/// jump back to `index`. #[derive(Clone, Debug, PartialEq, Eq)] pub struct Block { - pub pc: u32, - pub size: u32, - pub table: u8, + pub index: usize, + pub size: usize, + pub table: usize, } -/// A block's frame, as offsets from the frame pointer the interpreter gives it. -/// -/// The three the interpreter writes are the closing jump's destination (`g^{pc}` of the -/// block's own first instruction, which is a g-power and so doubles as the jump's nonzero -/// condition), the frame to go there in (this frame), and the pointer the `DEREF` dummy -/// follows. No instruction writes them, which is why a traversal costs its own rows and -/// nothing more. -/// -/// The rest is what the dummies use: a cell that is never written, so the `JUMP` table's -/// dummy reads a zero condition and falls through instead of leaving the block; the -/// scratch cell a dummy writes, which doubles as the `BLAKE2s` dummy's chaining value and -/// so spans `SCRATCH..SCRATCH+2`; and the digest, placed clear of it so that a digest -/// never becomes the next traversal's chaining value. `DIGEST+2..DIGEST+6` are the -/// message cells, never written, so every traversal compresses the same input. -pub mod frame { - /// Where the closing jump goes, and in which frame. - pub const DEST: u32 = 0; - pub const NEXT_FP: u32 = 1; - /// The pointer a `DEREF` dummy follows: `g^0`, memory cell `0`. - pub const PTR: u32 = 2; - /// Never written, so it reads as zero. - pub const ZERO: u32 = 3; - /// What a dummy writes. - pub const SCRATCH: u32 = 4; - /// The `BLAKE2s` dummy's output pair. - pub const DIGEST: u32 = 6; - /// Cells a block's frame occupies. - pub const CELLS: u32 = 12; +/// A no-op of `class`: every register is `x0`, which a padding row never touches for real. +fn nop(class: Class) -> u32 { + match class { + Class::Alu => asm::i_type(0x13, 0, 0, 0, 0), + // A load and a store of the byte at address zero, which clock zero never checks. + Class::Load => asm::i_type(0x03, 0, 0, 0, 0), + Class::Store => asm::s_type(0x23, 0, 0, 0, 0), + Class::Shift => asm::i_type(0x13, 1, 0, 0, 0), + Class::Mul => asm::r_type(0x33, 0, 1, 0, 0, 0), + Class::Mulh => asm::r_type(0x33, 3, 1, 0, 0, 0), + Class::Div => asm::r_type(0x33, 5, 1, 0, 0, 0), + // A compression of the block at address zero, which clock zero never checks. + Class::Hash => asm::r_type(crate::rv::hash::OPCODE, 0, 0, 0, 0, 0), + Class::Illegal => unreachable!("no fill block of an illegal entry"), + } +} + +/// Whether a text of `words` instructions still leaves room, inside the text region, +/// for the illegal word and the padding blocks [`crate::cpu::Program::new`] appends, +/// and for the pad to a power of two ([`crate::rv::Program::new`]). +pub fn text_fits(words: usize) -> bool { + let mut blocks = Vec::new(); + append_blocks(&mut blocks); + words + .checked_add(blocks.len() + 1 + 2) + .is_some_and(|total| total.next_power_of_two() <= 1 << crate::rv::MAX_LOG_TEXT) +} + +/// Append every table's blocks to `text`, returning where each one landed. +pub fn append_blocks(text: &mut Vec) -> Vec { + let mut blocks = Vec::new(); + for (t, spec) in CLASSES.iter().enumerate() { + for size in SIZES.into_iter().filter(|&size| size > 0 || t == JUMP) { + blocks.push(Block { + index: text.len(), + size, + table: t, + }); + text.extend(std::iter::repeat_n(nop(spec.class), size)); + // The closing jump, which a padding row takes back to the block's top. + text.push(asm::j_type(0, -4 * size as i32)); + } + } + blocks } /// Traversals per block: `plan[t][k]` is how many times the size-`SIZES[k]` block of /// table `t` is traversed. pub type Plan = [[usize; SIZES.len()]; N_TABLES]; -/// Traversals in total, which is the number of `JUMP` rows the fill costs. +/// Traversals in total, which is the number of jump rows the fill costs. pub fn traversals(plan: &Plan) -> usize { plan.iter().flatten().sum() } -/// Cycles in a plan, one frame each: every block it traverses at all. -pub fn frames(plan: &Plan) -> usize { - plan.iter().flatten().filter(|&&n| n > 0).count() -} - /// The fill a plan delivers to each table, not counting the closing jumps. fn delivered(plan: &Plan) -> [usize; N_TABLES] { let mut out = [0usize; N_TABLES]; @@ -116,11 +126,11 @@ fn delivered(plan: &Plan) -> [usize; N_TABLES] { fn decompose(fill: usize) -> [usize; SIZES.len()] { let mut out = [0usize; SIZES.len()]; let mut left = fill; - for (k, &s) in SIZES.iter().enumerate() { + for (k, &s) in SIZES.iter().enumerate().filter(|&(_, &s)| s > 0) { out[k] = left / s; left -= out[k] * s; } - debug_assert_eq!(left, 0, "the sizes end at 1, so nothing can be left over"); + debug_assert_eq!(left, 0, "the positive sizes end at 1, so nothing can be left over"); out } @@ -130,96 +140,61 @@ fn ceil_pow2(n: usize) -> usize { } /// Traversals whose rows land on `JUMP` itself, delivering exactly `gap` rows. A -/// traversal of the size-`s` block gives that table `s + 1` rows here, its dummies plus -/// its own closing jump, so the sizes to decompose over are `s + 1`, which are not powers -/// of two. `2` and `3` are among them, so every gap but `1` is reachable; a gap of `1` -/// returns `None` and the caller takes a larger target. -fn decompose_jump(gap: usize) -> Option<[usize; SIZES.len()]> { - if gap == 1 { - return None; - } +/// traversal of the size-`s` block gives that table `s + 1` rows here, its no-ops plus +/// its own closing jump, so the sizes to decompose over are `s + 1`, down to the lone +/// jump's `1`. +fn decompose_jump(gap: usize) -> [usize; SIZES.len()] { let mut out = [0usize; SIZES.len()]; let mut left = gap; for (k, &s) in SIZES.iter().enumerate() { - // Never leave exactly one row behind, which nothing can deliver. - while left > s && left - (s + 1) != 1 { - out[k] += 1; - left -= s + 1; - } + out[k] = left / (s + 1); + left -= out[k] * (s + 1); } - (left == 0).then_some(out) + out } -/// A plan taking every table from `base` to an exact power of two, at or above -/// `floors[t]`, or `None` if some table cannot be filled at all. +/// A plan taking every table from `base` to an exact power of two. /// /// Every table but `JUMP` is independent: its fill is the distance to its next power of /// two, decomposed into traversals. `JUMP` is not, because every traversal of the whole /// fill lands a row there, its own traversals included. Counting those first makes it a /// single decomposition rather than a fixpoint. -/// -/// A floor above a table's natural target is how a run too small to be provable at the -/// size its consumer needs buys the difference: see [`crate::cpu::Program::min_log_committed`]. -pub fn solve(base: [usize; N_TABLES], floors: [usize; N_TABLES]) -> Option { +pub fn solve(base: [usize; N_TABLES]) -> Plan { let mut plan: Plan = [[0; SIZES.len()]; N_TABLES]; for t in 0..N_TABLES { if t != JUMP { - plan[t] = decompose(ceil_pow2(base[t].max(MIN_ROWS[t]).max(floors[t])) - base[t]); + plan[t] = decompose(ceil_pow2(base[t].max(min_rows(t))) - base[t]); } } // What `JUMP` already owes: its own rows, plus one per traversal so far. let owed = base[JUMP] + traversals(&plan); - let mut target = ceil_pow2(owed.max(MIN_ROWS[JUMP]).max(floors[JUMP])); - loop { - if let Some(jump_steps) = decompose_jump(target - owed) { - plan[JUMP] = jump_steps; - debug_assert!(is_filled(filled(base, &plan))); - return Some(plan); - } - target = target.checked_mul(2)?; - } + plan[JUMP] = decompose_jump(ceil_pow2(owed.max(min_rows(JUMP))) - owed); + debug_assert!(is_filled(filled(base, &plan))); + plan } -/// The plan filling `base` whose stacked witness, as `committed_log` measures it, is -/// at least `2^min_log`: the natural one if it is, else the one growing [`PAD_TABLE`] -/// the least. The witness never shrinks as a table grows, so the first to reach wins. -pub fn plan(base: [usize; N_TABLES], min_log: usize, committed_log: impl Fn(&Plan) -> usize) -> Plan { - let mut floors = NO_FLOORS; - loop { - let plan = solve(base, floors).unwrap_or_else(|| panic!("no fill plan from {base:?}")); - if committed_log(&plan) >= min_log { - return plan; - } - floors[PAD_TABLE] = 2 * filled(base, &plan)[PAD_TABLE]; - assert!( - floors[PAD_TABLE] <= 1 << crate::cpu::MAX_LOG_ROWS, - "no fill reaches a 2^{min_log} witness" - ); - } -} - -/// The cycles a plan runs, in the order the interpreter should walk them: for each, the -/// block's first pc, its size, and how many times to traverse it. Panics if `blocks` is -/// missing one the plan calls for, which can only mean bytecode the compiler did not emit. -pub fn cycles(blocks: &[Block], plan: &Plan) -> Vec<(u32, u32, usize)> { +/// The cycles a run needs, in the order to walk them: for each, the block's first +/// entry, its size, and how many times to traverse it. +pub fn cycles(blocks: &[Block], base: [usize; N_TABLES]) -> Vec<(usize, usize, usize)> { + let plan = solve(base); let mut out = Vec::new(); for (t, row) in plan.iter().enumerate() { for (k, &n) in row.iter().enumerate() { if n == 0 { continue; } - let size = SIZES[k] as u32; + let size = SIZES[k]; let block = blocks .iter() - .find(|b| b.table as usize == t && b.size == size) + .find(|b| b.table == t && b.size == size) .unwrap_or_else(|| panic!("the program has no fill block for table {t}, size {size}")); - out.push((block.pc, size, n)); + out.push((block.index, size, n)); } } out } -/// The row counts a plan produces from `base`: its fill, plus one `JUMP` per traversal. +/// The row counts a plan produces from `base`: its fill, plus one jump per traversal. pub fn filled(base: [usize; N_TABLES], plan: &Plan) -> [usize; N_TABLES] { let mut out = base; for (t, add) in delivered(plan).into_iter().enumerate() { @@ -235,7 +210,7 @@ pub fn is_filled(counts: [usize; N_TABLES]) -> bool { counts .iter() .enumerate() - .all(|(u, &c)| c.is_power_of_two() && c >= MIN_ROWS[u]) + .all(|(u, &c)| c.is_power_of_two() && c >= min_rows(u)) } #[cfg(test)] @@ -243,50 +218,39 @@ mod tests { use super::*; /// Whatever the shape of the run, every table comes out an exact power of two at or - /// above its floor. + /// above its floor, and no further than the next one: a gap of a single row included, + /// which the closing jumps of the other tables' fills can leave `ALU` with. #[test] fn solve_reaches_power_of_two_floors() { - let cases: [[usize; N_TABLES]; 6] = [ - [0, 0, 0, 0, 0, 0], - [1, 1, 1, 1, 1, 1], - // Roughly the XMSS run's mix. - [125_000, 286_000, 341_000, 508_000, 114_000, 130_000], - // Tables already exactly on a power of two, the awkward case: the closing - // jumps of every other table's traversals still have to fit somewhere. - [1 << 17, 1 << 12, 1000, 1 << 19, 1 << 16, 8], - [1, 2, 3, 4, 5, 6], - [0, 0, 0, 0, 1 << 20, 0], - ]; + let mut cases = vec![[0; N_TABLES], [1; N_TABLES], [125_000; N_TABLES], [1 << 17; N_TABLES]]; + for alu in [(1 << 17) - 3, (1 << 17) - 2, (1 << 17) - 1] { + let mut base = [1 << 10; N_TABLES]; + base[JUMP] = alu; + cases.push(base); + } for base in cases { - let plan = solve(base, NO_FLOORS).unwrap_or_else(|| panic!("no plan for {base:?}")); + let plan = solve(base); let got = filled(base, &plan); assert!(is_filled(got), "{base:?} filled to {got:?}"); for t in 0..N_TABLES { - assert!(got[t] >= base[t], "rows cannot be removed"); + let owed = base[t] + + if t == JUMP { + traversals(&plan) - plan[JUMP].iter().sum::() + } else { + 0 + }; + assert_eq!(got[t], ceil_pow2(owed.max(min_rows(t))), "{base:?} filled to {got:?}"); } } } - /// A floor takes a table past its natural power of two, and every other table - /// still lands on one: what a run too small for its consumer buys with. - #[test] - fn floor_grows_target_table() { - let base = [1_000, 2_000, 3_000, 4_000, 500, 8]; - let mut floors = NO_FLOORS; - floors[PAD_TABLE] = 1 << 16; - let plan = solve(base, floors).expect("solvable"); - let got = filled(base, &plan); - assert!(is_filled(got), "{got:?}"); - assert_eq!(got[PAD_TABLE], 1 << 16); - } - /// The bulk of a fill rides the largest block, so the fill stays cheap: one closing /// jump per 128 rows, plus at most one traversal per size per table for the /// remainders. #[test] fn fill_uses_bulk_blocks() { - let base = [125_000, 286_000, 341_000, 508_000, 114_000, 130_000]; - let plan = solve(base, NO_FLOORS).expect("solvable"); + let base = [125_000; N_TABLES]; + let plan = solve(base); let fill: usize = delivered(&plan).iter().sum(); assert!( traversals(&plan) <= fill / SIZES[0] + SIZES.len() * N_TABLES, diff --git a/crates/lean_vm/src/cpu/hints.rs b/crates/lean_vm/src/cpu/hints.rs deleted file mode 100644 index a9161bdac..000000000 --- a/crates/lean_vm/src/cpu/hints.rs +++ /dev/null @@ -1,224 +0,0 @@ -//! Runtime hint machinery shared by the interpreter and the compiler: the -//! resolved hint ops a [`super::Program`] carries ([`RHint`]), and the g-power -//! table + reverse index the hint interpreter grows on demand. - -use primitives::field::F64; - -/// Frame-relative offset operand (matches the compiler's `ir::Off`). -pub type Off = u32; - -/// Cells a program can address: [`GPow`] covers no more. -pub(crate) const MAX_CELLS: u32 = 1 << 28; - -/// The `g^k` table paired with a reverse index `g^k ↦ k`, both grown on demand -/// (recursion depth, and so the address range, is unbounded). -/// -/// The interpreter needs both directions: a cell index becomes the address `g^k`, -/// and a pointer word read back out of memory must be inverted to the index it -/// addresses. Two things keep this cheap. -/// -/// The reverse index holds nothing but the exponent, confirming a candidate slot -/// against the forward table, so a slot is four bytes where a `HashMap` bucket is -/// sixteen plus a control byte. Keys are field elements, hence effectively -/// uniform, which is all the placement asks for: one multiplicative mix, then -/// linear probing. The table stays at most half full to keep probing short. -/// -/// And it indexes only the exponents [`Self::note`] announces: a run addresses -/// millions of cells but stores only the few thousand bases of its frames and -/// buffers, and one random write per addressable cell into a table far too large -/// for cache is what building the whole inverse costs. A lookup that misses the -/// announced set indexes everything and retries, so an address arriving by a -/// route the interpreter cannot label is slow, never wrong. -#[derive(Default)] -pub struct GPow { - pow: Vec, - /// The announced exponents, so widening rebuilds from them rather than - /// scanning the old table. - noted: Vec, - /// `k + 1` at the slot `g^k` mixes to; 0 marks an empty slot. Power-of-two - /// length. - index: Vec, - /// Set once a lookup missed: every exponent is announced from then on. - dense: bool, -} - -impl GPow { - /// Seeded with `g^0 = 1`, grown to cover index `upto`, every exponent up to - /// there announced. - pub fn new(upto: usize) -> Self { - let mut g = Self::default(); - g.pow.push(F64::ONE); - g.grow_to(upto); - g.index = vec![0; (2 * (upto + 1)).next_power_of_two().max(1 << 13)]; - g.noted.reserve((upto + 1).next_power_of_two()); - for k in 0..=upto { - g.note(k); - } - g - } - - /// `g^k`. Panics past the grown range, which is the honest outcome: every - /// index the interpreter reads has been covered by a [`Self::grow_to`]. - #[inline(always)] - pub fn pow(&self, k: usize) -> F64 { - self.pow[k] - } - - /// One past the largest index the forward table covers. - #[inline(always)] - pub fn covered(&self) -> usize { - self.pow.len() - } - - /// Extend the forward table to cover index `upto`. - pub fn grow_to(&mut self, upto: usize) { - assert!(upto < MAX_CELLS as usize, "address space overflow (program too large)"); - while self.pow.len() <= upto { - // ×g is ×x = `mul_by_g` (shift+fold), not a PMULL. - let next = primitives::field::mul_by_g(*self.pow.last().unwrap()); - self.pow.push(next); - if self.dense { - self.note(self.pow.len() - 1); - } - } - } - - /// Announce that `g^k` may come back for inversion, because the interpreter - /// is about to store it in memory as an address. Covers `k` first. Idempotent. - pub fn note(&mut self, k: usize) { - self.grow_to(k); - if self.noted.len() * 2 >= self.index.len() { - self.widen(); - } - let mask = self.index.len() - 1; - let mut i = slot_of(self.pow[k].0, mask); - loop { - let e = self.index[i]; - if e == 0 { - break; - } - // Exponents are distinct (`g` has order `2^64 − 1`), so finding `k` - // along its own probe chain means it is already announced. - if e - 1 == k as u32 { - return; - } - i = (i + 1) & mask; - } - self.index[i] = k as u32 + 1; - self.noted.push(k as u32); - } - - /// The discrete log of `x`, if it is a covered g-power. - #[inline] - pub fn log(&mut self, x: F64) -> Option { - if let Some(k) = self.lookup(x) { - return Some(k); - } - // Announced exponents are distinct and below `pow.len()`, so an equal - // count means the whole range is already announced and there is nothing - // a dense pass could add. - if self.dense || self.noted.len() == self.pow.len() { - return None; - } - // An address the interpreter never announced: index the lot and retry. - self.dense = true; - self.noted = (0..self.pow.len() as u32).collect(); - self.widen(); - self.lookup(x) - } - - #[inline] - fn lookup(&self, x: F64) -> Option { - let mask = self.index.len() - 1; - let mut i = slot_of(x.0, mask); - loop { - let e = self.index[i]; - if e == 0 { - return None; - } - if self.pow[(e - 1) as usize] == x { - return Some(e - 1); - } - i = (i + 1) & mask; - } - } - - /// Rebuild the index at a size that holds `noted` at most half full. - fn widen(&mut self) { - let cap = (self.noted.len() * 2 + 1).next_power_of_two().max(1 << 13); - let mut index = vec![0u32; cap]; - let mask = cap - 1; - for &k in &self.noted { - let mut i = slot_of(self.pow[k as usize].0, mask); - while index[i] != 0 { - i = (i + 1) & mask; - } - index[i] = k + 1; - } - self.index = index; - } -} - -/// Bits 32..64 of the mix are the well-distributed ones; the mask keeps as many -/// of them as the table is wide. -#[inline(always)] -fn slot_of(key: u64, mask: usize) -> usize { - (key.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> 32) as usize & mask -} - -/// Where a computed-advice bit buffer lives. Frame cells are ordinary memory, so -/// a run of them serves as well as a heap region and costs no `DEREF` to index at -/// a compile-time offset; the heap form stays for a buffer indexed at run time. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum BitsDest { - /// Frame cells `fp+base+k`. - Stack(Off), - /// Heap cells `m[fp+ptr]·g^k`, the pointer read at run time. - Heap(Off), -} - -/// A hint resolved to concrete offsets/sizes, keyed by global program counter. -#[derive(Clone, Debug)] -pub enum RHint { - /// Announce a pointer into a reserved run of frames. - FrameAddress { offset: Off }, - /// Reserve `(log_g(m[end] * start_inverse) + 1)` consecutive frames. - AllocFrames { - ptr: Off, - size: u32, - end: Off, - start_inverse: F64, - }, - /// Allocate a fresh region of `size` cells and write `g^{base}` to the cell. - Alloc { ptr: Off, size: u32 }, - /// `Alloc` with the cell count read at runtime as the g-power exponent of - /// `m[fp+size]`. - AllocDyn { ptr: Off, size: Off }, - /// Pop stream `name`'s next entry (`len` values) into frame cells `fp+base+k`. - WitnessStack { name: String, base: Off, len: u32 }, - /// Pop stream `name`'s next entry (`len` values) into heap cells `m[fp+ptr]·g^{lo+k}`. - WitnessHeap { name: String, ptr: Off, lo: u32, len: u32 }, - /// Write `g^max(log2_ceil(value), floor)` into `fp+dst`, where `value` is the - /// integer reconstructed from the `nbits` bits at `bits`. - Log2Ceil { - bits: BitsDest, - dst: Off, - nbits: u32, - floor: u32, - }, - /// Write the `nbits` bits of `m[fp+value]` into `bits`. - BitDecompose { value: Off, bits: BitsDest, nbits: u32 }, - /// Write the `nbits` bits of `n`, where `m[fp+value] = g^n` (a bounded - /// discrete log at witness generation), into `bits`. - BitDecomposeExp { value: Off, bits: BitsDest, nbits: u32 }, - /// Write the first `len` K-coordinate limbs of `m[fp+value]` to - /// `m[fp+base..]`. Computed advice; callers constrain the result. - FieldLimbs { value: Off, base: Off, len: u32 }, - /// Write `m[fp+value]⁻¹` to `m[fp+dst]`, or `0` when the value is zero. - /// Untrusted: `assert a != b` multiplies the two back together and asserts - /// `1`, which a zero value cannot satisfy (`FnLower::lower_assert_ne`). - Inverse { value: Off, dst: Off }, - /// Prover-side debug print (`print(...)` in the zkDSL): display the value - /// of `m[fp+cell]` at this program point. Witness generation only. - Print { label: String, cell: Off }, -} diff --git a/crates/lean_vm/src/cpu/isa.rs b/crates/lean_vm/src/cpu/isa.rs deleted file mode 100644 index 9b81cd90e..000000000 --- a/crates/lean_vm/src/cpu/isa.rs +++ /dev/null @@ -1,75 +0,0 @@ -//! The ISA and the `DEREF` store modes. - -use primitives::field::{F64, F192}; - -#[derive(Clone, Copy, Debug)] -pub enum Op { - Xor { - a: u32, - b: u32, - c: u32, - }, - Mul { - a: u32, - b: u32, - c: u32, - }, - Set { - o: u32, - /// The immediate stored into `mem[fp·o]`. A full 192-bit machine word - /// (`E = F192`); K-valued constants (addresses, small ints) ride the - /// low lane with `c1 = c2 = 0`. - k: F192, - }, - Deref { - o1: u32, - o2: u32, - o3: u32, - mode: DerefMode, - }, - Jump { - oc: u32, - od: u32, - of: u32, - }, - /// `BLAKE2s`: one standard BLAKE2s compression. The four 16-byte - /// message chunks `ins` (each a canonical 128-bit chunk in ONE 192-bit cell, - /// top limb zero) form the 64-byte block; the digest lands in the TWO - /// consecutive cells `out, out+1`. Each message chunk is addressed - /// independently, with no forced contiguity, so the caller need not assemble - /// its operands into adjacent cells. Every operand is a memory operand, the - /// metadata included. The compression relation is proven by flock. - Blake2s { - ins: [u32; 4], - /// Base of two consecutive cells holding the 256-bit chaining value - /// (canonical 128-bit chunks, top limbs zero). - cv: u32, - out: u32, - /// The cell holding the metadata `counter:u64 | f0:u32 | f1:u32`, - /// little-endian in its two low K-lanes (top lane zero, as for every - /// other cell this opcode reads). `f0` is the final-block flag and `f1` - /// the last-node flag. A memory operand like the rest, so a program can - /// hash any length: a compile-time counter is one pooled `SET` per - /// frame, a runtime one any cell the program computes. - md: u32, - }, -} - -/// The source `DEREF` stores at `mem[loc_o1·o2]`: a local cell, the return -/// address `g²·pc`, or the frame pointer. Encoded as two boolean flags `(f_pc, -/// f_fp)`: `Cell=(0,0)`, `Pc=(1,0)`, `Fp=(0,1)`, keeping the store constraint degree 2. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum DerefMode { - Cell, - Pc, - Fp, -} - -impl DerefMode { - pub(crate) fn f_pc(self) -> F64 { - if self == DerefMode::Pc { F64::ONE } else { F64::ZERO } - } - pub(crate) fn f_fp(self) -> F64 { - if self == DerefMode::Fp { F64::ONE } else { F64::ZERO } - } -} diff --git a/crates/lean_vm/src/cpu/layout.rs b/crates/lean_vm/src/cpu/layout.rs index 6d7c8811d..6713b8043 100644 --- a/crates/lean_vm/src/cpu/layout.rs +++ b/crates/lean_vm/src/cpu/layout.rs @@ -1,29 +1,41 @@ //! The public column schema and bus layout: the committed-column indices, and -//! the flush/count blocks the verifier reconstructs from the program + announced -//! sizes and public input. Plus the prover-side witness build. +//! the flush/count blocks the verifier reconstructs from the program and the +//! announced sizes. Plus the prover-side witness build. use super::*; +use crate::leaf::SparseColumn; +use crate::rv::{ADVICE_BASE, INPUT_WORDS, LOG_REGS, RAM_BASE, TEXT_BASE}; // ---- column schema ----------------------------------------------------------- -// Shared committed columns (indices `0..N_SHARED`). The program (opcode + -// operands) is PUBLIC, not committed: it rides the bytecode seed/finalize blocks -// as `Coord::Public`; only the witness-dependent finalize counts are committed. -// The data-memory image, a 192-bit word per cell committed as three K-lane columns. -pub const MEM_LO: usize = 0; -pub const MEM_HI: usize = 1; -pub const MEM_TOP: usize = 2; -pub const MFCNT: usize = 3; // per-cell memory access count, g^{A[i]} -pub const BFCNT: usize = 4; // per-pc bytecode execution count, g^{A[pc]} -// flock's packed BLAKE2s witness `q_flock`, committed in the SAME stack as every -// other column (single PCS). Size `2^(K_LOG+n_log-6)` F64 words, always ≥ 1 -// instance (a no-BLAKE2s program commits one full padding instance). It is the -// SOLE copy of the input/output words: the VM's BLAKE2s value columns are -// virtual and their memory-bus claims route to `q_flock` slots (§hash_flock), so -// nothing duplicates them. flock's R1CS validity is discharged by the single -// stacked WHIR opening over this commitment. -pub const QFLOCK: usize = 5; -pub const N_SHARED: usize = 6; +// Shared committed columns (indices `0..N_SHARED`). The program is PUBLIC, not +// committed: it rides the bytecode seed/finalize blocks as `Coord::Public`; only the +// witness-dependent finalize counts are committed. So are the registers and RAM before +// the run, zero and the program's image: what is committed is what they hold after it, +// and each cell's last timestamp (§sec:memchan). The advice is the one array whose +// initial words are committed as well: they are the prover's. +pub const REG_FIN: usize = 0; +pub const REG_FTS: usize = 1; // per-register final timestamp g^y, g^0 if never accessed +pub const MEM_FIN: usize = 2; +pub const MFTS: usize = 3; +pub const ADV_INIT: usize = 4; +pub const ADV_FIN: usize = 5; +pub const ADV_FTS: usize = 6; +pub const BFCNT: usize = 7; // per-pc bytecode execution count, g^{A[pc]} +// Per-entry read counts of the two range arrays (§sec:rangecheck). +pub const RLO_CNT: usize = 8; +pub const RHI_CNT: usize = 9; +/// Then one packed flock witness per table, committed in the SAME stack as every +/// other column (single PCS): `2^(k_log + tau - 6)` words, the SOLE copy of the +/// table's circuit words, whose columns are virtual and route their claims here +/// (§class_flock). +pub const Q_BASE: usize = 10; +pub const N_SHARED: usize = Q_BASE + tables::N_TABLES; + +/// The committed column holding table `t`'s packed witness. +pub(crate) const fn q_column(t: usize) -> usize { + Q_BASE + t +} /// Global column indexing: the shared columns occupy `0..N_SHARED`, then each /// table `t` (in [`tables::tables`] order) owns the contiguous block `[base[t], @@ -64,13 +76,13 @@ fn offset_coord(base: usize, c: Coord) -> Coord { } /// The public proof structure: everything the verifier reconstructs from the -/// program, the announced sizes, and the public input, with no witness values. The -/// flush blocks reference columns by INDEX (see [`crate::leaf::Coord`]), so they -/// are pure public structure. +/// program and the announced sizes, with no witness values. The flush blocks +/// reference columns by INDEX (see [`crate::leaf::Coord`]), so they are pure public +/// structure. pub struct Layout { pub push: Vec, pub pull: Vec, - /// Count channel: read-count columns whose product must be nonzero (§sec:memchan). + /// Count channel: the lookups' read-count columns, whose product must be nonzero (§sec:lookup). pub count: Vec, /// Per-column placement (offset + n_vars) in the stacked witness; from the /// columns' log-sizes alone, so reconstructable by the verifier. @@ -78,24 +90,21 @@ pub struct Layout { /// The stacked witness's shape: its announced `2^mu` size, plus how many lane /// blocks of it the prover actually commits (see [`witness::StackShape`]). pub shape: witness::StackShape, - /// Public input: the first two memory cells `m[0], m[1]` (each a 192-bit - /// word), bound to the committed memory at verification (§sec:e2e-pi). - pub pi: [F192; 2], pub taus: [usize; tables::N_TABLES], } /// The prover's witness: the stacked multilinear `q`, which holds every committed -/// column at its placed offset, plus the public [`Layout`] (and the sizes needed to -/// announce it). +/// column at its placed offset, plus the public [`Layout`]. pub(crate) struct Witness { pub(crate) q: zk_alloc::ArenaVec, /// The virtual columns' values as `(global column index, values)`. They carry /// data for the bus but are not committed, so they are not in `q`. pub(crate) virt: Vec<(usize, zk_alloc::ArenaVec)>, pub(crate) layout: Layout, - pub(crate) log_mem: usize, - /// Freed immediately after reduction, before the mixed PCS opening. - pub(crate) flock_reduction: crate::hash_flock::PreparedReductionWitness, + /// The clock the run ended on, which the prover announces. + pub(crate) ts_final: F64, + /// Each table's flock batch, freed right after its reduction. + pub(crate) reductions: Vec, } impl Witness { @@ -131,59 +140,86 @@ impl Witness { } } -/// The committed columns' kappa SOURCES, for the recursion guest's -/// in-circuit certification of the stacked size m = max(log2_ceil(sum of -/// 2^kappa), MIN_MU). Per committed column: `Some((source, adj))` with -/// kappa = value(source) + adj, where source 0 is the constant 0 (kappa = -/// adj; used for the fixed-size columns and the program bytecode length, -/// which the caller passes as `log_bytecode`), source 1 is log_mem, and -/// source 2 + t is tau_t. `None` = virtual (never committed). `col_kappas` -/// is derived from this, so the two cannot drift apart. -pub fn col_kappa_sources(log_bytecode: usize) -> Vec> { +/// The program's sizes the layout depends on: `log2` of its entries, of RAM's words, +/// and of the advice's. +#[derive(Clone, Copy)] +pub struct Sizes { + pub log_bytecode: usize, + pub log_ram: usize, + pub log_advice: usize, +} + +impl Sizes { + pub fn of(p: &rv::Program) -> Self { + Self { + log_bytecode: crate::log2_strict_usize(p.entries.len()), + log_ram: p.log_ram, + log_advice: p.log_advice, + } + } +} + +/// The committed columns' kappa SOURCES. Per committed column: `Some((source, adj))` with +/// kappa = value(source) + adj, where source 0 is the constant 0 (kappa = adj; used for +/// the fixed-size columns and the program's sizes, which the caller passes), and +/// source 1 + t is tau_t. `None` = virtual (never committed). `col_kappas` is derived +/// from this, so the two cannot drift apart. +pub fn col_kappa_sources(sizes: Sizes) -> Vec> { let sch = schema(); let mut k = vec![Some((0usize, 0usize)); sch.n]; - k[MEM_LO] = Some((1, 0)); - k[MEM_HI] = Some((1, 0)); - k[MEM_TOP] = Some((1, 0)); - k[MFCNT] = Some((1, 0)); - k[BFCNT] = Some((0, log_bytecode)); - // q_flock is `2^(K_LOG + n_blocks_log - LOG_PACKING)` F64 words, always ≥ 1 - // instance (a no-BLAKE2s program commits one padding instance), and tau_5 IS - // n_blocks_log (the announced-size certification uses the same floor), so this - // reproduces `qflock_kappa`. - k[QFLOCK] = Some((2 + tables::BLAKE2S_TABLE, flock::hash::K_LOG - ::pcs::LOG_PACKING)); + k[REG_FIN] = Some((0, LOG_REGS)); + k[REG_FTS] = Some((0, LOG_REGS)); + k[MEM_FIN] = Some((0, sizes.log_ram)); + k[MFTS] = Some((0, sizes.log_ram)); + k[ADV_INIT] = Some((0, sizes.log_advice)); + k[ADV_FIN] = Some((0, sizes.log_advice)); + k[ADV_FTS] = Some((0, sizes.log_advice)); + k[BFCNT] = Some((0, sizes.log_bytecode)); + k[RLO_CNT] = Some((0, tables::RANGE_LOG)); + k[RHI_CNT] = Some((0, tables::RANGE_LOG)); for (t, table) in tables::tables().iter().enumerate() { let base = sch.base[t]; - k[base..base + table.n_committed_columns()].fill(Some((2 + t, 0))); - } - // The BLAKE2s value columns are ALWAYS virtual: `q_flock` already holds those - // words at fixed packed slots, so committing them again is redundant. Their - // memory-bus claims route directly to `q_flock` slot evaluations (`slot_claims`), - // which both binds them to the proven witness AND removes the separate - // value-binding sub-protocol. - let b3 = sch.base[tables::BLAKE2S_TABLE]; - for &c in &tables::BLAKE2S_VALUE_COLS { - k[b3 + c] = None; + k[base..base + table.n_committed_columns()].fill(Some((1 + t, 0))); + // The circuit's words are ALWAYS virtual: the class's packed witness already + // holds them at fixed packed slots, so committing them again is redundant. + // Their bus claims route directly to slot evaluations of it (`slot_claims`), + // which is the whole binding. + k[q_column(t)] = Some((1 + t, crate::class_flock::stride_log(tables::CLASSES[t]))); + for (_, c) in tables::word_columns(t) { + k[base + c] = None; + } } k } +/// The framework blocks a side starts with, as `(source, adj)`: the state boundary, the +/// registers, RAM, the advice, the bytecode, the two range arrays. +fn framework_kappa_sources(sizes: Sizes) -> Vec<(usize, usize)> { + vec![ + (0, 0), + (0, LOG_REGS), + (0, sizes.log_ram), + (0, sizes.log_advice), + (0, sizes.log_bytecode), + (0, tables::RANGE_LOG), + (0, tables::RANGE_LOG), + ] +} + /// The bus flush blocks' kappa SOURCES, flattened in side order (push, pull, /// count) exactly as the blocks are constructed below: per block /// `(source, adj)` with kappa = value(source) + adj, source 0 = the constant -/// 0, 1 = log_mem, 2 + t = tau_t. For the recursion guest's in-circuit pin -/// of every hinted block kappa. Keep in lockstep with the block -/// construction in [`fn@layout`]. -pub fn block_kappa_sources(log_bytecode: usize) -> Vec<(usize, usize)> { - let mut push = vec![(0, 0), (1, 0), (0, log_bytecode)]; - let mut pull = vec![(0, 0), (1, 0), (0, log_bytecode)]; +/// 0, 1 + t = tau_t. Keep in lockstep with the block construction in [`fn@layout`]. +pub fn block_kappa_sources(sizes: Sizes) -> Vec<(usize, usize)> { + let mut push = framework_kappa_sources(sizes); + let mut pull = push.clone(); let mut count = Vec::new(); for (t, table) in tables::tables().iter().enumerate() { let mut fb = tables::FlushBuilder::new(); table.flushes(&mut fb); - push.extend(std::iter::repeat_n((2 + t, 0), fb.push.len())); - pull.extend(std::iter::repeat_n((2 + t, 0), fb.pull.len())); - count.extend(std::iter::repeat_n((2 + t, 0), table.count_columns().len())); + push.extend(std::iter::repeat_n((1 + t, 0), fb.push.len())); + pull.extend(std::iter::repeat_n((1 + t, 0), fb.pull.len())); + count.extend(std::iter::repeat_n((1 + t, 0), table.count_columns().len())); } push.extend(pull); push.extend(count); @@ -191,200 +227,147 @@ pub fn block_kappa_sources(log_bytecode: usize) -> Vec<(usize, usize)> { } /// Column → log-size (`kappa`) map, derived from [`col_kappa_sources`] by -/// substituting the announced sizes: source 0 is the constant 0, source 1 is -/// `log_mem`, source `2 + t` is `tau_t`. `None` marks a **virtual** -/// (uncommitted) column. Depends only on the public sizes, so the verifier can -/// reconstruct the placements. -fn col_kappas(log_mem: usize, log_bytecode: usize, taus: [usize; tables::N_TABLES]) -> Vec> { - let mut values = vec![0usize, log_mem]; +/// substituting the announced sizes. `None` marks a **virtual** (uncommitted) column. +/// Depends only on the public sizes, so the verifier can reconstruct the placements. +fn col_kappas(sizes: Sizes, taus: [usize; tables::N_TABLES]) -> Vec> { + let mut values = vec![0usize]; values.extend(taus); - col_kappa_sources(log_bytecode) + col_kappa_sources(sizes) .iter() .map(|s| s.map(|(source, adj)| values[source] + adj)) .collect() } -/// `log2` of the stacked witness the announced sizes imply. The one part of -/// [`layout`] a size floor needs, and far cheaper than the rest of it. -pub(crate) fn committed_log(log_mem: usize, log_bytecode: usize, taus: [usize; tables::N_TABLES]) -> usize { - crate::witness::placements_of(&col_kappas(log_mem, log_bytecode, taus)) - .1 - .mu -} +/// How many PUBLIC columns a bytecode entry is: the class tag, then `flags, a1, a2, +/// ad, imm, pc4, dt, link, jalr` (§sec:e2e-bc). +pub const N_BYTECODE_COLUMNS: usize = 10; -/// Build the public [`Layout`] from the program, the memory log-size `log_mem`, the -/// instruction tables' log heights `taus`, and the public input `pi`. The flush blocks -/// reference columns only by INDEX and the program only through its public columns, so -/// this needs no committed witness: both prover and verifier reconstruct exactly the same -/// structure. -/// -/// A table's height is its row count: the fill blocks bring every count up to a power of -/// two (`cpu::filler`), so `2^taus[t]` rows were all executed and no flush has padding -/// tuples to divide back out of the bus. -/// The eight PUBLIC bytecode columns over the program cube, in bytecode-slot -/// order: the opcode, then seven operand/immediate slots. The program is not -/// committed, so these ride the seed/finalize blocks as `Coord::Public` and -/// stack into the polynomial [`bytecode_table`] returns. -pub fn bytecode_columns(prog: &[Op]) -> [Vec; 8] { - let max_op = prog - .iter() - .map(|op| match *op { - Op::Xor { a, b, c } | Op::Mul { a, b, c } => a.max(b).max(c), - Op::Set { o, .. } => o, - Op::Deref { o1, o2, o3, .. } => o1.max(o2).max(o3), - Op::Jump { oc, od, of } => oc.max(od).max(of), - Op::Blake2s { ins, cv, out, md } => ins[0].max(ins[1]).max(ins[2]).max(ins[3]).max(cv).max(out).max(md), - }) - .max() - .unwrap_or(0) as usize; - let gpow = primitives::field::g_powers((max_op + 1).max(2)); - let g_at = |i: u32| gpow[i as usize]; // operand g-power - - let opcode = |op: &Op| match op { - Op::Xor { .. } => OP_XOR, - Op::Mul { .. } => OP_MUL, - Op::Set { .. } => OP_SET, - Op::Deref { .. } => OP_DEREF, - Op::Jump { .. } => OP_JUMP, - Op::Blake2s { .. } => OP_BLAKE2S, - }; - let operands = |op: &Op| -> (F64, F64, F64) { - match *op { - Op::Xor { a, b, c } | Op::Mul { a, b, c } => (g_at(a), g_at(b), g_at(c)), - // The immediate's first two K-limbs ride operand slots o2/o3; c2 - // rides the fpc slot below. - Op::Set { o, k } => (g_at(o), F64(k.c0), F64(k.c1)), - Op::Deref { o1, o2, o3, .. } => (g_at(o1), g_at(o2), g_at(o3)), - Op::Jump { oc, od, of } => (g_at(oc), g_at(od), g_at(of)), - // BLAKE2s's first three input-word offsets; the last two ride the - // fpc/ffp bytecode slots below. - Op::Blake2s { ins, .. } => (g_at(ins[0]), g_at(ins[1]), g_at(ins[2])), - } - }; - // The 4th/5th bytecode operand slots: the two DEREF store-mode flags, or - // BLAKE2s's remaining input word / chaining-value base (0 elsewhere). - let fpc = |op: &Op| match op { - Op::Deref { mode, .. } => mode.f_pc(), - Op::Blake2s { ins, .. } => g_at(ins[3]), - Op::Set { k, .. } => F64(k.c2), - _ => F64::ZERO, - }; - let ffp = |op: &Op| match op { - Op::Deref { mode, .. } => mode.f_fp(), - Op::Blake2s { cv, .. } => g_at(*cv), - _ => F64::ZERO, - }; - // The 6th/7th bytecode operand slots: BLAKE2s's output base and its metadata - // cell (0 elsewhere). - let extra0 = |op: &Op| match op { - Op::Blake2s { out, .. } => g_at(*out), - _ => F64::ZERO, - }; - let extra1 = |op: &Op| match op { - Op::Blake2s { md, .. } => g_at(*md), - _ => F64::ZERO, +/// The public bytecode columns over the program cube, in bytecode-slot order. The +/// program is not committed, so these ride the seed/finalize blocks as +/// `Coord::Public` and stack into the polynomial [`bytecode_table`] returns. +pub fn bytecode_columns(p: &rv::Program) -> [Vec; N_BYTECODE_COLUMNS] { + let column = |f: &(dyn Fn(usize, &rv::Entry) -> u64 + Sync)| { + parallel::map_collect(p.entries.len(), |i| F64(f(i, &p.entries[i]))) }; - // The program is PUBLIC (not committed): eight public columns over the - // program cube, embedded in the bytecode seed/finalize blocks below. - let column = |f: &(dyn Fn(&Op) -> F64 + Sync)| parallel::map_collect(prog.len(), |i| f(&prog[i])); - let prog_op: Vec = column(&opcode); - let prog_o1: Vec = column(&|o| operands(o).0); - let prog_o2: Vec = column(&|o| operands(o).1); - let prog_o3: Vec = column(&|o| operands(o).2); - let prog_fpc: Vec = column(&fpc); - let prog_ffp: Vec = column(&ffp); - let prog_extra0: Vec = column(&extra0); - let prog_extra1: Vec = column(&extra1); [ - prog_op, - prog_o1, - prog_o2, - prog_o3, - prog_fpc, - prog_ffp, - prog_extra0, - prog_extra1, + // An illegal entry's tag is zero, which is no table's: nothing can read it. + parallel::map_collect(p.entries.len(), |i| { + tables::table_of(p.entries[i].class).map_or(F64::ZERO, primitives::field::g_pow) + }), + column(&|_, e| e.flags), + column(&|_, e| e.a1 as u64), + column(&|_, e| e.a2 as u64), + column(&|_, e| e.ad as u64), + column(&|_, e| e.imm), + column(&|i, _| p.pc_of(i).wrapping_add(4)), + column(&|i, _| p.dt_of(i)), + column(&|_, e| e.link as u64), + column(&|_, e| e.jalr as u64), ] } -/// The stacked bytecode polynomial: the eight columns at their bus tuple -/// coordinates, which is what makes the program's whole share of a bus leaf one -/// evaluation at `(ζ, α⃗)` (see [`crate::leaf::stacked_bytecode_table`]). +/// The stacked bytecode polynomial: the columns at their bus tuple coordinates, +/// which is what makes the program's whole share of a bus leaf one evaluation at +/// `(ζ, α⃗)` (see [`crate::leaf::stacked_bytecode_table`]). /// /// This is the multilinear an outermost verifier is handed in place of a -/// structured program, and what the transcript seed binds ([`super::fs_seed`]). -pub fn bytecode_table(prog: &[Op]) -> Vec { - let coords = bytecode_columns(prog) +/// structured program, and what the program digest binds ([`Program::new`]). +pub fn bytecode_table(p: &rv::Program) -> Vec { + let coords = bytecode_columns(p) .map(|c| Coord::Public(std::sync::Arc::new(c))) .into(); let block = Block { - kappa: crate::log2_strict_usize(prog.len()), + kappa: crate::log2_strict_usize(p.entries.len()), coords, }; crate::leaf::stacked_bytecode_table(std::slice::from_ref(&block)) } -pub fn layout(prog: &[Op], log_mem: usize, taus: [usize; tables::N_TABLES], pi: [F192; 2]) -> Layout { - let bytecode_size = prog.len(); - let log_bytecode = crate::log2_strict_usize(bytecode_size); - - // Derived boundary: the run starts at (pc,fp) = (0,0) and, by convention, the - // final pc is the bytecode's last cell g^{B-1} (the compiler emits a halt jump - // there), with fp returned to 0. All public, no trace needed. - let final_pc = (bytecode_size - 1) as u32; - +/// Build the public [`Layout`] from the program, the run's public input, the tables' log +/// heights `taus` and the clock the prover says the run ended on. The flush blocks reference columns only +/// by INDEX and the program only through its public columns, so this needs no +/// committed witness: both prover and verifier reconstruct exactly the same structure. +/// +/// A table's height is its row count: the fill blocks bring every count up to a power of +/// two (`cpu::filler`), so `2^taus[t]` rows were all executed and no flush has padding +/// tuples to divide back out of the bus. +pub fn layout(p: &rv::Program, input: &[u64; INPUT_WORDS], taus: [usize; tables::N_TABLES], ts_final: F64) -> Layout { + let sizes = Sizes::of(p); + let log_bytecode = sizes.log_bytecode; let one = F64::ONE; - // The bytecode columns map operand *offsets* (small, ≤ frame size) to - // g-powers (not memory addresses), so precompute only up to the largest - // operand, an O(1) lookup each, rather than over the whole 2^log_mem memory. - // Shared between the seed and finalize blocks: at kbc = 19 a copy is tens of - // megabytes per column. - let prog_cols: [std::sync::Arc>; 8] = bytecode_columns(prog).map(std::sync::Arc::new); + // Shared between the seed and finalize blocks: a copy is tens of megabytes per + // column at production sizes. + let prog_cols: [std::sync::Arc>; N_BYTECODE_COLUMNS] = bytecode_columns(p).map(std::sync::Arc::new); // ---- bus blocks ---- - use Coord::{Col, Const, Index, Public}; + use Coord::{Col, Const, IntIndex, Powers, Public, Sparse}; let blk = |kappa: usize, coords: Vec| Block { kappa, coords }; let mut push: Vec = Vec::new(); let mut pull: Vec = Vec::new(); - // Shared blocks (cross-instruction infra, not owned by any single table). - // boundary state. - push.push(blk(0, vec![Const(SEP_STATE), Const(one), Const(one)])); - pull.push(blk( + // Shared blocks (cross-instruction infra, not owned by any single table). The + // boundary: the run starts at the entry point at cycle 1 and ends on the halt + // slot, at whatever clock the prover announced (`ts_final`), which nothing has to + // check: a wrong one unbalances the bus. + push.push(blk( 0, - vec![Const(SEP_STATE), Const(g_pow(final_pc as usize)), Const(one)], + vec![Const(SEP_STATE), Const(F64(p.entry_pc)), Const(tables::CLOCK_START)], + )); + pull.push(blk(0, vec![Const(SEP_STATE), Const(F64(p.halt_pc())), Const(ts_final)])); + // Register seed + finalize: every register starts at timestamp g^0 holding zero, + // and ends at its last timestamp holding its final word (§sec:memchan). + let cell = IntIndex { + base: F64::ZERO, + shift: 0, + }; + push.push(blk(LOG_REGS, vec![Const(tables::SEP_REG), cell.clone(), Const(one)])); + pull.push(blk( + LOG_REGS, + vec![Const(tables::SEP_REG), cell, Col(REG_FTS), Col(REG_FIN)], )); - // memory seed + finalize (every address real, no padding). The value is the - // full three-limb 192-bit word. + // RAM the same way, cell `z` at its byte address `RAM_BASE + 8z`. What it holds + // before the run is public: the input, the program's image, zeros. + let word = IntIndex { + base: F64(RAM_BASE), + shift: 3, + }; + let ram = SparseColumn::new(p.log_ram, &[(0, input), (INPUT_WORDS, &p.image)]); push.push(blk( - log_mem, + p.log_ram, vec![ - Const(SEP_MEM), - Index, + Const(tables::SEP_MEM), + word.clone(), Const(one), - Col(MEM_LO), - Col(MEM_HI), - Col(MEM_TOP), + Sparse(std::sync::Arc::new(ram)), ], )); pull.push(blk( - log_mem, - vec![ - Const(SEP_MEM), - Index, - Col(MFCNT), - Col(MEM_LO), - Col(MEM_HI), - Col(MEM_TOP), - ], + p.log_ram, + vec![Const(tables::SEP_MEM), word, Col(MFTS), Col(MEM_FIN)], + )); + // The advice, the one array seeded from a committed column: the prover's words. + let word = IntIndex { + base: F64(ADVICE_BASE), + shift: 3, + }; + push.push(blk( + p.log_advice, + vec![Const(tables::SEP_MEM), word.clone(), Const(one), Col(ADV_INIT)], + )); + pull.push(blk( + p.log_advice, + vec![Const(tables::SEP_MEM), word, Col(ADV_FTS), Col(ADV_FIN)], )); - // bytecode seed + finalize (program columns are public; padding entries - // self-cancel at count 1, so the whole 2^log_bytecode is "real"). + // Bytecode seed + finalize, entry `i` at its `pc` (the program columns are public). let bytecode_block = |count: Coord| { + let pc = IntIndex { + base: F64(TEXT_BASE), + shift: 2, + }; blk( log_bytecode, - [Const(SEP_BYTECODE), Index, count] + [Const(SEP_BYTECODE), pc, count] .into_iter() .chain(prog_cols.iter().cloned().map(Public)) .collect(), @@ -392,6 +375,18 @@ pub fn layout(prog: &[Op], log_mem: usize, taus: [usize; tables::N_TABLES], pi: }; push.push(bytecode_block(Const(one))); pull.push(bytecode_block(Col(BFCNT))); + // The two range arrays (§sec:rangecheck): entries with no value, so a read is a + // range check on its address. Neither is committed: their addresses are + // geometric, `g^{j+1}` and `g^{-2^16·j}`. + for (sep, first, ratio, count) in [ + (tables::SEP_RANGE_LO, tables::range_lo_first(), F64::G, RLO_CNT), + (tables::SEP_RANGE_HI, one, tables::range_hi_ratio(), RHI_CNT), + ] { + let addresses = Powers { first, ratio }; + push.push(blk(tables::RANGE_LOG, vec![Const(sep), addresses.clone(), Const(one)])); + pull.push(blk(tables::RANGE_LOG, vec![Const(sep), addresses, Col(count)])); + } + debug_assert_eq!(push.len(), framework_kappa_sources(sizes).len()); // Per-table blocks: each table declares its flushes and read-count columns in // local indices; offset them to the table's global columns. @@ -413,68 +408,69 @@ pub fn layout(prog: &[Op], log_mem: usize, taus: [usize; tables::N_TABLES], pi: } } - let (placements, shape) = witness::placements_of(&col_kappas(log_mem, log_bytecode, taus)); + let (placements, shape) = witness::placements_of(&col_kappas(sizes, taus)); Layout { push, pull, count: count_blocks, placements, shape, - pi, taus, } } impl Program { - pub(crate) fn build(&self, exec: &Execution) -> Witness { - assert!(self.prog.len().is_power_of_two()); - assert!(exec.mem.len().is_power_of_two()); - // The trace was emitted in the same walk as the memory image (no re-walk). - let tr = &exec.trace; - let cells = exec.mem.len(); - let bytecode_size = self.prog.len(); - let log_mem = crate::log2_strict_usize(cells); + /// `log2` of the stacked witness a run of these row counts commits: what one proof + /// can hold is capped ([`pcs::MAX_MU`]), so a run is checked before it is built. + pub(crate) fn stack_log(&self, row_counts: [usize; tables::N_TABLES]) -> usize { + let taus = row_counts.map(|rows| crate::log2_ceil_usize(rows.max(1))); + witness::placements_of(&col_kappas(Sizes::of(&self.rv), taus)).1.mu + } + pub(crate) fn build(&self, exec: &Execution, input: &[u64; INPUT_WORDS]) -> Witness { + let p = &self.rv; + // The trace was emitted in the same walk as the run (no re-walk). + let tr = &exec.trace; let sch = schema(); - // Precompute g^0..g^{span-1} once so every address/pc/operand fill is an - // O(1) lookup instead of an O(log) power. - let span = cells.max(bytecode_size); - let gpow = primitives::field::g_powers(span); // The public layout (flush/count blocks, placements, boundary, taus) is a pure - // function of the program + announced sizes + public input, with no committed - // witness; reconstruct it here so the prover and verifier share exactly the same + // function of the program and the announced sizes, with no committed witness; + // reconstruct it here so the prover and verifier share exactly the same // structure. It comes before the fill because it fixes each table's height // `2^tau`, which lets every column be allocated at its final length in one pass. - let row_counts = exec.trace.row_counts(); + let row_counts = tr.row_counts(); assert!( row_counts.iter().all(|&r| r <= 1 << MAX_LOG_ROWS), "a table exceeds 2^{MAX_LOG_ROWS} rows" ); // Every table's rows are real rows, so its height IS its row count: the fill - // blocks ran each count up to a power of two, and BLAKE2s up to flock's instance - // floor as well (`cpu::filler`). - let taus = row_counts.map(|r| { + // blocks ran each count up to a power of two, and up to flock's instance floor + // as well (`cpu::filler`). + let taus: [usize; tables::N_TABLES] = std::array::from_fn(|t| { + let r = row_counts[t]; assert!( r.is_power_of_two(), - "a table has {r} rows, not a power of two: the fill blocks did not fill \ - it (cpu::filler)" + "a table has {r} rows, not a power of two: the fill blocks did not fill it (cpu::filler)" + ); + let tau = crate::log2_strict_usize(r); + assert_eq!( + tau, + crate::class_flock::n_blocks_log(tables::CLASSES[t], r), + "the {} table must be filled to flock's instance floor", + tables::CLASSES[t].name ); - crate::log2_strict_usize(r) + tau }); - assert_eq!( - taus[tables::BLAKE2S_TABLE], - crate::hash_flock::n_blocks_log(row_counts[tables::BLAKE2S_TABLE]), - "the BLAKE2s table must be filled to flock's instance floor" - ); - let pi = [exec.mem[0], exec.mem[1]]; - let l = layout(&self.prog, log_mem, taus, pi); + let l = layout(p, input, taus, tr.ts_final); + // The range arrays' addresses, to turn a gap's chunks into column values. + let range_lo = primitives::field::geometric(tables::range_lo_first(), F64::G, 1 << tables::RANGE_LOG); + let range_hi = primitives::field::geometric(F64::ONE, tables::range_hi_ratio(), 1 << tables::RANGE_LOG); // The stacked witness is written exactly ONCE: allocate it, carve one window // per committed column, and have every fill write its column straight into // place. Copying columns in afterwards would move the whole witness a second - // time, a gigabyte at this scale, for no gain: nothing folds the K-columns - // in place, so the stack can be their only home. + // time for no gain: nothing folds the K-columns in place, so the stack can be + // their only home. // // SAFETY: the allocation is uninitialized. `split_stack` zeroes the pad tail // and hands out windows tiling the rest; `fill_table` checks that each table @@ -482,13 +478,13 @@ impl Program { let mut q = unsafe { witness::alloc_stack(l.shape) }; // A virtual column is not in the stack, so its values need storage of their // own: it carries data for the bus, and only its evaluation claims route - // elsewhere (to `q_flock`). + // elsewhere (to its class's packed witness). let mut virt: Vec<(usize, zk_alloc::ArenaVec)> = Vec::new(); for (t, table) in tables::tables().iter().enumerate() { for c in 0..table.n_committed_columns() { let i = sch.base[t] + c; if l.placements[i].is_virtual() { - // SAFETY: a virtual window is a table column, `FillCtx::cols_at` + // SAFETY: a virtual window is a table column, `FillCtx::cols` // writes every row of every window it is given, and `fill_table` // asserts each table wrote all of its columns. virt.push((i, unsafe { zk_alloc::ArenaVec::::uninitialized(1 << l.taus[t]) })); @@ -505,54 +501,39 @@ impl Program { crate::stage!("Fill columns", || { for (t, table) in tables::tables().iter().enumerate() { let (base, n) = (sch.base[t], table.n_committed_columns()); - let ctx = FillCtx::new(tr, &exec.mem, &gpow, &self.prog, 1 << l.taus[t]); + let ctx = FillCtx::new(tr, &range_lo, &range_hi, p, 1 << l.taus[t], n); tables::fill_table(*table, &ctx, &mut windows[base..base + n]); } - // Shared columns. The 192-bit memory image splits into three K-limbs. - // These five plus `QFLOCK` below are every shared column, and each has to - // be written: the stack is uninitialized, so one left out would be read - // as indeterminate bytes rather than caught by a length mismatch. - const _: () = assert!(N_SHARED == 6, "a new shared column needs a fill here"); - parallel::fill(windows[MEM_LO], |i| F64(exec.mem[i].c0)); - parallel::fill(windows[MEM_HI], |i| F64(exec.mem[i].c1)); - parallel::fill(windows[MEM_TOP], |i| F64(exec.mem[i].c2)); - parallel::fill(windows[MFCNT], |i| tr.mem_count[i]); // counts ended at g^{A[i]} - parallel::fill(windows[BFCNT], |i| tr.bytecode_count[i]); // … at g^{A[pc]} + // Shared columns. These ten plus the flock witnesses below are every + // shared column, and each has to be written: the stack is uninitialized, so + // one left out would be read as indeterminate bytes rather than caught by + // a length mismatch. + const _: () = assert!(Q_BASE == 10, "a new shared column needs a fill here"); + windows[REG_FIN].copy_from_slice(&tr.reg_fin); + windows[REG_FTS].copy_from_slice(&tr.reg_ts); + windows[MEM_FIN].copy_from_slice(&tr.ram_fin); + windows[MFTS].copy_from_slice(&tr.ram_ts); + windows[ADV_INIT].copy_from_slice(&tr.adv_init); + windows[ADV_FIN].copy_from_slice(&tr.adv_fin); + windows[ADV_FTS].copy_from_slice(&tr.adv_ts); + windows[BFCNT].copy_from_slice(&tr.bytecode_count); // counts ended at g^{A[pc]} + windows[RLO_CNT].copy_from_slice(&tr.range_lo_count); + windows[RHI_CNT].copy_from_slice(&tr.range_hi_count); }); - // flock's packed BLAKE2s witness q_flock, ALWAYS committed in this same stack: - // built from the executed BLAKE2s rows in order (row j = flock instance j), - // padded to `2^n_blocks_log(max(count,1))` all-padding instances, so a - // program with no BLAKE2s still carries a single padding instance. - let flock_reduction = crate::stage!("Build q_flock", || { - // The rows carry only their access counts; the compression's input - // words are the nine cells they read, in the finished (write-once) - // memory image. - let blocks: Vec<_> = parallel::map_collect(tr.blake2s.len(), |i| { - let r = &tr.blake2s[i]; - let a = tables::blake2s_addresses(&self.prog, r); - let chunk = |c0: u32, c1: u32| { - let (w0, w1) = (exec.mem[c0 as usize], exec.mem[c1 as usize]); - [F64(w0.c0), F64(w0.c1), F64(w1.c0), F64(w1.c1)] - }; - crate::hash_flock::compression( - chunk(a[0], a[1]), - chunk(a[2], a[3]), - chunk(a[4], a[4] + 1), - exec.mem[a[6] as usize], - ) - }); - crate::hash_flock::build_qflock_prepared(&blocks, windows[QFLOCK]) + // The classes' packed witnesses, one instance per row of their table. + let reductions = crate::stage!("Build flock witnesses", || { + (0..tables::N_TABLES) + .map(|t| crate::class_flock::Prepared::build(t, &tr.rows[t], &p.entries, windows[q_column(t)])) + .collect() }); - // (`execute` already asserts the run halts at the sentinel (pc, fp) = - // (g^{B-1}, 0), exactly the boundary the public layout derives.) drop(windows); // release the borrow of `q` and of the virtual buffers Witness { q, virt, layout: l, - log_mem, - flock_reduction, + ts_final: tr.ts_final, + reductions, } } } diff --git a/crates/lean_vm/src/cpu/mod.rs b/crates/lean_vm/src/cpu/mod.rs index 8de21934b..c0f46357a 100644 --- a/crates/lean_vm/src/cpu/mod.rs +++ b/crates/lean_vm/src/cpu/mod.rs @@ -1,133 +1,77 @@ //! Whole-program assembly over GF(2^64) (`doc/leanvm/main.tex`): the instruction tables -//! sharing the state / memory / bytecode buses, bound to one field-valued -//! commitment and verified oracle-free. Addresses, the program counter, and read -//! counts are g-powers, so every increment is a free ×g. Machine-word arithmetic -//! is over `E = F192 = K[y]/(y³+y+1)` (XOR degree 1, MUL_NATIVE degree 2), -//! with each word carried by three committed `K = F64` limbs. `BLAKE2s` -//! adds the memory/state/bytecode plumbing for a 64→32-byte compression -//! whose relation is discharged by flock (see [`crate::hash_flock`]). All -//! Challenges and transcript scalars live in the same tower E. - -use std::collections::HashMap; +//! sharing the state, register and bytecode buses, bound to one field-valued commitment +//! and verified oracle-free. The machine is RISC-V ([`crate::rv`]): `pc`, register +//! numbers and addresses are integers, read as the field element with those bits, while +//! timestamps and read counts are g-powers, so every increment is a free ×g. A register +//! is one `K = F64` element. What an instruction computes is a flock circuit +//! ([`crate::class_flock`]); the tables only move words between the bytecode, the +//! registers and those circuits. Challenges and transcript scalars live in `E = F192`. use crate::colval::ColVal; use crate::constraints; use crate::leaf::{self, Block, ColumnClaim, Coord}; use crate::pcs; -use crate::tables::{ - self, FillCtx, FlushBuilder, OP_BLAKE2S, OP_DEREF, OP_JUMP, OP_MUL, OP_SET, OP_XOR, SEP_BYTECODE, SEP_MEM, - SEP_STATE, -}; +use crate::rv; +use crate::tables::{self, FillCtx, FlushBuilder, SEP_BYTECODE, SEP_STATE}; use crate::transcript::{Challenger, ProverState, Receiver, Transmitter, VerifierState}; use crate::witness; -use primitives::field::{F64, F192, g_pow}; +use primitives::field::{F64, F192}; mod execute; pub mod filler; -pub mod hints; -mod isa; pub mod layout; mod trace; -pub use execute::{ExecError, Execution, Fault}; -pub use isa::{DerefMode, Op}; +pub use execute::Execution; pub use layout::*; -pub(crate) use trace::{Brow, Drow, Jrow, Srow, Trace, Xrow}; - -/// Witness-gen `BLAKE2s` compression: the four message cells' eight -/// words are laid out little-endian into 64 bytes, combined with the supplied -/// chaining value and metadata, and the 32-byte result is split back into the -/// four output words `c`. Flock proves this same compression relation -/// ([`crate::hash_flock`]). -fn blake2s_compress(va: [F64; 4], vb: [F64; 4], vcv: [F64; 4], metadata: F192) -> [F64; 4] { - crate::hash_flock::digest(&crate::hash_flock::compression(va, vb, vcv, metadata)) -} +pub(crate) use trace::{Access, HashRow, Row, Trace}; -/// Data-memory size bounds (doc §Memory): memory is `2^h` cells with -/// `MIN_LOG_MEM ≤ h ≤ MAX_LOG_MEM`. The prover pads up to the minimum; the -/// verifier rejects any announced `h` outside the range. `MIN_LOG_MEM` is also -/// the static cap on range-check bounds (`compiler::Stmt::AssertLt`): a bound -/// `≤ 2^MIN_LOG_MEM` keeps the complement argument sound for every memory size -/// the prover may announce. -pub const MIN_LOG_MEM: usize = 16; -const MAX_LOG_MEM: usize = 32; - -/// Each per-opcode table holds at most `2^MAX_LOG_ROWS` rows (executed -/// instructions of that opcode). Together with `MAX_LOG_MEM` and the bytecode -/// cap these are the instance caps from “Counts must not wrap” in `doc/leanvm/body/06-memory-and-bytecode-lookups.tex`: at `ord(g) = 2^64−1` +/// Each table holds at most `2^MAX_LOG_ROWS` rows (executed instructions of its +/// class). Together with the bytecode cap these are the instance caps from “Counts +/// must not wrap” in `doc/leanvm/body/06-bus-interactions.tex`: at `ord(g) = 2^64−1` /// the memory-soundness and count-non-wrap counting arguments are theorems only /// for instances whose total read-flush count stays far below `2^64`, so the /// verifier rejects any announcement exceeding them before running a reduction. -const MAX_LOG_ROWS: usize = 32; - -/// Bytecode-length instance cap (see [`MAX_LOG_ROWS`]): programs are at most -/// `2^32` instructions. -const MAX_LOG_BYTECODE: usize = 32; +pub const MAX_LOG_ROWS: usize = 32; -/// The Fiat-Shamir IV: ONE 32-byte digest, as two field words, committing to -/// everything fixed about the proving environment. -/// -/// Two things go in. [`flock::hash::R1CS_DIGEST`] names the flock BLAKE2s -/// circuit, independent of the instance count: the full instance is -/// block-diagonal and the count is announced and absorbed with the other sizes, -/// so one constant covers every shape. And the bytecode enters through the hash -/// cached on `Program`, BLAKE2s over the stacked multilinear -/// ([`layout::bytecode_table`]) rather than over an assembler digest, so a -/// verifier holding only that polynomial reproduces the seed; that inner hash is -/// cached, so the table is walked once per program rather than once per proof. -/// -/// The IV IS the transcript's starting chaining value ([`fiat_shamir::FiatShamirState::new`]), -/// so all challenges depend on the circuit version and the program before -/// anything else; a recursion guest carries the INNER program's IV in its public -/// input, pinning both with one word pair. -pub fn fs_seed(program: &Program) -> [F192; 2] { +/// The Fiat-Shamir IV: the program's digest, which commits to everything public and +/// fixed about the statement ([`Program::new`]), hashed with the run's public input. +/// All challenges depend on it before anything else; the run's public output seeds +/// the transcript beside it. +pub fn fs_seed(program: &Program, input: &[u64; rv::INPUT_WORDS]) -> [F64; 4] { let mut h = primitives::hash::Hasher::new(); - h.update(b"leanvm"); - // Length-framed so the preimage parses one way: the domain and the bytecode - // hash are fixed-width, so framing the digest between them is all it takes. - h.update(&(flock::hash::R1CS_DIGEST.len() as u64).to_le_bytes()); - h.update(&flock::hash::R1CS_DIGEST); - h.update(&program.bytecode_hash); - let d = h.finalize(); - let word = |o: usize| u64::from_le_bytes(d[o..o + 8].try_into().unwrap()); - [F192::new(word(0), word(8), 0), F192::new(word(16), word(24), 0)] -} - -/// The two 128-bit halves a digest travels in, as the four words the -/// Fiat-Shamir chain runs on. Only defined for a real digest, whose halves have -/// no third limb; [`read_public`] rejects a public input that has one, so a -/// third limb can never be silently dropped from what the transcript binds. -fn digest_words(halves: &[F192; 2]) -> [F64; 4] { - [ - F64(halves[0].c0), - F64(halves[0].c1), - F64(halves[1].c0), - F64(halves[1].c1), - ] + h.update(&program.digest); + for word in input { + h.update(&word.to_le_bytes()); + } + fiat_shamir::digest_words(&h.finalize()) } -/// Announce the prover's sizes (`log_mem`, every table's log height, the PCS rate) -/// by writing them onto the scalar stream, which binds them into the state and lets -/// the verifier reconstruct the layout. The public statement (program + input) is not -/// announced here; it seeds the transcript at construction (see [`fs_seed`]). -/// The boundary states are derived from the program, so they need no binding. +/// Announce the prover's sizes (every table's log height, the PCS rate) by writing +/// them onto the scalar stream, which binds them into the state and lets the verifier +/// reconstruct the layout. The public statement is not announced here; it seeds the +/// transcript at construction (see [`fs_seed`]). The boundary states are derived from +/// the program, so they need no binding. /// /// Log heights, not row counts: every table's rows are real rows, the fill blocks /// having run each count up to a power of two (`filler`), so a height is all there is -/// to say. That also spares both sides a `log2_ceil`, which -/// in-circuit is a bit decomposition against a hinted exponent rather than a shift. -fn announce_public(ps: &mut ProverState, log_mem: usize, taus: [usize; tables::N_TABLES], log_inv_rate: usize) { - ps.add_scalar(F192::new(log_mem as u64, 0, 0)); +/// to say. +fn announce_public(ps: &mut ProverState, taus: [usize; tables::N_TABLES], log_inv_rate: usize, ts_final: F64) { for t in taus { ps.add_scalar(F192::new(t as u64, 0, 0)); } ps.add_scalar(F192::new(log_inv_rate as u64, 0, 0)); + // The clock the run ended on: the final state's timestamp (§sec:state). + ps.add_scalar(F192::from(ts_final)); } -/// Verifier side of [`announce_public`]: read the announced sizes and PCS -/// rate from the stream, validate them, and reconstruct the public [`Layout`] -/// from the program + sizes + public input. (The public input was already bound -/// by seeding the transcript.) -fn read_public(vs: &mut VerifierState, prog: &Program, public_input: &[F192; 2]) -> Result<(Layout, usize), CpuError> { +/// Verifier side of [`announce_public`]: read the announced sizes and PCS rate from +/// the stream, validate them, and reconstruct the public [`Layout`] from the program +/// and those sizes. Nothing the program fixes is read from the prover. +fn read_public( + vs: &mut VerifierState, + prog: &Program, + input: &[u64; rv::INPUT_WORDS], +) -> Result<(Layout, usize), CpuError> { let read_size = |vs: &mut VerifierState| -> Result { let word = vs.next_scalar().map_err(CpuError::Transcript)?; if word.c1 != 0 || word.c2 != 0 { @@ -136,39 +80,32 @@ fn read_public(vs: &mut VerifierState, prog: &Program, public_input: &[F192; 2]) usize::try_from(word.c0).map_err(|_| CpuError::PublicInput) }; - // The transcript binds a public input as two 128-bit halves, so a third limb - // would be dropped and two statements would share a transcript. - if public_input.iter().any(|half| half.c2 != 0) { - return Err(CpuError::PublicInput); - } - let log_mem = read_size(vs)?; let mut taus = [0usize; tables::N_TABLES]; for t in &mut taus { *t = read_size(vs)?; } let log_inv_rate = read_size(vs)?; + let ts_final = vs.next_scalar().map_err(CpuError::Transcript)?; + if ts_final.c1 != 0 || ts_final.c2 != 0 { + return Err(CpuError::PublicInput); + } // The public instance caps ensure that, with `ord(g) = 2^64 − 1`, the // counting arguments (memory soundness, count non-wrap, exponent range checks) // are theorems only when the announced instance keeps the total read-flush // count provably below `2^64 − 1`, so reject any announcement exceeding the // caps BEFORE running any reduction. (A table's row count is the number of - // times its opcode runs, unbounded by the bytecode size since a small loop - // body runs many times, so it gets its own cap, not `bytecode_size`.) - let bytecode_size = prog.prog.len(); - if !bytecode_size.is_power_of_two() - || bytecode_size > (1usize << MAX_LOG_BYTECODE) - || !(MIN_LOG_MEM..=MAX_LOG_MEM).contains(&log_mem) - || taus.iter().any(|&t| t > MAX_LOG_ROWS) - // flock sizes its argument to at least `n_blocks_log(1)` instances, and the - // BLAKE2s table's value columns share that instance cube, so a height below the - // floor describes a layout the arithmetization cannot express. The other two - // verifiers reject it here too (`python-verifier`, `guests/lean_ethereum.py`). - || taus[tables::BLAKE2S_TABLE] < crate::hash_flock::n_blocks_log(1) - || ::pcs::whir::validate_log_inv_rate(log_inv_rate).is_err() - { + // times its class runs, unbounded by the bytecode size since a small loop + // body runs many times, so it gets its own cap.) + let floors_hold = (0..tables::N_TABLES) + // flock sizes its argument to at least `n_blocks_log(1)` instances, and a + // table's circuit words share that instance cube, so a height below the floor + // describes a layout the arithmetization cannot express. `python-verifier` + // rejects it here too. + .all(|t| (crate::class_flock::n_blocks_log(tables::CLASSES[t], 1)..=MAX_LOG_ROWS).contains(&taus[t])); + if !floors_hold || ::pcs::whir::validate_log_inv_rate(log_inv_rate).is_err() { return Err(CpuError::PublicInput); } - let l = layout(&prog.prog, log_mem, taus, *public_input); + let l = layout(&prog.rv, input, taus, F64(ts_final.c0)); // The caps bound each announced log on its own; what the PCS is configured for // is the stacked size they imply, which they do not bound. if !(pcs::MIN_MU..=pcs::MAX_MU).contains(&l.shape.mu) { @@ -177,177 +114,93 @@ fn read_public(vs: &mut VerifierState, prog: &Program, public_input: &[F192; 2]) Ok((l, log_inv_rate)) } +/// A program as the prover and the verifier hold it. #[derive(Clone)] pub struct Program { - pub prog: Vec, // bytecode (size B, power of two) - /// BLAKE2s over the stacked bytecode multilinear, computed once at assembly - /// so proving and verifying the same program do not rehash it (that table is - /// 16·2^kbc words, tens of megabytes at production sizes). Trusted to match - /// `prog`: always set by [`Program::assemble`] from the bytecode, so a - /// `Program` cannot carry a hash inconsistent with its own `prog`. - pub(crate) bytecode_hash: [u8; 32], - /// Prover-side frame/buffer allocation hints (keyed by global pc) and the - /// size of `main`'s frame: the nondeterminism [`Program::execute`] needs to - /// run the program. Public verification (§ `verify`) ignores them. - pub(crate) hints: HashMap>, - pub(crate) main_frame: u32, - /// Named prover witness streams for the program's `hint_witness` calls - /// ([`Program::set_witness`]): a stream is a sequence of *entries* (one - /// slice of values per `hint_witness` call; the same symbol may be - /// hinted many times); each call pops the next entry, whose length must - /// match its destination. Prover-side only; verification ignores them. - pub(crate) witness: HashMap>>, - /// The fill blocks in the bytecode ([`filler`]): the cycles the interpreter - /// traverses, after the program halts, to bring every table's row count to a power - /// of two. Set by the compiler, prover-side only, and no program code reaches them, - /// so a missing or wrong entry costs the prover a run that does not fill rather than - /// anything a verifier would accept. + /// The decoded text, the padding blocks included, and RAM as the run finds it. + pub rv: rv::Program, + /// BLAKE2s over everything public and fixed: the stacked bytecode multilinear + /// (that table is 16·2^kbc words, tens of megabytes at production sizes, so it is + /// hashed once here rather than per proof), the entry and halt `pc`, and RAM's + /// size and initial words. Always set by [`Program::new`], so a `Program` cannot + /// carry a digest inconsistent with itself. + pub(crate) digest: [u8; 32], + /// The padding blocks in the text ([`filler`]), whose rows bring every table's + /// row count to a power of two. Prover-side only, and no program code reaches + /// them, so a missing or wrong entry costs the prover a run that does not fill + /// rather than anything a verifier would accept. pub filler: Vec, - /// Function pc-ranges `(name, entry, len)` from the compiler, for the - /// `DBG_PROF=1` per-function cycle profile ([`Program::execute`]). Purely - /// diagnostic; empty for hand-assembled programs. - pub fn_ranges: Vec<(String, u32, u32)>, - /// Source line of the statement that emitted each pc. Prover-side only, and - /// outside `bytecode_hash`, so it costs nothing in the proof: a failed guest - /// check reports a line rather than a pc to disassemble around. Empty for a - /// hand-assembled program, and shorter than `prog`, which is padded. - pub src_lines: Vec, - /// The smallest stacked witness this program's proofs may commit to, as a - /// log2. Zero (the default) asks for nothing. - /// - /// A consumer can need a proof to be no smaller than some size even when the - /// run is: the recursion guest holds one WHIR opening arm per committed size - /// it was compiled for, and has none below the first. A run that falls short - /// buys the difference in fill rows ([`crate::cpu::filler`]) rather than in a - /// padded commitment, which keeps the committed size a function of the - /// announced table heights, so neither the verifier nor the guest needs a new - /// parameter to certify. Prover-side only. - pub min_log_committed: usize, } -/// The bytecode digest reinterprets the stacked table as bytes, which is its -/// `to_le_bytes` image only on a little-endian target. +/// The digest reinterprets tables of words as bytes, which is their `to_le_bytes` +/// image only on a little-endian target. const _: () = assert!(cfg!(target_endian = "little")); impl Program { - /// Assemble a [`Program`], computing its bytecode digest - /// from `prog`. The single funnel for construction, so the digest is always - /// consistent with the bytecode. - pub fn assemble(prog: Vec, hints: HashMap>, main_frame: u32) -> Self { - let bytecode_hash = { - let table = layout::bytecode_table(&prog); - // SAFETY: F64 is #[repr(transparent)] over u64, so the slice's byte image is - // exactly the concatenation of its `to_le_bytes` on little-endian targets. - let bytes: &[u8] = - unsafe { core::slice::from_raw_parts(table.as_ptr().cast::(), core::mem::size_of_val(&table[..])) }; - primitives::hash::Hasher::new().update(bytes).finalize() - }; - Self { - prog, - bytecode_hash, - hints, - main_frame, - witness: HashMap::new(), - filler: Vec::new(), - fn_ranges: Vec::new(), - src_lines: Vec::new(), - min_log_committed: 0, + /// The program of a guest's ELF executable ([`rv::Guest::from_elf`]). + pub fn from_elf(elf: &[u8]) -> Result { + let guest = rv::Guest::from_elf(elf)?; + // The loader's cap is on the text alone, and [`Self::new`] appends to it, so a + // text that only just fits the region would leave the padding blocks nowhere to + // go and panic there. Refuse it here, where a malformed file is still an error. + if !filler::text_fits(guest.text.len()) { + return Err(rv::ElfError("the text leaves no room for the padding blocks")); } + Ok(Self::new( + &guest.text, + guest.entry_pc, + guest.image, + guest.log_ram, + guest.log_advice, + )) } - /// Where `pc` came from: `"verify_sub (line 2204)"` when the compiler left a - /// line for it, the function name alone otherwise (a hand-assembled program, - /// a fill block, or padding). This is what a run-time failure reports, so - /// the reader gets a line instead of a pc to disassemble around. - pub fn site_at(&self, pc: u32) -> String { - match self.src_lines.get(pc as usize) { - Some(&line) if line != 0 => format!("{} (line {line})", self.fn_at(pc)), - _ => self.fn_at(pc).to_string(), - } - } - - /// Instructions before the pad to a power of two: the compiled functions, or - /// the whole bytecode of a hand-assembled program. - pub fn code_len(&self) -> usize { - match self.fn_ranges.as_slice() { - [] => self.prog.len(), - ranges => ranges.iter().map(|&(_, _, len)| len as usize).sum(), + /// A program from its text, whose first word sits at [`rv::TEXT_BASE`], where it + /// starts, RAM's first words, and `log2` of RAM's and of the advice's sizes in + /// words. An illegal word and then the padding blocks ([`filler`]) follow the text. + /// + /// Panics if an entry is malformed ([`rv::Entry::is_well_formed`]): the `x0` and + /// sink rules are semantics the proof system takes from the table as given, so + /// they are checked where a table enters, on the verifier's side too. + pub fn new(text: &[u32], entry_pc: u64, image: Vec, log_ram: usize, log_advice: usize) -> Self { + let mut text = text.to_vec(); + // A run falling off the program's own text must trap, not slide into a block. + text.push(0); + let filler = filler::append_blocks(&mut text); + let rv = rv::Program::new(&text, entry_pc, image, log_ram, log_advice); + assert!(rv.entries.iter().all(rv::Entry::is_well_formed), "a malformed entry"); + + let bytes = |words: &[u64]| -> Vec { words.iter().flat_map(|w| w.to_le_bytes()).collect() }; + let table = layout::bytecode_table(&rv); + // SAFETY: F64 is #[repr(transparent)] over u64, so the slice's byte image is + // exactly the concatenation of its `to_le_bytes` on little-endian targets. + let table_bytes: &[u8] = + unsafe { core::slice::from_raw_parts(table.as_ptr().cast::(), core::mem::size_of_val(&table[..])) }; + // Every variable-length part is length-framed, so the preimage parses one way. + let mut h = primitives::hash::Hasher::new(); + h.update(b"leanvm-rv64im-1"); + h.update(&bytes(&[table.len() as u64])); + h.update(table_bytes); + h.update(&bytes(&[ + rv.entry_pc, + rv.halt_pc(), + rv.log_ram as u64, + rv.log_advice as u64, + rv.image.len() as u64, + ])); + h.update(&bytes(&rv.image)); + Self { + digest: h.finalize(), + rv, + filler, } } - - /// The compiled function containing `pc`. [`Self::site_at`] wraps this with - /// the source line when one is known. - pub fn fn_at(&self, pc: u32) -> &str { - self.fn_ranges - .iter() - .find(|(_, entry, len)| pc >= *entry && pc < *entry + *len) - .map_or("", |(name, _, _)| name.as_str()) - } - - /// Supply the entries of witness stream `name`: one slice of values per - /// `hint_witness(dest, "name")` call, popped in order (the same symbol - /// may be hinted many times). Prover-side data: entirely unconstrained, - /// invisible to verification. - pub fn set_witness(&mut self, name: impl Into, entries: Vec>) { - self.witness.insert(name.into(), entries); - } - - /// Assemble a program directly from a fixed bytecode vector, starting at - /// `(pc, fp) = (0, 0)` with no allocation hints. Suitable for straight-line - /// programs that never change the frame pointer and touch only the first - /// `main_frame` memory cells (so the prover needs no nondeterministic frame - /// allocation). `prog.len()` must be a power of two with a never-executed - /// sentinel in its last slot: the run halts on reaching `g^{len-1}` (§sec:state). - #[cfg(test)] - pub fn from_bytecode(prog: Vec, main_frame: u32) -> Self { - Self::assemble(prog, HashMap::new(), main_frame) - } } /// The whole proof is the transcript: a scalar stream plus the PCS hint /// channels (see [`crate::transcript::Proof`]). pub use crate::transcript::Proof; -/// Why [`prove`] produced no proof. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum ProveError { - /// `log_inv_rate` is outside the range the WHIR configuration accepts (1, 2, 3 or 4). - InvalidRate { log_inv_rate: usize }, - /// The program has no execution on this input and advice: a failed `assert`, - /// a wild pointer, advice that does not fit, ... - Execution(ExecError), - /// The run read these cells before anything wrote them, so their values would be - /// the prover's ([`Execution::unconstrained_reads`]). - UnconstrainedReads(Vec), - /// The committed witness is `2^log_committed` words, outside the - /// `2^MIN_MU..=2^MAX_MU` every verifier accepts - WitnessOutOfRange { log_committed: usize }, -} - -impl std::fmt::Display for ProveError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::InvalidRate { log_inv_rate } => write!(f, "log_inv_rate {log_inv_rate} is not supported"), - Self::Execution(e) => write!(f, "{e}"), - Self::UnconstrainedReads(cells) => write!( - f, - "the program read {} cell(s) before anything wrote them, first at {:?}: a source read \ - of a cell nothing stores, or a store the lowering dropped or misplaced", - cells.len(), - &cells[..cells.len().min(8)] - ), - Self::WitnessOutOfRange { log_committed } => write!( - f, - "the committed witness would be 2^{log_committed} words, outside the verifiable 2^{}..=2^{}", - pcs::MIN_MU, - pcs::MAX_MU - ), - } - } -} - -impl std::error::Error for ProveError {} - #[derive(Clone, Debug, PartialEq, Eq)] pub enum CpuError { Bus(leaf::Error), @@ -355,10 +208,10 @@ pub enum CpuError { Open(pcs::Error), PublicInput, Transcript(crate::transcript::Error), - /// flock's BLAKE2s R1CS validity sub-proof failed to verify. (A missing or - /// malformed sub-proof surfaces as [`CpuError::Transcript`] when the shared - /// `stream`/`openings` fail to reconstruct or fully consume.) - Blake2s(flock::verifier::VerifyError), + /// A class's flock sub-proof failed to verify. (A missing or malformed one + /// surfaces as [`CpuError::Transcript`] when the shared `stream`/`openings` fail + /// to reconstruct or fully consume.) + Flock(flock::verifier::VerifyError), } /// Per side, which table (if any) owns each bus block, as `(table, column base)`. @@ -366,14 +219,14 @@ type BlockOwners = [Vec>; 3]; /// Each table's `(column base, committed column count)` in the global schema. type TableSpans = Vec<(usize, usize)>; -/// Blocks sourced from a table's height belong to it; the boundary, memory and -/// bytecode blocks belong to none and keep their own column claims at ζ. -fn block_owners(log_bytecode: usize, sides: [usize; 3]) -> BlockOwners { +/// Blocks sourced from a table's height belong to it; the boundary, register, memory, +/// bytecode and range blocks belong to none and keep their own column claims at ζ. +fn block_owners(sizes: Sizes, sides: [usize; 3]) -> BlockOwners { let sch = schema(); - let src = block_kappa_sources(log_bytecode); + let src = block_kappa_sources(sizes); let mut it = src .into_iter() - .map(|(source, _)| source.checked_sub(2).map(|t| (t, sch.base[t]))); + .map(|(source, _)| source.checked_sub(1).map(|t| (t, sch.base[t]))); sides.map(|n| it.by_ref().take(n).collect()) } @@ -381,10 +234,7 @@ fn block_owners(log_bytecode: usize, sides: [usize; 3]) -> BlockOwners { /// column span. Derived from the program and the announced layout alone, so prover /// and verifier build it identically. fn bus_wiring(program: &Program, l: &Layout) -> (BlockOwners, TableSpans) { - let owners = block_owners( - crate::log2_strict_usize(program.prog.len()), - [l.push.len(), l.pull.len(), l.count.len()], - ); + let owners = block_owners(Sizes::of(&program.rv), [l.push.len(), l.pull.len(), l.count.len()]); (owners, table_spans()) } @@ -407,16 +257,19 @@ fn table_spans() -> TableSpans { /// evaluated on the same values. The identities take the air's own `η`-range; the /// three forms take the shared powers at [`xi_form_base`], folded into the forms' /// coefficients once rather than multiplied onto every row's form value. -fn airs( - taus: &[usize; tables::N_TABLES], - forms: &[Vec; 3], - form_pows: [F192; 3], -) -> Vec> { +fn airs(taus: &[usize; tables::N_TABLES], forms: &[Vec; 3], xi: F192) -> Vec> { + let form_pows = xi_form_pows(xi); + // Each table's slice of the batch's `η`-powers, exactly as `constraints` cuts + // them, turned into the weights its identities want once rather than per row. + let pows = primitives::field::powers(xi, xi_form_base()); + let offsets = constraints::xi_offsets(tables::tables().iter().map(|t| t.n_constraints())); tables::tables() .iter() .zip(taus) .enumerate() .map(|(t, (&table, &tau))| { + let weights = table.constraint_weights(&pows[offsets[t]..offsets[t] + table.n_constraints()]); + let weights_k = weights.clone(); // One form, not three: the batch adds the three sides' evaluations // anyway, and summing them here is a setup cost against a dot product // and a product list per row per node. @@ -426,14 +279,14 @@ fn airs( tau, n_cols: table.n_committed_columns(), n_constraints: table.n_constraints(), - eval: Box::new(move |p, vals, quadratic| { - let air = ::lift(table.eval_constraint(p, vals, quadratic)); + eval: Box::new(move |_, vals, quadratic| { + let air = ::lift(table.eval_constraint(&weights, vals, quadratic)); ::reduce(air ^ bus.eval_unreduced(vals, quadratic)) }), // The same expression over K columns: the identity's K-only products // stay 64-bit and the bus form becomes a mixed dot product. - eval_k: Box::new(move |p, vals, quadratic| { - let air = ::lift(table.eval_constraint_k(p, vals, quadratic)); + eval_k: Box::new(move |_, vals, quadratic| { + let air = ::lift(table.eval_constraint_k(&weights_k, vals, quadratic)); ::reduce(air ^ bus_k.eval_unreduced(vals, quadratic)) }), } @@ -470,21 +323,24 @@ fn xi_form_pows(xi: F192) -> [F192; 3] { [pows[base], pows[base + 1], pows[base + 2]] } -/// If `col` is a BLAKE2s **value** column (global index), its `q_flock` packed slot. -/// These columns are virtual (uncommitted): their memory-bus evaluation claims -/// are re-routed to `q_flock` slot evaluations, which is the whole binding: the -/// bus-tied value IS the proven `q_flock` word, no separate check needed. -fn blake2s_value_slot(col: usize) -> Option { - let base = schema().base[tables::BLAKE2S_TABLE]; - tables::BLAKE2S_VALUE_COLS - .iter() - .position(|&c| base + c == col) - .map(|i| crate::hash_flock::SLOTS[i]) +/// If `col` is a circuit word's column (global index), where the word lives: the +/// class's committed packed witness, the word's place within an instance (its port), +/// and the stride between instances. These columns are virtual (uncommitted): their +/// bus evaluation claims are re-routed to slot evaluations of that witness, which is +/// the whole binding: the bus-tied value IS the flock-proven word, no separate check +/// needed. +fn flock_value_slot(col: usize) -> Option<(usize, usize, usize)> { + let sch = schema(); + (0..tables::N_TABLES).find_map(|t| { + let (port, _) = tables::word_columns(t) + .into_iter() + .find(|&(_, c)| sch.base[t] + c == col)?; + Some((q_column(t), port, crate::class_flock::stride_log(tables::CLASSES[t]))) + }) } /// Run statistics returned alongside the proof: the cycle count (total executed -/// instructions), the per-opcode counts -/// `[XOR, MUL, SET, DEREF, JUMP, BLAKE2s]`, and the +/// instructions), the per-table counts in [`tables::CLASSES`] order, and the /// committed witness size, the sum of the column lengths, i.e. the real data /// before the stacked witness is zero-padded to a power of two `2^m`. pub struct Stats { @@ -496,37 +352,26 @@ pub struct Stats { /// What a cost measurement wants. pub base_counts: [usize; tables::N_TABLES], pub committed: usize, - /// Data memory is `2^log_mem` cells (the padded write-once image). - pub log_mem: usize, - /// Cells actually touched, before the pad to `2^log_mem`, i.e. the real memory - /// footprint (`log2` is fractional). - pub mem_used: usize, - /// Bytecode instructions, before the pad to a power of two ([`Program::code_len`]). - pub bytecode: usize, } impl Stats { - /// Table names in `counts` order. - pub const TABLES: [&'static str; tables::N_TABLES] = ["XOR", "MUL", "SET", "DEREF", "JUMP", "BLAKE2S"]; - - /// One line of per-table instruction counts and shares, largest first, followed by memory, bytecode and committed-witness sizes. + /// One line of per-table instruction counts and shares, largest first, followed by + /// the committed-witness size. /// /// The counts are `base_counts`, the work the program itself does, since the proven /// `counts` are all exact powers of two once the fill blocks have run (`filler`) and - /// so say nothing about the workload. - /// `log_mem` holds the padded memory size the commitment covers. Zero-count - /// tables are omitted. + /// so say nothing about the workload. Zero-count tables are omitted. #[must_use] pub fn details(&self) -> String { if self.cycles == 0 { return "-".to_string(); } let base_cycles: usize = self.base_counts.iter().sum(); - let mut shares: Vec<(&str, usize)> = Self::TABLES + let mut shares: Vec<(&str, usize)> = tables::CLASSES .iter() .zip(&self.base_counts) .filter(|&(_, &c)| c > 0) - .map(|(&name, &c)| (name, c)) + .map(|(spec, &c)| (spec.name, c)) .collect(); shares.sort_unstable_by_key(|&(_, c)| std::cmp::Reverse(c)); let mut parts: Vec = shares @@ -537,68 +382,72 @@ impl Stats { }) .collect(); let log2 = |n: usize| primitives::pretty_f64((n.max(1) as f64).log2()); - parts.push(format!("MEMORY 2^{}", log2(self.mem_used))); - parts.push(format!("BYTECODE 2^{}", log2(self.bytecode))); parts.push(format!("TOTAL_COMMITTED 2^{}", log2(self.committed))); parts.join(" ") } } -/// Prove the program on the given public input: run it (witness generation), -/// then emit everything the verifier needs through the returned [`Proof`] -/// (scalar stream + PCS commitment / opening hints). Returns the proof and the -/// run [`Stats`], or why no proof was made ([`ProveError`]). `log_inv_rate` -/// selects the PCS rate and is announced in the Fiat-Shamir transcript before -/// the commitment. +/// Prove a run of the program on `input`, RAM's first words, and `advice`, the advice +/// region's first words, which the statement says nothing about: execute it (witness +/// generation), then emit everything the verifier needs through the returned [`Proof`] +/// (scalar stream + PCS commitment / opening hints). Returns the proof, the run's public +/// output (`a0..a3` at the exit) and its [`Stats`], or the trap if the run has no proof. +/// `log_inv_rate` selects the PCS rate and is announced in the Fiat-Shamir transcript +/// before the commitment. +/// +/// Panics if `advice` holds more than the program's region does (`2^log_advice` words, +/// which the program fixes) or if `log_inv_rate` is not a rate the PCS supports: both +/// are the caller's to get right, like the program itself, and neither is a trap of the run. #[tracing::instrument(name = "Prove", skip_all, fields(log_inv_rate))] -pub fn prove(program: &Program, public_input: [F192; 2], log_inv_rate: usize) -> Result<(Proof, Stats), ProveError> { - if ::pcs::whir::validate_log_inv_rate(log_inv_rate).is_err() { - return Err(ProveError::InvalidRate { log_inv_rate }); - } +pub fn prove( + program: &Program, + input: [u64; rv::INPUT_WORDS], + advice: &[u64], + log_inv_rate: usize, +) -> Result<(Proof, [u64; 4], Stats), rv::Trap> { + ::pcs::whir::validate_log_inv_rate(log_inv_rate).expect("valid log_inv_rate"); // One proof is one arena phase: every transient buffer below is bump-allocated // and reclaimed wholesale here, rather than faulted in and unmapped again per // proof. Bound first so it outlives them; inert unless `init_prover` opted in. // The returned `Proof` is system-allocated (`ps.into_proof()` builds `Vec`s), // so it survives the next phase. let _phase = zk_alloc::enter_phase(); - let exec = crate::stage!("Execute program", || program.execute(public_input)).map_err(ProveError::Execution)?; - // A live value that came from outside the constraint system would make the proof - // about a weaker statement than the program text, so it is refused here, on the one - // path every proof takes, rather than left to whichever test happens to look. - if !exec.unconstrained_reads.is_empty() { - return Err(ProveError::UnconstrainedReads(exec.unconstrained_reads)); + let exec = crate::stage!("Execute program", || program.execute(input, advice))?; + let log_words = program.stack_log(exec.trace.row_counts()); + if log_words > pcs::MAX_MU { + return Err(rv::Trap::TooLong { log_words }); } + let (proof, stats) = prove_execution(program, &exec, &input, log_inv_rate); + Ok((proof, exec.output, stats)) +} + +/// [`prove`] from a finished run. Split out so a test can hand it a run no honest +/// machine produced. +fn prove_execution( + program: &Program, + exec: &Execution, + input: &[u64; rv::INPUT_WORDS], + log_inv_rate: usize, +) -> (Proof, Stats) { let cycles = exec.cycles; - let w = crate::stage!("Build witness", || program.build(&exec)); - if !(pcs::MIN_MU..=pcs::MAX_MU).contains(&w.layout.shape.mu) { - return Err(ProveError::WitnessOutOfRange { - log_committed: w.layout.shape.mu, - }); - } + let w = crate::stage!("Build witness", || program.build(exec, input)); let counts = w.layout.taus.map(|t| 1usize << t); let committed_size = w.committed_size(); - // The public statement (program digest + input) seeds the transcript, so - // every challenge depends on the exact program and public input. - debug_assert!( - public_input.iter().all(|h| h.c2 == 0), - "a public input is a 256-bit digest" - ); - let mut ps = ProverState::new(digest_words(&fs_seed(program)), digest_words(&public_input)); + // The public statement (program digest + output) seeds the transcript, so + // every challenge depends on the exact program and what the run claims to return. + let mut ps = ProverState::new(fs_seed(program, input), exec.output.map(F64)); // Announce the prover's sizes, then commit, before sampling any challenge. - announce_public(&mut ps, w.log_mem, w.layout.taus, log_inv_rate); + announce_public(&mut ps, w.layout.taus, log_inv_rate, w.ts_final); let committed = crate::stage!("Commit", || { pcs::commit(&mut ps, &w.q, w.layout.shape, log_inv_rate) }); - // BLAKE2s to flock (§hash_flock), single PCS: q_flock is ALWAYS a column in - // `w.q` (≥1 instance, a program with no BLAKE2s carries one padding instance, - // so the proof shape is uniform and there is no has/hasn't-BLAKE2s fork). flock's - // R1CS validity and EVERY leanVM point claim are discharged together by ONE - // WHIR over this commitment (below). Message, chaining-value, and output words - // bind through the memory bus; counter and flags bind through bytecode. Their - // virtual value columns route to q_flock, so no separate pin claims are needed. - // Mirrored in `verify`. + // Single PCS: every class's packed flock witness is a column of `w.q`, so flock's + // R1CS validity and EVERY leanVM point claim are discharged together by ONE WHIR + // over this commitment (below). A circuit's words bind through the register and + // bytecode buses: their virtual columns route to that witness, so no separate pin + // claims are needed. Mirrored in `verify`. let (owners, spans) = bus_wiring(program, &w.layout); // The columns are windows into `w.q`, so both stages read them in place: the // table sumcheck lifts each K-column into a fresh `E` copy on the round it @@ -610,7 +459,7 @@ pub fn prove(program: &Program, public_input: [F192; 2], log_inv_rate: usize) -> leaf::prove_balance(&l.push, &l.pull, &l.count, &cols, &owners, &spans, &mut ps) }); let table_claims = crate::stage!("Prove constraints", || { - // One sumcheck for all six tables (§constraints). + // One sumcheck for all the tables (§constraints). let table_cols: Vec> = spans .iter() .map(|&(base, n)| (0..n).map(|c| cols[base + c]).collect()) @@ -621,7 +470,7 @@ pub fn prove(program: &Program, public_input: [F192; 2], log_inv_rate: usize) -> let form_pows = xi_form_pows(xi); let sigma = sigmas(&bus.sigmas, form_pows); constraints::prove( - &airs(&l.taus, &bus.forms, form_pows), + &airs(&l.taus, &bus.forms, xi), &table_cols, xi, &bus.point, @@ -633,60 +482,46 @@ pub fn prove(program: &Program, public_input: [F192; 2], log_inv_rate: usize) -> }; let l = &w.layout; - // The PI binding transmits the two LOW memory limbs' evaluations - // (§sec:e2e-pi); the verifier checks them against the public-input line at - // `r_pi`. The top limb of both public words is zero, so its evaluation is - // zero at every `r_pi` and rides no scalar. - let r_pi = ps.sample(); - let pi_limbs = [ - primitives::multilinear::interp_k(F64(l.pi[0].c0), F64(l.pi[1].c0), r_pi), - primitives::multilinear::interp_k(F64(l.pi[0].c1), F64(l.pi[1].c1), r_pi), - F192::ZERO, - ]; - for v in &pi_limbs[..2] { - ps.add_scalar(*v); - } - // Memory binds the message, chaining-value, and output words; bytecode binds - // the counter and flags. All corresponding value columns are virtual and route - // to q_flock through `slot_claims`. - let slots = finish_claims(l, bus.claims, &table_claims, r_pi, pi_limbs); + let slots = finish_claims(l, bus.claims, &table_claims, &exec.output); - // Run flock's reduction (zerocheck + lincheck) over the prepared native - // layouts retained from the fused q_flock build pass; it returns the - // validity claim on the committed `q_flock`, discharged by the PCS below in - // the SAME WHIR as every leanVM point claim (the point claims become the - // opener's `point_claims`). - let flock_reduction = w.flock_reduction; - let reduced = crate::stage!("Flock reduction", || { flock_reduction.prove(&mut ps) }); - let n_blocks = flock_reduction.n_blocks(); - drop(flock_reduction); - let offset = w.layout.placements[QFLOCK].offset; - let ring = crate::hash_flock::ring_switch_open(n_blocks, offset, &reduced); - crate::stage!("PCS open", || { pcs::open(&mut ps, &committed, &w.q, &slots, &ring) }); - Ok(( + // Run each class's flock reduction (zerocheck + lincheck) over the native layouts + // retained from the witness build; it returns the validity claim on the class's + // committed packed witness, discharged by the PCS below in the SAME WHIR as every + // leanVM point claim, through a ring-switched region of its own. + let reductions = w.reductions; + let rings: Vec<_> = crate::stage!("Flock reductions", || { + reductions + .iter() + .enumerate() + .map(|(t, prepared)| { + let placement = &w.layout.placements[q_column(t)]; + let reduced = prepared.prove(&mut ps); + flock::reduction::ring_switch_open(placement.n_vars, placement.offset, &reduced) + }) + .collect() + }); + drop(reductions); + crate::stage!("PCS open", || { pcs::open(&mut ps, &committed, &w.q, &slots, &rings) }); + ( ps.into_proof(), Stats { cycles, counts, base_counts: exec.base_counts, committed: committed_size, - log_mem: w.log_mem, - mem_used: exec.mem_used, - bytecode: program.code_len(), }, - )) + ) } /// Everything the PCS has to open, in the ORDER that feeds the batch's weights: /// the bus's framework claims, then the zerocheck's per-table column claims, then -/// the three public-input limb claims, each located in its committed slot. Both -/// sides assemble it here, so a claim can never shift by one element. +/// the exit's claims, each located in its committed slot. Both sides assemble +/// it here, so a claim can never shift by one element. fn finish_claims( l: &Layout, bus_claims: Vec, table_claims: &[constraints::Claims], - r_pi: F192, - pi_limbs: [F192; 3], + output: &[u64; 4], ) -> Vec { let mut claims = bus_claims; let sch = schema(); @@ -700,63 +535,53 @@ fn finish_claims( }); } } - claims.extend(bind_pi_claim(r_pi, &l.placements, pi_limbs)); + // The exit (§sec:e2e-pi): the run halted on `exit`, returning `output`. + claims.push(final_register_claim(rv::SYSCALL_REG, rv::SYS_EXIT)); + for (reg, &value) in rv::OUTPUT_REGS.into_iter().zip(output) { + claims.push(final_register_claim(reg, value)); + } slot_claims(l, claims) } -/// The public-input binding (§sec:e2e-pi): the committed `MEM` at `(r, 0,…,0)` must -/// equal `interp(pi[0], pi[1], r)`, one transmitted evaluation per physical `K` -/// limb. The caller has already checked the three against the line; here they -/// simply become the three claims the opening discharges. `placements` comes from -/// the prover's or verifier's layout, so both sides build byte-identical claims. -fn bind_pi_claim(r: F192, placements: &[witness::Placement], limbs: [F192; 3]) -> [ColumnClaim; 3] { - let mut point = vec![F192::ZERO; placements[MEM_LO].n_vars]; - point[0] = r; - [MEM_LO, MEM_HI, MEM_TOP].map(|col| ColumnClaim { - col, - point: point.clone(), - value: limbs[col - MEM_LO], - }) +/// What register `reg` holds when the run ends: the committed final registers at the +/// Boolean point naming `reg`. Both parties know the value, so the claim is computed +/// rather than transmitted, and the opening discharges it like any other. +fn final_register_claim(reg: u8, value: u64) -> ColumnClaim { + ColumnClaim { + col: REG_FIN, + point: (0..rv::LOG_REGS) + .map(|bit| if (reg >> bit) & 1 == 1 { F192::ONE } else { F192::ZERO }) + .collect(), + value: F192::from(F64(value)), + } } -/// Everything a recursion harness needs from an accepting verify run, named -/// and typed: the deferred bytecode claim, flock's -/// reduction claims, and the stacked-opening summary (ring-switch challenges + -/// WHIR fold/query data). The sub-proof scalars themselves live on -/// `proof.stream`, ending at `flock_stream_end`. Ordinary callers just -/// `?`-discard it. -pub struct VerifySummary { - /// Transcript-bound inverse-rate logarithm used by this proof's PCS. - pub log_inv_rate: usize, - pub bytecode_claim: leaf::BytecodeClaim, - pub zc_claim: flock::zerocheck::ZerocheckClaim, - pub lc_claim: flock::lincheck::LincheckClaim, - /// Stream cursor just after flock's reduction, i.e. where the PCS opening's - /// own scalars start. The recursion harness reads flock's lincheck tail - /// from here rather than counting back from the end of the stream. - pub flock_stream_end: usize, - /// The proof this run just verified, in the unpruned form the recursion - /// guest and the Python verifier consume. - pub raw: fiat_shamir::transcript::RawProof, +/// Verify a proof against the public statement, that the program run on `input` exits +/// returning `output`: replay the transcript, reconstruct the public layout from the announced +/// sizes, read every scalar the prover wrote and pull the PCS hints, then assert the +/// stream was fully consumed. Takes only public inputs, never the prover's witness. +pub fn verify( + program: &Program, + input: &[u64; rv::INPUT_WORDS], + output: &[u64; 4], + proof: &Proof, +) -> Result<(), CpuError> { + verify_to_raw(program, input, output, proof).map(|_| ()) } -/// Verify a proof against the public statement (program + public input): replay -/// the transcript, reconstruct the public layout from the announced sizes, read -/// every scalar the prover wrote and pull the PCS hints, then assert the stream -/// was fully consumed. Takes only public inputs, never the prover's witness. +/// [`verify`], returning the proof it accepted with every query's Merkle path +/// written out, the form `python-verifier` reads. #[tracing::instrument(name = "Verify", skip_all)] -pub fn verify(program: &Program, public_input: &[F192; 2], proof: &Proof) -> Result { - let mut vs = VerifierState::new(digest_words(&fs_seed(program)), proof, digest_words(public_input)); - let (l, log_inv_rate) = read_public(&mut vs, program, public_input)?; +pub fn verify_to_raw( + program: &Program, + input: &[u64; rv::INPUT_WORDS], + output: &[u64; 4], + proof: &Proof, +) -> Result { + let mut vs = VerifierState::new(fs_seed(program, input), proof, output.map(F64)); + let (l, log_inv_rate) = read_public(&mut vs, program, input)?; let root = pcs::read_commitment(&mut vs).map_err(CpuError::Transcript)?; - // BLAKE2s to flock (single PCS): flock's R1CS validity and every leanVM point - // claim are verified together by ONE WHIR opening at the end. The padded - // BLAKE2s table size is public and announced; its flock sub-proof rides the - // shared stream and openings. Memory and bytecode bind every compression input - // and output by routing their virtual value-column claims to q_flock. - let n_blake2s = 1usize << l.taus[tables::BLAKE2S_TABLE]; - let (owners, spans) = bus_wiring(program, &l); let bus = leaf::verify_balance(&l.push, &l.pull, &l.count, &owners, &spans, &mut vs).map_err(CpuError::Bus)?; @@ -770,72 +595,53 @@ pub fn verify(program: &Program, public_input: &[F192; 2], proof: &Proof) -> Res // sides. A transmitted target would be a free value in its own check, and the // tables' bus blocks would be settled by nothing at all. let target = (0..3).fold(F192::ZERO, |a, s| a + form_pows[s] * bus.totals[s]); - let table_claims = constraints::verify( - &airs(&l.taus, &bus.forms, form_pows), - zc_xi, - &bus.point, - target, - &mut vs, - ) - .map_err(CpuError::Constraint)?; - - let r_pi = vs.sample(); - let mut pi_limbs = [F192::ZERO; 3]; - for v in &mut pi_limbs[..2] { - *v = vs.next_scalar().map_err(CpuError::Transcript)?; - } - // The two claimed evaluations must sit on the public-input line, the top - // limb's being zero (§sec:e2e-pi). - let want = primitives::multilinear::interp(l.pi[0], l.pi[1], r_pi); - if pi_limbs[0] + F192::Y * pi_limbs[1] != want { - return Err(CpuError::PublicInput); + let table_claims = constraints::verify(&airs(&l.taus, &bus.forms, zc_xi), zc_xi, &bus.point, target, &mut vs) + .map_err(CpuError::Constraint)?; + + let slots = finish_claims(&l, bus.claims, &table_claims, output); + + // Replay each class's flock reduction straight off the shared stream (each scalar + // bound as it is read) to recover its validity claim on the class's packed + // witness, then verify them alongside every point claim in the ONE WHIR opening + // (mirroring `prove`). + let mut replays = Vec::with_capacity(tables::N_TABLES); + for (t, &tau) in l.taus.iter().enumerate() { + replays.push(crate::class_flock::verify_reduction(t, tau, &mut vs).map_err(CpuError::Flock)?); } - let slots = finish_claims(&l, bus.claims, &table_claims, r_pi, pi_limbs); - - // Replay flock's reduction straight off the shared stream (each scalar bound - // as it is read) to recover its validity claim on q_flock, then - // verify them alongside every point claim in the ONE WHIR opening - // (mirroring `prove`). The padding convention always supplies at least one - // instance, including programs that execute no BLAKE2s instruction. - let n_blocks = n_blake2s.max(1); - let offset = l.placements[QFLOCK].offset; - let replay = crate::hash_flock::verify_reduction(n_blocks, &mut vs).map_err(CpuError::Blake2s)?; - let flock_stream_end = vs.stream_offset(); - let ring = crate::hash_flock::ring_switch_verify(n_blocks, offset, &replay.claim); - pcs::verify(&mut vs, &slots, &ring, l.shape, log_inv_rate, &root).map_err(CpuError::Open)?; + let rings: Vec<_> = replays + .iter() + .enumerate() + .map(|(t, replay)| { + let placement = &l.placements[q_column(t)]; + flock::reduction::ring_switch_verify(placement.n_vars, placement.offset, &replay.claim) + }) + .collect(); + pcs::verify(&mut vs, &slots, &rings, l.shape, log_inv_rate, &root).map_err(CpuError::Open)?; vs.finish().map_err(CpuError::Transcript)?; - Ok(VerifySummary { - bytecode_claim: bus.bytecode_claim, - zc_claim: replay.zc_claim, - lc_claim: replay.lc_claim, - log_inv_rate, - flock_stream_end, - raw: vs.into_raw_proof(), - }) + Ok(vs.into_raw_proof()) } /// Lift `ColumnClaim`s to located PCS claims: a claim on column `c` lives in /// the slot at `placements[c].offset`, with the claim's point as the low point. /// -/// BLAKE2s value columns are virtual: they have no committed placement. A bus -/// claim `value_col(r) = v` (at the `n_log`-dim instance point `r`) is re-routed -/// to the equal `q_flock` slot evaluation: an ordinary claim on the committed -/// `QFLOCK` column at the point freezing the low 8 coords to the slot's bits and -/// the high coords to `r`. No downstream special-casing: it folds into the -/// one opening like every other point claim. +/// A circuit word's column is virtual: it has no committed placement. A claim +/// `word_col(r) = v` (at the table's row point `r`) is re-routed to the equal slot +/// evaluation of the class's packed witness: an ordinary claim on that committed +/// column at the point freezing the low coords to the port's bits and the high +/// coords to `r`. No downstream special-casing: it folds into the one opening like +/// every other point claim. fn slot_claims(l: &Layout, claims: Vec) -> Vec { claims .into_iter() .map(|c| { - // A virtual BLAKE2s value column (always virtual): its bus claim at - // instance point `c.point` is the q_flock slot value, a boolean-selector - // (strided) claim on QFLOCK, folded sparsely (2^n_log, not the 2^(8+n_log) - // dense QFLOCK block). - if let Some(slot) = blake2s_value_slot(c.col) { + // A circuit word: its claim at the row point `c.point` is a slot value of + // the packed witness, a boolean-selector (strided) claim, folded sparsely + // (the table's height, not the dense witness block). + if let Some((q_col, slot, stride_log)) = flock_value_slot(c.col) { return pcs::SlotClaim::Strided { - offset: l.placements[QFLOCK].offset, + offset: l.placements[q_col].offset, slot, - stride_log: crate::hash_flock::SLOT_STRIDE_LOG, + stride_log, point: c.point, value: c.value, }; @@ -852,202 +658,122 @@ fn slot_claims(l: &Layout, claims: Vec) -> Vec { #[cfg(test)] mod tests { use super::*; - - /// A K-embedded immediate (both extension limbs zero). - fn w(x: u64) -> F192 { - F192::new(x, 0, 0) - } - - /// Pack two 64-bit flock words into the canonical BLAKE2s subspace of F192. - fn cell(lo: F64, hi: F64) -> F192 { - F192::new(lo.0, hi.0, 0) - } - - /// The default one-block-root metadata for a hand-built BLAKE2s op. - fn md() -> F192 { - crate::hash_flock::metadata(crate::hash_flock::PINNED_T, crate::hash_flock::FINAL_FLAG, 0) - } - - /// The four chaining-value lanes of the two cv cells. - fn cv_lanes(cv0: F192, cv1: F192) -> [F64; 4] { - [F64(cv0.c0), F64(cv0.c1), F64(cv1.c0), F64(cv1.c1)] - } - - /// A hand-built straight-line program with one BLAKE2s row: set up the two - /// 256-bit inputs (`a` at cells 2,3, `b` at cells 4,5, one 128-bit word per - /// cell) and the metadata (cell 8), hash them into the output `c` (cells 6,7), - /// pad with filler SETs so the last executed instruction lands one before the - /// sentinel, and halt there. The flock validity sub-proof plus the memory / - /// state / bytecode bus interactions are verified end-to-end (the proof - /// carries the WHIR opening they assert on). - fn blake2s_program(a: [F64; 4], b: [F64; 4]) -> Program { - // a → cells 2,3 and b → cells 4,5 (two flock lanes per BLAKE2s cell). - let mut prog = vec![ - Op::Set { - o: 2, - k: cell(a[0], a[1]), - }, - Op::Set { - o: 3, - k: cell(a[2], a[3]), - }, - Op::Set { - o: 4, - k: cell(b[0], b[1]), - }, - Op::Set { - o: 5, - k: cell(b[2], b[3]), - }, - Op::Set { o: 8, k: md() }, - // The chaining value reads cells 0,1 (the public input); any - // canonical cv is legal. - Op::Blake2s { - ins: [2, 3, 4, 5], - cv: 0, - out: 6, - md: 8, - }, - ]; // c → cells 6,7 - // 16 slots: 6 executed, then 9 filler SETs step the pc to 15, whose slot is - // the never-executed sentinel. - for k in 0..9u32 { - prog.push(Op::Set { - o: 16 + k, - k: F192::ONE, - }); + use crate::rv::asm::*; + use primitives::field::g_pow; + + const INPUT: [u64; 4] = [0; 4]; + + /// Reassign every range read's count, as a prover would after changing a gap, so + /// that the range arrays balance and what is left to judge is the registers. + fn recount_range_reads(exec: &mut Execution) { + let mask = (1u32 << tables::RANGE_LOG) - 1; + let mut lo = vec![F64::ONE; 1 << tables::RANGE_LOG]; + let mut hi = lo.clone(); + let t = &mut exec.trace; + let accesses = t.rows.iter_mut().enumerate().flat_map(|(table, rows)| { + let n = tables::CLASSES[table].n_accesses(); + rows.iter_mut().flat_map(move |r| r.accesses_mut()[..n].iter_mut()) + }); + for a in accesses { + let (l, h) = ((a.gap & mask) as usize, (a.gap >> tables::RANGE_LOG) as usize); + (a.count_lo, a.count_hi) = (lo[l], hi[h]); + lo[l] = primitives::field::mul_by_g(lo[l]); + hi[h] = primitives::field::mul_by_g(hi[h]); } - prog.push(Op::Xor { a: 0, b: 0, c: 0 }); // sentinel - assert_eq!(prog.len(), 16); - Program::from_bytecode(prog, 32) + (t.range_lo_count, t.range_hi_count) = (lo, hi); } - /// The opcode's execution semantics: the digest of the two message pairs under - /// the public input's chaining value lands in the output pair. Proving a program - /// is exercised from `lean_compiler`'s tests, which can compile one whose tables - /// come out powers of two. + /// An honest run's bus balances tuple by tuple, which says more than the proof + /// failing would: the blocks left unmatched are named. #[test] - fn blake2s_computes_the_compression() { - let a: [F64; 4] = [ - F64(0x0123_4567_89ab_cdef), - F64(0xfedc_ba98_7654_3210), - F64(0x1111_2222_3333_4444), - F64(0x5555_6666_7777_8888), - ]; - let b: [F64; 4] = [ - F64(0xdead_beef_cafe_babe), - F64(0x0badf00d_0badf00d), - F64(0x9999_aaaa_bbbb_cccc), - F64(0xdddd_eeee_ffff_0000), - ]; - let program = blake2s_program(a, b); - - let pi = [w(7), w(11)]; - let exec = program.execute(pi).unwrap(); - - // The output cells hold the compression of the two inputs under the - // pi-supplied chaining value (two 128-bit chunks). - let d = blake2s_compress(a, b, cv_lanes(pi[0], pi[1]), md()); - assert_eq!(exec.mem[6], cell(d[0], d[1])); - assert_eq!(exec.mem[7], cell(d[2], d[3])); + fn an_honest_run_balances() { + let text = Asm::new().i("addi", A0, ZERO, 5).exit().finish(); + let program = Program::new(&text, rv::TEXT_BASE, vec![], 2, 0); + let w = program.build(&program.execute(INPUT, &[]).unwrap(), &INPUT); + let unmatched = leaf::unmatched_leaves(&w.layout.push, &w.layout.pull, &w.columns()); + assert!( + unmatched.is_empty(), + "unmatched (side, block, row): {:?}", + &unmatched[..unmatched.len().min(12)] + ); } - /// BLAKE consumes the `(c0,c1,0)` embedding. This is not an extra AIR - /// constraint: the full three-limb memory bus makes a request carrying a - /// literal zero in limb 2 match only such a stored word. + /// A load cannot return what its cell does not hold. The forged run is consistent + /// everywhere else (the circuit's instance, the register written, the output), so + /// what is left unmatched is RAM's: the load pulls a tuple no store pushed. #[test] - fn blake2s_requires_zero_third_limb() { - let mut program = blake2s_program([F64::ZERO; 4], [F64::ZERO; 4]); - program.prog[0] = Op::Set { - o: 2, - k: F192::new(0, 0, 1), - }; - let err = program.execute([w(7), w(11)]).err().expect("the run must fail"); - assert!(matches!(err.fault, Fault::NotCanonical { operand: "m0", .. }), "{err}"); + fn a_forged_load_unbalances_the_bus() { + let text = Asm::new() + .li(T0, rv::RAM_BASE + 32) + .i("addi", T1, ZERO, 5) + .store("sd", T1, 0, T0) + .load("ld", A0, 0, T0) + .exit() + .finish(); + let program = Program::new(&text, rv::TEXT_BASE, vec![], 3, 0); + let mut forged = program.execute(INPUT, &[]).unwrap(); + assert_eq!(forged.output, [5, 0, 0, 0]); + let load = tables::table_of(rv::Class::Load).unwrap(); + let row = forged.trace.rows[load].iter_mut().find(|r| !r.ts.is_zero()).unwrap(); + (row.ram.old, row.ram.new, row.out) = (7, 7, 7); + forged.trace.reg_fin[A0 as usize] = F64(7); + forged.trace.ram_fin[4] = F64(7); + let w = program.build(&forged, &INPUT); + let unmatched = leaf::unmatched_leaves(&w.layout.push, &w.layout.pull, &w.columns()); + // The load's pull and the store's push, which it should have met. + assert_eq!(unmatched.len(), 2, "{unmatched:?}"); } - /// A self-hash `BLAKE2s(h, h)` (the hash-chain step) passes the *same* input - /// chunks as both `a` and `b` (`ins[0..2] == ins[2..4]`), so one 256-bit quad - /// feeds both inputs with no copy. The row reads those cells twice; the - /// running access counts thread through and the bus still balances. This is - /// the aliasing the DSL's hash-chain lowering relies on. + /// The point of the timestamps: a register written twice cannot be read as of its + /// first write. The forged run is consistent everywhere else (the stale value + /// flows into the result, the final registers and the range reads), so what fails + /// is the register multiset itself: the first write's tuple is pulled twice. #[test] - fn blake2s_self_hash_aliased_operands() { - let h: [F64; 4] = [ - F64(0xfeed_face_dead_beef), - F64(0x0123_4567_89ab_cdef), - F64(0xcafe_d00d_1337_c0de), - F64(0x8877_6655_4433_2211), - ]; - // a == b: hash h ‖ h into cells 4,5, both input operands aliasing one pair - let mut prog = vec![ - Op::Set { - o: 2, - k: cell(h[0], h[1]), - }, - Op::Set { - o: 3, - k: cell(h[2], h[3]), - }, - Op::Set { o: 6, k: md() }, - Op::Blake2s { - ins: [2, 3, 2, 3], - cv: 0, - out: 4, - md: 6, - }, - ]; - // 8 slots: 4 executed, 3 filler SETs stepping the pc, then the sentinel. - for k in 0..3u32 { - prog.push(Op::Set { - o: 12 + k, - k: F192::ONE, - }); - } - prog.push(Op::Xor { a: 0, b: 0, c: 0 }); // sentinel - assert_eq!(prog.len(), 8); - let program = Program::from_bytecode(prog, 16); - let pi = [w(3), w(5)]; - - let exec = program.execute(pi).unwrap(); - let d = blake2s_compress(h, h, cv_lanes(pi[0], pi[1]), md()); - assert_eq!(exec.mem[4], cell(d[0], d[1])); - assert_eq!(exec.mem[5], cell(d[2], d[3])); - } - - /// A 192-bit-word MUL: the E-product of two full machine words. Full-limb - /// constants are why this one is hand-written bytecode: a source literal fills - /// only the low two limbs. - #[test] - fn mul_192bit_word() { - let x = F192::new(0x0123_4567_89ab_cdef, 0xfeed_face_dead_beef, 0x1111_2222_3333_4444); - let y = F192::new(0x9999_aaaa_bbbb_cccc, 0x1357_9bdf_2468_ace0, 0x5555_6666_7777_8888); - let prog = vec![ - Op::Set { o: 2, k: x }, - Op::Set { o: 3, k: y }, - Op::Mul { a: 2, b: 3, c: 4 }, - Op::Xor { a: 0, b: 0, c: 0 }, // sentinel (never executed) - ]; - let program = Program::from_bytecode(prog, 5); - let pi = [w(1), w(2)]; - let exec = program.execute(pi).unwrap(); - assert_eq!(exec.mem[4], x * y, "MUL computes the E product"); - } - - /// The `XOR` reads `m[2]` before the `SET` writes it, so it reads ZERO, and its - /// row, filled from the final image, would contradict any other value there. - #[test] - fn a_write_must_agree_with_an_earlier_read() { - let prog = vec![ - Op::Xor { a: 2, b: 0, c: 3 }, - Op::Set { o: 2, k: F192::ONE }, - Op::Set { o: 4, k: F192::ZERO }, - Op::Xor { a: 0, b: 0, c: 0 }, // sentinel (never executed) - ]; - let err = Program::from_bytecode(prog, 8) - .execute([F192::ZERO; 2]) - .err() - .expect("the run must fail"); - assert!(matches!(err.fault, Fault::Conflict { cell: 2, .. }), "{err}"); + fn a_stale_read_unbalances_the_bus() { + const RATE: usize = pcs::TEST_LOG_INV_RATE; + let text = Asm::new() + .i("addi", T0, ZERO, 5) + .i("addi", T0, ZERO, 9) + .i("addi", T1, ZERO, 3) + .r("add", A0, T0, T1) + .exit() + .finish(); + let program = Program::new(&text, rv::TEXT_BASE, vec![], 2, 0); + let honest = program.execute(INPUT, &[]).unwrap(); + assert_eq!(honest.output, [12, 0, 0, 0]); + let (proof, _) = prove_execution(&program, &honest, &INPUT, RATE); + verify(&program, &INPUT, &honest.output, &proof).expect("the honest run verifies"); + + let mut forged = program.execute(INPUT, &[]).unwrap(); + let row = &mut forged.trace.rows[0][3]; + // The read happens at cycle 4; the first write happened at cycle 1, the second at 2. + let (stride, write_slot) = (tables::CLOCK_STRIDE, tables::REG_SLOTS[2]); + assert_eq!( + (row.v1, row.acc[0].x, row.acc[0].gap), + ( + 9, + g_pow((2 * stride + write_slot) as usize), + 2 * stride - write_slot - 1 + ) + ); + (row.v1, row.out) = (5, 8); + (row.acc[0].x, row.acc[0].gap) = (g_pow((stride + write_slot) as usize), 3 * stride - write_slot - 1); + forged.output[0] = 8; + forged.trace.reg_fin[A0 as usize] = F64(8); + recount_range_reads(&mut forged); + let refused = std::panic::catch_unwind(|| prove_execution(&program, &forged, &INPUT, RATE).0) + .expect_err("a stale read was proven"); + let message = refused.downcast_ref::().map(String::as_str).unwrap_or(""); + assert!( + message.contains("two products to agree"), + "refused for another reason: {message}" + ); + + // The same forgery with an honest read is a no-op: recounting alone changes + // no product, so the refusal above is the stale read's. + let mut recounted = program.execute(INPUT, &[]).unwrap(); + recount_range_reads(&mut recounted); + let (proof, _) = prove_execution(&program, &recounted, &INPUT, RATE); + verify(&program, &INPUT, &recounted.output, &proof).expect("recounting is harmless"); } } diff --git a/crates/lean_vm/src/cpu/trace.rs b/crates/lean_vm/src/cpu/trace.rs index d83ca703f..b93bf334b 100644 --- a/crates/lean_vm/src/cpu/trace.rs +++ b/crates/lean_vm/src/cpu/trace.rs @@ -1,85 +1,108 @@ -//! Per-opcode trace rows, emitted during execution and assembled into a [`Trace`]. +//! The trace: one [`Row`] per executed instruction, emitted during execution and +//! grouped by table, plus what the run leaves behind for the finalize blocks. //! -//! A row carries only what the witness fill cannot recover: the step's -//! `(pc, fp)`, the access counts read at access time, and `DEREF`'s resolved -//! store index. Everything else is a function of those: -//! operands and immediates come from `prog[pc]`, addresses from `fp` plus those -//! operands, and values from the final memory image, which is write-once and so -//! still holds what each accessed cell held at the time it was accessed. The -//! rows are the interpreter's largest write stream, so what they do not carry -//! they do not pay for, twice: once writing them and once reading them back. +//! A row carries what its accesses saw, its clock, and per access what the memory +//! argument needs ([`Access`]). Everything else comes back from the program's entry +//! at `index`. use primitives::field::F64; -/// `XOR` or `MUL` row: the three cells are `fp·g^{a,b,c}`. -pub(crate) struct Xrow { - pub(crate) pc: u32, - pub(crate) fp: u32, // frame base: address = fp + offset, operand = g^offset - pub(crate) ra: F64, - pub(crate) rb: F64, - pub(crate) rc: F64, - pub(crate) bytecode_read: F64, +/// One access to a read-write array, as the memory argument sees it (§sec:memchan). +#[derive(Clone, Copy)] +pub(crate) struct Access { + /// `g^x`, the timestamp of the cell's previous access. Zero on a padding row. + pub(crate) x: F64, + /// `y - x - 1` for this access's timestamp `y`: what the two range reads certify + /// to be below `2^32`. + pub(crate) gap: u32, + /// The read counts of the two range-array entries the gap's chunks name. + pub(crate) count_lo: F64, + pub(crate) count_hi: F64, } -pub(crate) struct Srow { - pub(crate) pc: u32, - pub(crate) fp: u32, - pub(crate) r: F64, - pub(crate) bytecode_read: F64, + +/// What a hash row adds to a [`Row`]: its block's words as found and the four it +/// writes ([`crate::rv::hash`]), and every one of its accesses, the registers' first. +pub(crate) struct HashRow { + pub(crate) block: [u64; crate::rv::hash::WORDS], + pub(crate) out: [u64; 4], + pub(crate) acc: [Access; 2 + crate::rv::hash::WORDS], } -pub(crate) struct Drow { - pub(crate) pc: u32, - pub(crate) fp: u32, - pub(crate) r1: F64, - pub(crate) r2: F64, - pub(crate) r3: F64, - pub(crate) bytecode_read: F64, + +impl HashRow { + /// Word `k` of the block after the row. + pub(crate) fn word_after(&self, k: usize) -> u64 { + match k.wrapping_sub(crate::rv::hash::OUT as usize / 8) { + j if j < 4 => self.out[j], + _ => self.block[k], + } + } } -pub(crate) struct Jrow { - pub(crate) pc: u32, - pub(crate) fp: u32, - pub(crate) rc: F64, - pub(crate) rd: F64, - pub(crate) rf: F64, + +pub(crate) struct Row { + /// The entry executed. + pub(crate) index: u32, + /// The row's clock, zero on a padding row. + pub(crate) ts: F64, + pub(crate) v1: u64, + pub(crate) v2: u64, + /// What the class computed. + pub(crate) out: u64, + pub(crate) taken: bool, + /// What the destination register held before the write. + pub(crate) vd_old: u64, + /// The RAM cell a load or a store accessed: its bus address, what it held and + /// what it holds. Zeros for another class. + pub(crate) ram: crate::rv::machine::RamAccess, + /// `rs1`, `rs2`, `rd`, then the RAM access if the class has one. A hash row + /// keeps its accesses in `hash` instead. + pub(crate) acc: [Access; 4], + pub(crate) hash: Option>, pub(crate) bytecode_read: F64, } -/// `BLAKE2s` row: the nine per-cell memory access counts of the four -/// message-chunk cells, the chaining value's two cells, the output's two and the -/// metadata cell. The addresses are `fp·g^{ins[i]}`, `fp·g^{cv}`, `fp·g^{out}`, -/// `fp·g^{md}` and the successors of the middle two; the eighteen flock words are -/// those cells' lanes. -pub(crate) struct Brow { - pub(crate) pc: u32, - pub(crate) fp: u32, - pub(crate) ra: [F64; 2], // per-cell counts for the two a input cells - pub(crate) rb: [F64; 2], // … the two b input cells - pub(crate) rcv: [F64; 2], // … the two cv input cells - pub(crate) rc: [F64; 2], // … the two c output cells - pub(crate) rmd: F64, // … and the metadata cell - pub(crate) bytecode_read: F64, +impl Row { + /// The row's accesses in column order, at least the class's + /// [`n_accesses`](crate::tables::ClassSpec::n_accesses). + pub(crate) fn accesses(&self) -> &[Access] { + match &self.hash { + Some(hash) => &hash.acc, + None => &self.acc, + } + } + + #[cfg(test)] + pub(crate) fn accesses_mut(&mut self) -> &mut [Access] { + match &mut self.hash { + Some(hash) => &mut hash.acc, + None => &mut self.acc, + } + } } pub(crate) struct Trace { - pub(crate) xor: Vec, - pub(crate) mul: Vec, - pub(crate) set: Vec, - pub(crate) deref: Vec, - pub(crate) jump: Vec, - pub(crate) blake2s: Vec, - pub(crate) mem_count: Vec, // per-cell running access count g^{count}; final = g^{A[i]} + /// Per table, in [`crate::tables::CLASSES`] order. + pub(crate) rows: [Vec; crate::tables::N_TABLES], + /// The registers after the run, and each one's last timestamp `g^y`, `g^0` if + /// never touched. + pub(crate) reg_fin: Vec, + pub(crate) reg_ts: Vec, + /// The same for RAM, and for the advice, whose initial words are committed too. + pub(crate) ram_fin: Vec, + pub(crate) ram_ts: Vec, + pub(crate) adv_init: Vec, + pub(crate) adv_fin: Vec, + pub(crate) adv_ts: Vec, pub(crate) bytecode_count: Vec, // per-pc running execution count g^{count}; final = g^{A[pc]} + /// Final read counts of the two range arrays' entries. + pub(crate) range_lo_count: Vec, + pub(crate) range_hi_count: Vec, + /// The clock `g^{4·cycle}` the run ended on: the final state's timestamp. + pub(crate) ts_final: F64, } impl Trace { - /// Rows per instruction table, in [`crate::cpu::Stats::TABLES`] order. + /// Rows per instruction table. pub(crate) fn row_counts(&self) -> [usize; crate::tables::N_TABLES] { - [ - self.xor.len(), - self.mul.len(), - self.set.len(), - self.deref.len(), - self.jump.len(), - self.blake2s.len(), - ] + std::array::from_fn(|t| self.rows[t].len()) } } diff --git a/crates/lean_vm/src/hash_flock.rs b/crates/lean_vm/src/hash_flock.rs deleted file mode 100644 index 6c2b4ab98..000000000 --- a/crates/lean_vm/src/hash_flock.rs +++ /dev/null @@ -1,450 +0,0 @@ -//! Bridge to the flock BLAKE2s prover ([`flock::hash`]), single-PCS. -//! -//! `q_flock` (flock's packed BLAKE2s witness, 64 bits per `F64` word) is committed -//! as a column in leanVM's ONE stacked `F64` witness (§sec:stacking), with no separate flock -//! commitment. The VM's `BLAKE2s` table binds to it by point-eval equality (its -//! value columns and `q_flock`'s slots are point-evals of the same committed -//! stack), and flock's R1CS validity is discharged by the same stacked WHIR: -//! the reduction's two tower-field claims pass through -//! [`ring_switch_open`] / [`ring_switch_verify`] and join the batch-mixed -//! opening ([`::pcs::stack_open`]). -//! -//! ## The mapping -//! -//! The VM's `BLAKE2s(a, b, cv, metadata) -> c` is one standard BLAKE2s -//! compression. `metadata` packs `counter:u64 | f0:u32 | f1:u32` in -//! little-endian order. All inputs are witness values in `q_flock`, and the -//! memory interaction binds every one of them: `a`, `b`, `cv` and `metadata` -//! are cells the instruction reads. -//! -//! ## The layout (aligned re-layout, `MSG_BASE = 640`, 64-bit words) -//! -//! Each compression's `2^K_LOG` bits pack into `2^(K_LOG-6)` `F64` words (the -//! [`SLOT_STRIDE_LOG`] stride); each VM-visible 64-bit word is one whole packed -//! word at a fixed within-instance slot (bit position / 64): -//! -//! ```text -//! c0..c3 = slots 4..8 a0..a3 = slots 10..14 b0..b3 = slots 14..18 -//! cv0..cv3 = slots 0..4 counter = slot 18 f0‖f1 = slot 19 -//! ``` -//! -//! All compression inputs are free witness rows; the VM routes claims on all -//! eighteen aligned words directly to these slots. - -use crate::transcript::{ProverState, VerifierState}; -use ::pcs::pack::LOG_PACKING; -use flock::hash::{ - Blake2sSetup, Compression, K_LOG, ReductionReplay, blake2s_compress, generate_witness_with_ab_packed_and_lincheck, -}; -use flock::verifier::VerifyError; -use primitives::field::{F64, F192}; -use primitives::stream::Stream; -use zk_alloc::ArenaVec; - -pub use flock::hash::{ - SliceClaim, min_n_blocks_log as n_blocks_log, qflock_kappa, ring_switch_open, ring_switch_verify, -}; - -/// BLAKE2s's final-block flag `f0`, which RFC 7693 sets to all ones on the last -/// block. A hash-shaped compression is one 64-byte final block, so it is always -/// set there. -pub const FINAL_FLAG: u32 = flock::hash::PINNED_F0; -/// The byte counter of the one 64-byte block a hash-shaped compression absorbs. -pub const PINNED_T: u64 = flock::hash::PINNED_T; - -/// Flock-native reduction buffers emitted in the same fused pass as the -/// committed, flattened `q_flock`. They stay alive across commit, bus, and -/// constraint proving so reduction needs no second witness pass. -pub(crate) struct PreparedReductionWitness { - n_blocks: usize, - z_packed: ArenaVec, - a_packed: ArenaVec, - b_packed: ArenaVec, - z_lincheck: ArenaVec, -} - -impl PreparedReductionWitness { - pub(crate) fn n_blocks(&self) -> usize { - self.n_blocks - } - - pub(crate) fn prove(&self, ps: &mut ProverState) -> SliceClaim { - Blake2sSetup::new(self.n_blocks).prove_reduction_precomputed( - &self.z_packed, - &self.a_packed, - &self.b_packed, - &self.z_lincheck, - ps, - ) - } -} - -// Within-instance packed-word (slot) indices of the VM-visible words, fixed by -// the aligned flock layout (bit bases asserted by `layout_constants` there): -// `CV_BASE = 0` → cv words 0..4, `OUT_BASE = 256` → c words 4..8, `MSG_BASE -// = 640` → a words 10..14 and b words 14..18, metadata (counter, f0‖f1) -// words 18..20. -pub const SLOT_CV0: usize = 0; -pub const SLOT_C0: usize = 4; -pub const SLOT_A0: usize = 10; -pub const SLOT_B0: usize = 14; -pub const SLOT_METADATA: usize = 18; - -/// The eighteen within-instance value slots in canonical order -/// `[a0..a3, b0..b3, c0..c3, cv0..cv3, md_lo, md_hi]`, matching -/// `tables::BLAKE2S_VALUE_COLS`. -pub const SLOTS: [usize; 18] = [ - SLOT_A0, - SLOT_A0 + 1, - SLOT_A0 + 2, - SLOT_A0 + 3, - SLOT_B0, - SLOT_B0 + 1, - SLOT_B0 + 2, - SLOT_B0 + 3, - SLOT_C0, - SLOT_C0 + 1, - SLOT_C0 + 2, - SLOT_C0 + 3, - SLOT_CV0, - SLOT_CV0 + 1, - SLOT_CV0 + 2, - SLOT_CV0 + 3, - SLOT_METADATA, - SLOT_METADATA + 1, -]; - -/// Split a 64-bit field element into the two little-endian `u32` words flock's -/// message uses (the VM memory byte order). -fn words_of(x: F64) -> [u32; 2] { - [x.0 as u32, (x.0 >> 32) as u32] -} - -/// Inverse of `words_of`: pack two little-endian `u32` words into the `F64`. -fn pack_words(w: [u32; 2]) -> F64 { - F64((w[0] as u64) | ((w[1] as u64) << 32)) -} - -/// Pack BLAKE2s's compression metadata as one little-endian 128-bit value in -/// the two low K-lanes of a 192-bit word (top lane zero). -pub const fn metadata(counter: u64, f0: u32, f1: u32) -> F192 { - F192::new(counter, (f0 as u64) | ((f1 as u64) << 32), 0) -} - -/// Unpack `counter:u64 | f0:u32 | f1:u32` from a 192-bit word (the top lane -/// must be zero). -pub const fn unpack_metadata(x: F192) -> (u64, u32, u32) { - assert!(x.c2 == 0, "BLAKE2s metadata must have a zero top lane"); - (x.c0, x.c1 as u32, (x.c1 >> 32) as u32) -} - -/// BLAKE2s-256's initial chaining value as four flock words (the two -/// chaining-value cells' low lanes, in canonical lane order). This is the IV -/// with the parameter block (digest length 32, unkeyed, fanout and depth 1) -/// folded into word 0, which is what makes a 64-byte compression equal -/// `blake2s` of those 64 bytes. -/// -/// Derived from [`flock::hash::param_iv`] rather than written out: the -/// parameter block touches three bytes of word 0, and a hand-copied constant -/// that XORs only the low byte still looks plausible. -pub const IV: [F64; 4] = { - const fn w(lo: u32, hi: u32) -> F64 { - F64((lo as u64) | ((hi as u64) << 32)) - } - let h = flock::hash::param_iv(); - [w(h[0], h[1]), w(h[2], h[3]), w(h[4], h[5]), w(h[6], h[7])] -}; - -/// The initial chaining value as the two 192-bit VM memory cells a chaining -/// value occupies (canonical 128-bit chunks, top limbs zero). -pub const IV_CELLS: [F192; 2] = [F192::new(IV[0].0, IV[1].0, 0), F192::new(IV[2].0, IV[3].0, 0)]; - -/// The flock [`Compression`] for one VM instruction. -pub fn compression(a: [F64; 4], b: [F64; 4], cv: [F64; 4], meta: F192) -> Compression { - let mut m = [0u32; 16]; - for (i, &w) in a.iter().enumerate() { - m[2 * i..2 * i + 2].copy_from_slice(&words_of(w)); - } - for (i, &w) in b.iter().enumerate() { - m[8 + 2 * i..8 + 2 * i + 2].copy_from_slice(&words_of(w)); - } - let mut cv_words = [0u32; 8]; - for (i, &w) in cv.iter().enumerate() { - cv_words[2 * i..2 * i + 2].copy_from_slice(&words_of(w)); - } - let (counter, f0, f1) = unpack_metadata(meta); - (cv_words, m, counter, f0, f1) -} - -/// The 256-bit output chaining value `c = (c0..c3)` of an arbitrary -/// compression. This is `blake2s(a‖b)` only for the parameterized [`IV`] and -/// one-block final metadata (counter 64, `f0` set). -pub fn digest(block: &Compression) -> [F64; 4] { - let h = blake2s_compress(&block.0, &block.1, block.2, block.3, block.4); - std::array::from_fn(|k| pack_words([h[2 * k], h[2 * k + 1]])) -} - -/// Lift flock's packed witness (64 bits per word, bit `i` at position `i`) into -/// the committed `F64` column: word for word, which is exactly `pcs::pack`'s -/// LSB-first convention on the same bit string. -fn flatten_packed_into(packed: &[u64], out: &mut [F64]) { - assert_eq!(out.len(), packed.len(), "q_flock's window is the wrong size"); - // Write directly into the committed window and publish it with streaming stores. - // SAFETY: `F64` is `repr(transparent)` over `u64`, so the two slices are the - // same bytes. - let words: &mut [u64] = unsafe { std::slice::from_raw_parts_mut(out.as_mut_ptr().cast(), out.len()) }; - parallel::chunks_mut_zip(words, packed, 1 << 14, |_, dst, src| Stream::new().copy(dst, src)); -} - -/// Build the committed `q_flock` column (flock's packed witness) for `blocks`, padded -/// to `2^n_blocks_log(max(blocks.len(),1))` instances (the unused ones flock's own -/// `padding_block`), and retain the Flock-native layouts produced by that same fused pass so -/// reduction does not regenerate them later. Deterministic, so it matches what the -/// reduction regenerates. An empty `blocks` yields one padding cube (all instances are -/// padding). -pub(crate) fn build_qflock_prepared(blocks: &[Compression], q_flock: &mut [F64]) -> PreparedReductionWitness { - let n_blocks = blocks.len().max(1); - let (z_packed, a_packed, b_packed, z_lincheck) = - generate_witness_with_ab_packed_and_lincheck(blocks, n_blocks_log(n_blocks)); - flatten_packed_into(&z_packed, q_flock); - PreparedReductionWitness { - n_blocks, - z_packed, - a_packed, - b_packed, - z_lincheck, - } -} - -/// `log2` of the within-instance packed span (`2^8` words): the -/// number of low coords of a `q_flock` point that carry the slot's bits, and the -/// stride between consecutive instances' same-slot words in `q_flock`. A value -/// claim on `q_flock` is thus a boolean-selector (strided) claim with this stride. -pub const SLOT_STRIDE_LOG: usize = K_LOG - LOG_PACKING; - -/// **Flock reduction only** (prover): run flock's BLAKE2s zerocheck + lincheck -/// over `blocks` and return the one [`SliceClaim`] on the committed witness -/// `q_flock`, along with the regenerated packed witness (already flattened to -/// the committed `F64` packing). The sub-proof scalars ride the shared -/// transcript stream (`ps.add_scalar` at the protocol points); flock runs -/// natively in the tower field on the shared transcript. Does NOT open the PCS: the -/// caller discharges the returned claim via [`crate::pcs::open`] (as -/// [`crate::cpu`]'s prove does). -#[cfg(test)] -fn prove_reduction(blocks: &[Compression], ps: &mut ProverState) -> (Vec, SliceClaim) { - let setup = Blake2sSetup::new(blocks.len()); - let (z_packed, a_packed, b_packed, z_lincheck) = - generate_witness_with_ab_packed_and_lincheck(blocks, setup.n_blocks_log()); - let reduced = setup.prove_reduction_precomputed(&z_packed, &a_packed, &b_packed, &z_lincheck, ps); - let mut q_flock = vec![F64::ZERO; z_packed.len()]; - flatten_packed_into(&z_packed, &mut q_flock); - (q_flock, reduced) -} - -/// `q_flock` on its own, for the tests that only need the committed column. -#[cfg(test)] -fn build_qflock(blocks: &[Compression]) -> Vec { - let mut q_flock = vec![F64::ZERO; 1 << qflock_kappa(blocks.len())]; - build_qflock_prepared(blocks, &mut q_flock); - q_flock -} - -/// **Flock reduction only** (verifier): mirror of `prove_reduction`. Replay -/// the zerocheck + lincheck sub-proofs straight off the shared stream (each -/// scalar bound as it is read), and recover the one claim on `q_flock` -/// for the PCS to discharge, plus the reassembled reduction claims -/// ([`ReductionReplay`]). The statement is already bound (the seed, the announced -/// sizes, and the commitment root on the stream), so nothing else enters here. -pub fn verify_reduction(n_blocks: usize, vs: &mut VerifierState) -> Result { - Blake2sSetup::new(n_blocks).verify_reduction(vs) -} - -#[cfg(test)] -mod tests { - use super::*; - - fn f(x: u64) -> F64 { - F64(x) - } - - fn sample_blocks(n: usize) -> Vec { - (0..n as u64) - .map(|i| { - compression( - [ - f(0x11 * (i + 1)), - f(0x22 * (i + 1)), - f(0x33 * (i + 1)), - f(0x44 * (i + 1)), - ], - [ - f(0x55 * (i + 1)), - f(0x66 * (i + 1)), - f(0x77 * (i + 1)), - f(0x88 * (i + 1)), - ], - IV, - metadata(PINNED_T, FINAL_FLAG, 0), - ) - }) - .collect() - } - - /// `q_flock`'s aligned packed slots hold the VM's 64-bit words in our field - /// representation, and the digest matches the `blake2s` crate. - #[test] - fn qflock_words_match_layout() { - let inputs: Vec<([F64; 4], [F64; 4])> = (0..5u64) - .map(|i| { - ( - [f(0x1000 + i), f(0x2000 + i), f(0x3000 + i), f(0x4000 + i)], - [f(0x5000 + i), f(0x6000 + i), f(0x7000 + i), f(0x8000 + i)], - ) - }) - .collect(); - let blocks: Vec = inputs - .iter() - .map(|&(a, b)| compression(a, b, IV, metadata(PINNED_T, FINAL_FLAG, 0))) - .collect(); - let q_flock = build_qflock(&blocks); - assert_eq!(q_flock.len(), 1 << qflock_kappa(blocks.len())); - - let slot = |j: usize, s: usize| q_flock[j * (1 << SLOT_STRIDE_LOG) + s]; - for (j, (&(a, b), blk)) in inputs.iter().zip(&blocks).enumerate() { - for k in 0..4 { - assert_eq!(slot(j, SLOT_A0 + k), a[k]); - assert_eq!(slot(j, SLOT_B0 + k), b[k]); - } - let mut input = [0u8; 64]; - for (s, w) in input.as_chunks_mut::<8>().0.iter_mut().zip(a.into_iter().chain(b)) { - *s = w.0.to_le_bytes(); - } - let h = primitives::hash::hash(&input); - let word = |o: usize| F64(u64::from_le_bytes(h[o..o + 8].try_into().unwrap())); - let d: [F64; 4] = std::array::from_fn(|k| word(8 * k)); - assert_eq!(digest(blk), d); - for k in 0..4 { - assert_eq!(slot(j, SLOT_C0 + k), d[k]); - } - } - // Input slots for this one-block hash: the parameterized initial chaining - // value in slots 0..4, the byte counter in slot 18, and the packed - // f0‖f1 flag word in slot 19. - let iv = flock::hash::param_iv(); - for k in 0..4 { - assert_eq!(slot(0, k), pack_words([iv[2 * k], iv[2 * k + 1]])); - } - assert_eq!(slot(0, 18), pack_words([PINNED_T as u32, 0])); - assert_eq!(slot(0, 19), pack_words([FINAL_FLAG, 0])); - } - - /// The Flock reduction (zerocheck + lincheck) is a clean, self-contained - /// unit: run WITHOUT any PCS open, the prover's claim on the - /// committed witness `q_flock` are exactly what the verifier recovers by - /// replaying the sub-proofs. This is the seam the PCS builds on. - #[test] - fn reduction_roundtrip() { - let blocks = sample_blocks(4); - let q_flock = build_qflock(&blocks); - let dummy = vec![f(7); 8]; - let stacked = crate::witness::stack(&[q_flock.clone(), dummy]); - let offset = stacked.placements[0].offset; - - // Prover: commit, then run ONLY the reduction (no PCS open). - let mut ps = ProverState::from_label(b"reduce"); - let _committed = crate::pcs::commit(&mut ps, &stacked.q, stacked.shape, crate::pcs::TEST_LOG_INV_RATE); - let (z_packed, reduced) = prove_reduction(&blocks, &mut ps); - let bundle = ps.into_proof(); - - // The reduction regenerates exactly the committed `q_flock` sub-block. - assert_eq!(z_packed, q_flock, "reduction witness must equal committed q_flock"); - assert_eq!(&stacked.q[offset..offset + z_packed.len()], z_packed.as_slice()); - - // Verifier: replay the reduction and recover the claims. - let mut vs = VerifierState::from_label(b"reduce", &bundle); - let _root = crate::pcs::read_commitment(&mut vs).unwrap(); - let replay = verify_reduction(blocks.len(), &mut vs).expect("reduction verifies"); - - // Prover and verifier agree on the claim left for the PCS. - assert_eq!(reduced, replay.claim, "reduction claim mismatch"); - - // A mismatched transcript domain diverges the state, so the recovered - // claims must NOT match the prover's (the reduction is transcript-bound). - let mut vs_bad = VerifierState::from_label(b"different", &bundle); - let _root_b = crate::pcs::read_commitment(&mut vs_bad).unwrap(); - if let Ok(replay_b) = verify_reduction(blocks.len(), &mut vs_bad) { - assert!( - replay_b.claim != replay.claim, - "a diverged transcript must not reproduce the prover's claim" - ); - } - } - - /// flock's validity claims, discharged by ONE stacked WHIR over a - /// hand-stacked witness containing `q_flock` (plus a dummy column) together - /// with an ordinary point claim: the full prove_reduction → ring-switch → - /// stack_open seam without the VM pipeline. Proves and verifies on the - /// shared transcript; a mismatched domain and a tampered point value are - /// rejected. - #[test] - fn validity_stacked_roundtrip() { - let blocks = sample_blocks(4); - let q_flock = build_qflock(&blocks); - let dummy: Vec = (0..8u64).map(|i| f(0x9000 + i)).collect(); - let stacked = crate::witness::stack(&[q_flock.clone(), dummy.clone()]); - let offset = stacked.placements[0].offset; - - // One ordinary point claim on the dummy column (exercises the point-claim - // path of the single fused opening). - let dummy_pl = stacked.placements[1]; - let low_point: Vec = (0..dummy_pl.n_vars) - .map(|i| F192::new(0x100 + i as u64, 0x7, 0x55)) - .collect(); - let pd_value = primitives::multilinear::mle_eval(&dummy, &low_point); - let points = vec![crate::pcs::SlotClaim::Point { - offset: dummy_pl.offset, - low_point: low_point.clone(), - value: pd_value, - }]; - - let mut ps = ProverState::from_label(b"vstack"); - let committed = crate::pcs::commit(&mut ps, &stacked.q, stacked.shape, crate::pcs::TEST_LOG_INV_RATE); - let (_z, reduced) = prove_reduction(&blocks, &mut ps); - let ring = ring_switch_open(blocks.len(), offset, &reduced); - crate::pcs::open(&mut ps, &committed, &stacked.q, &points, &ring); - let bundle = ps.into_proof(); - - let run = |label: &'static [u8], points: &[crate::pcs::SlotClaim]| -> Result<(), &'static str> { - let mut vs = VerifierState::from_label(label, &bundle); - let root = crate::pcs::read_commitment(&mut vs).map_err(|_| "root")?; - let replay = verify_reduction(blocks.len(), &mut vs).map_err(|_| "reduction")?; - let ring = ring_switch_verify(blocks.len(), offset, &replay.claim); - crate::pcs::verify( - &mut vs, - points, - &ring, - stacked.shape, - crate::pcs::TEST_LOG_INV_RATE, - &root, - ) - .map_err(|_| "opening")?; - vs.finish().map_err(|_| "leftover") - }; - - run(b"vstack", &points).expect("validity verifies"); - - // A mismatched transcript (different domain) diverges the shared state, - // so the stacked opening must be rejected. - assert!( - run(b"different-domain", &points).is_err(), - "validity under a mismatched transcript must fail" - ); - - // A tampered point value must be rejected too. - let mut bad_points = points.clone(); - if let crate::pcs::SlotClaim::Point { value, .. } = &mut bad_points[0] { - *value += F192::ONE; - } - assert!(run(b"vstack", &bad_points).is_err(), "tampered point value must fail"); - } -} diff --git a/crates/lean_vm/src/leaf.rs b/crates/lean_vm/src/leaf.rs index 859a76b56..712373933 100644 --- a/crates/lean_vm/src/leaf.rs +++ b/crates/lean_vm/src/leaf.rs @@ -11,7 +11,7 @@ use crate::PAR_THRESHOLD; use crate::colval::ColVal; use crate::gkr; use crate::transcript::{Challenger, ProverState, Receiver, Transmitter, VerifierState}; -use primitives::field::{F64, F192, F192Unreduced, g_pow, index_mle}; +use primitives::field::{F64, F192, F192Unreduced, g_pow, index_mle, int_index_mle, powers_mle}; use primitives::multilinear::{eq_eval, eq_table_arena, mle_eval}; use std::collections::HashMap; use std::sync::Arc; @@ -27,27 +27,96 @@ pub enum Coord { /// The free increment `g^k · col[z]` (a virtual column, §sec:vm): `k = 1` for the /// count/state steps, `k ∈ {1,2,3}` for BLAKE2s's consecutive-word successors. GCol(usize, u32), - /// The product `g^k · col_a[z] · col_b[z]` of two committed columns. An address - /// is `fp·g^o`, so this carries one on the bus without committing it: the - /// coordinate IS the product, so no column can disagree with it and the binding - /// constraint that used to say so is unnecessary (§sec:m3). + /// The product `g^k · col_a[z] · col_b[z]` of two committed columns. The g-power + /// of an address `fp + o + k` is `g^fp·g^o·g^k`, so this carries one on the bus + /// without committing it: the coordinate IS the product, so no column can + /// disagree with it and the binding constraint that used to say so is unnecessary + /// (§sec:m3). Prod(usize, usize, u32), - /// The index column `g^z` (§sec:idxcol), free via the factored MLE. + /// The integer index column `base ^ (z << shift)` (§sec:idxcol), the element + /// whose bits are that integer's: what addresses a region whose cell `z` sits at + /// `base + (z << shift)`. Free, its MLE being linear. + IntIndex { base: F64, shift: u32 }, + /// The exponent column `g^z` (§sec:idxcol), free via the factored MLE: the + /// values of the `EXP` array. Index, + /// The geometric column `first·ratio^z`, free the same way: the addresses of a + /// range-check array (§sec:rangecheck). + Powers { first: F64, ratio: F64 }, /// A public column (the bytecode program, §sec:e2e-bc): not committed; both parties form /// its MLE directly, so it raises no claim. Shared rather than owned: push and - /// pull carry the same eight columns, tens of megabytes at production sizes. + /// pull carry the same ten columns, tens of megabytes at production sizes. Public(Arc>), + /// A public column that is zero outside a few blocks (RAM as the run finds it, + /// §sec:memchan): the verifier evaluates it in time proportional to the blocks, + /// not to the column. + Sparse(Arc), /// A sum of `Const`/`Col`/`GCol`/`Prod` terms: any degree-2 form over the /// table's columns, which is all §sec:m3 asks of a coordinate. This is what - /// carries a value a row DERIVES from its columns (an `XOR`/`MUL` result, a - /// `DEREF` store, a `JUMP` successor) without committing a column for it, and + /// carries a value a row DERIVES from its columns (a branch's successor, what a + /// jump writes to `rd`, a hash row's block addresses) without committing a column for it, and /// with it the identity that would have tied the two. Like [`Coord::Prod`], /// only a table's blocks may carry one: the table sumcheck settles them, /// while a framework block has to split into per-column openings. Sum(Vec), } +/// A public column of `2^log_len` words given by its nonzero stretches, each cut into +/// ALIGNED blocks: a power of two of words, at an offset that is a multiple of it. Such +/// a block's share of the column's multilinear extension is its own extension in the +/// low variables times the indicator of its offset's bits in the high ones. +#[derive(Debug)] +pub struct SparseColumn { + log_len: usize, + blocks: Vec<(usize, Vec)>, + /// The column written out, which only the prover needs. + dense: std::sync::OnceLock>, +} + +impl SparseColumn { + /// From `(offset, words)` stretches, which must not overlap. + pub fn new(log_len: usize, stretches: &[(usize, &[u64])]) -> Self { + let mut blocks = Vec::new(); + for &(mut at, mut words) in stretches { + assert!(at + words.len() <= 1 << log_len, "a stretch runs past the column"); + while !words.is_empty() { + // The largest aligned block starting here that the stretch still fills. + let aligned = if at == 0 { usize::MAX } else { 1 << at.trailing_zeros() }; + let size = aligned.min(1 << words.len().ilog2()); + blocks.push((at, words[..size].iter().map(|&w| F64(w)).collect())); + (at, words) = (at + size, &words[size..]); + } + } + Self { + log_len, + blocks, + dense: std::sync::OnceLock::new(), + } + } + + fn dense(&self) -> &[F64] { + self.dense.get_or_init(|| { + let mut column = vec![F64::ZERO; 1 << self.log_len]; + for (at, words) in &self.blocks { + column[*at..at + words.len()].copy_from_slice(words); + } + column + }) + } + + /// The column's multilinear extension at `point`. + fn eval(&self, point: &[F192]) -> F192 { + assert_eq!(point.len(), self.log_len); + self.blocks.iter().fold(F192::ZERO, |acc, (at, words)| { + let k = words.len().ilog2() as usize; + let selector = point[k..].iter().enumerate().fold(F192::ONE, |s, (j, &z)| { + s * if (at >> (k + j)) & 1 == 1 { z } else { z + F192::ONE } + }); + acc + selector * primitives::multilinear::mle_eval(words, &point[..k]) + }) + } +} + /// A flushing rule: `2^kappa` rows, each a tuple of coordinates. Every one of them is a /// row the program executed, since a table's height is its row count (§sec:e2e-pad), so a /// block has no padding rows to divide back out of the product. @@ -76,7 +145,7 @@ pub struct ColumnClaim { #[derive(Clone, Debug, PartialEq, Eq)] pub enum Error { Truncated, - /// A read count is zero, so a read self-cancels on the bus (§sec:memchan). + /// A lookup's read count is zero, so the read self-cancels on the bus (§sec:lookup). ZeroCount, Gkr(gkr::GkrError), } @@ -156,6 +225,45 @@ pub fn layout(blocks: &[Block]) -> Layout { } } +/// The leaves one side leaves unmatched on the other, as `(side, block, row)`, under +/// one fixed fingerprint: what to look at when a bus does not balance. +#[cfg(test)] +pub(crate) fn unmatched_leaves(push: &[Block], pull: &[Block], cols: &[&[F64]]) -> Vec<(&'static str, usize, usize)> { + let alphas: Vec = (0..N_TUPLE_BITS as u64) + .map(|i| F192::new(3 + i, 5 + 7 * i, 11)) + .collect(); + let (w, beta) = (fingerprint_weights(&alphas), F192::new(13, 17, 19)); + let powers = power_tables([push, pull, &[]]); + let longest = push.iter().chain(pull).map(|b| b.kappa).max().unwrap_or(0); + let gpow = primitives::field::g_powers(1 << longest); + let side = |blocks: &[Block]| { + let lay = layout(blocks); + let leaves = build_leaves(blocks, &lay, cols, &w, beta, &gpow, &powers); + let mut at = Vec::new(); + for (b, block) in blocks.iter().enumerate() { + at.extend((0..1usize << block.kappa).map(|z| (leaves[lay.offsets[b] + z], b, z))); + } + at + }; + let (pushed, pulled) = (side(push), side(pull)); + let key = |leaf: &F192| (leaf.c0, leaf.c1, leaf.c2); + let mut counts: HashMap<_, i64> = HashMap::new(); + for (leaf, ..) in &pushed { + *counts.entry(key(leaf)).or_default() += 1; + } + for (leaf, ..) in &pulled { + *counts.entry(key(leaf)).or_default() -= 1; + } + let unmatched = |name: &'static str, leaves: &[(F192, usize, usize)]| { + leaves + .iter() + .filter(|(leaf, ..)| counts[&key(leaf)] != 0) + .map(|&(_, b, z)| (name, b, z)) + .collect::>() + }; + [unmatched("push", &pushed), unmatched("pull", &pulled)].concat() +} + /// A non-constant coordinate as `(source, coefficient)`: its leaf contribution is /// the mixed product `coeff · source(z)` with `source(z) ∈ K`, `coeff ∈ E`. /// `GCol` folds the `g^k` factor into the coefficient. @@ -163,23 +271,57 @@ enum Term<'a> { Col(usize, F192), Prod(usize, usize, F192), Index(F192), + IntIndex(F192, u32), Public(&'a [F64], F192), } +/// The values of each distinct [`Coord::Powers`] column, prover-side. +pub type PowerTables = Vec<((F64, F64), Vec)>; + +fn power_tables(sides: [&[Block]; 3]) -> PowerTables { + let mut tables = PowerTables::new(); + for blk in sides.into_iter().flatten() { + for c in &blk.coords { + if let Coord::Powers { first, ratio } = *c + && !tables.iter().any(|(k, _)| *k == (first, ratio)) + { + tables.push(( + (first, ratio), + primitives::field::geometric(first, ratio, 1 << blk.kappa), + )); + } + } + } + tables +} + /// Flatten one coordinate into leaf terms at coefficient `w`. A [`Coord::Sum`] /// spreads its children over the SAME `w`: they are one coordinate, so they share /// its `α`-power. -fn push_terms<'a>(c: &'a Coord, w: F192, terms: &mut Vec>, constant: &mut F192) { +fn push_terms<'a>(c: &'a Coord, w: F192, powers: &'a PowerTables, terms: &mut Vec>, constant: &mut F192) { match c { Coord::Const(v) => *constant += w.mul_base(*v), Coord::Col(i) => terms.push(Term::Col(*i, w)), Coord::GCol(i, k) => terms.push(Term::Col(*i, w.mul_base(g_pow(*k as usize)))), Coord::Prod(i, j, k) => terms.push(Term::Prod(*i, *j, w.mul_base(g_pow(*k as usize)))), Coord::Index => terms.push(Term::Index(w)), + Coord::IntIndex { base, shift } => { + *constant += w.mul_base(*base); + terms.push(Term::IntIndex(w, *shift)); + } + Coord::Powers { first, ratio } => { + let table = &powers + .iter() + .find(|(k, _)| *k == (*first, *ratio)) + .expect("every geometric column was tabulated") + .1; + terms.push(Term::Public(table, w)); + } Coord::Public(vals) => terms.push(Term::Public(vals.as_slice(), w)), + Coord::Sparse(column) => terms.push(Term::Public(column.dense(), w)), Coord::Sum(cs) => { for c in cs { - push_terms(c, w, terms, constant); + push_terms(c, w, powers, terms, constant); } } } @@ -197,6 +339,7 @@ pub fn build_leaves( w: &[F192], beta: F192, gpow: &[F64], + powers: &PowerTables, ) -> ArenaVec { let explicit = blocks .iter() @@ -228,7 +371,7 @@ pub fn build_leaves( let mut const_part = beta; let mut terms: Vec = Vec::with_capacity(blk.coords.len()); for (i, c) in blk.coords.iter().enumerate() { - push_terms(c, w[i], &mut terms, &mut const_part); + push_terms(c, w[i], powers, &mut terms, &mut const_part); } let row = |z: usize| -> F192 { // The α-weighted coordinate sum defers its reductions: each mixed @@ -241,6 +384,7 @@ pub fn build_leaves( Term::Col(i, c) => c.mul_base_unreduced(cols[*i][z]), Term::Prod(i, j, c) => c.mul_base_unreduced(cols[*i][z] * cols[*j][z]), Term::Index(c) => c.mul_base_unreduced(gpow[z]), + Term::IntIndex(c, shift) => c.mul_base_unreduced(F64((z as u64) << shift)), Term::Public(vals, c) => c.mul_base_unreduced(vals[z]), }; } @@ -380,7 +524,7 @@ fn accumulate_form(c: &Coord, w: F192, base: usize, form: &mut BusForm) { accumulate_form(c, w, base, form); } } - Coord::Index | Coord::Public(_) => { + Coord::Index | Coord::IntIndex { .. } | Coord::Powers { .. } | Coord::Public(_) | Coord::Sparse(_) => { unreachable!("a table's bus block carries no virtual coordinate") } } @@ -448,12 +592,15 @@ fn decompose_formula Result>( let coord_val = match c { Coord::Const(v) => F192::from(*v), Coord::Index => index_mle(zeta_lo), + Coord::IntIndex { base, shift } => int_index_mle(*base, *shift, zeta_lo), + Coord::Powers { first, ratio } => powers_mle(*first, *ratio, zeta_lo), Coord::Col(i) => col_val(*i)?, Coord::GCol(i, k) => col_val(*i)?.mul_base(g_pow(*k as usize)), Coord::Prod(..) | Coord::Sum(..) => { unreachable!("only a table's bus block carries a degree-2 coordinate") } Coord::Public(vals) => public_eval(vals, zeta_lo, public), + Coord::Sparse(column) => column.eval(zeta_lo), }; inner += w[i] * coord_val; } @@ -575,20 +722,6 @@ fn decompose_verify( }) } -/// One reduced claim on the bytecode polynomial. The eight public encoding -/// columns (opcode plus seven operand/immediate slots), padded to sixteen slots -/// along four selector bits, form one multilinear polynomial B̃ in `κ_bc + 4` -/// variables. The native verifier combines its column evaluations at ζ with -/// the bus weights `eq(α⃗, ·)`, giving `B̃(ζ_lo, α⃗)`. The recursive verifier -/// defers this claim to its public input. -#[derive(Clone, Debug)] -pub struct BytecodeClaim { - /// `ζ_side_lo ++ s`, a point in `κ_bc + 4` variables. - pub point: Vec, - /// `B̃(point)`. - pub value: F192, -} - /// Selector bits of the stacked bytecode polynomial: the public encoding /// columns (opcode + seven operand/immediate slots = eight) stack along /// `2^N_BYTECODE_SELECTORS` slots. A column's slot is its bus tuple coordinate, @@ -601,9 +734,8 @@ pub const N_BYTECODE_SELECTORS: usize = 4; pub const BYTECODE_PUBLIC_SLOT: usize = 3; /// The stacked bytecode polynomial as a dense table: eight public encoding -/// columns at their tuple coordinates, padded to sixteen selector slots. This is -/// the polynomial [`BytecodeClaim`]s are claims about; the outermost verifier -/// evaluates it. +/// columns at their tuple coordinates, padded to sixteen selector slots. The +/// program's digest is taken over it ([`crate::cpu::Program`]). pub fn stacked_bytecode_table(blocks: &[Block]) -> Vec { let mut kbc = 0; let mut cols: Vec<&[F64]> = Vec::new(); @@ -643,33 +775,6 @@ fn sides<'a>( ] } -/// The program's whole share of a bus leaf, in ONE evaluation: a public column's -/// slot is its tuple coordinate and the weights are `eq(α⃗, ·)`, so the weighted -/// sum over the columns IS the stacked polynomial at `(ζ, α⃗)` (§sec:e2e-bc). -fn bytecode_claim(blocks: &[Block], point: &[F192], alphas: &[F192], public: &mut PublicEvals) -> BytecodeClaim { - let weights = fingerprint_weights(alphas); - let mut kbc = 0; - let mut slot = BYTECODE_PUBLIC_SLOT; - let mut value = F192::ZERO; - for blk in blocks { - for c in &blk.coords { - if let Coord::Public(vals) = c { - if slot == BYTECODE_PUBLIC_SLOT { - kbc = blk.kappa; - } - assert_eq!(vals.len(), 1 << kbc); - value += weights[slot] * public_eval(vals, &point[..kbc], public); - slot += 1; - } - } - } - let claim_point = [&point[..kbc], alphas].concat(); - BytecodeClaim { - value, - point: claim_point, - } -} - /// Prove the bus balances; returns the per-column claims to open (§sec:leafstack). `alpha`/ /// `beta` follow the witness commitment (the only ordering the grand product /// needs), and the block structure is public, so no shape is observed. @@ -716,14 +821,15 @@ pub fn prove_balance( .map(|b| b.kappa) .max(); let gpow = index_k.map_or_else(Vec::new, |k| primitives::field::g_powers(1usize << k)); + let powers = power_tables([push, pull, count]); // Three independent leaf vectors, built one after another: each `build_leaves` // already fans its own blocks out across the whole pool, so nesting a // three-way outer split on top would only add a barrier. let [push_leaves, pull_leaves, count_leaves] = crate::stage!("Bus leaves", || { [ - build_leaves(push, &push_lay, cols, &w, beta, &gpow), - build_leaves(pull, &pull_lay, cols, &w, beta, &gpow), - build_leaves(count, &count_lay, cols, &count_w, F192::ZERO, &gpow), + build_leaves(push, &push_lay, cols, &w, beta, &gpow, &powers), + build_leaves(pull, &pull_lay, cols, &w, beta, &gpow, &powers), + build_leaves(count, &count_lay, cols, &count_w, F192::ZERO, &gpow, &powers), ] }); // Leaf construction keeps the all-one padding implicit; decomposition uses the full logical depth. @@ -872,12 +978,10 @@ fn tables_and_prods_at( .unzip() } -/// What [`verify_balance`] establishes: the per-column claims to open, the -/// reduced bytecode claim (push and pull share ζ), -/// and the table forms with their claimed sums. +/// What [`verify_balance`] establishes: the per-column claims to open and the +/// table forms with their claimed sums. pub struct BusVerify { pub claims: Vec, - pub bytecode_claim: BytecodeClaim, /// The GKR point ζ, reused as the table sumcheck's eq point. pub point: Vec, /// `forms[side][table]`, for the zerocheck to settle. @@ -910,7 +1014,7 @@ pub fn verify_balance( let beta = vs.sample(); let bus_gkr = gkr::verify_product_triple(push_lay.mu, vs, gkr::RootShape::FirstTwoShared).map_err(Error::Gkr)?; let count_root = bus_gkr.roots[2]; - // Every read count is nonzero iff this product is (§sec:memchan); a zero would + // Every lookup's read count is nonzero iff this product is (§sec:lookup); a zero would // let a read self-cancel and free its value from memory. if count_root == F192::ZERO { return Err(Error::ZeroCount); @@ -956,7 +1060,6 @@ pub fn verify_balance( Ok(BusVerify { claims, - bytecode_claim: bytecode_claim(push, &bus_gkr.point, &alphas, &mut public), point: bus_gkr.point, forms, totals, @@ -965,29 +1068,24 @@ pub fn verify_balance( #[cfg(test)] mod tests { - use super::soundness_bits; + use super::{F64, F192, SparseColumn, soundness_bits}; + /// A sparse column's block-wise evaluation is its dense multilinear extension, + /// whatever the stretches' offsets and lengths. #[test] - fn bytecode_claim_matches_dense_stacking() { - use super::*; - - let columns: Vec<_> = (0..8) - .map(|col| Arc::new((0..32).map(|i| F64((i + 1) * (col + 1))).collect())) - .collect(); - let blocks = [Block { - kappa: 5, - coords: columns.iter().cloned().map(Coord::Public).collect(), - }]; - let point: Vec<_> = (0..5).map(|i| F192::new(i + 2, i + 17, i + 23)).collect(); - let alphas: Vec<_> = (0..N_TUPLE_BITS).map(|i| F192::new(i as u64 + 5, 3, 7)).collect(); - let table = stacked_bytecode_table(&blocks); - let mut public = PublicEvals::new(); - for col in columns.iter().rev() { - public_eval(col, &point, &mut public); - } - let claim = bytecode_claim(&blocks, &point, &alphas, &mut public); - assert_eq!(claim.point, [point, alphas].concat()); - assert_eq!(claim.value, mle_eval(&table, &claim.point)); + fn sparse_column_evaluates_as_its_dense_form() { + let words: Vec = (1..=37).map(|i| i * 0x9e37_79b9_7f4a_7c15).collect(); + let column = SparseColumn::new(9, &[(0, &words[..4]), (5, &words[4..17]), (300, &words[17..])]); + assert_eq!( + column.dense()[5..18], + words[4..17].iter().map(|&w| F64(w)).collect::>() + ); + assert_eq!(column.dense().iter().filter(|w| !w.is_zero()).count(), words.len()); + let point: Vec = (0..9).map(|i| F192::new(3 + i, 5 * i + 1, 7)).collect(); + assert_eq!( + column.eval(&point), + primitives::multilinear::mle_eval(column.dense(), &point) + ); } /// The bound is `(N_TUPLE_BITS + 1)·2^mu` plus the GKR terms: only the bus diff --git a/crates/lean_vm/src/lib.rs b/crates/lean_vm/src/lib.rs index 3103fc224..e4d2836f6 100644 --- a/crates/lean_vm/src/lib.rs +++ b/crates/lean_vm/src/lib.rs @@ -1,35 +1,36 @@ //! leanVM: arithmetization of a minimal zkVM (see `doc/leanvm/main.tex`). //! -//! Machine words are `c0 + c1*y + c2*y² ∈ E = K[y]/(y³ + y + 1)`. -//! Addresses, pc/fp, read counters, and logical indices live in -//! `K = GF(2^64)`; indices are powers of a fixed generator `g`, so incrementing -//! one is a multiplication by `g`, a free virtual operation. Every physical -//! witness column is K-valued (an E-valued word is three K-lane columns) and is -//! committed directly by a dense multilinear PCS. Challenges and transcript -//! scalars live in `E = GF(2^192)`, leaving ample margin for 128-bit soundness. +//! Machine words, addresses, the pc, timestamps and read counters live in `K = GF(2^64)`. +//! What the machine computes with is an integer, read as the element with those bits; +//! what the proof system only ever steps (a timestamp, a read count) is a power of a +//! fixed generator `g`, so incrementing one is a multiplication by `g`, a free virtual +//! operation. Every physical witness column is K-valued and is committed directly by a +//! dense multilinear PCS. +//! Challenges and transcript scalars live in `E = GF(2^192)`, leaving ample margin +//! for 128-bit soundness. //! //! - [`transcript`]: the shared Fiat-Shamir transcript (re-exported from `fiat_shamir`). //! - [`pcs`]: `K`-committed witness, `E`-opened, via the stacked WHIR (§sec:stacking, §annex:pcs). //! - [`witness`]: `K`-valued columns stacked into one committed witness. //! - [`gkr`]: the grand product via GKR (§sec:gkr), balancing the bus. //! - [`leaf`]: the shared bus: grand-product balance, decomposed to per-column claims (§sec:gp through §sec:leafstack, §sec:omc). -//! - [`constraints`]: one table sumcheck over all six tables' +//! - [`constraints`]: one table sumcheck over all the tables' //! degree-2 identities plus their three bus forms (§sec:air). -//! - [`tables`]: the six instruction tables (columns, flushes, constraints). -//! - [`cpu`]: whole-program assembly, control flow, and the prove/verify entry points. -//! - [`hash_flock`]: the `BLAKE2s` glue: flock's R1CS validity proof over the same commitment. -//! - [`vmhash`]: VM-native hashing (one-block compression and standard BLAKE2s slice hashing). +//! - [`rv`]: RISC-V (rv64im): the decoder, each instruction class's function and circuit, and the reference interpreter. +//! - [`tables`]: the instruction tables, one per class (columns, flushes, constraints). +//! - [`class_flock`]: the glue to flock: a class's circuit proven over its own packed witness, in the same commitment. +//! - [`cpu`]: whole-program assembly and the prove/verify entry points. +pub mod class_flock; pub mod colval; pub mod constraints; pub mod cpu; pub mod gkr; -pub mod hash_flock; pub mod leaf; pub mod pcs; +pub mod rv; pub mod tables; pub mod transcript; -pub mod vmhash; pub mod witness; /// Prepare the process for proving: the worker pool ([`init_prover_pool`]) plus diff --git a/crates/lean_vm/src/pcs.rs b/crates/lean_vm/src/pcs.rs index a13f5516b..d7b0ee5d0 100644 --- a/crates/lean_vm/src/pcs.rs +++ b/crates/lean_vm/src/pcs.rs @@ -47,7 +47,7 @@ pub use ::pcs::whir::{MAX_LOG_INV_RATE, MIN_LOG_INV_RATE}; const _: () = assert!(::pcs::whir::SECURITY_BITS == crate::SECURITY_BITS as usize); /// Minimum committed-witness log-size accepted by the WHIR level ladder, with one level of margin. pub const MIN_MU: usize = 15; -/// Largest committed size accepted by all verifiers and compiled into the recursion guest. +/// Largest committed size accepted by all verifiers. pub const MAX_MU: usize = 28; /// The shared WHIR config for a `2^μ`-word witness, memoized per `(μ, log_inv_rate)`. @@ -144,12 +144,12 @@ pub fn read_commitment(vs: &mut VerifierState) -> Result<[u8; 32], crate::transc /// /// There is no plain (non-ring-switch) path: the witness ALWAYS carries a `q_flock` /// sub-block (≥ 1 padding instance, §cpu), so every opening is stacked. -pub fn open(ps: &mut ProverState, c: &Committed, q: &[F64], points: &[SlotClaim], ring: &RingSwitchOpen) { +pub fn open(ps: &mut ProverState, c: &Committed, q: &[F64], points: &[SlotClaim], rings: &[RingSwitchOpen]) { let lane_block = 1usize << (c.mu - LOG_BATCH); assert_eq!(q.len() % lane_block, 0, "witness must be whole committed lanes"); assert!(q.len() <= 1usize << c.mu, "witness must fit the announced size"); let cfg = whir_config(c.mu, c.log_inv_rate); - open_batch_mixed_whir_stacked(ps, c.mu, q, &c.prover_data, &cfg, points, ring) + open_batch_mixed_whir_stacked(ps, c.mu, q, &c.prover_data, &cfg, points, rings) } /// Verify the opening (mirror of [`open`]): flock's ring-switched claim @@ -158,13 +158,13 @@ pub fn open(ps: &mut ProverState, c: &Committed, q: &[F64], points: &[SlotClaim] pub fn verify( vs: &mut VerifierState, points: &[SlotClaim], - ring: &RingSwitchVerify<'_>, + rings: &[RingSwitchVerify<'_>], shape: crate::witness::StackShape, log_inv_rate: usize, root: &[u8; 32], ) -> Result<(), Error> { let cfg = whir_config(shape.mu, log_inv_rate); - verify_opening_batch_mixed_whir_stacked(vs, &cfg, shape.mu, shape.n_lanes, root, points, ring).map_err(Error::Whir) + verify_opening_batch_mixed_whir_stacked(vs, &cfg, shape.mu, shape.n_lanes, root, points, rings).map_err(Error::Whir) } #[cfg(test)] diff --git a/crates/lean_vm/src/rv/asm.rs b/crates/lean_vm/src/rv/asm.rs new file mode 100644 index 000000000..746e7c001 --- /dev/null +++ b/crates/lean_vm/src/rv/asm.rs @@ -0,0 +1,265 @@ +//! A small assembler: instruction encoders, labels for branches and jumps, and `li`. +//! For hand-written programs and tests; real guests come from an ELF. + +use std::collections::HashMap; + +// Registers, by their ABI names. +pub const ZERO: u32 = 0; +pub const RA: u32 = 1; +pub const SP: u32 = 2; +pub const T0: u32 = 5; +pub const T1: u32 = 6; +pub const T2: u32 = 7; +pub const S0: u32 = 8; +pub const S1: u32 = 9; +pub const A0: u32 = 10; +pub const A1: u32 = 11; +pub const A2: u32 = 12; +pub const A3: u32 = 13; +pub const A4: u32 = 14; +pub const A5: u32 = 15; +pub const A6: u32 = 16; +pub const A7: u32 = 17; + +pub fn r_type(opcode: u32, f3: u32, f7: u32, rd: u32, rs1: u32, rs2: u32) -> u32 { + opcode | (rd << 7) | (f3 << 12) | (rs1 << 15) | (rs2 << 20) | (f7 << 25) +} +pub fn i_type(opcode: u32, f3: u32, rd: u32, rs1: u32, imm: i32) -> u32 { + opcode | (rd << 7) | (f3 << 12) | (rs1 << 15) | ((imm as u32 & 0xfff) << 20) +} +pub fn s_type(opcode: u32, f3: u32, rs1: u32, rs2: u32, imm: i32) -> u32 { + let imm = imm as u32; + opcode | ((imm & 31) << 7) | (f3 << 12) | (rs1 << 15) | (rs2 << 20) | ((imm >> 5 & 0x7f) << 25) +} +pub fn b_type(f3: u32, rs1: u32, rs2: u32, offset: i32) -> u32 { + let o = offset as u32; + 0x63 | ((o >> 11 & 1) << 7) + | ((o >> 1 & 0xf) << 8) + | (f3 << 12) + | (rs1 << 15) + | (rs2 << 20) + | ((o >> 5 & 0x3f) << 25) + | ((o >> 12 & 1) << 31) +} +pub fn u_type(opcode: u32, rd: u32, imm20: u32) -> u32 { + opcode | (rd << 7) | ((imm20 & 0xf_ffff) << 12) +} +pub fn j_type(rd: u32, offset: i32) -> u32 { + let o = offset as u32; + 0x6f | (rd << 7) + | ((o >> 12 & 0xff) << 12) + | ((o >> 11 & 1) << 20) + | ((o >> 1 & 0x3ff) << 21) + | ((o >> 20 & 1) << 31) +} + +/// `(mnemonic, opcode, f3, f7)` of every register-register instruction. +pub const R_OPS: [(&str, u32, u32, u32); 28] = [ + ("add", 0x33, 0, 0), + ("sub", 0x33, 0, 0x20), + ("sll", 0x33, 1, 0), + ("slt", 0x33, 2, 0), + ("sltu", 0x33, 3, 0), + ("xor", 0x33, 4, 0), + ("srl", 0x33, 5, 0), + ("sra", 0x33, 5, 0x20), + ("or", 0x33, 6, 0), + ("and", 0x33, 7, 0), + ("mul", 0x33, 0, 1), + ("mulh", 0x33, 1, 1), + ("mulhsu", 0x33, 2, 1), + ("mulhu", 0x33, 3, 1), + ("div", 0x33, 4, 1), + ("divu", 0x33, 5, 1), + ("rem", 0x33, 6, 1), + ("remu", 0x33, 7, 1), + ("addw", 0x3b, 0, 0), + ("subw", 0x3b, 0, 0x20), + ("sllw", 0x3b, 1, 0), + ("srlw", 0x3b, 5, 0), + ("sraw", 0x3b, 5, 0x20), + ("mulw", 0x3b, 0, 1), + ("divw", 0x3b, 4, 1), + ("divuw", 0x3b, 5, 1), + ("remw", 0x3b, 6, 1), + ("remuw", 0x3b, 7, 1), +]; +/// `(mnemonic, opcode, f3)` of the register-immediate instructions with a 12-bit immediate. +pub const I_OPS: [(&str, u32, u32); 7] = [ + ("addi", 0x13, 0), + ("slti", 0x13, 2), + ("sltiu", 0x13, 3), + ("xori", 0x13, 4), + ("ori", 0x13, 6), + ("andi", 0x13, 7), + ("addiw", 0x1b, 0), +]; +/// `(mnemonic, opcode, f3, top bits, amount bits)` of the shifts by an immediate. +pub const SHIFT_OPS: [(&str, u32, u32, u32, u32); 6] = [ + ("slli", 0x13, 1, 0, 6), + ("srli", 0x13, 5, 0, 6), + ("srai", 0x13, 5, 0x400, 6), + ("slliw", 0x1b, 1, 0, 5), + ("srliw", 0x1b, 5, 0, 5), + ("sraiw", 0x1b, 5, 0x400, 5), +]; +/// `(mnemonic, f3)`. +pub const LOAD_OPS: [(&str, u32); 7] = [ + ("lb", 0), + ("lh", 1), + ("lw", 2), + ("ld", 3), + ("lbu", 4), + ("lhu", 5), + ("lwu", 6), +]; +pub const STORE_OPS: [(&str, u32); 4] = [("sb", 0), ("sh", 1), ("sw", 2), ("sd", 3)]; +pub const BRANCH_OPS: [(&str, u32); 6] = [("beq", 0), ("bne", 1), ("blt", 4), ("bge", 5), ("bltu", 6), ("bgeu", 7)]; + +pub const ECALL: u32 = 0x73; + +fn lookup(table: &[T], name: impl Fn(&T) -> &str, mnemonic: &str) -> T { + *table + .iter() + .find(|t| name(t) == mnemonic) + .unwrap_or_else(|| panic!("no instruction {mnemonic}")) +} + +enum Fixup { + Branch(u32, u32, u32), + Jal(u32), +} + +/// A program under assembly. +#[derive(Default)] +pub struct Asm { + words: Vec, + labels: HashMap<&'static str, usize>, + fixups: Vec<(usize, &'static str, Fixup)>, +} + +impl Asm { + pub fn new() -> Self { + Self::default() + } + + pub fn word(&mut self, word: u32) -> &mut Self { + self.words.push(word); + self + } + + pub fn label(&mut self, name: &'static str) -> &mut Self { + assert!( + self.labels.insert(name, self.words.len()).is_none(), + "label {name} defined twice" + ); + self + } + + /// A register-register instruction of [`R_OPS`]. + pub fn r(&mut self, mnemonic: &str, rd: u32, rs1: u32, rs2: u32) -> &mut Self { + let (_, opcode, f3, f7) = lookup(&R_OPS, |t| t.0, mnemonic); + self.word(r_type(opcode, f3, f7, rd, rs1, rs2)) + } + + /// A register-immediate instruction of [`I_OPS`] or [`SHIFT_OPS`]. + pub fn i(&mut self, mnemonic: &str, rd: u32, rs1: u32, imm: i32) -> &mut Self { + if let Some(&(_, opcode, f3, top, bits)) = SHIFT_OPS.iter().find(|t| t.0 == mnemonic) { + assert!((0..1 << bits).contains(&imm), "{mnemonic} by {imm}"); + return self.word(i_type(opcode, f3, rd, rs1, top as i32 | imm)); + } + let (_, opcode, f3) = lookup(&I_OPS, |t| t.0, mnemonic); + assert!((-2048..2048).contains(&imm), "{mnemonic} immediate {imm}"); + self.word(i_type(opcode, f3, rd, rs1, imm)) + } + + /// `rd <- mem[rs1 + offset]`, an instruction of [`LOAD_OPS`]. + pub fn load(&mut self, mnemonic: &str, rd: u32, offset: i32, rs1: u32) -> &mut Self { + self.word(i_type(0x03, lookup(&LOAD_OPS, |t| t.0, mnemonic).1, rd, rs1, offset)) + } + + /// `mem[rs1 + offset] <- rs2`, an instruction of [`STORE_OPS`]. + pub fn store(&mut self, mnemonic: &str, rs2: u32, offset: i32, rs1: u32) -> &mut Self { + self.word(s_type(0x23, lookup(&STORE_OPS, |t| t.0, mnemonic).1, rs1, rs2, offset)) + } + + /// A branch of [`BRANCH_OPS`] to `label`. + pub fn branch(&mut self, mnemonic: &str, rs1: u32, rs2: u32, label: &'static str) -> &mut Self { + let f3 = lookup(&BRANCH_OPS, |t| t.0, mnemonic).1; + self.fixups.push((self.words.len(), label, Fixup::Branch(f3, rs1, rs2))); + self.word(0) + } + + pub fn jal(&mut self, rd: u32, label: &'static str) -> &mut Self { + self.fixups.push((self.words.len(), label, Fixup::Jal(rd))); + self.word(0) + } + + pub fn jalr(&mut self, rd: u32, rs1: u32, offset: i32) -> &mut Self { + self.word(i_type(0x67, 0, rd, rs1, offset)) + } + + pub fn lui(&mut self, rd: u32, imm20: u32) -> &mut Self { + self.word(u_type(0x37, rd, imm20)) + } + + pub fn auipc(&mut self, rd: u32, imm20: u32) -> &mut Self { + self.word(u_type(0x17, rd, imm20)) + } + + /// `rd <- value`, any 64-bit constant. + pub fn li(&mut self, rd: u32, value: u64) -> &mut Self { + let v = value as i64; + let lo = (v << 52) >> 52; + if v == v as i32 as i64 { + let hi = ((v.wrapping_sub(lo)) >> 12) as u32 & 0xf_ffff; + if hi == 0 { + return self.i("addi", rd, ZERO, lo as i32); + } + self.lui(rd, hi); + return if lo == 0 { + self + } else { + self.i("addiw", rd, rd, lo as i32) + }; + } + let hi = v.wrapping_sub(lo) >> 12; + let zeros = hi.trailing_zeros(); + self.li(rd, (hi >> zeros) as u64).i("slli", rd, rd, (12 + zeros) as i32); + if lo == 0 { + self + } else { + self.i("addi", rd, rd, lo as i32) + } + } + + /// `blake2s rs1, rs2` ([`super::hash`]): compress the block at `rs1` with the + /// counter `rs2`, `last` marking the final block. + pub fn blake2s(&mut self, rs1: u32, rs2: u32, last: bool) -> &mut Self { + self.word(r_type(super::hash::OPCODE, last as u32, 0, 0, rs1, rs2)) + } + + /// `exit(a0)`. + pub fn exit(&mut self) -> &mut Self { + self.i("addi", A7, ZERO, super::SYS_EXIT as i32).word(ECALL) + } + + /// The words, every label resolved. + pub fn finish(&mut self) -> Vec { + for (at, label, fixup) in self.fixups.drain(..) { + let to = *self + .labels + .get(label) + .unwrap_or_else(|| panic!("undefined label {label}")); + let offset = (to as i32 - at as i32) * 4; + self.words[at] = match fixup { + Fixup::Branch(f3, rs1, rs2) => { + assert!((-4096..4096).contains(&offset), "branch to {label} out of range"); + b_type(f3, rs1, rs2, offset) + } + Fixup::Jal(rd) => j_type(rd, offset), + }; + } + std::mem::take(&mut self.words) + } +} diff --git a/crates/lean_vm/src/rv/circuits.rs b/crates/lean_vm/src/rv/circuits.rs new file mode 100644 index 000000000..14791a3f4 --- /dev/null +++ b/crates/lean_vm/src/rv/circuits.rs @@ -0,0 +1,706 @@ +//! Each instruction class's function ([`super::semantics`]) as a flock gate list. +//! +//! A circuit's ports are whole 64-bit words, and they are the words its table puts +//! on the bus: the values read from the registers, the bytecode's immediate and +//! flags, the result. A circuit is defined on its class's legal flags only, which +//! are one-hot where they select, so a selection is an XOR of products. + +use flock::circuit::{Builder, Circuit, Wire}; + +/// A word of wires, low bit first. +type Word = Vec; + +fn xor_words(c: &mut Builder, x: &[Wire], y: &[Wire]) -> Word { + x.iter().zip(y).map(|(&x, &y)| c.xor(x, y)).collect() +} + +/// `s·x`, bit by bit. +fn gate_word(c: &mut Builder, s: Wire, x: &[Wire]) -> Word { + x.iter().map(|&x| c.and(s, x)).collect() +} + +/// Whether any bit of `x` is set. +fn any(c: &mut Builder, x: &[Wire]) -> Wire { + x.iter().fold(None, |acc, &bit| c.or(acc, bit)) +} + +/// `x + y + carry_in`, and the carry out of the top bit. +fn add_with_carry(c: &mut Builder, x: &[Wire], y: &[Wire], carry_in: Wire) -> (Word, Wire) { + let mut carry = carry_in; + let mut sum = Vec::with_capacity(x.len()); + for (&x, &y) in x.iter().zip(y) { + let xc = c.xor(x, carry); + let yc = c.xor(y, carry); + sum.push(c.xor(xc, y)); + let maj = c.and(xc, yc); + carry = c.xor(maj, carry); + } + (sum, carry) +} + +/// `x`, its bits from 32 up replaced by bit 31 when `word` is set. +fn sext32_if(c: &mut Builder, word: Wire, x: &[Wire]) -> Word { + (0..64) + .map(|i| if i < 32 { x[i] } else { c.mux(word, x[31], x[i]) }) + .collect() +} + +/// `x + y` over `x.len()` bits, the carry out of the top bit dropped: one product +/// per bit but the top, the carry into bit `i + 1` being `maj(x_i, y_i, c_i)`. +fn add(c: &mut Builder, x: &[Wire], y: &[Wire]) -> Word { + let (mut carry, width) = (None, x.len()); + let mut sum = Vec::with_capacity(width); + for (i, (&x, &y)) in x.iter().zip(y).enumerate() { + let xc = c.xor(x, carry); + let yc = c.xor(y, carry); + sum.push(c.xor(xc, y)); + // The carry out of the top bit falls off the modulus, so it is never a product. + if i + 1 < width { + let maj = c.and(xc, yc); + carry = c.xor(maj, carry); + } + } + sum +} + +/// `x` shifted by `8·amount` bits, `amount` being three bits: left, or right. +fn shift_bytes(c: &mut Builder, x: &[Wire], amount: &[Wire], left: bool) -> Word { + let mut x = x.to_vec(); + for (stage, &bit) in amount.iter().enumerate() { + let by = 8 << stage; + let from = |x: &[Wire], i: usize| { + if left { + i.checked_sub(by).and_then(|j| x[j]) + } else { + x.get(i + by).copied().flatten() + } + }; + x = (0..64).map(|i| c.mux(bit, from(&x, i), x[i])).collect(); + } + x +} + +/// The width thresholds of a load or a store, from the two bits of `log2` of its +/// width in bytes: at least 2, at least 4, exactly 8. +fn width_thresholds(c: &mut Builder, log_width: &[Wire]) -> [Wire; 3] { + [ + c.or(log_width[0], log_width[1]), + log_width[1], + c.and(log_width[0], log_width[1]), + ] +} + +/// [`super::semantics::bus_address`]: the low three bits are the ones misaligning the access. +fn bus_address(c: &mut Builder, address: &[Wire], thresholds: [Wire; 3]) -> Word { + (0..64) + .map(|i| { + if i < 3 { + c.and(address[i], thresholds[i]) + } else { + address[i] + } + }) + .collect() +} + +/// [`super::Class::Load`]'s circuit: `(v1, imm, flags, cell) -> (address, out)`, the +/// address being what goes on the memory bus and `cell` the word read there. +pub fn load() -> Circuit { + let mut c = Builder::new(&[64, 64, 3, 64], &[64, 64]); + let (v1, imm, flags, cell) = (c.input(0), c.input(1), c.input(2), c.input(3)); + let address = add(&mut c, &v1, &imm); + let [ge2, ge4, eq8] = width_thresholds(&mut c, &flags[..2]); + let bus = bus_address(&mut c, &address, [ge2, ge4, eq8]); + let value = shift_bytes(&mut c, &cell, &address[..3], false); + // The extension: the value's top bit, which the width places, if the load is signed. + let (w1, w2, w4) = (c.not(ge2), c.xor(ge2, ge4), c.xor(ge4, eq8)); + let sign = [(w1, 7), (w2, 15), (w4, 31)] + .into_iter() + .fold(None, |acc, (width, bit)| { + let term = c.and(width, value[bit]); + c.xor(acc, term) + }); + let extension = c.and(flags[2], sign); + for (i, &wire) in bus.iter().enumerate() { + c.output(0, i, wire); + } + for i in 0..64 { + let keeps = match i { + 0..8 => None, + 8..16 => Some(ge2), + 16..32 => Some(ge4), + _ => Some(eq8), + }; + let wire = keeps.map_or(value[i], |keeps| c.mux(keeps, value[i], extension)); + c.output(1, i, wire); + } + c.finish() +} + +/// [`super::Class::Store`]'s circuit: `(v1, v2, imm, flags, cell) -> (address, new cell, +/// out)`. `out` is what the row writes to its destination, the sink: zero. +pub fn store() -> Circuit { + let mut c = Builder::new(&[64, 64, 64, 2, 64], &[64, 64, 64]); + let (v1, v2, imm, flags, cell) = (c.input(0), c.input(1), c.input(2), c.input(3), c.input(4)); + let address = add(&mut c, &v1, &imm); + let [ge2, ge4, eq8] = width_thresholds(&mut c, &flags); + let bus = bus_address(&mut c, &address, [ge2, ge4, eq8]); + let value = shift_bytes(&mut c, &v2, &address[..3], true); + // Byte `j` is written when it shares the access's block: bit `k` of `j` equals bit + // `k` of the address wherever the width does not already span both. + let spans: [[Wire; 2]; 3] = std::array::from_fn(|k| { + let (is_zero, threshold) = (c.not(address[k]), [ge2, ge4, eq8][k]); + [c.or(is_zero, threshold), c.or(address[k], threshold)] + }); + for (i, &wire) in bus.iter().enumerate() { + c.output(0, i, wire); + } + for j in 0..8 { + let low = c.and(spans[0][j & 1], spans[1][(j >> 1) & 1]); + let written = c.and(low, spans[2][j >> 2]); + for i in 8 * j..8 * j + 8 { + let wire = c.mux(written, value[i], cell[i]); + c.output(1, i, wire); + } + } + c.finish() +} + +/// [`super::semantics::shift`]: `(v1, v2, imm, flags) -> out`. One right shifter serves +/// both directions, a left shift being a right shift of the reversed word. +pub fn shift() -> Circuit { + let mut c = Builder::new(&[64, 64, 64, 3], &[64]); + let (v1, v2, imm, f) = (c.input(0), c.input(1), c.input(2), c.input(3)); + let flag = |bit: u64| f[bit.trailing_zeros() as usize]; + use super::shift::*; + let (right, arith, word) = (flag(RIGHT), flag(ARITH), flag(WORD)); + + // The amount: six bits, five for a word shift. + let mut amount: Word = (0..6).map(|i| c.xor(v2[i], imm[i])).collect(); + let not_word = c.not(word); + amount[5] = c.and(not_word, amount[5]); + // A word shift takes the low 32 bits, extended as the shift is: by the sign for an + // arithmetic one, by zero otherwise. + let low_sign = c.and(arith, v1[31]); + let x: Word = (0..64) + .map(|i| if i < 32 { v1[i] } else { c.mux(word, low_sign, v1[i]) }) + .collect(); + // What a right shift brings in from the top. `ARITH` comes with `RIGHT`. + let fill = c.and(arith, x[63]); + + let reversed = |c: &mut Builder, x: &[Wire]| -> Word { (0..64).map(|i| c.mux(right, x[i], x[63 - i])).collect() }; + let mut y = reversed(&mut c, &x); + for (stage, &bit) in amount.iter().enumerate() { + let by = 1 << stage; + y = (0..64) + .map(|i| c.mux(bit, if i + by < 64 { y[i + by] } else { fill }, y[i])) + .collect(); + } + let y = reversed(&mut c, &y); + for (i, &wire) in sext32_if(&mut c, word, &y).iter().enumerate() { + c.output(0, i, wire); + } + c.finish() +} + +/// [`super::semantics::mul`]: `(v1, v2, flags) -> out`, the low word of the product. +pub fn mul() -> Circuit { + let mut c = Builder::new(&[64, 64, 1], &[64]); + let (v1, v2, f) = (c.input(0), c.input(1), c.input(2)); + let (product, _) = flock::arith::mul::Multiplier::build(&mut c, &v1, &v2, 64); + for (i, &wire) in sext32_if(&mut c, f[0], &product).iter().enumerate() { + c.output(0, i, wire); + } + c.finish() +} + +/// [`super::semantics::mulh`]: `(v1, v2, flags) -> out`, the high word of the product. +/// The unsigned product's high word, less `v2` if `v1` is signed and negative, less `v1` +/// if `v2` is: a negative operand is its unsigned reading minus `2^64`. +pub fn mulh() -> Circuit { + let mut c = Builder::new(&[64, 64, 2], &[64]); + let (v1, v2, f) = (c.input(0), c.input(1), c.input(2)); + let (product, _) = flock::arith::mul::Multiplier::build(&mut c, &v1, &v2, 128); + let mut high = product[64..].to_vec(); + for (signed, operand, other) in [(f[0], &v1, &v2), (f[1], &v2, &v1)] { + let negative = c.and(signed, operand[63]); + // `high - other` is `high + !other + 1`. + let subtrahend: Word = other + .iter() + .map(|&bit| { + let inverted = c.not(bit); + c.and(negative, inverted) + }) + .collect(); + (high, _) = add_with_carry(&mut c, &high, &subtrahend, negative); + } + for (i, &wire) in high.iter().enumerate() { + c.output(0, i, wire); + } + c.finish() +} + +/// `x` negated if `negative`: `(x ^ negative) + negative`. +fn negate_if(c: &mut Builder, negative: Wire, x: &[Wire]) -> Word { + let flipped: Word = x.iter().map(|&bit| c.xor(bit, negative)).collect(); + add_with_carry(c, &flipped, &[None; 64], negative).0 +} + +/// [`super::semantics::div`]: `(v1, v2, flags, q, r) -> (out, bad)`. The quotient's and +/// the remainder's magnitudes `q` and `r` are the prover's ([`super::semantics::div_hints`]), +/// and `bad` is set unless they are the ones: `|n| = q·|d| + r` over the integers (the +/// product's high word zero, the sum without a carry) and `r < |d|`. A row puts `bad` +/// where its bytecode entry holds zero, so it is zero. Dividing by zero checks nothing +/// and returns what the specification says, all ones or the dividend, and the one +/// overflow, `-2^63 / -1`, is no special case on magnitudes. +pub fn div() -> Circuit { + let mut c = Builder::new(&[64, 64, 3, 64, 64], &[64, 1]); + let (v1, v2, f, q, r) = (c.input(0), c.input(1), c.input(2), c.input(3), c.input(4)); + let (signed, rem, word) = (f[0], f[1], f[2]); + // A word form divides the low 32 bits, extended as the division is signed or not. + let mut extend = |x: &[Wire]| -> Word { + let sign = c.and(signed, x[31]); + (0..64) + .map(|i| if i < 32 { x[i] } else { c.mux(word, sign, x[i]) }) + .collect() + }; + let (n, d) = (extend(&v1), extend(&v2)); + let (n_negative, d_negative) = (c.and(signed, n[63]), c.and(signed, d[63])); + let (n_abs, d_abs) = (negate_if(&mut c, n_negative, &n), negate_if(&mut c, d_negative, &d)); + + let (product, _) = flock::arith::mul::Multiplier::build(&mut c, &q, &d_abs, 128); + let overflows = any(&mut c, &product[64..]); + let (sum, carries) = add_with_carry(&mut c, &product[..64], &r, None); + let difference = xor_words(&mut c, &sum, &n_abs); + let differs = any(&mut c, &difference); + // `r - |d|` does not borrow, which is `r + !|d| + 1` carrying out, when `r >= |d|`. + let d_inverted: Word = d_abs.iter().map(|&bit| c.not(bit)).collect(); + let one = c.one(); + let (_, too_large) = add_with_carry(&mut c, &r, &d_inverted, one); + let d_nonzero = any(&mut c, &d); + let wrong = [carries, differs, too_large] + .into_iter() + .fold(overflows, |acc, w| c.or(acc, w)); + let bad = c.and(d_nonzero, wrong); + + // The quotient is negative when the operands' signs differ, the remainder when + // the dividend is. + let q_negative = c.xor(n_negative, d_negative); + let (q_signed, r_signed) = (negate_if(&mut c, q_negative, &q), negate_if(&mut c, n_negative, &r)); + let out: Word = (0..64) + .map(|i| { + let result = c.mux(rem, r_signed[i], q_signed[i]); + let by_zero = c.mux(rem, n[i], one); + c.mux(d_nonzero, result, by_zero) + }) + .collect(); + for (i, &wire) in sext32_if(&mut c, word, &out).iter().enumerate() { + c.output(0, i, wire); + } + c.output(1, 0, bad); + c.finish() +} + +/// [`super::Class::Alu`]'s ports. +pub mod alu_ports { + /// Input words: the two register values, the immediate, the flags. + pub const V1: usize = 0; + pub const V2: usize = 1; + pub const IMM: usize = 2; + pub const FLAGS: usize = 3; + /// Output words. + pub const OUT: usize = 4; + pub const TAKEN: usize = 5; +} + +/// [`super::semantics::alu`]. +pub fn alu() -> Circuit { + let mut c = Builder::new(&[64, 64, 64, 15], &[64, 1]); + let (v1, v2, imm, f) = (c.input(0), c.input(1), c.input(2), c.input(3)); + let flag = |bit: u64| f[bit.trailing_zeros() as usize]; + use super::alu::*; + + let b = xor_words(&mut c, &v2, &imm); + // `v1 - b` is `v1 + !b + 1`, and it borrows exactly when that does not carry out. + let sub = flag(SUB); + let b_or_not: Word = b.iter().map(|&bit| c.xor(bit, sub)).collect(); + let (sum, carry_out) = add_with_carry(&mut c, &v1, &b_or_not, sub); + let ltu = c.not(carry_out); + let signs = c.xor(v1[63], b[63]); + let lt = c.xor(ltu, signs); + let diff = xor_words(&mut c, &v1, &b); + let ne = any(&mut c, &diff); + let eq = c.not(ne); + + // `out`: the sum unless a selector is set. AND is a product, OR is AND plus XOR. + let sum = sext32_if(&mut c, flag(WORD), &sum); + let selectors = [SEL_LT, SEL_LTU, SEL_AND, SEL_OR, SEL_XOR]; + let none = selectors.iter().fold(c.one(), |acc, &s| c.xor(acc, flag(s))); + let and_or = c.xor(flag(SEL_AND), flag(SEL_OR)); + let or_xor = c.xor(flag(SEL_OR), flag(SEL_XOR)); + let mut out = gate_word(&mut c, none, &sum); + for i in 0..64 { + let both = c.and(v1[i], b[i]); + let and_term = c.and(and_or, both); + let xor_term = c.and(or_xor, diff[i]); + let logic = c.xor(and_term, xor_term); + out[i] = c.xor(out[i], logic); + } + let lt_term = c.and(flag(SEL_LT), lt); + let ltu_term = c.and(flag(SEL_LTU), ltu); + let compared = c.xor(lt_term, ltu_term); + out[0] = c.xor(out[0], compared); + let keep_bit0 = c.not(flag(CLEAR_BIT0)); + out[0] = c.and(keep_bit0, out[0]); + + let (ge, geu) = (c.not(lt), c.not(ltu)); + let taken = [ + (BR_EQ, eq), + (BR_NE, ne), + (BR_LT, lt), + (BR_GE, ge), + (BR_LTU, ltu), + (BR_GEU, geu), + ] + .into_iter() + .fold(flag(ALWAYS), |acc, (when, holds)| { + let term = c.and(flag(when), holds); + c.xor(acc, term) + }); + + for (i, &wire) in out.iter().enumerate() { + c.output(0, i, wire); + } + c.output(1, 0, taken); + c.finish() +} + +/// [`super::Class::Hash`]'s circuit: `(t, f0, h, m) -> out`, the BLAKE2s compression +/// ([`super::semantics::blake2s`]) on the 32-bit halves of the block's words, `h` +/// and `out` four words each and `m` eight. Every G is six 32-bit additions, its +/// two three-operand ones chained, and the state is never materialized: only the +/// carries are products, and the result's words are copied out. +pub fn blake2s() -> Circuit { + use primitives::hash::{G_LANES, IV, SIGMA}; + let mut c = Builder::new( + &[64, 32, 64, 64, 64, 64, 64, 64, 64, 64, 64, 64, 64, 64], + &[64, 64, 64, 64], + ); + let half = |c: &Builder, port: usize, i: usize| -> Word { c.input(port)[32 * (i % 2)..32 * (i % 2) + 32].to_vec() }; + let (t, f0) = (c.input(0), c.input(1)); + let h: Vec = (0..8).map(|i| half(&c, 2 + i / 2, i)).collect(); + let m: Vec = (0..16).map(|i| half(&c, 6 + i / 2, i)).collect(); + let literal = |c: &Builder, x: u32| -> Word { (0..32).map(|i| c.one().filter(|_| x >> i & 1 == 1)).collect() }; + let rotr = |w: &[Wire], r: usize| -> Word { (0..32).map(|i| w[(i + r) % 32]).collect() }; + + let mut v = h.clone(); + v.extend(IV[..4].iter().map(|&x| literal(&c, x))); + for (i, x) in [&t[..32], &t[32..], &f0, &[None; 32]].into_iter().enumerate() { + let iv = literal(&c, IV[4 + i]); + v.push(xor_words(&mut c, &iv, x)); + } + for round in &SIGMA { + for (g, &[a, b, cc, d]) in G_LANES.iter().enumerate() { + for (x, r1, r2) in [(&m[round[2 * g]], 16, 12), (&m[round[2 * g + 1]], 8, 7)] { + let ab = add(&mut c, &v[a], &v[b]); + v[a] = add(&mut c, &ab, x); + let da = xor_words(&mut c, &v[d], &v[a]); + v[d] = rotr(&da, r1); + v[cc] = add(&mut c, &v[cc], &v[d]); + let bc = xor_words(&mut c, &v[b], &v[cc]); + v[b] = rotr(&bc, r2); + } + } + } + for i in 0..8 { + let hv = xor_words(&mut c, &h[i], &v[i]); + let out = xor_words(&mut c, &hv, &v[i + 8]); + for (bit, &wire) in out.iter().enumerate() { + c.output(i / 2, 32 * (i % 2) + bit, wire); + } + } + c.finish() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::rv::semantics; + use crate::transcript::{ProverState, VerifierState}; + + struct Rng(u64); + impl Rng { + fn next(&mut self) -> u64 { + self.0 ^= self.0 << 13; + self.0 ^= self.0 >> 7; + self.0 ^= self.0 << 17; + self.0 + } + fn word(&mut self) -> u64 { + match self.next() % 6 { + 0 => [ + 0, + 1, + u64::MAX, + 1 << 63, + (1 << 63) - 1, + 1 << 31, + 0xffff_ffff, + (1 << 31) - 1, + ][(self.next() % 8) as usize], + 1 => self.next() as i32 as i64 as u64, + _ => self.next(), + } + } + } + + /// The circuit's output words on `inputs`, read off the witness it generates. + fn run(circuit: &Circuit, inputs: &[u64], outputs: std::ops::Range) -> Vec { + let words = 1 << (circuit.k_log() - 6); + let (mut z, mut az, mut bz) = (vec![0; words], vec![0; words], vec![0; words]); + circuit.witness_instance(inputs, &mut z, &mut az, &mut bz); + z[outputs].to_vec() + } + + #[test] + fn alu_is_its_reference() { + let circuit = alu(); + assert_eq!(circuit.k_log(), 10, "an ALU instance is 16 packed words"); + let mut rng = Rng(0xA1); + for &flags in &crate::rv::alu::LEGAL { + for round in 0..400 { + let v1 = rng.word(); + // Equal operands now and then, which random words never are. + let v2 = if round % 7 == 0 { v1 } else { rng.word() }; + // One of `v2` and `imm` is zero, as the decoder has it. + let (v2, imm) = if round % 2 == 0 { (v2, 0) } else { (0, v2) }; + let (out, taken) = semantics::alu(v1, v2, imm, flags); + assert_eq!( + run(&circuit, &[v1, v2, imm, flags], alu_ports::OUT..alu_ports::TAKEN + 1), + [out, taken as u64], + "flags {flags:#x} on {v1:#x}, {v2:#x}, {imm:#x}" + ); + } + } + } + + #[test] + fn load_and_store_are_their_references() { + let (load, store) = (load(), store()); + assert_eq!( + (load.k_log(), store.k_log()), + (10, 10), + "an instance is 16 packed words" + ); + let mut rng = Rng(0xA3); + for round in 0..4000 { + let (v1, imm, v2, cell) = (rng.word(), rng.next() % 4096, rng.word(), rng.next()); + let address = semantics::address(v1, imm); + for &flags in &crate::rv::load::LEGAL { + let log_width = flags & crate::rv::load::LOG_WIDTH; + // Aligned every other round, which a random address seldom is. + let v1 = if round % 2 == 0 { + v1 & !((1 << log_width) - 1) + } else { + v1 + }; + let imm = if round % 2 == 0 { imm & !7 } else { imm }; + let address = if round % 2 == 0 { + semantics::address(v1, imm) + } else { + address + }; + assert_eq!( + run(&load, &[v1, imm, flags, cell], 4..6), + [ + semantics::bus_address(address, log_width), + semantics::load(cell, address, flags) + ], + "load {flags:#x} at {address:#x} of {cell:#x}" + ); + } + for &flags in &crate::rv::store::LEGAL { + let got = run(&store, &[v1, v2, imm, flags, cell], 5..8); + assert_eq!( + got[0], + semantics::bus_address(address, flags), + "store {flags:#x} at {address:#x}" + ); + assert_eq!(got[2], 0); + // A misaligned store names no cell, so what it would write is nobody's business. + if semantics::is_aligned(address, flags) { + assert_eq!( + got[1], + semantics::store(cell, address, v2, flags), + "store {flags:#x} at {address:#x}" + ); + } + } + } + } + + #[test] + fn shift_and_multiplications_are_their_references() { + let (shift, mul, mulh) = (shift(), mul(), mulh()); + assert_eq!((shift.k_log(), mul.k_log(), mulh.k_log()), (10, 12, 13)); + let mut rng = Rng(0xA4); + for round in 0..1500 { + let (v1, v2) = (rng.word(), rng.word()); + for &flags in &crate::rv::shift::LEGAL { + // Every amount now and then, which a random word's low bits cover slowly. + let amount = if round % 3 == 0 { round as u64 % 64 } else { v2 }; + let (v2, imm) = if round % 2 == 0 { (amount, 0) } else { (0, amount & 63) }; + assert_eq!( + run(&shift, &[v1, v2, imm, flags], 4..5), + [semantics::shift(v1, v2, imm, flags)], + "shift {flags:#x} of {v1:#x} by {v2:#x}, {imm:#x}" + ); + } + if round % 10 == 0 { + for &flags in &crate::rv::mul::LEGAL { + assert_eq!( + run(&mul, &[v1, v2, flags], 3..4), + [semantics::mul(v1, v2, flags)], + "mul {flags:#x}" + ); + } + for &flags in &crate::rv::mulh::LEGAL { + assert_eq!( + run(&mulh, &[v1, v2, flags], 3..4), + [semantics::mulh(v1, v2, flags)], + "mulh {flags:#x} of {v1:#x}, {v2:#x}" + ); + } + } + } + } + + #[test] + fn div_is_its_reference_and_refuses_other_hints() { + let div = div(); + assert_eq!(div.k_log(), 13); + let mut rng = Rng(0xA5); + for round in 0..300 { + let (v1, v2) = ( + rng.word(), + if round % 9 == 0 { + 0 + } else { + rng.word() >> (rng.next() % 64) + }, + ); + for &flags in &crate::rv::div::LEGAL { + let (q, r) = semantics::div_hints(v1, v2, flags); + let expected = [semantics::div(v1, v2, flags), 0]; + assert_eq!( + run(&div, &[v1, v2, flags, q, r], 5..7), + expected, + "div {flags:#x} of {v1:#x} by {v2:#x}" + ); + // Any other quotient or remainder is refused, unless the divisor is zero, + // where they are ignored and the result is still the specification's. + let by_zero = if flags & crate::rv::div::WORD != 0 { + v2 as u32 == 0 + } else { + v2 == 0 + }; + for (q, r) in [ + (q.wrapping_add(1), r), + (q, r.wrapping_add(1)), + (q ^ (1 << 63), r), + (rng.next(), rng.next()), + ] { + let got = run(&div, &[v1, v2, flags, q, r], 5..7); + if by_zero { + assert_eq!(got, expected); + } else { + assert_eq!(got[1], 1, "div {flags:#x} of {v1:#x} by {v2:#x} accepts {q:#x}, {r:#x}"); + } + } + } + } + // The forgery a check modulo 2^64 would accept: 1 / 3 with a quotient of (2^64 + 1) / 3. + assert_eq!(run(&div, &[1, 3, 0, 0x5555_5555_5555_5555, 2], 5..7)[1], 1); + // The overflow, and division by zero, as the specification has them. + let min = i64::MIN as u64; + let (q, r) = semantics::div_hints(min, u64::MAX, crate::rv::div::SIGNED); + assert_eq!( + run(&div, &[min, u64::MAX, crate::rv::div::SIGNED, q, r], 5..7), + [min, 0] + ); + assert_eq!(run(&div, &[7, 0, 0, 0, 0], 5..7), [u64::MAX, 0]); + assert_eq!(run(&div, &[7, 0, crate::rv::div::REM, 0, 0], 5..7), [7, 0]); + } + + #[test] + fn blake2s_is_its_reference() { + let circuit = blake2s(); + assert_eq!(circuit.k_log(), 14, "a compression is 256 packed words"); + let mut rng = Rng(0xA6); + for round in 0..40 { + let block: [u64; 16] = std::array::from_fn(|_| rng.word()); + let t = rng.word(); + // The legal flags, and now and then any finalization word, which the + // circuit XORs in like the reference does. + let flags = match round % 4 { + 0 => crate::rv::hash::FINAL, + 1 => rng.next() as u32 as u64, + _ => 0, + }; + let inputs: Vec = [t, flags] + .into_iter() + .chain(block[..4].iter().chain(&block[8..]).copied()) + .collect(); + let expected = if crate::rv::hash::LEGAL.contains(&flags) { + semantics::blake2s(&block, t, flags) + } else { + let half = |w: &[u64]| -> Vec { w.iter().flat_map(|&w| [w as u32, (w >> 32) as u32]).collect() }; + let out = flock::hash::blake2s_compress( + &half(&block[..4]).try_into().unwrap(), + &half(&block[8..]).try_into().unwrap(), + t, + flags as u32, + 0, + ); + std::array::from_fn(|i| out[2 * i] as u64 | (out[2 * i + 1] as u64) << 32) + }; + assert_eq!(run(&circuit, &inputs, 14..18), expected, "flags {flags:#x}"); + } + } + + /// flock proves a batch of honest instances, and refuses one with a flipped output bit. + #[test] + fn alu_reduction_roundtrip() { + const LABEL: &[u8] = b"rv-alu-reduction-test"; + let circuit = alu(); + let block = circuit.block(); + let n_log = 4; + let mut rng = Rng(0xA2); + let legal = crate::rv::alu::LEGAL; + let rows: Vec<[u64; 4]> = (0..1 << n_log) + .map(|i| [rng.word(), rng.word(), 0, legal[i % legal.len()]]) + .collect(); + let run = |tamper: Option| { + let (mut z, a, b, mut z_lincheck) = circuit.generate_witness(&rows, n_log); + if let Some(bit) = tamper { + z[bit / 64] ^= 1 << (bit % 64); + z_lincheck[bit] ^= 1; + } + let mut ps = ProverState::from_label(LABEL); + let stage = block.prove_zerocheck(n_log, &z, &a, &b, &mut ps); + let claim = block.prove_lincheck(n_log, stage, &z_lincheck, &mut ps); + let proof = ps.into_proof(); + let mut vs = VerifierState::from_label(LABEL, &proof); + block.verify(n_log, &mut vs).is_ok_and(|r| r.claim == claim) && vs.finish().is_ok() + }; + assert!(run(None)); + // An output bit, a spare bit of `taken`'s word, a product. + for bit in [ + 64 * alu_ports::OUT + 5, + 64 * alu_ports::TAKEN + 1, + circuit.useful_bits() - 1, + ] { + assert!(!run(Some(bit)), "flipping bit {bit} must reject"); + } + } +} diff --git a/crates/lean_vm/src/rv/decode.rs b/crates/lean_vm/src/rv/decode.rs new file mode 100644 index 000000000..e5341e58d --- /dev/null +++ b/crates/lean_vm/src/rv/decode.rs @@ -0,0 +1,211 @@ +//! The decoder: a 32-bit word at `pc` to its [`Entry`]. Anything rv64im does not +//! define, a reserved encoding included, is [`Entry::ILLEGAL`]. + +use super::{Class, Entry, SINK, Target, alu, div, hash, load, mul, mulh, shift, store}; + +/// Sign-extend the low `bits` bits of `x`. +fn sext(x: u32, bits: u32) -> u64 { + (((x as u64) << (64 - bits)) as i64 >> (64 - bits)) as u64 +} + +fn entry(class: Class, flags: u64, a1: u32, a2: u32, rd: u32, imm: u64) -> Entry { + Entry { + class, + flags, + a1: a1 as u8, + a2: a2 as u8, + ad: if rd == 0 { SINK } else { rd as u8 }, + imm, + target: Target::Next, + link: false, + jalr: false, + } +} + +pub fn decode(word: u32, pc: u64) -> Entry { + let (opcode, rd, f3) = (word & 0x7f, (word >> 7) & 31, (word >> 12) & 7); + let (rs1, rs2, f7) = ((word >> 15) & 31, (word >> 20) & 31, word >> 25); + let imm_i = sext(word >> 20, 12); + let imm_s = sext((f7 << 5) | rd, 12); + let imm_b = sext( + ((word >> 31) << 12) | (((word >> 7) & 1) << 11) | (((word >> 25) & 0x3f) << 5) | (((word >> 8) & 0xf) << 1), + 13, + ); + let imm_u = sext(word & 0xffff_f000, 32); + let imm_j = sext( + ((word >> 31) << 20) + | (((word >> 12) & 0xff) << 12) + | (((word >> 20) & 1) << 11) + | (((word >> 21) & 0x3ff) << 1), + 21, + ); + // A register-register and a register-immediate instruction are one entry: the + // first has no immediate, the second reads `x0`. + let reg = |class, flags| entry(class, flags, rs1, rs2, rd, 0); + let imm = |class, flags, imm| entry(class, flags, rs1, 0, rd, imm); + + match opcode { + // LUI and AUIPC: the constant, added to `x0`. + 0x37 => entry(Class::Alu, 0, 0, 0, rd, imm_u), + 0x17 => entry(Class::Alu, 0, 0, 0, rd, pc.wrapping_add(imm_u)), + // JAL + 0x6f => Entry { + target: Target::Abs(pc.wrapping_add(imm_j)), + link: true, + ..entry(Class::Alu, alu::ALWAYS, 0, 0, rd, 0) + }, + // JALR + 0x67 if f3 == 0 => Entry { + link: true, + jalr: true, + ..imm(Class::Alu, alu::CLEAR_BIT0, imm_i) + }, + // Branches + 0x63 => { + let when = match f3 { + 0 => alu::BR_EQ, + 1 => alu::BR_NE, + 4 => alu::BR_LT, + 5 => alu::BR_GE, + 6 => alu::BR_LTU, + 7 => alu::BR_GEU, + _ => return Entry::ILLEGAL, + }; + Entry { + target: Target::Abs(pc.wrapping_add(imm_b)), + ..entry(Class::Alu, alu::SUB | when, rs1, rs2, 0, 0) + } + } + // Loads + 0x03 => { + let flags = match f3 { + 0..=2 => load::SIGNED | f3 as u64, + // A 64-bit load has no extension. + 3 => 3, + 4..=6 => (f3 - 4) as u64, + _ => return Entry::ILLEGAL, + }; + imm(Class::Load, flags, imm_i) + } + // Stores + 0x23 if f3 <= 3 => entry(Class::Store, f3 as u64 & store::LOG_WIDTH, rs1, rs2, 0, imm_s), + // Register-immediate + 0x13 => match f3 { + 0 => imm(Class::Alu, 0, imm_i), + 2 => imm(Class::Alu, alu::SUB | alu::SEL_LT, imm_i), + 3 => imm(Class::Alu, alu::SUB | alu::SEL_LTU, imm_i), + 4 => imm(Class::Alu, alu::SEL_XOR, imm_i), + 6 => imm(Class::Alu, alu::SEL_OR, imm_i), + 7 => imm(Class::Alu, alu::SEL_AND, imm_i), + // The shift amount has six bits, so the function field is the other six. + 1 if word >> 26 == 0 => imm(Class::Shift, 0, (word >> 20 & 63) as u64), + 5 if word >> 26 == 0 => imm(Class::Shift, shift::RIGHT, (word >> 20 & 63) as u64), + 5 if word >> 26 == 0x10 => imm(Class::Shift, shift::RIGHT | shift::ARITH, (word >> 20 & 63) as u64), + _ => Entry::ILLEGAL, + }, + // Register-immediate, 32-bit: a shift amount of five bits. + 0x1b => match (f3, f7) { + (0, _) => imm(Class::Alu, alu::WORD, imm_i), + (1, 0) => imm(Class::Shift, shift::WORD, rs2 as u64), + (5, 0) => imm(Class::Shift, shift::WORD | shift::RIGHT, rs2 as u64), + (5, 0x20) => imm(Class::Shift, shift::WORD | shift::RIGHT | shift::ARITH, rs2 as u64), + _ => Entry::ILLEGAL, + }, + // Register-register + 0x33 => match (f7, f3) { + (0, 0) => reg(Class::Alu, 0), + (0x20, 0) => reg(Class::Alu, alu::SUB), + (0, 1) => reg(Class::Shift, 0), + (0, 2) => reg(Class::Alu, alu::SUB | alu::SEL_LT), + (0, 3) => reg(Class::Alu, alu::SUB | alu::SEL_LTU), + (0, 4) => reg(Class::Alu, alu::SEL_XOR), + (0, 5) => reg(Class::Shift, shift::RIGHT), + (0x20, 5) => reg(Class::Shift, shift::RIGHT | shift::ARITH), + (0, 6) => reg(Class::Alu, alu::SEL_OR), + (0, 7) => reg(Class::Alu, alu::SEL_AND), + (1, 0) => reg(Class::Mul, 0), + (1, 1) => reg(Class::Mulh, mulh::SIGNED_1 | mulh::SIGNED_2), + (1, 2) => reg(Class::Mulh, mulh::SIGNED_1), + (1, 3) => reg(Class::Mulh, 0), + (1, 4) => reg(Class::Div, div::SIGNED), + (1, 5) => reg(Class::Div, 0), + (1, 6) => reg(Class::Div, div::SIGNED | div::REM), + (1, 7) => reg(Class::Div, div::REM), + _ => Entry::ILLEGAL, + }, + // Register-register, 32-bit + 0x3b => match (f7, f3) { + (0, 0) => reg(Class::Alu, alu::WORD), + (0x20, 0) => reg(Class::Alu, alu::SUB | alu::WORD), + (0, 1) => reg(Class::Shift, shift::WORD), + (0, 5) => reg(Class::Shift, shift::WORD | shift::RIGHT), + (0x20, 5) => reg(Class::Shift, shift::WORD | shift::RIGHT | shift::ARITH), + (1, 0) => reg(Class::Mul, mul::WORD), + (1, 4) => reg(Class::Div, div::WORD | div::SIGNED), + (1, 5) => reg(Class::Div, div::WORD), + (1, 6) => reg(Class::Div, div::WORD | div::SIGNED | div::REM), + (1, 7) => reg(Class::Div, div::WORD | div::REM), + _ => Entry::ILLEGAL, + }, + // FENCE: a no-op, whatever its other fields hold. + 0x0f if f3 == 0 => entry(Class::Alu, 0, 0, 0, 0, 0), + // The BLAKE2s precompile: the block at rs1, the counter in rs2, no destination. + hash::OPCODE if f7 == 0 && rd == 0 && f3 <= 1 => { + entry(Class::Hash, if f3 == 1 { hash::FINAL } else { 0 }, rs1, rs2, 0, 0) + } + // ECALL: a jump to the halt slot. EBREAK and the CSR instructions are illegal. + 0x73 if word == 0x73 => Entry { + target: Target::Halt, + ..entry(Class::Alu, alu::ALWAYS, 0, 0, 0, 0) + }, + _ => Entry::ILLEGAL, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Every entry the decoder can produce is well formed, over a sweep dense in the + /// fields that select an instruction. + #[test] + fn decoded_entries_are_well_formed() { + for opcode in 0..128u32 { + for f3 in 0..8u32 { + for top in 0..(1u32 << 12) { + let word = opcode | (f3 << 12) | (top << 20) | (0x15 << 7) | (0x0a << 15); + let e = decode(word, super::super::TEXT_BASE); + assert!(e.is_well_formed(), "{word:#010x} decodes to {e:?}"); + } + } + } + } + + /// The encodings RV64 reserves among the shifts, and what is not rv64im at all. + #[test] + fn reserved_encodings_are_illegal() { + let illegal = [ + 0x0400_1013u32, // SLLI with a function bit set + 0x0200_101b, // SLLIW with bit 5 of the amount + 0x0200_501b, // SRLIW with bit 5 of the amount + 0x4200_501b, // SRAIW with bit 5 of the amount + 0x0010_0073, // EBREAK + 0x3000_2073, // CSRRS mstatus + 0x0000_100f, // FENCE.I + 0x0000_0000, + 0xffff_ffff, + 0x0000_7003, // a load of width 7 + 0x0000_4023, // a store of width 4 + 0x0000_2063, // a branch with function 2 + 0x0000_1067, // JALR with a nonzero function + 0x0000_208b, // BLAKE2S with function 2 + 0x0000_008b | 5 << 7, // BLAKE2S with a destination + 0x0200_000b, // BLAKE2S with a function-7 bit set + ]; + for word in illegal { + assert_eq!(decode(word, 0), Entry::ILLEGAL, "{word:#010x}"); + } + // SRAI by 63 is legal on RV64. + assert_eq!(decode(0x43f5_5513, 0).class, Class::Shift); + } +} diff --git a/crates/lean_vm/src/rv/elf.rs b/crates/lean_vm/src/rv/elf.rs new file mode 100644 index 000000000..af35236d3 --- /dev/null +++ b/crates/lean_vm/src/rv/elf.rs @@ -0,0 +1,235 @@ +//! Loading a [`Guest`]: a static RV64 ELF executable to a program's text, entry point +//! and RAM. Anything the machine could not run as the file means it to be run is +//! refused here, not discovered as a trap. + +use super::{ADVICE_BASE, INPUT_WORDS, MAX_LOG_ADVICE, MAX_LOG_RAM, MAX_LOG_TEXT, RAM_BASE, TEXT_BASE}; + +/// What a program is made from. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Guest { + /// The text's words, the first at [`TEXT_BASE`]. + pub text: Vec, + pub entry_pc: u64, + /// RAM's words after the input's. + pub image: Vec, + pub log_ram: usize, + pub log_advice: usize, +} + +/// Why a file is no guest. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ElfError(pub &'static str); + +impl std::fmt::Display for ElfError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "not a leanVM guest: {}", self.0) + } +} + +const ET_EXEC: u16 = 2; +const EM_RISCV: u16 = 243; +/// `e_flags`: compressed instructions, and the float ABIs. +const EF_RISCV_RVC: u32 = 1; +const EF_RISCV_FLOAT_ABI: u32 = 6; +const PT_LOAD: u32 = 1; +const PT_DYNAMIC: u32 = 2; +const PT_INTERP: u32 = 3; +const PT_TLS: u32 = 7; +const PF_X: u32 = 1; +const PF_W: u32 = 2; +const SHT_SYMTAB: u32 = 2; + +/// The symbols whose values are the ends of RAM and of the advice region, which the +/// linker script defines. +const RAM_END_SYMBOL: &[u8] = b"__stack_top"; +const ADVICE_END_SYMBOL: &[u8] = b"__advice_top"; + +/// `base + offset`, which a malformed header can wrap: a wrapped address would name +/// a field inside the file that the header did not point at. +fn at(base: u64, offset: u64) -> Result { + base.checked_add(offset).ok_or(ElfError("a field's address wraps")) +} + +/// Little-endian fields of `bytes`, every read checked. +struct Reader<'a>(&'a [u8]); + +impl Reader<'_> { + fn bytes(&self, at: u64, len: u64) -> Result<&[u8], ElfError> { + let end = at.checked_add(len).filter(|&end| end <= self.0.len() as u64); + end.map(|end| &self.0[at as usize..end as usize]) + .ok_or(ElfError("a field runs past the end of the file")) + } + fn u16(&self, at: u64) -> Result { + Ok(u16::from_le_bytes(self.bytes(at, 2)?.try_into().unwrap())) + } + fn u32(&self, at: u64) -> Result { + Ok(u32::from_le_bytes(self.bytes(at, 4)?.try_into().unwrap())) + } + fn u64(&self, at: u64) -> Result { + Ok(u64::from_le_bytes(self.bytes(at, 8)?.try_into().unwrap())) + } +} + +/// `bytes` written at `offset` of a buffer of whole `WORD`-byte words, grown to hold them. +fn place(buffer: &mut Vec, offset: u64, bytes: &[u8]) { + let end = offset as usize + bytes.len(); + if buffer.len() < end { + buffer.resize(end.next_multiple_of(WORD), 0); + } + buffer[offset as usize..end].copy_from_slice(bytes); +} + +impl Guest { + /// A static RV64 executable linked with the guests' linker script: its executable + /// segments are the text, every other loaded segment is RAM's image, RAM ends + /// where the script says the stack starts, and the advice region where it says. + pub fn from_elf(elf: &[u8]) -> Result { + let r = Reader(elf); + // 64-bit, little-endian, version 1. + if r.bytes(0, 7)? != [0x7f, b'E', b'L', b'F', 2, 1, 1] { + return Err(ElfError("not a little-endian ELF64 file")); + } + if r.u16(16)? != ET_EXEC { + return Err(ElfError("not a static executable (a PIE would need relocating)")); + } + if r.u16(18)? != EM_RISCV { + return Err(ElfError("not for RISC-V")); + } + let flags = r.u32(48)?; + if flags & EF_RISCV_RVC != 0 { + return Err(ElfError("compressed instructions")); + } + if flags & EF_RISCV_FLOAT_ABI != 0 { + return Err(ElfError("a hardware float ABI")); + } + let entry_pc = r.u64(24)?; + + let (mut text, mut image) = (Vec::new(), Vec::new()); + let (phoff, phentsize, phnum) = (r.u64(32)?, r.u16(54)? as u64, r.u16(56)? as u64); + // What a region may hold is capped by the map, but a buffer here is grown to a + // segment's own address, so a handful of bytes at the top of a region would + // allocate the whole of it. A program gets no more text and no more image than + // its file carries: the rest would be zeros, which is no instruction and no data + // the program did not write itself. + let file_len = elf.len() as u64; + for i in 0..phnum { + let ph = at( + phoff, + i.checked_mul(phentsize).ok_or(ElfError("a header's address wraps"))?, + )?; + let (kind, flags) = (r.u32(ph)?, r.u32(at(ph, 4)?)?); + if matches!(kind, PT_DYNAMIC | PT_INTERP | PT_TLS) { + return Err(ElfError("dynamic linking or thread-local storage")); + } + if kind != PT_LOAD { + continue; + } + let (offset, vaddr, filesz, memsz) = ( + r.u64(at(ph, 8)?)?, + r.u64(at(ph, 16)?)?, + r.u64(at(ph, 32)?)?, + r.u64(at(ph, 40)?)?, + ); + let end = vaddr + .checked_add(memsz) + .ok_or(ElfError("a segment wraps the address space"))?; + let bytes = r.bytes(offset, filesz.min(memsz))?; + if flags & PF_X != 0 { + if flags & PF_W != 0 { + return Err(ElfError("a writable executable segment")); + } + if vaddr < TEXT_BASE || vaddr % 4 != 0 || end > TEXT_BASE + (4 << MAX_LOG_TEXT) { + return Err(ElfError("an executable segment outside the text region")); + } + if vaddr - TEXT_BASE + bytes.len() as u64 > file_len { + return Err(ElfError("more text than the file carries")); + } + place::<4>(&mut text, vaddr - TEXT_BASE, bytes); + } else { + // RAM's first words are the input's: a segment may reserve them, not fill them. + let first = RAM_BASE + 8 * INPUT_WORDS as u64; + if vaddr < RAM_BASE || end > RAM_BASE + (8 << MAX_LOG_RAM) || (vaddr < first && !bytes.is_empty()) { + return Err(ElfError("a data segment outside RAM, or over the input words")); + } + if !bytes.is_empty() { + if vaddr - first + bytes.len() as u64 > file_len { + return Err(ElfError("more image than the file carries")); + } + place::<8>(&mut image, vaddr - first, bytes); + } + } + } + if text.is_empty() { + return Err(ElfError("no executable segment")); + } + + let ram_end = + symbol(&r, RAM_END_SYMBOL)?.ok_or(ElfError("no __stack_top symbol: not linked with the guests' script"))?; + let ram_bytes = ram_end.wrapping_sub(RAM_BASE); + if ram_end <= RAM_BASE || !ram_bytes.is_power_of_two() || ram_bytes < 8 * (INPUT_WORDS as u64 + 1) { + return Err(ElfError("RAM's size is not a power of two")); + } + let log_ram = (ram_bytes / 8).trailing_zeros() as usize; + let advice_end = symbol(&r, ADVICE_END_SYMBOL)?.ok_or(ElfError("no __advice_top symbol"))?; + let advice_bytes = advice_end.wrapping_sub(ADVICE_BASE); + if advice_end <= ADVICE_BASE || !advice_bytes.is_power_of_two() || advice_bytes < 8 { + return Err(ElfError("the advice region's size is not a power of two")); + } + let log_advice = (advice_bytes / 8).trailing_zeros() as usize; + if log_advice > MAX_LOG_ADVICE { + return Err(ElfError("the advice region exceeds its bounds")); + } + let text: Vec = text.as_chunks::<4>().0.iter().map(|&w| u32::from_le_bytes(w)).collect(); + let image: Vec = image + .as_chunks::<8>() + .0 + .iter() + .map(|&w| u64::from_le_bytes(w)) + .collect(); + if log_ram > MAX_LOG_RAM || INPUT_WORDS + image.len() > 1 << log_ram { + return Err(ElfError("the image does not fit RAM")); + } + Ok(Self { + text, + entry_pc, + image, + log_ram, + log_advice, + }) + } +} + +/// The value of the symbol `name`, from the file's symbol table. +fn symbol(r: &Reader, name: &[u8]) -> Result, ElfError> { + let (shoff, shentsize, shnum) = (r.u64(40)?, r.u16(58)? as u64, r.u16(60)? as u64); + for i in 0..shnum { + let sh = at( + shoff, + i.checked_mul(shentsize).ok_or(ElfError("a section's address wraps"))?, + )?; + if r.u32(at(sh, 4)?)? != SHT_SYMTAB { + continue; + } + let (offset, size, link, entsize) = ( + r.u64(at(sh, 24)?)?, + r.u64(at(sh, 32)?)?, + r.u32(at(sh, 40)?)? as u64, + r.u64(at(sh, 56)?)?, + ); + if entsize == 0 || link >= shnum { + return Err(ElfError("a malformed symbol table")); + } + let strings = at(shoff, link * shentsize)?; + let (str_offset, str_size) = (r.u64(at(strings, 24)?)?, r.u64(at(strings, 32)?)?); + for s in 0..size / entsize { + let sym = at(offset, s * entsize)?; + let name_at = r.u32(sym)? as u64; + if name_at + (name.len() as u64) < str_size + && r.bytes(at(str_offset, name_at)?, name.len() as u64 + 1)? == [name, &[0]].concat() + { + return r.u64(at(sym, 8)?).map(Some); + } + } + } + Ok(None) +} diff --git a/crates/lean_vm/src/rv/machine.rs b/crates/lean_vm/src/rv/machine.rs new file mode 100644 index 000000000..98c701dca --- /dev/null +++ b/crates/lean_vm/src/rv/machine.rs @@ -0,0 +1,730 @@ +//! The interpreter: the reference semantics of a [`Program`], one [`Step`] at a time. + +use super::{ + ADVICE_BASE, Class, Entry, INPUT_WORDS, LOG_REGS, MAX_LOG_ADVICE, MAX_LOG_RAM, MAX_LOG_TEXT, OUTPUT_REGS, RAM_BASE, + SYS_EXIT, SYSCALL_REG, TEXT_BASE, +}; +use super::{Target, decode, hash, load, semantics, store}; + +/// A decoded program: its text, where it starts, RAM as the run finds it, and the +/// advice's size. +#[derive(Clone, Debug)] +pub struct Program { + /// A power of two of entries, instruction `i` at [`Self::pc_of`]`(i)`. The last + /// is the halt slot, which is illegal, as is every slot holding no instruction. + pub entries: Vec, + pub entry_pc: u64, + /// RAM's words after the [`INPUT_WORDS`] the public input takes. The rest of its + /// `2^log_ram` words are zero. + pub image: Vec, + pub log_ram: usize, + /// The advice region holds `2^log_advice` words ([`ADVICE_BASE`]). + pub log_advice: usize, +} + +impl Program { + /// Decode `text`, whose first word sits at [`TEXT_BASE`]. + pub fn new(text: &[u32], entry_pc: u64, image: Vec, log_ram: usize, log_advice: usize) -> Self { + let mut entries: Vec = text + .iter() + .enumerate() + .map(|(i, &word)| decode(word, TEXT_BASE + 4 * i as u64)) + .collect(); + // Two more slots at least: the halt slot, and an illegal one before it, so that + // a run falling off the text traps instead of halting. + entries.resize((text.len() + 2).next_power_of_two(), Entry::ILLEGAL); + assert!( + entries.len() <= 1 << MAX_LOG_TEXT, + "the text exceeds 2^{MAX_LOG_TEXT} instructions" + ); + assert!( + (2..=MAX_LOG_RAM).contains(&log_ram) && INPUT_WORDS + image.len() <= 1 << log_ram, + "RAM is too small for its image, or exceeds its region" + ); + assert!(log_advice <= MAX_LOG_ADVICE, "the advice exceeds its region"); + Self { + entries, + entry_pc, + image, + log_ram, + log_advice, + } + } + + pub fn pc_of(&self, index: usize) -> u64 { + TEXT_BASE + 4 * index as u64 + } + + /// The entry at `pc`, if `pc` names one. + pub fn index_of(&self, pc: u64) -> Option { + let offset = pc.wrapping_sub(TEXT_BASE); + (offset.is_multiple_of(4) && offset / 4 < self.entries.len() as u64).then_some((offset / 4) as usize) + } + + /// Where a run ends: the last slot, which is never executed. + pub fn halt_pc(&self) -> u64 { + self.pc_of(self.entries.len() - 1) + } + + /// The bytecode's `dt` field: a taken entry's target, as a XOR against `pc + 4`. + pub fn dt_of(&self, index: usize) -> u64 { + self.target_of(index) + .map_or(0, |target| target ^ self.pc_of(index).wrapping_add(4)) + } + + /// Where entry `index` goes when its class takes the jump. + pub fn target_of(&self, index: usize) -> Option { + match self.entries[index].target { + Target::Next => None, + Target::Abs(target) => Some(target), + Target::Halt => Some(self.halt_pc()), + } + } +} + +/// Why a run stops without halting. No proof exists of a run that traps. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Trap { + /// `pc` names no instruction: outside the text, misaligned, or an illegal one. + Illegal { + pc: u64, + }, + Misaligned { + pc: u64, + address: u64, + }, + /// An access outside RAM. + Unmapped { + pc: u64, + address: u64, + }, + /// An `ECALL` that is not `exit`. + NotAnExit { + syscall: u64, + }, + CycleCap, + /// The run is sound but longer than one proof holds: its witness would be + /// `2^log_words` words. + TooLong { + log_words: usize, + }, +} + +impl std::fmt::Display for Trap { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match *self { + Self::Illegal { pc } => write!(f, "no legal instruction at pc {pc:#x}"), + Self::Misaligned { pc, address } => write!(f, "misaligned access to {address:#x} at pc {pc:#x}"), + Self::Unmapped { pc, address } => write!(f, "access outside RAM, to {address:#x}, at pc {pc:#x}"), + Self::NotAnExit { syscall } => write!(f, "ecall {syscall} is not exit"), + Self::CycleCap => write!(f, "the run exceeds its cycle cap"), + Self::TooLong { log_words } => write!( + f, + "the run's witness is 2^{log_words} words, more than one proof holds (continuations are not implemented)" + ), + } + } +} + +/// The RAM cell a step accessed. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct RamAccess { + /// What goes on the memory bus ([`semantics::bus_address`]): the word's byte + /// address, for an aligned access. + pub address: u64, + pub old: u64, + pub new: u64, +} + +/// The block a hash row works on ([`hash`]): its words as the row found them, and +/// the four it leaves in the result's. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct HashAccess { + pub block: [u64; hash::WORDS], + pub out: [u64; 4], +} + +/// One executed instruction, as a row of its class's table records it. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Step { + /// The entry executed. + pub index: usize, + pub v1: u64, + pub v2: u64, + /// What the class computed. + pub out: u64, + pub taken: bool, + /// What `ad` held, and what it holds now: `out`, or `pc + 4` for a link. Zeros + /// for a hash row, which writes no register. + pub vd_old: u64, + pub vd: u64, + pub ram: Option, + pub hash: Option>, + pub npc: u64, +} + +/// What entry `e` computes from the registers it reads and, for a load or a store, the +/// RAM cell it names: `(out, taken, RAM access)`, whether or not a run could make +/// that access. +pub fn compute(e: &Entry, v1: u64, v2: u64, cell: u64) -> (u64, bool, RamAccess) { + let address = semantics::address(v1, e.imm); + let ram = |new: u64, log_width: u64| RamAccess { + address: semantics::bus_address(address, log_width), + old: cell, + new, + }; + let none = RamAccess::default(); + match e.class { + Class::Alu => { + let (out, taken) = semantics::alu(v1, v2, e.imm, e.flags); + (out, taken, none) + } + Class::Shift => (semantics::shift(v1, v2, e.imm, e.flags), false, none), + Class::Mul => (semantics::mul(v1, v2, e.flags), false, none), + Class::Mulh => (semantics::mulh(v1, v2, e.flags), false, none), + Class::Div => (semantics::div(v1, v2, e.flags), false, none), + Class::Load => ( + semantics::load(cell, address, e.flags), + false, + ram(cell, e.flags & load::LOG_WIDTH), + ), + Class::Store => ( + 0, + false, + ram(semantics::store(cell, address, v2, e.flags), e.flags & store::LOG_WIDTH), + ), + Class::Hash | Class::Illegal => (0, false, none), + } +} + +/// What a hash row does to its block: [`semantics::blake2s`] of the block's words. +pub fn compute_hash(block: [u64; hash::WORDS], t: u64, flags: u64) -> HashAccess { + HashAccess { + block, + out: semantics::blake2s(&block, t, flags), + } +} + +pub struct Machine<'a> { + pub program: &'a Program, + /// `x0..x31`, then [`super::SINK`]. + pub regs: [u64; 1 << LOG_REGS], + /// RAM's cells, then the advice's. + mem: Vec, + pub pc: u64, +} + +impl<'a> Machine<'a> { + /// The machine about to run `program` on `input`, RAM's first words, and `advice`, + /// the advice region's first words. + pub fn new(program: &'a Program, input: [u64; INPUT_WORDS], advice: &[u64]) -> Self { + assert!( + advice.len() <= 1 << program.log_advice, + "the advice does not fit its region" + ); + let mut mem = input.to_vec(); + mem.extend(&program.image); + mem.resize(1 << program.log_ram, 0); + mem.extend(advice); + mem.resize((1 << program.log_ram) + (1 << program.log_advice), 0); + Self { + program, + regs: [0; 1 << LOG_REGS], + mem, + pc: program.entry_pc, + } + } + + pub fn ram(&self) -> &[u64] { + &self.mem[..1 << self.program.log_ram] + } + + pub fn advice(&self) -> &[u64] { + &self.mem[1 << self.program.log_ram..] + } + + /// The cell holding `address`: RAM's, or past them the advice's. + fn cell(&self, address: u64) -> Result { + let (ram, advice) = ( + address.wrapping_sub(RAM_BASE) / 8, + address.wrapping_sub(ADVICE_BASE) / 8, + ); + if address >= RAM_BASE && ram < 1 << self.program.log_ram { + Ok(ram as usize) + } else if address >= ADVICE_BASE && advice < 1 << self.program.log_advice { + Ok((1 << self.program.log_ram) + advice as usize) + } else { + Err(Trap::Unmapped { pc: self.pc, address }) + } + } + + pub fn halted(&self) -> bool { + self.pc == self.program.halt_pc() + } + + pub fn step(&mut self) -> Result { + let pc = self.pc; + let index = self.program.index_of(pc).ok_or(Trap::Illegal { pc })?; + let e = self.program.entries[index]; + let (v1, v2) = (self.regs[e.a1 as usize], self.regs[e.a2 as usize]); + let cell = match e.class { + Class::Load | Class::Store => { + let address = semantics::address(v1, e.imm); + if !semantics::is_aligned(address, e.flags & load::LOG_WIDTH) { + return Err(Trap::Misaligned { pc, address }); + } + Some(self.cell(address)?) + } + Class::Illegal => return Err(Trap::Illegal { pc }), + _ => None, + }; + let (out, taken, access) = compute(&e, v1, v2, cell.map_or(0, |cell| self.mem[cell])); + let ram = cell.map(|cell| { + self.mem[cell] = access.new; + access + }); + let hash = match e.class { + Class::Hash => Some(Box::new(self.hash(pc, v1, v2, e.flags)?)), + _ => None, + }; + let pc4 = pc.wrapping_add(4); + let vd = if e.link { pc4 } else { out }; + let vd_old = match e.class { + Class::Hash => 0, + _ => std::mem::replace(&mut self.regs[e.ad as usize], vd), + }; + let npc = match (e.jalr, taken) { + (true, _) => out, + (false, true) => self.program.target_of(index).expect("a taken entry has a target"), + (false, false) => pc4, + }; + self.pc = npc; + Ok(Step { + index, + v1, + v2, + out, + taken, + vd_old, + vd, + ram, + hash, + npc, + }) + } + + /// A hash row's block ([`hash`]): word `k` is the cell at `base ^ 8k`, so `base` + /// has to be a word address and every word of the block in RAM. + fn hash(&mut self, pc: u64, base: u64, t: u64, flags: u64) -> Result { + if !base.is_multiple_of(8) { + return Err(Trap::Misaligned { pc, address: base }); + } + let mut cells = [0usize; hash::WORDS]; + for (k, cell) in cells.iter_mut().enumerate() { + *cell = self.cell(base ^ (8 * k as u64))?; + } + let access = compute_hash(cells.map(|cell| self.mem[cell]), t, flags); + for (j, &out) in access.out.iter().enumerate() { + self.mem[cells[hash::OUT as usize / 8 + j]] = out; + } + Ok(access) + } + + /// Run to the halt slot, within `cycle_cap` steps, and return the public output `a0..a3`. + pub fn run(&mut self, cycle_cap: u64) -> Result<[u64; 4], Trap> { + for _ in 0..cycle_cap { + if self.halted() { + let syscall = self.regs[SYSCALL_REG as usize]; + return if syscall == SYS_EXIT { + Ok(OUTPUT_REGS.map(|r| self.regs[r as usize])) + } else { + Err(Trap::NotAnExit { syscall }) + }; + } + self.step()?; + } + Err(Trap::CycleCap) + } +} + +#[cfg(test)] +mod tests { + use super::super::asm::{self, *}; + use super::*; + + const LOG_RAM: usize = 8; + const RAM_BYTES: u64 = 8 << LOG_RAM; + + struct Rng(u64); + impl Rng { + fn next(&mut self) -> u64 { + self.0 ^= self.0 << 13; + self.0 ^= self.0 >> 7; + self.0 ^= self.0 << 17; + self.0 + } + fn below(&mut self, n: u64) -> u64 { + self.next() % n + } + /// A word biased toward the values arithmetic gets wrong. + fn word(&mut self) -> u64 { + match self.below(8) { + 0 => [ + 0, + 1, + u64::MAX, + 1 << 63, + (1 << 63) - 1, + 1 << 31, + (1 << 31) - 1, + 0xffff_ffff, + ][self.below(8) as usize], + 1 => self.next() as u32 as u64, + 2 => self.next() as i32 as i64 as u64, + 3 => self.below(65), + _ => self.next(), + } + } + } + + /// RISC-V straight from the specification, on registers and BYTES: shares no code + /// with the decoder, the class functions or the word-addressed RAM. + fn spec_step(word: u32, x: &mut [u64; 32], pc: &mut u64, mem: &mut [u8]) { + let (opcode, rd, f3) = (word & 0x7f, (word >> 7 & 31) as usize, word >> 12 & 7); + let (rs1, rs2, f7) = ((word >> 15 & 31) as usize, (word >> 20 & 31) as usize, word >> 25); + let (a, b) = (x[rs1], x[rs2]); + let imm_i = (word as i32 >> 20) as i64 as u64; + let at = |address: u64| (address - RAM_BASE) as usize; + let w = |v: u64| v as i32 as i64 as u64; + let mut next = pc.wrapping_add(4); + let mut result = None; + match opcode { + 0x37 => result = Some((word & 0xffff_f000) as i32 as i64 as u64), + 0x17 => result = Some(pc.wrapping_add((word & 0xffff_f000) as i32 as i64 as u64)), + 0x6f => { + let o = ((word >> 31) << 20) + | ((word >> 12 & 0xff) << 12) + | ((word >> 20 & 1) << 11) + | ((word >> 21 & 0x3ff) << 1); + result = Some(next); + next = pc.wrapping_add(((o << 11) as i32 >> 11) as i64 as u64); + } + 0x67 => { + result = Some(next); + next = a.wrapping_add(imm_i) & !1; + } + 0x63 => { + let o = ((word >> 31) << 12) + | ((word >> 7 & 1) << 11) + | ((word >> 25 & 0x3f) << 5) + | ((word >> 8 & 0xf) << 1); + let taken = match f3 { + 0 => a == b, + 1 => a != b, + 4 => (a as i64) < (b as i64), + 5 => (a as i64) >= (b as i64), + 6 => a < b, + _ => a >= b, + }; + if taken { + next = pc.wrapping_add(((o << 19) as i32 >> 19) as i64 as u64); + } + } + 0x03 => { + let p = at(a.wrapping_add(imm_i)); + let bytes = |n: usize| { + let mut le = [0u8; 8]; + le[..n].copy_from_slice(&mem[p..p + n]); + u64::from_le_bytes(le) + }; + result = Some(match f3 { + 0 => bytes(1) as i8 as i64 as u64, + 1 => bytes(2) as i16 as i64 as u64, + 2 => bytes(4) as i32 as i64 as u64, + 3 => bytes(8), + 4 => bytes(1), + 5 => bytes(2), + _ => bytes(4), + }); + } + 0x23 => { + let imm = (((f7 << 5) | rd as u32) << 20) as i32 >> 20; + let p = at(a.wrapping_add(imm as i64 as u64)); + let n = 1 << f3; + mem[p..p + n].copy_from_slice(&b.to_le_bytes()[..n]); + } + 0x13 | 0x1b | 0x33 | 0x3b => { + let word32 = opcode & 8 != 0; + let immediate = opcode & 0x20 == 0; + let b = if immediate { imm_i } else { b }; + let m_ext = !immediate && f7 == 1; + let alt = if immediate { + f3 == 5 && word >> 30 & 1 == 1 + } else { + f7 == 0x20 + }; + let sh = (b & if word32 { 31 } else { 63 }) as u32; + let r = if m_ext && !word32 { + match f3 { + 0 => a.wrapping_mul(b), + 1 => ((a as i64 as i128 * b as i64 as i128) >> 64) as u64, + 2 => ((a as i64 as i128 * b as i128) >> 64) as u64, + 3 => ((a as u128 * b as u128) >> 64) as u64, + 4 if b == 0 => u64::MAX, + 4 => (a as i64).wrapping_div(b as i64) as u64, + 5 if b == 0 => u64::MAX, + 5 => a / b, + 6 if b == 0 => a, + 6 => (a as i64).wrapping_rem(b as i64) as u64, + _ if b == 0 => a, + _ => a % b, + } + } else if m_ext { + let (a, b) = (a as i32, b as i32); + let (ua, ub) = (a as u32, b as u32); + w(match f3 { + 0 => a.wrapping_mul(b) as u64, + 4 if b == 0 => u64::MAX, + 4 => a.wrapping_div(b) as u64, + 5 if b == 0 => u64::MAX, + 5 => (ua / ub) as u64, + 6 if b == 0 => a as u64, + 6 => a.wrapping_rem(b) as u64, + _ if b == 0 => ua as u64, + _ => (ua % ub) as u64, + }) + } else if word32 { + let a = a as u32; + w(match f3 { + 0 if alt && !immediate => a.wrapping_sub(b as u32) as u64, + 0 => a.wrapping_add(b as u32) as u64, + 1 => (a << sh) as u64, + _ if alt => (a as i32 >> sh) as u64, + _ => (a >> sh) as u64, + }) + } else { + match f3 { + 0 if alt && !immediate => a.wrapping_sub(b), + 0 => a.wrapping_add(b), + 1 => a << sh, + 2 => ((a as i64) < (b as i64)) as u64, + 3 => (a < b) as u64, + 4 => a ^ b, + 5 if alt => (a as i64 >> sh) as u64, + 5 => a >> sh, + 6 => a | b, + _ => a & b, + } + }; + result = Some(r); + } + _ => unreachable!("the generator made {word:#010x}"), + } + if let Some(r) = result + && rd != 0 + { + x[rd] = r; + } + *pc = next; + } + + /// A random legal instruction. `base` is a register the caller points at + /// `address - imm` for a load or a store. + fn random_instruction(rng: &mut Rng) -> (u32, Option<(u32, i32, u32)>) { + let reg = |rng: &mut Rng| rng.below(32) as u32; + let (rd, rs1, rs2) = (reg(rng), reg(rng), reg(rng)); + let imm = rng.below(4096) as i32 - 2048; + let word = match rng.below(10) { + 0 | 1 => { + let (_, opcode, f3, f7) = R_OPS[rng.below(R_OPS.len() as u64) as usize]; + r_type(opcode, f3, f7, rd, rs1, rs2) + } + 2 => { + let (_, opcode, f3) = I_OPS[rng.below(I_OPS.len() as u64) as usize]; + i_type(opcode, f3, rd, rs1, imm) + } + 3 => { + let (_, opcode, f3, top, bits) = SHIFT_OPS[rng.below(SHIFT_OPS.len() as u64) as usize]; + i_type(opcode, f3, rd, rs1, (top | rng.below(1 << bits) as u32) as i32) + } + 4 => { + let f3 = LOAD_OPS[rng.below(LOAD_OPS.len() as u64) as usize].1; + let base = 1 + rng.below(31) as u32; + return (i_type(0x03, f3, rd, base, imm), Some((base, imm, f3 & 3))); + } + 5 => { + let f3 = STORE_OPS[rng.below(STORE_OPS.len() as u64) as usize].1; + let base = 1 + rng.below(31) as u32; + return (s_type(0x23, f3, base, rs2, imm), Some((base, imm, f3))); + } + 6 => b_type( + BRANCH_OPS[rng.below(6) as usize].1, + rs1, + rs2, + (rng.below(2048) as i32 - 1024) * 4, + ), + 7 => j_type(rd, (rng.below(1 << 18) as i32 - (1 << 17)) * 4), + 8 => i_type(0x67, 0, rd, rs1, imm), + _ => u_type(if rng.below(2) == 0 { 0x37 } else { 0x17 }, rd, rng.next() as u32), + }; + (word, None) + } + + /// One random instruction from one random state, through the decoder and the class + /// functions and through the specification: same registers, same `pc`, same RAM. + #[test] + fn every_instruction_matches_the_specification() { + let mut rng = Rng(0x9e37_79b9_7f4a_7c15); + for _ in 0..300_000 { + let (word, access) = random_instruction(&mut rng); + let program = Program::new( + &[word], + TEXT_BASE, + (INPUT_WORDS..1 << LOG_RAM).map(|_| rng.next()).collect(), + LOG_RAM, + 0, + ); + let mut m = Machine::new(&program, [0; 4], &[]); + for r in 1..32 { + m.regs[r] = rng.word(); + } + if let Some((base, imm, log_width)) = access { + let address = RAM_BASE + (rng.below(RAM_BYTES - 8) & !((1 << log_width) - 1)); + m.regs[base as usize] = address.wrapping_sub(imm as i64 as u64); + } + let mut x: [u64; 32] = m.regs[..32].try_into().unwrap(); + let mut mem: Vec = m.ram().iter().flat_map(|w| w.to_le_bytes()).collect(); + let mut pc = m.pc; + spec_step(word, &mut x, &mut pc, &mut mem); + m.step().unwrap_or_else(|trap| panic!("{word:#010x}: {trap}")); + assert_eq!(m.regs[..32], x, "{word:#010x}: registers"); + assert_eq!(m.pc, pc, "{word:#010x}: pc"); + let ram: Vec = m.ram().iter().flat_map(|w| w.to_le_bytes()).collect(); + assert_eq!(ram, mem, "{word:#010x}: RAM"); + } + } + + fn run(text: &[u32], image: Vec) -> Result<[u64; 4], Trap> { + Machine::new(&Program::new(text, TEXT_BASE, image, LOG_RAM, 0), [7, 0, 0, 0], &[]).run(1 << 20) + } + + #[test] + fn li_loads_any_constant() { + let mut rng = Rng(7); + for _ in 0..2000 { + let value = rng.word(); + let text = Asm::new().li(A0, value).exit().finish(); + assert_eq!(run(&text, vec![]), Ok([value, 0, 0, 0]), "{value:#x}"); + } + } + + /// A loop, a call and return through `jal`/`jalr`, a stack, and an in-place sort. + #[test] + fn programs_run() { + // Fibonacci: a0 <- F(90), iteratively. + let text = Asm::new() + .li(A0, 0) + .li(A1, 1) + .li(T0, 90) + .label("loop") + .r("add", A2, A0, A1) + .i("addi", A0, A1, 0) + .i("addi", A1, A2, 0) + .i("addi", T0, T0, -1) + .branch("bne", T0, ZERO, "loop") + .li(A1, 0) + .li(A2, 0) + .exit() + .finish(); + assert_eq!(run(&text, vec![]), Ok([2_880_067_194_370_816_120, 0, 0, 0])); + + // Bubble sort of eight words through a stack frame, then a0 <- the median pair's sum. + const DATA: u64 = RAM_BASE + 8 * INPUT_WORDS as u64; + let data = [5u64, 3, 9, 1, 8, 2, 7, 4]; + let text = Asm::new() + .li(SP, RAM_BASE + RAM_BYTES) + .li(A0, DATA) + .jal(RA, "sort") + .li(T0, DATA) + .load("ld", A0, 24, T0) + .load("ld", A1, 32, T0) + .r("add", A0, A0, A1) + .li(A1, 0) + .li(A2, 0) + .li(A3, 0) + .exit() + .label("sort") + .i("addi", SP, SP, -16) + .store("sd", RA, 8, SP) + .li(T2, 7) + .label("outer") + .i("addi", T0, A0, 0) + .i("addi", T1, T2, 0) + .label("inner") + .load("ld", A2, 0, T0) + .load("ld", A3, 8, T0) + .branch("bgeu", A3, A2, "ordered") + .store("sd", A3, 0, T0) + .store("sd", A2, 8, T0) + .label("ordered") + .i("addi", T0, T0, 8) + .i("addi", T1, T1, -1) + .branch("bne", T1, ZERO, "inner") + .i("addi", T2, T2, -1) + .branch("bne", T2, ZERO, "outer") + .load("ld", RA, 8, SP) + .i("addi", SP, SP, 16) + .jalr(ZERO, RA, 0) + .finish(); + let program = Program::new(&text, TEXT_BASE, data.to_vec(), LOG_RAM, 0); + let mut m = Machine::new(&program, [0; 4], &[]); + assert_eq!(m.run(1 << 20), Ok([4 + 5, 0, 0, 0])); + assert_eq!(m.ram()[INPUT_WORDS..][..8], [1, 2, 3, 4, 5, 7, 8, 9]); + assert_eq!(m.regs[0], 0, "x0"); + } + + #[test] + fn traps() { + let text = |f: &mut dyn FnMut(&mut Asm)| { + let mut a = Asm::new(); + f(&mut a); + a.exit().finish() + }; + let pc = TEXT_BASE + 4; + // A misaligned load, one outside RAM, one reaching for the text. + let t = text(&mut |a| { + a.li(T0, RAM_BASE).load("lw", A0, 2, T0); + }); + assert_eq!( + run(&t, vec![]), + Err(Trap::Misaligned { + pc, + address: RAM_BASE + 2 + }) + ); + let t = text(&mut |a| { + a.li(T0, RAM_BASE).load("ld", A0, -8, T0); + }); + assert_eq!( + run(&t, vec![]), + Err(Trap::Unmapped { + pc, + address: RAM_BASE - 8 + }) + ); + let t = text(&mut |a| { + a.li(T0, TEXT_BASE).store("sd", A0, 0, T0); + }); + assert_eq!(run(&t, vec![]), Err(Trap::Unmapped { pc, address: TEXT_BASE })); + // Falling off the text, a jump into the middle of an instruction, EBREAK. + assert_eq!(run(&[0x13], vec![]), Err(Trap::Illegal { pc })); + assert_eq!( + run(&[asm::i_type(0x67, 0, 0, 0, 0), 0x13], vec![]), + Err(Trap::Illegal { pc: 0 }) + ); + assert_eq!(run(&[0x0010_0073], vec![]), Err(Trap::Illegal { pc: TEXT_BASE })); + // An ecall that is not exit, and a loop that never ends. + assert_eq!(run(&[ECALL], vec![]), Err(Trap::NotAnExit { syscall: 0 })); + assert_eq!(run(&[asm::j_type(0, 0)], vec![]), Err(Trap::CycleCap)); + } +} diff --git a/crates/lean_vm/src/rv/mod.rs b/crates/lean_vm/src/rv/mod.rs new file mode 100644 index 000000000..554e52727 --- /dev/null +++ b/crates/lean_vm/src/rv/mod.rs @@ -0,0 +1,277 @@ +//! RISC-V (rv64im) as the VM sees it: a public program of DECODED entries. +//! +//! The program is public, so nothing is decoded in a table: [`decode()`] turns each +//! 32-bit word into an [`Entry`] once, and a row's bytecode read returns the entry's +//! fields. An entry names an instruction [`Class`], whose one function +//! ([`semantics`]) a `flags` word specializes, the three register cells the row +//! touches, and the immediate already sign-extended. `LUI`, `AUIPC` and `JAL` are +//! folded to constants here, `pc` being known. +//! +//! Every row reads two registers and writes one. An instruction with fewer reads +//! `x0`, and one with no destination, or with `rd = x0`, writes [`SINK`], a cell +//! nothing reads: that is what hardwires `x0` to zero. The one exception is the +//! BLAKE2s precompile ([`hash`]), whose row writes RAM instead of a register. +//! +//! A trap is the absence of a run: [`machine::Trap`]. + +pub mod asm; +pub mod circuits; +pub mod decode; +pub mod elf; +pub mod machine; +pub mod semantics; + +pub use decode::decode; +pub use elf::{ElfError, Guest}; +pub use machine::{Machine, Program, Trap}; + +/// Where the text sits. Nonzero (a function at 0 would be Rust's null), and a +/// multiple of the largest text, so that instruction `i` is at `TEXT_BASE ^ (i << 2)`. +pub const TEXT_BASE: u64 = 0x1000_0000; +/// `log2` of the most instructions a program holds. +pub const MAX_LOG_TEXT: usize = 26; +/// Where RAM sits: a multiple of the largest RAM, in the text's 2 GiB window (the +/// medany code model). A maximal RAM ends at `0x8000_0000`, and its last 2 KiB are past +/// what `LUI` can form, since `LUI` sign-extends bit 31 on RV64; medlow code stops short +/// of them. +pub const RAM_BASE: u64 = 0x4000_0000; +/// `log2` of the most 64-bit words RAM holds. +pub const MAX_LOG_RAM: usize = 27; +/// Where the advice sits: a second region of memory, read and written like RAM, whose +/// contents before the run are the prover's rather than the statement's. What a +/// guest reads from it, it has to check. +pub const ADVICE_BASE: u64 = 0x2000_0000; +pub const MAX_LOG_ADVICE: usize = 26; + +/// RAM's first words are the run's public input, and the program's image follows. +pub const INPUT_WORDS: usize = 4; + +/// The register array holds `2^LOG_REGS` cells: `x0..x31`, then [`SINK`]. +pub const LOG_REGS: usize = 6; +/// The cell written by an instruction with no destination. +pub const SINK: u8 = 32; + +/// The exit system call's number, which `a7` must hold when the run halts. +pub const SYS_EXIT: u64 = 93; +/// `a0..a3`, whose final values are the run's public output. +pub const OUTPUT_REGS: [u8; 4] = [10, 11, 12, 13]; +/// `a7`. +pub const SYSCALL_REG: u8 = 17; + +/// An instruction class: one table, one circuit. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub enum Class { + /// Add, subtract, compare, bitwise logic, branches and jumps. + Alu, + Shift, + Load, + Store, + /// The low word of a product. + Mul, + /// The high word of a product. + Mulh, + Div, + /// The BLAKE2s compression, a custom instruction ([`hash`]). + Hash, + /// No table runs it: reaching one is a trap. + Illegal, +} + +/// Where control goes after an instruction. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Target { + /// Fall through, or wherever the row computes (`JALR`). + Next, + /// A branch or `JAL` target, taken when the class says so. + Abs(u64), + /// The halt slot, whose address the program fixes: what `ECALL` jumps to. + Halt, +} + +/// One decoded instruction. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Entry { + pub class: Class, + /// Selector bits for the class's function, one of its `LEGAL` words. + pub flags: u64, + /// The two cells read, below 32, and the cell written, in `1..=32`. + pub a1: u8, + pub a2: u8, + pub ad: u8, + pub imm: u64, + pub target: Target, + /// `rd` receives `pc + 4` instead of the class's result. + pub link: bool, + /// The next `pc` is the class's result. + pub jalr: bool, +} + +impl Entry { + pub const ILLEGAL: Self = Self { + class: Class::Illegal, + flags: 0, + a1: 0, + a2: 0, + ad: SINK, + imm: 0, + target: Target::Next, + link: false, + jalr: false, + }; + + /// What both verifiers check of every entry: the `x0` and [`SINK`] rules are + /// semantics, and the proof system is sound for any table, a malformed one included. + pub fn is_well_formed(&self) -> bool { + if self.class == Class::Illegal { + return *self == Self::ILLEGAL; + } + let legal = legal_flags(self.class); + let control = self.class == Class::Alu; + // A hash row writes no register and reads no immediate: its table holds both at + // their constants. + let hash = self.class == Class::Hash; + self.a1 < 32 + && self.a2 < 32 + && (1..=SINK).contains(&self.ad) + && legal.contains(&self.flags) + && (control || (self.target == Target::Next && !self.link && !self.jalr)) + && (!hash || (self.ad == SINK && self.imm == 0)) + } +} + +/// The flag words a class defines, which are the only ones its circuit is written for +/// and the only ones an entry may carry ([`Entry::is_well_formed`]). Empty for +/// [`Class::Illegal`], which carries no flags at all. +pub fn legal_flags(class: Class) -> &'static [u64] { + match class { + Class::Alu => &alu::LEGAL, + Class::Shift => &shift::LEGAL, + Class::Load => &load::LEGAL, + Class::Store => &store::LEGAL, + Class::Mul => &mul::LEGAL, + Class::Mulh => &mulh::LEGAL, + Class::Div => &div::LEGAL, + Class::Hash => &hash::LEGAL, + Class::Illegal => &[], + } +} + +/// [`Class::Alu`]'s flags. `b` is `v2 ^ imm`, one of the two being zero. +pub mod alu { + /// `v1 - b`, which the comparisons and the branches need, instead of `v1 + b`. + pub const SUB: u64 = 1 << 0; + /// Sign-extend the low 32 bits of the sum. + pub const WORD: u64 = 1 << 1; + // What `out` is, the sum when none is set. + pub const SEL_LT: u64 = 1 << 2; + pub const SEL_LTU: u64 = 1 << 3; + pub const SEL_AND: u64 = 1 << 4; + pub const SEL_OR: u64 = 1 << 5; + pub const SEL_XOR: u64 = 1 << 6; + /// Clear bit 0 of `out` (`JALR`). + pub const CLEAR_BIT0: u64 = 1 << 7; + // When the branch is taken. + pub const BR_EQ: u64 = 1 << 8; + pub const BR_NE: u64 = 1 << 9; + pub const BR_LT: u64 = 1 << 10; + pub const BR_GE: u64 = 1 << 11; + pub const BR_LTU: u64 = 1 << 12; + pub const BR_GEU: u64 = 1 << 13; + pub const ALWAYS: u64 = 1 << 14; + + pub const LEGAL: [u64; 17] = [ + 0, + SUB, + WORD, + SUB | WORD, + SUB | SEL_LT, + SUB | SEL_LTU, + SEL_AND, + SEL_OR, + SEL_XOR, + CLEAR_BIT0, + SUB | BR_EQ, + SUB | BR_NE, + SUB | BR_LT, + SUB | BR_GE, + SUB | BR_LTU, + SUB | BR_GEU, + ALWAYS, + ]; +} + +/// [`Class::Shift`]'s flags. The amount is the low 6 bits of `v2 ^ imm`, 5 for a word shift. +pub mod shift { + pub const RIGHT: u64 = 1 << 0; + /// An arithmetic right shift. + pub const ARITH: u64 = 1 << 1; + /// Shift the low 32 bits, and sign-extend the low 32 bits of the result. + pub const WORD: u64 = 1 << 2; + + pub const LEGAL: [u64; 6] = [0, RIGHT, RIGHT | ARITH, WORD, WORD | RIGHT, WORD | RIGHT | ARITH]; +} + +/// [`Class::Load`]'s flags: `log2` of the width in bytes, then the extension. +pub mod load { + pub const LOG_WIDTH: u64 = 0b11; + pub const SIGNED: u64 = 1 << 2; + + pub const LEGAL: [u64; 7] = [SIGNED, SIGNED | 1, SIGNED | 2, 3, 0, 1, 2]; +} + +/// [`Class::Store`]'s flags: `log2` of the width in bytes. +pub mod store { + pub const LOG_WIDTH: u64 = 0b11; + + pub const LEGAL: [u64; 4] = [0, 1, 2, 3]; +} + +/// [`Class::Mul`]'s flags. +pub mod mul { + /// Sign-extend the low 32 bits of the product. + pub const WORD: u64 = 1 << 0; + + pub const LEGAL: [u64; 2] = [0, WORD]; +} + +/// [`Class::Mulh`]'s flags: which operands are signed. +pub mod mulh { + pub const SIGNED_1: u64 = 1 << 0; + pub const SIGNED_2: u64 = 1 << 1; + + pub const LEGAL: [u64; 3] = [SIGNED_1 | SIGNED_2, SIGNED_1, 0]; +} + +/// [`Class::Hash`]: `blake2s rs1, rs2`, the BLAKE2s compression of the 128-byte block +/// at `rs1`, with `rs2` the byte counter. The block is the chaining value (32 bytes), +/// where the result goes (32 bytes), then the message (64 bytes). Word `k` of the +/// block is the cell at `rs1 ^ 8k`, which is `rs1 + 8k` when `rs1` is aligned to the +/// block, as the guest library makes sure; an unaligned `rs1` permutes the words, +/// which is deterministic and provable, and a guest bug. The flags are the +/// finalization word `f0`: all ones on the last block. +pub mod hash { + /// The custom-0 opcode, R-type: `funct3` is 1 for the final block, `funct7` and `rd` are zero. + pub const OPCODE: u32 = 0x0b; + /// Byte offsets in the block. + pub const H: u64 = 0; + pub const OUT: u64 = 32; + pub const M: u64 = 64; + /// The block's words, and how many the row rewrites. + pub const WORDS: usize = 16; + pub const BLOCK_BYTES: u64 = 8 * WORDS as u64; + + pub const FINAL: u64 = u32::MAX as u64; + pub const LEGAL: [u64; 2] = [0, FINAL]; +} + +/// [`Class::Div`]'s flags. +pub mod div { + pub const SIGNED: u64 = 1 << 0; + /// The remainder instead of the quotient. + pub const REM: u64 = 1 << 1; + /// Operate on the low 32 bits, extended as `SIGNED` says, and sign-extend the + /// low 32 bits of the result. + pub const WORD: u64 = 1 << 2; + + pub const LEGAL: [u64; 8] = [0, 1, 2, 3, 4, 5, 6, 7]; +} diff --git a/crates/lean_vm/src/rv/semantics.rs b/crates/lean_vm/src/rv/semantics.rs new file mode 100644 index 000000000..60ec7d772 --- /dev/null +++ b/crates/lean_vm/src/rv/semantics.rs @@ -0,0 +1,173 @@ +//! What each [`Class`](super::Class) computes: the reference its circuit is tested +//! against, and what the interpreter runs. Defined on a class's legal flags only. + +use super::{alu, div, hash, load, mul, mulh, shift, store}; + +fn sext32(x: u64) -> u64 { + x as u32 as i32 as i64 as u64 +} + +/// `(out, taken)`. +pub fn alu(v1: u64, v2: u64, imm: u64, flags: u64) -> (u64, bool) { + let on = |flag: u64| flags & flag != 0; + let b = v2 ^ imm; + let sum = if on(alu::SUB) { + v1.wrapping_sub(b) + } else { + v1.wrapping_add(b) + }; + let (lt, ltu, eq) = ((v1 as i64) < (b as i64), v1 < b, v1 == b); + let mut out = if on(alu::SEL_LT) { + lt as u64 + } else if on(alu::SEL_LTU) { + ltu as u64 + } else if on(alu::SEL_AND) { + v1 & b + } else if on(alu::SEL_OR) { + v1 | b + } else if on(alu::SEL_XOR) { + v1 ^ b + } else if on(alu::WORD) { + sext32(sum) + } else { + sum + }; + if on(alu::CLEAR_BIT0) { + out &= !1; + } + let taken = on(alu::ALWAYS) + || (on(alu::BR_EQ) && eq) + || (on(alu::BR_NE) && !eq) + || (on(alu::BR_LT) && lt) + || (on(alu::BR_GE) && !lt) + || (on(alu::BR_LTU) && ltu) + || (on(alu::BR_GEU) && !ltu); + (out, taken) +} + +pub fn shift(v1: u64, v2: u64, imm: u64, flags: u64) -> u64 { + let (right, arith, word) = ( + flags & shift::RIGHT != 0, + flags & shift::ARITH != 0, + flags & shift::WORD != 0, + ); + let amount = (v2 ^ imm) & if word { 31 } else { 63 }; + let x = match (word, arith) { + (false, _) => v1, + (true, true) => sext32(v1), + (true, false) => v1 as u32 as u64, + }; + let out = match (right, arith) { + (false, _) => x << amount, + (true, false) => x >> amount, + (true, true) => ((x as i64) >> amount) as u64, + }; + if word { sext32(out) } else { out } +} + +/// A load or store's address, `v1 + imm`. +pub fn address(v1: u64, imm: u64) -> u64 { + v1.wrapping_add(imm) +} + +/// What a load or a store puts on the memory bus for `address`: the byte address of +/// its 64-bit cell, with the bits that misalign the access left in. A cell's address +/// is a multiple of 8, so a misaligned access names no cell at all. +pub fn bus_address(address: u64, log_width: u64) -> u64 { + (address & !7) | (address & ((1 << log_width) - 1)) +} + +/// Whether an access of `2^log_width` bytes at `address` is naturally aligned. +pub fn is_aligned(address: u64, log_width: u64) -> bool { + address & ((1 << log_width) - 1) == 0 +} + +/// The value a load at `address` returns, from the 64-bit cell holding it. +pub fn load(cell: u64, address: u64, flags: u64) -> u64 { + let bits = 8 << (flags & load::LOG_WIDTH); + let x = cell >> (8 * (address & 7)); + if bits == 64 { + x + } else if flags & load::SIGNED != 0 { + (((x << (64 - bits)) as i64) >> (64 - bits)) as u64 + } else { + x & ((1 << bits) - 1) + } +} + +/// The cell a store of `value` at `address` leaves. +pub fn store(cell: u64, address: u64, value: u64, flags: u64) -> u64 { + let bits = 8 << (flags & store::LOG_WIDTH); + if bits == 64 { + return value; + } + let mask = ((1u64 << bits) - 1) << (8 * (address & 7)); + (cell & !mask) | ((value << (8 * (address & 7))) & mask) +} + +pub fn mul(v1: u64, v2: u64, flags: u64) -> u64 { + let product = v1.wrapping_mul(v2); + if flags & mul::WORD != 0 { + sext32(product) + } else { + product + } +} + +pub fn mulh(v1: u64, v2: u64, flags: u64) -> u64 { + let widen = |v: u64, signed: bool| if signed { v as i64 as i128 } else { v as i128 }; + let product = widen(v1, flags & mulh::SIGNED_1 != 0).wrapping_mul(widen(v2, flags & mulh::SIGNED_2 != 0)); + (product >> 64) as u64 +} + +/// What the prover tells [`super::circuits::div`] beyond the operands: the magnitudes of +/// the quotient and the remainder, which the circuit checks rather than computes. +pub fn div_hints(v1: u64, v2: u64, flags: u64) -> (u64, u64) { + let (signed, word) = (flags & div::SIGNED != 0, flags & div::WORD != 0); + let magnitude = |v: u64| match (word, signed) { + (false, false) => v, + (false, true) => (v as i64).unsigned_abs(), + (true, false) => v as u32 as u64, + (true, true) => (v as i32 as i64).unsigned_abs(), + }; + let (n, d) = (magnitude(v1), magnitude(v2)); + n.checked_div(d).map_or((0, 0), |q| (q, n % d)) +} + +/// One rule for the word forms: extend the low 32 bits of both operands, divide as +/// 64-bit, sign-extend the low 32 bits of the result. +pub fn div(v1: u64, v2: u64, flags: u64) -> u64 { + let (signed, rem, word) = (flags & div::SIGNED != 0, flags & div::REM != 0, flags & div::WORD != 0); + let extend = |v: u64| match (word, signed) { + (false, _) => v, + (true, true) => sext32(v), + (true, false) => v as u32 as u64, + }; + let (n, d) = (extend(v1), extend(v2)); + let (q, r) = if d == 0 { + (u64::MAX, n) + } else if signed { + let (n, d) = (n as i64, d as i64); + (n.wrapping_div(d) as u64, n.wrapping_rem(d) as u64) + } else { + (n / d, n % d) + }; + let out = if rem { r } else { q }; + if word { sext32(out) } else { out } +} + +/// The 32-bit words of `words`, little-endian. +fn halves(words: &[u64]) -> [u32; N] { + std::array::from_fn(|i| (words[i / 2] >> (32 * (i % 2))) as u32) +} + +/// [`super::Class::Hash`]: the compression of the block's chaining value and message +/// with the counter `t` and the finalization word `flags`, as the four words the row +/// writes back. `block` is the block's sixteen words. +pub fn blake2s(block: &[u64; hash::WORDS], t: u64, flags: u64) -> [u64; 4] { + let mut h: [u32; 8] = halves(&block[..4]); + let m: [u32; 16] = halves(&block[8..]); + debug_assert!(hash::LEGAL.contains(&flags)); + primitives::hash::compress(&mut h, &m, t, flags == hash::FINAL); + std::array::from_fn(|i| h[2 * i] as u64 | (h[2 * i + 1] as u64) << 32) +} diff --git a/crates/lean_vm/src/tables.rs b/crates/lean_vm/src/tables.rs index 8dd5ef14e..4b0ac260b 100644 --- a/crates/lean_vm/src/tables.rs +++ b/crates/lean_vm/src/tables.rs @@ -1,80 +1,49 @@ -//! Per-instruction tables (`doc/leanvm/body/07-instruction-tables.tex`). Each opcode is one [`Table`] impl that declares, -//! in one place, its committed columns, how to fill them from the trace, its bus -//! interactions (flushes), the read-count columns that feed the count channel, -//! and its degree-2 constraint. Column indices here are *local* (`0..n_committed_columns`); -//! `cpu`'s schema offsets them to global witness columns. +//! The instruction tables (`doc/leanvm/body/07-instruction-tables.tex`): one per +//! instruction class ([`crate::rv::Class`]), all of them the same table, which a +//! [`ClassSpec`] specializes. Column indices here are *local* +//! (`0..n_committed_columns`); `cpu`'s schema offsets them to global witness columns. //! -//! Columns are `K`-valued (`F64`). The pc/fp, operands, counts, opcodes and -//! separators are single `K`-columns; a **machine word** (memory value) is -//! 192-bit (`E = F192`), committed as THREE `K`-lane columns. Nothing a row -//! DERIVES is a column at all: an operand address `fp·o`, an `XOR`/`MUL` result, -//! the `DEREF` store, the `JUMP` successors are each written out as the degree-2 -//! bus coordinate that carries them (§sec:m3), which leaves `JUMP`'s is-nonzero -//! indicator as the one identity any table still has. Every identity is `K`-valued, -//! so a relation on machine words is written out lane by lane; after the round a -//! table joins the batch its columns are `E`-valued, which is what -//! `eval_constraint` takes. +//! A table does plumbing only. What its class computes is a flock circuit +//! ([`crate::rv::circuits`]), and every word that circuit reads or writes is a +//! VIRTUAL column here: it lives in the circuit's packed witness +//! ([`crate::class_flock`]) and rides the bus from there, in the register and RAM +//! tuples and the bytecode tuple, which is all that binds the circuit to the machine. +//! What a row commits of its own is the rest of its bytecode entry, the old value of +//! what it writes, and per access the argument that orders it in time +//! (§sec:memchan): the timestamp `X` of the cell's previous access and the two chunks +//! of the gap to the row's own, each a read of a range array. use crate::colval::ColVal; -use crate::cpu::{Brow, Drow, Jrow, Op, Srow, Trace}; +use crate::cpu::{Access, HashRow, Row, Trace}; use crate::leaf::Coord::{self, Col, Const, GCol, Prod}; +use crate::rv::{self, Class, SINK, hash}; +use flock::circuit::Circuit; use primitives::field::{F64, F192, mul_by_g}; // ---- the identities ---------------------------------------------------------- // -// Each is written ONCE, generic over the column type: `F64` in the round a table -// joins the batch, `F192` afterwards (see [`ColVal`]). Products of two `K` -// columns stay 64-bit and an `η`-power multiplies through `mul_e`. -// -// EVERY identity is `K`-valued, like the columns it reads and like the bus -// coordinates (§sec:air). A relation between machine WORDS is therefore written out -// lane by lane, with the tower multiplication unrolled by hand ([`TOWER_LANES`]), -// rather than assembled into one `E` equation. A table's identity slice is then a -// mixed dot product against its `η`-range: in the round a table joins the batch (the -// batch's largest) that is one 64-bit product per identity plus three PMULL for its -// `η`-power, with ONE reduction for the whole slice ([`ColVal::dot`]). One `E`-valued -// identity would instead cost a full `E×E` product against its `η`-power, plus a -// reduction of its own, and the lanes it bundles are the same lane polynomials -// either way. - -/// The tower product `x·y` in `E = K[y]/(y³+y+1)` as three lane sums: lane `i` is -/// `Σ x_j·y_k` over `TOWER_LANES[i]`. The five partial sums of §sec:tab-mul fold -/// into `c0 = p0+p3`, `c1 = p1+p3+p4`, `c2 = p2+p4`; written once here because both -/// `MUL`'s result coordinate ([`arith_result`]) and `JUMP`'s inverse identity need -/// the same unrolling. -const TOWER_LANES: [&[(usize, usize)]; 3] = [ - &[(0, 0), (1, 2), (2, 1)], - &[(0, 1), (1, 0), (1, 2), (2, 1), (2, 2)], - &[(0, 2), (1, 1), (2, 0), (2, 2)], -]; - -/// One lane of the tower product, in the columns' own field. Only the test that -/// checks [`TOWER_LANES`] against the field needs it: `MUL`'s result rides the bus -/// as a coordinate ([`arith_result`]), and `JUMP`'s condition is `K`-valued, so no -/// identity assembles a tower product any more. -#[cfg(test)] -fn tower_lane(lane: usize, x: [T; 3], y: [T; 3]) -> T { - TOWER_LANES[lane].iter().fold(T::ZERO, |acc, &(j, k)| acc + x[j] * y[k]) +// Written ONCE, generic over the column type: `F64` in the round a table joins the +// batch, `F192` afterwards (see [`ColVal`]). Products of two `K` columns stay +// 64-bit and an `η`-power multiplies through `mul_e`. + +/// Every access's `X·LO = g^{slot}·TS·HI` (§sec:memchan), folded: `w` is +/// [`access_weights`], the identities' `η`-powers then the same times `g^{slot}`, +/// so the clock factors out of the second half. Homogeneous of degree two, so the +/// round coefficient and the value are the same expression. +fn access_identities(w: &[F192], cols: &[T], ts: usize, acc: Acc) -> F192 { + let n = acc.n; + let left = (0..n).fold(T::lift(F192::ZERO), |sum, i| { + sum ^ (cols[acc.x(i)] * cols[acc.lo(i)]).mul_e_unreduced(w[i]) + }); + let hi = T::dot(&w[n..2 * n], &cols[acc.hi(0)..acc.hi(0) + n], F192::ZERO); + T::reduce(left ^ T::lift(cols[ts].mul_e(hi))) } -/// `JUMP`'s two identities: `b = cond·w` and `cond·(b+1) = 0` (§sec:tab-jump). -/// -/// The two relations together force `b = [cond ≠ 0]`: when `cond ≠ 0` the second -/// gives `b = 1` (and the first `w = cond⁻¹`); when `cond = 0` the first gives -/// `b = 0`. The two selections need no identity: the state push carries each as its -/// own degree-2 coordinate (§sec:m3). -/// -/// The condition is `K`-valued, so both identities are single-lane. Its memory -/// flush carries literal zeros above the low limb (`memory_k`), so a word outside -/// `K` cannot balance the bus; the interpreter rejects one outright. -fn jump_identity(pows: &[F192], cols: &[T], quadratic: bool) -> F192 { - use jump::*; - let (b, b1) = if quadratic { - (T::ZERO, cols[B]) - } else { - (cols[B], cols[B] + T::ONE) - }; - T::dot(pows, &[b + cols[V_COND] * cols[W], cols[V_COND] * b1], F192::ZERO) +/// [`access_identities`]' weights from the `n` identities' `η`-powers. +fn access_weights(pows: &[F192], slots: &[u32]) -> Vec { + assert_eq!(pows.len(), slots.len()); + let shifted = pows.iter().zip(slots).map(|(p, &s)| p.mul_base(g_pow(s as usize))); + pows.iter().copied().chain(shifted).collect() } // ---- shared bus vocabulary --------------------------------------------------- @@ -90,25 +59,75 @@ const fn g_pow(k: usize) -> F64 { acc } -// Domain separators (coordinate 0 of every bus tuple): the g-powers g^0, g^1, g^2. +// Domain separators (coordinate 0 of every bus tuple). pub(crate) const SEP_STATE: F64 = g_pow(0); pub(crate) const SEP_MEM: F64 = g_pow(1); pub(crate) const SEP_BYTECODE: F64 = g_pow(2); +pub(crate) const SEP_RANGE_LO: F64 = g_pow(3); +pub(crate) const SEP_RANGE_HI: F64 = g_pow(4); +pub(crate) const SEP_REG: F64 = g_pow(5); + +/// Each range array holds `2^RANGE_LOG` entries, so a gap is below `2^(2·RANGE_LOG)`. +pub const RANGE_LOG: usize = 16; + +/// The clock advances by this much per instruction, which leaves one timestamp per +/// access slot: `ts = 4·cycle + slot` (§sec:memchan). A hash row, with more accesses, +/// advances it further ([`ClassSpec::stride`]). +pub const CLOCK_STRIDE: u32 = 4; +/// The clock the run starts on: cycle 1, so the first access is strictly after the +/// seeds' `g^0`. +pub const CLOCK_START: F64 = g_pow(CLOCK_STRIDE as usize); +/// A row's clock slots: `rs1`, `rs2`, then `rd` last, after the RAM access if there is one. +pub const REG_SLOTS: [u32; 3] = [0, 1, 3]; +pub const RAM_SLOT: u32 = 2; +/// A hash row reads its two registers, then accesses its block's words in order. +pub const fn block_slot(k: usize) -> u32 { + 2 + k as u32 +} -// Opcodes (coordinate 3 of a bytecode tuple). -pub(crate) const OP_XOR: F64 = g_pow(0); -pub(crate) const OP_MUL: F64 = g_pow(1); -pub(crate) const OP_SET: F64 = g_pow(2); -pub(crate) const OP_DEREF: F64 = g_pow(3); -pub(crate) const OP_JUMP: F64 = g_pow(4); -pub(crate) const OP_BLAKE2S: F64 = g_pow(5); +/// The low range array's addresses `g^{j+1}`: the `+1` is the strictness of `x < y`. +pub fn range_lo_first() -> F64 { + F64::G +} +/// The high range array's addresses are the powers of `g^{-2^16}`, from `g^0`. +pub fn range_hi_ratio() -> F64 { + static RATIO: std::sync::OnceLock = std::sync::OnceLock::new(); + *RATIO.get_or_init(|| primitives::field::g_pow(1 << RANGE_LOG).inv()) +} + +/// Where a table keeps its `n` accesses' columns, grouped by kind so that each kind +/// is contiguous: the previous timestamps `X`, the gap's low and high chunks, then +/// the counts of the two range reads. +#[derive(Clone, Copy)] +pub(crate) struct Acc { + base: usize, + n: usize, +} + +impl Acc { + const fn x(&self, i: usize) -> usize { + self.base + i + } + const fn lo(&self, i: usize) -> usize { + self.base + self.n + i + } + const fn hi(&self, i: usize) -> usize { + self.base + 2 * self.n + i + } + const fn count_lo(&self, i: usize) -> usize { + self.base + 3 * self.n + i + } + const fn count_hi(&self, i: usize) -> usize { + self.base + 4 * self.n + i + } + const fn end(&self) -> usize { + self.base + 5 * self.n + } +} // ---- flush builder ----------------------------------------------------------- -/// Collects a table's push/pull bus interactions in *local* column indices. The -/// push/pull of a memory-checked entry differ only by one coordinate carrying the -/// post-increment `g·count` (`GCol`) instead of the pre-increment (`Col`); these -/// helpers encode that pairing so each table reads declaratively. +/// Collects a table's push/pull bus interactions in *local* column indices. pub struct FlushBuilder { pub(crate) push: Vec>, pub(crate) pull: Vec>, @@ -127,81 +146,61 @@ impl FlushBuilder { self.pull.push(pull); } - /// Fall-through state step: the next pc is `g·pc`, fp unchanged. - pub(crate) fn state_step(&mut self, pc: usize, fp: usize) { - self.pair( - vec![Const(SEP_STATE), GCol(pc, 1), Col(fp)], - vec![Const(SEP_STATE), Col(pc), Col(fp)], - ); - } - - /// Explicit state transition (JUMP): push the next state, which the row - /// DERIVES from its columns rather than committing, and pull `(pc, fp)`. - pub(crate) fn state_derived(&mut self, pc: usize, fp: usize, npc: Coord, nfp: Coord) { + /// Pull the current state and push the next: `npc`, which a row DERIVES from its + /// columns rather than committing, and the clock advanced by the class's stride. + fn state(&mut self, pc: usize, ts: usize, npc: Coord, stride: u32) { self.pair( - vec![Const(SEP_STATE), npc, nfp], - vec![Const(SEP_STATE), Col(pc), Col(fp)], + vec![Const(SEP_STATE), npc, GCol(ts, stride)], + vec![Const(SEP_STATE), Col(pc), Col(ts)], ); } - /// Bytecode read at `pc`: the program tuple (opcode + seven operand slots), - /// with the per-pc execution count advanced by ×g on the push side. - pub(crate) fn bytecode(&mut self, pc: usize, count: usize, opcode: F64, operands: &[Coord]) { - let mut push = vec![Const(SEP_BYTECODE), Col(pc), GCol(count, 1), Const(opcode)]; - let mut pull = vec![Const(SEP_BYTECODE), Col(pc), Col(count), Const(opcode)]; - push.extend_from_slice(operands); - pull.extend_from_slice(operands); - self.pair(push, pull); - } - - /// The shape every memory interaction shares: the word at `addr` carried as - /// three value coordinates, with the cell's access count advanced by ×g on the - /// push side. A value the row DERIVES rather than commits (an `XOR`/`MUL` - /// result, a `DEREF` store) is passed here as its form: the cell then holds - /// whatever the form says, which removes both the value columns and the - /// identity that used to tie them (§sec:m3). - pub(crate) fn memory_coords(&mut self, addr: Coord, count: usize, vals: [Coord; 3]) { - let mut push = vec![Const(SEP_MEM), addr.clone(), GCol(count, 1)]; - let mut pull = vec![Const(SEP_MEM), addr, Col(count)]; - push.extend_from_slice(&vals); - pull.extend_from_slice(&vals); - self.pair(push, pull); - } - - /// Memory access: read the three-limb word at `addr`. - pub(crate) fn memory(&mut self, addr: Coord, count: usize, val0: usize, val1: usize, val2: usize) { - self.memory_coords(addr, count, [Col(val0), Col(val1), Col(val2)]); + /// A read of a lookup array (§sec:lookup): `tuple` as pulled, `tuple[2]` being + /// its count column `count`, pushed back with the count advanced by ×g. + fn counted(&mut self, tuple: Vec, count: usize) { + let mut push = tuple.clone(); + push[2] = GCol(count, 1); + self.pair(push, tuple); } - /// Memory read of a K-valued word: both higher limbs are literal zero. Used where - /// the word is carried by a single K column (e.g. the DEREF pointer). Sound - /// because the bus balances only if the stored value's HI lane is likewise 0. - pub(crate) fn memory_k(&mut self, addr: Coord, count: usize, val: usize) { - self.memory_coords(addr, count, [Col(val), Const(F64::ZERO), Const(F64::ZERO)]); - } - - /// Memory access to a canonical 128-bit word `(lo, hi, 0)`. - pub(crate) fn memory_128(&mut self, addr: Coord, count: usize, lo: usize, hi: usize) { - self.memory_coords(addr, count, [Col(lo), Col(hi), Const(F64::ZERO)]); + /// Access `i` of the row, at clock slot `slot` (§sec:memchan), to the cell `addr` + /// of the array `sep`: pull the cell as its previous access left it, `(X, old)`, + /// push it back as `(g^{slot}·ts, new)`, and read the gap's two chunks off the + /// range arrays. A value the row DERIVES rather than commits is passed as its + /// form (§sec:m3). + #[allow(clippy::too_many_arguments)] + fn access(&mut self, sep: F64, addr: Coord, ts: usize, acc: Acc, i: usize, slot: u32, old: Coord, new: Coord) { + self.pair( + vec![Const(sep), addr.clone(), GCol(ts, slot), new], + vec![Const(sep), addr, Col(acc.x(i)), old], + ); + self.counted( + vec![Const(SEP_RANGE_LO), Col(acc.lo(i)), Col(acc.count_lo(i))], + acc.count_lo(i), + ); + self.counted( + vec![Const(SEP_RANGE_HI), Col(acc.hi(i)), Col(acc.count_hi(i))], + acc.count_hi(i), + ); } } // ---- fill context ------------------------------------------------------------ -/// Inputs a table needs to fill its columns: the trace rows, the final memory -/// image (for read values), and `g^0..` for O(1) address/operand lookups. +/// Inputs a table needs to fill its columns. pub struct FillCtx<'a> { pub(crate) trace: &'a Trace, - pub(crate) mem: &'a [F192], - pub(crate) gpow: &'a [F64], - pub(crate) prog: &'a [Op], + /// The two range arrays' addresses, by chunk. + pub(crate) range_lo: &'a [F64], + pub(crate) range_hi: &'a [F64], + pub(crate) program: &'a rv::Program, /// This table's height `2^tau`, the length of every window in `out`, and its row - /// count too: every row is a row the program executed (`cpu::filler`). + /// count too (`cpu::filler`). pub(crate) rows: usize, /// Which local columns [`Self::col`] / [`Self::cols`] have written. A fill that /// misses one would leave the stacked witness holding uninitialized slots, so /// [`fill_table`] checks the whole set was covered. - written: std::sync::atomic::AtomicU64, + written: Vec, } /// Where one column's values go: its window in the stacked witness, or a private @@ -209,73 +208,48 @@ pub struct FillCtx<'a> { pub type ColumnOut<'a> = &'a mut [F64]; impl<'a> FillCtx<'a> { - pub(crate) fn new(trace: &'a Trace, mem: &'a [F192], gpow: &'a [F64], prog: &'a [Op], rows: usize) -> Self { + pub(crate) fn new( + trace: &'a Trace, + range_lo: &'a [F64], + range_hi: &'a [F64], + program: &'a rv::Program, + rows: usize, + n_cols: usize, + ) -> Self { Self { trace, - mem, - gpow, - prog, + range_lo, + range_hi, + program, rows, - written: std::sync::atomic::AtomicU64::new(0), + written: (0..n_cols).map(|_| false.into()).collect(), } } - fn g_at(&self, i: u32) -> F64 { - self.gpow[i as usize] - } - - /// The three frame offsets of an `XOR`/`MUL` row. A row records - /// only its `(pc, fp)`; the operands are the instruction's, so they are read - /// back from the bytecode rather than copied into every row (§the trace rows - /// in `cpu::trace`). - fn ternary_operands(&self, pc: u32) -> (u32, u32, u32) { - match self.prog[pc as usize] { - Op::Xor { a, b, c } | Op::Mul { a, b, c } => (a, b, c), - op => unreachable!("a three-operand row's pc {pc} holds {op:?}"), - } - } - - /// Write local column `at`: `f` over the trace rows, then its pad value to the - /// end of the window. + /// Write local column `at`: `f` over the trace rows. fn col(&self, out: &mut [ColumnOut], rows: &[R], at: usize, f: impl Fn(&R) -> F64 + Sync) { self.cols(out, rows, at, |r| [f(r)]); } /// Write the `N` local columns at `at..at + N` from one closure per row. - /// Columns fed by the same read fill together: the three lanes of a 192-bit - /// memory word are one random access, and splitting them across `N` passes - /// pays for it `N` times. fn cols( &self, out: &mut [ColumnOut], rows: &[R], at: usize, f: impl Fn(&R) -> [F64; N] + Sync, - ) { - self.cols_at(out, rows.len(), at, |i| f(&rows[i])); - } - - /// [`Self::cols`] over row indices, for values held in a side buffer rather - /// than read off the row. - fn cols_at( - &self, - out: &mut [ColumnOut], - n_rows: usize, - at: usize, - f: impl Fn(usize) -> [F64; N] + Sync, ) { let n = self.rows; let dst: [parallel::SendPtr; N] = std::array::from_fn(|k| { assert_eq!(out[at + k].len(), n, "column {} has the wrong window length", at + k); - self.written - .fetch_or(1 << (at + k), std::sync::atomic::Ordering::Relaxed); + self.written[at + k].store(true, std::sync::atomic::Ordering::Relaxed); parallel::SendPtr(out[at + k].as_mut_ptr()) }); - // Every row is a row the program executed: a table's height is its row - // count (`cpu::filler`), so there is nothing to pad with. - assert_eq!(n_rows, n, "a table's rows must fill its cube"); + // A table's height is its row count (`cpu::filler`), so there is nothing to + // pad with. + assert_eq!(rows.len(), n, "a table's rows must fill its cube"); parallel::for_each(n, |i| { - let v = f(i); + let v = f(&rows[i]); for (k, p) in dst.iter().enumerate() { // SAFETY: distinct `i` write disjoint in-bounds slots of each of the // `N` windows, each exactly once, and the dispatch blocks until @@ -285,10 +259,18 @@ impl<'a> FillCtx<'a> { }); } - /// The three `K`-lanes of the 192-bit word in memory cell `addr`. - fn limbs(&self, addr: u32) -> [F64; 3] { - let w = self.mem[addr as usize]; - [F64(w.c0), F64(w.c1), F64(w.c2)] + /// The `5·n` columns of a table's accesses, one kind at a time. + fn accesses(&self, out: &mut [ColumnOut], rows: &[R], acc: Acc, f: impl Fn(&R) -> &[Access] + Sync) { + let mask = (1u32 << RANGE_LOG) - 1; + for i in 0..acc.n { + self.col(out, rows, acc.x(i), |r| f(r)[i].x); + self.col(out, rows, acc.lo(i), |r| self.range_lo[(f(r)[i].gap & mask) as usize]); + self.col(out, rows, acc.hi(i), |r| { + self.range_hi[(f(r)[i].gap >> RANGE_LOG) as usize] + }); + self.col(out, rows, acc.count_lo(i), |r| f(r)[i].count_lo); + self.col(out, rows, acc.count_hi(i), |r| f(r)[i].count_hi); + } } } @@ -297,11 +279,9 @@ impl<'a> FillCtx<'a> { /// indeterminate bytes rather than caught by a length mismatch. pub(crate) fn fill_table(table: &dyn Table, ctx: &FillCtx, out: &mut [ColumnOut]) { table.fill(ctx, out); - let n = table.n_committed_columns(); - assert!(n <= 64, "the write mask covers at most 64 columns per table"); - let all = if n == 64 { u64::MAX } else { (1u64 << n) - 1 }; - let written = ctx.written.load(std::sync::atomic::Ordering::Relaxed); - assert_eq!(written, all, "a table left one of its columns unwritten"); + assert_eq!(ctx.written.len(), table.n_committed_columns()); + let all = ctx.written.iter().all(|w| w.load(std::sync::atomic::Ordering::Relaxed)); + assert!(all, "a table left one of its columns unwritten"); } // ---- the trait --------------------------------------------------------------- @@ -309,649 +289,560 @@ pub(crate) fn fill_table(table: &dyn Table, ctx: &FillCtx, out: &mut [ColumnOut] /// One instruction table. Indices in [`flushes`](Table::flushes) and /// [`count_columns`](Table::count_columns) are local to this table. pub trait Table: Sync { - /// Number of committed columns (local indices `0..n_committed_columns`). + /// Number of columns (local indices `0..n_committed_columns`), the virtual ones included. fn n_committed_columns(&self) -> usize; - /// Local indices of this table's read-count columns: the `g^{count}` values - /// recording how many times each accessed cell (and the pc) was read. The - /// framework treats them specially: each gets its own single-column "count" - /// bus block, and padding rows fill them with `1` (= g^0) instead of `0`. - fn count_columns(&self) -> &'static [usize]; + /// Local indices of this table's read-count columns: the `g^{count}` values of + /// its lookups into the read-only arrays (the bytecode, the two range arrays). + /// The framework treats them specially: each gets its own single-column "count" + /// bus block. + fn count_columns(&self) -> &[usize]; /// How many identities [`eval_constraint`](Table::eval_constraint) folds. /// Sizes this table's slice of the batch's disjoint `xi`-range (§constraints). - /// Defaults to none, which is every table but `JUMP`: a relation whose value - /// rides the bus as a coordinate needs no identity to tie it (§sec:m3). - fn n_constraints(&self) -> usize { - 0 - } - /// Evaluate the table's degree-2 constraint at one row, reading column values - /// by local index from `cols` (e.g. `cols[jump::V_COND]`) and weighting identity - /// `i` by `pows[i]`, this table's slice of the batch's `xi`-powers. The slice is - /// is exactly [`n_constraints`](Table::n_constraints) long: an identity indexed - /// past its end panics rather than silently reaching into the next table's - /// range. The table sumcheck carries every committed column of a table, in - /// local order, so `cols` is indexed directly. With `quadratic=false` it returns - /// `0` on every valid row (§sec:air); `true` selects only the degree-two terms. - /// The default is the constraint-free case; a table that declares constraints - /// and forgets to evaluate them trips the assert instead of dropping them. - fn eval_constraint(&self, pows: &[F192], _cols: &[F192], _quadratic: bool) -> F192 { - assert!(pows.is_empty(), "a table with constraints must evaluate them"); - F192::ZERO - } + fn n_constraints(&self) -> usize; + /// What [`eval_constraint`](Table::eval_constraint) is handed, from this table's + /// slice of the batch's `xi`-powers: the powers themselves, plus whatever + /// constant multiples of them the identities need, computed once per proof + /// rather than per row. + fn constraint_weights(&self, pows: &[F192]) -> Vec; + /// Evaluate the table's degree-2 constraints at one row, reading column values + /// by local index from `cols` and weighting them by `weights` + /// ([`constraint_weights`](Table::constraint_weights)). The table sumcheck + /// carries every column of a table, in local order, so `cols` is indexed + /// directly. With `quadratic=false` it returns `0` on every valid row + /// (§sec:air); `true` selects only the degree-two terms. + fn eval_constraint(&self, weights: &[F192], cols: &[F192], quadratic: bool) -> F192; /// The same identity over `K`-valued columns, for the round a table joins the /// batch, before its columns have been folded into `E` (§sec:air). Both entry - /// points delegate to one generic definition per table, so they cannot drift. - fn eval_constraint_k(&self, pows: &[F192], _cols: &[F64], _quadratic: bool) -> F192 { - assert!(pows.is_empty(), "a table with constraints must evaluate them"); - F192::ZERO - } + /// points delegate to one generic definition, so they cannot drift. + fn eval_constraint_k(&self, weights: &[F192], cols: &[F64], quadratic: bool) -> F192; /// Declare the table's bus interactions. fn flushes(&self, f: &mut FlushBuilder); /// Fill this table's columns from the trace: `out[i]` is local column `i`'s - /// window, already at its padded length. Every window must be written in full; - /// use `FillCtx::col` / `FillCtx::cols`, which append the column's pad value - /// past the last trace row and record the coverage `fill_table` checks. + /// window, already at its final length. Every window must be written in full; + /// use `FillCtx::col` / `FillCtx::cols`, which record the coverage `fill_table` + /// checks. fn fill(&self, ctx: &FillCtx, out: &mut [ColumnOut]); } -/// The tables in fixed order `[XOR, MUL, SET, DEREF, JUMP, BLAKE2S]`, the -/// order of `row_counts` / `taus` throughout `cpu`. -pub const N_TABLES: usize = 6; - -pub fn tables() -> [&'static dyn Table; N_TABLES] { - [ - &Arith { is_xor: true }, - &Arith { is_xor: false }, - &SetTable, - &DerefTable, - &JumpTable, - &Blake2sTable, - ] -} - -/// Index of the BLAKE2s table in [`tables`]. -pub(crate) const BLAKE2S_TABLE: usize = 5; - -/// The seven base addresses a `BLAKE2s` row reads: the four message cells, the -/// chaining-value base, the output base (each of those two spans that cell and -/// its successor) and the metadata cell. Recovered from the instruction, not -/// stored per row. -pub(crate) fn blake2s_addresses(prog: &[Op], r: &Brow) -> [u32; 7] { - match prog[r.pc as usize] { - Op::Blake2s { ins, cv, out, md } => [ - r.fp + ins[0], - r.fp + ins[1], - r.fp + ins[2], - r.fp + ins[3], - r.fp + cv, - r.fp + out, - r.fp + md, - ], - op => unreachable!("a BLAKE2s row's pc {} holds {op:?}", r.pc), - } +// ---- the classes ------------------------------------------------------------- + +/// A word a class's circuit reads or writes, which is a virtual column of its table. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Word { + /// The bytecode's selector bits and immediate. + Flags, + Imm, + /// The two registers read. + V1, + V2, + /// What the class computes. + Out, + /// Whether the branch is taken, 0 or 1. + Taken, + /// A load's or a store's bus address. + Address, + /// Word `k` of the RAM cells the row names, as the row found it: the one cell of + /// a load or a store, or one of the hash's block ([`Ram::Block`]). + Cell(u8), + /// What the row leaves in word `k`. + CellNew(u8), + /// What a circuit asserts to be zero: the row puts it in its bytecode tuple, in a + /// slot where the program holds zero, so the lookup is what makes it zero. + Bad, + /// What the prover tells the circuit beyond the row: in its witness, in no column. + HintQ, + HintR, } -/// BLAKE2s value-column LOCAL indices in canonical slot order -/// `[a0..a3, b0..b3, c0..c3, cv0..cv3, md_lo, md_hi]` (matches -/// `hash_flock::SLOTS`). These columns are -/// VIRTUAL (never committed): `q_flock` already holds those words at fixed packed -/// slots, so `cpu` routes their memory-bus evaluation claims straight to `q_flock` -/// (`slot_claims`): the value the bus flushes IS the flock-proven word. -pub const BLAKE2S_VALUE_COLS: [usize; 18] = [ - blake2st::V_M0, - blake2st::V_M0 + 1, - blake2st::V_M0 + 2, - blake2st::V_M0 + 3, - blake2st::V_M2, - blake2st::V_M2 + 1, - blake2st::V_M2 + 2, - blake2st::V_M2 + 3, - blake2st::V_OUT0, - blake2st::V_OUT0 + 1, - blake2st::V_OUT0 + 2, - blake2st::V_OUT0 + 3, - blake2st::V_CV0, - blake2st::V_CV0 + 1, - blake2st::V_CV0 + 2, - blake2st::V_CV0 + 3, - blake2st::MD0, - blake2st::MD1, -]; -// The eighteen value lanes are laid out contiguously (V_M0..V_M0+17), so they map -// 1:1 onto `hash_flock::SLOTS`. -const _: () = assert!( - blake2st::V_M2 == blake2st::V_M0 + 4 - && blake2st::V_OUT0 == blake2st::V_M0 + 8 - && blake2st::V_CV0 == blake2st::V_M0 + 12 - && blake2st::MD0 == blake2st::V_M0 + 16 - && blake2st::MD1 == blake2st::V_M0 + 17 -); - -// ---- XOR / MUL --------------------------------------------------------------- - -/// `XOR` and `MUL_NATIVE` share their column layout, flushes, and fill; they -/// differ only in the opcode tag and in how the destination cell's value rides -/// the bus (`v_A + v_B` for `XOR`, `v_A·v_B` in `E = K[y]/(y³+y+1)` for `MUL`). -/// Neither commits that value and neither has an identity: the destination's -/// memory flush carries the result as a degree-≤2 coordinate over the operand -/// lanes, so bus balance IS the assertion (§sec:m3). -struct Arith { - is_xor: bool, +/// How a class uses RAM. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Ram { + None, + /// One cell read, in clock slot 2, at the address the circuit computes. + Read, + /// One cell read and rewritten. + Write, + /// The hash's block ([`hash`]): word `k` is the cell at `v1 ^ 8k`, in clock slot + /// [`block_slot`]`(k)`, and the result's four words are rewritten. The row writes + /// no register. + Block, } -mod arith { - pub const PC: usize = 0; - pub const FP: usize = 1; - pub const OA: usize = 2; - pub const OB: usize = 3; - pub const OC: usize = 4; - // No absolute-address columns: the memory bus carries `fp·o` as a product - // coordinate (§sec:m3), which is why there is no address binding below. - // The two read words, each three K-limbs. The third (the result) is DERIVED. - pub const VA_LO: usize = 5; - pub const VA_HI: usize = 6; - pub const VA_TOP: usize = 7; - pub const VB_LO: usize = 8; - pub const VB_HI: usize = 9; - pub const VB_TOP: usize = 10; - pub const RA: usize = 11; - pub const RB: usize = 12; - pub const RC: usize = 13; - pub const RBC: usize = 14; - pub const N: usize = 15; +/// What specializes the class table to one instruction class. +pub struct ClassSpec { + pub class: Class, + pub name: &'static str, + /// Branches and jumps: the bytecode's `dt`, `link` and `jalr` fields, and the + /// circuit's `taken` word. Without them the next `pc` is `pc + 4` and `rd` + /// receives `out`. + pub control: bool, + pub ram: Ram, + pub circuit: fn() -> Circuit, + /// `log2` of the bits one instance of the circuit occupies. A constant, because + /// the layout needs it before any circuit is built; [`crate::class_flock`] checks it. + pub k_log: usize, + /// The circuit's port words in order, the first `n_inputs` of them its inputs. + pub ports: &'static [Word], + pub n_inputs: usize, } -/// The result word's three K-lanes as forms over the operand lanes. For `XOR` -/// that is the lane-wise sum; for `MUL` it is the tower product, unrolled through -/// [`TOWER_LANES`]. -fn arith_result(is_xor: bool) -> [Coord; 3] { - use arith::*; - if is_xor { - return [ - Coord::Sum(vec![Col(VA_LO), Col(VB_LO)]), - Coord::Sum(vec![Col(VA_HI), Col(VB_HI)]), - Coord::Sum(vec![Col(VA_TOP), Col(VB_TOP)]), - ]; +impl ClassSpec { + /// Whether the row writes a register, in clock slot 3. + pub const fn writes_register(&self) -> bool { + !matches!(self.ram, Ram::Block) } - let (a, b) = ([VA_LO, VA_HI, VA_TOP], [VB_LO, VB_HI, VB_TOP]); - let lane = |i: usize| Coord::Sum(TOWER_LANES[i].iter().map(|&(j, k)| Prod(a[j], b[k], 0)).collect()); - [lane(0), lane(1), lane(2)] -} -impl Table for Arith { - fn n_committed_columns(&self) -> usize { - arith::N - } - fn count_columns(&self) -> &'static [usize] { - use arith::*; - &[RA, RB, RC, RBC] + /// Accesses per row: the registers', then RAM's. + pub const fn n_accesses(&self) -> usize { + match self.ram { + Ram::None => 3, + Ram::Read | Ram::Write => 4, + Ram::Block => 2 + hash::WORDS, + } } - fn flushes(&self, f: &mut FlushBuilder) { - use arith::*; - f.state_step(PC, FP); - f.bytecode( - PC, - RBC, - if self.is_xor { OP_XOR } else { OP_MUL }, - &[Col(OA), Col(OB), Col(OC), Const(F64::ZERO), Const(F64::ZERO)], - ); - f.memory(Prod(FP, OA, 0), RA, VA_LO, VA_HI, VA_TOP); - f.memory(Prod(FP, OB, 0), RB, VB_LO, VB_HI, VB_TOP); - f.memory_coords(Prod(FP, OC, 0), RC, arith_result(self.is_xor)); + + /// The clock slots of the row's accesses, in the order of their columns. + pub fn slots(&self) -> Vec { + let [s1, s2, sd] = REG_SLOTS; + match self.ram { + Ram::None => vec![s1, s2, sd], + Ram::Read | Ram::Write => vec![s1, s2, sd, RAM_SLOT], + Ram::Block => [s1, s2].into_iter().chain((0..hash::WORDS).map(block_slot)).collect(), + } } - fn fill(&self, ctx: &FillCtx, out: &mut [ColumnOut]) { - use arith::*; - let rows = if self.is_xor { &ctx.trace.xor } else { &ctx.trace.mul }; - ctx.col(out, rows, PC, |r| ctx.g_at(r.pc)); - ctx.col(out, rows, FP, |r| ctx.g_at(r.fp)); - // The offsets and both operand words come out of ONE bytecode decode: split - // across passes, the row's instruction is fetched once per pass. - ctx.cols(out, rows, OA, |r| { - let (a, b, c) = ctx.ternary_operands(r.pc); - let (va, vb) = (ctx.limbs(r.fp + a), ctx.limbs(r.fp + b)); - [ - ctx.g_at(a), - ctx.g_at(b), - ctx.g_at(c), - va[0], - va[1], - va[2], - vb[0], - vb[1], - vb[2], - ] - }); - ctx.cols(out, rows, RA, |r| [r.ra, r.rb, r.rc]); - ctx.col(out, rows, RBC, |r| r.bytecode_read); + + /// How far the clock advances per row: past its last slot. + pub const fn stride(&self) -> u32 { + match self.ram { + Ram::Block => block_slot(hash::WORDS), + _ => CLOCK_STRIDE, + } } } -// ---- SET --------------------------------------------------------------------- - -struct SetTable; - -mod set { - pub const PC: usize = 0; - pub const FP: usize = 1; - pub const O: usize = 2; - // The stored immediate's three K-limbs ride the bytecode's spare slots. - pub const K_LO: usize = 3; - pub const K_HI: usize = 4; - pub const K_TOP: usize = 5; - pub const R: usize = 6; - pub const RBC: usize = 7; - pub const N: usize = 8; +pub static ALU: ClassSpec = ClassSpec { + class: Class::Alu, + name: "ALU", + control: true, + ram: Ram::None, + circuit: rv::circuits::alu, + k_log: 10, + ports: &[Word::V1, Word::V2, Word::Imm, Word::Flags, Word::Out, Word::Taken], + n_inputs: 4, +}; +pub static LOAD: ClassSpec = ClassSpec { + class: Class::Load, + name: "LOAD", + control: false, + ram: Ram::Read, + circuit: rv::circuits::load, + k_log: 10, + ports: &[ + Word::V1, + Word::Imm, + Word::Flags, + Word::Cell(0), + Word::Address, + Word::Out, + ], + n_inputs: 4, +}; +pub static STORE: ClassSpec = ClassSpec { + class: Class::Store, + name: "STORE", + control: false, + ram: Ram::Write, + circuit: rv::circuits::store, + k_log: 10, + ports: &[ + Word::V1, + Word::V2, + Word::Imm, + Word::Flags, + Word::Cell(0), + Word::Address, + Word::CellNew(0), + Word::Out, + ], + n_inputs: 5, +}; + +pub static SHIFT: ClassSpec = ClassSpec { + class: Class::Shift, + name: "SHIFT", + control: false, + ram: Ram::None, + circuit: rv::circuits::shift, + k_log: 10, + ports: &[Word::V1, Word::V2, Word::Imm, Word::Flags, Word::Out], + n_inputs: 4, +}; +pub static MUL: ClassSpec = ClassSpec { + class: Class::Mul, + name: "MUL", + control: false, + ram: Ram::None, + circuit: rv::circuits::mul, + k_log: 12, + ports: &[Word::V1, Word::V2, Word::Flags, Word::Out], + n_inputs: 3, +}; +pub static MULH: ClassSpec = ClassSpec { + class: Class::Mulh, + name: "MULH", + control: false, + ram: Ram::None, + circuit: rv::circuits::mulh, + k_log: 13, + ports: &[Word::V1, Word::V2, Word::Flags, Word::Out], + n_inputs: 3, +}; + +pub static DIV: ClassSpec = ClassSpec { + class: Class::Div, + name: "DIV", + control: false, + ram: Ram::None, + circuit: rv::circuits::div, + k_log: 13, + ports: &[ + Word::V1, + Word::V2, + Word::Flags, + Word::HintQ, + Word::HintR, + Word::Out, + Word::Bad, + ], + n_inputs: 5, +}; + +/// The BLAKE2s precompile ([`hash`]): the counter is `v2`, the finalization word the +/// flags, and the block's words are the row's cells, the result's four rewritten. +pub static HASH: ClassSpec = ClassSpec { + class: Class::Hash, + name: "HASH", + control: false, + ram: Ram::Block, + circuit: rv::circuits::blake2s, + k_log: 14, + ports: &[ + Word::V2, + Word::Flags, + Word::Cell(0), + Word::Cell(1), + Word::Cell(2), + Word::Cell(3), + Word::Cell(8), + Word::Cell(9), + Word::Cell(10), + Word::Cell(11), + Word::Cell(12), + Word::Cell(13), + Word::Cell(14), + Word::Cell(15), + Word::CellNew(4), + Word::CellNew(5), + Word::CellNew(6), + Word::CellNew(7), + ], + n_inputs: 14, +}; + +/// The tables, in the order of `row_counts` / `taus` throughout `cpu`. Table `t`'s +/// class tag in the bytecode is `g^t`. +pub const N_TABLES: usize = 8; +pub static CLASSES: [&ClassSpec; N_TABLES] = [&ALU, &LOAD, &STORE, &SHIFT, &MUL, &MULH, &DIV, &HASH]; + +/// The table running `class`, if it has one yet. +pub fn table_of(class: Class) -> Option { + CLASSES.iter().position(|spec| spec.class == class) } -impl Table for SetTable { - fn n_committed_columns(&self) -> usize { - set::N - } - fn count_columns(&self) -> &'static [usize] { - use set::*; - &[R, RBC] - } - fn flushes(&self, f: &mut FlushBuilder) { - use set::*; - f.state_step(PC, FP); - // The immediate's three limbs occupy bytecode operand slots o2..o4 - // (matching layout::operands for SET). - f.bytecode( - PC, - RBC, - OP_SET, - &[Col(O), Col(K_LO), Col(K_HI), Col(K_TOP), Const(F64::ZERO)], - ); - // The stored constant K is the cell's value. - f.memory(Prod(FP, O, 0), R, K_LO, K_HI, K_TOP); - } - fn fill(&self, ctx: &FillCtx, out: &mut [ColumnOut]) { - use set::*; - let rows = &ctx.trace.set; - // The offset and the stored immediate are the instruction's. - let imm = |r: &Srow| match ctx.prog[r.pc as usize] { - Op::Set { o, k } => (o, k), - op => unreachable!("a SET row's pc {} holds {op:?}", r.pc), - }; - ctx.col(out, rows, PC, |r| ctx.g_at(r.pc)); - ctx.col(out, rows, FP, |r| ctx.g_at(r.fp)); - ctx.col(out, rows, O, |r| ctx.g_at(imm(r).0)); - ctx.cols(out, rows, K_LO, |r| { - let k = imm(r).1; - [F64(k.c0), F64(k.c1), F64(k.c2)] - }); - ctx.col(out, rows, R, |r| r.r); - ctx.col(out, rows, RBC, |r| r.bytecode_read); - } +pub fn tables() -> [&'static dyn Table; N_TABLES] { + static TABLES: std::sync::OnceLock> = std::sync::OnceLock::new(); + let tables = TABLES.get_or_init(|| (0..N_TABLES).map(ClassTable::new).collect()); + std::array::from_fn(|t| &tables[t] as &dyn Table) } -// ---- DEREF ------------------------------------------------------------------- - -struct DerefTable; - -mod deref { - pub const PC: usize = 0; - pub const FP: usize = 1; - pub const O1: usize = 2; - pub const O2: usize = 3; - pub const O3: usize = 4; - pub const FPC: usize = 5; - pub const FFP: usize = 6; - // The pointer word is a SINGLE K-lane, so its extension limbs are provably - // zero: they are NOT committed, and the memory read carries literal zeros - // there. Being a column is what puts it in K, and the pointer-relative - // address it forms on the bus, `p·obe`, is a K product for the same reason. - pub const P: usize = 7; - // The local cell, a full 192-bit word. The store target is DERIVED from it, - // the two flags, `pc` and `fp`, so it is no column. - pub const V3_LO: usize = 8; - pub const V3_HI: usize = 9; - pub const V3_TOP: usize = 10; - pub const R1: usize = 11; - pub const R2: usize = 12; - pub const R3: usize = 13; - pub const RBC: usize = 14; - pub const N: usize = 15; +/// Table `t`'s circuit words that are columns, as `(port, local column)`. +pub(crate) fn word_columns(t: usize) -> Vec<(usize, usize)> { + let cols = Cols::new(CLASSES[t]); + let ports = CLASSES[t].ports.iter().enumerate(); + ports.filter_map(|(port, &w)| Some((port, cols.word(w)?))).collect() } -/// The stored word's three K-lanes as forms: -/// `v_2 = (1+f_pc+f_fp)·v_3 + f_pc·(g²·pc) + f_fp·fp`, the flag-selected source -/// of §sec:tab-deref. The `pc` source is the virtual return target `g²·pc`, a -/// free `×g²` on the product coordinate. Only the low lane takes the two K-valued -/// sources; the upper two are the gated local lanes alone. -fn deref_store() -> [Coord; 3] { - use deref::*; - let gated = |v: usize| vec![Col(v), Prod(FPC, v, 0), Prod(FFP, v, 0)]; - let mut lo = gated(V3_LO); - lo.push(Prod(FPC, PC, 2)); - lo.push(Prod(FFP, FP, 0)); - [Coord::Sum(lo), Coord::Sum(gated(V3_HI)), Coord::Sum(gated(V3_TOP))] +/// The slot of a bytecode tuple that holds a row's [`Word::Bad`]: past every field of +/// an entry, where the program is zero. +pub const BAD_SLOT: usize = 13; + +/// A class table's local columns: `pc, ts, a1, a2, pc4, v1, v2, flags`, then the +/// optional groups in the order of the fields below, then the accesses and the +/// bytecode read's count. +#[derive(Clone, Copy)] +struct Cols { + pc: usize, + ts: usize, + // The bytecode entry's fields that are no circuit word. + a1: usize, + a2: usize, + pc4: usize, + v1: usize, + v2: usize, + flags: usize, + /// The register write: `ad`, what it held, and `out`. A hash row has none. + rd: Option, + /// `dt`, `link`, `jalr`, then `taken`. + control: Option, + /// The immediate, which a hash row has not. + imm: Option, + /// The bus address and the cell, then what a store leaves in the cell. + ram: Option, + /// The hash's block: its sixteen words as found, then the four the row writes. + block: Option, + bad: Option, + acc: Acc, + /// The bytecode read's count, the last column. + rbc: usize, } -impl Table for DerefTable { - fn n_committed_columns(&self) -> usize { - deref::N - } - fn count_columns(&self) -> &'static [usize] { - use deref::*; - &[R1, R2, R3, RBC] - } - fn flushes(&self, f: &mut FlushBuilder) { - use deref::*; - f.state_step(PC, FP); - f.bytecode(PC, RBC, OP_DEREF, &[Col(O1), Col(O2), Col(O3), Col(FPC), Col(FFP)]); - // The pointer cell and the local cell are frame-relative; the store target - // is pointer-relative, so its address is `p·obe`, and its value is the - // flag-selected source rather than a column. - f.memory_k(Prod(FP, O1, 0), R1, P); - f.memory_coords(Prod(P, O2, 0), R2, deref_store()); - f.memory(Prod(FP, O3, 0), R3, V3_LO, V3_HI, V3_TOP); - } - fn fill(&self, ctx: &FillCtx, out: &mut [ColumnOut]) { - use deref::*; - let rows = &ctx.trace.deref; - // Offsets and store mode are the instruction's. Neither the store target's - // address nor its value is committed. - let ins = |r: &Drow| match ctx.prog[r.pc as usize] { - Op::Deref { o1, o2, o3, mode } => (o1, o2, o3, mode), - op => unreachable!("a DEREF row's pc {} holds {op:?}", r.pc), +impl Cols { + fn new(spec: &ClassSpec) -> Self { + let mut next = 0; + let mut take = |n: usize| { + next += n; + next - n }; - ctx.col(out, rows, PC, |r| ctx.g_at(r.pc)); - ctx.col(out, rows, FP, |r| ctx.g_at(r.fp)); - debug_assert!( - rows.iter().all(|r| { - let p = ctx.mem[(r.fp + ins(r).0) as usize]; - p.c1 == 0 && p.c2 == 0 - }), - "deref pointer must be K-valued" - ); - // The three offsets, the two mode flags, the pointer lane and the local - // word all follow from ONE bytecode decode. - ctx.cols(out, rows, O1, |r| { - let (o1, o2, o3, mode) = ins(r); - let v3 = ctx.limbs(r.fp + o3); - [ - ctx.g_at(o1), - ctx.g_at(o2), - ctx.g_at(o3), - mode.f_pc(), - mode.f_fp(), - F64(ctx.mem[(r.fp + o1) as usize].c0), - v3[0], - v3[1], - v3[2], - ] - }); - ctx.cols(out, rows, R1, |r| [r.r1, r.r2, r.r3]); - ctx.col(out, rows, RBC, |r| r.bytecode_read); + let (pc, ts, a1, a2, pc4) = (take(1), take(1), take(1), take(1), take(1)); + let (v1, v2, flags) = (take(1), take(1), take(1)); + let rd = spec.writes_register().then(|| take(3)); + let control = spec.control.then(|| take(4)); + let imm = spec.ports.contains(&Word::Imm).then(|| take(1)); + let (ram, block) = match spec.ram { + Ram::None => (None, None), + Ram::Read => (Some(take(2)), None), + Ram::Write => (Some(take(3)), None), + Ram::Block => (None, Some(take(hash::WORDS + 4))), + }; + let bad = spec.ports.contains(&Word::Bad).then(|| take(1)); + let n = spec.n_accesses(); + let acc = Acc { base: take(5 * n), n }; + Self { + pc, + ts, + a1, + a2, + pc4, + v1, + v2, + flags, + rd, + control, + imm, + ram, + block, + bad, + acc, + rbc: take(1), + } + } + + /// The word's column, if it has one: a hint has none. + fn word(&self, word: Word) -> Option { + let out_word = hash::OUT as usize / 8; + Some(match word { + Word::Flags => self.flags, + Word::Imm => self.imm.expect("a hash row has no immediate"), + Word::V1 => self.v1, + Word::V2 => self.v2, + Word::Out => self.rd.expect("a hash row has no result word") + 2, + Word::Taken => self.control.expect("only a control class has a taken word") + 3, + Word::Address => self.ram.expect("only a load or a store has an address"), + Word::Cell(k) => match (self.ram, self.block) { + (Some(address), _) => address + 1, + (_, Some(block)) => block + k as usize, + _ => panic!("the class names no cell"), + }, + Word::CellNew(k) => match (self.ram, self.block) { + (Some(address), _) => address + 2, + (_, Some(block)) => block + hash::WORDS + k as usize - out_word, + _ => panic!("the class rewrites no cell"), + }, + Word::Bad => self.bad.expect("the class asserts nothing"), + Word::HintQ | Word::HintR => return None, + }) } } -// ---- JUMP -------------------------------------------------------------------- - -struct JumpTable; - -mod jump { - pub const PC: usize = 0; - pub const FP: usize = 1; - pub const OC: usize = 2; - pub const OD: usize = 3; - pub const OF: usize = 4; - // The condition, destination and frame words are all K-valued, so each is a - // SINGLE lane read through `memory_k`: bus balance forces the stored words - // into K, exactly as for the DEREF pointer. A guest branches on g-powers, - // never on an arbitrary word: `assert a != b` takes an inverse hint instead - // of a branch (§sec:prog-div-ne). - pub const V_COND: usize = 5; - pub const V_PC: usize = 6; - pub const V_FP: usize = 7; - pub const RC: usize = 8; - pub const RD: usize = 9; - pub const RF: usize = 10; - pub const RBC: usize = 11; - // Local witness columns (committed, never flushed): the inverse hint `w = c⁻¹` - // and the taken indicator `b = [c ≠ 0]` it certifies (the `JUMP` table in - // `doc/leanvm/body/07-instruction-tables.tex`). Both are single K lanes. - pub const W: usize = 12; - pub const B: usize = 13; - pub const N: usize = 14; +/// The table of one instruction class (§sec:tables): the state step, the bytecode +/// read, two register reads, the RAM access of a load or a store, and one register write. +struct ClassTable { + index: usize, + spec: &'static ClassSpec, + cols: Cols, + counts: Vec, } -impl Table for JumpTable { +impl ClassTable { + fn new(index: usize) -> Self { + let spec = CLASSES[index]; + let cols = Cols::new(spec); + assert_eq!(cols.rbc, cols.acc.end(), "the counts are the table's last columns"); + Self { + index, + spec, + cols, + // The accesses' counts end `acc`, and the bytecode's follows. + counts: (cols.acc.count_lo(0)..=cols.rbc).collect(), + } + } + + fn eval(&self, w: &[F192], cols: &[T]) -> F192 { + access_identities(w, cols, self.cols.ts, self.cols.acc) + } +} + +impl Table for ClassTable { fn n_committed_columns(&self) -> usize { - jump::N + self.cols.rbc + 1 } - fn count_columns(&self) -> &'static [usize] { - use jump::*; - &[RC, RD, RF, RBC] + fn count_columns(&self) -> &[usize] { + &self.counts } fn n_constraints(&self) -> usize { - 2 // the two indicator identities; the selections ride the state push + self.cols.acc.n } - fn eval_constraint(&self, pows: &[F192], cols: &[F192], quadratic: bool) -> F192 { - jump_identity(pows, cols, quadratic) + fn constraint_weights(&self, pows: &[F192]) -> Vec { + access_weights(pows, &self.spec.slots()) } - fn eval_constraint_k(&self, pows: &[F192], cols: &[F64], quadratic: bool) -> F192 { - jump_identity(pows, cols, quadratic) + // `quadratic` is ignored because `access_identities` is homogeneous of degree two: + // every term is a product of two columns, so the quadratic part IS the identity. A + // linear or constant term added there would have to be split out here. + fn eval_constraint(&self, w: &[F192], cols: &[F192], _quadratic: bool) -> F192 { + self.eval(w, cols) } - fn flushes(&self, f: &mut FlushBuilder) { - use jump::*; - // The successor state is DERIVED: `b·d + (b+1)·g·pc` and `b·f + (b+1)·fp`, - // each degree 2 in K columns, so neither successor is committed. Written - // out in characteristic 2 as `b·d + b·(g·pc) + g·pc`. - f.state_derived( - PC, - FP, - Coord::Sum(vec![Prod(B, V_PC, 0), Prod(B, PC, 1), GCol(PC, 1)]), - Coord::Sum(vec![Prod(B, V_FP, 0), Prod(B, FP, 0), Col(FP)]), - ); - f.bytecode( - PC, - RBC, - OP_JUMP, - &[Col(OC), Col(OD), Col(OF), Const(F64::ZERO), Const(F64::ZERO)], - ); - f.memory_k(Prod(FP, OC, 0), RC, V_COND); - f.memory_k(Prod(FP, OD, 0), RD, V_PC); - f.memory_k(Prod(FP, OF, 0), RF, V_FP); + fn eval_constraint_k(&self, w: &[F192], cols: &[F64], _quadratic: bool) -> F192 { + self.eval(w, cols) } - fn fill(&self, ctx: &FillCtx, out: &mut [ColumnOut]) { - use jump::*; - let rows = &ctx.trace.jump; - let ins = |r: &Jrow| match ctx.prog[r.pc as usize] { - Op::Jump { oc, od, of } => (oc, od, of), - op => unreachable!("a JUMP row's pc {} holds {op:?}", r.pc), - }; - let cell = |r: &Jrow, o: u32| ctx.mem[(r.fp + o) as usize]; - let cond = |r: &Jrow| cell(r, ins(r).0); - ctx.col(out, rows, PC, |r| ctx.g_at(r.pc)); - ctx.col(out, rows, FP, |r| ctx.g_at(r.fp)); - // The three offsets and the three cells they name come out of ONE decode. - // Those cells are K-valued on every row, taken or not (`cpu::execute` - // rejects anything else), so each is one lane and the memory flush carries - // literal zeros above it. - ctx.cols(out, rows, OC, |r| { - let (oc, od, of) = ins(r); - [ - ctx.g_at(oc), - ctx.g_at(od), - ctx.g_at(of), - F64(cell(r, oc).c0), - F64(cell(r, od).c0), - F64(cell(r, of).c0), - ] - }); - // The is-nonzero witness `w = c⁻¹` (0 where c = 0) for every row, in one - // batched Montgomery inversion. `prefix[i]` is the - // running product of the nonzero conditions before row `i`, so `acc` ends - // as their full product (nonzero, hence invertible). The taken indicator - // `b = [c ≠ 0]` falls out of the same pass, so it costs no extra decode. - let (w, b) = { - let mut acc = F192::ONE; - let mut prefix: Vec = Vec::with_capacity(rows.len()); - let mut b = vec![F64::ZERO; rows.len()]; - for (i, r) in rows.iter().enumerate() { - prefix.push(acc); - let c = cond(r); - if !c.is_zero() { - acc *= c; - b[i] = F64::ONE; - } - } - let mut inv = acc.inv(); - let mut w = vec![F192::ZERO; rows.len()]; - for (i, r) in rows.iter().enumerate().rev() { - let c = cond(r); - if !c.is_zero() { - w[i] = inv * prefix[i]; - inv *= c; - } + fn flushes(&self, f: &mut FlushBuilder) { + let c = &self.cols; + // What the row derives: the next `pc`, `pc4 + taken·dt + jalr·(out + pc4)`, and + // what `rd` receives, `out + link·(out + pc4)`, each of degree 2 (§sec:m3). + let (npc, vd, control) = match (c.control, c.rd) { + (Some(dt), Some(ad)) => { + let (link, jalr, taken, out) = (dt + 1, dt + 2, dt + 3, ad + 2); + ( + Coord::Sum(vec![ + Col(c.pc4), + Prod(taken, dt, 0), + Prod(jalr, out, 0), + Prod(jalr, c.pc4, 0), + ]), + Some(Coord::Sum(vec![Col(out), Prod(link, out, 0), Prod(link, c.pc4, 0)])), + vec![Col(dt), Col(link), Col(jalr)], + ) } - (w, b) + (_, rd) => (Col(c.pc4), rd.map(|ad| Col(ad + 2)), Vec::new()), }; - ctx.cols_at(out, rows.len(), W, |i| [F64(w[i].c0), b[i]]); - ctx.cols(out, rows, RC, |r| [r.rc, r.rd, r.rf]); - ctx.col(out, rows, RBC, |r| r.bytecode_read); - } -} - -// ---- BLAKE2s ------------------------------------------------------------------ - -/// `BLAKE2s` (“BLAKE2s” in `doc/leanvm/body/07-instruction-tables.tex`): one standard compression. The four 128-bit message -/// chunks are addressed *independently* at `fp·o_i` (`o_i = g^{ins[i]}`), each a -/// single cell, with no forced contiguity between chunks, so a caller hashing e.g. -/// `(tweak, pp)` need not copy them into adjacent cells. The chaining value and the -/// 32-byte output each occupy two consecutive cells, based at `fp·o_cv` and -/// `fp·o_c`, and the metadata is one more cell at `fp·o_md`, so the row reads nine -/// cells in all. No address is committed: each rides the bus as the product `fp·o_X` -/// (§sec:m3). The compression relating output words to input words carries no -/// table constraint either: it is proven by flock's R1CS validity via `q_flock` -/// (§hash_flock), which leaves this table with no identity of its own. -/// -/// A 128-bit chunk is two flock 64-bit words (lo, hi lanes), so the eighteen -/// memory-borne flock words are eighteen value LANE columns over the nine cells. -/// They are listed in `n_committed_columns` (they need a local index for the -/// flushes and are filled from the trace for the bus), but `cpu` treats them as -/// VIRTUAL (not committed) and routes their bus claims to `q_flock`, which already -/// holds those words (see [`BLAKE2S_VALUE_COLS`]). -struct Blake2sTable; - -pub(crate) mod blake2st { - pub const PC: usize = 0; - pub const FP: usize = 1; - pub const O_M0: usize = 2; // operand g-powers (offsets) of the four message cells … - pub const O_M1: usize = 3; - pub const O_M2: usize = 4; - pub const O_M3: usize = 5; - pub const O_CV: usize = 6; // … the chaining-value base … - pub const O_OUT: usize = 7; // … the output base … - pub const O_MD: usize = 8; // … and the metadata cell - // The eighteen flock lanes: a (lo, hi) pair for each of the nine cells, - // the four message cells first, then the output pair, the chaining-value - // pair, and last the metadata cell's counter and flag lanes. - pub const V_M0: usize = 9; // m0.lo, m0.hi, m1.lo, m1.hi - pub const V_M2: usize = 13; // m2.lo, m2.hi, m3.lo, m3.hi - pub const V_OUT0: usize = 17; // out0.lo, out0.hi, out1.lo, out1.hi - pub const V_CV0: usize = 21; // cv0.lo, cv0.hi, cv1.lo, cv1.hi - pub const MD0: usize = 25; // metadata: the counter lane … - pub const MD1: usize = 26; // … and the final ‖ last_node lane - pub const R_M0: usize = 27; // one read count per cell: the four message cells … - pub const R_M1: usize = 28; - pub const R_M2: usize = 29; - pub const R_M3: usize = 30; - pub const R_CV0: usize = 31; // … the two chaining-value cells … - pub const R_CV1: usize = 32; - pub const R_OUT0: usize = 33; // … the two output cells … - pub const R_OUT1: usize = 34; - pub const R_MD: usize = 35; // … and the metadata cell. - pub const RBC: usize = 36; - pub const N: usize = 37; -} - -impl Table for Blake2sTable { - fn n_committed_columns(&self) -> usize { - blake2st::N - } - fn count_columns(&self) -> &'static [usize] { - use blake2st::*; - &[R_M0, R_M1, R_M2, R_M3, R_CV0, R_CV1, R_OUT0, R_OUT1, R_MD, RBC] - } - fn flushes(&self, f: &mut FlushBuilder) { - use blake2st::*; - f.state_step(PC, FP); - f.bytecode( - PC, - RBC, - OP_BLAKE2S, - &[ - Col(O_M0), - Col(O_M1), - Col(O_M2), - Col(O_M3), - Col(O_CV), - Col(O_OUT), - Col(O_MD), - ], - ); - // Nine cell reads: four independent 128-bit message cells, the chaining - // value's two consecutive cells (ACV, g·ACV), the output's two - // consecutive cells (AC, g·AC), and the metadata cell. Each carries its - // chunk's two lanes with a literal-zero top limb (`memory_128`), so the - // canonical embedding is proof-enforced and the zero limbs are never - // committed. A consecutive cell is a free ×g on the product's g-power. - f.memory_128(Prod(FP, O_M0, 0), R_M0, V_M0, V_M0 + 1); - f.memory_128(Prod(FP, O_M1, 0), R_M1, V_M0 + 2, V_M0 + 3); - f.memory_128(Prod(FP, O_M2, 0), R_M2, V_M2, V_M2 + 1); - f.memory_128(Prod(FP, O_M3, 0), R_M3, V_M2 + 2, V_M2 + 3); - f.memory_128(Prod(FP, O_CV, 0), R_CV0, V_CV0, V_CV0 + 1); - f.memory_128(Prod(FP, O_CV, 1), R_CV1, V_CV0 + 2, V_CV0 + 3); - f.memory_128(Prod(FP, O_OUT, 0), R_OUT0, V_OUT0, V_OUT0 + 1); - f.memory_128(Prod(FP, O_OUT, 1), R_OUT1, V_OUT0 + 2, V_OUT0 + 3); - // The metadata rides the memory bus like every other operand: the read - // is what binds flock's counter and flag inputs, so a compile-time - // counter is pinned by the `SET` immediate that wrote the cell. - f.memory_128(Prod(FP, O_MD, 0), R_MD, MD0, MD1); + f.state(c.pc, c.ts, npc, self.spec.stride()); + // A row without a register write or an immediate reads their constants off + // the entry: the sink, and zero. + let mut entry = vec![ + Const(SEP_BYTECODE), + Col(c.pc), + Col(c.rbc), + Const(g_pow(self.index)), + Col(c.flags), + Col(c.a1), + Col(c.a2), + c.rd.map_or(Const(F64(SINK as u64)), Col), + c.imm.map_or(Const(F64::ZERO), Col), + Col(c.pc4), + ]; + entry.extend(control); + if let Some(bad) = c.bad { + entry.resize(BAD_SLOT, Const(F64::ZERO)); + entry.push(Col(bad)); + } + f.counted(entry, c.rbc); + let [s1, s2, sd] = REG_SLOTS; + f.access(SEP_REG, Col(c.a1), c.ts, c.acc, 0, s1, Col(c.v1), Col(c.v1)); + f.access(SEP_REG, Col(c.a2), c.ts, c.acc, 1, s2, Col(c.v2), Col(c.v2)); + if let (Some(ad), Some(vd)) = (c.rd, vd) { + f.access(SEP_REG, Col(ad), c.ts, c.acc, 2, sd, Col(ad + 1), vd); + } + // The cell a load or a store names is the circuit's word, so an access outside + // RAM, or a misaligned one, pulls a tuple nothing pushed. + if let Some(address) = c.ram { + let (cell, new) = (address + 1, address + if self.spec.ram == Ram::Write { 2 } else { 1 }); + f.access(SEP_MEM, Col(address), c.ts, c.acc, 3, RAM_SLOT, Col(cell), Col(new)); + } + // The hash's block: word `k` at `v1 ^ 8k`, which is `v1 + 8k` in the field. + if let Some(block) = c.block { + let out_word = hash::OUT as usize / 8; + for k in 0..hash::WORDS { + let addr = Coord::Sum(vec![Col(c.v1), Const(F64(8 * k as u64))]); + let new = match k.wrapping_sub(out_word) { + j if j < 4 => Col(block + hash::WORDS + j), + _ => Col(block + k), + }; + f.access(SEP_MEM, addr, c.ts, c.acc, 2 + k, block_slot(k), Col(block + k), new); + } + } } fn fill(&self, ctx: &FillCtx, out: &mut [ColumnOut]) { - use blake2st::*; - let rows = &ctx.trace.blake2s; - let ad = |r: &Brow| blake2s_addresses(ctx.prog, r); - ctx.col(out, rows, PC, |r| ctx.g_at(r.pc)); - ctx.col(out, rows, FP, |r| ctx.g_at(r.fp)); - // O_M0..O_MD are the seven base addresses' offsets, from the instruction decode. - ctx.cols(out, rows, O_M0, |r| ad(r).map(|a| ctx.g_at(a - r.fp))); - // The eighteen memory-borne flock words are the nine cells' lo/hi lanes: - // the four message cells, then the cv pair, the output pair and the - // metadata. A cell's two lanes are one read, so each group of four takes two. - let word_pair = |c0: u32, c1: u32| { - let (w0, w1) = (ctx.mem[c0 as usize], ctx.mem[c1 as usize]); - [F64(w0.c0), F64(w0.c1), F64(w1.c0), F64(w1.c1)] - }; - ctx.cols(out, rows, V_M0, |r| { - let a = ad(r); - word_pair(a[0], a[1]) - }); - ctx.cols(out, rows, V_M2, |r| { - let a = ad(r); - word_pair(a[2], a[3]) - }); - ctx.cols(out, rows, V_OUT0, |r| { - let a = ad(r); - word_pair(a[5], a[5] + 1) - }); - ctx.cols(out, rows, V_CV0, |r| { - let a = ad(r); - word_pair(a[4], a[4] + 1) - }); - ctx.cols(out, rows, MD0, |r| { - let md = ctx.mem[ad(r)[6] as usize]; - [F64(md.c0), F64(md.c1)] - }); - ctx.cols(out, rows, R_M0, |r| { + let c = &self.cols; + let rows: &[Row] = &ctx.trace.rows[self.index]; + let p = ctx.program; + let entry = |r: &Row| &p.entries[r.index as usize]; + ctx.cols(out, rows, c.pc, |r| { + let (e, pc) = (entry(r), p.pc_of(r.index as usize)); [ - r.ra[0], r.ra[1], r.rb[0], r.rb[1], r.rcv[0], r.rcv[1], r.rc[0], r.rc[1], r.rmd, + F64(pc), + r.ts, + F64(e.a1 as u64), + F64(e.a2 as u64), + F64(pc.wrapping_add(4)), + F64(r.v1), + F64(r.v2), + F64(e.flags), ] }); - ctx.col(out, rows, RBC, |r| r.bytecode_read); + if let Some(ad) = c.rd { + ctx.cols(out, rows, ad, |r| [F64(entry(r).ad as u64), F64(r.vd_old), F64(r.out)]); + } + if let Some(dt) = c.control { + ctx.cols(out, rows, dt, |r| { + let e = entry(r); + [ + F64(p.dt_of(r.index as usize)), + F64(e.link as u64), + F64(e.jalr as u64), + F64(r.taken as u64), + ] + }); + } + if let Some(imm) = c.imm { + ctx.col(out, rows, imm, |r| F64(entry(r).imm)); + } + if let Some(address) = c.ram { + ctx.cols(out, rows, address, |r| [F64(r.ram.address), F64(r.ram.old)]); + if self.spec.ram == Ram::Write { + ctx.col(out, rows, address + 2, |r| F64(r.ram.new)); + } + } + if let Some(block) = c.block { + fn hash(r: &Row) -> &HashRow { + r.hash.as_ref().expect("a hash row has its block") + } + ctx.cols(out, rows, block, |r| hash(r).block.map(F64)); + ctx.cols(out, rows, block + hash::WORDS, |r| hash(r).out.map(F64)); + } + if let Some(bad) = c.bad { + ctx.col(out, rows, bad, |_| F64::ZERO); + } + ctx.accesses(out, rows, c.acc, Row::accesses); + ctx.col(out, rows, c.rbc, |r| r.bytecode_read); } } @@ -960,54 +851,30 @@ mod tests { use super::*; use primitives::field::powers; - const C: F192 = F192::new(0x0123_4567_89ab_cdef, 0xfeed_face_dead_beef, 0x1111_2222_3333_4444); - - /// The hand-unrolled tower product IS `E`'s multiplication, lane by lane. Both - /// `MUL`'s result coordinate and `JUMP`'s inverse identity are written out from - /// [`TOWER_LANES`], and neither can be checked against the field at run time (one - /// is a bus coordinate, the other a `K` identity), so pin the unrolling here. + /// An access's identity vanishes exactly when the previous timestamp, the two + /// chunks and the row's own timestamp agree: `x + gap + 1 = 4·cycle + slot`. #[test] - fn unrolled_tower_product_matches_field_product() { - let lanes = |v: F192| [F64(v.c0), F64(v.c1), F64(v.c2)]; - let (x, y) = (C, C * C + F192::ONE); - let got = [0, 1, 2].map(|i| tower_lane(i, lanes(x), lanes(y)).0); - assert_eq!(F192::new(got[0], got[1], got[2]), x * y); - } - - /// `JUMP`'s two identities vanish on an honest row, taken or not, and reject a - /// wrong indicator or, for a taken jump, an inverse that is not `cond⁻¹`. - #[test] - fn jump_identities_bind_indicator_and_inverse() { - let pows = powers(F192::new(0x9e37_79b9_7f4a_7c15, 0x1234_5678_9abc_def0, 7), 2); - // The condition is K-valued (`memory_k` on its read, §sec:tab-jump), so the - // pair is single-lane: `w = c⁻¹` in K too. - for cond in [F64::ZERO, F64(0x9e37_79b9_7f4a_7c15)] { - let mut row = vec![F64::ZERO; jump::N]; - let w = if cond.is_zero() { - F64::ZERO - } else { - F64(F192::from(cond).inv().c0) - }; - row[jump::V_COND] = cond; - row[jump::W] = w; - row[jump::B] = if cond.is_zero() { F64::ZERO } else { F64::ONE }; - assert_eq!(jump_identity(&pows, &row, false), F192::ZERO, "cond = {cond:?}"); - // On a zero condition the inverse is unconstrained, being multiplied by - // zero: what has to be pinned there is the indicator alone. - let forgeable: &[usize] = if cond.is_zero() { - &[jump::B] - } else { - &[jump::B, jump::W] - }; - for &col in forgeable { - let mut forged = row.clone(); - forged[col] += F64::ONE; - assert_ne!( - jump_identity(&pows, &forged, false), - F192::ZERO, - "column {col}, cond = {cond:?}" - ); - } - } + fn access_identity_is_the_strict_gap() { + use primitives::field::g_pow; + let table = ClassTable::new(0); + let (x, cycle, access) = (41usize, 70_000usize, 2usize); + let slot = REG_SLOTS[access] as usize; + let gap = CLOCK_STRIDE as usize * cycle + slot - x - 1; + assert!(gap >> RANGE_LOG > 0, "both chunks are exercised"); + let w = table.constraint_weights(&powers(F192::new(3, 5, 7), 3)); + let row = |gap: usize| { + // Accesses 0 and 1 stay all-zero, which the identity accepts. + let mut row = vec![F64::ZERO; table.n_committed_columns()]; + row[table.cols.ts] = g_pow(CLOCK_STRIDE as usize * cycle); + row[table.cols.acc.x(access)] = g_pow(x); + row[table.cols.acc.lo(access)] = range_lo_first() * g_pow(gap & 0xffff); + row[table.cols.acc.hi(access)] = (0..gap >> RANGE_LOG).fold(F64::ONE, |h, _| h * range_hi_ratio()); + row + }; + assert_eq!(table.eval(&w, &row(gap)), F192::ZERO); + let lifted: Vec = row(gap).into_iter().map(F192::from).collect(); + assert_eq!(table.eval(&w, &lifted), F192::ZERO); + assert_ne!(table.eval(&w, &row(gap + 1)), F192::ZERO); + assert_ne!(table.eval(&w, &row(gap + (1 << RANGE_LOG))), F192::ZERO); } } diff --git a/crates/lean_vm/src/vmhash.rs b/crates/lean_vm/src/vmhash.rs deleted file mode 100644 index 800007085..000000000 --- a/crates/lean_vm/src/vmhash.rs +++ /dev/null @@ -1,13 +0,0 @@ -//! VM-provable standard BLAKE2s hashing. -//! -//! The VM instruction exposes one BLAKE2s compression whose chaining value, byte -//! counter and finalization flags are all memory-supplied. A guest can therefore -//! implement the standard sequential compression chain for any input length, -//! including a zero-padded partial final block, and the length need not be known -//! when the program is compiled. - -/// Standard BLAKE2s of exactly 64 bytes (two 256-bit halves laid out -/// little-endian, the `Blake2s` opcode's default metadata), which is also the -/// PCS Merkle parent. Lives in [`fiat_shamir`] (the shared -/// [`fiat_shamir::FiatShamirState`] state is built on it). -pub use fiat_shamir::compress; diff --git a/crates/lean_vm/src/witness.rs b/crates/lean_vm/src/witness.rs index d198e26b0..8444c2237 100644 --- a/crates/lean_vm/src/witness.rs +++ b/crates/lean_vm/src/witness.rs @@ -27,14 +27,6 @@ impl Placement { } } -/// The stacked witness and the per-column placements (in input order). -#[cfg(test)] -pub(crate) struct Stacked { - pub shape: StackShape, - pub q: ArenaVec, - pub placements: Vec, -} - /// The committed stack's shape. `mu` is the log size everything public is derived /// from (the claims' selector coords, the PCS level ladder), while `n_lanes` counts /// the `2^(mu - LOG_BATCH)`-word lane blocks that actually carry data: the PCS @@ -155,27 +147,6 @@ pub fn split_stack<'a>(q: &'a mut [F64], placements: &[Placement]) -> Vec<&'a mu windows } -/// Stack the columns for a test: build the placements from their lengths, then -/// copy each into its window. -#[cfg(test)] -pub(crate) fn stack(cols: &[Vec]) -> Stacked { - let kappas: Vec> = cols - .iter() - .map(|c| { - assert!(!c.is_empty(), "column must be non-empty"); - Some(crate::log2_strict_usize(c.len())) - }) - .collect(); - let (placements, shape) = placements_of(&kappas); - // SAFETY: `split_stack` covers the whole allocation, and every window is - // copied into below. - let mut q = unsafe { alloc_stack(shape) }; - for (w, c) in split_stack(&mut q, &placements).into_iter().zip(cols) { - w.copy_from_slice(c); - } - Stacked { shape, q, placements } -} - #[cfg(test)] mod tests { use super::*; diff --git a/crates/lean_vm/tests/verifiers/act4.rs b/crates/lean_vm/tests/verifiers/act4.rs new file mode 100644 index 000000000..479c27aa3 --- /dev/null +++ b/crates/lean_vm/tests/verifiers/act4.rs @@ -0,0 +1,178 @@ +//! ACT4, the RISC-V architectural tests (riscv-arch-test 4.1.0), its I and M suites: +//! generated for leanVM's memory map by `conformance/act4/generate.sh`, which runs every +//! test on the Sail reference model and builds it again with Sail's results inside, +//! so that it checks itself. Not checked in: `generate.sh` writes them to the ignored +//! `conformance/act4/elf/`, or `LEANVM_ACT4` names another directory, which is how CI runs +//! them. Run on the interpreter, proven, and checked by both verifiers. A test exits +//! with the output zero when every check passes. Both tests are `#[ignore]`d, since +//! they need the generated files: `cargo test --release -p lean_vm --test verifiers +//! -- --ignored act4`. + +use super::python_verifier::PythonStatement; +use lean_vm::cpu::{Program, prove, verify_to_raw}; +use lean_vm::rv::{Guest, Machine, RAM_BASE, TEXT_BASE, Trap}; +use lean_vm::tables::N_TABLES; +use std::path::{Path, PathBuf}; + +/// Every test of the two suites, as `(extension, instruction)`, the file being +/// `/--00.elf`. +const TESTS: [(&str, &[&str]); 2] = [ + ( + "I", + &[ + "add", "addi", "addiw", "addw", "and", "andi", "auipc", "beq", "bge", "bgeu", "blt", "bltu", "bne", + "fence", "jal", "jalr", "lb", "lbu", "ld", "lh", "lhu", "lui", "lw", "lwu", "nop", "or", "ori", "sb", "sd", + "sh", "sll", "slli", "slliw", "sllw", "slt", "slti", "sltiu", "sltu", "sra", "srai", "sraiw", "sraw", + "srl", "srli", "srliw", "srlw", "sub", "subw", "sw", "xor", "xori", + ], + ), + ( + "M", + &[ + "div", "divu", "divuw", "divw", "mul", "mulh", "mulhsu", "mulhu", "mulw", "rem", "remu", "remuw", "remw", + ], + ), +]; + +const PASS: [u64; 4] = [0; 4]; +const CYCLE_CAP: u64 = 1 << 20; + +struct Test { + name: String, + text: Vec, + program: Program, +} + +/// Every test, from the directory holding exactly their files. +fn suite() -> Vec { + let root = std::env::var_os("LEANVM_ACT4").map_or_else( + || PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../conformance/act4/elf"), + PathBuf::from, + ); + let listing = |directory: &Path| -> Vec { + let mut names: Vec = std::fs::read_dir(directory) + .unwrap_or_else(|error| { + panic!( + "{}: {error}; generate the files with conformance/act4/generate.sh", + directory.display() + ) + }) + .map(|entry| entry.unwrap().file_name().into_string().unwrap()) + .collect(); + names.sort(); + names + }; + assert_eq!( + listing(&root), + TESTS.map(|(extension, _)| extension), + "{}", + root.display() + ); + let mut suite = Vec::new(); + for (extension, instructions) in TESTS { + let directory = root.join(extension); + let names: Vec = instructions + .iter() + .map(|name| format!("{extension}-{name}-00")) + .collect(); + let files: Vec = names.iter().map(|name| format!("{name}.elf")).collect(); + assert_eq!(listing(&directory), files, "{}", directory.display()); + for name in names { + let elf = std::fs::read(directory.join(format!("{name}.elf"))).unwrap(); + let guest = Guest::from_elf(&elf).unwrap_or_else(|error| panic!("{name}: {error}")); + let program = Program::from_elf(&elf).unwrap(); + suite.push(Test { + name, + text: guest.text, + program, + }); + } + } + suite +} + +/// What ACT4's failure handler knows of a failing check. It reads the check's +/// instructions back from the text, which leanVM cannot read, so the handler traps +/// there, loading from `x5 - 6`, with `x5` the return address of the check's `jal`, +/// behind which the text holds the check's label and the address of its description, +/// and `x4` its register save area, where the registers are. +fn failed_check(test: &Test, machine: &Machine, trap: &Trap) -> Option { + let link = machine.regs[5]; + if !matches!(*trap, Trap::Unmapped { address, .. } if address == link.wrapping_sub(6)) { + return None; + } + let word = |pc: u64| test.text.get(pc.checked_sub(TEXT_BASE)? as usize / 4).copied(); + let ram = |address: u64| machine.ram().get(address.checked_sub(RAM_BASE)? as usize / 8).copied(); + // `ld expected, offset(signature)`, `beq expected, actual`, `jal x5, handler`. + let (load, beq) = (word(link.wrapping_sub(12))?, word(link.wrapping_sub(8))?); + if load & 0x707f != 0x3003 || beq & 0x707f != 0x63 { + return None; + } + let saved = |register: u32| ram(machine.regs[4].wrapping_add(8 * register as u64)); + let actual = (beq >> 20) & 31; + let signature = saved((load >> 15) & 31)?.wrapping_add((load as i32 >> 20) as u64); + let description = u64::from(word(link.wrapping_add(8))?) | u64::from(word(link.wrapping_add(12))?) << 32; + let description: Vec = (description..) + .map_while(|address| Some((ram(address)? >> (8 * (address % 8))) as u8).filter(|&byte| byte != 0)) + .collect(); + Some(format!( + "{}: x{actual} = {:#x}, Sail's is {:#x}", + String::from_utf8_lossy(&description), + saved(actual)?, + ram(signature)?, + )) +} + +/// What the interpreter says of a test, as a failure message if it is not a pass. +fn check_run(test: &Test) -> Result<(), String> { + let name = &test.name; + let mut machine = Machine::new(&test.program.rv, [0; 4], &[]); + match machine.run(CYCLE_CAP) { + Ok(PASS) => Ok(()), + Ok([1, called_from, ..]) => Err(format!("{name}: fails, halting from {called_from:#x}")), + Ok(output) => Err(format!("{name}: exits with {output:?}")), + Err(trap) => Err(match failed_check(test, &machine, &trap) { + Some(check) => format!("{name}: {check}"), + None => format!("{name}: {trap}"), + }), + } +} + +#[test] +#[ignore = "needs the ELF files of conformance/act4/generate.sh"] +fn act4_on_the_interpreter() { + let failures: Vec = suite().iter().filter_map(|test| check_run(test).err()).collect(); + assert!(failures.is_empty(), "{}", failures.join("\n")); +} + +/// Every test proven and checked by the Rust verifier. Python, far slower per proof, +/// checks the first test to use each table, which between them reach every table the +/// suite does, side by side. +#[test] +#[ignore = "needs the ELF files of conformance/act4/generate.sh"] +fn act4_proven() { + let mut covered = [false; N_TABLES]; + let mut python = Vec::new(); + for Test { name, program, .. } in suite() { + let (proof, output, stats) = prove(&program, [0; 4], &[], 1).unwrap_or_else(|trap| panic!("{name}: {trap}")); + assert_eq!(output, PASS, "{name}: the prover's output"); + let raw = verify_to_raw(&program, &[0; 4], &output, &proof).unwrap_or_else(|error| panic!("{name}: {error:?}")); + let used = stats.base_counts.map(|rows| rows > 0); + if used.iter().zip(&covered).any(|(&used, &seen)| used && !seen) { + covered = std::array::from_fn(|t| covered[t] || used[t]); + python.push(( + name.clone(), + PythonStatement::new(&name, &program, &[0; 4], &output), + raw, + )); + } + } + std::thread::scope(|scope| { + for (name, statement, raw) in &python { + std::thread::Builder::new() + .name(name.clone()) + .spawn_scoped(scope, move || statement.assert_accepts(raw)) + .unwrap(); + } + }); +} diff --git a/crates/lean_vm/tests/verifiers/constants.rs b/crates/lean_vm/tests/verifiers/constants.rs new file mode 100644 index 000000000..7266ce0b2 --- /dev/null +++ b/crates/lean_vm/tests/verifiers/constants.rs @@ -0,0 +1,135 @@ +//! The two verifiers agree on their constants. +//! +//! `python-verifier/verifier.py` writes out the same protocol a second time, which means +//! writing out its constants a second time: the region bases and caps, the clock's slots +//! and strides, the flock and WHIR parameters, and every class's block size and legal +//! flags. The end-to-end tests catch a Python circuit that computes the wrong thing, but +//! a constant that drifts changes what each side ACCEPTS, and only a statement they both +//! reject would show it. So the two lists are rendered the same way and diffed here. + +use lean_vm::rv::hash; +use lean_vm::tables::CLASSES; +use std::fmt::Write; + +/// What the Rust verifier's constants come to, in the format `protocol_constants()` +/// prints: sorted `name value` lines, a list being comma-separated. +fn rust_constants() -> String { + let mut lines: Vec = Vec::new(); + let mut scalar = |name: &str, value: u64| lines.push(format!("{name} {value}")); + + scalar("ADVICE_BASE", lean_vm::rv::ADVICE_BASE); + scalar("BAD_SLOT", lean_vm::tables::BAD_SLOT as u64); + scalar("BUS_BITS", lean_vm::leaf::N_TUPLE_BITS as u64); + scalar("CLOCK_STRIDE", lean_vm::tables::CLOCK_STRIDE as u64); + scalar("FLOCK_K_SKIP", flock::zerocheck::K_SKIP as u64); + scalar("FLOCK_MIN_LOG_SIZE", lean_vm::class_flock::MIN_CUBE_LOG as u64); + scalar("HASH_OUT_WORD", hash::OUT / 8); + scalar("HASH_STRIDE", lean_vm::tables::HASH.stride() as u64); + scalar("HASH_WORDS", hash::WORDS as u64); + scalar( + "INITIAL_FOLDING_FACTOR", + pcs::whir_config::INITIAL_FOLDING_FACTOR as u64, + ); + scalar("INPUT_WORDS", lean_vm::rv::INPUT_WORDS as u64); + scalar("LOG_PACKING", pcs::pack::LOG_PACKING as u64); + scalar("LOG_REGISTERS", lean_vm::rv::LOG_REGS as u64); + scalar("MAX_LOG_ADVICE", lean_vm::rv::MAX_LOG_ADVICE as u64); + scalar("MAX_LOG_RAM", lean_vm::rv::MAX_LOG_RAM as u64); + scalar("MAX_LOG_ROWS", lean_vm::cpu::MAX_LOG_ROWS as u64); + scalar("MAX_LOG_TEXT", lean_vm::rv::MAX_LOG_TEXT as u64); + scalar("MAX_STACKED_LOG", lean_vm::pcs::MAX_MU as u64); + scalar("MIN_STACKED_LOG", lean_vm::pcs::MIN_MU as u64); + scalar("NUM_FRAMEWORK_COLUMNS", lean_vm::cpu::Q_BASE as u64); + scalar("QUERY_GRINDING_BITS", pcs::whir_config::QUERY_GRINDING_BITS as u64); + scalar("RAM_BASE", lean_vm::rv::RAM_BASE); + scalar("RAM_SLOT", lean_vm::tables::RAM_SLOT as u64); + scalar("RANGE_LOG", lean_vm::tables::RANGE_LOG as u64); + scalar("RESIDUAL_MAX_LOG", pcs::whir_config::RESIDUAL_MAX_LOG as u64); + let rs_domain = pcs::whir_config::RS_DOMAIN_INITIAL_REDUCTION_FACTOR; + scalar("RS_DOMAIN_INITIAL_REDUCTION_FACTOR", rs_domain as u64); + let rs_domain_rest = pcs::whir_config::RS_DOMAIN_SUBSEQUENT_REDUCTION_FACTOR; + scalar("RS_DOMAIN_SUBSEQUENT_REDUCTION_FACTOR", rs_domain_rest as u64); + scalar("SINK", lean_vm::rv::SINK as u64); + scalar( + "SUBSEQUENT_FOLDING_FACTOR", + pcs::whir_config::SUBSEQUENT_FOLDING_FACTOR as u64, + ); + scalar("SYSCALL_REGISTER", lean_vm::rv::SYSCALL_REG as u64); + scalar("SYS_EXIT", lean_vm::rv::SYS_EXIT); + scalar("TEXT_BASE", lean_vm::rv::TEXT_BASE); + + let list = |values: &[u64]| values.iter().map(u64::to_string).collect::>().join(","); + lines.push(format!( + "OUTPUT_REGISTERS {}", + list(&lean_vm::rv::OUTPUT_REGS.map(u64::from)) + )); + lines.push(format!( + "REGISTER_SLOTS {}", + list(&lean_vm::tables::REG_SLOTS.map(u64::from)) + )); + + for (t, spec) in CLASSES.iter().enumerate() { + let circuit = lean_vm::class_flock::circuit(t); + let prefix = format!("TABLE.{}", spec.name.to_lowercase()); + let mut line = String::new(); + for (field, value) in [ + ("opcode", t as u64), + ("k_log", spec.k_log as u64), + ("const_pos", circuit.const_pos() as u64), + ("slot_bits", lean_vm::class_flock::stride_log(spec) as u64), + ("min_log_height", lean_vm::class_flock::n_blocks_log(spec, 1) as u64), + ("ports", spec.ports.len() as u64), + ("width", lean_vm::tables::tables()[t].n_committed_columns() as u64), + ] { + line.clear(); + write!(line, "{prefix}.{field} {value}").unwrap(); + lines.push(line.clone()); + } + lines.push(format!( + "{prefix}.slots {}", + list(&spec.slots().iter().map(|&s| s as u64).collect::>()) + )); + let mut flags = lean_vm::rv::legal_flags(spec.class).to_vec(); + flags.sort_unstable(); + lines.push(format!("{prefix}.legal_flags {}", list(&flags))); + } + lines.sort(); + lines.join("\n") +} + +#[test] +fn constants_match_the_python_verifier() { + let verifier = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("../../python-verifier/verifier.py"); + let dumped = std::process::Command::new("python3") + .arg(&verifier) + .arg("--constants") + .output() + .expect("run the Python verifier"); + assert!(dumped.status.success(), "{}", String::from_utf8_lossy(&dumped.stderr)); + let python = String::from_utf8(dumped.stdout).expect("utf-8").trim().to_string(); + let rust = rust_constants(); + + let parse = |dump: &str| -> std::collections::BTreeMap { + dump.lines() + .filter_map(|line| line.split_once(' ')) + .map(|(name, value)| (name.to_string(), value.to_string())) + .collect() + }; + let (theirs, ours) = (parse(&python), parse(&rust)); + let names: std::collections::BTreeSet<&String> = theirs.keys().chain(ours.keys()).collect(); + let differences: Vec = names + .into_iter() + .filter(|name| theirs.get(*name) != ours.get(*name)) + .map(|name| { + let missing = "(missing)".to_string(); + let (them, us) = (theirs.get(name).unwrap_or(&missing), ours.get(name).unwrap_or(&missing)); + format!(" {name}: python {them}, rust {us}") + }) + .collect(); + assert!( + differences.is_empty(), + "the two verifiers disagree on {} constant(s):\n{}", + differences.len(), + differences.join("\n") + ); +} diff --git a/crates/lean_vm/tests/verifiers/corrupted.rs b/crates/lean_vm/tests/verifiers/corrupted.rs new file mode 100644 index 000000000..df135aee4 --- /dev/null +++ b/crates/lean_vm/tests/verifiers/corrupted.rs @@ -0,0 +1,86 @@ +//! A corrupted proof is refused, and refused the way a verifier must refuse: with an +//! error, never with a panic and never with acceptance. Everything here is the +//! prover's to choose, so every path the verifier takes through it has to end in +//! [`CpuError`], not in an index out of bounds. + +use lean_vm::cpu::{Proof, prove, verify}; + +struct Rng(u64); + +impl Rng { + fn next(&mut self) -> u64 { + self.0 ^= self.0 << 13; + self.0 ^= self.0 >> 7; + self.0 ^= self.0 << 17; + self.0 + } + fn below(&mut self, n: usize) -> usize { + (self.next() % n as u64) as usize + } +} + +/// One corruption of `proof`, chosen by `round`: a scalar's bit, a truncated stream, +/// a leaf word, a sibling digest, or a missing Merkle hint. +fn corrupt(proof: &Proof, round: usize, rng: &mut Rng) -> Proof { + let mut forged = proof.clone(); + match round % 5 { + 0 => { + let scalar = &mut forged.stream[rng.below(proof.stream.len())]; + let bit = 1u64 << (rng.next() % 64); + match rng.next() % 3 { + 0 => scalar.c0 ^= bit, + 1 => scalar.c1 ^= bit, + _ => scalar.c2 ^= bit, + } + } + // At least one scalar short, so the stream really is cut. + 1 => forged.stream.truncate(rng.below(proof.stream.len())), + 2 => { + let paths = &mut forged.merkle[rng.below(proof.merkle.len())]; + let row = rng.below(paths.leaf_data.len()); + let word = rng.below(paths.leaf_data[row].len()); + paths.leaf_data[row][word].0 ^= 1 << (rng.next() % 64); + } + 3 => { + let paths = &mut forged.merkle[rng.below(proof.merkle.len())]; + let hash = rng.below(paths.sibling_hashes.len()); + paths.sibling_hashes[hash][rng.below(32)] ^= 1; + } + _ => { + forged.merkle.remove(rng.below(proof.merkle.len())); + } + } + forged +} + +#[test] +fn a_corrupted_proof_is_rejected_and_never_panics() { + let (program, expected) = super::programs::fibonacci(); + let input = [0; 4]; + let (proof, output, _) = prove(&program, input, &[], 1).expect("the run halts"); + assert_eq!(output, expected); + verify(&program, &input, &output, &proof).expect("the honest proof verifies"); + assert!( + proof + .merkle + .iter() + .all(|p| !p.leaf_data.is_empty() && !p.sibling_hashes.is_empty()), + "the corruptions below index into every opening" + ); + + let mut rng = Rng(0x5eed_1234_5678_9abc); + for round in 0..250 { + let forged = corrupt(&proof, round, &mut rng); + if forged == proof { + continue; // the one no-op a random truncation can draw + } + let verified = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + verify(&program, &input, &output, &forged) + })); + match verified { + Ok(Ok(())) => panic!("round {round}: a corrupted proof was accepted"), + Ok(Err(_)) => {} + Err(_) => panic!("round {round}: the verifier panicked instead of rejecting"), + } + } +} diff --git a/crates/lean_vm/tests/verifiers/guests.rs b/crates/lean_vm/tests/verifiers/guests.rs new file mode 100644 index 000000000..012b8eec7 --- /dev/null +++ b/crates/lean_vm/tests/verifiers/guests.rs @@ -0,0 +1,195 @@ +//! Rust guests (`guests/`), from their ELF files: compiled by `rustc` for +//! `riscv64im-unknown-none-elf`, loaded, run, proven, and checked by both verifiers. +//! The files are built by `guests/build.sh` and checked in, the target needing a +//! nightly toolchain. + +use super::python_verifier::PythonStatement; +use lean_vm::cpu::{Program, prove, verify, verify_to_raw}; +use lean_vm::rv::{self, Guest, Machine}; + +fn proves_and_verifies(tag: &str, elf: &[u8], input: [u64; 4], expected: [u64; 4]) { + proves_and_verifies_with(tag, elf, input, &[], expected); +} + +fn proves_and_verifies_with(tag: &str, elf: &[u8], input: [u64; 4], advice: &[u64], expected: [u64; 4]) { + let program = Program::from_elf(elf).expect("a guest"); + let ran = Machine::new(&program.rv, input, advice) + .run(1 << 24) + .expect("the run halts"); + assert_eq!(ran, expected, "{tag}: the interpreter"); + + let (proof, output, stats) = prove(&program, input, advice, 1).expect("the run halts"); + assert_eq!(output, expected); + let raw = verify_to_raw(&program, &input, &output, &proof).expect("honest proof verifies"); + PythonStatement::new(tag, &program, &input, &output).assert_accepts(&raw); + let mut wrong = output; + wrong[3] ^= 1; + assert!(verify(&program, &input, &wrong, &proof).is_err()); + println!("{tag}: {} instructions, {}", program.rv.entries.len(), stats.details()); +} + +#[test] +fn fibonacci_guest() { + let (mut a, mut b) = (0u64, 1u64); + for _ in 0..5000 { + (a, b) = (b, a.wrapping_add(b)); + } + proves_and_verifies( + "fibonacci", + include_bytes!("../../../../guests/elf/fibonacci.elf"), + [5000, 0, 0, 0], + [a, 0, 0, 0], + ); +} + +/// BLAKE2s-256 in plain Rust on the VM, against the prover's own BLAKE2s: shifts, +/// rotations, 32-bit arithmetic, byte loads and stores, a message that is not a whole +/// number of blocks. +#[test] +fn blake2s_guest() { + let length = 150u64; + let message: Vec = (0..length).map(|i| (i % 251) as u8).collect(); + let digest = primitives::hash::Hasher::new().update(&message).finalize(); + let expected = std::array::from_fn(|i| u64::from_le_bytes(digest[8 * i..8 * i + 8].try_into().unwrap())); + proves_and_verifies( + "blake2s", + include_bytes!("../../../../guests/elf/blake2s.elf"), + [length, 0, 0, 0], + expected, + ); +} + +/// The same digest through the `blake2s` instruction, from the runtime's hasher: the +/// precompile as a guest reaches it, on a message of several blocks. +#[test] +fn hash_guest() { + let length = 1000u64; + let message: Vec = (0..length).map(|i| (i % 251) as u8).collect(); + let digest = primitives::hash::Hasher::new().update(&message).finalize(); + let expected = std::array::from_fn(|i| u64::from_le_bytes(digest[8 * i..8 * i + 8].try_into().unwrap())); + proves_and_verifies( + "hash", + include_bytes!("../../../../guests/elf/hash.elf"), + [length, 0, 0, 0], + expected, + ); +} + +/// The lengths a block-based hash gets wrong: nothing, one byte, and the exact +/// multiples of the block either side. Proving each would cost minutes, so these run on +/// the interpreter, which is what the proofs above are checked against anyway. +#[test] +fn the_hash_guests_agree_with_the_host_at_every_block_boundary() { + let guests = [ + ( + "blake2s", + include_bytes!("../../../../guests/elf/blake2s.elf").as_slice(), + ), + ("hash", include_bytes!("../../../../guests/elf/hash.elf").as_slice()), + ]; + for (name, elf) in guests { + let program = Program::from_elf(elf).expect("a guest"); + for length in [0u64, 1, 55, 63, 64, 65, 127, 128, 129, 256] { + let message: Vec = (0..length).map(|i| (i % 251) as u8).collect(); + let digest = primitives::hash::hash(&message); + let expected: [u64; 4] = + std::array::from_fn(|i| u64::from_le_bytes(digest[8 * i..8 * i + 8].try_into().unwrap())); + let ran = Machine::new(&program.rv, [length, 0, 0, 0], &[]) + .run(1 << 24) + .unwrap_or_else(|trap| panic!("{name} on {length} bytes: {trap}")); + assert_eq!(ran, expected, "{name} on {length} bytes"); + } + } +} + +/// A digest of a message only the prover has: the advice region, read through the +/// runtime, and the precompile on it. +#[test] +fn preimage_guest() { + let message: Vec = (0..300u32).map(|i| (i * 7 + 3) as u8).collect(); + let mut advice = vec![message.len() as u64]; + advice.extend(message.chunks(8).map(|chunk| { + let mut word = [0u8; 8]; + word[..chunk.len()].copy_from_slice(chunk); + u64::from_le_bytes(word) + })); + let digest = primitives::hash::hash(&message); + let expected = std::array::from_fn(|i| u64::from_le_bytes(digest[8 * i..8 * i + 8].try_into().unwrap())); + proves_and_verifies_with( + "preimage", + include_bytes!("../../../../guests/elf/preimage.elf"), + [0; 4], + &advice, + expected, + ); +} + +/// Multiplications and divisions as `rustc` emits them, 128-bit arithmetic included. +#[test] +fn numbers_guest() { + let (base, exponent, modulus) = (0x1234_5678_9abc_def1u64, 65_537u64, 0xffff_ffff_0000_0001u64); + let pow_mod = { + let (mut result, mut b, mut e) = (1u128, base as u128 % modulus as u128, exponent); + while e > 0 { + if e & 1 == 1 { + result = result * b % modulus as u128; + } + b = b * b % modulus as u128; + e >>= 1; + } + result as u64 + }; + let gcd = { + let (mut a, mut b) = (base, modulus); + while b != 0 { + (a, b) = (b, a % b); + } + a + }; + let signed = (base as i64).wrapping_neg() / (exponent as i64 | 1); + let mixed = ((base as i32) / (exponent as i32 | 1)) as i64 % 1000; + proves_and_verifies( + "numbers", + include_bytes!("../../../../guests/elf/numbers.elf"), + [base, exponent, modulus, 0], + [pow_mod, gcd, signed as u64, mixed as u64], + ); +} + +/// What is not a guest is refused by name, not run. +#[test] +fn malformed_elf_files_are_refused() { + let elf = include_bytes!("../../../../guests/elf/fibonacci.elf"); + assert!(Guest::from_elf(elf).is_ok()); + assert!(Guest::from_elf(&elf[..40]).is_err(), "a truncated header"); + for (at, value, what) in [ + (4usize, 1u8, "32-bit"), + (16, 3, "a PIE"), + (18, 62, "x86-64"), + (48, 1, "compressed"), + ] { + let mut bad = elf.to_vec(); + bad[at] = value; + assert!(Guest::from_elf(&bad).is_err(), "{what}"); + } + + // A file is read into buffers the size of what it says it holds, so what it says has + // to be bounded by what it carries: a segment at the top of a region would otherwise + // allocate the whole region, gigabytes from a few kilobytes. The two headers a + // malformed file can wrap are checked too, since a wrapped address reads a field the + // header never pointed at. + let word_at = |file: &[u8], at: usize| u64::from_le_bytes(file[at..at + 8].try_into().unwrap()); + let phoff = word_at(elf, 32) as usize; + for (field, value, what) in [ + (16, rv::TEXT_BASE + (4 << 20), "a segment past what the file carries"), + (16, rv::TEXT_BASE - 4, "a segment below the text"), + (16, rv::TEXT_BASE + 1, "a segment that is no instruction address"), + ] { + let mut bad = elf.to_vec(); + bad[phoff + field..phoff + field + 8].copy_from_slice(&value.to_le_bytes()); + assert!(Guest::from_elf(&bad).is_err(), "{what}"); + } + let mut wrapped = elf.to_vec(); + wrapped[40..48].copy_from_slice(&u64::MAX.to_le_bytes()); // the section headers' offset + assert!(Guest::from_elf(&wrapped).is_err(), "a wrapped section-header address"); +} diff --git a/crates/lean_vm/tests/verifiers/main.rs b/crates/lean_vm/tests/verifiers/main.rs index effc3737a..a3e3592d2 100644 --- a/crates/lean_vm/tests/verifiers/main.rs +++ b/crates/lean_vm/tests/verifiers/main.rs @@ -1,9 +1,10 @@ -//! The other two verifiers of this protocol, pinned against `cpu::verify`. -//! -//! An integration test rather than a `src` module: pinning the Python verifier -//! needs a proof of a real program, so it needs the zkDSL compiler, and -//! `lean_compiler` depends on this crate. Cargo allows that cycle through -//! dev-dependencies, but only for a target that links the ordinary library. +//! The Python verifier of this protocol, pinned against `cpu::verify` on +//! hand-assembled programs and on Rust guests. +mod act4; +mod constants; +mod corrupted; +mod guests; +mod programs; mod python_verifier; mod whir_query_table; diff --git a/crates/lean_vm/tests/verifiers/programs.rs b/crates/lean_vm/tests/verifiers/programs.rs new file mode 100644 index 000000000..bc3c2f849 --- /dev/null +++ b/crates/lean_vm/tests/verifiers/programs.rs @@ -0,0 +1,352 @@ +//! RISC-V programs, proven and checked by both verifiers. + +use super::python_verifier::PythonStatement; +use lean_vm::cpu::{Program, prove, verify, verify_to_raw}; +use lean_vm::rv::asm::*; +use lean_vm::rv::{ADVICE_BASE, RAM_BASE, TEXT_BASE}; + +const STEPS: u64 = 1000; + +/// Fibonacci mod 2^64, iteratively, and the output it proves: `a0 = F(STEPS)`. +pub fn fibonacci() -> (Program, [u64; 4]) { + let text = Asm::new() + .li(A0, 0) + .li(A1, 1) + .li(T0, STEPS) + .label("loop") + .r("add", A2, A0, A1) + .i("addi", A0, A1, 0) + .i("addi", A1, A2, 0) + .i("addi", T0, T0, -1) + .branch("bne", T0, ZERO, "loop") + .li(A1, 0) + .li(A2, 0) + .exit() + .finish(); + let (mut a, mut b) = (0u64, 1u64); + for _ in 0..STEPS { + (a, b) = (b, a.wrapping_add(b)); + } + (Program::new(&text, TEXT_BASE, vec![], 2, 0), [a, 0, 0, 0]) +} + +fn proves_and_verifies(tag: &str, program: &Program, input: [u64; 4], expected: [u64; 4]) { + proves_and_verifies_with(tag, program, input, &[], expected); +} + +fn proves_and_verifies_with(tag: &str, program: &Program, input: [u64; 4], advice: &[u64], expected: [u64; 4]) { + let (proof, output, _) = prove(program, input, advice, 1).expect("the run halts"); + assert_eq!(output, expected); + let raw = verify_to_raw(program, &input, &output, &proof).expect("honest proof verifies"); + PythonStatement::new(tag, program, &input, &output).assert_accepts(&raw); + + // The proof is about this input and this output. + for (wrong_input, wrong_output) in [(1, 0), (0, 1)] { + let (mut input, mut output) = (input, output); + input[0] ^= wrong_input; + output[0] ^= wrong_output; + assert!(verify(program, &input, &output, &proof).is_err()); + } +} + +#[test] +fn fibonacci_proves_and_verifies() { + let (program, output) = fibonacci(); + proves_and_verifies("fibonacci", &program, [0; 4], output); +} + +/// Every instruction of the `ALU` class at least once: the arithmetic and its 32-bit +/// forms, the comparisons, the logic, the constants, all six branches taken and not, +/// and a call and return. +#[test] +fn alu_instructions_prove_and_verify() { + let mut a = Asm::new(); + // Constants `li` builds without a shift, which is another class's. + a.li(S0, 0xffff_ffff_8000_0001) + .li(S1, 0x7fff_ffff) + .r("add", A0, S0, S1) + .r("sub", A1, S0, S1) + .r("addw", A2, S1, S1) + .r("subw", A3, S0, S1) + .i("addiw", A4, S1, 1) + .r("slt", T0, S0, S1) + .r("sltu", T1, S0, S1) + .i("slti", T2, S0, -1) + .i("sltiu", A5, S1, -1) + .r("and", A6, S0, S1) + .r("or", A6, A6, T0) + .r("xor", A6, A6, T1) + .i("andi", T0, S1, 0x555) + .i("ori", T0, T0, -0x800) + .i("xori", T0, T0, 0x2aa) + .r("add", A0, A0, T0) + .r("add", A0, A0, T2) + .r("add", A0, A0, A5) + .lui(T0, 0xfffff) + .auipc(T1, 0x12345) + .r("add", A1, A1, T0) + .r("add", A1, A1, T1); + // Each branch twice, operands swapped, so that one of the two is taken. A branch + // taken skips an increment of a4. + for (i, op) in ["beq", "bne", "blt", "bge", "bltu", "bgeu"].into_iter().enumerate() { + for (j, (x, y)) in [(S0, S1), (S1, S0)].into_iter().enumerate() { + let label: &'static str = Box::leak(format!("skip{i}{j}").into_boxed_str()); + a.branch(op, x, y, label).i("addi", A4, A4, 1).label(label); + } + } + a.jal(RA, "double") + .jal(RA, "double") + .li(A3, 0) + .exit() + .label("double") + .r("add", A2, A2, A2) + .r("add", A4, A4, A4) + .jalr(ZERO, RA, 0); + let program = Program::new(&a.finish(), TEXT_BASE, vec![], 2, 0); + let expected = lean_vm::rv::Machine::new(&program.rv, [0; 4], &[]) + .run(1 << 20) + .expect("the run halts"); + assert_ne!(expected, [0; 4]); + proves_and_verifies("alu", &program, [0; 4], expected); +} + +/// Every load and store, through a stack frame and over the program's image: a +/// bubble sort of eight words in place, then a checksum of the sorted bytes read back +/// at every width, signed and not, and the public input folded in. +#[test] +fn loads_and_stores_prove_and_verify() { + const LOG_RAM: usize = 6; + const DATA: u64 = RAM_BASE + 8 * 4; + let image = vec![5u64, 3, 0xffff_ffff_ffff_fff9, 1, 8, 0x8877_6655_4433_2211, 7, 4]; + let mut a = Asm::new(); + a.li(SP, RAM_BASE + (8 << LOG_RAM)) + .li(A0, DATA) + .jal(RA, "sort") + .li(T0, DATA) + .load("ld", A0, 56, T0) + .load("lw", T1, 56, T0) + .r("add", A0, A0, T1) + .load("lwu", T1, 60, T0) + .r("add", A0, A0, T1) + .load("lh", T1, 62, T0) + .r("add", A0, A0, T1) + .load("lhu", T1, 58, T0) + .r("add", A0, A0, T1) + .load("lb", T1, 63, T0) + .r("add", A0, A0, T1) + .load("lbu", T1, 57, T0) + .r("add", A0, A0, T1) + // Narrow stores into the first sorted word, then the public input. + .store("sb", T1, 1, T0) + .store("sh", T1, 2, T0) + .store("sw", T1, 4, T0) + .load("ld", A1, 0, T0) + .li(T0, RAM_BASE) + .load("ld", A2, 0, T0) + .load("ld", A3, 24, T0) + .exit() + .label("sort") + .i("addi", SP, SP, -16) + .store("sd", RA, 8, SP) + .li(T2, 7) + .label("outer") + .i("addi", T0, A0, 0) + .i("addi", T1, T2, 0) + .label("inner") + .load("ld", A2, 0, T0) + .load("ld", A3, 8, T0) + .branch("bgeu", A3, A2, "ordered") + .store("sd", A3, 0, T0) + .store("sd", A2, 8, T0) + .label("ordered") + .i("addi", T0, T0, 8) + .i("addi", T1, T1, -1) + .branch("bne", T1, ZERO, "inner") + .i("addi", T2, T2, -1) + .branch("bne", T2, ZERO, "outer") + .load("ld", RA, 8, SP) + .i("addi", SP, SP, 16) + .jalr(ZERO, RA, 0); + let program = Program::new(&a.finish(), TEXT_BASE, image, LOG_RAM, 0); + let input = [0x1111, 0x2222, 0x3333, 0x4444]; + let expected = lean_vm::rv::Machine::new(&program.rv, input, &[]) + .run(1 << 20) + .expect("the run halts"); + assert_eq!(expected[2..], [0x1111, 0x4444], "the public input is RAM's first words"); + proves_and_verifies("memory", &program, input, expected); +} + +/// Every shift and every multiplication, registers and immediates, 64-bit and 32-bit +/// forms, folded into the output. +#[test] +fn shifts_and_multiplications_prove_and_verify() { + let mut a = Asm::new(); + a.li(S0, 0x8765_4321_fedc_ba98).li(S1, 0xffff_ffff_0000_0025); + for (i, op) in [ + "sll", "srl", "sra", "sllw", "srlw", "sraw", "mul", "mulh", "mulhsu", "mulhu", "mulw", + ] + .into_iter() + .enumerate() + { + a.r(op, T0, S0, S1) + .r("xor", A0, A0, T0) + .i("addi", A1, A1, i as i32 + 1) + .r("add", A1, A1, T0); + } + for (op, amount) in [ + ("slli", 63), + ("srli", 1), + ("srai", 40), + ("slliw", 31), + ("srliw", 0), + ("sraiw", 17), + ] { + a.i(op, T0, S0, amount).r("xor", A2, A2, T0).r("sub", A3, A3, T0); + } + let program = Program::new(&a.exit().finish(), TEXT_BASE, vec![], 2, 0); + let expected = lean_vm::rv::Machine::new(&program.rv, [0; 4], &[]) + .run(1 << 20) + .expect("the run halts"); + assert!(expected.iter().all(|&word| word != 0)); + proves_and_verifies("shift-mul", &program, [0; 4], expected); +} + +/// Every division and remainder, 64-bit and 32-bit, on operands of both signs, by zero, +/// and the one that overflows. +#[test] +fn divisions_prove_and_verify() { + let mut a = Asm::new(); + let operands = [ + (0x8765_4321_fedc_ba98u64, 0xffff_ffff_ffff_ff85u64), + (1_000_000_007, 13), + (5, 0), + (i64::MIN as u64, u64::MAX), + (0xffff_ffff_8000_0000, 0xffff_ffff_ffff_ffff), + ]; + for (n, d) in operands { + a.li(S0, n).li(S1, d); + for op in ["div", "divu", "rem", "remu", "divw", "divuw", "remw", "remuw"] { + a.r(op, T0, S0, S1) + .r("xor", A0, A0, T0) + .r("add", A1, A1, T0) + .r("sub", A2, A2, A1); + } + } + let program = Program::new(&a.exit().finish(), TEXT_BASE, vec![], 2, 0); + let expected = lean_vm::rv::Machine::new(&program.rv, [0; 4], &[]) + .run(1 << 20) + .expect("the run halts"); + proves_and_verifies("div", &program, [0; 4], expected); +} + +/// BLAKE2s of 100 bytes through the precompile: two compressions of the block at +/// `RAM_BASE + 128`, the chaining value copied forward between them and the second +/// message block loaded from the image, checked against the host's hash. +#[test] +fn blake2s_precompile_proves_and_verifies() { + use lean_vm::rv::hash::{H, M, OUT}; + const BLOCK: u64 = RAM_BASE + 128; + let data: Vec = (0..100u32).map(|i| (i * 37 + 11) as u8).collect(); + let words = |bytes: &[u8]| -> Vec { + let mut padded = bytes.to_vec(); + padded.resize(64, 0); + padded + .chunks(8) + .map(|w| u64::from_le_bytes(w.try_into().unwrap())) + .collect() + }; + let iv: Vec = primitives::hash::PARAM_IV + .chunks(2) + .map(|w| w[0] as u64 | (w[1] as u64) << 32) + .collect(); + // The image: the block (its chaining value seeded, its message the first 64 bytes), + // then the second message block. + let mut image = vec![0u64; ((BLOCK - RAM_BASE) / 8 - 4) as usize]; + image.extend(&iv); + image.extend([0; 4]); + image.extend(words(&data[..64])); + image.extend(words(&data[64..])); + let second = BLOCK + 128; + + let mut a = Asm::new(); + a.li(S0, BLOCK).li(S1, 64).blake2s(S0, S1, false); + for k in 0..4 { + a.load("ld", T0, (OUT + 8 * k) as i32, S0) + .store("sd", T0, (H + 8 * k) as i32, S0); + } + a.li(T1, second); + for k in 0..8 { + a.load("ld", T0, 8 * k, T1) + .store("sd", T0, (M + 8 * k as u64) as i32, S0); + } + a.li(S1, data.len() as u64).blake2s(S0, S1, true); + for (i, reg) in [A0, A1, A2, A3].into_iter().enumerate() { + a.load("ld", reg, (OUT + 8 * i as u64) as i32, S0); + } + let program = Program::new(&a.exit().finish(), TEXT_BASE, image, 7, 0); + let expected: [u64; 4] = words(&primitives::hash::hash(&data))[..4].try_into().unwrap(); + proves_and_verifies("blake2s", &program, [0; 4], expected); + + // A block pointer that is no word address traps, like a misaligned load. + let text = Asm::new().li(S0, BLOCK + 4).blake2s(S0, ZERO, true).exit().finish(); + let program = Program::new(&text, TEXT_BASE, vec![], 7, 0); + assert_eq!( + prove(&program, [0; 4], &[], 1).err(), + Some(lean_vm::rv::Trap::Misaligned { + pc: TEXT_BASE + 8, + address: BLOCK + 4 + }) + ); +} + +/// The advice region: words the prover supplies, read and written like RAM, which the +/// statement says nothing about, so one program proves a different output per advice. +#[test] +fn advice_proves_and_verifies() { + const LOG_ADVICE: usize = 3; + let mut a = Asm::new(); + a.li(T0, ADVICE_BASE) + .load("ld", A0, 0, T0) + .load("ld", T1, 8, T0) + .r("add", A0, A0, T1) + .load("lw", A1, 20, T0) + .store("sd", A0, 56, T0) + .load("ld", A2, 56, T0) + .li(A3, 0) + .exit(); + let program = Program::new(&a.finish(), TEXT_BASE, vec![], 2, LOG_ADVICE); + for advice in [ + vec![3, 4, 0xdead_beef_0000_0005u64], + vec![u64::MAX, 1, 0xffff_ffff_ffff_ffff, 9, 9, 9, 9, 9], + ] { + let expected = [ + advice[0].wrapping_add(advice[1]), + (advice[2] >> 32) as i32 as i64 as u64, + advice[0].wrapping_add(advice[1]), + 0, + ]; + proves_and_verifies_with("advice", &program, [0; 4], &advice, expected); + } + // Past the region is nowhere, like past RAM. + let text = Asm::new() + .li(T0, ADVICE_BASE + (8 << LOG_ADVICE)) + .load("ld", A0, 0, T0) + .exit() + .finish(); + let program = Program::new(&text, TEXT_BASE, vec![], 2, LOG_ADVICE); + assert!(matches!( + prove(&program, [0; 4], &[], 1).err(), + Some(lean_vm::rv::Trap::Unmapped { .. }) + )); +} + +/// A run that traps has no proof, and says why. +#[test] +fn a_trap_is_reported() { + let text = Asm::new().word(0x0010_0073).exit().finish(); + let program = Program::new(&text, TEXT_BASE, vec![], 2, 0); + assert_eq!( + prove(&program, [0; 4], &[], 1).err(), + Some(lean_vm::rv::Trap::Illegal { pc: TEXT_BASE }) + ); +} diff --git a/crates/lean_vm/tests/verifiers/python_verifier.rs b/crates/lean_vm/tests/verifiers/python_verifier.rs index a5581422f..e0b785a4f 100644 --- a/crates/lean_vm/tests/verifiers/python_verifier.rs +++ b/crates/lean_vm/tests/verifiers/python_verifier.rs @@ -1,175 +1,189 @@ //! Pins `python-verifier/verifier.py` against `lean_vm::cpu::verify`: the same -//! protocol is written out in Rust, in Python, and in zkDSL, so any protocol -//! change must land in all three, and this is what catches the Python one -//! drifting. +//! protocol is written out in Rust and in Python, so any protocol change must land +//! in both, and this is what catches the Python one drifting. use fiat_shamir::transcript::RawProof; -use lean_compiler::{compile, parse_with_replacements}; -use lean_vm::cpu::{prove, verify}; -use primitives::field::{F64, F192, g_pow}; -use std::collections::BTreeMap; -use std::path::Path; +use lean_vm::cpu::{prove, verify, verify_to_raw}; +use std::path::{Path, PathBuf}; use std::process::{Command, Output}; use std::time::Instant; -const SOURCE: &str = r#" -from snark_lib import * - -LOOP_STEPS = LOOP_STEPS_PLACEHOLDER - -def mix(value, tag): - # A JUMP condition is K-valued (a g-power here), never an arbitrary word. - if tag == 0: - return value - return value * GEN + value - -def main(): - seed = [5, 7] - digest = StackBuf(2) - blake2s(seed, seed, digest) - - chain = HeapBuf(LOOP_STEPS + 1) - chain[1] = digest[0] - for index in mul_range(1, GEN ** LOOP_STEPS): - chain[index * GEN] = mix(chain[index] + index, index) + index - - public = GEN ** 0 - public[1] = chain[GEN ** LOOP_STEPS] - public[GEN] = mix(digest[1], GEN ** 0) - return -"#; - -const LOOP_STEPS: usize = 16_384; +/// One statement laid out the way the Python verifier takes it: the bytecode +/// multilinear, then what else is public (where the run starts, RAM's size and first +/// words, the input, the output), not a structured program. +pub struct PythonStatement { + directory: PathBuf, + bytecode: PathBuf, + public: PathBuf, +} -/// Write `raw` as the two files Python reads and run the verifier on it: the -/// scalar stream as 24-byte little-endian elements, and every opening's leaf -/// words followed by its sibling digests. Neither file carries a length, the -/// reader deriving every leaf width and tree height from the protocol it is -/// replaying. -fn python_verify(directory: &Path, bytecode: &Path, public_input: &Path, raw: &RawProof) -> Output { - let mut stream = Vec::new(); - for scalar in &raw.stream { - for limb in [scalar.c0, scalar.c1, scalar.c2] { - stream.extend(limb.to_le_bytes()); - } +impl PythonStatement { + pub fn new(tag: &str, program: &lean_vm::cpu::Program, input: &[u64; 4], output: &[u64; 4]) -> Self { + // One directory per statement, not per tag: the tests share a process, so two of + // them naming the same tag would write each other's files and check the wrong + // proof, which python would ACCEPT, silently proving nothing. + static NEXT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + let unique = NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + let directory = + std::env::temp_dir().join(format!("leanvm-python-verifier-{tag}-{}-{unique}", std::process::id())); + std::fs::create_dir_all(&directory).expect("create test directory"); + let statement = Self { + bytecode: directory.join("bytecode.bin"), + public: directory.join("public.bin"), + directory, + }; + let rv = &program.rv; + let table: Vec = lean_vm::cpu::layout::bytecode_table(rv) + .iter() + .flat_map(|w| w.0.to_le_bytes()) + .collect(); + std::fs::write(&statement.bytecode, &table).expect("write bytecode"); + let public: Vec = [ + rv.entry_pc, + rv.log_ram as u64, + rv.log_advice as u64, + rv.image.len() as u64, + ] + .iter() + .chain(&rv.image) + .chain(input) + .chain(output) + .flat_map(|w| w.to_le_bytes()) + .collect(); + std::fs::write(&statement.public, &public).expect("write the public words"); + statement } - let mut openings = Vec::new(); - for opening in &raw.merkle { - for word in &opening.leaf_data { - openings.extend(word.0.to_le_bytes()); + + /// Write `raw` as the two files Python reads and run the verifier on it: the + /// scalar stream as 24-byte little-endian elements, and every opening's leaf + /// words followed by its sibling digests. Neither file carries a length, the + /// reader deriving every leaf width and tree height from the protocol it is + /// replaying. + pub fn verify(&self, raw: &RawProof) -> Output { + let mut stream = Vec::new(); + for scalar in &raw.stream { + for limb in [scalar.c0, scalar.c1, scalar.c2] { + stream.extend(limb.to_le_bytes()); + } } - for digest in &opening.path { - openings.extend(digest); + let mut openings = Vec::new(); + for opening in &raw.merkle { + for word in &opening.leaf_data { + openings.extend(word.0.to_le_bytes()); + } + for digest in &opening.path { + openings.extend(digest); + } } + let stream_path = self.directory.join("stream.bin"); + let openings_path = self.directory.join("merkle_openings.bin"); + std::fs::write(&stream_path, stream).expect("write scalar stream"); + std::fs::write(&openings_path, openings).expect("write Merkle openings"); + Command::new("python3") + .arg(Path::new(env!("CARGO_MANIFEST_DIR")).join("../../python-verifier/verifier.py")) + .arg(&self.bytecode) + .arg(&self.public) + .arg(stream_path) + .arg(openings_path) + .output() + .expect("run native Python verifier") } - let stream_path = directory.join("stream.bin"); - let openings_path = directory.join("merkle_openings.bin"); - std::fs::write(&stream_path, stream).expect("write scalar stream"); - std::fs::write(&openings_path, openings).expect("write Merkle openings"); - Command::new("python3") - .arg(Path::new(env!("CARGO_MANIFEST_DIR")).join("../../python-verifier/verifier.py")) - .arg(bytecode) - .arg(public_input) - .arg(stream_path) - .arg(openings_path) - .output() - .expect("run native Python verifier") -} -fn public_input() -> [F192; 2] { - use lean_vm::hash_flock::{FINAL_FLAG, IV, PINNED_T, compression, digest, metadata}; + /// Python refused, and refused the way it should: through its own error path, not + /// through a traceback, which exits nonzero just the same and would hide a crash. + pub fn assert_rejects(output: &Output, what: &str) { + let stderr = String::from_utf8_lossy(&output.stderr); + assert!(!output.status.success(), "Python accepted {what}"); + assert!( + stderr.starts_with("verification failed:"), + "Python crashed on {what} rather than rejecting it:\n{stderr}" + ); + } - let seed = [F64(5), F64::ZERO, F64(7), F64::ZERO]; - let metadata = metadata(PINNED_T, FINAL_FLAG, 0); - let digest = digest(&compression(seed, seed, IV, metadata)); - let digest = [ - F192::new(digest[0].0, digest[1].0, 0), - F192::new(digest[2].0, digest[3].0, 0), - ]; - let mut value = digest[0]; - let mut index = F192::ONE; - let generator = F192::from(g_pow(1)); - for _ in 0..LOOP_STEPS { - let candidate = value + index; - let product = candidate * generator; - value = (if product == F192::ZERO { - candidate - } else { - product + candidate - }) + index; - index *= generator; + pub fn assert_accepts(&self, raw: &RawProof) { + let output = self.verify(raw); + assert!( + output.status.success(), + "native Python verification failed:\n{}", + String::from_utf8_lossy(&output.stderr), + ); + assert_eq!(String::from_utf8_lossy(&output.stdout).trim(), "verification succeeded"); } - [value, digest[1] * generator + digest[1]] } +impl Drop for PythonStatement { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.directory); + } +} + +/// Both verifiers reject a proof whose announcement or commitment root is not a +/// canonical encoding, and agree on everything before that. #[test] fn test_python_verifier() { - let replacements = BTreeMap::from([("LOOP_STEPS_PLACEHOLDER".to_string(), LOOP_STEPS.to_string())]); - let ast = parse_with_replacements(SOURCE, &replacements).expect("parse zkDSL program"); - let program = compile(&ast); - let public_input = public_input(); - let (proof, stats) = prove(&program, public_input, 1).unwrap(); + let (program, _) = super::programs::fibonacci(); + let input = [0; 4]; + let (proof, output, stats) = prove(&program, input, &[], 1).expect("the run halts"); // Python reads the RAW proof: same protocol, each query carrying its own // full Merkle path instead of one octopus over the batch. A Rust verify // expands the wire form, so the pruning is written once. - let raw = verify(&program, &public_input, &proof) - .expect("honest proof verifies") - .raw; - - let directory = std::env::temp_dir().join(format!("leanvm-python-verifier-test-{}", std::process::id())); - std::fs::create_dir_all(&directory).expect("create test directory"); - let bytecode_path = directory.join("bytecode.bin"); - let public_input_path = directory.join("public_input.bin"); + let raw = verify_to_raw(&program, &input, &output, &proof).expect("honest proof verifies"); let encoded = bincode::serialize(&proof).expect("serialize proof"); - // The statement the verifier takes is the bytecode multilinear plus 256 bits - // of public input, not a structured program. - let table: Vec = lean_vm::cpu::layout::bytecode_table(&program.prog) - .iter() - .flat_map(|w| w.0.to_le_bytes()) - .collect(); - std::fs::write(&bytecode_path, &table).expect("write bytecode"); - let pi: Vec = public_input - .iter() - .flat_map(|v| [v.c0.to_le_bytes(), v.c1.to_le_bytes()].concat()) - .collect(); - std::fs::write(&public_input_path, &pi).expect("write public input"); - + let statement = PythonStatement::new("tamper", &program, &input, &output); let verification_started = Instant::now(); - let output = python_verify(&directory, &bytecode_path, &public_input_path, &raw); + statement.assert_accepts(&raw); let verification_time = verification_started.elapsed(); - assert!( - output.status.success(), - "native Python verification failed:\n{}", - String::from_utf8_lossy(&output.stderr), - ); - assert_eq!(String::from_utf8_lossy(&output.stdout).trim(), "verification succeeded",); let mut malformed_announcement = proof.clone(); malformed_announcement.stream[0].c1 = 1; - assert!(verify(&program, &public_input, &malformed_announcement).is_err()); + assert!(verify(&program, &input, &output, &malformed_announcement).is_err()); let mut raw_announcement = raw.clone(); raw_announcement.stream[0].c1 = 1; - let output = python_verify(&directory, &bytecode_path, &public_input_path, &raw_announcement); - assert!(!output.status.success(), "Python accepted a noncanonical announcement"); + PythonStatement::assert_rejects(&statement.verify(&raw_announcement), "a noncanonical announcement"); let mut malformed_root = proof.clone(); + // Past the announcement: the table heights, the rate, the final clock. let root_offset = lean_vm::tables::N_TABLES + 2; malformed_root.stream[root_offset].c2 = 1; - assert!(verify(&program, &public_input, &malformed_root).is_err()); + assert!(verify(&program, &input, &output, &malformed_root).is_err()); let mut raw_root = raw.clone(); raw_root.stream[root_offset].c2 = 1; - let output = python_verify(&directory, &bytecode_path, &public_input_path, &raw_root); + PythonStatement::assert_rejects(&statement.verify(&raw_root), "a noncanonical commitment root"); + + // A decoded table is RISC-V only if it says so: one whose first entry writes `x0` + // is refused before anything is verified. + let table = std::fs::read(&statement.bytecode).expect("read bytecode"); + let (ad_slot, entries) = (7, table.len() / 8 / 16); + let mut writes_x0 = table.clone(); + writes_x0[8 * ad_slot * entries..][..8].copy_from_slice(&0u64.to_le_bytes()); + std::fs::write(&statement.bytecode, writes_x0).expect("write bytecode"); + let python = statement.verify(&raw); + PythonStatement::assert_rejects(&python, "a table that writes x0"); assert!( - !output.status.success(), - "Python accepted a noncanonical commitment root" + String::from_utf8_lossy(&python.stderr).contains("misnames a register"), + "Python refused a table that writes x0 for the wrong reason" ); + std::fs::write(&statement.bytecode, table).expect("restore bytecode"); println!( - "zkDSL compiled to {} instructions; proved {} cycles in {} bytes; Python verified in {:.2?}", - program.prog.len(), + "{} instructions; proved {} cycles in {} bytes; Python verified in {:.2?}", + program.rv.entries.len(), stats.cycles, encoded.len(), verification_time, ); - std::fs::remove_dir_all(directory).expect("remove test directory"); +} + +/// The PCS rate changes WHIR's ladder: how many levels it folds through, how wide a leaf +/// is and how many queries each level takes. Every other cross-check runs at the fastest +/// rate, so the slowest one is checked here, where the two verifiers would otherwise +/// agree only by never being asked. +#[test] +fn the_python_verifier_follows_the_slowest_rate() { + let (program, _) = super::programs::fibonacci(); + let input = [0; 4]; + let rate = lean_vm::pcs::MAX_LOG_INV_RATE; + let (proof, output, _) = prove(&program, input, &[], rate).expect("the run halts"); + let raw = verify_to_raw(&program, &input, &output, &proof).expect("honest proof verifies"); + PythonStatement::new("rate", &program, &input, &output).assert_accepts(&raw); } diff --git a/crates/pcs/src/ring_switch.rs b/crates/pcs/src/ring_switch.rs index dd7518b26..78a648fe9 100644 --- a/crates/pcs/src/ring_switch.rs +++ b/crates/pcs/src/ring_switch.rs @@ -95,11 +95,10 @@ pub const COMPOSITION_SHIFTS: [usize; 6] = [32, 16, 8, 4, 2, 1]; /// `sum_w weights[w]·t_w = sum_j x^j·Phi(y_j)` for the row /// view `y` and the column view `t` of the same tensor-algebra element. That /// identity is what lets a verifier evaluate the batched claim from `Phi`'s -/// six challenges instead of the 192 weights; the recursion guest does exactly -/// that. Expanding the composition puts a distinct monomial at every Frobenius +/// six challenges instead of the 192 weights. Expanding the composition puts a distinct monomial at every Frobenius /// exponent `0..64`: writing `k = sum_p k_p·2^(5-p)` for the binary digits of /// `k`, the coefficient is `C_k = prod_{p : k_p = 1} f_p^(2^(k mod 2^(5-p)))`, -/// which is what the guest's coefficient table builds. Applying the composed +/// which is what `python-verifier`'s coefficient table builds. Applying the composed /// form directly costs only 63 squarings and six multiplications. /// /// ## Soundness @@ -552,10 +551,9 @@ mod tests { } /// The contract between the native opener and every verifier that batches - /// from `Phi`'s coefficients (the recursion guest, the Python reference - /// verifier): weighting the COLUMN view by `build_coordinate_weights` must + /// from `Phi`'s coefficients (the Python reference verifier): weighting the COLUMN view by `build_coordinate_weights` must /// equal applying `Phi` to the ROW view and combining with `x^j`. If this - /// drifts, the guest computes a different opening target than the prover. + /// drifts, that verifier computes a different opening target than the prover. #[test] fn column_weights_match_the_row_side_linearized_map() { let mut rng = Rng::new(0xF00D_BEEF_1234_5678); @@ -567,7 +565,7 @@ mod tests { let columns = transpose_s_hat(&s_hat_v); let lhs = inner_product_base_ext(&columns, &build_coordinate_weights(&challenges)); - // Row side: sum_j x^j * Phi(y_j), the guest's loop. + // Row side: sum_j x^j * Phi(y_j). let x = F192::new(2, 0, 0); let mut rhs = F192::ZERO; let mut x_pow = F192::ONE; diff --git a/crates/pcs/src/stack_open.rs b/crates/pcs/src/stack_open.rs index d231ad87a..5bb573cb1 100644 --- a/crates/pcs/src/stack_open.rs +++ b/crates/pcs/src/stack_open.rs @@ -11,7 +11,8 @@ //! `eq(low_point, .)` supported on `[offset, offset + 2^|low_point|)`; a //! `Strided` claim freezes the low `stride_log` in-block coords to `slot`'s //! bits, so its weight is nonzero only at `offset + slot + j * 2^stride_log`), -//! - **ring-switched claims** ([`RingSwitchOpen`]): bit-MLE evaluation claims +//! - **ring-switched claims** ([`RingSwitchOpen`], one per packed sub-block, each +//! circuit committing its own): bit-MLE evaluation claims //! on the packed sub-block `q_flock = stack[offset .. offset + 2^qflock_vars]`, //! reduced per claim by [`super::ring_switch::prove_prepare`] and the //! deferred finish path to an inner-product @@ -34,9 +35,10 @@ //! //! ## The combined weight //! -//! With `sel = offset >> qflock_vars` the selector coords of the q_flock slice, +//! With `sel = offset >> qflock_vars` the selector coords of a q_flock slice, //! the lifted weight at a full-stack point `x = (x_lo, x_hi)` (split at -//! `qflock_vars`, LSB-first) is +//! `qflock_vars`, LSB-first) is, for one such slice (several add up, each with +//! its own selector and its own run of `lambda` powers), //! //! ```text //! b(x) = eq(sel, x_hi) * sum_i lambda^i * MLE(rs_eq_ind_i)(x_lo) @@ -224,25 +226,25 @@ pub fn open_batch_mixed_whir_stacked( prover_data: &ProverData, config: &ProverConfig, point_claims: &[StackClaim], - ring: &RingSwitchOpen, + rings: &[RingSwitchOpen], ) { - let qflock_len = 1usize << ring.qflock_vars; - assert!( - ring.offset.is_multiple_of(qflock_len), - "q_flock offset must be 2^qflock_vars-aligned" - ); - assert!( - ring.offset + qflock_len <= stack.len(), - "q_flock slice must fit inside the stack" - ); + for ring in rings { + let qflock_len = 1usize << ring.qflock_vars; + assert!( + ring.offset.is_multiple_of(qflock_len), + "q_flock offset must be 2^qflock_vars-aligned" + ); + assert!( + ring.offset + qflock_len <= stack.len(), + "q_flock slice must fit inside the stack" + ); + } assert!( point_claims.iter().all(|c| claim_range(c).1 <= stack.len()), "every claim must live inside the committed lanes" ); - assert!( - !ring.claims.is_empty(), - "stacked PCS opening carries at least one ring-switched claim" - ); + let n_rs: usize = rings.iter().map(|ring| ring.claims.len()).sum(); + assert!(n_rs > 0, "stacked PCS opening carries at least one ring-switched claim"); // Optional phase timing, answering to the same env var as the WHIR // prover/commit tracing (one env lookup per open, no work when unset). let trace = std::env::var_os("WHIR_TRACE").is_some(); @@ -256,19 +258,21 @@ pub fn open_batch_mixed_whir_stacked( // 1. Ring-switch reduction: prepare every claim's s_hat_v (the caller bound // them upstream), then sample one shared linear map. - let qflock = &stack[ring.offset..ring.offset + qflock_len]; - let mut rs_states = Vec::with_capacity(ring.claims.len()); - for claim in &ring.claims { - assert_eq!( - claim.suffix_point.len(), - ring.qflock_vars, - "ring-switch suffix point must have qflock_vars coords" - ); - rs_states.push(ring_switch::prove_prepare( - qflock, - &claim.suffix_point, - claim.s_hat_v.as_deref(), - )); + let mut rs_states = Vec::with_capacity(n_rs); + for ring in rings { + let qflock = &stack[ring.offset..ring.offset + (1usize << ring.qflock_vars)]; + for claim in &ring.claims { + assert_eq!( + claim.suffix_point.len(), + ring.qflock_vars, + "ring-switch suffix point must have qflock_vars coords" + ); + rs_states.push(ring_switch::prove_prepare( + qflock, + &claim.suffix_point, + claim.s_hat_v.as_deref(), + )); + } } let map_challenges = ring_switch::sample_map_challenges(ps); let coordinate_weights = ring_switch::build_coordinate_weights(&map_challenges); @@ -276,8 +280,8 @@ pub fn open_batch_mixed_whir_stacked( // 2. The ONE batching challenge both families take disjoint power ranges of. Nothing is // observed first: every claim value reached the caller through a binding stream read, so // the challenge already depends on all of them (`lean_vm::pcs::open`). - let lambdas = powers(ps.sample(), ring.claims.len() + point_claims.len()); - let (lambdas_rs, lambdas_pd) = lambdas.split_at(ring.claims.len()); + let lambdas = powers(ps.sample(), n_rs + point_claims.len()); + let (lambdas_rs, lambdas_pd) = lambdas.split_at(n_rs); let rs_outputs: Vec<_> = rs_states .into_iter() @@ -301,7 +305,7 @@ pub fn open_batch_mixed_whir_stacked( // filled from the ring-switch outputs and the point claims, then feeds // round 0's message while it is still hot, so nothing re-reads the buffer. let lane_block = 1usize << (log_n - config.initial_k); - let (b_stack, message) = basis::build(stack, lane_block, point_claims, lambdas_pd, ring, &rs_outputs); + let (b_stack, message) = basis::build(stack, lane_block, point_claims, lambdas_pd, rings, &rs_outputs); mark("basis + initial message", &mut t); // 4. One WHIR over the full stack against the combined claim (the @@ -335,23 +339,24 @@ pub fn verify_opening_batch_mixed_whir_stacked( n_lanes: usize, root: &Hash, point_claims: &[StackClaim], - ring: &RingSwitchVerify<'_>, + rings: &[RingSwitchVerify<'_>], ) -> Result<(), VerifyError> { - let n_rs = ring.claims.len(); - let qflock_vars = ring.qflock_vars; - // Caller (statement) invariants: panic on misuse, like the extension-field layer. - assert!(qflock_vars <= log_n); - assert!( - ring.offset.is_multiple_of(1usize << qflock_vars), - "q_flock offset must be 2^qflock_vars-aligned" - ); + let n_rs: usize = rings.iter().map(|ring| ring.claims.len()).sum(); assert!(n_rs > 0, "stacked PCS opening carries at least one ring-switched claim"); - for claim in &ring.claims { - assert_eq!(claim.suffix_point.len(), qflock_vars); + // Caller (statement) invariants: panic on misuse, like the extension-field layer. + for ring in rings { + assert!(ring.qflock_vars <= log_n); + assert!( + ring.offset.is_multiple_of(1usize << ring.qflock_vars), + "q_flock offset must be 2^qflock_vars-aligned" + ); + for claim in &ring.claims { + assert_eq!(claim.suffix_point.len(), ring.qflock_vars); + } + // Every claim's support must lie inside the cube, or its selector coords would + // run off the end of the fold point (mirror of the opener's own bound). + assert!(ring.offset + (1usize << ring.qflock_vars) <= 1usize << log_n); } - // Every claim's support must lie inside the cube, or its selector coords would - // run off the end of the fold point (mirror of the opener's own bound). - assert!(ring.offset + (1usize << qflock_vars) <= 1usize << log_n); assert!( point_claims.iter().all(|c| claim_range(c).1 <= 1usize << log_n), "every claim must live inside the committed cube" @@ -369,7 +374,7 @@ pub fn verify_opening_batch_mixed_whir_stacked( let (lambdas_rs, lambdas_pd) = lambdas.split_at(n_rs); let mut target = F192::ZERO; - for (claim, g) in ring.claims.iter().zip(lambdas_rs.iter()) { + for (claim, g) in rings.iter().flat_map(|ring| &ring.claims).zip(lambdas_rs.iter()) { target += *g * ring_switch::verify_finish(claim.s_hat_v, &coordinate_weights); } for (claim, g) in point_claims.iter().zip(lambdas_pd.iter()) { @@ -377,18 +382,22 @@ pub fn verify_opening_batch_mixed_whir_stacked( } // 3. Evaluate the lifted weight once, at the terminal sumcheck point. - let sel = ring.offset >> qflock_vars; let eval_b_at = |x: &[F192]| -> F192 { - let (x_lo, x_hi) = x.split_at(qflock_vars); - let mut sel_eq = F192::ONE; - for (k, &xi) in x_hi.iter().enumerate() { - sel_eq *= if (sel >> k) & 1 == 1 { xi } else { F192::ONE + xi }; - } - let mut rs_part = F192::ZERO; - for (claim, g) in ring.claims.iter().zip(lambdas_rs.iter()) { - rs_part += *g * ring_switch::eval_rs_eq(claim.suffix_point, x_lo, &coordinate_weights); + let mut acc = F192::ZERO; + let mut lambdas_rs = lambdas_rs.iter(); + for ring in rings { + let (x_lo, x_hi) = x.split_at(ring.qflock_vars); + let sel = ring.offset >> ring.qflock_vars; + let mut sel_eq = F192::ONE; + for (k, &xi) in x_hi.iter().enumerate() { + sel_eq *= if (sel >> k) & 1 == 1 { xi } else { F192::ONE + xi }; + } + let mut rs_part = F192::ZERO; + for (claim, g) in ring.claims.iter().zip(lambdas_rs.by_ref()) { + rs_part += *g * ring_switch::eval_rs_eq(claim.suffix_point, x_lo, &coordinate_weights); + } + acc += rs_part * sel_eq; } - let mut acc = rs_part * sel_eq; for (claim, g) in point_claims.iter().zip(lambdas_pd.iter()) { acc += *g * stack_claim_eq_at(claim, x); } @@ -486,7 +495,14 @@ mod tests { expected[base + (j << stride_log)] += w; } } - let (actual, message) = basis::build(&stack, lane_block, &claims, &lambdas, &ring, &rs_outputs); + let (actual, message) = basis::build( + &stack, + lane_block, + &claims, + &lambdas, + std::slice::from_ref(&ring), + &rs_outputs, + ); assert_eq!(&*actual, expected, "lane_vars={lane_vars}, lanes={lanes}"); let (_, expected_message) = super::super::whir::build_initial_basis(&stack, lane_block, |start, dst| { dst.copy_from_slice(&expected[start..start + dst.len()]); @@ -598,7 +614,15 @@ mod tests { ); let (cm, pd) = commit(&stack, log_n, pc.initial_k, pc.log_inv_rates[0]); let mut ps = fiat_shamir::transcript::ProverState::from_label(DOMAIN); - open_batch_mixed_whir_stacked(&mut ps, log_n, &stack, &pd, &pc, &point_claims, &ring); + open_batch_mixed_whir_stacked( + &mut ps, + log_n, + &stack, + &pd, + &pc, + &point_claims, + std::slice::from_ref(&ring), + ); Instance { vc: pc, @@ -640,7 +664,7 @@ mod tests { 1 << inst.vc.initial_k, &inst.root, point_claims, - &ring, + std::slice::from_ref(&ring), ) .is_ok() } @@ -757,7 +781,15 @@ mod tests { claims, }; let mut ps = fiat_shamir::transcript::ProverState::from_label(DOMAIN); - open_batch_mixed_whir_stacked(&mut ps, log_n, &stack, &pd, &pc, &point_claims, &ring); + open_batch_mixed_whir_stacked( + &mut ps, + log_n, + &stack, + &pd, + &pc, + &point_claims, + std::slice::from_ref(&ring), + ); let fs = ps.into_proof(); let ring_v = RingSwitchVerify { @@ -774,7 +806,7 @@ mod tests { 1 << pc.initial_k, &cm.root, &point_claims, - &ring_v + std::slice::from_ref(&ring_v) ) .is_ok(), "honest crossing-regime opening rejected" @@ -797,7 +829,7 @@ mod tests { 1 << pc.initial_k, &cm.root, &point_claims, - &bad_ring + std::slice::from_ref(&bad_ring) ) .is_err(), "tampered crossing-regime ring slice accepted" diff --git a/crates/pcs/src/stack_open/basis.rs b/crates/pcs/src/stack_open/basis.rs index 08fc34b4f..80aa8e737 100644 --- a/crates/pcs/src/stack_open/basis.rs +++ b/crates/pcs/src/stack_open/basis.rs @@ -78,7 +78,7 @@ pub(super) fn build( lane_block: usize, claims: &[StackClaim], lambdas: &[F192], - ring: &RingSwitchOpen, + rings: &[RingSwitchOpen], rs_outputs: &[DeferredRingSwitchOutput], ) -> (ArenaVec, SumcheckMessage) { assert_eq!(claims.len(), lambdas.len()); @@ -94,13 +94,25 @@ pub(super) fn build( lane.push(index); } } - let ring_end = ring.offset + (1 << ring.qflock_vars); + // Each ring's outputs are its own run of `rs_outputs`, in ring order. + let mut first = 0; + let regions: Vec<_> = rings + .iter() + .map(|ring| { + let outputs = &rs_outputs[first..first + ring.claims.len()]; + first += ring.claims.len(); + (ring.offset, ring.offset + (1usize << ring.qflock_vars), outputs) + }) + .collect(); + assert_eq!(first, rs_outputs.len()); build_initial_basis(stack, lane_block, |start, dst| { dst.fill(F192::ZERO); - let lo = start.max(ring.offset); - let hi = (start + dst.len()).min(ring_end); - if lo < hi { - combine_deferred_chunk(rs_outputs, lo - ring.offset, &mut dst[lo - start..hi - start]); + for &(offset, end, outputs) in ®ions { + let lo = start.max(offset); + let hi = (start + dst.len()).min(end); + if lo < hi { + combine_deferred_chunk(outputs, lo - offset, &mut dst[lo - start..hi - start]); + } } let mut scratch = [MaybeUninit::uninit(); INITIAL_BASIS_CHUNK]; for &index in &by_lane[start / lane_block] { diff --git a/crates/pcs/src/whir.rs b/crates/pcs/src/whir.rs index e3d4d7f87..a3c18034a 100644 --- a/crates/pcs/src/whir.rs +++ b/crates/pcs/src/whir.rs @@ -1046,7 +1046,7 @@ impl<'a> SumcheckProver<'a> { /// Sample `count` query positions in transcript order: no dedup, no sort. /// `block_len = 2^d`; each squeezed field element yields `⌊192/d⌋` positions as -/// its disjoint d-bit chunks (low bits first), matching the scheme the recursive guest derives (fixed `192/d` per +/// its disjoint d-bit chunks (low bits first) (fixed `192/d` per /// squeeze, dup-tolerant: soundness matches the deployed PCS with the same /// `config.queries`). Duplicates are harmless, a repeated position re-opens the /// same Merkle-authenticated row. diff --git a/crates/pcs/src/whir_config.rs b/crates/pcs/src/whir_config.rs index 0058561d5..26b2cc9bc 100644 --- a/crates/pcs/src/whir_config.rs +++ b/crates/pcs/src/whir_config.rs @@ -75,7 +75,7 @@ pub const RS_DOMAIN_INITIAL_REDUCTION_FACTOR: usize = 3; /// After each subsequent fold, shrink the total Reed--Solomon domain by one /// bit. This mirrors WHIR's recursive-domain schedule; unlike the initial /// reduction, it is deliberately fixed rather than a tuning parameter. -const RS_DOMAIN_SUBSEQUENT_REDUCTION_FACTOR: usize = 1; +pub const RS_DOMAIN_SUBSEQUENT_REDUCTION_FACTOR: usize = 1; const _: () = assert!(RS_DOMAIN_INITIAL_REDUCTION_FACTOR <= INITIAL_FOLDING_FACTOR); const _: () = assert!(RS_DOMAIN_SUBSEQUENT_REDUCTION_FACTOR <= SUBSEQUENT_FOLDING_FACTOR); @@ -85,10 +85,9 @@ const _: () = assert!(RS_DOMAIN_SUBSEQUENT_REDUCTION_FACTOR <= SUBSEQUENT_FOLDIN /// clear instead of committed and folded further. pub const RESIDUAL_MAX_LOG: usize = 5; -// The recursion guest rotates the terminal point left by the lane fold to index it -// by witness coordinate, and the residual segment is what the last lane challenges -// rotate past, so the residual may never be longer than that fold -// (`rec_aggregation`'s placeholder emitter re-checks this per derived candidate). +// A verifier rotates the terminal point left by the lane fold to index it by +// witness coordinate, and the residual segment is what the last lane challenges +// rotate past, so the residual may never be longer than that fold. const _: () = assert!(RESIDUAL_MAX_LOG <= INITIAL_FOLDING_FACTOR); /// Shape plus per-level soundness parameters for one WHIR opening. Prover @@ -119,7 +118,7 @@ pub type VerifierConfig = ProverConfig; /// The per-level shape table a [`VerifierConfig`] implies for a /// `log_n`-variable opening: the numbers every consumer of the multilevel -/// protocol (the verifier itself, recursion harnesses) otherwise re-derives. +/// protocol otherwise re-derives. #[derive(Clone, Debug)] pub struct LevelShapes { /// Level count (`level_steps + 1`). diff --git a/crates/pcs/src/whir_induce.rs b/crates/pcs/src/whir_induce.rs index 271a7fcf0..c34b18982 100644 --- a/crates/pcs/src/whir_induce.rs +++ b/crates/pcs/src/whir_induce.rs @@ -26,9 +26,7 @@ fn next_s(s: F64, s_at_root: F64) -> F64 { } /// `sks_vks[k] = s_k(v_k)` for `k = 0..=log_n`, over K. Mirror of -/// `whir::eval_sk_at_vks`. Public for the recursion harness, which dumps -/// these vanishing-polynomial values as guest hints (base-field, embedded into -/// the tower with both extension limbs zero). +/// `whir::eval_sk_at_vks`. pub fn eval_sk_at_vks(log_n: usize) -> Vec { let mut sks_vks = vec![F64::ZERO; log_n + 1]; sks_vks[0] = F64::ONE; diff --git a/crates/primitives/src/field/mod.rs b/crates/primitives/src/field/mod.rs index 617db13dc..91c2b7626 100644 --- a/crates/primitives/src/field/mod.rs +++ b/crates/primitives/src/field/mod.rs @@ -96,18 +96,46 @@ pub fn g_pow(i: usize) -> F64 { /// The fixed generator `g = x ∈ K`, with `ord(g) = 2^64 - 1` (pinned by a /// field test), larger than every index any admissible /// instance uses (the verifier's instance caps, §cpu). For `k < 64`, `g^k` is -/// the monomial `x^k` (bit `k`), which the XMSS encoding check relies on. +/// the monomial `x^k` (bit `k`). pub const G: F64 = F64::G; /// MLE of the index column `[g^0, …, g^{2^n−1}]` over the `n`-variable cube, /// evaluated at an `E`-point: `∏_k (1 + ζ_k·(1 + g^{2^k}))` in `O(n)` (§sec:idxcol). /// The `g^{2^k}` factors are `K`-constants, so each term is one mixed product. pub fn index_mle(zeta: &[F192]) -> F192 { - let mut acc = F192::ONE; - let mut g2k = G; // g^{2^0} = g + powers_mle(F64::ONE, G, zeta) +} + +/// MLE of the integer column `[base ^ (z << shift)]_z`, entry `z` being the element +/// whose bits are that integer's: `base + Σ_k ζ_k·x^{k+shift}`, linear, since bit `k` +/// contributes the monomial `x^k` (§sec:idxcol). What addresses the registers, RAM +/// and the bytecode: with `base` a multiple of the region's size, the XOR is the sum. +pub fn int_index_mle(base: F64, shift: u32, zeta: &[F192]) -> F192 { + zeta.iter().enumerate().fold(F192::from(base), |acc, (k, z)| { + acc + z.mul_base(F64(1 << (k as u32 + shift))) + }) +} + +/// MLE of the geometric column `[first·ratio^z]_z` over the `n`-variable cube: +/// `first·∏_k (1 + ζ_k·(1 + ratio^{2^k}))`, the index column's formula for any ratio +/// (§sec:idxcol). What makes a range-check table free: it is never committed. +pub fn powers_mle(first: F64, ratio: F64, zeta: &[F192]) -> F192 { + let mut acc = F192::from(first); + let mut r2k = ratio; for &z in zeta { - acc *= F192::ONE + z.mul_base(F64::ONE + g2k); - g2k = g2k * g2k; + acc *= F192::ONE + z.mul_base(F64::ONE + r2k); + r2k = r2k * r2k; } acc } + +/// `[first·ratio^z]_{z < n}`. +pub fn geometric(first: F64, ratio: F64, n: usize) -> Vec { + let mut out = Vec::with_capacity(n); + let mut acc = first; + for _ in 0..n { + out.push(acc); + acc *= ratio; + } + out +} diff --git a/crates/primitives/src/test_rng.rs b/crates/primitives/src/test_rng.rs index 1531e018a..1f62c008c 100644 --- a/crates/primitives/src/test_rng.rs +++ b/crates/primitives/src/test_rng.rs @@ -2,8 +2,8 @@ //! kernel and its scalar oracle see the same stream on every host. //! //! Exported normally rather than under `#[cfg(test)]`: integration tests in -//! `pcs`, `flock`, `primitives` and `rec_aggregation` are separate crates and -//! cannot reach a test-only module. +//! `pcs`, `flock` and `primitives` are separate crates and cannot reach a +//! test-only module. use rand::{RngCore, SeedableRng, rngs::StdRng}; diff --git a/crates/rec_aggregation/Cargo.toml b/crates/rec_aggregation/Cargo.toml deleted file mode 100644 index 963b74665..000000000 --- a/crates/rec_aggregation/Cargo.toml +++ /dev/null @@ -1,26 +0,0 @@ -[package] -name = "rec_aggregation" -version.workspace = true -edition.workspace = true -publish = false - -[lints] -workspace = true - -[dependencies] -primitives.workspace = true -parallel.workspace = true -pcs.workspace = true -flock.workspace = true -lean_vm.workspace = true -lean_compiler.workspace = true -xmss.workspace = true -sphincs.workspace = true -lean_da.workspace = true -rand.workspace = true -bincode.workspace = true -serde.workspace = true -tracing.workspace = true - -[dev-dependencies] -zk_alloc.workspace = true diff --git a/crates/rec_aggregation/guests/lean_ethereum.py b/crates/rec_aggregation/guests/lean_ethereum.py deleted file mode 100644 index 65d66ae32..000000000 --- a/crates/rec_aggregation/guests/lean_ethereum.py +++ /dev/null @@ -1,3724 +0,0 @@ -# The recursive aggregation guest: zkDSL, not runnable Python (see -# crates/lean_compiler/zkDSL.md). One node of an aggregation tree verifies raw XMSS -# and SPHINCS+ signatures and sub-proofs OF THIS SAME BYTECODE, and publishes the -# statement digest binding its signature claims and a possibly empty list of LeanDA roots. -# It may check a blob matrix directly and retain any roots proved by its children. -# Reading order: `main` is the -# node, `verify_sub` the in-circuit copy of `lean_vm::cpu::verify`, and -# `open_stacked` the WHIR opening it dispatches into. -# -# Every `*_PLACEHOLDER` below is filled by the host at compile time -# (rec_aggregation::aggregation::placeholder_map), so one source serves every inner -# shape. Names follow doc/leanvm/preamble/macros.tex. -from snark_lib import * - -# ---------------------------------------------------------------- proof stream -# The proof stream rides ONE padded witness hint (the guest walks only the prefix -# the shape dictates); binding always comes from the per-word absorbs. -STREAM_CAP = STREAM_CAP_PLACEHOLDER -MIN_LOG_MEM = MIN_LOG_MEM_PLACEHOLDER -INV_GEN = INV_GEN_PLACEHOLDER - -# ------------------------------------------------------------- the field, GF(2^192) -# Tower F192 = F64[Y]/(Y^3+Y+1). Y_TOWER embeds Y for reassembling -# e192(lo,hi,top) = lo + hi*Y + top*Y², and Y_INV also deduces a top limb at the -# opening boundary once the low and high limbs are transmitted. COORD_BASIS is the -# coordinate basis e_i of F192 (which spans the WHOLE field, unlike the g-power -# basis GEN**i, which spans only F64): hint_decompose_bits emits a word's -# coordinate bits, so a value reconstructs as Σ b_i·COORD_BASIS[i]. -FIELD_BITS = 192 -BASE_FIELD_BITS = 64 -Y_TOWER = Y_TOWER_PLACEHOLDER -Y_INV = Y_INV_PLACEHOLDER -COORD_BASIS = COORD_BASIS_PLACEHOLDER -# Six challenges compose the F2-linear map that batches the 192 transposed -# ring-switch coordinates. -RING_MAP_SHIFTS = [32, 16, 8, 4, 2, 1] -# Exponent bit-widths: an announced 32-bit count decomposes into COUNT_BITS bits, -# its top bit constrained to zero so the native strict 32-bit bound holds; any -# structural size (sums of 2^kappa, packing offsets) fits SIZE_BITS bits; and a -# structural LOG (log_mem, tau_t, log_inv_rate), announced as an integer word and -# raised to a g-power by g_power_of_word, is below SIZE_BITS, so LOG_WORD_BITS bits -# are enough and the reconstruction IS the bound (a larger announced log cannot -# reproduce itself from this many bits). -COUNT_BITS = 33 -SIZE_BITS = 34 -LOG_WORD_BITS = 6 - -# ------------------------------------------------------------------- Fiat-Shamir -# Every absorbed block carries its domain tag in lane 3, which is exactly -# `fs_compress`'s `tail` argument, so a role is never smuggled through the data -# lanes. The seeding block has none, being fixed at the head of the chain. -DS_OBSERVE = 1 -DS_SQ = 2 -DS_POW_BASE = 3 -DS_POW_NONCE = 4 - -# ----------------------------------------------------------- loop-carried chains -# A loop whose state is several fields keeps ONE heap run of N cells per iteration, -# so a step is one pointer multiply rather than one per field. -PAIR_SLOTS = 2 # the Fiat-Shamir state pair alone -# A Fiat-Shamir chain carrying one accumulator (the claim batching loops). -ACC_FS0 = 0 -ACC_FS1 = 1 -ACC_VALUE = 2 -ACC_SLOTS = 3 -# One sumcheck round of the bus GKR, the table batch, or flock's multilinear rounds. -ROUND_FS0 = 0 -ROUND_FS1 = 1 -ROUND_CURSOR = 2 -ROUND_CLAIM = 3 -ROUND_SLOTS = 4 -# One product layer of the bus GKR. -LAYER_FS0 = 0 -LAYER_FS1 = 1 -LAYER_CURSOR = 2 -LAYER_PUSH = 3 -LAYER_PULL = 4 -LAYER_COUNT = 5 -LAYER_LAMBDA = 6 -LAYER_ROW = 7 -LAYER_POS = 8 -LAYER_SLOTS = 9 -# One OOD sample of a WHIR level. -OOD_BETA = 0 -OOD_Y = 1 -OOD_C0 = 2 -OOD_C2 = 3 -OOD_SLOTS = 4 - -# ---------------------------------------------------- the bus: sides and blocks -# GKR sides. The layer counts mu_s are hinted and certified from the block kappas; -# GKR_ROUNDS_CAP caps the per-tree round positions (triangle rounds plus one slot -# per layer) and GKR_POINTS_CAP the point triangle (rows x MU_CAP). -PUSH_SIDE = 0 -PULL_SIDE = 1 -COUNT_SIDE = 2 -N_GKR_SIDES = 3 -GKR_ROUNDS_CAP = GKR_ROUNDS_CAP_PLACEHOLDER -MU_CAP = MU_CAP_PLACEHOLDER -GKR_POINTS_CAP = GKR_POINTS_CAP_PLACEHOLDER -# Bus blocks, flattened across the 3 sides (side s covers blocks -# [SIDE_BLOCK_START[s], SIDE_BLOCK_START[s+1])). The block STRUCTURE is -# protocol-fixed and baked: each block's coord range [BLOCK_COORD_OFF, -# +BLOCK_COORD_COUNT), per coord its COORD_TYPE kind (mirroring leaf.rs::Coord), -# COORD_CONST (the const value, a product's or gcol's g^k, else 0), and the kappa -# SOURCE map (BLOCK_KAPPA_SRC/ADJ: 0 = const adj, 1 = log_mem, 2+t = tau_t). The -# block SHAPES are all reconstructed at runtime from the certified logs: kappa -# directly, the selector bits by pinned advice-decompositions. BLOCK_TABLE names -# the table a block's flush belongs to, or NO_TABLE for the framework blocks -# (boundary, memory seed/finalize, bytecode seed/finalize); it is also what marks a -# block as owned, an owned block's fingerprint being settled by the table sumcheck -# off its table's column evaluations. -COORD_KIND_CONST = 0 -COORD_KIND_COL = 1 -COORD_KIND_GCOL = 2 -COORD_KIND_INDEX = 3 -COORD_KIND_PUBLIC = 4 -COORD_KIND_PROD = 5 -NO_TABLE = NO_TABLE_PLACEHOLDER -SIDE_BLOCK_START = SIDE_BLOCK_START_PLACEHOLDER -N_BLOCKS = N_BLOCKS_PLACEHOLDER -BLOCK_KAPPA_SRC = BLOCK_KAPPA_SRC_PLACEHOLDER -BLOCK_KAPPA_ADJ = BLOCK_KAPPA_ADJ_PLACEHOLDER -BLOCK_TABLE = BLOCK_TABLE_PLACEHOLDER -BLOCK_SIDE = BLOCK_SIDE_PLACEHOLDER -BLOCK_COORD_OFF = BLOCK_COORD_OFF_PLACEHOLDER -BLOCK_COORD_COUNT = BLOCK_COORD_COUNT_PLACEHOLDER -COORD_TYPE = COORD_TYPE_PLACEHOLDER -COORD_CONST = COORD_CONST_PLACEHOLDER -# Claim dedup: push/pull share their GKR point, so a column read by two blocks with -# the same kappa (across OR within the sides) is streamed and opened ONCE. -# COORD_FRESH = 1 on the first occurrence (read the stream, fill pool slot -# COORD_CLAIM_SLOT), 0 on a duplicate (reuse that slot). The count side has its own -# point, so its claims never dedup against the pair's. -COORD_FRESH = COORD_FRESH_PLACEHOLDER -COORD_CLAIM_SLOT = COORD_CLAIM_SLOT_PLACEHOLDER -# A TABLE block's coordinates, flattened into TERMS: coord c is -# Σ_{j < COORD_TERM_COUNT[c]} term(COORD_TERM_OFF[c] + j), each term a TERM_TYPE -# kind over that table's LOCAL column indices TERM_COL_A/TERM_COL_B, scaled by -# TERM_CONST. Those coords raise no claim (the table sumcheck settles them), which -# is what lets one carry a value the row DERIVES from its columns: an XOR/MUL -# result, a DEREF store, a JUMP successor. A framework coord has no terms. -COORD_TERM_OFF = COORD_TERM_OFF_PLACEHOLDER -COORD_TERM_COUNT = COORD_TERM_COUNT_PLACEHOLDER -TERM_TYPE = TERM_TYPE_PLACEHOLDER -TERM_CONST = TERM_CONST_PLACEHOLDER -TERM_COL_A = TERM_COL_A_PLACEHOLDER -TERM_COL_B = TERM_COL_B_PLACEHOLDER -N_BUS_CLAIMS = N_BUS_CLAIMS_PLACEHOLDER -INDEX_MLE_FACTORS = INDEX_MLE_FACTORS_PLACEHOLDER # 1 + g^(2^i) -# Committed-coordinate claims (Col/GCol coords across all sides) and the deferred -# bytecode values (Public coords). -N_CLAIMS = N_CLAIMS_PLACEHOLDER -# A bus tuple's coordinates index the 2^N_TUPLE_BITS fingerprint slots (doc -# sec:gp). The stacked bytecode has BYTECODE_COLS encoding columns, stacked along -# LOG2_BYTECODE_COLS selector bits into ONE multilinear; push and pull share their -# GKR point, so the columns are opened ONCE. -N_TUPLE_BITS = 4 -N_TUPLE_SLOTS = 16 -BYTECODE_COLS = BYTECODE_COLS_PLACEHOLDER -LOG2_BYTECODE_COLS = LOG2_BYTECODE_COLS_PLACEHOLDER - -# --------------------------------------------------------------- the six tables -# The table sumcheck's batch carries EVERY committed column of a table, because its -# bus forms read the flushed ones and its constraint the rest; TABLE_COLS_CAP caps -# the evaluation frame. ETA_OFFSET[t] starts table t's disjoint range of zc_xi -# powers; the three bus forms take ETA_FORM_BASE + side, the SAME three powers for -# every table, and that sharing is what makes the batch's target derivable from the -# three leaf claims. FLOORS[t] is the table's tau floor (BLAKE2s is sized to -# flock's instance count, >= 2^3). -TABLE_XOR = 0 -TABLE_MUL = 1 -TABLE_SET = 2 -TABLE_DEREF = 3 -TABLE_JUMP = 4 -TABLE_BLAKE2s = 5 -N_TABLES = N_TABLES_PLACEHOLDER -FLOORS = [0, 0, 0, 0, 0, 3] -N_TABLE_COLS = N_TABLE_COLS_PLACEHOLDER -TABLE_COLS_CAP = TABLE_COLS_CAP_PLACEHOLDER -ETA_OFFSET = ETA_OFFSET_PLACEHOLDER -ETA_FORM_BASE = ETA_FORM_BASE_PLACEHOLDER -N_ETA_POWS = N_ETA_POWS_PLACEHOLDER - -# ------------------------------------------------------------ flock (the R1CS) -# Univariate skip: K_SKIP variables fold in one skip round (half-domain 2^K_SKIP -# phi8 nodes), then N_FIXED_CHALLENGE_ROUNDS fixed inner rounds (FIXED_CHALLENGES), -# then sampled outer rounds. LAGRANGE_INV_* are the one baked inverse barycentric -# denominator per domain (combined, S). The zerocheck point/round buffers are sized -# at runtime in the exponent (m = K_LOG + tau_5 and m - 6, both certified); -# LINCHECK_ROUNDS = K_LOG - K_SKIP is protocol-fixed and PIN_COLUMN is the -# const-pin column. -K_SKIP = K_SKIP_PLACEHOLDER -N_FIXED_CHALLENGE_ROUNDS = N_FIXED_CHALLENGE_ROUNDS_PLACEHOLDER -FIXED_CHALLENGES = FIXED_CHALLENGES_PLACEHOLDER -PHI8_NODES = PHI8_NODES_PLACEHOLDER -LAGRANGE_INV_COMBINED = LAGRANGE_INV_COMBINED_PLACEHOLDER -LAGRANGE_INV_S = LAGRANGE_INV_S_PLACEHOLDER -LINCHECK_ROUNDS = LINCHECK_ROUNDS_PLACEHOLDER -PIN_COLUMN = PIN_COLUMN_PLACEHOLDER -K_LOG = K_LOG_PLACEHOLDER -SLOT_STRIDE_LOG = SLOT_STRIDE_LOG_PLACEHOLDER # = K_LOG - LOG_PACKING (=8); the q_flock slot stride - -# ------------------------------------------------- the stacked WHIR opening -# The opening is dispatched by the certified committed log-size m through `match`. -# The LIG_* tables carry one row per (rate, m), emitted from the same -# derive_profile/level_shapes the prover uses: scalars index as TBL[m_idx], -# per-level values as TBL[m_idx * LIG_MAX_LEVELS + lvl] where m_idx is the -# flattened rate-major configuration index, and the subspace vanishing constants -# with the LIG_MAX_VANISH_LEN stride. -LIG_MIN_LOG_SIZE = LIG_MIN_LOG_SIZE_PLACEHOLDER -LIG_N_LOG_SIZES = LIG_N_LOG_SIZES_PLACEHOLDER -LIG_N_RATES = LIG_N_RATES_PLACEHOLDER -# Committed-column kappa sources (0 = const COL_KAPPA_ADJ, 1 = log_mem, 2+t = tau_t) -# and the PCS floor for the stacked size. -N_COMMITTED_COLS = N_COMMITTED_COLS_PLACEHOLDER -N_COLUMN_LOGS = N_COLUMN_LOGS_PLACEHOLDER -COL_KAPPA_SRC = COL_KAPPA_SRC_PLACEHOLDER -COL_KAPPA_ADJ = COL_KAPPA_ADJ_PLACEHOLDER -PCS_MIN_MU = PCS_MIN_MU_PLACEHOLDER -# Global maxima; StackBuf frame sizes are parse-time, so they must be baked. -LIG_MAX_LEVELS = LIG_MAX_LEVELS_PLACEHOLDER -LIG_MAX_VANISH_LEN = LIG_MAX_VANISH_LEN_PLACEHOLDER -LIG_MAX_OOD_SAMPLES = LIG_MAX_OOD_SAMPLES_PLACEHOLDER -LIG_LOG_MSG_COLS_CAP = LIG_LOG_MSG_COLS_CAP_PLACEHOLDER -YR_LOG_CAP = YR_LOG_CAP_PLACEHOLDER -MAX_STACK_LOG = LIG_MIN_LOG_SIZE + LIG_N_LOG_SIZES - 1 -LIG_N_LEVELS = LIG_N_LEVELS_PLACEHOLDER -LIG_YR_LEVEL = LIG_YR_LEVEL_PLACEHOLDER -LIG_YR_LOG_LEN = LIG_YR_LOG_LEN_PLACEHOLDER -LIG_YR_LEN = LIG_YR_LEN_PLACEHOLDER -LIG_TOTAL_FOLDS = LIG_TOTAL_FOLDS_PLACEHOLDER -LIG_MAX_QUERIES = LIG_MAX_QUERIES_PLACEHOLDER -LIG_MAX_SQUEEZES = LIG_MAX_SQUEEZES_PLACEHOLDER -LIG_MAX_INTERLEAVE = LIG_MAX_INTERLEAVE_PLACEHOLDER -LIG_POSITIONS_LEN = LIG_POSITIONS_LEN_PLACEHOLDER -LIG_CAP_DEPTH = LIG_CAP_DEPTH_PLACEHOLDER -LIG_CAP_OFF = LIG_CAP_OFF_PLACEHOLDER -LIG_CAP_LEN = LIG_CAP_LEN_PLACEHOLDER -LIG_QUERY_GRIND_BITS = LIG_QUERY_GRIND_BITS_PLACEHOLDER -LIG_OOD_SAMPLES = LIG_OOD_SAMPLES_PLACEHOLDER -LIG_QUERIES = LIG_QUERIES_PLACEHOLDER -LIG_FOLDS = LIG_FOLDS_PLACEHOLDER -LIG_INTERLEAVE = LIG_INTERLEAVE_PLACEHOLDER -LIG_LEAF_BLOCKS = LIG_LEAF_BLOCKS_PLACEHOLDER -LIG_PACKED_ROW_CAP = LIG_PACKED_ROW_CAP_PLACEHOLDER -LIG_ROW_CAP = LIG_ROW_CAP_PLACEHOLDER -LIG_PATH_CAP = LIG_PATH_CAP_PLACEHOLDER -LIG_TREE_DEPTH = LIG_TREE_DEPTH_PLACEHOLDER -LIG_SQUEEZES = LIG_SQUEEZES_PLACEHOLDER -LIG_POSITIONS_OFF = LIG_POSITIONS_OFF_PLACEHOLDER -LIG_LOG_MSG_COLS = LIG_LOG_MSG_COLS_PLACEHOLDER -LIG_RESIDUAL_FOLD_OFF = LIG_RESIDUAL_FOLD_OFF_PLACEHOLDER -LIG_RESIDUAL_PREFIX_LEN = LIG_RESIDUAL_PREFIX_LEN_PLACEHOLDER -LIG_FOLDS_OFF = LIG_FOLDS_OFF_PLACEHOLDER -LIG_VANISH_OFF = LIG_VANISH_OFF_PLACEHOLDER -LIG_VANISH_VALS = LIG_VANISH_VALS_PLACEHOLDER -LIG_VANISH_INVS = LIG_VANISH_INVS_PLACEHOLDER -LIG_N_CANDIDATES = LIG_N_CANDIDATES_PLACEHOLDER -LIG_MIN_SHIFT_INV = LIG_MIN_SHIFT_INV_PLACEHOLDER -# eval_b claim descriptors. CLAIM_POINT_BUF says which point buffer a pooled -# claim's x-part lives in, CLAIM_COMMITTED_COL maps it to the compact index of the -# committed column it must open (a virtual BLAKE2s value claim maps to QFLOCK), -# CLAIM_QFLOCK_SLOT_BITS holds the fixed packed-slot bits of every logical claim -# (zero for a non-virtual one), and QFLOCK_COMMITTED_COL is the ring-switch target. -POINT_BUF_ZETA = 0 -POINT_BUF_RHO = 1 -POINT_BUF_PI = 2 -POINT_BUF_QFLOCK_RHO = 3 -CLAIM_POINT_BUF = CLAIM_POINT_BUF_PLACEHOLDER -CLAIM_COMMITTED_COL = CLAIM_COMMITTED_COL_PLACEHOLDER -CLAIM_QFLOCK_SLOT_BITS = CLAIM_QFLOCK_SLOT_BITS_PLACEHOLDER -QFLOCK_COMMITTED_COL = QFLOCK_COMMITTED_COL_PLACEHOLDER -QFLOCK_VARS_CAP = QFLOCK_VARS_CAP_PLACEHOLDER - -# ------------------------------------------------------ statements and deferral -# A node defers three claims on fixed polynomials. DEFER_SIZE is the region one -# sub-proof's verification exports (a bytecode point plus the flock lincheck data, -# see verify_sub's defer_out layout); DEFER_STMT_* index the batched claims a -# node's OWN statement carries. The Fiat-Shamir seed rides the public input rather -# than being baked, so one compiled guest verifies proofs of any inner program. -BYTECODE_LOG = BYTECODE_LOG_PLACEHOLDER # log rows of the bytecode blocks -DEFER_SIZE = DEFER_SIZE_PLACEHOLDER -BYTECODE_VARS = BYTECODE_VARS_PLACEHOLDER # = BYTECODE_LOG + LOG2_BYTECODE_COLS -# The exported record's layout: the shared bytecode point, then the flock data. -FRESH_BC_VALUE = BYTECODE_VARS -FRESH_ALPHA = BYTECODE_VARS + 1 -FRESH_Z_SKIP = BYTECODE_VARS + 2 -FRESH_ZCHI = BYTECODE_VARS + 3 -FRESH_LINCHECK_RS = FRESH_ZCHI + LINCHECK_ROUNDS -FRESH_Z_PARTIAL = FRESH_LINCHECK_RS + LINCHECK_ROUNDS -FRESH_MATPART = FRESH_Z_PARTIAL + 2 ** K_SKIP -DEFER_STMT_CELLS = BYTECODE_VARS + 1 + 2 * K_LOG + 2 -DEFER_STMT_BC_VALUE = BYTECODE_VARS -DEFER_STMT_MAT_POINT = BYTECODE_VARS + 1 -DEFER_STMT_A_VALUE = BYTECODE_VARS + 1 + 2 * K_LOG -DEFER_STMT_B_VALUE = BYTECODE_VARS + 2 + 2 * K_LOG -AGG_SEED_0 = AGG_SEED_0_PLACEHOLDER -AGG_SEED_1 = AGG_SEED_1_PLACEHOLDER -# The statement digest's preimage: the STMT_HEADER header values as the 16-byte -# cells they already are (the seed and the signer-set digest, which itself binds -# the epoch groups and every count), then the deferred cells' tower limbs, two to -# a cell and four cells to a 64-byte block. No domain tag: the seed leads, and it -# binds this bytecode and flock's R1CS. -STMT_HEADER = STMT_HEADER_PLACEHOLDER -STMT_DEFER_OFF = STMT_HEADER -STMT_ODD = STMT_ODD_PLACEHOLDER -STMT_PAIRS = STMT_PAIRS_PLACEHOLDER -STMT_PAD_CELLS = STMT_PAD_CELLS_PLACEHOLDER -STMT_BLOCKS = STMT_BLOCKS_PLACEHOLDER -# The declared lists are hashed with plain BLAKE2s over a flat run of cells, 64 -# bytes a compression. A block's byte counter is a runtime value and the ISA has no -# integer addition, so it splits as in doc §sec:prog-byte-counter: a window of -# SIGNERS_WINDOW blocks shares one base 64·SIGNERS_WINDOW·q, whose set bits all sit -# above the window's own offsets 64(j+1), so a block's metadata cell is one XOR. The -# base comes from the window loop's own counter, and the one block whose offset -# overlaps it takes the next window's base instead. -SIGNERS_WINDOW = SIGNERS_WINDOW_PLACEHOLDER -SIGNERS_WINDOW_LOG = SIGNERS_WINDOW_LOG_PLACEHOLDER -SIGNERS_MAX_WINDOWS = SIGNERS_MAX_WINDOWS_PLACEHOLDER -SIGNERS_COUNT_BITS = SIGNERS_COUNT_BITS_PLACEHOLDER -# BLAKE2s's parameterized initial chaining value, which every hash here starts from, -# and the metadata of a final block with a zero counter, to add a length into. -BLAKE2S_IV_0 = BLAKE2S_IV_0_PLACEHOLDER -BLAKE2S_IV_1 = BLAKE2S_IV_1_PLACEHOLDER -MD_FINAL = MD_FINAL_PLACEHOLDER - -# ---------------------------------------------------------- XMSS (host-supplied) -# Every 16-byte native value (tweak, digest, chain tip, sibling, public parameter) -# is one canonical 128-bit cell. XM_* are the tweaks minus their index field, as -# the host read them out of the native `make_tweak`, so the guest holds no byte -# layout of its own; XM_INDEX_WEIGHT[b] is what bit b of an index weighs, an index -# being its set bits summed. -V = V_PLACEHOLDER -W = W_PLACEHOLDER -TARGET_SUM = TARGET_SUM_PLACEHOLDER -LOG_LIFETIME = LOG_LIFETIME_PLACEHOLDER -CHAIN_LENGTH = 2 ** W -CHAIN_STEPS = CHAIN_LENGTH - 1 -WORDS_PER_VALUE = 1 -WORDS_PER_BLOCK = 2 -# Tweak table (one 1-cell tweak per index): encoding | V·CHAIN_STEPS chain | -# wots-pk | merkle. Derived in-circuit, once per epoch group the statement carries. -N_TWEAKS = 1 + V * CHAIN_STEPS + 1 + LOG_LIFETIME -N_TWEAK_CELLS = WORDS_PER_VALUE * N_TWEAKS -WOTS_PK_TWEAK_IDX = 1 + V * CHAIN_STEPS -MERKLE_TWEAK_IDX = WOTS_PK_TWEAK_IDX + 1 -MERKLE_BIT_CELLS = WORDS_PER_VALUE * LOG_LIFETIME # one 1-cell bit word per level -XM_ENC_TWEAK = XM_ENC_TWEAK_PLACEHOLDER -XM_PK_TWEAK = XM_PK_TWEAK_PLACEHOLDER -XM_CHAIN_TWEAKS = XM_CHAIN_TWEAKS_PLACEHOLDER # indexed CHAIN_STEPS·i + s -XM_MERKLE_TWEAKS = XM_MERKLE_TWEAKS_PLACEHOLDER # indexed by level -XM_INDEX_WEIGHT = XM_INDEX_WEIGHT_PLACEHOLDER -# Digits packed per digest lane: W bits each in GF(2^64)'s monomial budget (the -# lane's leftover top bits are ground to zero by the signer). -DIGITS_PER_WORD = V / 2 -TIP_CELLS = 2 * V # tip i at cell 2i, beside the unread upper half of its last hash -WOTS_PK_BLOCKS = (2 + V) / 4 # prefix (tweak, pp) + V tips, four cells a block - -# ------------------------------------------------------ SPHINCS+ (host-supplied) -# The scheme's own letters, prefixed SP_ where XMSS has the same one. -SP_V = SP_V_PLACEHOLDER -SP_W = SP_W_PLACEHOLDER -SP_TARGET_SUM = SP_TARGET_SUM_PLACEHOLDER -SP_D = SP_D_PLACEHOLDER -SP_HEIGHTS = SP_HEIGHTS_PLACEHOLDER # h_lay, one per hypertree layer, top first -SP_SUFFIX = SP_SUFFIX_PLACEHOLDER # SP_SUFFIX[lay] = sum of h_j for j >= lay -SP_A = SP_A_PLACEHOLDER -SP_K = SP_K_PLACEHOLDER -SP_H = SP_H_PLACEHOLDER # the total hypertree height, SP_SUFFIX[0] -SP_CHAIN_LENGTH = 2 ** SP_W -SP_CHAIN_STEPS = SP_CHAIN_LENGTH - 1 -SP_DIGITS_PER_WORD = SP_V / 2 -SP_TIP_CELLS = SP_V -SP_LEAF_BLOCKS = (2 + SP_V) / 4 # prefix (tweak, pp) + V tips, four cells a block -SP_N_FTS = SP_K - 1 # the forest drops the last index's tree -SP_ROOT_BLOCKS = (2 + SP_N_FTS) / 4 -# The message digest is h + k*a bits of a BLAKE2s output: the whole low cell and -# the low 48 bits of the high one. Decomposing the high cell's low lane covers -# them, so the buffer holds three lanes and the top 16 are never read. -SP_BIT_LANES = 3 -SP_BIT_CELLS = SP_BIT_LANES * BASE_FIELD_BITS -# Native tweak prefixes, including the protocol domain separator and type. -SP_TW_CHAIN = SP_TW_CHAIN_PLACEHOLDER -SP_TW_LEAF = SP_TW_LEAF_PLACEHOLDER -SP_TW_NODE = SP_TW_NODE_PLACEHOLDER -SP_TW_ENC = SP_TW_ENC_PLACEHOLDER -SP_TW_FTS_LEAF = SP_TW_FTS_LEAF_PLACEHOLDER -SP_TW_FTS_NODE = SP_TW_FTS_NODE_PLACEHOLDER -SP_TW_FTS_ROOTS = SP_TW_FTS_ROOTS_PLACEHOLDER -SP_TW_MSG = SP_TW_MSG_PLACEHOLDER -# Tweak layout: protocol_domain_sep | type | layer | zero | p | tree | index. -# Each 32-bit field stays within one 64-bit lane. -SP_LAY_MUL = 2 ** 16 -SP_P_MUL = 2 ** 32 -SP_TAU_POS = BASE_FIELD_BITS -SP_J_POS = BASE_FIELD_BITS + 32 -SP_CHAIN_MUL = SP_CHAIN_LENGTH * SP_P_MUL # chain i's tweaks start at p = 2^w * i -# The encoding counter, LE_32 in the low four bytes of its cell: bounded by -# decomposing exactly that many bits, so the guest accepts no preimage the native -# verifier cannot parse. -SP_COUNTER_BITS = 32 - -# --------------------------------------------------------------- node capacities -# MAX_KEYS caps the coverage table's slots, both schemes' declared keys and their -# duplicates, which is what the coverage range check needs below 2^MIN_LOG_MEM; -# MAX_RECURSIONS is the arity of an aggregation tree; MAX_EPOCHS caps the runtime -# number of XMSS epoch groups. -MAX_KEYS = MAX_KEYS_PLACEHOLDER -MAX_DA_ROOTS = MAX_DA_ROOTS_PLACEHOLDER -DA_ROOT_COUNTS = DA_ROOT_COUNTS_PLACEHOLDER -MAX_RECURSIONS = MAX_RECURSIONS_PLACEHOLDER -MAX_EPOCHS = MAX_EPOCHS_PLACEHOLDER - - -# ---------------------------------- LeanDA ------------------------------------------ -# Blob and cell widths are fixed. The row count is hinted and bounded by DA_MAX_ROWS; -# the Merkle trees dispatch on the checked log of its padded value. -DA_LOG_K = DA_LOG_K_PLACEHOLDER -DA_LOG_CELL = DA_LOG_CELL_PLACEHOLDER -DA_MAX_ROWS = DA_MAX_ROWS_PLACEHOLDER -DA_LOG_MAX_ROWS = DA_LOG_MAX_ROWS_PLACEHOLDER -DA_PAD_CELL_0 = DA_PAD_CELL_0_PLACEHOLDER -DA_PAD_CELL_1 = DA_PAD_CELL_1_PLACEHOLDER -DA_PAD_ROW_0 = DA_PAD_ROW_0_PLACEHOLDER -DA_PAD_ROW_1 = DA_PAD_ROW_1_PLACEHOLDER - -DA_CELL = 2 ** DA_LOG_CELL # symbols in a cell -DA_BLOCK_BITS = DA_LOG_K + 1 - DA_LOG_CELL # log of the cells per row -DA_CELLS = 2 ** DA_BLOCK_BITS # cells per row -DA_PREFIX_CELLS = 2 ** (DA_LOG_K - DA_LOG_CELL) # cells in the first half -DA_CELL_BLOCKS = DA_CELL // 8 # BLAKE2s blocks in one cell -DA_ROW_BLOCKS = DA_PREFIX_CELLS // 2 # BLAKE2s blocks in one row digest -DA_TREE_ARMS = DA_LOG_MAX_ROWS + 1 # tree depths the row count can dispatch to - -# =================================== field packing ================================== -# Serializing a word means exposing its three K limbs, and the tower representation -# is unique, so proving that the exposed lanes are in K and weight back to the word -# is the whole check: no equality to make, one less hint to distrust. - - -@inline -def pack64x2(a, b): - assert_in_k(a, b) - return a + Y_TOWER * b - - -def assert_canonical(word): - # Hint the low limb and derive the quotient by Y. Requiring both in K proves - # `word = lo + Y*hi`, hence that its top limb is zero. - lo = StackBuf(1) - hint_f192_limbs(lo, word) - hi = (word + lo[0]) * Y_INV - assert_in_k(lo[0], hi) - return 0 - - -@inline -def challenge_from_state(state): - # Both words are BLAKE2s outputs with zero top limbs. - # Hint d2 and derive d3 = (state[1] + d2)/Y. - # Requiring both in K binds d2 by the tower representation. - # The challenge uses d2; d3 is checked and discarded. - d2 = StackBuf(1) - hint_f192_limbs(d2, state[1]) - d3 = (state[1] + d2[0]) * Y_INV - assert_in_k(d2[0], d3) - return state[0] + Y_TOWER * Y_TOWER * d2[0] - - -# ==================================== Fiat-Shamir =================================== - - -@inline -def fs_compress(state, scalar, tail, out): - # BLAKE2s requires block[0] to have zero top limb, binding the hinted top. - limbs = StackBuf(3) - hint_f192_limbs(limbs, scalar) - assert_in_k(limbs[2], tail) - block = StackBuf(2) - block[0] = scalar + Y_TOWER * Y_TOWER * limbs[2] - block[1] = limbs[2] + Y_TOWER * tail - blake2s(state, block, out) - return - - -@inline -def obs(state, x): - # Bind one scalar into the chain: state <- compress(state, (x, DS_OBSERVE)). - # Returns the successor StackBuf; the call site aliases it (zero copies). - nb = StackBuf(2) - fs_compress(state, x, DS_OBSERVE, nb) - return nb - - -@inline -def fs_next(state, cursor): - # Fetch, observe and advance in one act: read the word under `cursor`, fold it - # into the state, and hand back the successor state, the word, AND the cursor - # stepped one word on. Reading and absorbing are inseparable here, so no - # proof-stream word can enter the computation unbound: the soundness invariant - # the whole guest rests on. All three returns alias into the caller for free. - x = cursor[GEN ** 0] - nb = obs(state, x) - return nb, x, cursor * GEN - - -@inline -def squeeze_state(state): - nb = StackBuf(2) - blake2s(state, [0, Y_TOWER * DS_SQ], nb) - return nb - - -@inline -def squeeze(state): - # Ratchet: the canonical 128+128 digest is the new state; its first three K - # lanes are reassembled as the F192 challenge. - nb = squeeze_state(state) - challenge = challenge_from_state(nb) - return nb, challenge - - -@inline -def absorb_nonce(state, x): - # Full-field grinding nonce absorb: [x.c0, x.c1, x.c2, DS_POW_NONCE]. - nb = StackBuf(2) - fs_compress(state, x, DS_POW_NONCE, nb) - return nb - - -@inline -def squeeze_step(state_0, state_1): - # `squeeze` exposing BOTH output words, so a query-squeeze loop can chain the - # state through a heap buffer. Returns (challenge, next_state_0, next_state_1). - state = [state_0, state_1] - next_state = squeeze_state(state) - challenge = challenge_from_state(next_state) - return challenge, next_state[0], next_state[1] - - -# ============================ bits, logs, and the exponent ========================== - - -@inline -def bind_bits(bits_ptr, value, n: Const): - # Tie an advice bit run back to the value it decomposes. Booleanity is a - # write-once pin: the cell already holds the bit, so storing its square IS the - # assert, one instruction shorter than a separate equality. - acc = 0 - for i in unroll(0, n): - b = bits_ptr[GEN ** i] - bits_ptr[GEN ** i] = b * b - acc += b * COORD_BASIS[i] - assert acc == value - return - - -def exponent_tables(): - # Read-only lookup tables over the exponent domain, indexed at runtime g-powers - # (so they must be heap, not stack): g_logs_pow2[g^j] = 2^j raises a g-power's - # log, and g_squares[g^j] = g^(2^j) turns integer sums of powers of two into - # field products. Both span SIZE_BITS because verify_log2_ceil bounds its result - # there, so g_log reaches g^(SIZE_BITS-1) and indexes g_logs_pow2 at it; sizing - # to COUNT_BITS would leave that lookup reading a prover-chosen cell. - g_logs_pow2 = HeapBuf(SIZE_BITS) - for j in unroll(0, SIZE_BITS): - g_logs_pow2[GEN ** j] = 2 ** j - g_squares = HeapBuf(SIZE_BITS) - sq_run = GEN - for j in unroll(0, SIZE_BITS): - g_squares[GEN ** j] = sq_run - sq_run *= sq_run - return g_logs_pow2, g_squares - - -def g_power_of_word(value, g_squares, nbits: Const): - # g^value for a concrete integer `value` < 2^nbits: advice-decompose its bits, - # tie them back to the word, and assemble Π g^(bit_j·2^j). - bits = HeapBuf(GEN ** nbits) - hint_decompose_bits(bits, value, nbits) - word = 0 - g_value = GEN ** 0 - for j in unroll(0, nbits): - bit = bits[GEN ** j] - assert bit * bit == bit - word += bit * (2 ** j) - g_value *= (1 + bit * (g_squares[GEN ** j] + 1)) - assert word == value - return g_value - - -def verify_log2_ceil(bits_buf, g_logs_pow2, g_squares, floor: Const, nbits: Const): - # Given `nbits` bits already in bits_buf, return (g_log, exp_prod) for - # word = Σ bit_j 2^j: exp_prod = g^word and g_log = g^max(log2_ceil(word), - # floor). g_log is prover advice, pinned to log2_ceil(word) by psum[g_log] == - # word (word < 2^log, the == 2^log case via g_logs_pow2) and word > 2^(log-1) - # (waived at floor). Callers fill the bits and tie word or exp_prod to their - # value. NB: log2 here is the base-2 log of the integer word, not the discrete - # log base g that `log(...)` means. - psum_buf = HeapBuf(SIZE_BITS + 1) # psum_buf[g^j] = value of bits [0, j) - psum_buf[GEN ** 0] = 0 - word = 0 - exp_prod = GEN ** 0 - for j in unroll(0, nbits): - bit = bits_buf[GEN ** j] - assert bit * bit == bit - exp_prod *= (1 + bit * (g_squares[GEN ** j] + 1)) - word += bit * (2 ** j) - psum_buf[GEN ** (j + 1)] = word - for j in unroll(nbits + 1, SIZE_BITS + 1): - psum_buf[GEN ** j] = word - g_log = hint_log2_ceil(bits_buf, nbits, floor) # prover advice; verified below - assert log(g_log) < SIZE_BITS - assert log(g_log / (GEN ** floor)) < SIZE_BITS - low_bits = psum_buf[g_log] # value of bits [0, log) - high_bits = low_bits + word # value of bits [log, nbits) - assert high_bits * low_bits == 0 # word < 2^log (high bits clear) OR ... - assert high_bits * (word + g_logs_pow2[g_log]) == 0 # ... word == 2^log - if g_log != GEN ** floor: - # minimality (word > 2^(log-1)); skip at g_log == g^0, where word is in - # {0,1}, its ceil-log 0 already minimal and psum_buf[g^-1] out of range. - if g_log != GEN ** 0: - low_bits_prev = psum_buf[g_log * INV_GEN] # bits [0, log-1) - word_vs_2logprev = word + g_logs_pow2[g_log * INV_GEN] # 0 iff word == 2^(log-1) - assert (low_bits_prev + word) * word_vs_2logprev != 0 - return g_log, exp_prod - - -def log2_ceil_in_the_exponent(g_N, g_logs_pow2, g_squares, floor: Const, nbits: Const): - # g^log2_ceil(N) given g_N = g^N (N < 2^nbits). There is no in-circuit log, so - # the prover hints N's bits; they are verified and tied back, the value they - # decode to having to equal g_N. - bits = HeapBuf(GEN ** nbits) - hint_decompose_bits_exponent(bits, g_N, nbits) - g_log, g_bits_value = verify_log2_ceil(bits, g_logs_pow2, g_squares, floor, nbits) - assert g_bits_value == g_N - return g_log - - -def decode_query_bits(squeezed_word, positions_out, bit_ptrs_out, depth: Const): - # The squeezed word's bits are advice-decomposed HERE, boolean-constrained and - # tied back by reconstruction; each depth-bit group also becomes a query - # position (little-endian), with a pointer to its bit run (the Merkle direction - # bits). Each field word packs FIELD_BITS // depth positions. The bits live in - # FRAME cells, every index into them being compile-time, and `addr` names the - # run so the direction-bit pointers still reach it. - per_word = FIELD_BITS // depth - bits = StackBuf(FIELD_BITS) - hint_decompose_bits(bits, squeezed_word, FIELD_BITS) - bits_ptr = addr(bits) - reconstructed = 0 - for j in unroll(0, per_word): - base_bit = j * depth - # A group inside one 64-bit limb shifts as a WHOLE, the coordinate basis - # being the polynomial basis there: COORD_BASIS[base_bit + b] == - # COORD_BASIS[base_bit] * COORD_BASIS[b] below 64, so the group contributes - # one multiply rather than one per bit. A group straddling the boundary - # splits into the two runs that do stay inside a limb; `b // cut == 0` IS - # `b < cut`, the DSL's `if` comparing for equality only. - cut = 64 - base_bit % 64 # bits of this group below the next limb - p_lo = 0 - p_hi = 0 - for b in unroll(0, depth): - t = bits[base_bit + b] - bits[base_bit + b] = t * t # booleanity, as a write-once pin - if b // cut == 0: - p_lo += t * COORD_BASIS[b] - else: - p_hi += t * COORD_BASIS[b - cut] - # position = p_lo + 2^cut * p_hi: multiplying by X^cut concatenates the two - # runs, since both degrees stay below 64. - if cut // depth == 0: # `cut < depth`: this group straddles the boundary - positions_out[GEN ** j] = p_lo + COORD_BASIS[cut] * p_hi - reconstructed += COORD_BASIS[base_bit] * p_lo + COORD_BASIS[base_bit + cut] * p_hi - else: - positions_out[GEN ** j] = p_lo - reconstructed += COORD_BASIS[base_bit] * p_lo - bit_ptrs_out[GEN ** j] = bits_ptr * GEN ** base_bit - for i in unroll(per_word * depth, FIELD_BITS): - t = bits[i] - bits[i] = t * t - reconstructed += t * COORD_BASIS[i] - assert reconstructed == squeezed_word - return - - -def grind_check(state_0, state_1, nonce, nbits_g): - # WHIR fold/query grinding: digest = H(H(state, POW_BASE), (nonce, POW_NONCE)), - # whose low nbits (nbits_g = g^nbits) must be zero. The PoW window of - # transcript::pow_bits_ok is `digest.0 & ((1 << bits) - 1)` with nbits < 64, so - # it lives entirely in the digest's FIRST 64-bit lane: only that lane is - # advice-decomposed and verified, not all FIELD_BITS of the cell. The caller - # absorbs the full field nonce afterwards. The honest prover searches the - # deterministic u64 subset while verification permits the full field: each - # candidate still costs one hash and succeeds with probability 2^-bits. - if nbits_g == GEN ** 0: - assert nonce == 0 # native canonical zero-work nonce - st = [state_0, state_1] - base = StackBuf(2) - blake2s(st, [0, Y_TOWER * DS_POW_BASE], base) - out = StackBuf(2) - fs_compress(base, nonce, DS_POW_NONCE, out) - lanes = StackBuf(1) - hint_f192_limbs(lanes, out[0]) - high = (out[0] + lanes[0]) * Y_INV - assert_in_k(lanes[0], high) - # Frame cells for the unrolled pass (no DEREF per bit), named by `addr` for the - # zero-check walk, whose bound is runtime and so must index a pointer. - lane_bits = StackBuf(BASE_FIELD_BITS) - hint_decompose_bits(lane_bits, lanes[0], BASE_FIELD_BITS) - acc = 0 - for i in unroll(0, BASE_FIELD_BITS): - b = lane_bits[i] - lane_bits[i] = b * b # booleanity, as a write-once pin - acc += b * COORD_BASIS[i] - assert acc == lanes[0] # the bits ARE the lane's coordinates, so the pins below bind it - lane_ptr = addr(lane_bits) - for xb in mul_range(1, nbits_g): - assert lane_ptr[xb] == 0 - return - - -# =============================== multilinear primitives ============================= - - -@inline -def eq_weight(ch, count: Const, idx: Const, msb_span: Const): - # The eq-tensor weight of compile-time index `idx` against the challenge run - # ch[0..count): prod_c eq(bit(idx), ch[c]), where the bit is bit c of idx - # (msb_span == 0) or bit (msb_span - 1 - c) (an MSB-first walk over an - # msb_span-bit index). - w = GEN ** 0 - for c in unroll(0, count): - cv = ch[GEN ** c] - if msb_span == 0: - if (idx // (2 ** c)) % 2 == 1: - w *= cv - else: - w *= (1 + cv) - else: - if (idx // (2 ** (msb_span - 1 - c))) % 2 == 1: - w *= cv - else: - w *= (1 + cv) - return w - - -@inline -def eqtree(point_ptr, out, n_coords: Const): - # The eq tensor of the n_coords challenges at point_ptr[0..n_coords), built by - # doubling into out (size 2^(n_coords+1) - 2); the final 2^n_coords values start - # at offset 2^n_coords - 2. - r0 = point_ptr[GEN ** 0] - out[GEN ** 0] = 1 + r0 - out[GEN ** 1] = r0 - for t in unroll(1, n_coords): - rt = point_ptr[GEN ** t] - one_plus_rt = 1 + rt - for i in unroll(0, 2 ** t): - pw = out[GEN ** (2 ** t - 2 + i)] - out[GEN ** (2 ** (t + 1) - 2 + i)] = pw * one_plus_rt - out[GEN ** (2 ** (t + 1) - 2 + 2 ** t + i)] = pw * rt - return - - -@inline -def lag64(z, out, node_base: Const): - # The 64 phi8-domain Lagrange NUMERATORS at z over nodes - # PHI8_NODES[node_base .. node_base + 64]: out[i] = prod_{j != i} (z + - # PHI8_NODES[node_base + j]). Every barycentric denominator over an aligned phi8 - # window is the same element, so callers scale the finished sum once by - # LAGRANGE_INV_S / LAGRANGE_INV_COMBINED instead of the numerators one by one. - pre = StackBuf(65) - pre[0] = 1 - for i in unroll(0, 64): - pre[i + 1] = pre[i] * (z + PHI8_NODES[node_base + i]) - suf = StackBuf(65) - suf[64] = 1 - for i in unroll(0, 64): - suf[63 - i] = suf[64 - i] * (z + PHI8_NODES[node_base + 63 - i]) - for i in unroll(0, 64): - out[i] = pre[i] * suf[i + 1] - return - - -def eq_prefix_chain(chain, seed, a, b, count_g): - # Prefix products of eq(a_k, b_k) = 1 + a_k + b_k from `seed`, so a reader picks - # the partial product up at its own certified length. Entry t is written from - # inputs with index < t only, so a garbage tail past a buffer's written extent - # cannot corrupt any shorter prefix. - chain[GEN ** 0] = seed - for xk in mul_range(1, count_g): - chain[xk * GEN] = chain[xk] * (1 + a[xk] + b[xk]) - return - - -def rs_eq_run(chain, z_vals, point, count_g): - # One run of the telescoped ring-switch product E = sum_k c_k * prod_j - # (z_j^(2^k) + 1 + ris_j): coordinate x multiplies row k by (z^(2^k) + 1 + - # point_x), z evolving by squaring per row. The runtime coordinates walk - # OUTSIDE and the fixed Frobenius powers inside, so nothing stores a z-power - # table. - for xk in mul_range(1, count_g): - zv = z_vals[xk] - one_plus = 1 + point[xk] - row = chain * xk ** BASE_FIELD_BITS - nxt = row * GEN ** BASE_FIELD_BITS - for k in unroll(0, BASE_FIELD_BITS): - nxt[GEN ** k] = row[GEN ** k] * (zv + one_plus) - if k != BASE_FIELD_BITS - 1: - zv *= zv - return - - -def fold_final_msg(msg, point, log_len: Const): - weights = StackBuf(2 * YR_LOG_CAP) - for j in unroll(0, log_len): - weights[2 * j] = 1 + point[GEN ** j] - weights[2 * j + 1] = point[GEN ** j] - # Weighted fold of the final_msg multilinear over 2^log_len values (log_len is - # the candidate's yr_log_n; the frame buffers use the global max size). - l0 = StackBuf(2 ** YR_LOG_CAP) - for t in unroll(0, 2 ** log_len // 2): - l0[t] = weights[0] * msg[GEN ** (2 * t)] + weights[1] * msg[GEN ** (2 * t + 1)] - cursor = l0 - n = 2 ** log_len // 2 - for j in unroll(1, log_len): - nxt = StackBuf(2 ** YR_LOG_CAP) - for t in unroll(0, n // 2): - nxt[t] = weights[2 * j] * cursor[2 * t] + weights[2 * j + 1] * cursor[2 * t + 1] - cursor = nxt - n = n // 2 - return cursor[0] - - -def sumcheck_round4(state_0, state_1, msg_cursor, claim): - # One PLAIN sumcheck round. The prover sends the round polynomial's coefficients - # bar the one the split h(0) + h(1) == claim fixes, so the verifier derives that - # one and reads h at the challenge by Horner. Nothing is reapplied: no eq - # factor, no separate term for the tables still waiting, and the eq point is not - # read here at all. - fs = [state_0, state_1] - fs, c0, msg_cursor = fs_next(fs, msg_cursor) - fs, c2, msg_cursor = fs_next(fs, msg_cursor) - fs, c3, msg_cursor = fs_next(fs, msg_cursor) - c1 = claim + c2 + c3 # the split identity fixes it, so it is neither sent nor bound - fs, y = squeeze(fs) - return fs[0], fs[1], msg_cursor, c0 + y * (c1 + y * (c2 + y * c3)), y - - -def sumcheck_round5(state_0, state_1, msg_cursor, claim, prev_challenge): - # One GKR round. The prover sends every coefficient but c0, which the round's - # pulled-out eq factor leaves fixed: `c0 + prev_challenge * (c1 + ... + c4) == - # claim`. - fs = [state_0, state_1] - fs, c1, msg_cursor = fs_next(fs, msg_cursor) - fs, c2, msg_cursor = fs_next(fs, msg_cursor) - fs, c3, msg_cursor = fs_next(fs, msg_cursor) - fs, c4, msg_cursor = fs_next(fs, msg_cursor) - fs, y = squeeze(fs) - c0 = claim + prev_challenge * (c1 + c2 + c3 + c4) - return fs[0], fs[1], msg_cursor, c0 + y * (c1 + y * (c2 + y * (c3 + y * c4))), y - - -def batch_sumcheck(fs0, fs1, msgs, running, point, n_rounds: Const): - # The rounds of a claim-batching sumcheck: two hinted values per round - # (g(1) and g(inf)), the split fixing the third against the running claim, and - # the challenges collected into `point`. - fs = [fs0, fs1] - for rd in unroll(0, n_rounds): - fs, msg_g1, c = fs_next(fs, msgs * GEN ** (2 * rd)) - fs, msg_ginf, c = fs_next(fs, c) - fs, rv = squeeze(fs) - point[GEN ** rd] = rv - g_zero = running + msg_g1 - c_one = g_zero + msg_g1 + msg_ginf - running = (msg_ginf * rv + c_one) * rv + g_zero # fold the degree-2 round at rv - return fs[0], fs[1], running - - -# ==================================== Merkle paths ================================== - - -@inline -def order_children(node, sibling, bit): - # Branchless child ordering: `bit` is boolean-pinned wherever it comes from, so - # `m = bit*(node + sibling)` selects rather than branches, leaving (node, - # sibling) at bit 0 and (sibling, node) at bit 1. - m = bit * (node + sibling) - kids = [node + m, sibling + m] - return kids - - -@inline -def verify_merkle_path(leaf_0, leaf_1, direction_bits, depth: Const): - # Hinted child pairs are hashed in order; the boolean query bit selects the - # child that must equal the running node, binding each link of the path. - path = StackBuf(LIG_PATH_CAP) - hint_witness(path[0:4 * depth], "merkle_children") - path_ptr = addr(path) - node_0 = leaf_0 - node_1 = leaf_1 - for level in unroll(0, depth): - dir_bit = direction_bits[GEN ** level] - selected = path_ptr * GEN ** (4 * level) * (1 + dir_bit * (1 + GEN ** 2)) - selected[1] = node_0 - selected[GEN] = node_1 - parent = StackBuf(2) - blake2s(path[4 * level:4 * level + 2], path[4 * level + 2:4 * level + 4], parent) - node_0 = parent[0] - node_1 = parent[1] - return node_0, node_1 - - -@inline -def hash_cap_node(cap, index): - children = cap * index ** 4 - parent = StackBuf(2) - blake2s([children[1], children[GEN]], [children[GEN ** 2], children[GEN ** 3]], parent) - cap[index * index] = parent[0] - cap[GEN * index * index] = parent[1] - return - - -def verify_merkle_cap(cap, flags, depth: Const): - if depth != 0: - hash_cap_node(cap, GEN) - flags[GEN] = 1 - for parent in mul_range(GEN, GEN ** (2 ** (depth - 1))): - for side in unroll(0, 2): - child = parent * parent * GEN ** side - if flags[child] != 0: - flags[parent] = 1 # every active node forces its parent to be hashed - hash_cap_node(cap, child) - return cap[GEN ** 2], cap[GEN ** 3] - - -# ============================== the stacked WHIR opening ============================ - - -def opening_fold(fs0, fs1, cursor, c0, c1, c2): - fs = [fs0, fs1] - fs, r = squeeze(fs) - claim = (c2 * r + c1) * r + c0 - fs, c0, cursor = fs_next(fs, cursor) - fs, c2, cursor = fs_next(fs, cursor) - return fs[0], fs[1], cursor, claim, c0, c2, r - - -def opening_ood_point(fs0, fs1, point, n_g): - # Share the squeeze loop across point dimensions. - states = HeapBuf((n_g * GEN) ** 2) - states[1] = fs0 - states[GEN] = fs1 - for x in mul_range(1, n_g): - state = states * x * x - r, f0, f1 = squeeze_step(state[1], state[GEN]) - point[x] = r - state[GEN ** 2] = f0 - state[GEN ** 3] = f1 - last = states * n_g * n_g - return last[1], last[GEN] - - -def opening_final_message(fs0, fs1, cursor, out, n: Const): - fs = [fs0, fs1] - for i in unroll(0, n): - fs, value, cursor = fs_next(fs, cursor) - out[GEN ** i] = value - return fs[0], fs[1], cursor - - -def opening_row_weights(point, out, folds: Const, reverse: Const): - for i in unroll(0, 2 ** folds): - if reverse == 1: - slot = 2 ** folds - 1 - i - else: - slot = i - out[GEN ** i] = eq_weight(point, folds, slot, 0) - return - - -def opening_queries(cap, flags, query_weights, query_bit_ptrs, row_eq_weights, n_queries_g, base: Const, interleave: Const, blocks: Const, depth: Const, cap_depth: Const): - # Specialize by row and path shape so opening configurations share query code. - query_sum_chain = HeapBuf(n_queries_g * GEN) - query_sum_chain[GEN ** 0] = 0 - for xe in mul_range(1, n_queries_g): - if base == 1: - row_len = interleave - else: - row_len = 3 * interleave - row = StackBuf(LIG_ROW_CAP) - hint_witness(row[0:row_len], "merkle_leaf_rows") - row_dot = 0 - packed_row = StackBuf(LIG_PACKED_ROW_CAP) - if base == 1: - # Packing proves each hinted lane is in K before hashing or folding it. - for jb in unroll(0, interleave // 4): - e0 = row[4 * jb] - e1 = row[4 * jb + 1] - e2 = row[4 * jb + 2] - e3 = row[4 * jb + 3] - packed_row[2 * jb] = pack64x2(e0, e1) - packed_row[2 * jb + 1] = pack64x2(e2, e3) - row_dot += e0 * row_eq_weights[GEN ** (4 * jb)] + e1 * row_eq_weights[GEN ** (4 * jb + 1)] + e2 * row_eq_weights[GEN ** (4 * jb + 2)] + e3 * row_eq_weights[GEN ** (4 * jb + 3)] - else: - # Pack the checked tower limbs into the leaf's contiguous byte image. - for jb in unroll(0, 3 * interleave // 4): - packed_row[2 * jb] = pack64x2(row[4 * jb], row[4 * jb + 1]) - packed_row[2 * jb + 1] = pack64x2(row[4 * jb + 2], row[4 * jb + 3]) - for jw in unroll(0, interleave): - if 3 * jw % 2 == 0: - # limbs (3w, 3w+1) are a pack; add Y^2 * limb(3w+2). - row_word = packed_row[3 * jw // 2] + Y_TOWER * Y_TOWER * row[3 * jw + 2] - else: - # limbs (3w+1, 3w+2) are a pack; shift it by Y and add limb(3w). - row_word = row[3 * jw] + Y_TOWER * packed_row[(3 * jw + 1) // 2] - row_dot += row_word * row_eq_weights[GEN ** jw] - # Hash the packed row as full BLAKE2s blocks. - leaf_hash_state = StackBuf(2) - blake2s(packed_row[0:2], packed_row[2:4], leaf_hash_state, counter=64, final=1 // blocks) - for jb in unroll(1, blocks): - leaf_digest = StackBuf(2) - blake2s(packed_row[4 * jb:4 * jb + 2], packed_row[4 * jb + 2:4 * jb + 4], leaf_digest, cv=leaf_hash_state, counter=64 * (jb + 1), final=(jb + 1) // blocks) - leaf_hash_state = leaf_digest - query_sum_chain[xe * GEN] = query_sum_chain[xe] + query_weights[xe] * row_dot - direction_bits = query_bit_ptrs[xe] - path_depth = depth - cap_depth - node_0, node_1 = verify_merkle_path(leaf_hash_state[0], leaf_hash_state[1], direction_bits, path_depth) - if cap_depth != 0: - parent = GEN ** (2 ** (cap_depth - 1)) - for bit in unroll(0, cap_depth - 1): - parent *= 1 + direction_bits[GEN ** (path_depth + 1 + bit)] * (1 + GEN ** (2 ** bit)) - flags[parent] = 1 # the cap check propagates this obligation to the root - cap_index = parent * parent * (1 + direction_bits[GEN ** path_depth] * (1 + GEN)) - else: - cap_index = GEN - cap[cap_index * cap_index] = node_0 - cap[GEN * cap_index * cap_index] = node_1 - return query_sum_chain[n_queries_g] - - -def open_stacked(m_idx: Const, fs0, fs1, target, commit_root_0, commit_root_1, cursor): - # The stacked WHIR opening, one specialization per (rate, committed log-size) - # candidate: every LIG_* table reads row m_idx, per level row `ml`, and all - # opening proof data is hinted here, so only the executed arm pops its streams. - # - # The returned point is in witness order. Round-order challenges stay local - # for the induced weights, OOD claims, and residual evaluation. - n_levels = LIG_N_LEVELS[m_idx] - yr_level = LIG_YR_LEVEL[m_idx] - yr_log = LIG_YR_LOG_LEN[m_idx] - yr_len = LIG_YR_LEN[m_idx] - n_folds = LIG_TOTAL_FOLDS[m_idx] - max_q = LIG_MAX_QUERIES[m_idx] - ood_stride = LIG_MAX_OOD_SAMPLES * OOD_SLOTS - - # The opening's scalars (sumcheck messages, level roots, nonces, final message) - # ride the SHARED stream, walked on in protocol order. The K opener binds a - # Merkle root as its two F192 scalars, not as a byte string, and every digest - # uses ONE 128/128 encoding, so those scalars are exactly the root's two cells. - fs = [fs0, fs1] - msg_cursor = cursor - fs, round_quad_c, msg_cursor = fs_next(fs, msg_cursor) # the round polynomial in coefficients - fs, round_quad_a, msg_cursor = fs_next(fs, msg_cursor) # bar the linear one, which the split fixes - round_quad_b = target + round_quad_a - sumcheck_target = target - - # Caps are shared across queries; rows and paths are hinted in each query's frame. - merkle_caps = HeapBuf(GEN ** (4 * LIG_CAP_LEN[m_idx])) - hint_witness(merkle_caps[0:4 * LIG_CAP_LEN[m_idx]], "merkle_caps") - cap_flags = HeapBuf(GEN ** LIG_CAP_LEN[m_idx]) - hint_witness(cap_flags[0:LIG_CAP_LEN[m_idx]], "merkle_cap_active") - final_msg = HeapBuf(GEN ** yr_len) # filled from the stream at the last level - # Each cap root is checked against its transcript-bound level root. - level_roots = HeapBuf(GEN ** (2 * n_levels)) - level_roots[GEN ** 0] = commit_root_0 - level_roots[GEN ** 1] = commit_root_1 - # ...and guest-filled accumulators (one slot per fold / per level / per query): - fold_challenges = HeapBuf(GEN ** n_folds) - level_betas = HeapBuf(GEN ** n_levels) - query_weights = HeapBuf(GEN ** (n_levels * max_q)) - query_positions = HeapBuf(GEN ** (LIG_POSITIONS_LEN[m_idx])) - query_bit_ptrs = HeapBuf(GEN ** (LIG_POSITIONS_LEN[m_idx])) - # Explicit OOD claims bind every recursive Johnson-list commitment. L0 needs - # none: the opening claim itself is its post-commit binding value. An OOD claim - # is read before its level's query positions but batched after them (one - # challenge per level), so its value and intro message wait here. - ood_z = HeapBuf(GEN ** (n_levels * LIG_MAX_OOD_SAMPLES * LIG_LOG_MSG_COLS_CAP)) - ood = HeapBuf(GEN ** (n_levels * ood_stride)) - - for lvl in unroll(0, n_levels): - ml = m_idx * LIG_MAX_LEVELS + lvl - n_queries = LIG_QUERIES[ml] - depth = LIG_TREE_DEPTH[ml] - interleave = LIG_INTERLEAVE[ml] - folds_off = LIG_FOLDS_OFF[ml] - pos_off = LIG_POSITIONS_OFF[ml] - for j in unroll(0, LIG_FOLDS[ml]): - f0, f1, msg_cursor, sumcheck_target, round_quad_c, round_quad_a, fold_challenge = opening_fold(fs[0], fs[1], msg_cursor, round_quad_c, round_quad_b, round_quad_a) - fs = [f0, f1] - fold_challenges[GEN ** (folds_off + j)] = fold_challenge - round_quad_b = sumcheck_target + round_quad_a - - if lvl == yr_level: - f0, f1, msg_cursor = opening_final_message(fs[0], fs[1], msg_cursor, final_msg, yr_len) - fs = [f0, f1] - else: - fs, next_root_a, msg_cursor = fs_next(fs, msg_cursor) - fs, next_root_b, msg_cursor = fs_next(fs, msg_cursor) - # A non-canonical half is rejected (merkle.rs `scalars_to_hash`), as - # the commitment root was at its own read. - canon = StackBuf(2) - canon[0] = assert_canonical(next_root_a) - canon[1] = assert_canonical(next_root_b) - level_roots[GEN ** (2 * lvl + 2)] = next_root_a - level_roots[GEN ** (2 * lvl + 3)] = next_root_b - # OOD binding for the newly observed level-(lvl+1) commitment. The - # random point has the just-folded witness dimension, namely this - # level's message-column dimension. - for os in unroll(0, LIG_OOD_SAMPLES[ml + 1]): - oz = ood_z * GEN ** (((lvl + 1) * LIG_MAX_OOD_SAMPLES + os) * LIG_LOG_MSG_COLS_CAP) - f0, f1 = opening_ood_point(fs[0], fs[1], oz, GEN ** LIG_LOG_MSG_COLS[ml]) - fs = [f0, f1] - sample = ood * GEN ** ((lvl + 1) * ood_stride + os * OOD_SLOTS) - fs, ood_y, msg_cursor = fs_next(fs, msg_cursor) - fs, ood_c0, msg_cursor = fs_next(fs, msg_cursor) - fs, ood_c2, msg_cursor = fs_next(fs, msg_cursor) - sample[GEN ** OOD_Y] = ood_y - sample[GEN ** OOD_C0] = ood_c0 - sample[GEN ** OOD_C2] = ood_c2 # the split fixes c1 = y + c2 - q_nonce = msg_cursor[GEN ** 0] # raw transport word: bound by the DS_POW_NONCE absorb below - msg_cursor = msg_cursor * GEN - if LIG_QUERY_GRIND_BITS[ml] != 0: - grind_check(fs[0], fs[1], q_nonce, GEN ** LIG_QUERY_GRIND_BITS[ml]) - else: - assert q_nonce == 0 - fs = absorb_nonce(fs, q_nonce) - - sqz = HeapBuf((GEN ** (LIG_MAX_SQUEEZES[m_idx] + 1)) ** PAIR_SLOTS) - sqz[GEN ** 0] = fs[0] - sqz[GEN ** 1] = fs[1] - for xs in mul_range(1, GEN ** LIG_SQUEEZES[ml]): - # a loop body captures free names BY VALUE, so the compile-time aliases - # are rebound here (m_idx and lvl are substituted literals) - depth = LIG_TREE_DEPTH[m_idx * LIG_MAX_LEVELS + lvl] - pos_off = LIG_POSITIONS_OFF[m_idx * LIG_MAX_LEVELS + lvl] - row = sqz * xs ** PAIR_SLOTS - packed_word, next_c0, next_c1 = squeeze_step(row[GEN ** 0], row[GEN ** 1]) - row[GEN ** PAIR_SLOTS] = next_c0 - row[GEN ** (PAIR_SLOTS + 1)] = next_c1 - query_ptr = xs ** (FIELD_BITS // depth) - decode_query_bits(packed_word, query_positions * GEN ** pos_off * query_ptr, query_bit_ptrs * GEN ** pos_off * query_ptr, depth) - sqz_end = sqz * (GEN ** LIG_SQUEEZES[ml]) ** PAIR_SLOTS - fs = [sqz_end[GEN ** 0], sqz_end[GEN ** 1]] - - # One batching challenge for the level, drawn once every claim it batches is - # fixed: its OOD claims above and these query positions. Claim tau of the - # level is weighted lam^tau, the running claim keeping lam^0 (Annex B, - # Protocol 1 step 1): query i is claim n_ood + 1 + i, so its weight splits - # into lam^i here and the level scalar lam^(n_ood+1) below. - fs, lam = squeeze(fs) - lam_pow = 1 - for i in unroll(0, n_queries): - query_weights[GEN ** (lvl * max_q + i)] = lam_pow - lam_pow = lam_pow * lam - # At level 0, slot i of a leaf image is interleaving index n-1-i: the image - # reads its lanes from the top down, so the lanes a padding-free commitment - # leaves out are its LEADING words, whose whole blocks the committer hashes - # once for all leaves. The flip is a compile-time index and the guest still - # hashes the full image. Deeper levels commit every lane, ascending. - row_eq_weights = HeapBuf(GEN ** (LIG_MAX_INTERLEAVE[m_idx])) - opening_row_weights(fold_challenges * GEN ** folds_off, row_eq_weights, LIG_FOLDS[ml], 1 // (lvl + 1)) - - cap_depth = LIG_CAP_DEPTH[ml] - cap = merkle_caps * GEN ** (4 * LIG_CAP_OFF[ml]) - flags = cap_flags * GEN ** LIG_CAP_OFF[ml] - root_0, root_1 = verify_merkle_cap(cap, flags, cap_depth) - level_roots[GEN ** (2 * lvl)] = root_0 - level_roots[GEN ** (2 * lvl + 1)] = root_1 - - level_query_sum = opening_queries(cap, flags, query_weights * GEN ** (lvl * max_q), query_bit_ptrs * GEN ** pos_off, row_eq_weights, GEN ** n_queries, 1 // (lvl + 1), interleave, LIG_LEAF_BLOCKS[ml], depth, cap_depth) - - # Every level, including the last, ties its commitment in through an intro - # message. The level's claims then enter the running one with powers of - # `lam`: the OOD claims held above first, then this query batch. - fs, intro_c0, msg_cursor = fs_next(fs, msg_cursor) - fs, intro_c2, msg_cursor = fs_next(fs, msg_cursor) - intro_c1 = level_query_sum + intro_c2 # the split fixes the linear coefficient - if lvl == yr_level: - beta_lvl = lam # no OOD claim at the last level: no new oracle - else: - ood_scalar = lam - for os in unroll(0, LIG_OOD_SAMPLES[ml + 1]): - sample = ood * GEN ** ((lvl + 1) * ood_stride + os * OOD_SLOTS) - ood_y = sample[GEN ** OOD_Y] - ood_c2 = sample[GEN ** OOD_C2] - sample[GEN ** OOD_BETA] = ood_scalar - round_quad_c += ood_scalar * sample[GEN ** OOD_C0] - round_quad_b += ood_scalar * (ood_y + ood_c2) - round_quad_a += ood_scalar * ood_c2 - sumcheck_target += ood_scalar * ood_y - ood_scalar = ood_scalar * lam - beta_lvl = ood_scalar - level_betas[GEN ** lvl] = beta_lvl - round_quad_c += beta_lvl * intro_c0 - round_quad_b += beta_lvl * intro_c1 - round_quad_a += beta_lvl * intro_c2 - sumcheck_target += beta_lvl * level_query_sum - - # ---- finish the sumcheck over the tail coordinates ---- - tail_challenges = HeapBuf(GEN ** YR_LOG_CAP) - for j in unroll(0, yr_log - 1): - f0, f1, msg_cursor, sumcheck_target, round_quad_c, round_quad_a, tail_c = opening_fold(fs[0], fs[1], msg_cursor, round_quad_c, round_quad_b, round_quad_a) - fs = [f0, f1] - tail_challenges[GEN ** j] = tail_c - round_quad_b = sumcheck_target + round_quad_a - # The closing round sends no following message. - fs, tail_last = squeeze(fs) - tail_challenges[GEN ** (yr_log - 1)] = tail_last - sumcheck_target = round_quad_c + tail_last * round_quad_b + tail_last * tail_last * round_quad_a - for j in unroll(yr_log, YR_LOG_CAP): - tail_challenges[GEN ** j] = 0 - - yr_at_tail = fold_final_msg(final_msg, tail_challenges, yr_log) - - # ---- the same point, indexed by committed-witness coordinate ---- - # The folds bind coordinates in ROUND order, and level 0's folds are the lane - # fold, binding the witness's TOP k coordinates: lane l of the commitment is the - # stack block q[l * 2^(mu-k) ...], which is what makes the witness's zero - # padding whole lanes for the committer to leave out of the encode. Every - # transparent weight downstream is written in witness coordinates, so rotate the - # point left by those k rounds here, while the level shape is still - # compile-time. Padding is zero so shared prefix chains may compute unused - # entries above m without reading uninitialized cells. - lane_folds = LIG_FOLDS[m_idx * LIG_MAX_LEVELS] - fold_head = n_folds - lane_folds - point = HeapBuf(SIZE_BITS + SLOT_STRIDE_LOG) - for j in unroll(0, fold_head): - point[GEN ** j] = fold_challenges[GEN ** (lane_folds + j)] - for j in unroll(0, yr_log): - point[GEN ** (fold_head + j)] = tail_challenges[GEN ** j] - for j in unroll(0, lane_folds): - point[GEN ** (fold_head + yr_log + j)] = fold_challenges[GEN ** j] - for j in unroll(n_folds + yr_log, SIZE_BITS + SLOT_STRIDE_LOG): - point[GEN ** j] = 0 - - # ---- per-level induced bases at the single terminal point ---- - # Every query of a level runs the SAME product shape over its message-column - # coordinates; only the novel-basis chain (the query position's - # subspace-vanishing walk) differs. Since - # 1 + c_t * (1 + chain_t * inv_t) == (1 + c_t) + (c_t * inv_t) * chain_t - # a coordinate's two coefficients depend on the challenge and the baked - # vanishing inverse alone, so they hoist out of the query loop (one row a level, - # fold coords then tail coords) and each query is one multiply-add a - # coordinate. - basis_a = HeapBuf(GEN ** (n_levels * LIG_LOG_MSG_COLS_CAP)) - basis_b = HeapBuf(GEN ** (n_levels * LIG_LOG_MSG_COLS_CAP)) - for lvl in unroll(0, n_levels): - ml = m_idx * LIG_MAX_LEVELS + lvl - prefix_len = LIG_RESIDUAL_PREFIX_LEN[ml] - vanish = m_idx * LIG_MAX_VANISH_LEN + LIG_VANISH_OFF[ml] - for t in unroll(0, prefix_len): - fold_c = fold_challenges[GEN ** (LIG_RESIDUAL_FOLD_OFF[ml] + t)] - basis_a[GEN ** (lvl * LIG_LOG_MSG_COLS_CAP + t)] = 1 + fold_c - basis_b[GEN ** (lvl * LIG_LOG_MSG_COLS_CAP + t)] = fold_c * LIG_VANISH_INVS[vanish + t] - for j in unroll(0, yr_log): - tail_c = tail_challenges[GEN ** j] - basis_a[GEN ** (lvl * LIG_LOG_MSG_COLS_CAP + prefix_len + j)] = 1 + tail_c - basis_b[GEN ** (lvl * LIG_LOG_MSG_COLS_CAP + prefix_len + j)] = tail_c * LIG_VANISH_INVS[vanish + prefix_len + j] - inner_chain = HeapBuf(GEN ** (n_levels + 1)) - inner_chain[GEN ** 0] = 0 - for lvl in unroll(0, n_levels): - ml = m_idx * LIG_MAX_LEVELS + lvl - vanish = m_idx * LIG_MAX_VANISH_LEN + LIG_VANISH_OFF[ml] - basis_row = lvl * LIG_LOG_MSG_COLS_CAP - residual_chain = HeapBuf(GEN ** (max_q + 1)) - residual_chain[GEN ** 0] = 0 - for xr in mul_range(1, GEN ** LIG_QUERIES[ml]): - ml = m_idx * LIG_MAX_LEVELS + lvl # rebound: the body captures by value - vanish = m_idx * LIG_MAX_VANISH_LEN + LIG_VANISH_OFF[ml] - basis_row = lvl * LIG_LOG_MSG_COLS_CAP - max_q = LIG_MAX_QUERIES[m_idx] - basis_chain = query_positions[GEN ** LIG_POSITIONS_OFF[ml] * xr] - prefix_eq = basis_a[GEN ** basis_row] + basis_b[GEN ** basis_row] * basis_chain - for t in unroll(1, LIG_LOG_MSG_COLS[ml]): - # subspace-vanishing recurrence for the novel-basis point - basis_chain *= (basis_chain + LIG_VANISH_VALS[vanish + t - 1]) - prefix_eq *= basis_a[GEN ** (basis_row + t)] + basis_b[GEN ** (basis_row + t)] * basis_chain - residual_chain[xr * GEN] = residual_chain[xr] + query_weights[GEN ** (lvl * max_q)* xr] * prefix_eq - # accumulate beta_lvl * (per-level residual sum) into the grand residual - inner_chain[GEN ** (lvl + 1)] = inner_chain[GEN ** lvl] + level_betas[GEN ** lvl] * residual_chain[GEN ** LIG_QUERIES[ml]] - - # Explicit OOD eq bases at the same terminal point. - ood_inner = 0 - for ood_lvl in unroll(1, n_levels): - ml = m_idx * LIG_MAX_LEVELS + ood_lvl - z_folded = LIG_LOG_MSG_COLS[ml - 1] - yr_log - ris_start = LIG_FOLDS_OFF[ml] - for os in unroll(0, LIG_OOD_SAMPLES[ml]): - oz = ood_z * GEN ** ((ood_lvl * LIG_MAX_OOD_SAMPLES + os) * LIG_LOG_MSG_COLS_CAP) - scalar = ood[GEN ** (ood_lvl * ood_stride + os * OOD_SLOTS + OOD_BETA)] - for t in unroll(0, z_folded): - scalar *= (1 + oz[GEN ** t] + fold_challenges[GEN ** (ris_start + t)]) - for t in unroll(0, yr_log): - scalar *= (1 + oz[GEN ** (z_folded + t)] + tail_challenges[GEN ** t]) - ood_inner += scalar - return sumcheck_target, point, inner_chain[GEN ** n_levels] + ood_inner, yr_at_tail - - -# ============================== inner-proof verification ============================ -# The phases of `verify_sub`, in the order it runs them. Each takes and returns the -# Fiat-Shamir state pair and the stream cursor, so the sequence is what binds them. - - -def verify_bus_gkr(fs0, fs1, cursor, g_bus_mu, zeta): - # ONE GKR grand product over push, pull and count, RLC-batched. Push and pull - # have equal depth (matched blocks) and the count tree is padded with identity - # leaves up to it (product unchanged), so a single sumcheck serves all three. - # Radix four contracts two binary levels per layer; after checking the combined - # product identity, a fresh λ pins the individual values. All three trees reduce - # to the one shared point `zeta`, which this fills, returning their three leaf - # values with the walked Fiat-Shamir state and stream cursor. - layers = HeapBuf((g_bus_mu * GEN ** 2) ** LAYER_SLOTS) # mu + 2 layers - rounds = HeapBuf(GKR_ROUNDS_CAP * ROUND_SLOTS) - gkr_pts = HeapBuf(GKR_POINTS_CAP) - assert log(g_bus_mu) < COUNT_BITS - fs = [fs0, fs1] - fs, root_push, cursor = fs_next(fs, cursor) - root_pull = root_push - fs, root_count, cursor = fs_next(fs, cursor) - assert root_count != 0 # count-tree root nonzero: no read count self-cancels - fs, root_lambda = squeeze(fs) - layers[GEN ** LAYER_FS0] = fs[0] - layers[GEN ** LAYER_FS1] = fs[1] - layers[GEN ** LAYER_CURSOR] = cursor - layers[GEN ** LAYER_PUSH] = root_push - layers[GEN ** LAYER_PULL] = root_pull - layers[GEN ** LAYER_COUNT] = root_count - layers[GEN ** LAYER_LAMBDA] = root_lambda # λ over the three roots - layers[GEN ** LAYER_ROW] = gkr_pts - layers[GEN ** LAYER_POS] = GEN ** 0 - - # Contract two binary product levels at a time. pair_bounds[g^d] = g^(d//2) is - # the radix-four layer count, and shift is g exactly when the depth is odd, in - # which case the root-most BINARY layer runs first. - pair_bounds = HeapBuf(COUNT_BITS) - depth_shift = HeapBuf(COUNT_BITS) - for depth in unroll(0, COUNT_BITS): - pair_bounds[GEN ** depth] = GEN ** (depth // 2) - depth_shift[GEN ** depth] = GEN ** (depth % 2) - - if depth_shift[g_bus_mu] != 1: - # The odd layer is layer 0, so its round state would be written and read - # back at the same position: read it straight off the layer instead. - lam = layers[GEN ** LAYER_LAMBDA] - tail_fs = [layers[GEN ** LAYER_FS0], layers[GEN ** LAYER_FS1]] - tcur = layers[GEN ** LAYER_CURSOR] - tclaim = layers[GEN ** LAYER_PUSH] + lam * (layers[GEN ** LAYER_PULL] + lam * layers[GEN ** LAYER_COUNT]) - nextrow = layers[GEN ** LAYER_ROW] * GEN ** MU_CAP - evals = StackBuf(2 * N_GKR_SIDES) # the two children of each side, in side order - for i in unroll(0, 2 * N_GKR_SIDES): - tail_fs, ev, tcur = fs_next(tail_fs, tcur) - evals[i] = ev - combined = 0 - for i in unroll(0, N_GKR_SIDES): - side = N_GKR_SIDES - 1 - i # Horner in lam, so the top side lands last - combined = evals[2 * side] * evals[2 * side + 1] + lam * combined - assert tclaim == combined - tail_fs, c0 = squeeze(tail_fs) - nextrow[GEN ** 0] = c0 - tail_fs, tail_lambda = squeeze(tail_fs) # fresh λ pins the tail individuals - nxt = layers * GEN ** LAYER_SLOTS - nxt[GEN ** LAYER_FS0] = tail_fs[0] - nxt[GEN ** LAYER_FS1] = tail_fs[1] - nxt[GEN ** LAYER_CURSOR] = tcur - for side in unroll(0, N_GKR_SIDES): - nxt[GEN ** (LAYER_PUSH + side)] = evals[2 * side] + c0 * (evals[2 * side] + evals[2 * side + 1]) - nxt[GEN ** LAYER_LAMBDA] = tail_lambda - nxt[GEN ** LAYER_ROW] = nextrow - nxt[GEN ** LAYER_POS] = GEN - - for x_pair in mul_range(1, pair_bounds[g_bus_mu]): - x_layer = x_pair * x_pair * depth_shift[g_bus_mu] - layer = layers * x_layer ** LAYER_SLOTS - lam = layer[GEN ** LAYER_LAMBDA] - point_row = layer[GEN ** LAYER_ROW] - round_pos = layer[GEN ** LAYER_POS] - nextrow = point_row * GEN ** MU_CAP - head = rounds * round_pos ** ROUND_SLOTS - head[GEN ** ROUND_FS0] = layer[GEN ** LAYER_FS0] - head[GEN ** ROUND_FS1] = layer[GEN ** LAYER_FS1] - head[GEN ** ROUND_CURSOR] = layer[GEN ** LAYER_CURSOR] - head[GEN ** ROUND_CLAIM] = layer[GEN ** LAYER_PUSH] + lam * (layer[GEN ** LAYER_PULL] + lam * layer[GEN ** LAYER_COUNT]) - for x_round in mul_range(1, x_layer): - rd = rounds * (round_pos * x_round) ** ROUND_SLOTS - nfs0, nfs1, ncur, nclaim, rk = sumcheck_round5(rd[GEN ** ROUND_FS0], rd[GEN ** ROUND_FS1], rd[GEN ** ROUND_CURSOR], rd[GEN ** ROUND_CLAIM], point_row[x_round]) - nextrow[x_round * GEN ** 2] = rk - rd_next = rd * GEN ** ROUND_SLOTS - rd_next[GEN ** ROUND_FS0] = nfs0 - rd_next[GEN ** ROUND_FS1] = nfs1 - rd_next[GEN ** ROUND_CURSOR] = ncur - rd_next[GEN ** ROUND_CLAIM] = nclaim - tail = rounds * (round_pos * x_layer) ** ROUND_SLOTS - tail_fs = [tail[GEN ** ROUND_FS0], tail[GEN ** ROUND_FS1]] - tcur = tail[GEN ** ROUND_CURSOR] - tclaim = tail[GEN ** ROUND_CLAIM] - evals = StackBuf(4 * N_GKR_SIDES) # the four children of each side, in side order - for i in unroll(0, 4 * N_GKR_SIDES): - tail_fs, ev, tcur = fs_next(tail_fs, tcur) - evals[i] = ev - combined = 0 - for i in unroll(0, N_GKR_SIDES): - side = N_GKR_SIDES - 1 - i # Horner in lam, so the top side lands last - combined = evals[4 * side] * evals[4 * side + 1] * evals[4 * side + 2] * evals[4 * side + 3] + lam * combined - assert tclaim == combined - tail_fs, c0 = squeeze(tail_fs) - tail_fs, c1 = squeeze(tail_fs) - nextrow[GEN ** 0] = c0 - nextrow[GEN ** 1] = c1 - tail_fs, tail_lambda = squeeze(tail_fs) - nxt = layer * GEN ** (2 * LAYER_SLOTS) - nxt[GEN ** LAYER_FS0] = tail_fs[0] - nxt[GEN ** LAYER_FS1] = tail_fs[1] - nxt[GEN ** LAYER_CURSOR] = tcur - for side in unroll(0, N_GKR_SIDES): - lo = evals[4 * side] + c0 * (evals[4 * side] + evals[4 * side + 1]) - hi = evals[4 * side + 2] + c0 * (evals[4 * side + 2] + evals[4 * side + 3]) - nxt[GEN ** (LAYER_PUSH + side)] = lo + c1 * (lo + hi) - nxt[GEN ** LAYER_LAMBDA] = tail_lambda - nxt[GEN ** LAYER_ROW] = nextrow - nxt[GEN ** LAYER_POS] = round_pos * x_layer * GEN - last = layers * g_bus_mu ** LAYER_SLOTS - fs = [last[GEN ** LAYER_FS0], last[GEN ** LAYER_FS1]] - cursor = last[GEN ** LAYER_CURSOR] - final_point_row = last[GEN ** LAYER_ROW] - for xt in mul_range(1, g_bus_mu): - zeta[xt] = final_point_row[xt] # the ONE shared point - return fs[0], fs[1], cursor, last[GEN ** LAYER_PUSH], last[GEN ** LAYER_PULL], last[GEN ** LAYER_COUNT] - - -def verify_tables(fs0, fs1, cursor, pi_0, pi_1, zeta, g_bus_mu, dims_g, block_kappa, g_squares, fp_w, beta, claim_pool, claim_cplen_g, chi, claim_push, claim_pull, claim_count): - # Settle the bus against the six tables, in four steps: certify each side's - # leaf-cube tiling, decompose the three GKR leaf values over it, run the ONE - # table sumcheck they all reduce to, and bind the public input. Pooled claims - # land in claim_pool/claim_cplen_g and the reduced point in `chi`; the batch's - # round count g^n, the deferred bytecode share and the PI point come back. - # - # Bus-leaf packing offsets, for the selector certification. Each side's blocks - # tile its leaf cube, block b at offset_b; the hinted order is only - # PERMUTATION-checked and offsets accumulate as g^offset = Π_{earlier} g^(2^κ). - # The decompose below pins each block's selector bits against this offset, - # forcing κ-alignment, and no sort or tie-break check is needed: alignment plus - # consecutive offsets force a valid tiling, and the grand product is - # position-independent, so any tiling is sound. Pull's blocks mirror push's and - # share zeta, so only push and count need offsets (pull's slots go unread). - fs = [fs0, fs1] - gkr_claims = StackBuf(N_GKR_SIDES) - gkr_claims[PUSH_SIDE] = claim_push - gkr_claims[PULL_SIDE] = claim_pull - gkr_claims[COUNT_SIDE] = claim_count - sort_order = HeapBuf(N_BLOCKS) - hint_witness(sort_order[0:N_BLOCKS], "sort_order") - block_side_tab = HeapBuf(N_BLOCKS) # global block -> its side - for b in unroll(0, N_BLOCKS): - block_side_tab[GEN ** b] = BLOCK_SIDE[b] - block_off_g = HeapBuf(N_BLOCKS) # g^offset per block, keyed by global index - for cert in unroll(0, 2): - s = COUNT_SIDE * cert # PUSH_SIDE (0), then COUNT_SIDE (2) - g_off = GEN ** 0 - for r in unroll(SIDE_BLOCK_START[s], SIDE_BLOCK_START[s + 1]): - global_g = sort_order[GEN ** r] # g^{global block index at this rank} - assert log(global_g) < N_BLOCKS # a valid block index - assert block_side_tab[global_g] == s # ...belonging to THIS side - # write-once: a repeat collides, and an omission fails the decompose's - # offset read below - block_off_g[global_g] = g_off - g_off *= g_squares[block_kappa[global_g]] - - # ---- 3x leaf decomposition (claims pooled; bytecode Public DEFERRED) ---- - # Reconstruct Ṽ₀(ζ) per side and assert it equals the GKR leaf value. The - # committed-coordinate values ride the stream (observed, pooled); Index - # coordinates use the factored index MLE; and the program's whole share of a - # bytecode leaf is ONE evaluation of the stacked polynomial, since its slots are - # aligned with the tuple and the weights are eq(α⃗, ·), so the share IS that - # polynomial at (ζ_lo, α⃗) (doc sec:e2e-bc): one hinted value, exported as a - # deferred claim, with no per-coordinate values and no selector challenge. - # - # Pull's blocks mirror push's (same kappas, same offsets, generator-asserted - # pairing) and share zeta, so each pull block REUSES its push twin's eq_hi and - # Index-MLE value instead of recomputing them; its column values are mostly - # deduped pool reads (COORD_FRESH). The identity check against pull's own GKR - # claim still binds everything. - bc_share = hint_witness("bytecode_val") - idxc_tab = HeapBuf(SIZE_BITS) # INDEX_MLE_FACTORS[t] = 1 + g^(2^t) - for t in unroll(0, SIZE_BITS): - idxc_tab[GEN ** t] = INDEX_MLE_FACTORS[t] - bus_table_total = StackBuf(N_GKR_SIDES) # per side, what its tables' blocks owe - block_eq_hi = StackBuf(N_BLOCKS) # every block's eq_hi, reused below - block_index_mle = HeapBuf(N_BLOCKS) # per push block with an Index coord - for s in unroll(0, N_GKR_SIDES): - acc = 0 - selector_sum = 0 - for b in unroll(SIDE_BLOCK_START[s], SIDE_BLOCK_START[s + 1]): - block_has_public = 0 - kappa_g = block_kappa[GEN ** b] - assert log(kappa_g) < SIZE_BITS - if s == PULL_SIDE: - eq_hi = block_eq_hi[b - SIDE_BLOCK_START[PULL_SIDE]] - else: - # eq_hi over the ζ coords above κ against the selector bits, whose - # run is mu_s − κ = g^mu_s / g^κ long. Selector bits = offset >> κ: - # advice-decompose the offset and read it shifted by κ. Rebuilding - # g^offset from those high bits alone (weights g^(2^(κ+k))) and - # asserting it equals block_off_g pins the bits AND the κ-alignment - # in one shot; the low κ bit cells are written but never read. - sel_len_g = g_bus_mu / kappa_g # g^(mu - κ) - assert log(sel_len_g) < SIZE_BITS - zeta_hi = zeta * kappa_g - offset_bits = HeapBuf(GEN ** SIZE_BITS) - hint_decompose_bits_exponent(offset_bits, block_off_g[GEN ** b], SIZE_BITS) - sel_bits = offset_bits * kappa_g - eq_chain = HeapBuf(MU_CAP + 2) - goff_chain = HeapBuf(MU_CAP + 2) # rebuild g^offset from the high bits - eq_chain[GEN ** 0] = 1 - goff_chain[GEN ** 0] = 1 - for xk in mul_range(1, sel_len_g): - sbit = sel_bits[xk] - sel_bits[xk] = sbit * sbit # booleanity as a write-once pin - eq_chain[xk * GEN] = eq_chain[xk] * (1 + sbit + zeta_hi[xk]) # eq over GF(2) is 1 + b + z - goff_chain[xk * GEN] = goff_chain[xk] * (1 + sbit * (g_squares[kappa_g * xk] + 1)) - eq_hi = eq_chain[sel_len_g] - assert goff_chain[sel_len_g] == block_off_g[GEN ** b] # bits == offset >> κ, κ-aligned - selector_sum += eq_hi - block_eq_hi[b] = eq_hi - # A TABLE's block streams no value here: the table sumcheck settles its - # fingerprint from that table's column evaluations. Only the framework - # blocks (boundary, memory, bytecode) still decompose. - if BLOCK_TABLE[b] == NO_TABLE: - # inner fingerprint Σ_i w_i · coord_i(ζ_lo); the count side weighs - # slot 0 alone (α⃗ = 0), γ = 0. - inner_sum = 0 - for i in unroll(0, BLOCK_COORD_COUNT[b]): - ci = BLOCK_COORD_OFF[b] + i # a compile-time index, so `ci` costs nothing - if COORD_TYPE[ci] == COORD_KIND_CONST: - coord_val = COORD_CONST[ci] - if COORD_TYPE[ci] == COORD_KIND_COL: - if COORD_FRESH[ci] == 1: - fs, coord_val, cursor = fs_next(fs, cursor) - claim_pool[GEN ** COORD_CLAIM_SLOT[ci]] = coord_val - claim_cplen_g[GEN ** COORD_CLAIM_SLOT[ci]] = kappa_g # cplen = block kappa - else: - coord_val = claim_pool[GEN ** COORD_CLAIM_SLOT[ci]] - if COORD_TYPE[ci] == COORD_KIND_GCOL: - if COORD_FRESH[ci] == 1: - fs, rawv, cursor = fs_next(fs, cursor) - claim_pool[GEN ** COORD_CLAIM_SLOT[ci]] = rawv - claim_cplen_g[GEN ** COORD_CLAIM_SLOT[ci]] = kappa_g - else: - rawv = claim_pool[GEN ** COORD_CLAIM_SLOT[ci]] - coord_val = COORD_CONST[ci] * rawv - if COORD_TYPE[ci] == COORD_KIND_INDEX: - if s == PULL_SIDE: - coord_val = block_index_mle[GEN ** (b - SIDE_BLOCK_START[PULL_SIDE])] - else: - # Index-coord MLE: prod_t (1 + zeta_t * (1 + g^(2^t))) - idx_chain = HeapBuf(MU_CAP + 2) - idx_chain[GEN ** 0] = 1 - for xt in mul_range(1, kappa_g): - idx_chain[xt * GEN] = idx_chain[xt] * (1 + zeta[xt] * idxc_tab[xt]) - coord_val = idx_chain[kappa_g] - if s == PUSH_SIDE: - block_index_mle[GEN ** b] = coord_val - if COORD_TYPE[ci] == COORD_KIND_PUBLIC: - # The public slots carry no value of their own here: their - # alpha-weighted sum IS bc_share, added once per block below - # (push and pull share zeta, so both get the same one). - coord_val = 0 - block_has_public = 1 - if s == COUNT_SIDE: - inner_sum += coord_val - else: - inner_sum += fp_w[GEN ** i] * coord_val - inner_sum += block_has_public * bc_share # the bytecode blocks' public slots - if s == COUNT_SIDE: - acc += eq_hi * inner_sum - else: - acc += eq_hi * (beta + inner_sum) - acc += 1 + selector_sum - # What the tables' blocks owe this side: its GKR leaf value less the - # framework decomposition. DERIVED, not read: a transmitted total would be a - # free value in its own check. The table sumcheck's target pins it below. - bus_table_total[s] = acc + gkr_claims[s] - claim_idx = N_BUS_CLAIMS # AIR/PI/pin claims pool after the deduped bus claims - - # ---- ONE table sumcheck for all six tables ---- - # Mirrors lean_vm::constraints::verify. zc_xi ONCE, each table folding its own - # identities with a DISJOINT range of its powers (ETA_OFFSET[t]); one shared - # point zeta (the bus GKR's); n = max_t tau_t rounds. Rounds bind the HIGHEST - # variable first, so a 2^tau table sits out the first n - tau and joins carrying - # the challenges it sat out, weighing cprod[n - tau] * peq[tau] where peq[tau] = - # eq(zeta[..tau], chi[..tau]); the zc_xi-powers are inside constraint_eval. - # - # g^n is hinted, then pinned exactly: the range-checked division slacks force it - # to dominate every certified tau and the product identity forces it to BE one. - g_zc_n = hint_witness("zc_tau_max") - zc_is_a_tau = 1 - for t in unroll(0, N_TABLES): - tau_g = dims_g[GEN ** (t + 1)] - assert log(g_zc_n / tau_g) < COUNT_BITS - zc_is_a_tau *= g_zc_n + tau_g - assert zc_is_a_tau == 0 - # n <= mu, the `Error::Truncated` of constraints.rs. Every table pushes at - # kappa = tau, so it holds structurally, but zc_peq below reads zeta[..n] and - # zeta only holds mu coords: unwritten heap there is prover-chosen. - assert log(g_bus_mu / g_zc_n) < COUNT_BITS - fs, zc_xi = squeeze(fs) - zc_xi_pows = StackBuf(N_ETA_POWS) - zc_xi_pows[0] = 1 - for k in unroll(1, N_ETA_POWS): - zc_xi_pows[k] = zc_xi_pows[k - 1] * zc_xi - # The eq point is the bus GKR's zeta, NOT a fresh one, which is what lets the - # batch settle the bus forms alongside the constraints. It is also why no target - # is read: what the three sides' tables owe, each in its own shared power of - # zc_xi, IS the sum the batch must reach, and zc_xi is squeezed after those - # totals are fixed, so hitting one number forces all three side equations. - bus_target = 0 - for sd in unroll(0, N_GKR_SIDES): - bus_target += zc_xi_pows[ETA_FORM_BASE + sd] * bus_table_total[sd] - # n vanilla sumcheck rounds: the round polynomial arrives whole, so a round is - # `h(0) + h(1) == claim` and a fold, with no eq to reapply. The tables still - # waiting ride inside h, so nothing here is indexed by height; the heights enter - # only the per-table weights below. - zc_rounds = HeapBuf((g_zc_n * GEN) ** ROUND_SLOTS) - zc_cprod = HeapBuf(g_zc_n * GEN) # the challenges bound so far, multiplied - zc_rounds[GEN ** ROUND_FS0] = fs[0] - zc_rounds[GEN ** ROUND_FS1] = fs[1] - zc_rounds[GEN ** ROUND_CURSOR] = cursor - zc_rounds[GEN ** ROUND_CLAIM] = bus_target - zc_cprod[GEN ** 0] = 1 - for xk in mul_range(1, g_zc_n): - rd = zc_rounds * xk ** ROUND_SLOTS - nfs0, nfs1, ncur, nclaim, rk = sumcheck_round4(rd[GEN ** ROUND_FS0], rd[GEN ** ROUND_FS1], rd[GEN ** ROUND_CURSOR], rd[GEN ** ROUND_CLAIM]) - chi[g_zc_n * INV_GEN / xk] = rk # g^(n-1-j): the variable round j binds - nxt = rd * GEN ** ROUND_SLOTS - nxt[GEN ** ROUND_FS0] = nfs0 - nxt[GEN ** ROUND_FS1] = nfs1 - nxt[GEN ** ROUND_CURSOR] = ncur - nxt[GEN ** ROUND_CLAIM] = nclaim - # cprod is the weight of a table that joins here; peq below is the rest - zc_cprod[xk * GEN] = zc_cprod[xk] * rk - zc_last = zc_rounds * g_zc_n ** ROUND_SLOTS - fs = [zc_last[GEN ** ROUND_FS0], zc_last[GEN ** ROUND_FS1]] - cursor = zc_last[GEN ** ROUND_CURSOR] - claim = zc_last[GEN ** ROUND_CLAIM] - # peq[g^tau] = eq(zeta[..tau], chi[..tau]), as a prefix chain. - zc_peq = HeapBuf(g_zc_n * GEN) - zc_peq[GEN ** 0] = 1 - for xi in mul_range(1, g_zc_n): - zc_peq[xi * GEN] = zc_peq[xi] * (1 + zeta[xi] + chi[xi]) - # Per table: every committed column's evaluation (pooled), its AIR constraint - # at its own reduced point chi[..tau_t], weighted into the batch's final claim. - air_acc = 0 - for t in unroll(0, N_TABLES): - tau_g = dims_g[GEN ** (t + 1)] - col_evals = StackBuf(TABLE_COLS_CAP) - for k in unroll(0, N_TABLE_COLS[t]): - fs, e, cursor = fs_next(fs, cursor) - col_evals[k] = e - claim_pool[GEN ** claim_idx] = e - claim_cplen_g[GEN ** claim_idx] = tau_g # cplen = tau_t - claim_idx += 1 - # The table's AIR constraint at the final point (col_evals is indexed by - # local column index; the formulas mirror tables.rs eval_constraint). Every - # value relation now rides the bus as a degree-2 coordinate, so only JUMP's - # is-nonzero indicator is left with an identity of its own. - constraint_eval = 0 - if t == TABLE_JUMP: - # `b = c*w` and `c*(b+1) = 0`. The condition is K-valued, its memory read - # carrying literal zeros above the low limb, so both identities are - # single-lane (tables.rs jump_identity). Local columns: v_cond at 5, w at - # 12, the indicator b at 13. - c = col_evals[5] - b = col_evals[13] - constraint_eval = zc_xi_pows[ETA_OFFSET[t] + 0] * (b + c * col_evals[12]) - constraint_eval += zc_xi_pows[ETA_OFFSET[t] + 1] * (c * (b + 1)) - # The table's three bus forms, evaluated at the SAME column evaluations: - # Σ_b eq_hi(b) · (γ + Σ_i α^i · coord_i), the coords read off col_evals at - # their local index. This is what replaces opening those columns at ζ. - for sd in unroll(0, N_GKR_SIDES): - form = 0 - for b in unroll(0, N_BLOCKS): - if BLOCK_SIDE[b] == sd: - if BLOCK_TABLE[b] == t: - inner = 0 - for i in unroll(0, BLOCK_COORD_COUNT[b]): - # Each coord is the sum of its terms, over this table's - # column evaluations. A product term (an address, an - # arithmetic result) is degree 2, which the batch's - # round polynomial already allows. - ci = BLOCK_COORD_OFF[b] + i # a compile-time index, so `ci` costs nothing - cv = 0 - for j in unroll(0, COORD_TERM_COUNT[ci]): - tj = COORD_TERM_OFF[ci] + j - if TERM_TYPE[tj] == COORD_KIND_CONST: - cv += TERM_CONST[tj] - if TERM_TYPE[tj] == COORD_KIND_COL: - cv += col_evals[TERM_COL_A[tj]] - if TERM_TYPE[tj] == COORD_KIND_GCOL: - cv += TERM_CONST[tj] * col_evals[TERM_COL_A[tj]] - if TERM_TYPE[tj] == COORD_KIND_PROD: - cv += TERM_CONST[tj] * (col_evals[TERM_COL_A[tj]] * col_evals[TERM_COL_B[tj]]) - if sd == COUNT_SIDE: - inner += cv - else: - inner += fp_w[GEN ** i] * cv - if sd == COUNT_SIDE: - form += block_eq_hi[b] * inner - else: - form += block_eq_hi[b] * (beta + inner) - constraint_eval += zc_xi_pows[ETA_FORM_BASE + sd] * form - air_acc += zc_cprod[g_zc_n / tau_g] * zc_peq[tau_g] * constraint_eval # cprod[n - tau] * peq[tau] - assert air_acc == claim - - # ---- public-input binding claim: MEM as ONE logical E-column ---- - # The VM's bind_pi_claim makes a SINGLE E-claim at [rm, 0..]: - # MEM(rm) = interp(pi_0, pi_1, rm) = pi_0 + rm*(pi_0 + pi_1) - # over the E-valued public input (no lane splitting, no Frobenius). Both - # public words have a zero top limb, so that limb's evaluation is zero at - # every rm; only the two low ones ride the stream and must reassemble it: - # MEM = v_lo + Y*v_hi (doc sec:e2e-pi). - fs, rm = squeeze(fs) - mem = pi_0 + rm * (pi_0 + pi_1) - fs, mem_lo, cursor = fs_next(fs, cursor) - fs, mem_hi, cursor = fs_next(fs, cursor) - assert mem == mem_lo + mem_hi * Y_TOWER - claim_pool[GEN ** claim_idx] = mem_lo - claim_idx += 1 - claim_pool[GEN ** claim_idx] = mem_hi - claim_idx += 1 - claim_pool[GEN ** claim_idx] = 0 - claim_idx += 1 - return fs[0], fs[1], cursor, g_zc_n, bc_share, rm - - -def verify_flock(fs0, fs1, cursor, tau_blake2s_g, zerocheck_chis, lincheck_rs, z_partial): - # Flock's zerocheck (univariate skip, k_skip = 6) then its lincheck, whose - # matrix evaluation is DEFERRED to the caller's statement. The three run buffers - # come in pre-sized; the point z, lincheck's alpha and the deferred matrix part - # come back with the walked Fiat-Shamir state and stream cursor. - # - # tau's reach is bounded: the count gadget gives tau < 34 (every flock buffer is - # sized for that), and q_flock's committed kappa = K_LOG + tau feeds the - # certified size m, whose dispatch bound caps tau below any baked structure. The - # first K_SKIP Boolean rounds are replaced by the univariate skip and consume no - # equality challenges; the remaining r coordinates are N_FIXED_CHALLENGE_ROUNDS - # fixed inner values then sampled outer ones. The prover builds round 1 from - # this equality tail, so its sampled part is squeezed before round 1 is fetched, - # and round 1 before z, which evaluates it. - fs = [fs0, fs1] - mr1cs_g = tau_blake2s_g * GEN ** K_LOG # runtime m = K_LOG + tau_5, in the exponent - zerocheck_r = HeapBuf(mr1cs_g) - for i in unroll(0, N_FIXED_CHALLENGE_ROUNDS): - zerocheck_r[GEN ** (K_SKIP + i)] = FIXED_CHALLENGES[i] - flock_pts = HeapBuf((mr1cs_g * GEN ** 2) ** PAIR_SLOTS) - seed = flock_pts * (GEN ** (K_SKIP + N_FIXED_CHALLENGE_ROUNDS)) ** PAIR_SLOTS - seed[GEN ** 0] = fs[0] - seed[GEN ** 1] = fs[1] - for xi in mul_range(GEN ** (K_SKIP + N_FIXED_CHALLENGE_ROUNDS), mr1cs_g): - row = flock_pts * xi ** PAIR_SLOTS - point_fs = [row[GEN ** 0], row[GEN ** 1]] - point_fs, zerocheck_challenge = squeeze(point_fs) - zerocheck_r[xi] = zerocheck_challenge - row[GEN ** PAIR_SLOTS] = point_fs[0] - row[GEN ** (PAIR_SLOTS + 1)] = point_fs[1] - pts_last = flock_pts * mr1cs_g ** PAIR_SLOTS - fs = [pts_last[GEN ** 0], pts_last[GEN ** 1]] - # round-1 message (P = P^AB + P^C on Lambda, 2^K_SKIP words): fetch + - # observe each word as it comes off the stream, then sample z. - zc_round1 = HeapBuf(2 ** K_SKIP) - for i in unroll(0, 2 ** K_SKIP): - fs, w, cursor = fs_next(fs, cursor) - zc_round1[GEN ** i] = w - fs, zerocheck_z = squeeze(fs) # cursor now sits at the multilinear round messages, walked below - # P(z), interpolated at z over ALL 128 phi8 nodes: the transmitted Lambda values - # (nodes 64..128) plus the S half, zero by the zerocheck identity. The finished - # sum is scaled once by the domain's inverse denominator; the full-domain - # product only adds the S-half factor to the Lambda numerators. - lagrange_nums = StackBuf(2 ** K_SKIP) - lag64(zerocheck_z, lagrange_nums, 2 ** K_SKIP) - s_half_product = GEN ** 0 - zc_running = 0 # the zerocheck running claim entering the multilinear rounds - for i in unroll(0, 2 ** K_SKIP): - s_half_product *= (zerocheck_z + PHI8_NODES[i]) - zc_running += lagrange_nums[i] * zc_round1[GEN ** i] - zc_running *= s_half_product * LAGRANGE_INV_COMBINED - mr1cs_rounds_g = mr1cs_g * INV_GEN ** 6 # the multilinear rounds: m - 6 - for i in unroll(0, N_FIXED_CHALLENGE_ROUNDS): - r_eq = zerocheck_r[GEN ** (K_SKIP + i)] - fs, g_1, cursor = fs_next(fs, cursor) # G's coefficients, bar the constant one - fs, g_2, cursor = fs_next(fs, cursor) - g_0 = zc_running + r_eq * (g_1 + g_2) # the eq-weighted split fixes it - fs, chi_v = squeeze(fs) - zerocheck_chis[GEN ** i] = chi_v - zc_running = g_0 + chi_v * (g_1 + chi_v * g_2) - # the sampled rounds: K_LOG + tau_5 - K_SKIP in all, certified - nmlv_g = tau_blake2s_g * GEN ** (K_LOG - K_SKIP) - flock_rounds = HeapBuf((mr1cs_rounds_g * GEN ** 2) ** ROUND_SLOTS) - seed = flock_rounds * (GEN ** N_FIXED_CHALLENGE_ROUNDS) ** ROUND_SLOTS - seed[GEN ** ROUND_FS0] = fs[0] - seed[GEN ** ROUND_FS1] = fs[1] - seed[GEN ** ROUND_CURSOR] = cursor - seed[GEN ** ROUND_CLAIM] = zc_running - for xi in mul_range(GEN ** N_FIXED_CHALLENGE_ROUNDS, nmlv_g): - rd = flock_rounds * xi ** ROUND_SLOTS - round_fs = [rd[GEN ** ROUND_FS0], rd[GEN ** ROUND_FS1]] - r_eq = zerocheck_r[GEN ** K_SKIP * xi] - cur_i = rd[GEN ** ROUND_CURSOR] - round_fs, g_1, cur_i = fs_next(round_fs, cur_i) # coefficients, bar the constant one - round_fs, g_2, cur_i = fs_next(round_fs, cur_i) - g_0 = rd[GEN ** ROUND_CLAIM] + r_eq * (g_1 + g_2) # the eq-weighted split fixes it - round_fs, chi_v = squeeze(round_fs) - zerocheck_chis[xi] = chi_v - nxt = rd * GEN ** ROUND_SLOTS - nxt[GEN ** ROUND_FS0] = round_fs[0] - nxt[GEN ** ROUND_FS1] = round_fs[1] - nxt[GEN ** ROUND_CURSOR] = cur_i - nxt[GEN ** ROUND_CLAIM] = g_0 + chi_v * (g_1 + chi_v * g_2) - fr_last = flock_rounds * nmlv_g ** ROUND_SLOTS - fs = [fr_last[GEN ** ROUND_FS0], fr_last[GEN ** ROUND_FS1]] - zc_running = fr_last[GEN ** ROUND_CLAIM] - cursor = fr_last[GEN ** ROUND_CURSOR] # walked past all 2*n_mlv round words, now at a_eval - # final: observe a_eval, b_eval; the terminal identity is what defines - # c_eval, so nothing is checked here. C rode the rounds above, so all three - # claims sit at the same point and lincheck pins all three at once. - fs, a_eval, cursor = fs_next(fs, cursor) - fs, b_eval, cursor = fs_next(fs, cursor) - c_eval = zc_running + a_eval * b_eval - # The phi8 Lagrange weights at z over the S nodes: the quirky extension's - # own combination, which the lincheck terminal applies to the 64 slices. - claim_nums = StackBuf(2 ** K_SKIP) - lag64(zerocheck_z, claim_nums, 0) - - # ---- flock lincheck (matrix evaluation DEFERRED) ---- - matrix_eval = hint_witness("matpart") - fs, lincheck_alpha = squeeze(fs) - lincheck_beta = lincheck_alpha * lincheck_alpha - lincheck_cube = lincheck_beta * lincheck_alpha - lc_running = a_eval + lincheck_alpha * b_eval + lincheck_beta * c_eval + lincheck_cube # seed: a + alpha*b + alpha^2*c + alpha^3 (the two matrix claims, C, and the pin) - for i in unroll(0, LINCHECK_ROUNDS): - fs, c0, cursor = fs_next(fs, cursor) # q's coefficients, bar the linear one - fs, c2, cursor = fs_next(fs, cursor) - c1 = lc_running + c2 # the split fixes it against the running claim - fs, rv = squeeze(fs) - lincheck_rs[GEN ** i] = rv - lc_running = c0 + rv * (c1 + rv * c2) # fold the degree-2 round poly at the challenge rv - # post-sumcheck collapse: fetch + observe each word - for i in unroll(0, 2 ** K_SKIP): - fs, w, cursor = fs_next(fs, cursor) - z_partial[GEN ** i] = w - # final consistency: running == matpart (DEFERRED) + beta * pin term. The - # const-pin column folds through the top-variable bindings: weight = - # prod_j (bit_{klog-1-j}(PIN_COLUMN) ? r_j : 1+r_j), surviving z_partial index - # = PIN_COLUMN low 6 bits. - pin_term = lincheck_cube * eq_weight(lincheck_rs, LINCHECK_ROUNDS, PIN_COLUMN, K_LOG) - pin_term *= z_partial[GEN ** (PIN_COLUMN % 2 ** K_SKIP)] - # The C term. Its column weight is the row weight itself (C = I), and both - # sides are tensors, so it collapses to eq(chi_in, chi_in_prime) times the - # phi8 Lagrange combination of the 64 slices: no second matrix walk, no - # second family. - c_point_eq = GEN ** 0 - for t in unroll(0, LINCHECK_ROUNDS): - c_point_eq *= (1 + zerocheck_chis[GEN ** t] + lincheck_rs[GEN ** (LINCHECK_ROUNDS - 1 - t)]) - c_slice_value = 0 - for i in unroll(0, 2 ** K_SKIP): - c_slice_value += claim_nums[i] * z_partial[GEN ** i] - c_slice_value *= LAGRANGE_INV_S - # deferred matrix eval + pin + C - assert lc_running == matrix_eval + pin_term + lincheck_beta * c_point_eq * c_slice_value - # z_partial IS the claim: the terminal identity above pins its 64 slices, and - # ring switching binds every one of them. - return fs[0], fs[1], cursor, zerocheck_z, lincheck_alpha, matrix_eval - - -def certify_placement(kappa_base, g_squares): - # Certify the native order: descending kappa, then ascending column index. - # Accumulate g^offset, with each column advancing it by g^(2^kappa). - col_kappa_g = HeapBuf(N_COMMITTED_COLS) - for c in unroll(0, N_COMMITTED_COLS): - col_kappa_g[GEN ** c] = kappa_base[GEN ** COL_KAPPA_SRC[c]] * GEN ** COL_KAPPA_ADJ[c] - col_sort_order = HeapBuf(N_COMMITTED_COLS) - hint_witness(col_sort_order[0:N_COMMITTED_COLS], "col_sort_order") - col_off_g = HeapBuf(N_COMMITTED_COLS) - g_total = GEN ** 0 - prev_col = GEN ** 0 - prev_kappa = GEN ** 0 - for rank in unroll(0, N_COMMITTED_COLS): - col = col_sort_order[GEN ** rank] - assert log(col) < N_COMMITTED_COLS - kappa_g = col_kappa_g[col] - if rank != 0: - # A negative exponent wraps around the order of GEN and cannot pass this - # small range check, hence prev_kappa >= kappa. - assert log(prev_kappa / kappa_g) < SIZE_BITS - if prev_kappa == kappa_g: - # Equal-sized columns use their compact (native column-order) index - # as the deterministic ascending tie-break. - assert log(col / prev_col) < N_COMMITTED_COLS - col_off_g[col] = g_total # write-once: a duplicate permutation entry collides - # g_squares spans SIZE_BITS, and every kappa is under it: a certified log - # <= 32 (log_mem or a tau), the baked bytecode log, or q_flock's tau_5 + 8. - # Bound it before the lookup, independently of m derived from this product. - assert log(kappa_g) < SIZE_BITS - g_total *= g_squares[kappa_g] - prev_col = col - prev_kappa = kappa_g - - return g_total, col_off_g, col_kappa_g - - -def column_selector(offset, point, kappa: Const): - # Rebuild the offset using only bits above the column's kappa, certifying its - # alignment. The same bits select the column at the complete opening point. - # Both offset and point are zero above m, so extending to MAX_STACK_LOG adds - # only factors eq(0, 0) = 1. - bits = StackBuf(MAX_STACK_LOG) - hint_decompose_bits_exponent(bits, offset, MAX_STACK_LOG) - rebuilt = GEN ** 0 - selector = GEN ** 0 - for k in unroll(kappa, MAX_STACK_LOG): - bit = bits[k] - bits[k] = bit * bit - rebuilt *= 1 + bit * (1 + GEN ** (2 ** k)) - selector *= 1 + bit + point[GEN ** k] - assert rebuilt == offset - return selector - - -def check_opening_terminal(zeta, chi, rm, g_bus_mu, g_zc_n, g_log_mem, tau_blake2s_g, claim_cplen_g, lam_pool, col_offsets, col_kappas, z_vals, c_table, point, inner_total, yr_at_tail, sumcheck_target): - # Evaluate each transparent weight at the complete point in witness order. - # A claim is its low point followed by the certified column's selector bits; - # q_flock slots prepend their fixed slot bits to the low point. - zeta_eq_chain = HeapBuf(SIZE_BITS + 1) - eq_prefix_chain(zeta_eq_chain, 1, zeta, point, g_bus_mu) - chi_eq_chain = HeapBuf(SIZE_BITS + 1) - eq_prefix_chain(chi_eq_chain, 1, chi, point, g_zc_n) - chi_slot_eq_chain = HeapBuf(SIZE_BITS + 1) - eq_prefix_chain(chi_slot_eq_chain, 1, chi, point * GEN ** SLOT_STRIDE_LOG, tau_blake2s_g) - pi_chain = HeapBuf(SIZE_BITS + 1) - pi_chain[GEN ** 0] = 1 - pi_chain[GEN ** 1] = 1 + rm + point[GEN ** 0] - for xk in mul_range(GEN, g_log_mem): - pi_chain[xk * GEN] = pi_chain[xk] * (1 + point[xk]) - pi_eq = pi_chain[g_log_mem] - - selectors = StackBuf(N_COMMITTED_COLS) - for c in unroll(0, N_COMMITTED_COLS): - offset = col_offsets[GEN ** c] - kappa_g = col_kappas[GEN ** c] - selectors[c] = match(log(kappa_g), range(0, N_COLUMN_LOGS), lambda kappa: column_selector(offset, point, kappa)) - - inner_sum = inner_total - for j in unroll(0, N_CLAIMS): - if CLAIM_POINT_BUF[j] == POINT_BUF_PI: - cplen_g = g_log_mem - low_eq = pi_eq - else: - cplen_g = claim_cplen_g[GEN ** j] - nlow = cplen_g - if CLAIM_POINT_BUF[j] == POINT_BUF_ZETA: - low_eq = zeta_eq_chain[cplen_g] - if CLAIM_POINT_BUF[j] == POINT_BUF_RHO: - low_eq = chi_eq_chain[cplen_g] - if CLAIM_POINT_BUF[j] == POINT_BUF_QFLOCK_RHO: - slot_eq = GEN ** 0 - for k in unroll(0, SLOT_STRIDE_LOG): - slot_eq *= 1 + CLAIM_QFLOCK_SLOT_BITS[SLOT_STRIDE_LOG * j + k] + point[GEN ** k] - low_eq = slot_eq * chi_slot_eq_chain[cplen_g] - nlow = cplen_g * GEN ** SLOT_STRIDE_LOG - assert nlow == col_kappas[GEN ** CLAIM_COMMITTED_COL[j]] - inner_sum += lam_pool[GEN ** j] * low_eq * selectors[CLAIM_COMMITTED_COL[j]] - - qflockv_g = tau_blake2s_g * GEN ** SLOT_STRIDE_LOG - assert qflockv_g == col_kappas[GEN ** QFLOCK_COMMITTED_COL] - prod_chains = HeapBuf((qflockv_g * GEN) ** BASE_FIELD_BITS) - for k in unroll(0, BASE_FIELD_BITS): - prod_chains[GEN ** k] = 1 - rs_eq_run(prod_chains, z_vals, point, qflockv_g) - prod_final = prod_chains * qflockv_g ** BASE_FIELD_BITS - rs_weight = 0 - for k in unroll(0, BASE_FIELD_BITS): - rs_weight += c_table[GEN ** k] * prod_final[GEN ** k] - inner_sum += rs_weight * selectors[QFLOCK_COMMITTED_COL] - assert inner_sum * yr_at_tail == sumcheck_target - - -def verify_sub(pi_0, pi_1, seed_0, seed_1, g_logs_pow2, g_squares, defer_out): - # In-circuit verification of ONE inner proof for the statement (pi_0, pi_1), - # mirroring cpu::verify step for step; the `# ---- ... ----` headers below run - # in that order. All proof data is hinted HERE, so each call pops the next - # sub-proof's entry of every witness stream and the body lowers once. The - # exponent tables are shared read-only; the deferred claims go to `defer_out`. - # - # The pool holds every committed-coordinate claim's value in decompose order - # (the points are the GKR zetas, resolvable from the baked block structure) and - # its certified low dimension, which the terminal pins its lengths against. - claim_pool = HeapBuf(N_CLAIMS) - claim_cplen_g = HeapBuf(N_CLAIMS) - - # ---- seed (statement pre-bound: hinted sub pi + baked program digest) ---- - fs = StackBuf(2) - blake2s([seed_0, seed_1], [pi_0, pi_1], fs) - stream = HeapBuf(STREAM_CAP) - hint_witness(stream[0:STREAM_CAP], "stream") - cursor = stream # the proof stream, replayed word by word (advance = * g) - - # ---- announced layout and PCS rate (observed, then certified) ---- - # The stream announces the sizes as integer WORDS, one log each; the - # shape-generic phases need them as G-POWERS (loop bounds, match scrutinees), so - # each is reassembled from its advice-decomposed bits, with no hint and no - # g^j -> j lookup. Every table's rows are real rows (the prover's fill blocks - # bring each count up to a power of two), so a height is all there is to - # announce. - sizes = StackBuf(N_TABLES + 1) - for i in unroll(0, N_TABLES + 1): - fs, x, cursor = fs_next(fs, cursor) - sizes[i] = x - fs, log_inv_rate, cursor = fs_next(fs, cursor) - rate_sel = g_power_of_word(log_inv_rate, g_squares, LOG_WORD_BITS) / GEN - assert log(rate_sel) < LIG_N_RATES - dims_g = HeapBuf(N_TABLES + 1) # [g^log_mem, g^tau_0 .. g^tau_{N_TABLES-1}] - g_log_mem = g_power_of_word(sizes[0], g_squares, LOG_WORD_BITS) - assert log(g_log_mem) < COUNT_BITS - assert log(g_log_mem / GEN ** MIN_LOG_MEM) < COUNT_BITS # native MIN_LOG_MEM <= log_mem - dims_g[GEN ** 0] = g_log_mem - for t in unroll(0, N_TABLES): - g_tau = g_power_of_word(sizes[t + 1], g_squares, LOG_WORD_BITS) - assert log(g_tau) < COUNT_BITS - # A table's floor: flock sizes its BLAKE2s argument to at least 2^3 instances. - assert log(g_tau / GEN ** FLOORS[t]) < COUNT_BITS - dims_g[GEN ** (t + 1)] = g_tau - # kappa_base maps a kappa source index to its certified announced log (source 0 - # = const via the baked adj). Each block's kappa then DERIVES from its - # structural source as a compile-time offset off a certified log: no hint, and - # nothing left free. - kappa_base = HeapBuf(N_TABLES + 2) - kappa_base[GEN ** 0] = 1 - kappa_base[GEN ** 1] = g_log_mem - for t in unroll(0, N_TABLES): - kappa_base[GEN ** (2 + t)] = dims_g[GEN ** (t + 1)] - block_kappa = HeapBuf(N_BLOCKS) - for b in unroll(0, N_BLOCKS): - block_kappa[GEN ** b] = kappa_base[GEN ** BLOCK_KAPPA_SRC[b]] * GEN ** BLOCK_KAPPA_ADJ[b] - # The ONE bus depth, COMPUTED (not hinted): mu = log2_ceil(Σ_b 2^κ_b) over - # PUSH's blocks; pull matches by pairing, the count tree is padded to it. - push_total = GEN ** 0 - for b in unroll(SIDE_BLOCK_START[PUSH_SIDE], SIDE_BLOCK_START[PUSH_SIDE + 1]): - push_total *= g_squares[block_kappa[GEN ** b]] # g^(sum of 2^kappa) - g_bus_mu = log2_ceil_in_the_exponent(push_total, g_logs_pow2, g_squares, 0, SIZE_BITS) - zeta = HeapBuf(g_bus_mu) # the ONE shared GKR point: exactly mu coords - - # ---- commitment root (2 words), kept for the opening phase ---- - fs, commit_root_0, cursor = fs_next(fs, cursor) - fs, commit_root_1, cursor = fs_next(fs, cursor) - # A non-canonical half is rejected here (merkle.rs `scalars_to_hash`); the level - # roots get the same treatment at their own read. - root_cells = StackBuf(2) - root_cells[0] = assert_canonical(commit_root_0) - root_cells[1] = assert_canonical(commit_root_1) - - # ---- bus challenges (F192 provides the soundness margin without grinding) ---- - # A tuple is fingerprinted multilinearly: slot x weighs eq(alphas, x), so a leaf - # factor has total degree N_TUPLE_BITS in the challenges and the aligned bytecode - # polynomial is read off at the challenge vector itself (doc sec:gp, sec:e2e-bc). - bus_alpha = HeapBuf(N_TUPLE_BITS) - for t in unroll(0, N_TUPLE_BITS): - fs, av = squeeze(fs) - bus_alpha[GEN ** t] = av - fp_w = HeapBuf(N_TUPLE_SLOTS) - for x in unroll(0, N_TUPLE_SLOTS): - fp_w[GEN ** x] = eq_weight(bus_alpha, N_TUPLE_BITS, x, 0) - fs, beta = squeeze(fs) - - # ---- ONE GKR grand product: push, pull, and count RLC-batched ---- - fs0, fs1, cursor, claim_push, claim_pull, claim_count = verify_bus_gkr(fs[0], fs[1], cursor, g_bus_mu, zeta) - fs = [fs0, fs1] - - # ---- the bus leaves, the table sumcheck, and the public-input claim ---- - chi = HeapBuf(SIZE_BITS) # chi[i] = the challenge that bound variable i - fs0, fs1, cursor, g_zc_n, bc_share, rm = verify_tables(fs[0], fs[1], cursor, pi_0, pi_1, zeta, g_bus_mu, dims_g, block_kappa, g_squares, fp_w, beta, claim_pool, claim_cplen_g, chi, claim_push, claim_pull, claim_count) - fs = [fs0, fs1] - - # ---- flock zerocheck and lincheck (the matrix evaluation is DEFERRED) ---- - tau_blake2s_g = dims_g[GEN ** (TABLE_BLAKE2s + 1)] # the BLAKE2s table's certified tau - zerocheck_chis = HeapBuf(tau_blake2s_g * GEN ** (K_LOG - K_SKIP)) # m - 6 rounds - lincheck_rs = HeapBuf(LINCHECK_ROUNDS) - z_partial = HeapBuf(2 ** K_SKIP) - fs0, fs1, cursor, zerocheck_z, lincheck_alpha, matrix_eval = verify_flock(fs[0], fs[1], cursor, tau_blake2s_g, zerocheck_chis, lincheck_rs, z_partial) - fs = [fs0, fs1] - - # ---- stacked mixed opening: ring-switch front + claim combination ---- - # The ring-switch slices are z_partial, read and bound above; this block only - # binds them to the commitment. Compose six two-term F2-linear maps with shifts - # 32,16,8,4,2,1: their expansion has all 64 Frobenius terms soundness needs, - # while direct application costs 63 squarings and only six general - # multiplications. - map_challenges = HeapBuf(6) # len(RING_MAP_SHIFTS) - c_table = HeapBuf(BASE_FIELD_BITS) - z_vals = HeapBuf(QFLOCK_VARS_CAP) - for stage in unroll(0, len(RING_MAP_SHIFTS)): - fs, map_challenge = squeeze(fs) - map_challenges[GEN ** stage] = map_challenge - # Expand the same composition once for the later transparent-weight evaluation. - # Before shift d, the populated coefficients are exactly at multiples of 2d; the - # new branch fills the adjacent d-offset entries. - c_table[GEN ** 0] = 1 - for stage in unroll(0, len(RING_MAP_SHIFTS)): - shift = RING_MAP_SHIFTS[stage] - map_challenge = map_challenges[GEN ** stage] - for slot in unroll(0, BASE_FIELD_BITS // (2 * shift)): - coefficient = c_table[GEN ** (slot * 2 * shift)] - for k in unroll(0, shift): - coefficient *= coefficient - c_table[GEN ** (slot * 2 * shift + shift)] = map_challenge * coefficient - # Evaluate the claim and combine its 64 packing rows: the running x-power and - # the running sum ride one two-slot chain. - rs_chain = HeapBuf(((2 ** K_SKIP) + 1) * PAIR_SLOTS) - rs_chain[GEN ** 0] = GEN ** 0 # x^i - rs_chain[GEN ** 1] = 0 # the running sum - for x_round in mul_range(1, GEN ** (2 ** K_SKIP)): - lin_eval = z_partial[x_round] - for stage in unroll(0, len(RING_MAP_SHIFTS)): - frobenius = lin_eval - for k in unroll(0, RING_MAP_SHIFTS[stage]): - frobenius *= frobenius - lin_eval += map_challenges[GEN ** stage] * frobenius - row = rs_chain * x_round ** PAIR_SLOTS - x_pow = row[GEN ** 0] - row[GEN ** PAIR_SLOTS] = x_pow * 2 - row[GEN ** (PAIR_SLOTS + 1)] = row[GEN ** 1] + x_pow * lin_eval - rs_end = rs_chain * (GEN ** (2 ** K_SKIP)) ** PAIR_SLOTS - transposed_claim = rs_end[GEN ** 1] - # Suffix point for the transparent weight. - for t in unroll(0, LINCHECK_ROUNDS): - z_vals[GEN ** t] = lincheck_rs[GEN ** (LINCHECK_ROUNDS - 1 - t)] - zv_lo = z_vals * GEN ** LINCHECK_ROUNDS - zr_hi = zerocheck_chis * GEN ** LINCHECK_ROUNDS - for xt in mul_range(1, tau_blake2s_g): - zv_lo[xt] = zr_hi[xt] - # ONE batching challenge for the whole pool: N_CLAIMS - 1 fewer Fiat-Shamir - # compressions than a challenge per claim, and none for the values themselves, - # `fs_next` having bound every one of them as it read it, so `lam_cl` already - # depends on all of them. Disjoint power ranges, as for the zc_xi-powers above: - # the ring-switch claim takes lam_cl^0, the pool lam_cl^1 onward. - fs, lam_cl = squeeze(fs) - target = transposed_claim - lam_pool = HeapBuf(N_CLAIMS) - lam_pow = lam_cl - for j in unroll(0, N_CLAIMS): - lam_pool[GEN ** j] = lam_pow - target += lam_pow * claim_pool[GEN ** j] - lam_pow *= lam_cl - - g_total, col_offsets, col_kappas = certify_placement(kappa_base, g_squares) - - # ---- certify g^m: m = max(log2_ceil(sum_cols 2^kappa), PCS_MIN_MU) ---- - # g_total is g^(sum 2^kappa) from the certified placement walk above. - gmv = log2_ceil_in_the_exponent(g_total, g_logs_pow2, g_squares, PCS_MIN_MU, SIZE_BITS) # g^m - size_sel = gmv * LIG_MIN_SHIFT_INV # g^(m - MIN) - assert log(size_sel) < LIG_N_LOG_SIZES - # Flatten (rate-1, m-MIN) in rate-major order. Both coordinates are - # transcript-bound and range-checked above, so a single compiled guest can - # dispatch independently for every inner proof in a mixed-rate batch. - config_sel = size_sel * rate_sel ** LIG_N_LOG_SIZES - assert log(config_sel) < LIG_N_CANDIDATES - sumcheck_target, point, inner_total, yr_at_tail = match(log(config_sel), range(0, LIG_N_CANDIDATES), lambda m_idx: open_stacked(m_idx, fs[0], fs[1], target, commit_root_0, commit_root_1, cursor)) - # `stream` is a fixed-capacity witness transport. The shape fixes the exact - # consumed prefix, whose every word is transcript-bound; the unused suffix - # is outside the recursively verified proof and intentionally unconstrained. - - # ---- generalized eval_b terminal (runtime claim shapes) ---- - check_opening_terminal(zeta, chi, rm, g_bus_mu, g_zc_n, g_log_mem, tau_blake2s_g, claim_cplen_g, lam_pool, col_offsets, col_kappas, z_vals, c_table, point, inner_total, yr_at_tail, sumcheck_target) - - # ---- export this sub-proof's deferred-claim data to the caller (FRESH_*) ---- - for k in unroll(0, BYTECODE_LOG): - defer_out[GEN ** k] = zeta[GEN ** k] - for k in unroll(0, LOG2_BYTECODE_COLS): - defer_out[GEN ** (BYTECODE_LOG + k)] = bus_alpha[GEN ** k] - defer_out[GEN ** FRESH_BC_VALUE] = bc_share - defer_out[GEN ** FRESH_ALPHA] = lincheck_alpha - defer_out[GEN ** FRESH_Z_SKIP] = zerocheck_z - for k in unroll(0, LINCHECK_ROUNDS): - defer_out[GEN ** (FRESH_ZCHI + k)] = zerocheck_chis[GEN ** k] - defer_out[GEN ** (FRESH_LINCHECK_RS + k)] = lincheck_rs[GEN ** k] - for k in unroll(0, 2 ** K_SKIP): - defer_out[GEN ** (FRESH_Z_PARTIAL + k)] = z_partial[GEN ** k] - defer_out[GEN ** FRESH_MATPART] = matrix_eval - return - - -# ============================ XMSS signature verification =========================== -# One signature of its epoch group's (epoch, message), against the signer's public -# key at `pk_ptr[g^0..g^1]` = (merkle_root, public_param). Every 16-byte native -# value (tweak, digest, chain tip, sibling, pp) is one canonical 128-bit cell. -# Tweak table layout (tweak index t at cell g^t): -# 0 : encoding tweak -# 1 + CHAIN_STEPS·i + s : chain tweak, chain i < V, step s < CHAIN_STEPS -# WOTS_PK_TWEAK_IDX : wots-pk tweak -# MERKLE_TWEAK_IDX + l : merkle tweak, level l < LOG_LIFETIME - - -def fill_xmss_epoch_tables(epoch, merkle_bits, tweak_table): - # The tweak table and the Merkle direction bits at `epoch`, shared by every XMSS - # signature this node verifies at it. One bit decomposition gives all three uses: - # a tweak's index field is the epoch (encoding, chain, wots-pk) or the parent - # index `epoch >> lvl` at Merkle level lvl - 1, and the direction bit at that - # level IS bit lvl - 1. Booleanity is a write-once pin and the reconstruction - # ties the bits back to the epoch, which also bounds it to LOG_LIFETIME bits. - # SPHINCS shares none of this, deriving every tweak from the index its own - # digest picks, which is neither public nor shared between signers. - hint_decompose_bits(merkle_bits, epoch, LOG_LIFETIME) - bits = StackBuf(LOG_LIFETIME) - reconstructed = 0 - index = 0 - for b in unroll(0, LOG_LIFETIME): - bit = merkle_bits[GEN ** b] - merkle_bits[GEN ** b] = bit * bit - bits[b] = bit - reconstructed += bit * COORD_BASIS[b] - index += bit * XM_INDEX_WEIGHT[b] - assert reconstructed == epoch - tweak_table[1] = index + XM_ENC_TWEAK - for i in unroll(0, V): - for s in unroll(0, CHAIN_STEPS): - tweak_table[GEN ** (1 + CHAIN_STEPS * i + s)] = index + XM_CHAIN_TWEAKS[CHAIN_STEPS * i + s] - tweak_table[GEN ** WOTS_PK_TWEAK_IDX] = index + XM_PK_TWEAK - # Merkle level lvl - 1 hashes the parent at `epoch >> lvl`: the epoch's bits from - # lvl up, each weighed lvl places down. The top level gets the empty sum. - for lvl in unroll(1, LOG_LIFETIME + 1): - parent = 0 - for b in unroll(lvl, LOG_LIFETIME): - parent += bits[b] * XM_INDEX_WEIGHT[b - lvl] - tweak_table[GEN ** (MERKLE_TWEAK_IDX + lvl - 1)] = parent + XM_MERKLE_TWEAKS[lvl - 1] - return - - -def verify_sig(message, tweak_table, merkle_bits, pk_ptr): - pp = pk_ptr[GEN] - - # Encoding digest D = BLAKE2s(tweak | pp | msg | randomness | zero-pad), 96 - # bytes: one full 64-byte block then a 32-byte final block (24 bytes of - # randomness and the specified 8-byte zero pad). A packing helper source is read - # as (lo, 0, 0) where BLAKE2s reads (lo, hi, 0), which is what pins that pad. - after_msg = StackBuf(WORDS_PER_BLOCK) - blake2s([tweak_table[1], pp], [message[1], message[GEN]], after_msg, counter=64, final=0) - rand_block = StackBuf(WORDS_PER_BLOCK) - hint_witness(rand_block, "rand") - assert_in_k(rand_block[1], 0) - digest = StackBuf(WORDS_PER_BLOCK) - blake2s(rand_block, [0, 0], digest, cv=after_msg, counter=96, final=1) - - # V WOTS chains. Per chain the digit is hinted in the exponent (g^{e_i}), range - # checked and dispatched once into this frame; arm k walks the remaining - # CHAIN_STEPS-k steps into `tips` and returns e_i = k weighted by CHAIN_LENGTH^i - # inside its own 64-bit lane (DIGITS_PER_WORD digits a lane, GF(2^64)'s monomial - # budget, each lane's leftover top bits ground to zero by the signer). The product - # of the digits is the target sum (g^{Σe_i}), and the weighted digits reconstruct - # D's first cell as `acc_lo + acc_hi·Y`. - tips = StackBuf(TIP_CELLS) - step_md = StackBuf(1) - step_md[0] = MD_FINAL + 48 # SET once here, not in every arm - chain_tweaks = tweak_table * GEN ** WORDS_PER_VALUE # chain i at cell 1 + CHAIN_STEPS·i - digit_product = 1 - acc_lo = 0 - acc_hi = 0 - for i in unroll(0, V): - digit = hint_witness("digits") - assert log(digit) < CHAIN_LENGTH - chain_start = hint_witness("chain_starts") - term = match(log(digit), range(0, CHAIN_LENGTH), lambda k: walk(chain_start, chain_tweaks, pp, step_md[0], tips, i, k)) - digit_product = digit_product * digit - if i // DIGITS_PER_WORD == 0: - acc_lo = acc_lo + term - else: - acc_hi = acc_hi + term - chain_tweaks = chain_tweaks * GEN ** (WORDS_PER_VALUE * CHAIN_STEPS) - assert digit_product == GEN ** TARGET_SUM - assert acc_lo + acc_hi * Y_TOWER == digest[0] - - # WOTS public-key leaf = standard BLAKE2s over prefix + V tips: WOTS_PK_BLOCKS - # full blocks, carrying the chaining value between instructions. - leaf = StackBuf(WORDS_PER_BLOCK) - blake2s([tweak_table[GEN ** (WORDS_PER_VALUE * WOTS_PK_TWEAK_IDX)], pp], [tips[0], tips[2]], leaf, counter=64, final=0) - for q in unroll(1, WOTS_PK_BLOCKS): - next_leaf = StackBuf(WORDS_PER_BLOCK) - blake2s([tips[8 * q - 4], tips[8 * q - 2]], [tips[8 * q], tips[8 * q + 2]], next_leaf, cv=leaf, counter=64 * (q + 1), final=(q + 1) // WOTS_PK_BLOCKS) - leaf = next_leaf - - # Merkle path from the leaf to the root: the epoch bit orders the two children at - # each level, and the tweak carries that level's parent index. - node = leaf[0] - for lvl in unroll(0, LOG_LIFETIME): - sibling = hint_witness("siblings") - children = order_children(node, sibling, merkle_bits[GEN ** (WORDS_PER_VALUE * lvl)]) - parent = StackBuf(WORDS_PER_BLOCK) - blake2s([tweak_table[GEN ** (WORDS_PER_VALUE * (MERKLE_TWEAK_IDX + lvl))], pp], children, parent) - node = parent[0] - assert node == pk_ptr[1] - return - - -@inline -def walk(value, chain_tweaks, pp, md, tips, i: Const, k: Const): - # Walk chain i's steps k..CHAIN_STEPS-1, value' = H(tweak|pp, value|0), the tip - # landing in tips[2i]. Step s reads its tweak at cell s off the chain's subtable, - # a compile-time offset. - if const(k == CHAIN_STEPS): - tips[2 * i] = value - else: - word = value - for s in unroll(k, CHAIN_STEPS - 1): - out = StackBuf(WORDS_PER_BLOCK) - blake2s([chain_tweaks[GEN ** (WORDS_PER_VALUE * s)], pp], [word, 0], out, md=md) - word = out[0] - blake2s([chain_tweaks[GEN ** (WORDS_PER_VALUE * (CHAIN_STEPS - 1))], pp], [word, 0], tips[2 * i:2 * i + 2], md=md) - return const(k * CHAIN_LENGTH ** (i % DIGITS_PER_WORD)) - - -# ========================== SPHINCS+ signature verification ========================= - - -@inline -def sp_bit_field(bits_ptr, off: Const, n: Const, pos: Const): - # The integer held by bits [off, off+n) of the digest, weighed into the - # coordinate basis at `pos`: a tweak field placed where the tweak wants it, one - # fused multiply-add a bit, whatever lane the bits came from. - acc = 0 - for i in unroll(0, n): - acc += bits_ptr[GEN ** (off + i)] * COORD_BASIS[pos + i] - return acc - - -def sp_walk(value, tw_base, pp, k: Const): - # Walk chain steps k..SP_CHAIN_STEPS-1: value' = Th(P, tw_chain, value). - # `tw_base` already carries the type byte, the layer, 2^w*i and the position - # (tau, e), so step s's tweak is one addition of a compile-time literal. - word = value - for s in unroll(k, SP_CHAIN_STEPS): - out = StackBuf(WORDS_PER_BLOCK) - blake2s([tw_base + s * SP_P_MUL, pp], [word, 0], out, counter=48, final=1) - word = out[0] - return word, k - - -def sp_ots_leaf(tw_pos, pp, msg): - # One layer's one-time verification: the encoding of `msg` under the hinted - # counter, the V chains walked from the revealed values, and the leaf they hash - # to. `tw_pos` is the position's tweak base (layer, tau, e); this function is - # called once per layer, so the V dispatch tables are compiled once for the - # whole scheme. - ctr = hint_witness("sp_counter") - ctr_bits = HeapBuf(GEN ** SP_COUNTER_BITS) - hint_decompose_bits(ctr_bits, ctr, SP_COUNTER_BITS) - bind_bits(ctr_bits, ctr, SP_COUNTER_BITS) # LE_32: four counter bytes, twelve of padding - - # D = Th(P, tw_enc, msg | LE_32(c)), a 52-byte one-block hash. - digest = StackBuf(WORDS_PER_BLOCK) - blake2s([tw_pos + SP_TW_ENC, pp], [msg, ctr], digest, counter=52, final=1) - - # The codeword, as in XMSS: each digit hinted in the exponent, range checked and - # dispatched once, arm k walking the remaining steps; the product of the digits - # is the target sum, and the digits weighted by 2^w within each 64-bit lane - # reconstruct D, which pins each lane's leftover top bits to zero. - tips = StackBuf(SP_TIP_CELLS) - digit_product = 1 - acc_lo = 0 - acc_hi = 0 - for i in unroll(0, SP_V): - digit = hint_witness("sp_digits") - assert log(digit) < SP_CHAIN_LENGTH - chain_start = hint_witness("sp_chain_starts") - tw_chain = tw_pos + SP_TW_CHAIN + i * SP_CHAIN_MUL - tips[i], e = match(log(digit), range(0, SP_CHAIN_LENGTH), lambda k: sp_walk(chain_start, tw_chain, pp, k)) - digit_product = digit_product * digit - term = e * SP_CHAIN_LENGTH ** (i % SP_DIGITS_PER_WORD) - if i // SP_DIGITS_PER_WORD == 0: - acc_lo = acc_lo + term - else: - acc_hi = acc_hi + term - assert digit_product == GEN ** SP_TARGET_SUM - assert acc_lo + acc_hi * Y_TOWER == digest[0] - - leaf = StackBuf(WORDS_PER_BLOCK) - blake2s([tw_pos + SP_TW_LEAF, pp], tips[0:2], leaf, counter=64, final=0) - for q in unroll(1, SP_LEAF_BLOCKS): - next_leaf = StackBuf(WORDS_PER_BLOCK) - blake2s(tips[4 * q - 2:4 * q], tips[4 * q:4 * q + 2], next_leaf, cv=leaf, counter=64 * (q + 1), final=(q + 1) // SP_LEAF_BLOCKS) - leaf = next_leaf - return leaf[0] - - -def verify_sig_sphincs(signer): - # `signer` is one 4-cell entry of the SPHINCS coverage table: the key's root and - # public parameter, then the message THAT signer signed. Where XMSS's message is - # one statement field for the whole node, a SPHINCS message rides its own slot, - # and the signer-set digest binds the two together. - pp = signer[GEN] - - # ---- the message digest, which chooses the few-time key ---- - # D = Truncate(H(tw_msg | P | rho | root | m)), 96 bytes in two blocks. - rho_root = StackBuf(WORDS_PER_BLOCK) - hint_witness(rho_root[0:1], "sp_rand") - rho_root[1] = signer[1] - prefix = StackBuf(WORDS_PER_BLOCK) - blake2s([SP_TW_MSG, pp], rho_root, prefix, counter=64, final=0) - digest = StackBuf(WORDS_PER_BLOCK) - blake2s([signer[GEN ** 2], signer[GEN ** 3]], [0, 0], digest, cv=prefix, counter=96, final=1) - - # The index and the k leaf indices are bit fields of that digest, so its bits are - # advice-decomposed here and bound lane by lane. Nothing else derives them: every - # tweak below is built from these bits. - bits = HeapBuf(GEN ** SP_BIT_CELLS) - lo = StackBuf(1) - hint_f192_limbs(lo, digest[0]) - hi = (digest[0] + lo[0]) * Y_INV - assert_in_k(lo[0], hi) - tail = StackBuf(1) - hint_f192_limbs(tail, digest[1]) - tail_hi = (digest[1] + tail[0]) * Y_INV - assert_in_k(tail[0], tail_hi) - lanes = [lo[0], hi, tail[0]] - for lane in unroll(0, SP_BIT_LANES): - run = bits * GEN ** (lane * BASE_FIELD_BITS) - hint_decompose_bits(run, lanes[lane], BASE_FIELD_BITS) - bind_bits(run, lanes[lane], BASE_FIELD_BITS) - - # The digest is admissible only if its last leaf index is zero, which is what - # lets the forest drop that tree. - for b in unroll(0, SP_A): - assert bits[GEN ** (SP_H + (SP_K - 1) * SP_A + b)] == 0 - - # ---- the few-time signature: one opened leaf per tree of the forest ---- - idx_tau = sp_bit_field(bits, 0, SP_H, SP_TAU_POS) - roots = StackBuf(SP_N_FTS) - for kappa in unroll(0, SP_N_FTS): - leaf_off = SP_H + kappa * SP_A - secret = StackBuf(WORDS_PER_BLOCK) - hint_witness(secret[0:1], "sp_fts_secrets") - fts_leaf = StackBuf(WORDS_PER_BLOCK) - node_index = sp_bit_field(bits, leaf_off, SP_A, SP_J_POS) - blake2s([SP_TW_FTS_LEAF + kappa * SP_LAY_MUL + idx_tau + node_index, pp], [secret[0], 0], fts_leaf, counter=48, final=1) - node = fts_leaf[0] - for level in unroll(0, SP_A): - sibling = hint_witness("sp_fts_paths") - children = order_children(node, sibling, bits[GEN ** (leaf_off + level)]) - parent = StackBuf(WORDS_PER_BLOCK) - if const(level + 1 == SP_A): - node_index = 0 - else: - # The index fits in one lane; clearing its low bit makes division by GEN a right shift. - node_index = (node_index + bits[GEN ** (leaf_off + level)] * COORD_BASIS[SP_J_POS]) / GEN - blake2s([SP_TW_FTS_NODE + kappa * SP_LAY_MUL + const((level + 1) * SP_P_MUL) + idx_tau + node_index, pp], children, parent) - node = parent[0] - roots[kappa] = node - fts_key = StackBuf(WORDS_PER_BLOCK) - blake2s([SP_TW_FTS_ROOTS + idx_tau, pp], roots[0:2], fts_key, counter=64, final=0) - for q in unroll(1, SP_ROOT_BLOCKS): - next_key = StackBuf(WORDS_PER_BLOCK) - blake2s(roots[4 * q - 2:4 * q], roots[4 * q:4 * q + 2], next_key, cv=fts_key, counter=64 * (q + 1), final=(q + 1) // SP_ROOT_BLOCKS) - fts_key = next_key - signed = fts_key[0] - - # ---- the hypertree, bottom layer first ---- - # Layer lay signs what the layer below produced: the few-time key at the bottom, - # that layer's root above it, and the public key's root at the top. - for step in unroll(0, SP_D): - lay = SP_D - 1 - step - leaf_index_off = SP_SUFFIX[lay + 1] - tau_field = sp_bit_field(bits, SP_SUFFIX[lay], SP_H - SP_SUFFIX[lay], SP_TAU_POS) - node_index = sp_bit_field(bits, leaf_index_off, SP_HEIGHTS[lay], SP_J_POS) - tw_pos = tau_field + node_index + lay * SP_LAY_MUL - node = sp_ots_leaf(tw_pos, pp, signed) - for level in unroll(0, SP_HEIGHTS[lay]): - sibling = hint_witness("sp_siblings") - children = order_children(node, sibling, bits[GEN ** (leaf_index_off + level)]) - parent = StackBuf(WORDS_PER_BLOCK) - if const(level + 1 == SP_HEIGHTS[lay]): - node_index = 0 - else: - node_index = (node_index + bits[GEN ** (leaf_index_off + level)] * COORD_BASIS[SP_J_POS]) / GEN - blake2s([SP_TW_NODE + lay * SP_LAY_MUL + const((level + 1) * SP_P_MUL) + tau_field + node_index, pp], children, parent) - node = parent[0] - signed = node - assert signed == signer[1] - return - - -# =========================== statements and the signer set ========================== - - -def statement_digest(seed_0, seed_1, signers_hash, da_0, da_1, defer): - # A node's statement, hashed to the two words the VM publishes, over the - # proving environment's Fiat-Shamir seed (flock's R1CS and this bytecode), the - # two-cell signer-set digest, the digest of its possibly empty DA root list, - # and the DEFER_STMT_CELLS deferred-claim cells. A - # parent rebuilds a child's statement with this same call, over a signer-set - # digest it re-absorbed itself, which forces the child to be a proof of THIS - # bytecode over groups checked against the parent's own. - # - # The preimage is fixed-length, so a plain BLAKE2s beats the Fiat-Shamir chain. - # A header value is a canonical cell and needs no check, the BLAKE2s table - # reading only cells whose top limb is zero. A deferred cell is a full field - # element, so two fill three cells as (s0,s1) (s2,t0) (t1,t2), each top limb - # derived from the two hinted below it and each pack proving its lanes in K. - cells = StackBuf(4 * STMT_BLOCKS) - cells[0] = seed_0 # the STMT_HEADER header cells - cells[1] = seed_1 - cells[2] = signers_hash[1] - cells[3] = signers_hash[GEN] - cells[4] = da_0 - cells[5] = da_1 - for p in unroll(0, STMT_PAIRS): - s = defer[GEN ** (2 * p)] - if const(2 * p + 1 == DEFER_STMT_CELLS): - t = 0 # an odd cell count pairs the last one with a zero partner - else: - t = defer[GEN ** (2 * p + 1)] - s_lo = StackBuf(2) - t_lo = StackBuf(2) - hint_f192_limbs(s_lo, s) - hint_f192_limbs(t_lo, t) - cells[STMT_DEFER_OFF + 3 * p] = pack64x2(s_lo[0], s_lo[1]) - cells[STMT_DEFER_OFF + 3 * p + 1] = pack64x2(((s + s_lo[0]) * Y_INV + s_lo[1]) * Y_INV, t_lo[0]) - cells[STMT_DEFER_OFF + 3 * p + 2] = pack64x2(t_lo[1], ((t + t_lo[0]) * Y_INV + t_lo[1]) * Y_INV) - for k in unroll(0, STMT_PAD_CELLS): - cells[STMT_DEFER_OFF + 3 * STMT_PAIRS + k] = 0 - st = StackBuf(2) - blake2s(cells[0:2], cells[2:4], st, counter=64, final=1 // STMT_BLOCKS) - for b in unroll(1, STMT_BLOCKS): - nxt = StackBuf(2) - blake2s(cells[4 * b:4 * b + 2], cells[4 * b + 2:4 * b + 4], nxt, cv=st, counter=64 * (b + 1), final=(b + 1) // STMT_BLOCKS) - st = nxt - return st[0], st[1] - - -def keys_window(state_0, state_1, base, keys_ptr, x_q, g_squares): - # One window of an epoch group's key hash: SIGNERS_WINDOW blocks, two declared - # keys each (a key is two cells, a block four). Counters as in `sphincs_window`. - nxt = scaled_log(x_q * GEN, g_squares, const(6 + SIGNERS_WINDOW_LOG)) - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, SIGNERS_WINDOW): - pair = keys_ptr * (GEN ** (4 * j)) - hint_witness(pair[0:4], "pubkeys") - out = StackBuf(2) - if const(j + 1 == SIGNERS_WINDOW): - blake2s(pair[0:2], pair[2:4], out, cv=st, md=nxt) - else: - blake2s(pair[0:2], pair[2:4], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], nxt - - -def keys_tail(state_0, state_1, base, keys_ptr, k: Const): - # The key pairs past the last whole window, all non-final, so every offset stays - # below the base's lowest set bit. - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, k): - pair = keys_ptr * (GEN ** (4 * j)) - hint_witness(pair[0:4], "pubkeys") - out = StackBuf(2) - blake2s(pair[0:2], pair[2:4], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], keys_ptr * (GEN ** (4 * k)) - - -def key_list_digest(keys_ptr, half_g, odd_g, n_keys_g, g_squares): - # BLAKE2s of one epoch group's declared key list: 32 bytes a key, so the hashed - # string is 32·n bytes and its last block is the only partial one. The n // 2 - # pairs and the odd key out make half + odd blocks; all but the last run in - # windows plus a tail (doc §sec:prog-byte-counter), and the last carries the - # total length as its counter and the final-block flag. - split = StackBuf(2) - hint_witness(split, "signers_split") # g^windows, g^tail_blocks - windows = split[0] - tail = split[1] - assert log(tail) < SIGNERS_WINDOW - assert log(windows) < SIGNERS_MAX_WINDOWS - assert windows ** SIGNERS_WINDOW * tail == half_g * odd_g * INV_GEN - chain = HeapBuf((windows * GEN) ** 4) # state pair, base, first key of the window - chain[1] = BLAKE2S_IV_0 - chain[GEN] = BLAKE2S_IV_1 - chain[GEN ** 2] = 0 - chain[GEN ** 3] = keys_ptr - for xq in mul_range(1, windows): - slot = chain * (xq ** 4) - s0, s1, nb = keys_window(slot[1], slot[GEN], slot[GEN ** 2], slot[GEN ** 3], xq, g_squares) - step = chain * ((xq * GEN) ** 4) - step[1] = s0 - step[GEN] = s1 - step[GEN ** 2] = nb - step[GEN ** 3] = slot[GEN ** 3] * (GEN ** (4 * SIGNERS_WINDOW)) - end = chain * (windows ** 4) - t0, t1, last = match(log(tail), range(0, SIGNERS_WINDOW), lambda k: keys_tail(end[1], end[GEN], end[GEN ** 2], end[GEN ** 3], k)) - final = scaled_log(n_keys_g, g_squares, 5) + MD_FINAL - digest = StackBuf(2) - if odd_g == 1: - hint_witness(last[0:4], "pubkeys") - blake2s(last[0:2], last[2:4], digest, cv=[t0, t1], md=final) - else: - # The odd key out fills half its block, the rest being the zero bytes the - # counter already accounts for. - hint_witness(last[0:2], "pubkeys") - blake2s(last[0:2], [0, 0], digest, cv=[t0, t1], md=final) - return digest[0], digest[1] - - -def child_keys_window(state_0, state_1, base, keys_ptr, cover, marks, origin_g, limit_g, x_q, g_squares): - # One window of a child's key hash, absorbed exactly as the child absorbed it, - # but with both keys of a block read at hinted indices into THIS node's table and - # marked in the coverage table. Each index is an offset into the parent group the - # caller mapped this child group to, bounded by that group's size, so a child's - # key can only ever land on an XMSS slot of the right epoch. - nxt = scaled_log(x_q * GEN, g_squares, const(6 + SIGNERS_WINDOW_LOG)) - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, SIGNERS_WINDOW): - two = StackBuf(2) - hint_witness(two, "child_index") - assert log(two[0]) < log(limit_g) # precondition as in the raw loops - assert log(two[1]) < log(limit_g) - cover[origin_g * two[0]] = marks * (GEN ** (2 * j)) - cover[origin_g * two[1]] = marks * (GEN ** (2 * j + 1)) - key_a = keys_ptr * (two[0] * two[0]) - key_b = keys_ptr * (two[1] * two[1]) - out = StackBuf(2) - if const(j + 1 == SIGNERS_WINDOW): - blake2s(key_a[0:2], key_b[0:2], out, cv=st, md=nxt) - else: - blake2s(key_a[0:2], key_b[0:2], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], nxt - - -def child_keys_tail(state_0, state_1, base, keys_ptr, cover, marks, origin_g, limit_g, k: Const): - # The child's key pairs past its last whole window, all non-final. - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, k): - two = StackBuf(2) - hint_witness(two, "child_index") - assert log(two[0]) < log(limit_g) - assert log(two[1]) < log(limit_g) - cover[origin_g * two[0]] = marks * (GEN ** (2 * j)) - cover[origin_g * two[1]] = marks * (GEN ** (2 * j + 1)) - key_a = keys_ptr * (two[0] * two[0]) - key_b = keys_ptr * (two[1] * two[1]) - out = StackBuf(2) - blake2s(key_a[0:2], key_b[0:2], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], marks * (GEN ** (2 * k)) - - -def child_key_list_digest(keys_ptr, cover, base, origin_g, limit_g, half_g, odd_g, n_keys_g, g_squares): - # BLAKE2s of one epoch group of a child's keys, over the same 32·n bytes the - # child hashed (`key_list_digest`), so the digest it rebuilds is the one the - # child's statement carries. `base` prefixes the coverage write values, which - # count the keys off as they are marked. - split = StackBuf(2) - hint_witness(split, "signers_split") - windows = split[0] - tail = split[1] - assert log(tail) < SIGNERS_WINDOW - assert log(windows) < SIGNERS_MAX_WINDOWS - assert windows ** SIGNERS_WINDOW * tail == half_g * odd_g * INV_GEN - chain = HeapBuf((windows * GEN) ** 4) - chain[1] = BLAKE2S_IV_0 - chain[GEN] = BLAKE2S_IV_1 - chain[GEN ** 2] = 0 - chain[GEN ** 3] = base - for xq in mul_range(1, windows): - slot = chain * (xq ** 4) - s0, s1, nb = child_keys_window(slot[1], slot[GEN], slot[GEN ** 2], keys_ptr, cover, slot[GEN ** 3], origin_g, limit_g, xq, g_squares) - step = chain * ((xq * GEN) ** 4) - step[1] = s0 - step[GEN] = s1 - step[GEN ** 2] = nb - step[GEN ** 3] = slot[GEN ** 3] * (GEN ** (2 * SIGNERS_WINDOW)) - end = chain * (windows ** 4) - t0, t1, marks = match(log(tail), range(0, SIGNERS_WINDOW), lambda k: child_keys_tail(end[1], end[GEN], end[GEN ** 2], keys_ptr, cover, end[GEN ** 3], origin_g, limit_g, k)) - final = scaled_log(n_keys_g, g_squares, 5) + MD_FINAL - digest = StackBuf(2) - if odd_g == 1: - two = StackBuf(2) - hint_witness(two, "child_index") - assert log(two[0]) < log(limit_g) - assert log(two[1]) < log(limit_g) - cover[origin_g * two[0]] = marks - cover[origin_g * two[1]] = marks * GEN - key_a = keys_ptr * (two[0] * two[0]) - key_b = keys_ptr * (two[1] * two[1]) - blake2s(key_a[0:2], key_b[0:2], digest, cv=[t0, t1], md=final) - else: - tail_idx = hint_witness("child_index") - assert log(tail_idx) < log(limit_g) - cover[origin_g * tail_idx] = marks - key_last = keys_ptr * (tail_idx * tail_idx) - blake2s(key_last[0:2], [0, 0], digest, cv=[t0, t1], md=final) - return digest[0], digest[1] - - -def scaled_log(x, g_squares, shift: Const): - # 2^shift times the exponent of `x`, as a bit pattern (doc §sec:prog-byte-counter). - # The exponent's bits are advice, tied back by the g-power product; weighing them - # at COORD_BASIS[j] assembles the exponent itself and the final multiply is the - # shift, exact because nothing reduces below degree 64. Both sides of the product - # stay under the order of g, so the bits ARE that exponent, hence below - # 2^SIGNERS_COUNT_BITS, which every count and window index here is. - bits = StackBuf(SIGNERS_COUNT_BITS) - hint_decompose_bits_exponent(bits, x, SIGNERS_COUNT_BITS) - value = 0 - rebuilt = GEN ** 0 - for j in unroll(0, SIGNERS_COUNT_BITS): - b = bits[j] - bits[j] = b * b # booleanity, as a write-once pin - value += b * COORD_BASIS[j] - rebuilt *= (1 + b * (g_squares[GEN ** j] + 1)) - assert rebuilt == x - return value * COORD_BASIS[shift] - - -def sphincs_window(state_0, state_1, base, entries_ptr, x_q, g_squares): - # One window of the SPHINCS list's hash: SIGNERS_WINDOW claims, one 64-byte block - # each (the claimed key, then the message it signed). Block j's counter is - # base + 64(j+1), one XOR, except the last, whose offset is the base's own lowest - # bit and which therefore takes the NEXT window's base as its whole counter. That - # base is derived here and carried out for the following window. - nxt = scaled_log(x_q * GEN, g_squares, const(6 + SIGNERS_WINDOW_LOG)) - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, SIGNERS_WINDOW): - entry = entries_ptr * (GEN ** (4 * j)) - hint_witness(entry[0:4], "sphincs_signers") - out = StackBuf(2) - if const(j + 1 == SIGNERS_WINDOW): - blake2s(entry[0:2], entry[2:4], out, cv=st, md=nxt) - else: - blake2s(entry[0:2], entry[2:4], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], nxt - - -def sphincs_tail(state_0, state_1, base, entries_ptr, k: Const): - # The blocks the window loop leaves over, fewer than a window, so every offset - # 64(j+1) stays below the base's lowest set bit and needs no next base. - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, k): - entry = entries_ptr * (GEN ** (4 * j)) - hint_witness(entry[0:4], "sphincs_signers") - out = StackBuf(2) - blake2s(entry[0:2], entry[2:4], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], entries_ptr * (GEN ** (4 * k)) - - -def sphincs_list_digest(entries_ptr, n_g, g_squares): - # BLAKE2s of the declared SPHINCS claims: n blocks of 64 bytes, so the hash is - # over exactly 64n bytes and no block is partial. The last block is absorbed - # apart, carrying the total length as its counter and the final-block flag; the - # n - 1 before it run in windows plus a tail (doc §sec:prog-byte-counter). - digest = StackBuf(2) - if n_g == 1: - # No claims: the hash of the empty string, one compression of a zero block. - blake2s([0, 0], [0, 0], digest, md=MD_FINAL) - else: - split = StackBuf(2) - hint_witness(split, "signers_split") # g^windows, g^tail_blocks - windows = split[0] - tail = split[1] - assert log(tail) < SIGNERS_WINDOW - assert log(windows) < SIGNERS_MAX_WINDOWS - assert windows ** SIGNERS_WINDOW * tail == n_g * INV_GEN - # Four cells a window: the state pair, the window's base, its first entry. - chain = HeapBuf((windows * GEN) ** 4) - chain[1] = BLAKE2S_IV_0 - chain[GEN] = BLAKE2S_IV_1 - chain[GEN ** 2] = 0 - chain[GEN ** 3] = entries_ptr - for xq in mul_range(1, windows): - slot = chain * (xq ** 4) - s0, s1, nb = sphincs_window(slot[1], slot[GEN], slot[GEN ** 2], slot[GEN ** 3], xq, g_squares) - step = chain * ((xq * GEN) ** 4) - step[1] = s0 - step[GEN] = s1 - step[GEN ** 2] = nb - step[GEN ** 3] = slot[GEN ** 3] * (GEN ** (4 * SIGNERS_WINDOW)) - end = chain * (windows ** 4) - t0, t1, last = match(log(tail), range(0, SIGNERS_WINDOW), lambda k: sphincs_tail(end[1], end[GEN], end[GEN ** 2], end[GEN ** 3], k)) - hint_witness(last[0:4], "sphincs_signers") - final = scaled_log(n_g, g_squares, 6) + MD_FINAL - blake2s(last[0:2], last[2:4], digest, cv=[t0, t1], md=final) - return digest[0], digest[1] - - -def child_sphincs_window(state_0, state_1, base, entries_ptr, cover, marks, origin_g, limit_g, x_q, g_squares): - # One window of a child's SPHINCS list, absorbed exactly as the child absorbed - # it, but with each block's claim read at a hinted index into THIS node's table - # and marked in the coverage table. The index is an offset into the SPHINCS - # region and bounded by that region's size, so a child's claim can only ever - # land on a SPHINCS slot. Counters as in `sphincs_window`. - nxt = scaled_log(x_q * GEN, g_squares, const(6 + SIGNERS_WINDOW_LOG)) - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, SIGNERS_WINDOW): - off_hint = hint_witness("child_sphincs_index") - assert log(off_hint) < log(limit_g) # precondition as in the raw loops - cover[origin_g * off_hint] = marks * (GEN ** j) - entry = entries_ptr * (off_hint ** 4) - out = StackBuf(2) - if const(j + 1 == SIGNERS_WINDOW): - blake2s(entry[0:2], entry[2:4], out, cv=st, md=nxt) - else: - blake2s(entry[0:2], entry[2:4], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], nxt - - -def child_sphincs_tail(state_0, state_1, base, entries_ptr, cover, marks, origin_g, limit_g, k: Const): - # The blocks past the child's last whole window, all of them non-final, so every - # offset stays below this base's lowest set bit. - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, k): - off_hint = hint_witness("child_sphincs_index") - assert log(off_hint) < log(limit_g) - cover[origin_g * off_hint] = marks * (GEN ** j) - entry = entries_ptr * (off_hint ** 4) - out = StackBuf(2) - blake2s(entry[0:2], entry[2:4], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1] - - -def child_sphincs_list_digest(entries_ptr, cover, base, origin_g, limit_g, n_g, g_squares): - # BLAKE2s of a child's declared SPHINCS claims, over the same 64n bytes the - # child hashed (`sphincs_list_digest`), so the digest it rebuilds is the one the - # child's statement carries. `base` prefixes the coverage write values, which - # count the claims off as they are marked. - digest = StackBuf(2) - if n_g == 1: - blake2s([0, 0], [0, 0], digest, md=MD_FINAL) - else: - split = StackBuf(2) - hint_witness(split, "signers_split") - windows = split[0] - tail = split[1] - assert log(tail) < SIGNERS_WINDOW - assert log(windows) < SIGNERS_MAX_WINDOWS - assert windows ** SIGNERS_WINDOW * tail == n_g * INV_GEN - chain = HeapBuf((windows * GEN) ** 4) - chain[1] = BLAKE2S_IV_0 - chain[GEN] = BLAKE2S_IV_1 - chain[GEN ** 2] = 0 - for xq in mul_range(1, windows): - slot = chain * (xq ** 4) - marks = base * (xq ** SIGNERS_WINDOW) - s0, s1, nb = child_sphincs_window(slot[1], slot[GEN], slot[GEN ** 2], entries_ptr, cover, marks, origin_g, limit_g, xq, g_squares) - step = chain * ((xq * GEN) ** 4) - step[1] = s0 - step[GEN] = s1 - step[GEN ** 2] = nb - end = chain * (windows ** 4) - marks = base * (windows ** SIGNERS_WINDOW) - t0, t1 = match(log(tail), range(0, SIGNERS_WINDOW), lambda k: child_sphincs_tail(end[1], end[GEN], end[GEN ** 2], entries_ptr, cover, marks, origin_g, limit_g, k)) - off_hint = hint_witness("child_sphincs_index") - assert log(off_hint) < log(limit_g) - cover[origin_g * off_hint] = base * (n_g * INV_GEN) - entry = entries_ptr * (off_hint ** 4) - final = scaled_log(n_g, g_squares, 6) + MD_FINAL - blake2s(entry[0:2], entry[2:4], digest, cv=[t0, t1], md=final) - return digest[0], digest[1] - - -def plain_window(state_0, state_1, base, run_ptr, x_q, g_squares): - # One window over a run of cells already in memory, four to a block, hinting - # nothing. Counters as in `sphincs_window`. - nxt = scaled_log(x_q * GEN, g_squares, const(6 + SIGNERS_WINDOW_LOG)) - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, SIGNERS_WINDOW): - block = run_ptr * (GEN ** (4 * j)) - out = StackBuf(2) - if const(j + 1 == SIGNERS_WINDOW): - blake2s(block[0:2], block[2:4], out, cv=st, md=nxt) - else: - blake2s(block[0:2], block[2:4], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], nxt - - -def plain_tail(state_0, state_1, base, run_ptr, k: Const): - # The blocks of the run past its last whole window, all non-final. - st = StackBuf(2) - st[0] = state_0 - st[1] = state_1 - for j in unroll(0, k): - block = run_ptr * (GEN ** (4 * j)) - out = StackBuf(2) - blake2s(block[0:2], block[2:4], out, cv=st, md=base + const(64 * (j + 1))) - st = out - return st[0], st[1], run_ptr * (GEN ** (4 * k)) - - -def signer_set_digest(run_ptr, n_epochs_g, g_squares): - # BLAKE2s of the signer set: both list lengths and the SPHINCS list's digest - # in the first block, then two blocks a group, its (epoch, count, message) - # and its key list's digest. Every block is full, so the hash is over exactly - # 64·(1 + 2·epochs) bytes, and leading with both lengths makes the encoding - # prefix-free: no set's string is a prefix of another's. - blocks = n_epochs_g * n_epochs_g * GEN # g^(1 + 2·epochs) - split = StackBuf(2) - hint_witness(split, "signers_split") - windows = split[0] - tail = split[1] - assert log(tail) < SIGNERS_WINDOW - assert log(windows) < SIGNERS_MAX_WINDOWS - assert windows ** SIGNERS_WINDOW * tail == blocks * INV_GEN - chain = HeapBuf((windows * GEN) ** 4) - chain[1] = BLAKE2S_IV_0 - chain[GEN] = BLAKE2S_IV_1 - chain[GEN ** 2] = 0 - chain[GEN ** 3] = run_ptr - for xq in mul_range(1, windows): - slot = chain * (xq ** 4) - s0, s1, nb = plain_window(slot[1], slot[GEN], slot[GEN ** 2], slot[GEN ** 3], xq, g_squares) - step = chain * ((xq * GEN) ** 4) - step[1] = s0 - step[GEN] = s1 - step[GEN ** 2] = nb - step[GEN ** 3] = slot[GEN ** 3] * (GEN ** (4 * SIGNERS_WINDOW)) - end = chain * (windows ** 4) - t0, t1, last = match(log(tail), range(0, SIGNERS_WINDOW), lambda k: plain_tail(end[1], end[GEN], end[GEN ** 2], end[GEN ** 3], k)) - final = scaled_log(blocks, g_squares, 6) + MD_FINAL - digest = StackBuf(2) - blake2s(last[0:2], last[2:4], digest, cv=[t0, t1], md=final) - return digest[0], digest[1] - - -def rebuild_child_groups(nsub_e_g, run_ptr, base, epochs, msgs, group_base, group_slots, n_epochs_g, xmss_table, cover, g_squares): - # The child's epoch groups, written into the run its own signer-set hash covers, - # two blocks a group exactly as the child laid them out: its (epoch, count, - # message), then the digest of its keys, read from THIS node's table through - # hinted indices. A hinted map ties each group to the parent group holding the - # same epoch AND message, whose region its keys land in. Everything hinted here - # is pinned by the digest, which the child's statement carries. The running - # product of the group counts prefixes each group's coverage writes and ends as - # the child's XMSS claim count. - counts = HeapBuf(nsub_e_g * GEN) - counts[GEN ** 0] = 1 - for xj in mul_range(1, nsub_e_g): - grp = StackBuf(4) - hint_witness(grp, "child_group") # epoch, msg_lo, msg_hi, count - n_keys = grp[3] - assert log(n_keys) < MAX_KEYS - parent = hint_witness("child_group_map") - assert log(parent) < log(n_epochs_g) - assert epochs[parent] == grp[0] - parent_msg = msgs * (parent * parent) - assert parent_msg[1] == grp[1] - assert parent_msg[GEN] == grp[2] - halves = StackBuf(2) - hint_witness(halves, "child_halves") - assert log(halves[1]) < 2 - assert log(halves[0]) < MAX_KEYS - assert halves[0] * halves[0] * halves[1] == n_keys - gb = group_base[parent] - prefix = counts[xj] - kd_0, kd_1 = child_key_list_digest(xmss_table * (gb * gb), cover, base * prefix, gb, group_slots[parent], halves[0], halves[1], n_keys, g_squares) - slot = run_ptr * (xj ** 8) * (GEN ** 4) - slot[1] = grp[0] - slot[GEN] = n_keys - slot[GEN ** 2] = grp[1] - slot[GEN ** 3] = grp[2] - slot[GEN ** 4] = kd_0 - slot[GEN ** 5] = kd_1 - slot[GEN ** 6] = 0 - slot[GEN ** 7] = 0 - counts[xj * GEN] = prefix * n_keys - return counts[nsub_e_g] - - -# ================================ LeanDA =============================== -# Hash the encoded rows and the hinted membership vector, then check their inner products. -# The external verifier derives the vector hash from the root; recursion preserves both. - - -def da_verify(g_squares): - # Returns the matrix root and membership-vector hash, two canonical cells each. - # The row count is a run-time parameter. The trees need a compile-time depth, - # so the rows are padded to a power of two and the two `match`es below dispatch - # on its log; everything else walks the real rows only. A padding row is the - # zero codeword, so its cell digest and its row digest are constants, and the - # gap loop writes them without hashing anything. - shape = StackBuf(2) - hint_witness(shape, "da_shape") # n_blob, then log2 of the padded count - n_blob_g = shape[0] - g_log_pad = shape[1] - n_pad_g, gap_g = da_row_shape(n_blob_g, g_log_pad, g_squares) - - prefix_digests = HeapBuf(n_pad_g ** (2 * DA_PREFIX_CELLS)) - prefix_bases = HeapBuf(n_blob_g) - for xi in mul_range(1, n_blob_g): - prefix_bases[xi] = prefix_digests * (xi ** (2 * DA_PREFIX_CELLS)) - - hashes = HeapBuf(2 * (DA_CELLS + 1)) - hashes[1] = BLAKE2S_IV_0 - hashes[GEN] = BLAKE2S_IV_1 - counter_values = StackBuf(3 * DA_CELLS + 1) - for w in unroll(0, 3 * DA_CELLS + 1): - counter_values[w] = const(w * 2 ** (DA_LOG_CELL + 3)) - counters = addr(counter_values) - - # Each row's running sum uses a fresh write-once cell per column. - acc = HeapBuf(n_blob_g ** (DA_CELLS + 1)) - rowbase = HeapBuf(n_blob_g) - for xi in mul_range(1, n_blob_g): - chain = acc * (xi ** (DA_CELLS + 1)) - rowbase[xi] = chain - chain[1] = 0 - - # Prefix blocks first, so the digests the row branch needs are stored by a - # loop that knows it is inside the prefix, with no per-block branch. - coltree = HeapBuf(4 * DA_CELLS) - for xb in mul_range(1, GEN ** DA_PREFIX_CELLS): - da_verify_column(xb, n_blob_g, n_pad_g, gap_g, g_log_pad, prefix_bases, coltree, rowbase, hashes, counters, 1) - for xb in mul_range(GEN ** DA_PREFIX_CELLS, GEN ** DA_CELLS): - da_verify_column(xb, n_blob_g, n_pad_g, gap_g, g_log_pad, prefix_bases, coltree, rowbase, hashes, counters, 0) - - for xi in mul_range(1, n_blob_g): - chain = rowbase[xi] - assert chain[GEN ** DA_CELLS] == 0 - rc0, rc1 = da_levels(coltree, DA_CELLS, DA_BLOCK_BITS) - - # The row branch: each row's prefix digests, hashed as one string. - rowtree = HeapBuf(n_pad_g ** 4) - for xi in mul_range(1, n_blob_g): - run = prefix_bases[xi] - st = StackBuf(2) - blake2s(run[0:2], run[2:4], st, counter=64, final=1 // DA_ROW_BLOCKS) - for b in unroll(1, DA_ROW_BLOCKS): - nxt = StackBuf(2) - blake2s(run[4 * b:4 * b + 2], run[4 * b + 2:4 * b + 4], nxt, cv=st, counter=64 * (b + 1), final=(b + 1) // DA_ROW_BLOCKS) - st = nxt - slot = rowtree * (xi ** 2) - slot[1] = st[0] - slot[GEN] = st[1] - for xd in mul_range(1, gap_g): - pad = rowtree * (n_blob_g ** 2) * (xd ** 2) - pad[1] = DA_PAD_ROW_0 - pad[GEN] = DA_PAD_ROW_1 - rr0, rr1 = match(log(g_log_pad), range(0, DA_TREE_ARMS), lambda k: da_levels(rowtree, 2 ** k, k)) - - root = StackBuf(2) - blake2s([rr0, rr1], [rc0, rc1], root) - return root[0], root[1], hashes[GEN ** (2 * DA_CELLS)], hashes[GEN ** (2 * DA_CELLS + 1)] - - -def da_row_shape(n_blob_g, g_log_pad, g_squares): - assert n_blob_g != 1 - assert log(n_blob_g) < DA_MAX_ROWS + 1 - assert log(g_log_pad) < DA_LOG_MAX_ROWS + 1 - n_pad_g = g_squares[g_log_pad] - # n_blob <= n_pad < 2*n_blob: both differences must be nonnegative. - gap_g = n_pad_g / n_blob_g - assert log(gap_g) < DA_MAX_ROWS + 1 - assert log(n_blob_g * n_blob_g / (n_pad_g * GEN)) < DA_MAX_ROWS - return n_pad_g, gap_g - - -def da_verify_column(xb, n_blob_g, n_pad_g, gap_g, g_log_pad, prefix_bases, coltree, rowbase, hashes, counters, store: Const): - st = hashes * (xb ** 2) - h0, h1, lvals = da_vector_dispatch(xb, st[1], st[GEN], counters) - nxt = hashes * ((xb * GEN) ** 2) - nxt[1] = h0 - nxt[GEN] = h1 - node = HeapBuf(n_pad_g ** 4) - for xi in mul_range(1, n_blob_g): - d0, d1, s = da_verify_cell(lvals) - chain = rowbase[xi] * xb - chain[GEN] = chain[1] + s - leaf = node * (xi ** 2) - leaf[1] = d0 - leaf[GEN] = d1 - if const(store == 1): - slot = prefix_bases[xi] * (xb * xb) - slot[1] = d0 - slot[GEN] = d1 - for xd in mul_range(1, gap_g): - pad = node * (n_blob_g ** 2) * (xd ** 2) - pad[1] = DA_PAD_CELL_0 - pad[GEN] = DA_PAD_CELL_1 - c0, c1 = match(log(g_log_pad), range(0, DA_TREE_ARMS), lambda k: da_levels(node, 2 ** k, k)) - out = coltree * (xb * xb) - out[1] = c0 - out[GEN] = c1 - return - - -def da_verify_cell(lvals): - # Packing checks that each symbol used by both the hash and dot product is in K. - sym = StackBuf(DA_CELL) - hint_witness(sym, "da_symbols") - packed = StackBuf(DA_CELL // 2) - s = 0 - for e in unroll(0, DA_CELL // 2): - packed[e] = pack64x2(sym[2 * e], sym[2 * e + 1]) - s = s + lvals[GEN ** (2 * e)] * sym[2 * e] - s = s + lvals[GEN ** (2 * e + 1)] * sym[2 * e + 1] - st = StackBuf(2) - blake2s(packed[0:2], packed[2:4], st, counter=64, final=1 // DA_CELL_BLOCKS) - for b in unroll(1, DA_CELL_BLOCKS): - nxt = StackBuf(2) - blake2s(packed[4 * b:4 * b + 2], packed[4 * b + 2:4 * b + 4], nxt, cv=st, counter=64 * (b + 1), final=(b + 1) // DA_CELL_BLOCKS) - st = nxt - return st[0], st[1], s - - -def da_levels(tree, n: Const, log_n: Const): - # The internal levels of a Merkle tree whose leaves already sit at level 0 of - # `tree` (2 cells a node, level lvl at 4n - 4n//2**lvl). Returns the root pair. - for lvl in unroll(0, log_n): - for xp in mul_range(1, GEN ** (n // 2 ** (lvl + 1))): - a = tree * (GEN ** (4 * n - 4 * n // 2 ** lvl)) * (xp ** 4) - b = tree * (GEN ** (4 * n - 4 * n // 2 ** (lvl + 1))) * (xp * xp) - blake2s(a[0:2], a[2:4], b[0:2]) - return tree[GEN ** (4 * n - 4)], tree[GEN ** (4 * n - 3)] - - -def da_vector_dispatch(xb, h0, h1, counters): - out = StackBuf(3) - if xb == GEN ** (DA_CELLS - 1): - a, b, weights = da_vector_cell(xb, h0, h1, counters, 1) - out[0] = a - out[1] = b - out[2] = weights - else: - a, b, weights = da_vector_cell(xb, h0, h1, counters, 0) - out[0] = a - out[1] = b - out[2] = weights - return out[0], out[1], out[2] - - -def da_vector_cell(xb, h0, h1, counters, final: Const): - weights = StackBuf(DA_CELL) - hint_witness(weights[0:DA_CELL], "da_weights") - packed = StackBuf(3 * DA_CELL // 2) - # Two extension-field entries become three canonical 128-bit cells, without padding. - for p in unroll(0, DA_CELL // 2): - s = weights[2 * p] - t = weights[2 * p + 1] - slo = StackBuf(2) - tlo = StackBuf(2) - hint_f192_limbs(slo, s) - hint_f192_limbs(tlo, t) - packed[3 * p] = pack64x2(slo[0], slo[1]) - packed[3 * p + 1] = pack64x2(((s + slo[0]) * Y_INV + slo[1]) * Y_INV, tlo[0]) - packed[3 * p + 2] = pack64x2(tlo[1], ((t + tlo[0]) * Y_INV + tlo[1]) * Y_INV) - st = StackBuf(2) - st[0] = h0 - st[1] = h1 - # Three power-of-two byte windows per cell keep counter offsets disjoint from the base. - base = counters[xb ** 3] - for w in unroll(0, 3): - window = xb ** 3 * GEN ** w - end = counters[window * GEN] - for b in unroll(0, DA_CELL_BLOCKS): - out = StackBuf(2) - if const(b + 1 == DA_CELL_BLOCKS): - md = end - else: - md = base + const(64 * (b + 1)) - if const(final == 1): - if const(w == 2): - if const(b + 1 == DA_CELL_BLOCKS): - md = md + MD_FINAL - blake2s(packed[w * (DA_CELL // 2) + 4 * b:w * (DA_CELL // 2) + 4 * b + 2], packed[w * (DA_CELL // 2) + 4 * b + 2:w * (DA_CELL // 2) + 4 * b + 4], out, cv=st, md=md) - st = out - base = end - weights_ptr = addr(weights) - return st[0], st[1], weights_ptr - -# ================================ the aggregation node ============================== - - -def da_list_digest(roots, n_g): - # At most 16 roots: every BLAKE2s byte counter and final flag is constant. - a, b = match(log(n_g), range(0, DA_ROOT_COUNTS), lambda n: da_hash_roots(roots, n)) - return a, b - - -def da_hash_roots(roots, n: Const): - digest = StackBuf(2) - if const(n == 0): - blake2s([0, 0], [0, 0], digest, counter=0, final=1) - else: - st = StackBuf(2) - st[0] = BLAKE2S_IV_0 - st[1] = BLAKE2S_IV_1 - for i in unroll(0, n): - claim = roots * GEN ** (4 * i) - out = StackBuf(2) - blake2s(claim[0:2], claim[2:4], out, cv=st, counter=64 * (i + 1), final=(i + 1) // n) - st = out - digest[0] = st[0] - digest[1] = st[1] - return digest[0], digest[1] - - -def cover_da_root(roots, cover, n_slots_g, mark): - index = hint_witness("da_index") - assert log(index) < log(n_slots_g) - cover[index] = mark - root = roots * (index ** 4) - return root[1], root[GEN], root[GEN ** 2], root[GEN ** 3] - - -def main(): - # One node of an aggregation tree: raw XMSS signatures grouped by the (epoch, - # message) they claim (a RUNTIME number of groups), n_raw_sphincs SPHINCS signatures - # and n_children sub-proofs OF THIS SAME BYTECODE. Each XMSS group carries its - # own (epoch, message) pair, and each SPHINCS signature is against the message - # in its own coverage slot. DA roots occupy a separate region of the same table. - # - # The declared lists are the signer set; the duplicate slots absorb keys a child - # covers that the set does not declare. The coverage table is one region per - # epoch group, each its declared keys then its own duplicates, then SPHINCS - # and DA regions shaped the same way: - # - # [group 0: declared | dup]...[group n_epochs-1: declared | dup][SPHINCS: declared | dup][DA: declared | dup] - # - # The first n_decl groups are the signer set's; the n_drop after them declare - # nothing, holding a child group's (epoch, message) without publishing it. - # - # so one range check per write keeps each writer inside its own region: that is - # what makes the statement's split mean which scheme verified which key against - # which (epoch, message). An XMSS slot is two cells, a SPHINCS slot four: a key - # and the message that key signed. A DA root occupies two cells. - meta = StackBuf(7) - hint_witness(meta, "meta") # every count in the exponent - n_decl_g = meta[0] - n_drop_g = meta[1] - n_sphincs_g = meta[2] - n_sdup_g = meta[3] - n_raw_s_g = meta[4] - n_children_g = meta[5] - n_direct_da_g = meta[6] - # Declared plus dropped, so bounding each side pins n_decl <= n_epochs. - assert log(n_decl_g) < MAX_EPOCHS + 1 - assert log(n_drop_g) < MAX_EPOCHS + 1 - n_epochs_g = n_decl_g * n_drop_g - assert log(n_epochs_g) < MAX_EPOCHS + 1 - assert log(n_sphincs_g) < MAX_KEYS - assert log(n_sdup_g) < MAX_KEYS - assert log(n_raw_s_g) < MAX_KEYS - assert log(n_children_g) < MAX_RECURSIONS + 1 - assert log(n_direct_da_g) < 2 - da_meta = StackBuf(2) - hint_witness(da_meta, "da_meta") - n_da_g = da_meta[0] - n_da_dup_g = da_meta[1] - assert log(n_da_g) < MAX_DA_ROOTS + 1 - assert log(n_da_dup_g) < MAX_RECURSIONS * MAX_DA_ROOTS + 2 - da_slots_g = n_da_g * n_da_dup_g - assert log(da_slots_g) < MAX_RECURSIONS * MAX_DA_ROOTS + 2 - - # ---- the epoch groups: geometry pass ---- - # Per group: its epoch, its two message cells, and its declared, duplicate and - # raw-signature counts, each count bounded before it enters a product (up to - # 2^16 factors of exponent < 2^17 stay far from the order 2^64 - 1, so nothing - # wraps). Region bases and the three totals ride a stride-4 chain; the per-group - # values land in heap buffers the later passes and the children's hinted group - # maps read back at runtime. - epochs = HeapBuf(n_epochs_g) - msgs = HeapBuf(n_epochs_g * n_epochs_g) - group_n_keys = HeapBuf(n_epochs_g) - group_n_dups = HeapBuf(n_epochs_g) - group_n_raw = HeapBuf(n_epochs_g) - group_base = HeapBuf(n_epochs_g) - group_slots = HeapBuf(n_epochs_g) - geo = HeapBuf((n_epochs_g * GEN) ** 4) # [base, n_xmss product, n_raw product] - geo[1] = 1 - geo[GEN] = 1 - geo[GEN ** 2] = 1 - for xe in mul_range(1, n_epochs_g): - grp = StackBuf(6) - hint_witness(grp, "group") # epoch, msg_lo, msg_hi, n, n_dup, n_raw - assert log(grp[3]) < MAX_KEYS - assert log(grp[4]) < MAX_KEYS - assert log(grp[5]) < MAX_KEYS - epochs[xe] = grp[0] - msg = msgs * (xe * xe) - msg[1] = grp[1] - msg[GEN] = grp[2] - group_n_keys[xe] = grp[3] - group_n_dups[xe] = grp[4] - group_n_raw[xe] = grp[5] - state = geo * (xe ** 4) - base = state[1] - group_base[xe] = base - slots = grp[3] * grp[4] - group_slots[xe] = slots - nxt = geo * ((xe * GEN) ** 4) - nxt[1] = base * slots - nxt[GEN] = state[GEN] * grp[3] - nxt[GEN ** 2] = state[GEN ** 2] * grp[5] - geo_end = geo * (n_epochs_g ** 4) - xmss_slots_g = geo_end[1] - n_raw_x_g = geo_end[GEN ** 2] - sphincs_slots_g = n_sphincs_g * n_sdup_g - # The sum of every region bounds the coverage indices, so it is what has to sit - # below the minimum memory size. - da_base_g = xmss_slots_g * sphincs_slots_g - n_total_g = da_base_g * da_slots_g - assert log(n_total_g) < MAX_KEYS - - # The proving environment (flock's R1CS and this bytecode) as one digest. It - # rides the statement rather than the bytecode, so nothing here has to know its - # own hash; the outer verifier pins it, and every child statement rebuilt below - # copies it, which is what keeps a whole tree on one bytecode. - fs_seed = StackBuf(2) - hint_witness(fs_seed, "fs_seed") - seed_0 = fs_seed[0] - seed_1 = fs_seed[1] - - # ---- the signer set ---- - g_logs_pow2, g_squares = exponent_tables() - - # ---- data availability ---- - da_roots = HeapBuf(da_slots_g ** 4) - for xd in mul_range(1, da_slots_g): - root = da_roots * (xd ** 4) - hint_witness(root[0:4], "da_roots") - da_0, da_1 = da_list_digest(da_roots, n_da_g) - # One table per scheme, the XMSS one an epoch group at a time: each group its - # declared list (strictly sorted, checked by the outer verifier, which holds it) - # followed by its own duplicate slots. The coverage indices below run over one - # space: the group regions in order, then SPHINCS, then DA. - # - # The digest is a plain BLAKE2s of one string, in whole blocks: both lengths - # and the SPHINCS list's digest, then per group its (epoch, count, - # message) and its key list's digest, each list hashed plainly in turn. Leading - # with both lengths makes the encoding prefix-free, so no set's string is a - # prefix of another's and the digest binds its own lengths. `half` and `odd` are - # hinted per group and pinned by half*half*odd == n with odd in {0, 1}, which - # leaves half = n // 2 and odd = n % 2 as the only solution. - xmss_table = HeapBuf(xmss_slots_g * xmss_slots_g) - sphincs_table = HeapBuf(sphincs_slots_g ** 4) - # The run the set's hash covers: both lengths and the SPHINCS list's digest - # in one block, then two a group. Eight cells a group, so a group's - # header and its key digest are one block each. - signers_run = HeapBuf(n_decl_g ** 8 * GEN ** 4) - signers_run[1] = n_decl_g - signers_run[GEN] = n_sphincs_g - decl_keys = HeapBuf(n_decl_g * GEN) - decl_keys[GEN ** 0] = 1 - for xe in mul_range(1, n_decl_g): - n_keys = group_n_keys[xe] - base = group_base[xe] - decl_keys[xe * GEN] = decl_keys[xe] * n_keys - halves = StackBuf(2) - hint_witness(halves, "pk_halves") - assert log(halves[1]) < 2 - assert log(halves[0]) < MAX_KEYS - assert halves[0] * halves[0] * halves[1] == n_keys - kd_0, kd_1 = key_list_digest(xmss_table * (base * base), halves[0], halves[1], n_keys, g_squares) - group_msg = msgs * (xe * xe) - slot = signers_run * (xe ** 8) * (GEN ** 4) - slot[1] = epochs[xe] - slot[GEN] = n_keys - slot[GEN ** 2] = group_msg[1] - slot[GEN ** 3] = group_msg[GEN] - slot[GEN ** 4] = kd_0 - slot[GEN ** 5] = kd_1 - slot[GEN ** 6] = 0 - slot[GEN ** 7] = 0 - # The duplicate slots ride the same table but outside the hashed prefix, past - # each group's declared keys. Over the whole table: a dropped group has only these. - for xe in mul_range(1, n_epochs_g): - n_keys = group_n_keys[xe] - base = group_base[xe] - dup_ptr = xmss_table * (base * base * n_keys * n_keys) - for xd in mul_range(1, group_n_dups[xe]): - dup = dup_ptr * (xd * xd) - hint_witness(dup[0:2], "dup_pubkeys") - sp_0, sp_1 = sphincs_list_digest(sphincs_table, n_sphincs_g, g_squares) - signers_run[GEN ** 2] = sp_0 - signers_run[GEN ** 3] = sp_1 - # At least one published signature claim or DA root also ensures a nonempty coverage table. - assert decl_keys[n_decl_g] * n_sphincs_g * n_da_g != 1 - set_0, set_1 = signer_set_digest(signers_run, n_decl_g, g_squares) - signers_hash = HeapBuf(WORDS_PER_BLOCK) - signers_hash[1] = set_0 - signers_hash[GEN] = set_1 - for xd in mul_range(1, n_sdup_g): - dup = sphincs_table * ((n_sphincs_g * xd) ** 4) - hint_witness(dup[0:4], "dup_sphincs") - - # ---- coverage ---- - # Every one of the n_total slots is written exactly once: write-once memory - # rejects a second write (the value written is the running count, so two writes - # to one slot disagree), and the count below rejects a missed one. So every - # declared claim is covered by a direct check of its own kind or by a verified - # child. Each writer is confined to its own region, including DA roots. - # The raw XMSS walk runs one loop per epoch group, each signature verified - # against that group's tables (built here, only for a group that holds raw - # signatures); a stride-1 chain threads the running count across the groups. - cover = HeapBuf(n_total_g) - da_cover = cover * da_base_g - merkle_bits = HeapBuf(n_epochs_g ** MERKLE_BIT_CELLS) - tweak_tables = HeapBuf(n_epochs_g ** N_TWEAK_CELLS) - raw_count = HeapBuf(n_epochs_g * GEN) - raw_count[GEN ** 0] = 1 - for xe in mul_range(1, n_epochs_g): - n_raw = group_n_raw[xe] - prefix = raw_count[xe] - if n_raw != 1: - tweak_table = tweak_tables * (xe ** N_TWEAK_CELLS) - group_bits = merkle_bits * (xe ** MERKLE_BIT_CELLS) - fill_xmss_epoch_tables(epochs[xe], group_bits, tweak_table) - slots = group_slots[xe] - base = group_base[xe] - keys = xmss_table * (base * base) - group_msg = msgs * (xe * xe) - for xi in mul_range(1, n_raw): - idx = hint_witness("raw_index") - # A runtime bound, whose `n_total < 2^MIN_LOG_MEM` precondition is - # discharged by `assert log(n_total_g) < MAX_KEYS` above. Without it - # this degenerates to what DEREF alone gives and an index could - # reach past `cover`, which is the whole bijection. The bound is - # this GROUP's region, so a signature verified at this (epoch, - # message) covers no other group's declared key. - assert log(idx) < log(slots) - cover[base * idx] = prefix * xi - verify_sig(group_msg, tweak_table, group_bits, keys * (idx * idx)) - raw_count[xe * GEN] = prefix * n_raw - else: - raw_count[xe * GEN] = prefix - for xj in mul_range(1, n_raw_s_g): - off_hint = hint_witness("sp_raw_index") - assert log(off_hint) < log(sphincs_slots_g) - cover[xmss_slots_g * off_hint] = n_raw_x_g * xj - verify_sig_sphincs(sphincs_table * (off_hint ** 4)) - - if n_direct_da_g == GEN: - root_0, root_1, vector_0, vector_1 = da_verify(g_squares) - expected_0, expected_1, expected_v0, expected_v1 = cover_da_root(da_roots, da_cover, da_slots_g, n_raw_x_g * n_raw_s_g) - assert root_0 == expected_0 - assert root_1 == expected_1 - assert vector_0 == expected_v0 - assert vector_1 == expected_v1 - - # ---- children ---- - child_pi = HeapBuf(n_children_g * n_children_g) - child_fresh = HeapBuf(n_children_g ** DEFER_SIZE) - child_carried = HeapBuf(n_children_g ** DEFER_STMT_CELLS) - written = HeapBuf(n_children_g * GEN) # loop-carried write count, one per child - written[GEN ** 0] = n_raw_x_g * n_raw_s_g * n_direct_da_g - for xc in mul_range(1, n_children_g): - base = written[xc] - # The child's two list lengths, then its groups, rebuilt into its signer-set - # chain by rebuild_child_groups: everything hinted there is pinned by the - # chain, which the child's statement digest carries, so a lie about any of - # it changes the public input its proof has to satisfy. Nothing demands a - # mid-tree statement be canonical (sorted, distinct groups); it still binds - # every claim to its (epoch, message), which is all the group map relies on. - child_meta = StackBuf(2) - hint_witness(child_meta, "child_meta") # n_epochs, n_sphincs - nsub_e_g = child_meta[0] - nsub_s_g = child_meta[1] - assert log(nsub_e_g) < MAX_EPOCHS + 1 - assert log(nsub_s_g) < MAX_KEYS - sub_run = HeapBuf(nsub_e_g ** 8 * GEN ** 4) - sub_run[1] = nsub_e_g - sub_run[GEN] = nsub_s_g - nsub_x_g = rebuild_child_groups(nsub_e_g, sub_run, base, epochs, msgs, group_base, group_slots, n_epochs_g, xmss_table, cover, g_squares) - # Implied by the per-group bounds and the child's own n_total assert; stands - # as documentation. - assert log(nsub_x_g) < MAX_KEYS - nsub_g = nsub_x_g * nsub_s_g - csp_0, csp_1 = child_sphincs_list_digest(sphincs_table, cover, base * nsub_x_g, xmss_slots_g, sphincs_slots_g, nsub_s_g, g_squares) - sub_run[GEN ** 2] = csp_0 - sub_run[GEN ** 3] = csp_1 - sub_set_0, sub_set_1 = signer_set_digest(sub_run, nsub_e_g, g_squares) - sub_hash = HeapBuf(WORDS_PER_BLOCK) - sub_hash[1] = sub_set_0 - sub_hash[GEN] = sub_set_1 - carried = child_carried * xc ** DEFER_STMT_CELLS - hint_witness(carried[0:DEFER_STMT_CELLS], "child_defer") - nsub_da_g = hint_witness("child_da_count") - assert log(nsub_da_g) < MAX_DA_ROOTS + 1 - child_da = HeapBuf(nsub_da_g ** 4) - for xd in mul_range(1, nsub_da_g): - root = child_da * (xd ** 4) - root_0, root_1, vector_0, vector_1 = cover_da_root(da_roots, da_cover, da_slots_g, base * nsub_g * xd) - root[1] = root_0 - root[GEN] = root_1 - root[GEN ** 2] = vector_0 - root[GEN ** 3] = vector_1 - child_da_0, child_da_1 = da_list_digest(child_da, nsub_da_g) - pi_0, pi_1 = statement_digest(seed_0, seed_1, sub_hash, child_da_0, child_da_1, carried) - pi = xc * xc - child_pi[pi] = pi_0 - child_pi[pi * GEN] = pi_1 - verify_sub(pi_0, pi_1, seed_0, seed_1, g_logs_pow2, g_squares, child_fresh * xc ** DEFER_SIZE) - written[xc * GEN] = base * nsub_g * nsub_da_g - assert written[n_children_g] == n_total_g - - # ---- this node's own deferred claims ---- - defer_stmt = HeapBuf(DEFER_STMT_CELLS) - if n_children_g == 1: - # A leaf has nothing to batch, so it defers the three fixed polynomials at - # the all-zeros point. Their values ride a hint and are checked nowhere - # here: the outer verifier recomputes them and rebuilds the statement, so a - # lie changes the public input rather than the claim. - leaf_values = StackBuf(3) - hint_witness(leaf_values, "leaf_defer") - for k in unroll(0, BYTECODE_VARS): - defer_stmt[GEN ** k] = 0 - defer_stmt[GEN ** DEFER_STMT_BC_VALUE] = leaf_values[0] - for k in unroll(0, 2 * K_LOG): - defer_stmt[GEN ** (DEFER_STMT_MAT_POINT + k)] = 0 - defer_stmt[GEN ** DEFER_STMT_A_VALUE] = leaf_values[1] - defer_stmt[GEN ** DEFER_STMT_B_VALUE] = leaf_values[2] - else: - aggregate_claims(n_children_g, child_pi, child_fresh, child_carried, defer_stmt) - - own_0, own_1 = statement_digest(seed_0, seed_1, signers_hash, da_0, da_1, defer_stmt) - pub_ptr = GEN ** 0 - assert pub_ptr[1] == own_0 - assert pub_ptr[GEN] == own_1 - return - - -def aggregate_claims(n_children_g, child_pi, child_fresh, child_carried, defer_stmt): - # Every child contributes two claims per fixed polynomial: the one IT deferred - # (carried in its statement) and the fresh one raised by verifying its proof. A - # fresh transcript binds all of them, samples the batching coefficients, and two - # sumchecks (one for the bytecode, one shared by the two matrices) reduce the - # lot to one claim each, which is what the node then defers in its own - # statement. - # - # A carried claim is a plain point, so its weight is an eq product; a fresh one - # carries flock's zerocheck/lincheck structure and keeps the succinct weight the - # sub-verifier exported. That is the only asymmetry. - bc_msgs = HeapBuf(2 * BYTECODE_VARS) - hint_witness(bc_msgs[0:2 * BYTECODE_VARS], "bc_sumcheck_msgs") - mat_msgs = HeapBuf(4 * K_LOG) - hint_witness(mat_msgs[0:4 * K_LOG], "mat_sumcheck_msgs") - bytecode_star = hint_witness("bc_star_hint") - mat_stars = StackBuf(2) - hint_witness(mat_stars[0:2], "mat_stars_hint") - - # ---- one transcript over every child's statement and both its claim sets ---- - fresh_row = HeapBuf(n_children_g) - carried_row = HeapBuf(n_children_g) - agg_fs = [AGG_SEED_0, AGG_SEED_1] - agg_fs = obs(agg_fs, n_children_g) - absorb = HeapBuf((n_children_g * GEN) ** PAIR_SLOTS) - absorb[GEN ** 0] = agg_fs[0] - absorb[GEN ** 1] = agg_fs[1] - for xc in mul_range(1, n_children_g): - row = absorb * xc ** PAIR_SLOTS - st = [row[GEN ** 0], row[GEN ** 1]] - pi = xc * xc - st = obs(st, child_pi[pi]) - st = obs(st, child_pi[pi * GEN]) - fresh = child_fresh * xc ** DEFER_SIZE - fresh_row[xc] = fresh - for k in unroll(0, DEFER_SIZE): - st = obs(st, fresh[GEN ** k]) - carried = child_carried * xc ** DEFER_STMT_CELLS - carried_row[xc] = carried - for k in unroll(0, DEFER_STMT_CELLS): - st = obs(st, carried[GEN ** k]) - row[GEN ** PAIR_SLOTS] = st[0] - row[GEN ** (PAIR_SLOTS + 1)] = st[1] - absorbed = absorb * n_children_g ** PAIR_SLOTS - - # ---- bytecode batching sumcheck (BYTECODE_VARS variables, 2 per child) ---- - # Fresh and carried share the bytecode layout (point, then value), so the two - # differ only in which buffer they come from. - lam_bc = HeapBuf(n_children_g * n_children_g) - bc_chain = HeapBuf((n_children_g * GEN) ** ACC_SLOTS) - bc_chain[GEN ** ACC_FS0] = absorbed[GEN ** 0] - bc_chain[GEN ** ACC_FS1] = absorbed[GEN ** 1] - bc_chain[GEN ** ACC_VALUE] = 0 - for xc in mul_range(1, n_children_g): - row = bc_chain * xc ** ACC_SLOTS - st = [row[GEN ** ACC_FS0], row[GEN ** ACC_FS1]] - st, lam_fresh = squeeze(st) - st, lam_carried = squeeze(st) - pair = xc * xc - lam_bc[pair] = lam_fresh - lam_bc[pair * GEN] = lam_carried - fresh = fresh_row[xc] - carried = carried_row[xc] - nxt = row * GEN ** ACC_SLOTS - nxt[GEN ** ACC_FS0] = st[0] - nxt[GEN ** ACC_FS1] = st[1] - nxt[GEN ** ACC_VALUE] = row[GEN ** ACC_VALUE] + lam_fresh * fresh[GEN ** FRESH_BC_VALUE] + lam_carried * carried[GEN ** DEFER_STMT_BC_VALUE] - bc_end = bc_chain * n_children_g ** ACC_SLOTS - agg_fs = [bc_end[GEN ** ACC_FS0], bc_end[GEN ** ACC_FS1]] - bc_running = bc_end[GEN ** ACC_VALUE] - bc_point = HeapBuf(BYTECODE_VARS) - fs0, fs1, bc_running = batch_sumcheck(agg_fs[0], agg_fs[1], bc_msgs, bc_running, bc_point, BYTECODE_VARS) - agg_fs = [fs0, fs1] - bc_wsum = HeapBuf(n_children_g * GEN) - bc_wsum[GEN ** 0] = 0 - for xc in mul_range(1, n_children_g): - fresh = fresh_row[xc] - carried = carried_row[xc] - eq_fresh = GEN ** 0 - eq_carried = GEN ** 0 - for k in unroll(0, BYTECODE_VARS): - rk = bc_point[GEN ** k] - eq_fresh *= (1 + fresh[GEN ** k] + rk) - eq_carried *= (1 + carried[GEN ** k] + rk) - pair = xc * xc - bc_wsum[xc * GEN] = bc_wsum[xc] + lam_bc[pair] * eq_fresh + lam_bc[pair * GEN] * eq_carried - assert bc_running == bytecode_star * bc_wsum[n_children_g] - - # ---- matrix batching sumcheck (2*K_LOG variables, 3 claims per child) ---- - # The fresh claim is one value against A0 weighted by lincheck's alpha plus B0; - # a carried claim is one value per matrix at a shared point. - lam_mat = HeapBuf(n_children_g ** 3) - mat_chain = HeapBuf((n_children_g * GEN) ** ACC_SLOTS) - mat_chain[GEN ** ACC_FS0] = agg_fs[0] - mat_chain[GEN ** ACC_FS1] = agg_fs[1] - mat_chain[GEN ** ACC_VALUE] = 0 - for xc in mul_range(1, n_children_g): - row = mat_chain * xc ** ACC_SLOTS - st = [row[GEN ** ACC_FS0], row[GEN ** ACC_FS1]] - st, lam_fresh = squeeze(st) - st, lam_a = squeeze(st) - st, lam_b = squeeze(st) - triple = xc ** 3 - lam_mat[triple] = lam_fresh - lam_mat[triple * GEN] = lam_a - lam_mat[triple * GEN ** 2] = lam_b - fresh = fresh_row[xc] - carried = carried_row[xc] - nxt = row * GEN ** ACC_SLOTS - nxt[GEN ** ACC_FS0] = st[0] - nxt[GEN ** ACC_FS1] = st[1] - nxt[GEN ** ACC_VALUE] = row[GEN ** ACC_VALUE] + lam_fresh * fresh[GEN ** FRESH_MATPART] + lam_a * carried[GEN ** DEFER_STMT_A_VALUE] + lam_b * carried[GEN ** DEFER_STMT_B_VALUE] - mat_end = mat_chain * n_children_g ** ACC_SLOTS - agg_fs = [mat_end[GEN ** ACC_FS0], mat_end[GEN ** ACC_FS1]] - mat_running = mat_end[GEN ** ACC_VALUE] - mat_point = HeapBuf(2 * K_LOG) - fs0, fs1, mat_running = batch_sumcheck(agg_fs[0], agg_fs[1], mat_msgs, mat_running, mat_point, 2 * K_LOG) - agg_fs = [fs0, fs1] - # Terminal weights. A fresh claim's is U_t(r*) = urow_t(r*_row) * wcol_t(r*_col), - # with row_weight = (sum_i L_i(zz_t) eq(r*[0..6], i)) * eq(zchi_t, r*[6..K_LOG]) - # and col_weight = (sum_i z_partial_t[i] eq(r*[K_LOG..K_LOG+6], i)) * prod_j (1 + - # lrr_j + r*[2*K_LOG-1-j]) (the lincheck binds column variables top-down). A - # carried claim's is a plain eq over all 2*K_LOG coordinates. - eq_rows = HeapBuf(2 ** (K_SKIP + 1) - 2) - eqtree(mat_point, eq_rows, K_SKIP) - eq_cols = HeapBuf(2 ** (K_SKIP + 1) - 2) - eqtree(mat_point * GEN ** K_LOG, eq_cols, K_SKIP) - w_sums = HeapBuf((n_children_g * GEN) ** PAIR_SLOTS) # the A and B weight sums - w_sums[GEN ** 0] = 0 - w_sums[GEN ** 1] = 0 - for xc in mul_range(1, n_children_g): - fresh = fresh_row[xc] - row_nums = StackBuf(2 ** K_SKIP) - lag64(fresh[GEN ** FRESH_Z_SKIP], row_nums, 0) - row_weight = 0 - for i in unroll(0, 2 ** K_SKIP): - row_weight += row_nums[i] * eq_rows[GEN ** (2 ** K_SKIP - 2 + i)] - row_weight *= LAGRANGE_INV_S - for k in unroll(0, LINCHECK_ROUNDS): - row_weight *= (1 + fresh[GEN ** (FRESH_ZCHI + k)] + mat_point[GEN ** (K_SKIP + k)]) - col_weight = 0 - for i in unroll(0, 2 ** K_SKIP): - col_weight += fresh[GEN ** (FRESH_Z_PARTIAL + i)] * eq_cols[GEN ** (2 ** K_SKIP - 2 + i)] - for j in unroll(0, LINCHECK_ROUNDS): - col_weight *= (1 + fresh[GEN ** (FRESH_LINCHECK_RS + j)] + mat_point[GEN ** (2 * K_LOG - 1 - j)]) - weight_u = row_weight * col_weight - carried = carried_row[xc] - eq_carried = GEN ** 0 - for k in unroll(0, 2 * K_LOG): - eq_carried *= (1 + carried[GEN ** (DEFER_STMT_MAT_POINT + k)] + mat_point[GEN ** k]) - triple = xc ** 3 - lam_fresh = lam_mat[triple] - row = w_sums * xc ** PAIR_SLOTS - row[GEN ** PAIR_SLOTS] = row[GEN ** 0] + lam_fresh * weight_u + lam_mat[triple * GEN] * eq_carried - row[GEN ** (PAIR_SLOTS + 1)] = row[GEN ** 1] + lam_fresh * fresh[GEN ** FRESH_ALPHA] * weight_u + lam_mat[triple * GEN ** 2] * eq_carried - w_end = w_sums * n_children_g ** PAIR_SLOTS - a_star = mat_stars[0] - b_star = mat_stars[1] - assert mat_running == a_star * w_end[GEN ** 0] + b_star * w_end[GEN ** 1] - - for k in unroll(0, BYTECODE_VARS): - defer_stmt[GEN ** k] = bc_point[GEN ** k] - defer_stmt[GEN ** DEFER_STMT_BC_VALUE] = bytecode_star - for k in unroll(0, 2 * K_LOG): - defer_stmt[GEN ** (DEFER_STMT_MAT_POINT + k)] = mat_point[GEN ** k] - defer_stmt[GEN ** DEFER_STMT_A_VALUE] = a_star - defer_stmt[GEN ** DEFER_STMT_B_VALUE] = b_star - return diff --git a/crates/rec_aggregation/src/aggregation.rs b/crates/rec_aggregation/src/aggregation.rs deleted file mode 100644 index 4a8182a83..000000000 --- a/crates/rec_aggregation/src/aggregation.rs +++ /dev/null @@ -1,5284 +0,0 @@ -//! Recursive proofs of XMSS and SPHINCS signature claims and LeanDA blob well-formedness. -//! One bytecode (`guests/lean_ethereum.py`) serves every node of an aggregation tree. -//! -//! A node verifies `n_raw_xmss` XMSS signatures, `n_raw_sphincs` SPHINCS -//! signatures and `n_children` sub-proofs **of this same bytecode**, and -//! by default publishes the sorted deduplicated union of their signer sets. The XMSS -//! signers are grouped by `(epoch, message)`, so one epoch may carry several -//! messages, each its own group. A SPHINCS -//! signer carries its own message, so that half of the statement is a list of -//! `(key, message)` pairs. Coverage is what carries the security claim: a write-once slot per -//! declared signer, written once by each raw signature and each child key, plus -//! a final count, so every declared signer is backed by a real signature or a -//! verified child. -//! -//! Those slots are one contiguous region per XMSS group and one for -//! SPHINCS, so the one range check a write already needs also keeps a -//! signature off another group's declared keys, of either scheme: that is what -//! makes the published split mean which scheme verified which key against -//! which `(epoch, message)`, at every level of the tree. -//! -//! A duplicate slot sits outside the prefix the digest hashes, so a key in one is -//! covered and not claimed. `aggregate`'s `declare` rests on that, the table also -//! holding undeclared groups so a whole `(epoch, message)` can go unpublished. -//! A child's groups need -//! not equal its parent's: a hinted map, checked by the guest, ties each -//! non-empty child group to a parent group with the same epoch and message. -//! An XMSS slot holds the -//! key's two cells and a SPHINCS slot four, its key and its message, so the -//! guest reads each SPHINCS signature's message out of the slot it verifies. -//! -//! The bytecode is compiled to a fixed point on its own size -//! ([`unified_guest`]): the recursion placeholders depend on the inner bytecode -//! size, and here the inner bytecode is this one. Its digest does not need a -//! fixed point, riding the statement instead of the code. -//! -//! Three fixed polynomials (the stacked bytecode and flock's `A0`/`B0`) are too -//! big to evaluate in-circuit, so each node exports one deferred claim on each -//! and batches its children's carried claims with the fresh ones its -//! verifications raise (`doc/leanvm/main.tex` §Deferred evaluation claims). Only -//! the root's are discharged natively, by [`EthereumProof::verify`]. -//! -//! `gen_verify` derives the guest's whole witness for a child from the real -//! `cpu::layout` of the inner program and the summary of a real `cpu::verify` -//! run, so there is no hand-mirrored copy of the protocol to drift. - -use bincode::Options as _; -use pcs::whir::{MAX_LOG_INV_RATE, MIN_LOG_INV_RATE}; -use std::collections::{BTreeMap, BTreeSet}; -use std::ops::Range; - -use lean_compiler::{compile, parse_with_replacements}; -use lean_da::{BLOB_SYMBOLS, CELL_SYMBOLS, CELLS_PER_ROW, CODEWORD_SYMBOLS}; -pub use lean_da::{DA_LOG_CELL, DA_LOG_K, DA_MAX_ROWS}; -use lean_vm::cpu::{Program, ProveError, prove, verify}; -use lean_vm::leaf::{Block, Coord}; -use lean_vm::transcript::FiatShamirState; -use primitives::field::{F64, F192, G, g_pow}; -use primitives::multilinear::mle_eval_par; -use xmss::{XmssPublicKey, XmssSignature}; - -use sphincs::{SphincsPublicKey, SphincsSignature}; - -/// One SPHINCS claim: a key, and the message it signed. Where an XMSS group -/// shares one message, every SPHINCS signer carries its own. -pub type SphincsClaim = (SphincsPublicKey, sphincs::Message); - -/// The XMSS signers sharing one epoch: the epoch, the message they all signed -/// at it, and their strictly sorted keys. -#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub struct XmssClaimGroup { - pub epoch: xmss::Epoch, - pub message: xmss::Message, - pub keys: Vec, -} - -/// Why the guest reads every `q_flock` slot claim's instance point off `chi`: a -/// virtual value column is referenced only by its own table's bus blocks, which -/// the table sumcheck settles, so no framework block can raise one at `zeta`. -const VALCOL_FRAMEWORK: &str = "a framework block must not reference a virtual value column"; -const RECURSION_AGG_LABEL: &[u8] = b"leanvm/recursion-aggregation/v1"; - -/// The most earlier aggregates one [`aggregate`] call can take, so the arity of -/// an aggregation tree. -pub const MAX_RECURSIONS: usize = 16; - -/// Exclusive cap on coverage slots: signatures of both schemes and DA roots, -/// including duplicates and omitted claims. Exactly this many is already too many. -/// -/// It is where the coverage indices' range check has to sit to stay below -/// `2^MIN_LOG_MEM`, so the bound means the same at every announced memory size. -/// The guest checks it against a COMPILE-TIME bound, so raising it past -/// `2^MIN_LOG_MEM` fails the build rather than weakening the index bound. -pub const MAX_KEYS: usize = 1 << 16; - -/// Maximum number of LeanDA roots in one statement, including inherited and direct roots. -pub const MAX_DA_ROOTS: usize = 16; -const _: () = assert!(MAX_DA_ROOTS < MAX_KEYS); - -/// The most [`XmssClaimGroup`]s one aggregate can carry: headroom over the few epochs -/// expected in practice. An empty group is in-circuit unprovable, its list hash -/// having no valid window split, so this is a bound on cost, not on soundness. -pub const MAX_EPOCHS: usize = 1024; - -/// Blocks the guest absorbs per loop frame when it hashes a declared list, and so -/// how many share one byte-counter base (`doc/leanvm` §Byte counters for a hash of -/// runtime length). Larger amortizes the base's bit decomposition over more blocks -/// and costs bytecode in the tail's dispatch arms; the split it induces is what -/// [`signers_split`] hands the guest. -const SIGNERS_WINDOW: usize = 32; -// The counter split is a bit split: a window's base is `64·SIGNERS_WINDOW·q`, which -// has to be a power of two for its bits to sit clear of the window's own offsets. -const _: () = assert!(SIGNERS_WINDOW.is_power_of_two()); - -/// The window bound the guest range-checks its hinted window count against. A -/// range check takes a COUNT (`assert log(x) < k` bounds the exponent by `k`), so -/// this has to cover the most windows a list can hold, which a bit width would not: -/// under the bound every list still hashes, so the mistake shows up only past it. -const SIGNERS_MAX_WINDOWS: usize = MAX_KEYS / SIGNERS_WINDOW; -// The widest list is one block a claim, so at most MAX_KEYS blocks, of which the -// last is absorbed apart. The set's own string is 2 + 2·MAX_EPOCHS blocks, hashed -// the same way, so it needs the bound too. -const _: () = assert!((MAX_KEYS - 1) / SIGNERS_WINDOW < SIGNERS_MAX_WINDOWS); -const _: () = assert!((2 * MAX_EPOCHS + 1) / SIGNERS_WINDOW < SIGNERS_MAX_WINDOWS); -// The guest decomposes a count into SIGNERS_COUNT_BITS bits and shifts the result -// left to make a byte counter, so a count has to fit and the shift must not reduce. -const SIGNERS_COUNT_BITS: u32 = MAX_KEYS.ilog2(); -const _: () = assert!(MAX_KEYS.is_power_of_two() && 2 * MAX_EPOCHS + 2 < 1 << SIGNERS_COUNT_BITS); -const _: () = assert!(SIGNERS_COUNT_BITS + 6 + SIGNERS_WINDOW.ilog2() <= 64); - -// The guest bakes a bytecode claim's width from `N_TUPLE_BITS` while `bytecode_vars` -// reads it off the stacked table, which is `N_BYTECODE_SELECTORS` wide. Two constants -// that happen to agree: were they to drift, a leaf's claim point would be one length in -// the guest and another in the statement, and nothing else would notice. -const _: () = assert!(lean_vm::leaf::N_TUPLE_BITS == lean_vm::leaf::N_BYTECODE_SELECTORS); -// The epoch fills a tweak's four-byte index field, so a longer lifetime would -// need a weight per bit that `xmss::make_tweak` cannot express. -const _: () = assert!(xmss::LOG_LIFETIME <= 32); -// The guest's `WOTS_PK_BLOCKS = (2 + V) / 4` truncates, so a bad `V` would drop -// the last tips. -const _: () = assert!((2 + xmss::V).is_multiple_of(4)); -// The SPHINCS side of the same shape. `SP_LEAF_BLOCKS = (2 + V) / 4` and -// `SP_ROOT_BLOCKS = (2 + NUM_FTS_TREES) / 4` truncate, and a truncated loop -// would leave the last tips or roots out of the hash while the signature still -// carries them: revealed values no longer bound by the leaf they belong to. -const _: () = assert!((2 + sphincs::V).is_multiple_of(4)); -const _: () = assert!((2 + sphincs::NUM_FTS_TREES).is_multiple_of(4)); -// The guest reads the message digest's bits out of three 64-bit lanes, and a -// dynamically sized `HeapBuf` gets no compile-time index check, so a wider -// digest would read leaf indices from cells nothing writes. -const _: () = assert!(sphincs::DIGEST_BITS <= 3 * 64); -// The guest packs each tweak field into its own 32-bit word: p at bit 32, -// tau at bit 64, and j at bit 96. -const _: () = assert!(sphincs::H <= 32); -const _: () = assert!(sphincs::CHAIN_LEN * sphincs::V < 1 << 32); -const _: () = assert!(sphincs::A <= 32 && sphincs::HEIGHTS[0] <= 32); - -/// A count as the guest carries it: in the exponent, `g^n`. -fn count(n: usize) -> F192 { - F192::new(g_pow(n).0, 0, 0) -} - -/// A field element as the decimal `u128` literal the zkDSL parser accepts. -fn dsl_u128(value: F192) -> u128 { - assert_eq!(value.c2, 0, "u128 DSL literal cannot encode the top F192 limb"); - (value.c0 as u128) | ((value.c1 as u128) << 64) -} - -fn f192_literal(f: F192) -> String { - format!("f192({},{},{})", f.c0, f.c1, f.c2) -} - -/// Pack the Fiat-Shamir state's four K lanes as two canonical 128-bit VM cells. -fn pack_state(state: [F64; 4]) -> [F192; 2] { - [ - F192::new(state[0].0, state[1].0, 0), - F192::new(state[2].0, state[3].0, 0), - ] -} - -/// Pack a 32-byte Merkle node as the same canonical 128+128 cell pair used by -/// the VM's sole BLAKE2s representation. -fn pack_hash_state(hash: &[u8; 32]) -> [F192; 2] { - let word_at = |offset: usize| u64::from_le_bytes(hash[offset..offset + 8].try_into().unwrap()); - [ - F192::new(word_at(0), word_at(8), 0), - F192::new(word_at(16), word_at(24), 0), - ] -} - -/// A 16-byte native value as one canonical 128-bit cell. -fn pack_16_bytes(bytes: &[u8]) -> F192 { - let word_at = |offset: usize| u64::from_le_bytes(bytes[offset..offset + 8].try_into().unwrap()); - F192::new(word_at(0), word_at(8), 0) -} - -/// A public key as the two cells the guest hashes and `verify_sig` reads: the -/// root then the public parameter. Both schemes lay a key out the same way, and -/// the statement keeps them in separate lists rather than telling them apart by -/// their bytes. -fn key_cells(pk: &XmssPublicKey) -> [F192; 2] { - [pack_16_bytes(&pk.merkle_root), pack_16_bytes(&pk.public_param)] -} - -/// A run of canonical 128-bit cells as the byte string BLAKE2s hashes: each cell -/// is its two low limbs, little-endian, which is the order the VM's compression -/// reads a memory cell in. -fn cell_bytes(cells: impl IntoIterator) -> Vec { - let mut bytes = Vec::new(); - for cell in cells { - bytes.extend_from_slice(&cell.c0.to_le_bytes()); - bytes.extend_from_slice(&cell.c1.to_le_bytes()); - } - bytes -} - -/// How the guest splits a list's `n - 1` non-final blocks: whole windows, then the -/// tail. Its own product identity and range check pin both, so this only has to -/// agree with them (`sphincs_list_digest` in the guest). A list is never empty -/// here: a group holds at least one key, and both claim lists are guarded at the -/// call, since the guest hashes an empty one without a split at all. -fn signers_split(blocks: usize) -> Vec { - assert!(blocks > 0, "an empty list has no window split"); - let leading = blocks - 1; - vec![count(leading / SIGNERS_WINDOW), count(leading % SIGNERS_WINDOW)] -} - -/// One epoch group's declared keys under plain BLAKE2s: 32 bytes a key, so the -/// hashed string is `32n` bytes and only its last block is partial. The guest -/// computes this same digest a window of blocks at a time (`key_list_digest`). -fn key_list_digest(keys: &[XmssPublicKey]) -> [F192; 2] { - let cells = keys.iter().flat_map(key_cells); - pack_hash_state(&primitives::hash::hash(&cell_bytes(cells))) -} - -/// The declared SPHINCS claims under plain BLAKE2s: one 64-byte block per claim, -/// its key then the message it signed, so the hashed string is exactly `64n` bytes -/// and an empty list hashes the empty string. The guest computes this same digest a -/// window of blocks at a time (`sphincs_list_digest`). -fn sphincs_list_digest(signers: &[SphincsClaim]) -> [F192; 2] { - let cells = signers.iter().flat_map(sphincs_signer_cells); - pack_hash_state(&primitives::hash::hash(&cell_bytes(cells))) -} - -/// A SPHINCS signer as the four cells the guest hashes and `verify_sig_sphincs` -/// reads: the key, then the message that key signed. -fn sphincs_signer_cells((pk, message): &SphincsClaim) -> [F192; 4] { - [ - pack_16_bytes(&pk.root), - pack_16_bytes(&pk.public_param), - pack_16_bytes(&message[..16]), - pack_16_bytes(&message[16..]), - ] -} - -/// One XMSS tweak as the cell the guest adds into: `xmss::make_tweak`'s own -/// output, packed. The guest holds no byte layout of its own, building a tweak -/// as this constant half plus one [`tweak_index_weight`] per set epoch bit, so a -/// field that moves in `make_tweak` moves both halves together. -fn tweak_cell(tweak_type: u8, sub_position: u32) -> F192 { - pack_16_bytes(&xmss::make_tweak(tweak_type, sub_position, 0)) -} - -/// What bit `b` of the epoch weighs in a tweak's index field, so an index is its -/// set bits summed. The one property of the layout this assumes is that the -/// index field is linear in the index. Subtract the constant protocol prefix. -fn tweak_index_weight(b: usize) -> F192 { - pack_16_bytes(&xmss::make_tweak(0, 0, 1 << b)) + pack_16_bytes(&xmss::make_tweak(0, 0, 0)) -} -/// The signer-set digest: plain BLAKE2s of one byte string, laid out in whole -/// 64-byte blocks so the guest can absorb it four cells at a time -/// (`signer_set_digest` there). The first block carries both list lengths and the -/// SPHINCS list's own digest, followed by two blocks a group: its `(epoch, -/// count, message)`, then its key list's digest. Leading with both lengths makes -/// the encoding prefix-free, so no set's string is a prefix of another's, and the -/// digest binds its own lengths, the groups' epochs and messages, and every split. -/// The two list digests carry the bulk, each a stock hash of its own -/// ([`key_list_digest`], [`sphincs_list_digest`]). -fn signers_hash(xmss_signers: &[XmssClaimGroup], sphincs_signers: &[SphincsClaim]) -> [F192; 2] { - let sphincs = sphincs_list_digest(sphincs_signers); - let mut cells = vec![ - count(xmss_signers.len()), - count(sphincs_signers.len()), - sphincs[0], - sphincs[1], - ]; - for XmssClaimGroup { epoch, message, keys } in xmss_signers { - cells.extend([ - F192::new(*epoch as u64, 0, 0), - count(keys.len()), - pack_16_bytes(&message[..16]), - pack_16_bytes(&message[16..]), - ]); - let keys = key_list_digest(keys); - cells.extend([keys[0], keys[1], F192::ZERO, F192::ZERO]); - } - pack_hash_state(&primitives::hash::hash(&cell_bytes(cells))) -} - -/// The claims on the three fixed polynomials that a node defers rather than -/// evaluating in-circuit: one point and value on the stacked bytecode, one point -/// and two values on flock's `A0`/`B0` (`doc/leanvm/main.tex` §Deferred -/// evaluation claims). -/// -/// Only the points are transmitted; the values are derived from them on receipt, -/// so a prover that lies about a value changes the statement its proof has to -/// satisfy rather than the claim anyone checks. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct DeferredClaim { - bytecode_point: Vec, - bytecode_value: F192, - matrix_point: Vec, - matrix_a_value: F192, - matrix_b_value: F192, -} - -impl DeferredClaim { - /// What a leaf defers: the all-zeros point on each polynomial, where the - /// value is just the table's first entry. - fn leaf() -> Self { - Self::recompute( - vec![F192::ZERO; bytecode_vars()], - vec![F192::ZERO; 2 * flock::hash::K_LOG], - ) - .expect("the all-zeros point has the right shape") - } - - /// Evaluate the three fixed polynomials at `bytecode_point` / `matrix_point`. - fn recompute(bytecode_point: Vec, matrix_point: Vec) -> Result { - let klog = flock::hash::K_LOG; - if bytecode_point.len() != bytecode_vars() || matrix_point.len() != 2 * klog { - return Err(AggregateVerifyError::MalformedClaim); - } - // Every leaf defers the all-zeros point, where the bytecode polynomial is - // just its table's first entry. Still worth special-casing that half: the - // general path is a pass over 2^23 entries, and a leaf is the aggregate - // people verify most. The matrix half needs no special case, its walk - // being O(circuit) either way. - let zero_point = bytecode_point.iter().chain(&matrix_point).all(|x| *x == F192::ZERO); - let bytecode_value = if zero_point { - F192::from(stacked_bytecode()[0]) - } else { - let sp = tracing::info_span!("bytecode mle").entered(); - let value = mle_eval_par(stacked_bytecode(), &bytecode_point); - drop(sp); - value - }; - let sp = tracing::info_span!("matrix walk").entered(); - let eq_r = pcs::whir::build_eq_table_ext(&matrix_point[..klog]); - let eq_c = pcs::whir::build_eq_table_ext(&matrix_point[klog..]); - let (matrix_a_value, matrix_b_value) = flock::hash::bilinear_walk_pair(&eq_r, &eq_c); - drop(sp); - Ok(Self { - bytecode_point, - bytecode_value, - matrix_point, - matrix_a_value, - matrix_b_value, - }) - } - - /// The cells the statement digest absorbs, in the guest's `defer_stmt` order. - fn cells(&self) -> Vec { - let mut cells = self.bytecode_point.clone(); - cells.push(self.bytecode_value); - cells.extend_from_slice(&self.matrix_point); - cells.push(self.matrix_a_value); - cells.push(self.matrix_b_value); - cells - } -} - -/// The statement's fixed header, ahead of the deferred cells: the seed, the -/// signer-set digest (which itself binds the epoch groups and every count), and -/// the DA root-list digest. Fed to the guest as `STMT_HEADER`, so the two cannot drift. -const STATEMENT_HEADER: usize = 6; - -/// A plain BLAKE2s over a lane stream, zero-filled to a whole 64-byte block: -/// what the guest gets by streaming four 128-bit cells a block. -fn lane_hash(lanes: impl Iterator) -> [F192; 2] { - let mut bytes: Vec = lanes.flat_map(u64::to_le_bytes).collect(); - bytes.resize(bytes.len().next_multiple_of(64), 0); - pack_hash_state(&primitives::hash::hash(&bytes)) -} - -/// A node's public statement, hashed to the two words the VM publishes. The -/// guest's `statement_digest` computes exactly this, both for itself and when it -/// rebuilds a child's, which is what forces a whole tree onto one bytecode and -/// each child's `(epoch, message)` groups, bound by the signer-set digest, -/// onto its parent's list. -/// -/// Fixed-length preimage, so a plain BLAKE2s, with no domain tag of its own: the -/// header leads with the environment digest, which binds this bytecode and -/// flock's R1CS and so already separates the preimage from every other use of -/// BLAKE2s here. The header is hashed as the canonical cells it already is (two -/// lanes each, whence the assert, the guest being unable to hash a third), then -/// all three lanes of each deferred cell. -fn statement_digest(signers_hash: [F192; 2], da_digest: [u8; 32], defer: &DeferredClaim) -> [F192; 2] { - let seed = lean_vm::cpu::fs_seed(unified_guest()); - let da_digest = pack_hash_state(&da_digest); - let header = [ - seed[0], - seed[1], - signers_hash[0], - signers_hash[1], - da_digest[0], - da_digest[1], - ]; - assert_eq!(header.len(), STATEMENT_HEADER); - let mut cells = defer.cells(); - if !cells.len().is_multiple_of(2) { - cells.push(F192::ZERO); // the guest pairs the odd cell with a zero scalar - } - let head = header.iter().flat_map(|x| { - assert_eq!(x.c2, 0, "a header value is a canonical cell"); - [x.c0, x.c1] - }); - lane_hash(head.chain(cells.iter().flat_map(|x| [x.c0, x.c1, x.c2]))) -} - -/// The deferred-claim data the guest binds to the outer public input: the outer -/// verifier checks each claim natively (`doc/leanvm/main.tex` §Deferred evaluation claims; -/// n_rec = 1 forwards fresh claims without batching). -struct DeferredSubproof { - public_input: [F192; 2], - bytecode_row_point: Vec, - bytecode_selector_point: Vec, - bytecode_value: F192, - matrix_a_coefficient: F192, - skip_point: F192, - zerocheck_row_point: Vec, - lincheck_round_point: Vec, - lincheck_terminal_values: Vec, - matrix_claim: F192, -} - -/// A proof that every key in [`Self::xmss_signers`] signed its group's message -/// at its group's epoch under XMSS, and that every `(key, message)` in -/// [`Self::sphincs_signers`] is backed by a valid SPHINCS signature. Each root in -/// [`Self::da_commitments`] also attests that the committed blob rows are valid Reed-Solomon codewords. -/// These claims can be established directly or carried from verified child proofs. -/// -/// The two lists describe the signature claims and remain separate because the proof says which scheme -/// verified which key (the module docs give the coverage argument). Until -/// [`Self::verify`] returns `Ok` they are claims, not attestation, and even then -/// the epochs and messages are the prover's, so a caller that reads either list -/// as attestation of something must compare it against what it expected. -/// -/// **Neither length counts signers, only claims.** One key may appear in several -/// XMSS groups or under several SPHINCS messages, so a committee threshold has -/// to count distinct keys itself. -#[derive(Clone, Debug)] -pub struct EthereumProof { - /// The XMSS signers: strictly increasing `(epoch, message)` pairs, - /// each group non-empty and strictly sorted. May be empty. The claims of both schemes together are - /// strictly fewer than [`MAX_KEYS`]. - xmss_signers: Vec, - /// Strictly sorted and deduplicated on the whole `(key, message)` pair. - sphincs_signers: Vec, - /// Strictly sorted LeanDA roots. Their list digest rides the public statement. - da_roots: Vec<[u8; 32]>, - /// What this aggregate defers to whoever discharges it: its parent, in - /// circuit, or [`Self::verify`], natively. - defer: DeferredClaim, - proof: lean_vm::cpu::Proof, -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum AggregateVerifyError { - /// The signer set is unsorted, holds a duplicate, or has [`MAX_KEYS`] keys or more. - MalformedSignerSet, - /// The DA root list is unsorted, holds duplicates, or exceeds [`MAX_DA_ROOTS`]. - MalformedDaCommitments, - /// A deferred claim's point has the wrong number of coordinates. - MalformedClaim, - /// The bytes are not a valid encoding of an aggregate. - MalformedEncoding, - /// The snark itself did not verify. - Snark(lean_vm::cpu::CpuError), -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum AggregationError { - /// The union of the `(epoch, message)` groups, over the raw signatures and - /// the children, exceeds [`MAX_EPOCHS`]. - TooManyEpochs, - /// A child aggregate does not verify. - InvalidChild(AggregateVerifyError), - /// No signature claims or DA roots to publish. - Empty, - /// A declared claim is not one the contributions cover. A claim is a key, an - /// epoch and a message, so another epoch or another message is another claim. - NotCovered, - /// A raw signature's randomness does not decode to a target-sum encoding, - /// so there is no witness to build for it. - MalformedRawSignature, - /// More than [`MAX_RECURSIONS`] children, or [`MAX_KEYS`] signers or more - /// once the duplicate slots are counted, or more than [`MAX_DA_ROOTS`] DA roots. - TooLarge, - /// The payload has a partial row or exceeds [`DA_MAX_ROWS`]. - InvalidBlobSize { symbols: usize }, - /// A requested DA commitment is absent from the children and the direct blob check. - BlobNotCovered, - /// `log_inv_rate` is outside the range the WHIR configuration accepts. - InvalidRate { log_inv_rate: usize }, - /// A child's committed witness falls outside the opening arms the guest was - /// compiled with. Small aggregates are padded up to the floor, so this means - /// a child too big: more signatures than one node can hold. - ChildOutOfRange { log_committed: usize }, - /// The prover made no proof of this node: the guest rejects its input (a - /// forged signature fails the guest's run), or its witness is larger than any - /// verifier accepts (too many signatures for one proof). - ProofError(ProveError), -} - -impl std::fmt::Display for AggregateVerifyError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::MalformedSignerSet => write!(f, "malformed signer set"), - Self::MalformedDaCommitments => write!(f, "malformed DA commitment list"), - Self::MalformedClaim => write!(f, "malformed deferred claim"), - Self::MalformedEncoding => write!(f, "not a valid aggregate encoding"), - Self::Snark(e) => write!(f, "the snark did not verify: {e:?}"), - } - } -} - -impl std::error::Error for AggregateVerifyError {} - -impl std::fmt::Display for AggregationError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::TooManyEpochs => write!(f, "more than MAX_EPOCHS ({MAX_EPOCHS}) XMSS groups"), - Self::InvalidChild(_) => write!(f, "invalid child aggregate"), - Self::Empty => write!(f, "no signature claims or DA roots to publish"), - Self::NotCovered => write!(f, "a declared claim is not covered by the children and raw signatures"), - Self::MalformedRawSignature => write!(f, "a raw signature does not decode to a target-sum encoding"), - Self::TooLarge => { - write!(f, "too many children, signature claims, or DA roots") - } - Self::InvalidBlobSize { symbols } => { - write!( - f, - "{symbols} blob symbols do not form 1..={DA_MAX_ROWS} rows of {} symbols", - 1 << DA_LOG_K - ) - } - Self::BlobNotCovered => write!(f, "the requested blob commitment is not carried by any child"), - Self::InvalidRate { log_inv_rate } => { - write!( - f, - "log_inv_rate {log_inv_rate} is outside {}..={}", - MIN_LOG_INV_RATE, MAX_LOG_INV_RATE - ) - } - Self::ChildOutOfRange { log_committed } => { - write!( - f, - "a child commits 2^{log_committed} words, outside the guest's opening arms" - ) - } - Self::ProofError(e) => write!(f, "proving failed: {e}"), - } - } -} - -impl std::error::Error for AggregationError { - fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { - match self { - Self::InvalidChild(e) => Some(e), - Self::ProofError(e) => Some(e), - _ => None, - } - } -} - -/// Everything but the signer set, which a receiver may already hold. -type WireCore = (Vec<[u8; 32]>, Vec, Vec, lean_vm::cpu::Proof); - -/// Signature claims grouped by scheme: XMSS epoch/message groups, then SPHINCS key/message pairs. -#[derive(Clone, Debug, Default, PartialEq, Eq, serde::Serialize, serde::Deserialize)] -pub struct SignatureClaims { - pub xmss: Vec, - pub sphincs: Vec, -} - -/// Exactly the claims to publish. Empty lists publish no claims of that kind. -#[derive(Clone, Copy)] -pub struct ClaimSelection<'a> { - pub signatures: &'a SignatureClaims, - pub da_commitments: &'a [[u8; 32]], -} - -/// The wire encoding: bincode's fixed-width integers, as the free functions use, -/// but rejecting trailing bytes, which they do not. Without that an accepted -/// aggregate has unboundedly many encodings, so anything downstream that dedupes -/// or indexes on the serialized bytes can be made to see one aggregate as many. -fn wire() -> impl bincode::Options { - bincode::DefaultOptions::new().with_fixint_encoding() -} - -/// Reject a signer set that the coverage argument does not cover: strict sorting -/// within each list is what stops one signer being counted many times: the XMSS -/// list's length is a count of distinct `(epoch, message, key)` claims, the SPHINCS -/// list's of distinct `(key, message)` claims. The groups are strictly increasing -/// on `(epoch, message)`, non-empty (an absent pair is an absent group, the one -/// encoding of each set) and at most [`MAX_EPOCHS`]. Either list may be empty; -/// both may be empty for a blob proof. [`MAX_KEYS`] is exclusive here, as in the guest. -fn check_signer_set( - xmss_signers: &[XmssClaimGroup], - sphincs_signers: &[SphincsClaim], -) -> Result<(), AggregateVerifyError> { - let total = xmss_signers.iter().map(|group| group.keys.len()).sum::() + sphincs_signers.len(); - if total >= MAX_KEYS - || xmss_signers.len() > MAX_EPOCHS - || !xmss_signers - .windows(2) - .all(|w| (w[0].epoch, w[0].message) < (w[1].epoch, w[1].message)) - || xmss_signers - .iter() - .any(|group| group.keys.is_empty() || !group.keys.windows(2).all(|w| w[0] < w[1])) - || !sphincs_signers.windows(2).all(|w| w[0] < w[1]) - { - return Err(AggregateVerifyError::MalformedSignerSet); - } - Ok(()) -} - -fn check_da_roots(roots: &[[u8; 32]]) -> Result<(), AggregateVerifyError> { - if roots.len() > MAX_DA_ROOTS || roots.windows(2).any(|pair| pair[0] >= pair[1]) { - return Err(AggregateVerifyError::MalformedDaCommitments); - } - Ok(()) -} - -fn da_claim_cells(root: &[u8; 32]) -> Vec { - let vector_hash = lean_da::vector_digest(&lean_da::membership_vector(root)); - [pack_hash_state(root), pack_hash_state(&vector_hash)].concat() -} - -fn da_list_digest(roots: &[[u8; 32]]) -> [u8; 32] { - // Recompute vector hashes from the roots, including when verifying a received proof. - // Trusting prover-supplied hashes would let zero weights certify any matrix. - let cells = roots.iter().flat_map(da_claim_cells); - primitives::hash::hash(&cell_bytes(cells)) -} - -impl EthereumProof { - /// This aggregate's own public statement, as the VM publishes it. - fn public_input(&self) -> [F192; 2] { - statement_digest( - signers_hash(&self.xmss_signers, &self.sphincs_signers), - self.da_commitments_digest(), - &self.defer, - ) - } - - /// Strictly increasing `(epoch, message)` pairs, each group non-empty and - /// strictly sorted. May be empty, including in a blob-only proof. - pub fn xmss_signers(&self) -> &[XmssClaimGroup] { - &self.xmss_signers - } - - /// Strictly sorted and deduplicated on the whole `(key, message)` pair. - pub fn sphincs_signers(&self) -> &[SphincsClaim] { - &self.sphincs_signers - } - - /// Strictly sorted, distinct LeanDA roots proved directly or inherited from children. - /// An empty list makes no blob claim. - pub fn da_commitments(&self) -> &[[u8; 32]] { - &self.da_roots - } - - /// BLAKE2s of the concatenated `(root, vector hash)` pairs, or of empty input. - /// Each vector hash is derived from its root outside the SNARK. - /// This digest is bound into the public statement. - pub fn da_commitments_digest(&self) -> [u8; 32] { - da_list_digest(&self.da_roots) - } - - /// The declared claims, as many as the coverage table's declared slots. - /// - /// NOT a count of distinct signers: a SPHINCS key may hold several claims, - /// one per message it signed, and an XMSS key one per `(epoch, message)` it signed - /// (see the notes on the two lists). A caller that wants signers has to - /// deduplicate by key itself. - pub fn num_signature_claims(&self) -> usize { - self.xmss_signers.iter().map(|group| group.keys.len()).sum::() + self.sphincs_signers.len() - } - - /// The wire format: the signer set (each group with its epoch and - /// message), the DA root list, the two deferred points, and the VM proof. The claim *values* are not transmitted; - /// [`Self::from_bytes`] recomputes them, so there is nothing to lie about. - pub fn to_bytes(&self) -> Vec { - wire() - .serialize(&((&self.xmss_signers, &self.sphincs_signers), self.core())) - .expect("an aggregate serializes") - } - - /// Parsing does NOT verify: the proof is untouched, only shapes are checked. - /// Call [`Self::verify`] before believing any of it. - pub fn from_bytes(bytes: &[u8]) -> Result { - let (keys, core): (SignatureClaims, WireCore) = wire() - .deserialize(bytes) - .map_err(|_| AggregateVerifyError::MalformedEncoding)?; - Self::from_parts(keys, core) - } - - /// Without the signer set, for a receiver that already knows it. A set other - /// than the one aggregated fails verification. - pub fn to_bytes_without_pubkeys(&self) -> Vec { - wire().serialize(&self.core()).expect("an aggregate serializes") - } - - pub(crate) fn proof(&self) -> &lean_vm::cpu::Proof { - &self.proof - } - - /// Inverse of [`Self::to_bytes_without_pubkeys`], verifying nothing either. - pub fn from_bytes_without_pubkeys(bytes: &[u8], keys: SignatureClaims) -> Result { - Self::from_parts( - keys, - wire() - .deserialize(bytes) - .map_err(|_| AggregateVerifyError::MalformedEncoding)?, - ) - } - - fn core(&self) -> (&[[u8; 32]], &[F192], &[F192], &lean_vm::cpu::Proof) { - ( - &self.da_roots, - &self.defer.bytecode_point, - &self.defer.matrix_point, - &self.proof, - ) - } - - fn from_parts(keys: SignatureClaims, core: WireCore) -> Result { - let SignatureClaims { - xmss: xmss_signers, - sphincs: sphincs_signers, - } = keys; - let (da_roots, bytecode_point, matrix_point, proof) = core; - // Cheap rejections first. `recompute` below is a pass over the whole stacked - // bytecode plus a walk of the BLAKE2s circuit, on points a peer chose, so - // anything decidable without it has to be decided before it. - check_signer_set(&xmss_signers, &sphincs_signers)?; - check_da_roots(&da_roots)?; - Ok(Self { - xmss_signers, - sphincs_signers, - da_roots, - // The wire carries no value, only the points to derive it from. - defer: DeferredClaim::recompute(bytecode_point, matrix_point)?, - proof, - }) - } - - /// Verify the aggregate's internal consistency: the signer set and DA root list are well - /// formed, the three deferred fixed-polynomial claims hold at their - /// transmitted points, and the VM proof satisfies the statement built from - /// all of it. - /// - /// This says "every key in `xmss_signers` signed its group's message at its - /// group's epoch, and every `(key, message)` in `sphincs_signers` is a valid - /// SPHINCS signature", with the epochs and messages chosen by whoever - /// produced the aggregate: an aggregate over the same keys at different - /// epochs, or under different messages, verifies just as well. A caller that - /// expects particular pairs has to check the two lists against them. Every root in - /// [`Self::da_commitments`] also attests to a well-formed blob matrix; callers check - /// that this list contains the commitments they require. - pub fn verify(&self) -> Result<(), AggregateVerifyError> { - check_signer_set(&self.xmss_signers, &self.sphincs_signers)?; - check_da_roots(&self.da_roots)?; - // Discharging the deferred claims IS recomputing them: their values have - // to be the true evaluations, or the recursion below proves nothing about - // the fixed polynomials. - let _s = tracing::info_span!("Recompute deferred claims").entered(); - let defer = DeferredClaim::recompute(self.defer.bytecode_point.clone(), self.defer.matrix_point.clone())?; - drop(_s); - if defer != self.defer { - return Err(AggregateVerifyError::MalformedClaim); - } - let _s = tracing::info_span!("Statement digest").entered(); - let pi = self.public_input(); - drop(_s); - verify(unified_guest(), &pi, &self.proof).map_err(AggregateVerifyError::Snark)?; - Ok(()) - } -} -/// The stacked bytecode polynomial of the aggregation guest: the one fixed -/// table every node's bytecode claims are about. Cached, because verification -/// evaluates it and building it walks the whole program. -fn stacked_bytecode() -> &'static [F64] { - static TABLE: std::sync::OnceLock> = std::sync::OnceLock::new(); - TABLE.get_or_init(|| lean_vm::cpu::layout::bytecode_table(&unified_guest().prog)) -} - -/// The slots of the stacked bytecode that are not structurally zero. -/// -/// Eight encoding columns sit inside sixteen stacking slots, so half the -/// table is zero and contributes nothing to any round of the batching sumcheck. -/// Read off the table rather than from the column count, so an all-zero column -/// at the edge only ever shrinks the window. -fn bytecode_window() -> Range { - static WINDOW: std::sync::OnceLock> = std::sync::OnceLock::new(); - WINDOW - .get_or_init(|| { - let table = stacked_bytecode(); - let kbc = bytecode_vars() - lean_vm::leaf::N_BYTECODE_SELECTORS; - let live = |s: usize| table[s << kbc..(s + 1) << kbc].iter().any(|v| *v != F64::ZERO); - let slots = 1 << lean_vm::leaf::N_BYTECODE_SELECTORS; - let start = (0..slots).find(|&s| live(s)).expect("the bytecode is not all zero"); - let end = (0..slots).rfind(|&s| live(s)).expect("the bytecode is not all zero") + 1; - start..end - }) - .clone() -} - -/// One round message against a K-valued table: `bt` is the bytecode itself, so -/// the products are base-by-extension. -fn round_msg_base(bytecode: &[F64], weights: &[F192]) -> (F192, F192) { - let half = bytecode.len() / 2; - let term = |i: usize| { - let (bytecode_0, bytecode_1) = (bytecode[2 * i], bytecode[2 * i + 1]); - let (weight_0, weight_1) = (weights[2 * i], weights[2 * i + 1]); - ( - weight_1.mul_base(bytecode_1), - (weight_0 + weight_1).mul_base(bytecode_0 + bytecode_1), - ) - }; - if half >= PAR_MIN { - parallel::fold_reduce( - half, - || (F192::ZERO, F192::ZERO), - |acc: &mut (F192, F192), i| { - let (x, y) = term(i); - acc.0 += x; - acc.1 += y; - }, - |a: (F192, F192), b: (F192, F192)| (a.0 + b.0, a.1 + b.1), - ) - } else { - (0..half).fold((F192::ZERO, F192::ZERO), |acc, i| { - let (x, y) = term(i); - (acc.0 + x, acc.1 + y) - }) - } -} - -/// The first fold of a K-valued table, which is where it becomes extension-valued. -fn fold_lsb_base(table: &[F64], challenge: F192) -> Vec { - parallel::map_collect(table.len() / 2, |i| { - let (left, right) = (table[2 * i], table[2 * i + 1]); - F192::from(left) + challenge.mul_base(left + right) - }) -} - -/// `Σ_t γ_t · eq(points[t], (r_row, ·))` over the stacking slots, once the row -/// variables are bound. A closed form, so the row rounds never have to carry the -/// slot half of a `2^kbcv` weight table. -fn slot_weights(points: &[Vec], lambdas: &[F192], r_row: &[F192], kbc: usize) -> Vec { - let slots = lean_vm::leaf::N_BYTECODE_SELECTORS; - let mut weights = vec![F192::ZERO; 1 << slots]; - for (point, &lambda) in points.iter().zip(lambdas) { - let row_weight: F192 = (0..kbc).fold(lambda, |acc, k| acc * (F192::ONE + point[k] + r_row[k])); - for (slot, weight) in weights.iter_mut().enumerate() { - let slot_weight = (0..slots).fold(row_weight, |acc, bit| { - let coordinate = point[kbc + bit]; - acc * if (slot >> bit) & 1 == 1 { - coordinate - } else { - F192::ONE + coordinate - } - }); - *weight += slot_weight; - } - } - weights -} - -/// Variables of a bytecode claim's point: the log row count plus the stacking -/// selectors. -fn bytecode_vars() -> usize { - stacked_bytecode().len().trailing_zeros() as usize -} - -/// Below this a parallel dispatch costs more than the loop it replaces. -const PAR_MIN: usize = 1 << 16; - -fn fold_lsb(table: &mut Vec, challenge: F192) { - let half = table.len() / 2; - if half >= PAR_MIN { - let source: &[F192] = table; - let folded = parallel::map_collect(half, |i| { - let (left, right) = (source[2 * i], source[2 * i + 1]); - left + challenge * (left + right) - }); - *table = folded; - return; - } - for i in 0..half { - table[i] = table[2 * i] + challenge * (table[2 * i] + table[2 * i + 1]); - } - table.truncate(half); -} - -/// Compressed product-sumcheck round message over γ-weighted table pairs: -/// (g1, g∞) with g0 recovered from the running claim. -fn round_msg(pairs: &[(&[F192], &[F192], F192)]) -> (F192, F192) { - let (mut g1, mut gi) = (F192::ZERO, F192::ZERO); - for &(u, m, lambda) in pairs { - let half = u.len() / 2; - let term = |i: usize| { - let (u0, u1) = (u[2 * i], u[2 * i + 1]); - let (m0, m1) = (m[2 * i], m[2 * i + 1]); - (u1 * m1, (u0 + u1) * (m0 + m1)) - }; - let (a1, ai) = if half >= PAR_MIN { - parallel::fold_reduce( - half, - || (F192::ZERO, F192::ZERO), - |acc: &mut (F192, F192), i| { - let (x, y) = term(i); - acc.0 += x; - acc.1 += y; - }, - |a: (F192, F192), b: (F192, F192)| (a.0 + b.0, a.1 + b.1), - ) - } else { - (0..half).fold((F192::ZERO, F192::ZERO), |acc, i| { - let (x, y) = term(i); - (acc.0 + x, acc.1 + y) - }) - }; - g1 += lambda * a1; - gi += lambda * ai; - } - (g1, gi) -} - -/// One round of a batching sumcheck, in the transcript order the guest mirrors -/// word for word: observe `(g1, g_inf)`, squeeze the fold challenge, record -/// both, and advance the running claim through the compressed round polynomial. -/// Returns the challenge, which the caller folds its tables with. -fn absorb_round( - transcript: &mut FiatShamirState, - messages: &mut Vec, - challenges: &mut Vec, - claim: &mut F192, - (g1, gi): (F192, F192), -) -> F192 { - transcript.observe(g1); - transcript.observe(gi); - let challenge = transcript.sample(); - messages.extend([g1, gi]); - challenges.push(challenge); - let g0 = *claim + g1; - let c1 = g0 + g1 + gi; - *claim = (gi * challenge + c1) * challenge + g0; - challenge -} - -/// `Σ_t γ_t · eq(points[t], ·)`, over the `active` window of `2^vars` entries. -/// -/// Each point splits in half; the halves' eq tables are small and serial, and -/// the full table is their outer product, one fused multiply-add per entry, -/// parallel over the high index. Materializing a `2^vars` eq table per claim and -/// summing them is the same arithmetic done twice, serially. -fn weighted_eq_table(points: &[Vec], lambdas: &[F192], vars: usize, active: Range) -> Vec { - let lo_vars = vars / 2; - let lo_len = 1usize << lo_vars; - debug_assert!(active.start.is_multiple_of(lo_len) && active.end.is_multiple_of(lo_len)); - // `build_eq_table_ext` is LSB-first, so the low variables index the low bits - // and entry `hi * lo_len + lo` is `eq_lo[lo] * eq_hi[hi]`. - let halves: Vec<(Vec, Vec)> = points - .iter() - .zip(lambdas) - .map(|(point, &lambda)| { - let low = pcs::whir::build_eq_table_ext(&point[..lo_vars]); - let mut high = pcs::whir::build_eq_table_ext(&point[lo_vars..]); - high.iter_mut().for_each(|weight| *weight *= lambda); - (low, high) - }) - .collect(); - let mut weights = vec![F192::ZERO; active.len()]; - let first = active.start / lo_len; - parallel::chunks_mut(&mut weights, lo_len, |high_index, chunk| { - for (low, high) in &halves { - let scale = high[first + high_index]; - for (output, &low_weight) in chunk.iter_mut().zip(low) { - *output += scale * low_weight; - } - } - }); - weights -} - -/// Mirror the guest's `aggregate_claims` transcript and prove the two batching -/// sumchecks: dense for the bytecode, two-phase sparse for the matrices. -/// -/// Each child brings two claims per fixed polynomial: the one it deferred and -/// the fresh one its verification raised. They differ -/// only in the weight they enter with. A fresh matrix claim carries flock's -/// zerocheck/lincheck structure; a carried one is a plain point, so its weight -/// is an eq table on each side. Returns the guest hints and the single claim -/// per polynomial they reduce to. -fn aggregate_deferred_claims( - subproofs: &[DeferredSubproof], - carried_claims: &[&DeferredClaim], -) -> (SubHints, DeferredClaim) { - let child_count = subproofs.len(); - assert_eq!(child_count, carried_claims.len(), "one carried claim per child"); - let kbcv = bytecode_vars(); - let klog = flock::hash::K_LOG; - - let mut transcript = FiatShamirState::from_label(RECURSION_AGG_LABEL); - transcript.observe(count(child_count)); - for (subproof, carried) in subproofs.iter().zip(carried_claims) { - transcript.observe(subproof.public_input[0]); - transcript.observe(subproof.public_input[1]); - for &value in &subproof.bytecode_row_point { - transcript.observe(value); - } - for &value in &subproof.bytecode_selector_point { - transcript.observe(value); - } - transcript.observe(subproof.bytecode_value); - transcript.observe(subproof.matrix_a_coefficient); - transcript.observe(subproof.skip_point); - for &value in &subproof.zerocheck_row_point { - transcript.observe(value); - } - for &value in &subproof.lincheck_round_point { - transcript.observe(value); - } - for &value in &subproof.lincheck_terminal_values { - transcript.observe(value); - } - transcript.observe(subproof.matrix_claim); - for value in carried.cells() { - transcript.observe(value); - } - } - - let _span = tracing::info_span!("Bytecode batch", vars = kbcv).entered(); - let gbc: Vec = (0..2 * child_count).map(|_| transcript.sample()).collect(); - let points: Vec> = subproofs - .iter() - .zip(carried_claims) - .flat_map(|(subproof, carried)| { - [ - subproof - .bytecode_row_point - .iter() - .chain(&subproof.bytecode_selector_point) - .copied() - .collect::>(), - carried.bytecode_point.clone(), - ] - }) - .collect(); - let values: Vec = subproofs - .iter() - .zip(carried_claims) - .flat_map(|(subproof, carried)| [subproof.bytecode_value, carried.bytecode_value]) - .collect(); - let mut brun: F192 = (0..2 * child_count) - .map(|index| gbc[index] * values[index]) - .fold(F192::ZERO, |acc, value| acc + value); - let mut bscr = Vec::new(); - let mut r_bc = Vec::new(); - // The row variables, over the populated slot window only: the rest of the - // stacked table is structurally zero and contributes nothing to any round - // message, and folding LSB-first pairs entries within a slot, so the window's - // blocks stay aligned all the way down. - let n_slots = 1 << lean_vm::leaf::N_BYTECODE_SELECTORS; - let slot_window = bytecode_window(); - let kbc = kbcv - lean_vm::leaf::N_BYTECODE_SELECTORS; - let mut wt = weighted_eq_table( - &points, - &gbc, - kbcv, - (slot_window.start << kbc)..(slot_window.end << kbc), - ); - // Round zero runs against the K-valued table itself: the bytecode never needs - // to exist as `2^kbcv` extension elements (three times the memory traffic), - // and base-by-extension is cheaper than extension-by-extension. Every later - // round is extension-valued anyway. - let bc = &stacked_bytecode()[(slot_window.start << kbc)..(slot_window.end << kbc)]; - let mut bt = { - let msg = round_msg_base(bc, &wt); - let r = absorb_round(&mut transcript, &mut bscr, &mut r_bc, &mut brun, msg); - let folded = fold_lsb_base(bc, r); - fold_lsb(&mut wt, r); - folded - }; - for _ in 1..kbc { - let msg = round_msg(&[(&bt, &wt, F192::ONE)]); - let r = absorb_round(&mut transcript, &mut bscr, &mut r_bc, &mut brun, msg); - fold_lsb(&mut bt, r); - fold_lsb(&mut wt, r); - } - // The slot variables. The window has folded to one entry per populated slot; - // put those back among the zeros. The weights come from a closed form rather - // than from having carried a `2^kbcv` table through the rows. - let mut bt_slots = vec![F192::ZERO; n_slots]; - bt_slots[slot_window.clone()].copy_from_slice(&bt); - let wt_slots = slot_weights(&points, &gbc, &r_bc, kbc); - let (mut bt, mut wt) = (bt_slots, wt_slots); - for _ in 0..lean_vm::leaf::N_BYTECODE_SELECTORS { - let msg = round_msg(&[(&bt, &wt, F192::ONE)]); - let r = absorb_round(&mut transcript, &mut bscr, &mut r_bc, &mut brun, msg); - fold_lsb(&mut bt, r); - fold_lsb(&mut wt, r); - } - let v_bc = bt[0]; - assert_eq!(brun, v_bc * wt[0], "bytecode sumcheck terminal"); - - drop(_span); - let _span = tracing::info_span!("Matrix batch").entered(); - // One group per claim: the row weights, the column weights, and the - // coefficient each matrix enters with. Downstream is shape-blind. - let mut us: Vec> = Vec::with_capacity(2 * child_count); - let mut ws: Vec> = Vec::with_capacity(2 * child_count); - let mut ga: Vec = Vec::with_capacity(2 * child_count); - let mut gb: Vec = Vec::with_capacity(2 * child_count); - let mut mrun = F192::ZERO; - for (subproof, carried) in subproofs.iter().zip(carried_claims) { - let (gf, cga, cgb) = (transcript.sample(), transcript.sample(), transcript.sample()); - us.push(flock::lincheck::build_quirky_eq_table( - subproof.skip_point, - &subproof.zerocheck_row_point, - 6, - )); - ws.push( - (0..1usize << klog) - .map(|col| { - let mut w = subproof.lincheck_terminal_values[col & 63]; - for (j, &rj) in subproof.lincheck_round_point.iter().enumerate() { - let bit = (col >> (klog - 1 - j)) & 1; - w *= if bit == 1 { rj } else { F192::ONE + rj }; - } - w - }) - .collect(), - ); - ga.push(gf); - gb.push(gf * subproof.matrix_a_coefficient); - us.push(pcs::whir::build_eq_table_ext(&carried.matrix_point[..klog])); - ws.push(pcs::whir::build_eq_table_ext(&carried.matrix_point[klog..])); - ga.push(cga); - gb.push(cgb); - mrun += gf * subproof.matrix_claim + cga * carried.matrix_a_value + cgb * carried.matrix_b_value; - } - let _cols = tracing::info_span!("Contract columns").entered(); - // One forward walk of the circuit per claim yields that claim's two row - // tables `(A_0 w, B_0 w)` directly, in O(circuit): no matrix, and no pass - // over the ~89M nonzeros. A before B, the order `ga`/`gb` index. - let mut ms: Vec> = Vec::with_capacity(2 * ws.len()); - for w in &ws { - let (ra, rb) = flock::hash::row_values_walk(w); - ms.push(ra); - ms.push(rb); - } - drop(_cols); - // sanity: every claim really is the bilinear form over the matrices. - #[cfg(debug_assertions)] - for t in 0..2 * child_count { - let form = |m: &[F192]| { - m.iter() - .zip(&us[t]) - .map(|(&m, &u)| m * u) - .fold(F192::ZERO, |a, x| a + x) - }; - let (fa, fb) = (form(&ms[2 * t]), form(&ms[2 * t + 1])); - if t % 2 == 0 { - let subproof = &subproofs[t / 2]; - assert_eq!( - fa + subproof.matrix_a_coefficient * fb, - subproof.matrix_claim, - "fresh matrix claim, child {}", - t / 2 - ); - } else { - let carried = &carried_claims[t / 2]; - assert_eq!( - (fa, fb), - (carried.matrix_a_value, carried.matrix_b_value), - "carried matrix claim" - ); - } - } - let mut mscr = Vec::new(); - let mut r_row = Vec::new(); - let _rounds = tracing::info_span!("Row rounds").entered(); - for _ in 0..klog { - let pairs: Vec<(&[F192], &[F192], F192)> = (0..2 * child_count) - .flat_map(|t| { - [ - (&us[t][..], &ms[2 * t][..], ga[t]), - (&us[t][..], &ms[2 * t + 1][..], gb[t]), - ] - }) - .collect(); - let msg = round_msg(&pairs); - let r = absorb_round(&mut transcript, &mut mscr, &mut r_row, &mut mrun, msg); - for u in us.iter_mut() { - fold_lsb(u, r); - } - for m in ms.iter_mut() { - fold_lsb(m, r); - } - } - drop(_rounds); - let eq_rstar = pcs::whir::build_eq_table_ext(&r_row); - let _rows = tracing::info_span!("Contract rows").entered(); - // `A_0ᵀ eq` and `B_0ᵀ eq` are the column marginals, which the circuit walks - // backwards (`gf2`'s `back_*`) in O(circuit). - let (mut acol, mut bcol) = flock::hash::marginal_walk_pair(&eq_rstar); - drop(_rows); - let mut wa = vec![F192::ZERO; 1 << klog]; - let mut wb = vec![F192::ZERO; 1 << klog]; - for t in 0..2 * child_count { - let (sa, sb) = (ga[t] * us[t][0], gb[t] * us[t][0]); - for j in 0..1 << klog { - wa[j] += sa * ws[t][j]; - wb[j] += sb * ws[t][j]; - } - } - let mut r_col = Vec::new(); - for _ in 0..klog { - let pairs: Vec<(&[F192], &[F192], F192)> = vec![(&acol, &wa, F192::ONE), (&bcol, &wb, F192::ONE)]; - let msg = round_msg(&pairs); - let r = absorb_round(&mut transcript, &mut mscr, &mut r_col, &mut mrun, msg); - for tb in [&mut acol, &mut bcol, &mut wa, &mut wb] { - fold_lsb(tb, r); - } - } - let (v_a, v_b) = (acol[0], bcol[0]); - assert_eq!(mrun, v_a * wa[0] + v_b * wb[0], "matrix sumcheck terminal"); - // The guest reaches the same two weights by a succinct formula rather than by - // folding these tables, and nothing else compares the two: the aggregation - // layer has no third implementation the way `cpu::verify` does. So this runs - // unconditionally (it is O(n) against a 2^28 sumcheck), and it compares the - // weights COMPONENT-WISE. Checking only the combination `v_a·wa + v_b·wb` - // would let two correlated errors through. - { - let eqr = pcs::whir::build_eq_table_ext(&r_row[..6]); - let eqc = pcs::whir::build_eq_table_ext(&r_col[..6]); - let (mut wam, mut wbm) = (F192::ZERO, F192::ZERO); - for (t, (subproof, carried)) in subproofs.iter().zip(carried_claims).enumerate() { - let lam = primitives::multilinear::lagrange_weights_naive(6, subproof.skip_point); - let mut urow: F192 = (0..64).map(|i| lam[i] * eqr[i]).fold(F192::ZERO, |a, x| a + x); - for (k, &z) in subproof.zerocheck_row_point.iter().enumerate() { - urow *= F192::ONE + z + r_row[6 + k]; - } - let mut wcol: F192 = (0..64) - .map(|i| subproof.lincheck_terminal_values[i] * eqc[i]) - .fold(F192::ZERO, |a, x| a + x); - for (j, &rj) in subproof.lincheck_round_point.iter().enumerate() { - wcol *= F192::ONE + rj + r_col[klog - 1 - j]; - } - let fresh = urow * wcol; - let mut plain = F192::ONE; - for (k, &p) in carried.matrix_point.iter().enumerate() { - let r = if k < klog { r_row[k] } else { r_col[k - klog] }; - plain *= F192::ONE + p + r; - } - wam += ga[2 * t] * fresh + ga[2 * t + 1] * plain; - wbm += gb[2 * t] * fresh + gb[2 * t + 1] * plain; - } - assert_eq!(wa[0], wam, "guest row-weight formula for A0"); - assert_eq!(wb[0], wbm, "guest row-weight formula for B0"); - } - - drop(_span); - let hints = vec![ - ("bc_sumcheck_msgs", bscr), - ("mat_sumcheck_msgs", mscr), - ("bc_star_hint", vec![v_bc]), - ("mat_stars_hint", vec![v_a, v_b]), - ]; - ( - hints, - DeferredClaim { - bytecode_point: r_bc, - bytecode_value: v_bc, - matrix_point: r_row.iter().chain(&r_col).copied().collect(), - matrix_a_value: v_a, - matrix_b_value: v_b, - }, - ) -} -/// The verifier-side WHIR config for one committed size and rate, plus the -/// query packing derived from it. The hint builder needs it for the real -/// opening and the placeholder map for every candidate size, so it lives here: -/// a candidate whose shape differed from the real one would compile a guest -/// that cannot open the proof it is handed. -struct WhirShape { - config: pcs::whir::VerifierConfig, - levels: pcs::whir::LevelShapes, - /// Merkle tree depth per level. - depth: Vec, - /// Query positions carried by one squeezed F192 word, per level. - per_squeeze: Vec, -} - -fn whir_shape(mu: usize, log_inv_rate: usize) -> WhirShape { - let config = pcs::whir::config_for_rate(mu, log_inv_rate).expect("stacked whir config"); - let levels = config.level_shapes(mu); - let depth: Vec = levels.block_len.iter().map(|b| b.trailing_zeros() as usize).collect(); - let per_squeeze = depth.iter().map(|&d| 192 / d).collect(); - WhirShape { - config, - levels, - depth, - per_squeeze, - } -} - -fn merkle_cap_depth(queries: usize, depth: usize) -> usize { - (queries.next_power_of_two().ilog2() as usize).min(depth) -} - -/// The BLAKE2s table's virtual value columns, in `hash_flock::SLOTS` order. -fn blake2s_value_columns() -> Vec { - let base = lean_vm::cpu::schema().base[5]; - lean_vm::tables::BLAKE2S_VALUE_COLS.iter().map(|&c| base + c).collect() -} - -/// One entry of the guest's claim pool. -enum ClaimSite { - /// A committed column read by a framework bus block. - Framework { column: usize }, - /// A table column; `is_virtual` marks the q_flock-backed value - /// columns, whose claim is a strided slot rather than a plain column. - TableColumn { column: usize, is_virtual: bool }, - /// One of the three PI memory limbs (MEM_LO, MEM_HI, MEM_TOP). - MemoryLimb { column: usize }, -} - -/// The guest's `COORD_KIND_*` code for a coordinate (`guests/lean_ethereum.py`), -/// shared by its `COORD_TYPE` and `TERM_TYPE` arrays. -fn coord_kind(c: &Coord) -> usize { - match c { - Coord::Const(_) => 0, - Coord::Col(_) => 1, - Coord::GCol(..) => 2, - Coord::Index => 3, - Coord::Public(_) => 4, - Coord::Prod(..) => 5, - Coord::Sum(..) => 6, - } -} - -/// The `K` scalar a coordinate carries beside its columns: the constant itself, -/// or the `g^k` a `GCol`/`Prod` scales by. Zero for every other kind. -fn coord_scale(c: &Coord) -> F192 { - match c { - Coord::Const(v) => F192::new(v.0, 0, 0), - Coord::GCol(_, k) | Coord::Prod(_, _, k) => F192::new(g_pow(*k as usize).0, 0, 0), - _ => F192::ZERO, - } -} - -/// Flatten one table-block coordinate into the guest's term arrays, in local -/// column indices. A [`Coord::Sum`]'s children are its terms; every other kind is -/// one term. `Index`/`Public` never reach a table block. -fn push_coord_terms(c: &Coord, base: usize, terms: &mut Vec) { - let (column_a, column_b) = match c { - Coord::Const(_) => (0, 0), - Coord::Col(i) | Coord::GCol(i, _) => (*i - base, 0), - Coord::Prod(i, j, _) => (*i - base, *j - base), - Coord::Sum(cs) => { - for c in cs { - push_coord_terms(c, base, terms); - } - return; - } - Coord::Index | Coord::Public(_) => unreachable!("a table's bus block carries no virtual coordinate"), - }; - terms.push(Term { - kind: coord_kind(c), - constant: dsl_u128(coord_scale(c)), - column_a, - column_b, - }); -} - -/// Visit the claim pool in the exact order the guest indexes it: the framework -/// bus claims (deduped by `(column, kappa)`, as `leaf.rs` pools them), then every -/// table's committed columns, then the PI memory triple. The placeholder map's -/// claim descriptors follow this order. -fn walk_claims(layout: &lean_vm::cpu::Layout, kbc: usize, mut visit: impl FnMut(ClaimSite)) { - let sides: [&[Block]; 3] = [&layout.push, &layout.pull, &layout.count]; - let valcols = blake2s_value_columns(); - // Only the framework blocks raise claims: a table's coords are settled inside - // the table sumcheck. - let is_framework: Vec = lean_vm::cpu::block_kappa_sources(kbc) - .into_iter() - .map(|(src, _)| src < 2) - .collect(); - let mut seen: std::collections::HashSet<(usize, usize)> = Default::default(); - let mut bi = 0usize; - for blocks in sides.iter() { - for blk in blocks.iter() { - let framework = is_framework[bi]; - bi += 1; - if !framework { - continue; - } - for c in &blk.coords { - if let Coord::Col(i) | Coord::GCol(i, _) = c { - if !seen.insert((*i, blk.kappa)) { - continue; // deduped: pooled once at its first occurrence - } - assert!(!valcols.contains(i), "{VALCOL_FRAMEWORK}"); - visit(ClaimSite::Framework { column: *i }); - } - } - } - } - let sch = lean_vm::cpu::schema(); - for (t, table) in lean_vm::tables::tables().iter().enumerate() { - for c in 0..table.n_committed_columns() { - let column = sch.base[t] + c; - visit(ClaimSite::TableColumn { - column, - is_virtual: layout.placements[column].is_virtual(), - }); - } - } - for &column in &[lean_vm::cpu::MEM_LO, lean_vm::cpu::MEM_HI, lean_vm::cpu::MEM_TOP] { - visit(ClaimSite::MemoryLimb { column }); - } -} - -/// Config + hints for the recursion guest (`guests/lean_ethereum.py`), built -/// from the REAL `cpu::layout` of the inner program and the summary of a real -/// `cpu::verify` run (zero hand-mirroring drift). -fn gen_verify( - program: &Program, - public_input: [F192; 2], - summary: lean_vm::cpu::VerifySummary, -) -> Result<(SubHints, DeferredSubproof), AggregationError> { - let proof_stream = &summary.raw.stream; - let layout = lean_vm::cpu::layout( - &program.prog, - proof_stream[0].c0 as usize, - std::array::from_fn(|i| proof_stream[1 + i].c0 as usize), - public_input, - ); - let sides: [&[Block]; 3] = [&layout.push, &layout.pull, &layout.count]; - let side_layouts = sides.map(lean_vm::leaf::layout); - // Fixed capacities: every buffer/stride placeholder is a global cap so - // the placeholder map is SHAPE-INDEPENDENT (the definition of generic). - assert!(side_layouts.iter().all(|side| side.mu <= MU_CAP) && proof_stream.len() <= STREAM_CAP); - // The guest holds one opening arm per candidate committed size, so a child - // outside that window has no arm to dispatch to. `min_log_committed` keeps - // every aggregate above the low end, leaving only the ceiling reachable. - if !(MU_MIN..=MU_MAX).contains(&layout.shape.mu) { - return Err(AggregationError::ChildOutOfRange { - log_committed: layout.shape.mu, - }); - } - - // ---- typed extraction: proof structs + the verifier's summary ---- - // Push and pull share the bytecode point. - let kbc = summary.bytecode_claim.point.len() - lean_vm::leaf::N_BYTECODE_SELECTORS; - - let taus = layout.taus; - // Flock replay data, all named struct fields. - let lcrounds = flock::hash::K_LOG - 6; - let zcf = [summary.zc_claim.a_eval, summary.zc_claim.b_eval]; - let zc_z = summary.zc_claim.z; - let zchi = &summary.zc_claim.mlv_challenges; - let lc_alpha = summary.lc_claim.alpha; - let lc_beta = summary.lc_claim.beta; - let lrr = &summary.lc_claim.r_rounds; - - // ---- the stacked opening: config + the opening summary ---- - let stack = whir_shape(layout.shape.mu, summary.log_inv_rate); - - // flock's reduction ends at `flock_stream_end`, where the WHIR opening's own - // scalars start: its last 64 scalars are lincheck's `z_partial` (which the - // summary already carries as `s_hat_v`), immediately preceded by the - // coefficient PAIRS of the `lcrounds` lincheck rounds: the linear one is not - // sent, the running claim fixing it. - let ns = summary.flock_stream_end; - let lcr = &proof_stream[ns - 64 - 2 * lcrounds..ns - 64]; - let lcz = &summary.lc_claim.s_hat_v; - - // matpart = the deferred weighted matrix evaluation: the lincheck running - // claim minus (= plus, char 2) the const-pin and c-claim contributions. - // α² from α, not from β: `LincheckClaim::beta` (the pin, at α³) is zero for - // a circuit with no const-pin column, while every verifier draws the - // c-claim's coefficient unconditionally. - let lc_sq = lc_alpha.square(); - let mut lrun = zcf[0] + lc_alpha * zcf[1] + lc_sq * summary.zc_claim.c_eval + lc_beta; - for i in 0..lcrounds { - let (c0, c2) = (lcr[2 * i], lcr[2 * i + 1]); - lrun = primitives::multilinear::poly_eval(&[c0, lrun + c2, c2], lrr[i]); - } - let mut pinw = lc_beta; - for (j, &rv) in lrr.iter().enumerate() { - let bit = (flock::hash::Z_CONST_POS >> (flock::hash::K_LOG - 1 - j)) & 1; - pinw *= if bit == 1 { rv } else { F192::ONE + rv }; - } - pinw *= lcz[flock::hash::Z_CONST_POS % 64]; - // The c term: eq(ρ_in, ρ'_in) times the φ8-Lagrange combination of the 64 - // slices, ρ'_in being the lincheck challenges read back in coordinate order. - let mut c_point_eq = F192::ONE; - for (t, &rin) in zchi[..lcrounds].iter().enumerate() { - c_point_eq *= F192::ONE + rin + lrr[lcrounds - 1 - t]; - } - let c_slice_value = primitives::multilinear::lagrange_weights_naive(6, zc_z) - .iter() - .zip(lcz) - .fold(F192::ZERO, |acc, (&w, &s)| acc + w * s); - let matpart = lrun + pinw + lc_sq * c_point_eq * c_slice_value; - - // ---- hints ---- - // The program's whole share of a bytecode leaf: ONE value, the stacked - // polynomial at (ζ_lo, α⃗), the slot coordinates of the claim's own point being - // the fingerprint challenges (§sec:e2e-bc). - let bytecode_value = summary.bytecode_claim.value; - let bcv = vec![bytecode_value]; - - // ---- per-sub HINT data (the placeholder map is built once, elsewhere) ---- - // Per side, the packing order read straight off `leaf::layout`'s offsets: - // sort_order[side_base + rank] = g^{side-local index of the rank-r block}. - // The guest only perm-checks it and derives offsets; any aligned tiling is - // sound, so this canonical order just has to match the committed leaf. - let mut sort_order: Vec = Vec::new(); - let mut gbase = 0usize; - for (s, blocks) in sides.iter().enumerate() { - let mut order: Vec = (0..blocks.len()).collect(); - order.sort_by_key(|&i| side_layouts[s].offsets[i]); - for &i in &order { - sort_order.push(F192::new(g_pow(gbase + i).0, 0, 0)); // g^{global block index} - } - gbase += blocks.len(); - } - // The stacked commitment uses witness::placements_of: committed columns - // sorted by descending kappa, then by their native column index. Transport - // compact committed-column indices; the guest certifies the permutation, - // ordering, and accumulated offsets before using them as claim selectors. - let col_sources = lean_vm::cpu::col_kappa_sources(kbc); - let committed_globals: Vec = col_sources - .iter() - .enumerate() - .filter_map(|(i, source)| source.map(|_| i)) - .collect(); - let mut compact_col = vec![usize::MAX; col_sources.len()]; - for (compact, &global) in committed_globals.iter().enumerate() { - compact_col[global] = compact; - } - let mut col_order = committed_globals; - col_order.sort_by_key(|&global| layout.placements[global].offset); - let col_sort_order: Vec = col_order - .iter() - .map(|&global| F192::new(g_pow(compact_col[global]).0, 0, 0)) - .collect(); - - // ---- Phase E2 hints (the stacked WHIR opening) ---- - // Share the upper Merkle tree across queries. Unknown, unused subtrees stay - // opaque; the guest authenticates every cap leaf a query reaches. - let mut query_hints = Vec::new(); - let (mut caps, mut cap_active) = (Vec::new(), Vec::new()); - let mut openings = summary.raw.merkle.iter(); - for (&queries, &depth) in stack.config.queries.iter().zip(&stack.depth) { - let cap_depth = merkle_cap_depth(queries, depth); - let n = 1 << cap_depth; - let path_depth = depth - cap_depth; - let mut nodes = vec![[0u8; 32]; 2 * n]; - let mut active = vec![F192::ZERO; n]; - for opening in openings.by_ref().take(queries) { - query_hints.push(( - "merkle_leaf_rows", - opening.leaf_data.iter().map(|x| F192::from(*x)).collect(), - )); - let mut path_children = Vec::with_capacity(4 * path_depth); - let bytes: Vec<_> = opening.leaf_data.iter().flat_map(|x| x.0.to_le_bytes()).collect(); - let mut node = pcs::merkle::hash_leaf(&bytes); - let mut index = (1 << depth) + opening.leaf_index; - for (height, sibling) in opening.path.iter().enumerate() { - if height >= path_depth { - nodes[index] = node; - nodes[index ^ 1] = *sibling; - active[index >> 1] = F192::ONE; - } - let (left, right) = if index & 1 == 0 { - (&node, sibling) - } else { - (sibling, &node) - }; - if height < path_depth { - path_children.extend(pack_hash_state(left)); - path_children.extend(pack_hash_state(right)); - } - node = pcs::merkle::hash_pair(left, right); - index >>= 1; - } - nodes[1] = node; - query_hints.push(("merkle_children", path_children)); - } - caps.extend(nodes.iter().flat_map(pack_hash_state)); - cap_active.extend(active); - } - let mut bytecode_row_point = summary.bytecode_claim.point; - let bytecode_selector_point = bytecode_row_point.split_off(kbc); - let deferred = DeferredSubproof { - public_input, - bytecode_row_point, - bytecode_selector_point, - bytecode_value, - matrix_a_coefficient: lc_alpha, - skip_point: zc_z, - zerocheck_row_point: zchi[..lcrounds].to_vec(), - lincheck_round_point: summary.lc_claim.r_rounds, - lincheck_terminal_values: summary.lc_claim.s_hat_v, - matrix_claim: matpart, - }; - - let mut hints = vec![ - ("stream", { - // The guest replays the WHIR opening off the same stream the native - // verifier reads: every transmitted scalar (sumcheck messages, level - // roots, OOD claims, grind nonces, `yr`) is already there in protocol - // order, so there is nothing to reassemble. The guest's `open_stacked` - // picks it up at `msg_cursor = cursor`, which sits where the flock - // reduction stopped; the ring-switch messages are struct-observed and - // still do not advance that cursor. - let mut stream = summary.raw.stream; - assert!( - stream.len() <= STREAM_CAP, - "stream {} exceeds cap {STREAM_CAP}", - stream.len() - ); - stream.resize(STREAM_CAP, F192::ZERO); - stream - }), - ("bytecode_val", bcv), - ("matpart", vec![matpart]), - ("merkle_caps", caps), - ("merkle_cap_active", cap_active), - // the table sumcheck's round count: max_t tau_t, certified in-guest as a - // maximum (one of the taus, and dominating them all). - ( - "zc_tau_max", - vec![F192::new(g_pow(*taus.iter().max().unwrap()).0, 0, 0)], - ), - ("col_sort_order", col_sort_order), - ("sort_order", sort_order), - ]; - hints.extend(query_hints); - Ok((hints, deferred)) -} - -/// The guest's stacked-size dispatch range: one `match_range` opening arm per -/// candidate `mu` in `MU_MIN..=MU_MAX` (mirrored by the soundness test's -/// residual-log cap). -const MU_MIN: usize = 22; -const MU_MAX: usize = lean_vm::pcs::MAX_MU; - -const _: () = assert!(MU_MIN >= lean_vm::pcs::MIN_MU); - -/// The guest's baked buffer caps, which `placeholder_map` compiles in and -/// `gen_verify` admits against: one definition, so a hinted shape can never -/// outgrow the buffer the guest was compiled with. -const MU_CAP: usize = 40; -const STREAM_CAP: usize = 8192; -/// Named hint entries for a single sub-proof, ordered within each stream. -type SubHints = Vec<(&'static str, Vec)>; - -/// One `hint_witness` stream: a name and its entries, in the order the guest -/// pops them. -#[derive(Default)] -pub(crate) struct Hints(Vec<(String, Vec>)>); - -impl Hints { - #[cfg(test)] - fn is_empty(&self) -> bool { - self.0.is_empty() - } - - fn push(&mut self, name: &str, entry: Vec) { - match self.0.iter_mut().find(|(n, _)| n == name) { - Some((_, entries)) => entries.push(entry), - None => self.0.push((name.to_string(), vec![entry])), - } - } - - /// The entries of one stream, for the adversarial test to corrupt. - #[cfg(test)] - fn entries(&mut self, name: &str) -> &mut Vec> { - &mut self - .0 - .iter_mut() - .find(|(n, _)| n == name) - .unwrap_or_else(|| panic!("no hint stream `{name}`")) - .1 - } - - fn install(self, guest: &mut Program) { - for (name, entries) in self.0 { - guest.set_witness(name, entries); - } - } -} - -/// The coverage slot each write in the guest's coverage walk targets, in walk -/// order: the raw signatures first (grouped by epoch, as the guest loops over -/// them), then each child's key lists. -/// -/// The table is one contiguous region per XMSS epoch group, then the SPHINCS -/// pair, each region its declared keys followed by its own duplicate slots: -/// -/// | slots | holds | -/// | --- | --- | -/// | `[0, n_0 + d_0)` | epoch group 0: declared keys, then duplicates | -/// | ... | one such region per declared group, in epoch order | -/// | ... | then one per covered-but-undeclared group, all duplicates | -/// | `[X, X + n_sphincs)` | the declared SPHINCS keys | -/// | `[X + n_sphincs, n_total)` | SPHINCS duplicate slots | -/// -/// with `X` the sum of every group's slots. A key the declared set holds takes its -/// slot there, one it does not takes a fresh duplicate slot in the same region, so -/// the walk hits every one of the `n_total` slots exactly once. -/// That bijection, enforced in-circuit by write-once memory plus the final -/// count, is what makes every declared key covered by a real signature or a -/// verified child. -/// -/// Keeping each region contiguous is what binds the scheme and the epoch: the -/// guest addresses every writer as an offset into one region, bounded by that -/// region's size, one range check per write, so no signature can reach a key -/// declared under another epoch or the other scheme. -struct Coverage { - /// Declared groups in statement order, then undeclared ones, all duplicates. - xmss_groups: Vec, - /// How many of `xmss_groups` the signer set declares. - n_declared: usize, - /// One duplicate list per group, aligned with `xmss_groups`. - xmss_dups: Vec>, - sphincs_signers: Vec, - sphincs_dups: Vec, - /// Offsets within each raw signature's own group region, indexed as `raw_xmss`. - raw_xmss: Vec, - /// `raw_xmss` indices in the guest's walk order: it walks the table, whose - /// groups are declared-first rather than epoch-sorted. - raw_walk: Vec, - /// Offsets past `X`, in the SPHINCS region. - raw_sphincs: Vec, - /// Per child, per child group: the parent group it maps to, and the offset - /// of each child key within that parent group's region. - child_xmss: Vec)>>, - child_sphincs: Vec>, -} - -impl Coverage { - fn declared(&self) -> &[XmssClaimGroup] { - &self.xmss_groups[..self.n_declared] - } - - fn n_keys(&self) -> usize { - self.xmss_groups.iter().map(|group| group.keys.len()).sum::() + self.sphincs_signers.len() - } - - fn n_total(&self) -> usize { - self.n_keys() + self.xmss_dups.iter().map(Vec::len).sum::() + self.sphincs_dups.len() - } -} - -/// A claim takes its declared slot once; subsequent or omitted occurrences take -/// fresh slots outside the hashed prefix. -fn take_slot(claims: &[K], claimed: &mut [bool], duplicates: &mut Vec, claim: &K) -> usize { - match claims.binary_search(claim) { - Ok(pos) if !claimed[pos] => { - claimed[pos] = true; - pos - } - _ => { - duplicates.push(claim.clone()); - claims.len() + duplicates.len() - 1 - } - } -} - -fn plan_coverage( - raw_xmss: &[(XmssPublicKey, xmss::Epoch, xmss::Message)], - raw_sphincs: &[SphincsClaim], - children: &[EthereumProof], - declare: Option<&SignatureClaims>, -) -> Result { - // The union, as `(epoch, message, key)` claims, then grouped: consecutive - // equal `(epoch, message)` pairs of the sorted deduplicated list are one group. - let mut claims: Vec<(xmss::Epoch, xmss::Message, XmssPublicKey)> = Vec::with_capacity(raw_xmss.len()); - for (pk, epoch, message) in raw_xmss { - claims.push((*epoch, *message, pk.clone())); - } - let mut sphincs_signers = raw_sphincs.to_vec(); - for child in children { - for XmssClaimGroup { epoch, message, keys } in &child.xmss_signers { - claims.extend(keys.iter().map(|pk| (*epoch, *message, pk.clone()))); - } - sphincs_signers.extend_from_slice(&child.sphincs_signers); - } - claims.sort(); - claims.dedup(); - // On the whole pair, so one key signing two messages is two claims. - sphincs_signers.sort(); - sphincs_signers.dedup(); - let mut union_groups: Vec = Vec::new(); - for (epoch, message, pk) in claims { - match union_groups.last_mut() { - Some(group) if (group.epoch, group.message) == (epoch, message) => group.keys.push(pk), - _ => union_groups.push(XmssClaimGroup { - epoch, - message, - keys: vec![pk], - }), - } - } - // Groups the declaration holds nothing of go last, so the declared ones are the - // prefix the digest hashes. Claims are struck off, so leftovers are uncovered. - let mut wanted: BTreeSet<(xmss::Epoch, xmss::Message, XmssPublicKey)> = BTreeSet::new(); - let mut wanted_sphincs: BTreeSet = BTreeSet::new(); - if let Some(SignatureClaims { xmss, sphincs }) = declare { - for XmssClaimGroup { epoch, message, keys } in xmss { - wanted.extend(keys.iter().map(|key| (*epoch, *message, key.clone()))); - } - wanted_sphincs.extend(sphincs.iter().copied()); - } - let mut xmss_groups: Vec = Vec::new(); - let mut covered_only: Vec = Vec::new(); - for mut group in union_groups { - group - .keys - .retain(|key| declare.is_none() || wanted.remove(&(group.epoch, group.message, key.clone()))); - if group.keys.is_empty() { - covered_only.push(group); - } else { - xmss_groups.push(group); - } - } - let n_declared = xmss_groups.len(); - xmss_groups.append(&mut covered_only); - if xmss_groups.len() > MAX_EPOCHS { - return Err(AggregationError::TooManyEpochs); - } - if declare.is_some() { - sphincs_signers.retain(|signer| wanted_sphincs.remove(signer)); - if !wanted.is_empty() || !wanted_sphincs.is_empty() { - return Err(AggregationError::NotCovered); - } - } - // The table is no longer sorted by `(epoch, message)`. - let region_of: BTreeMap<(xmss::Epoch, xmss::Message), usize> = xmss_groups - .iter() - .enumerate() - .map(|(j, group)| ((group.epoch, group.message), j)) - .collect(); - let mut xmss_claimed: Vec> = xmss_groups.iter().map(|group| vec![false; group.keys.len()]).collect(); - let mut xmss_dups: Vec> = vec![Vec::new(); xmss_groups.len()]; - let mut sphincs_claimed = vec![false; sphincs_signers.len()]; - let mut sphincs_dups = Vec::new(); - let raw_xmss_slots: Vec = raw_xmss - .iter() - .map(|(pk, epoch, message)| { - let g = region_of[&(*epoch, *message)]; - take_slot(&xmss_groups[g].keys, &mut xmss_claimed[g], &mut xmss_dups[g], pk) - }) - .collect(); - let raw_sphincs_slots: Vec = raw_sphincs - .iter() - .map(|signer| take_slot(&sphincs_signers, &mut sphincs_claimed, &mut sphincs_dups, signer)) - .collect(); - let mut child_xmss = Vec::with_capacity(children.len()); - let mut child_sphincs = Vec::with_capacity(children.len()); - for child in children { - child_xmss.push( - child - .xmss_signers - .iter() - .map(|group| { - let g = region_of[&(group.epoch, group.message)]; - let offsets = group - .keys - .iter() - .map(|pk| take_slot(&xmss_groups[g].keys, &mut xmss_claimed[g], &mut xmss_dups[g], pk)) - .collect(); - (g, offsets) - }) - .collect::)>>(), - ); - child_sphincs.push( - child - .sphincs_signers - .iter() - .map(|signer| take_slot(&sphincs_signers, &mut sphincs_claimed, &mut sphincs_dups, signer)) - .collect(), - ); - } - // Stable, so signatures within a group keep the order their slots were taken in. - let mut raw_walk: Vec = (0..raw_xmss.len()).collect(); - raw_walk.sort_by_key(|&i| region_of[&(raw_xmss[i].1, raw_xmss[i].2)]); - let cover = Coverage { - xmss_groups, - n_declared, - xmss_dups, - sphincs_signers, - sphincs_dups, - raw_xmss: raw_xmss_slots, - raw_walk, - raw_sphincs: raw_sphincs_slots, - child_xmss, - child_sphincs, - }; - if cover.n_total() >= MAX_KEYS { - return Err(AggregationError::TooLarge); - } - Ok(cover) -} -/// One signature's witness: the WOTS randomness, the encoding digits (in the -/// exponent), the chain tips they start from, and the Merkle siblings. -fn push_signature_hints( - hints: &mut Hints, - pk: &XmssPublicKey, - sig: &XmssSignature, - message: &xmss::Message, - xmss_epoch: xmss::Epoch, -) -> Result<(), AggregationError> { - let wots = &sig.wots_signature; - let encoding = xmss::wots_encode(message, xmss_epoch, &pk.public_param, &wots.randomness) - .ok_or(AggregationError::MalformedRawSignature)?; - let mut randomness = [0u8; xmss::STATE_LEN]; - randomness[..xmss::RANDOMNESS_LEN].copy_from_slice(&wots.randomness); - hints.push( - "rand", - vec![pack_16_bytes(&randomness[..16]), pack_16_bytes(&randomness[16..])], - ); - for &e in &encoding { - hints.push("digits", vec![count(e as usize)]); - } - for tip in &wots.chain_tips { - hints.push("chain_starts", vec![pack_16_bytes(tip)]); - } - for sibling in &sig.merkle_proof { - hints.push("siblings", vec![pack_16_bytes(sibling)]); - } - Ok(()) -} - -/// One SPHINCS signature's witness: the randomizer, the few-time opening, and -/// per layer the encoding counter, the codeword digits (in the exponent), the -/// chain values they start from, and the Merkle siblings. -/// -/// The guest derives the index and the leaf indices from the digest itself, so -/// nothing here carries them; what it does carry is the per-layer message, which -/// this walk recomputes exactly as the guest will. The signer's own message is -/// not hinted either: it rides its slot in the coverage table. -fn push_sphincs_hints( - hints: &mut Hints, - (pk, message): &SphincsClaim, - sig: &SphincsSignature, -) -> Result<(), AggregationError> { - let pp = &pk.public_param; - hints.push("sp_rand", vec![pack_16_bytes(&sig.randomizer)]); - let (idx, u) = sphincs::message_digest(pp, &pk.root, &sig.randomizer, message); - for kappa in 0..sphincs::NUM_FTS_TREES { - hints.push("sp_fts_secrets", vec![pack_16_bytes(&sig.fts.secrets[kappa])]); - for sibling in &sig.fts.paths[kappa] { - hints.push("sp_fts_paths", vec![pack_16_bytes(sibling)]); - } - } - let mut signed = sphincs::fts_recover(pp, idx, &u, &sig.fts); - for lay in (0..sphincs::D).rev() { - let pos = sphincs::Pos::new(lay, sphincs::tree_of(idx, lay), sphincs::leaf_of(idx, lay)); - let counter = sig.counters[lay]; - let codeword = sphincs::encode(pp, pos, &signed, counter).ok_or(AggregationError::MalformedRawSignature)?; - hints.push("sp_counter", vec![F192::new(u64::from(counter), 0, 0)]); - for (&digit, opened) in codeword.iter().zip(&sig.ots[lay]) { - hints.push("sp_digits", vec![count(digit as usize)]); - hints.push("sp_chain_starts", vec![pack_16_bytes(opened)]); - } - let path = &sig.paths[sphincs::path_range(lay)]; - for sibling in path { - hints.push("sp_siblings", vec![pack_16_bytes(sibling)]); - } - let leaf = sphincs::ots_leaf(pp, pos, &signed, counter, &sig.ots[lay]) - .ok_or(AggregationError::MalformedRawSignature)?; - signed = sphincs::tree_fold(pp, pos, leaf, path); - } - debug_assert_eq!(signed, pk.root, "the hinted walk reaches the public key"); - Ok(()) -} -#[derive(Clone, Copy, Default)] -pub(crate) struct DaInput<'a> { - pub rows: &'a [u64], - pub roots: Option<&'a [[u8; 32]]>, -} - -/// Prove existence of signatures and valid encoding of PQ-blobs, potentially using recursive children. -/// -/// - `children`: child proofs; at most [`MAX_RECURSIONS`]. -/// - `raw_xmss`: list of `(public_key, epoch, message, signature)`, any order; at most -/// [`MAX_EPOCHS`] distinct `(epoch, message)` pairs across the whole result. -/// - `raw_sphincs`: list of `(public_key, message, signature)`, any order. -/// - `blobs`: concatenated blobs, each containing [`BLOB_SYMBOLS`] little-endian `u64` symbols; -/// at most [`DA_MAX_ROWS`] blobs. -/// - `declare`: `None` keeps all claims; `Some` specifies exactly the signatures and DA roots -/// to publish. Every declared claim must be covered by the inputs above. -/// - `log_inv_rate`: PCS code rate `2^-log_inv_rate`, higher `log_inv_rate` means a smaller proof -/// but slower proving; in [`MIN_LOG_INV_RATE`]..=[`MAX_LOG_INV_RATE`]. -/// -/// The combined XMSS and SPHINCS claim count, including duplicate coverage slots, must be strictly below [`MAX_KEYS`]. -/// -/// IMPORTANT: -/// - `aggregate` should not be called more than once at a time in parallel per process. -/// - Raw signatures are assumed valid (otherwise `aggregate` will panic). -/// -/// XMSS Performance: it is optimized for a small set of different (epoch, message), and many XMSS -/// sharing each such pair. -pub fn aggregate( - children: &[EthereumProof], - raw_xmss: Vec<(XmssPublicKey, xmss::Epoch, xmss::Message, XmssSignature)>, - raw_sphincs: Vec<(SphincsPublicKey, sphincs::Message, SphincsSignature)>, - blobs: &[u64], - declare: Option>, - log_inv_rate: usize, -) -> Result { - let da_input = DaInput { - rows: blobs, - roots: declare.map(|claims| claims.da_commitments), - }; - aggregate_with_stats( - children, - raw_xmss, - raw_sphincs, - declare.map(|claims| claims.signatures), - da_input, - log_inv_rate, - ) - .map(|(sig, _)| sig) -} - -/// [`aggregate`], keeping the prover statistics the benchmark reports. -pub(crate) fn aggregate_with_stats( - children: &[EthereumProof], - raw_xmss: Vec<(XmssPublicKey, xmss::Epoch, xmss::Message, XmssSignature)>, - raw_sphincs: Vec<(SphincsPublicKey, sphincs::Message, SphincsSignature)>, - declare: Option<&SignatureClaims>, - da_input: DaInput<'_>, - log_inv_rate: usize, -) -> Result<(EthereumProof, lean_vm::cpu::Stats), AggregationError> { - aggregate_tampered(children, raw_xmss, raw_sphincs, declare, da_input, log_inv_rate, |_| {}) -} - -/// [`aggregate`], with a hook to corrupt the witness before proving. -/// -/// The coverage argument and the claim batching are enforced entirely by guest -/// asserts over prover advice, so the only way to test them is to lie in a hint -/// and require the guest to notice. That is what `tamper` is for -/// (`aggregate_hints_bind`); with an empty hook this is the production path. -pub(crate) fn aggregate_tampered( - children: &[EthereumProof], - raw_xmss: Vec<(XmssPublicKey, xmss::Epoch, xmss::Message, XmssSignature)>, - raw_sphincs: Vec<(SphincsPublicKey, sphincs::Message, SphincsSignature)>, - declare: Option<&SignatureClaims>, - da_input: DaInput<'_>, - log_inv_rate: usize, - tamper: impl FnOnce(&mut Hints), -) -> Result<(EthereumProof, lean_vm::cpu::Stats), AggregationError> { - // Otherwise this reaches `cpu::prove`, which asserts rather than reporting. - if !(lean_vm::pcs::MIN_LOG_INV_RATE..=lean_vm::pcs::MAX_LOG_INV_RATE).contains(&log_inv_rate) { - return Err(AggregationError::InvalidRate { log_inv_rate }); - } - if children.len() > MAX_RECURSIONS { - return Err(AggregationError::TooLarge); - } - let rows = da_input.rows; - if !rows.len().is_multiple_of(BLOB_SYMBOLS) || rows.len() / BLOB_SYMBOLS > DA_MAX_ROWS { - return Err(AggregationError::InvalidBlobSize { symbols: rows.len() }); - } - let mut available_roots = BTreeSet::new(); - for child in children { - check_da_roots(&child.da_roots).map_err(AggregationError::InvalidChild)?; - available_roots.extend(child.da_roots.iter().copied()); - } - let mut da_roots = match da_input.roots { - Some(roots) => roots.iter().copied().collect::>().into_iter().collect(), - None => available_roots.iter().copied().collect::>(), - }; - if da_roots.len() > MAX_DA_ROOTS { - return Err(AggregationError::TooLarge); - } - if rows.is_empty() && da_roots.iter().any(|root| !available_roots.contains(root)) { - return Err(AggregationError::BlobNotCovered); - } - - let guest = unified_guest(); - // Sorted by `(epoch, message, key)` to group them; `Coverage::raw_walk` then - // puts the groups in the guest's order. - let mut raw_xmss = raw_xmss; - raw_xmss.sort_by(|(a, ae, am, _), (b, be, bm, _)| (ae, am, a).cmp(&(be, bm, b))); - raw_xmss.dedup_by(|(a, ae, am, _), (b, be, bm, _)| (ae, am, a) == (be, bm, b)); - // On the whole (key, message) pair, so a signer may appear once per message. - let mut raw_sphincs = raw_sphincs; - raw_sphincs.sort_by_key(|(pk, message, _)| (*pk, *message)); - raw_sphincs.dedup_by(|(a, am, _), (b, bm, _)| (a, am) == (b, bm)); - - // Verifying a child here is not a courtesy: `gen_verify` derives the guest's - // whole witness for it from a real verification's summary. Its deferred - // claim is deliberately NOT recomputed, which would cost a full pass over - // each fixed polynomial per child: the batching sumcheck below already - // forces every batched value to be the true evaluation, and the root - // discharges the one claim they reduce to. - let mut verified = Vec::with_capacity(children.len()); - let _span = tracing::info_span!("Verify children").entered(); - for child in children { - check_signer_set(&child.xmss_signers, &child.sphincs_signers).map_err(AggregationError::InvalidChild)?; - let pi = child.public_input(); - let summary = verify(guest, &pi, &child.proof) - .map_err(|e| AggregationError::InvalidChild(AggregateVerifyError::Snark(e)))?; - verified.push((pi, summary)); - } - - drop(_span); - - let _span = tracing::info_span!("Build witness").entered(); - let raw_xmss_claims: Vec<(XmssPublicKey, xmss::Epoch, xmss::Message)> = raw_xmss - .iter() - .map(|(pk, epoch, message, _)| (pk.clone(), *epoch, *message)) - .collect(); - let raw_sphincs_keys: Vec = raw_sphincs.iter().map(|(pk, message, _)| (*pk, *message)).collect(); - let cover = plan_coverage(&raw_xmss_claims, &raw_sphincs_keys, children, declare)?; - let da_contributions = - usize::from(!rows.is_empty()) + children.iter().map(|child| child.da_roots.len()).sum::(); - if cover.n_total() + da_contributions >= MAX_KEYS { - return Err(AggregationError::TooLarge); - } - let n_sphincs = cover.sphincs_signers.len(); - let group_cells = |group: &XmssClaimGroup| { - [ - F192::new(group.epoch as u64, 0, 0), - pack_16_bytes(&group.message[..16]), - pack_16_bytes(&group.message[16..]), - ] - }; - - let mut hints = Hints::default(); - hints.push( - "meta", - vec![ - count(cover.n_declared), - count(cover.xmss_groups.len() - cover.n_declared), - count(n_sphincs), - count(cover.sphincs_dups.len()), - count(raw_sphincs.len()), - count(children.len()), - count(usize::from(!rows.is_empty())), - ], - ); - let fs_seed = lean_vm::cpu::fs_seed(guest); - hints.push("fs_seed", vec![fs_seed[0], fs_seed[1]]); - // Per group: its epoch, its two message cells, and its declared, duplicate - // and raw-signature counts, in the guest's geometry-pass order. The keys - // then ride two per `pubkeys` entry, so the guest can halve its loop - // frames, the odd key out on a final one-key entry; each group's - // duplicates follow its keys. - for (j, group) in cover.xmss_groups.iter().enumerate() { - let mut entry = group_cells(group).to_vec(); - entry.extend([ - count(group.keys.len()), - count(cover.xmss_dups[j].len()), - count( - raw_xmss - .iter() - .filter(|(_, e, m, _)| (*e, *m) == (group.epoch, group.message)) - .count(), - ), - ]); - hints.push("group", entry); - } - for XmssClaimGroup { keys, .. } in cover.declared() { - hints.push("pk_halves", vec![count(keys.len() / 2), count(keys.len() % 2)]); - hints.push("signers_split", signers_split(keys.len().div_ceil(2))); - for pair in keys.chunks(2) { - let mut entry = key_cells(&pair[0]).to_vec(); - if let Some(second) = pair.get(1) { - entry.extend_from_slice(&key_cells(second)); - } - hints.push("pubkeys", entry); - } - } - for dups in &cover.xmss_dups { - for pk in dups { - hints.push("dup_pubkeys", key_cells(pk).to_vec()); - } - } - if !cover.sphincs_signers.is_empty() { - hints.push("signers_split", signers_split(cover.sphincs_signers.len())); - } - hints.push("signers_split", signers_split(1 + 2 * cover.n_declared)); - for signer in &cover.sphincs_signers { - hints.push("sphincs_signers", sphincs_signer_cells(signer).to_vec()); - } - for signer in &cover.sphincs_dups { - hints.push("dup_sphincs", sphincs_signer_cells(signer).to_vec()); - } - // Group-major over the table, not over the epochs `raw_xmss` is sorted by: a - // declaration puts the undeclared groups last. Each index is an offset within - // the signature's own group region. - for &i in &cover.raw_walk { - let (pk, epoch, message, sig) = &raw_xmss[i]; - hints.push("raw_index", vec![count(cover.raw_xmss[i])]); - push_signature_hints(&mut hints, pk, sig, message, *epoch)?; - } - // A SPHINCS slot is hinted as an offset into the SPHINCS region, which is - // how one range check keeps the scheme's writers off the other's keys. - for (&offset, (pk, message, sig)) in cover.raw_sphincs.iter().zip(&raw_sphincs) { - hints.push("sp_raw_index", vec![count(offset)]); - push_sphincs_hints(&mut hints, &(*pk, *message), sig)?; - } - - let mut subs = Vec::with_capacity(children.len()); - let mut carried = Vec::with_capacity(children.len()); - for (i, (child, (pi, summary))) in children.iter().zip(verified).enumerate() { - hints.push( - "child_meta", - vec![count(child.xmss_signers.len()), count(child.sphincs_signers.len())], - ); - for (group, (parent_group, offsets)) in child.xmss_signers.iter().zip(&cover.child_xmss[i]) { - let mut entry = group_cells(group).to_vec(); - entry.push(count(group.keys.len())); - hints.push("child_group", entry); - hints.push("child_group_map", vec![count(*parent_group)]); - hints.push("child_halves", vec![count(offsets.len() / 2), count(offsets.len() % 2)]); - hints.push("signers_split", signers_split(offsets.len().div_ceil(2))); - for pair in offsets.chunks(2) { - hints.push("child_index", pair.iter().map(|&idx| count(idx)).collect()); - } - } - if !cover.child_sphincs[i].is_empty() { - hints.push("signers_split", signers_split(cover.child_sphincs[i].len())); - } - for &offset in &cover.child_sphincs[i] { - hints.push("child_sphincs_index", vec![count(offset)]); - } - hints.push("signers_split", signers_split(1 + 2 * child.xmss_signers.len())); - hints.push("child_defer", child.defer.cells()); - hints.push("child_da_count", vec![count(child.da_roots.len())]); - let (sub_hints, defer) = gen_verify(guest, pi, summary)?; - for (name, entry) in sub_hints { - hints.push(name, entry); - } - subs.push(defer); - carried.push(&child.defer); - } - - drop(_span); - - let defer = if children.is_empty() { - let leaf = DeferredClaim::leaf(); - hints.push( - "leaf_defer", - vec![leaf.bytecode_value, leaf.matrix_a_value, leaf.matrix_b_value], - ); - leaf - } else { - let _span = tracing::info_span!("Batch deferred claims").entered(); - let (agg_hints, reduced) = aggregate_deferred_claims(&subs, &carried); - for (name, entry) in agg_hints { - hints.push(name, entry); - } - reduced - }; - - let direct_root = if rows.is_empty() { - None - } else { - let _span = tracing::info_span!("LeanDA commit").entered(); - let n_rows = rows.len() / BLOB_SYMBOLS; - let (commitment, witness) = lean_da::commit(rows); - for block in lean_da::membership_vector(&commitment.root) - .as_chunks::() - .0 - { - hints.push("da_weights", block.to_vec()); - } - hints.push( - "da_shape", - vec![count(n_rows), count(n_rows.next_power_of_two().ilog2() as usize)], - ); - // Padding rows are constants the guest bakes, so only the real rows' - // symbols ride the stream. - let (c, m) = (CELL_SYMBOLS, CODEWORD_SYMBOLS); - for j in 0..CELLS_PER_ROW { - for i in 0..n_rows { - hints.push( - "da_symbols", - witness.codewords[i * m + j * c..i * m + (j + 1) * c] - .iter() - .map(|&w| F192::from(F64(w))) - .collect(), - ); - } - } - Some(commitment.root) - }; - if let Some(root) = direct_root { - available_roots.insert(root); - if da_input.roots.is_none() { - da_roots = available_roots.iter().copied().collect(); - } - if da_roots.len() > MAX_DA_ROOTS { - return Err(AggregationError::TooLarge); - } - } - if cover.n_declared == 0 && cover.sphincs_signers.is_empty() && da_roots.is_empty() { - return Err(AggregationError::Empty); - } - let mut claimed = vec![false; da_roots.len()]; - let mut da_dups = Vec::new(); - for root in direct_root - .iter() - .chain(children.iter().flat_map(|child| &child.da_roots)) - { - let slot = take_slot(&da_roots, &mut claimed, &mut da_dups, root); - hints.push("da_index", vec![count(slot)]); - } - if claimed.contains(&false) { - return Err(AggregationError::BlobNotCovered); - } - hints.push("da_meta", vec![count(da_roots.len()), count(da_dups.len())]); - for root in da_roots.iter().chain(&da_dups) { - hints.push("da_roots", da_claim_cells(root)); - } - - let public_input = statement_digest( - signers_hash(cover.declared(), &cover.sphincs_signers), - da_list_digest(&da_roots), - &defer, - ); - let mut program = guest.clone(); - // Every aggregate is a potential child, and the guest has no opening arm below - // `2^MU_MIN`. A run smaller than that (a leaf of a few dozen signatures) grows - // its SET table until it clears the floor. - program.min_log_committed = MU_MIN; - tamper(&mut hints); - hints.install(&mut program); - let (proof, stats) = prove(&program, public_input, log_inv_rate).map_err(AggregationError::ProofError)?; - Ok(( - EthereumProof { - xmss_signers: cover.declared().to_vec(), - sphincs_signers: cover.sphincs_signers, - da_roots, - defer, - proof, - }, - stats, - )) -} -struct CoordinateDescriptor { - kind: usize, - constant: u128, - fresh: usize, - claim_slot: usize, - terms: Range, -} - -struct Term { - kind: usize, - constant: u128, - column_a: usize, - column_b: usize, -} - -struct ClaimDescriptor { - buffer: usize, - column: usize, - qflock_slot: usize, -} - -fn literals(values: impl IntoIterator) -> String { - format!( - "[{}]", - values.into_iter().map(|v| v.to_string()).collect::>().join(", ") - ) -} - -struct OpeningShape { - n_levels: usize, - yr_level: usize, - yr_log_len: usize, - folds: Vec, - log_message_columns: Vec, - queries: Vec, - tree_depths: Vec, - positions_per_squeeze: Vec, - squeezes: Vec, - interleaving: Vec, - query_grinding_bits: Vec, - cap_depths: Vec, - cap_offsets: Vec, - positions_offsets: Vec, - vanish_offsets: Vec, - fold_offsets: Vec, - residual_fold_offsets: Vec, - vanish_values: Vec, - vanish_inverses: Vec, - ood_samples: Vec, -} - -/// The recursion program's placeholder map depends on table structure and bytecode -/// size, so one compiled guest serves every proof layout. -fn placeholder_map(kbc: usize) -> BTreeMap { - // Only block and coordinate structure is used here; dummy instructions and - // table sizes let us derive it before the guest's bytecode exists. - let stand_in = vec![lean_vm::cpu::Op::Xor { a: 0, b: 0, c: 0 }; 1 << kbc]; - let layout = lean_vm::cpu::layout(&stand_in, 20, [10; lean_vm::tables::N_TABLES], [F192::ZERO, F192::ZERO]); - let sides: [&[Block]; 3] = [&layout.push, &layout.pull, &layout.count]; - let lcrounds = flock::hash::K_LOG - 6; - - // ---- flattened block/coord descriptors (structural) ---- - let mut sblk = vec![0usize]; - let mut block_coords = Vec::new(); - let mut coordinates = Vec::new(); - let mut terms = Vec::new(); - let (mut nclaims, mut nbcv, mut nblocks) = (0usize, 0usize, 0usize); - // Claim dedup (mirrors leaf.rs): per coord, fresh = first (group, col, - // kappa) occurrence gets the next pool slot; duplicates point at it. - let mut slot_of: std::collections::HashMap<(usize, usize), usize> = Default::default(); - // A TABLE block's coordinates, flattened into terms: the guest rebuilds each as - // `Σ_terms`, so a derived value (an XOR/MUL result, a DEREF store, a JUMP - // successor) costs terms rather than columns. A framework coordinate has none: - // it decomposes into pooled claims instead. - // The table sumcheck settles table claims; only framework blocks stream column values. - let sch_pm = lean_vm::cpu::schema(); - let owner_pm: Vec> = lean_vm::cpu::block_kappa_sources(kbc) - .into_iter() - .map(|(src, _)| src.checked_sub(2)) - .collect(); - for blocks in sides.iter() { - for blk in blocks.iter() { - block_coords.push(coordinates.len()..coordinates.len() + blk.coords.len()); - let owner = owner_pm[nblocks]; - nblocks += 1; - for c in &blk.coords { - // One COORD_FRESH/COORD_CLAIM_SLOT entry PER coord (the guest - // indexes them by global coord offset); only a framework block's - // Col/GCol raises a claim. - let (mut fresh, mut slot) = (0usize, 0usize); - if let (Coord::Col(i) | Coord::GCol(i, _), None) = (c, owner) { - let key = (*i, blk.kappa); - if let Some(&known) = slot_of.get(&key) { - slot = known; - } else { - slot_of.insert(key, nclaims); - fresh = 1; - slot = nclaims; - nclaims += 1; - } - } - let start = terms.len(); - if let Some(t) = owner { - push_coord_terms(c, sch_pm.base[t], &mut terms); - } - nbcv += usize::from(matches!(c, Coord::Public(_))); - coordinates.push(CoordinateDescriptor { - kind: coord_kind(c), - constant: dsl_u128(coord_scale(c)), - fresh, - claim_slot: slot, - terms: start..terms.len(), - }); - } - } - sblk.push(nblocks); - } - let evtot: usize = lean_vm::tables::tables().iter().map(|t| t.n_committed_columns()).sum(); - let ncl = nclaims + evtot + 3; // bus + constraint + the three PI memory-limb claims - - // ---- claim descriptor buffer ids (structural) ---- - let valcols = blake2s_value_columns(); - let col_sources_pm = lean_vm::cpu::col_kappa_sources(kbc); - let mut compact_col_pm = vec![usize::MAX; col_sources_pm.len()]; - let mut n_committed = 0usize; - for (global, source) in col_sources_pm.iter().enumerate() { - if source.is_some() { - compact_col_pm[global] = n_committed; - n_committed += 1; - } - } - let qflock_compact = compact_col_pm[lean_vm::cpu::QFLOCK]; - assert_ne!(qflock_compact, usize::MAX, "QFLOCK must be committed"); - // Buffer codes are the guest's POINT_BUF_*: zeta, chi, pi, qflock-chi. - let mut claims = Vec::new(); - walk_claims(&layout, kbc, |site| { - let descriptor = match site { - ClaimSite::Framework { column, .. } => { - let column = compact_col_pm[column]; - assert_ne!(column, usize::MAX, "framework claim must target a committed column"); - ClaimDescriptor { - buffer: 0, - column, - qflock_slot: 0, - } - } - ClaimSite::TableColumn { column, is_virtual, .. } => ClaimDescriptor { - buffer: if is_virtual { 3 } else { 1 }, - column: if is_virtual { - qflock_compact - } else { - compact_col_pm[column] - }, - qflock_slot: if is_virtual { - lean_vm::hash_flock::SLOTS[valcols.iter().position(|&v| v == column).unwrap()] - } else { - 0 - }, - }, - ClaimSite::MemoryLimb { column } => ClaimDescriptor { - buffer: 2, - column: compact_col_pm[column], - qflock_slot: 0, - }, - }; - claims.push(descriptor); - }); - assert_eq!(claims.len(), ncl, "descriptor count == pool size"); - - // ---- the placeholder map ---- - let flds = |v: &[F192]| { - format!( - "[{}]", - v.iter().map(|&x| f192_literal(x)).collect::>().join(", ") - ) - }; - let mut rep = BTreeMap::new(); - let mut ps = |k: &str, v: String| { - rep.insert(format!("{k}_PLACEHOLDER"), v); - }; - ps("STREAM_CAP", STREAM_CAP.to_string()); - ps("MIN_LOG_MEM", lean_vm::cpu::MIN_LOG_MEM.to_string()); - ps("INV_GEN", dsl_u128(F192::new(G.inv().0, 0, 0)).to_string()); - ps("MU_CAP", MU_CAP.to_string()); - ps("NO_TABLE", layout.taus.len().to_string()); - ps("GKR_ROUNDS_CAP", (MU_CAP * (MU_CAP + 1) / 2 + MU_CAP + 2).to_string()); - ps("GKR_POINTS_CAP", ((MU_CAP + 1) * MU_CAP).to_string()); - ps("SIDE_BLOCK_START", literals(&sblk)); - ps("N_BLOCKS", nblocks.to_string()); - let bks = lean_vm::cpu::block_kappa_sources(kbc); - // Push and pull emit bus blocks in matched pairs, so their baked kappa-source - // segments are identical; the guest computes only push's side total and - // aliases pull's mu to push's on this basis. - assert_eq!( - bks[sblk[0]..sblk[1]], - bks[sblk[1]..sblk[2]], - "push/pull kappa sources must match" - ); - ps("BLOCK_KAPPA_SRC", literals(bks.iter().map(|&(s, _)| s))); - ps("BLOCK_KAPPA_ADJ", literals(bks.iter().map(|&(_, a)| a))); - ps( - "BLOCK_TABLE", - literals(bks.iter().map(|&(s, _)| if s >= 2 { s - 2 } else { layout.taus.len() })), - ); - let mut block_side = Vec::new(); - for (s, blocks) in sides.iter().enumerate() { - block_side.extend(std::iter::repeat_n(s, blocks.len())); - } - ps("BLOCK_SIDE", literals(&block_side)); - ps("BLOCK_COORD_OFF", literals(block_coords.iter().map(|r| r.start))); - ps("BLOCK_COORD_COUNT", literals(block_coords.iter().map(|r| r.len()))); - ps("COORD_TYPE", literals(coordinates.iter().map(|c| c.kind))); - ps("COORD_CONST", literals(coordinates.iter().map(|c| c.constant))); - ps("COORD_FRESH", literals(coordinates.iter().map(|c| c.fresh))); - ps("COORD_CLAIM_SLOT", literals(coordinates.iter().map(|c| c.claim_slot))); - ps("COORD_TERM_OFF", literals(coordinates.iter().map(|c| c.terms.start))); - ps("COORD_TERM_COUNT", literals(coordinates.iter().map(|c| c.terms.len()))); - ps("TERM_TYPE", literals(terms.iter().map(|t| t.kind))); - ps("TERM_CONST", literals(terms.iter().map(|t| t.constant))); - ps("TERM_COL_A", literals(terms.iter().map(|t| t.column_a))); - ps("TERM_COL_B", literals(terms.iter().map(|t| t.column_b))); - ps("N_BUS_CLAIMS", nclaims.to_string()); - let idxc: Vec = (0..34) - .map(|i| { - let mut g2k = F192::new(G.0, 0, 0); - for _ in 0..i { - g2k = g2k * g2k; - } - dsl_u128(F192::ONE + g2k) - }) - .collect(); - ps("INDEX_MLE_FACTORS", literals(&idxc)); - ps("N_CLAIMS", ncl.to_string()); - ps("N_TABLES", layout.taus.len().to_string()); - // The table sumcheck's xi layout, from the native verifier's own numbers: - // a disjoint range of identities per table, then THREE powers shared by every - // table, one per bus side. Sharing is what lets the target be derived from the - // three leaf claims instead of trusted (lean_vm::cpu::xi_form_base). - let n_id: Vec = lean_vm::tables::tables().iter().map(|t| t.n_constraints()).collect(); - let form_base = lean_vm::cpu::xi_form_base(); - ps( - "ETA_OFFSET", - literals(lean_vm::constraints::xi_offsets(n_id.iter().copied())), - ); - ps("ETA_FORM_BASE", form_base.to_string()); - ps("N_ETA_POWS", (form_base + 3).to_string()); - let committed: Vec = lean_vm::tables::tables() - .iter() - .map(|t| t.n_committed_columns()) - .collect(); - ps("N_TABLE_COLS", literals(&committed)); - ps("TABLE_COLS_CAP", (committed.iter().max().unwrap() + 1).to_string()); - let fixed_challenges: Vec = flock::zerocheck::univariate_skip_optimized::small_challenges() - .into_iter() - .chain(flock::zerocheck::univariate_skip_optimized::medium_challenges()) - .collect(); - ps("FIXED_CHALLENGES", flds(&fixed_challenges)); - // Flock univariate skip: 6 skipped variables, then the fixed inner rounds. - ps("K_SKIP", "6".to_string()); - ps("N_FIXED_CHALLENGE_ROUNDS", fixed_challenges.len().to_string()); - ps("PHI8_NODES", flds(&primitives::field::PHI_8_TABLE_192[..128])); - // Tower F192 = F64[Y]/(Y^3+Y+1), Y = new(0,1,0). Y_TOWER embeds Y for - // AIR lane reassembly; Y_INV helps derive the top PI-memory limb. - let y_tower = F192::new(0, 1, 0); - ps("Y_TOWER", dsl_u128(y_tower).to_string()); - ps("Y_INV", f192_literal(y_tower.inv())); - // Coordinate basis e_i of F192 over F2 (spans the whole field): the 64 - // binary basis vectors in each of the three tower limbs. The guest uses - // these vectors to reconstruct a word from its 192 coordinate bits. - let coord_basis: Vec = (0..192) - .map(|i| match i / 64 { - 0 => F192::new(1u64 << i, 0, 0), - 1 => F192::new(0, 1u64 << (i - 64), 0), - 2 => F192::new(0, 0, 1u64 << (i - 128)), - _ => unreachable!(), - }) - .collect(); - ps("COORD_BASIS", flds(&coord_basis)); - // One constant per domain, not one per node: every barycentric denominator over an aligned φ₈ - // window is the same element (`primitives::multilinear::window_denominator`). - ps( - "LAGRANGE_INV_COMBINED", - f192_literal(primitives::multilinear::window_denominator(128)), - ); - ps( - "LAGRANGE_INV_S", - f192_literal(primitives::multilinear::window_denominator(64)), - ); - ps("LINCHECK_ROUNDS", lcrounds.to_string()); - ps("PIN_COLUMN", flock::hash::Z_CONST_POS.to_string()); - ps("K_LOG", flock::hash::K_LOG.to_string()); - // The q_flock Strided-claim slot stride is K_LOG - LOG_PACKING (= 8), so the - // qflock point-claim slot must use THIS, not LOG2_FIELD_BITS. - ps("SLOT_STRIDE_LOG", lean_vm::hash_flock::SLOT_STRIDE_LOG.to_string()); - - // ---- LIG candidate tables (fixed [minm, maxm] range; open_stacked config) ---- - let oshape = |m: usize, log_inv_rate: usize| { - let shape = whir_shape(m, log_inv_rate); - let (vc, sh) = (&shape.config, &shape.levels); - let (cn, cr) = (sh.levels, vc.level_steps); - // Every cap root must match a transcript-bound root, including the final level. - assert_eq!(cr, cn - 1, "the yr level must be the last one"); - let (ck, cl, cyr) = (&sh.ks, &sh.log_msg_cols, sh.yr_log_n); - let cq = &vc.queries; - let (cd, cp) = (&shape.depth, &shape.per_squeeze); - let cs: Vec = (0..cn).map(|i| cq[i].div_ceil(cp[i])).collect(); - let cni: Vec = ck.iter().map(|&k| 1usize << k).collect(); - assert!( - cni.iter().enumerate().all(|(lv, &n)| { - let (bytes, whole_blocks) = if lv == 0 { - (8 * n, n % 8 == 0) - } else { - (24 * n, (3 * n) % 8 == 0) - }; - bytes <= 1024 && whole_blocks - }), - "recursive WHIR guest supports whole-block Merkle rows of at most one 1024-byte BLAKE2s chunk" - ); - let psum = |f: &dyn Fn(usize) -> usize| -> Vec { - let mut offsets = Vec::with_capacity(cn); - let mut acc = 0; - for lv in 0..cn { - offsets.push(acc); - acc += f(lv); - } - offsets - }; - let cap_depths: Vec<_> = (0..cn).map(|lv| merkle_cap_depth(cq[lv], cd[lv])).collect(); - let cap_offsets = psum(&|lv| 1 << cap_depths[lv]); - let c_qpoff = psum(&|lv| cs[lv] * cp[lv]); - let c_svkoff = psum(&|lv| cl[lv] + 1); - let c_foldbase = psum(&|lv| ck[lv]); - let c_risstart: Vec = (0..cn).map(|k| c_foldbase[k] + ck[k]).collect(); - let mut c_svk = Vec::new(); - let mut c_ivk = Vec::new(); - for &cl_lv in cl.iter().take(cn) { - for &v in &pcs::whir::eval_sk_at_vks(cl_lv) { - c_svk.push(F192::new(v.0, 0, 0)); - c_ivk.push(if v == F64::ZERO { - F192::ZERO - } else { - F192::new(v.inv().0, 0, 0) - }); - } - } - OpeningShape { - n_levels: cn, - yr_level: cr, - yr_log_len: cyr, - folds: shape.levels.ks, - log_message_columns: shape.levels.log_msg_cols, - queries: shape.config.queries, - tree_depths: shape.depth, - positions_per_squeeze: shape.per_squeeze, - squeezes: cs, - interleaving: cni, - query_grinding_bits: shape.config.grinding_bits, - cap_depths, - cap_offsets, - positions_offsets: c_qpoff, - vanish_offsets: c_svkoff, - fold_offsets: c_foldbase, - residual_fold_offsets: c_risstart, - vanish_values: c_svk, - vanish_inverses: c_ivk, - ood_samples: shape.config.ood_samples, - } - }; - let (minm, maxm) = (MU_MIN, MU_MAX); - let rates = pcs::whir::MIN_LOG_INV_RATE..=pcs::whir::MAX_LOG_INV_RATE; - let cands: Vec<_> = rates - .clone() - .flat_map(|r| (minm..=maxm).map(move |m| oshape(m, r))) - .collect(); - let maxlev = cands.iter().map(|c| c.n_levels).max().unwrap(); - let maxsvk = cands.iter().map(|c| c.vanish_values.len()).max().unwrap(); - let maxood = cands.iter().flat_map(|c| &c.ood_samples).copied().max().unwrap_or(0); - ps("LIG_MAX_LEVELS", maxlev.to_string()); - ps("LIG_MAX_VANISH_LEN", maxsvk.to_string()); - ps("LIG_MAX_OOD_SAMPLES", maxood.to_string()); - ps("LIG_MIN_LOG_SIZE", minm.to_string()); - let cks: Vec<(usize, usize)> = lean_vm::cpu::col_kappa_sources(kbc).into_iter().flatten().collect(); - ps("N_COMMITTED_COLS", cks.len().to_string()); - ps("N_COLUMN_LOGS", (MU_MAX + 1).to_string()); - ps("COL_KAPPA_SRC", literals(cks.iter().map(|&(s, _)| s))); - ps("COL_KAPPA_ADJ", literals(cks.iter().map(|&(_, a)| a))); - ps("PCS_MIN_MU", lean_vm::pcs::MIN_MU.to_string()); - ps( - "LIG_LOG_MSG_COLS_CAP", - cands - .iter() - .map(|c| *c.log_message_columns.iter().max().unwrap()) - .max() - .unwrap() - .to_string(), - ); - ps( - "YR_LOG_CAP", - cands.iter().map(|c| c.yr_log_len).max().unwrap().to_string(), - ); - { - let flat = |f: &dyn Fn(&OpeningShape) -> Vec| { - let rows: Vec = cands - .iter() - .flat_map(|c| { - let mut row = f(c); - row.resize(maxlev, 0); - row - }) - .collect(); - literals(rows) - }; - let scal = |f: &dyn Fn(&OpeningShape) -> usize| literals(cands.iter().map(f)); - ps("LIG_N_LEVELS", scal(&|c| c.n_levels)); - ps("LIG_YR_LEVEL", scal(&|c| c.yr_level)); - // The guest rotates the terminal point by the lane-fold count to index it by - // witness coordinate, and the residual segment is what it rotates the last - // lane challenges past, so the residual may never be longer than that fold - // (`RESIDUAL_MAX_LOG` < `INITIAL_FOLDING_FACTOR` keeps this true by a margin). - assert!( - cands.iter().all(|c| c.yr_log_len <= c.folds[0]), - "residual longer than the lane fold: the guest's point rotation has no room" - ); - ps("LIG_YR_LOG_LEN", scal(&|c| c.yr_log_len)); - ps("LIG_YR_LEN", scal(&|c| 1usize << c.yr_log_len)); - ps("LIG_TOTAL_FOLDS", scal(&|c| c.folds.iter().sum())); - ps("LIG_MAX_QUERIES", scal(&|c| *c.queries.iter().max().unwrap())); - ps("LIG_MAX_SQUEEZES", scal(&|c| *c.squeezes.iter().max().unwrap())); - ps("LIG_MAX_INTERLEAVE", scal(&|c| *c.interleaving.iter().max().unwrap())); - ps( - "LIG_POSITIONS_LEN", - scal(&|c| { - (0..c.n_levels) - .map(|level| c.squeezes[level] * c.positions_per_squeeze[level]) - .sum() - }), - ); - let row_cap = cands - .iter() - .flat_map(|c| { - c.interleaving - .iter() - .enumerate() - .map(|(level, &n)| n * if level == 0 { 1 } else { 3 }) - }) - .max() - .unwrap(); - ps("LIG_ROW_CAP", row_cap.to_string()); - ps("LIG_PACKED_ROW_CAP", (row_cap / 2).to_string()); - ps( - "LIG_PATH_CAP", - cands - .iter() - .flat_map(|c| { - c.tree_depths - .iter() - .zip(&c.cap_depths) - .map(|(depth, cap)| 4 * (depth - cap)) - }) - .max() - .unwrap() - .to_string(), - ); - ps("LIG_QUERY_GRIND_BITS", flat(&|c| c.query_grinding_bits.clone())); - ps("LIG_OOD_SAMPLES", flat(&|c| c.ood_samples.clone())); - ps("LIG_QUERIES", flat(&|c| c.queries.clone())); - ps("LIG_FOLDS", flat(&|c| c.folds.clone())); - ps("LIG_INTERLEAVE", flat(&|c| c.interleaving.clone())); - // 64-byte BLAKE2s blocks per leaf row: level 0's committed rows are - // base-field F64 (8 bytes/lane); deeper levels are native F192 - // (24 bytes/word, received as three embedded K limbs each). Rows are - // whole blocks only (asserted at candidate construction). - ps( - "LIG_LEAF_BLOCKS", - flat(&|c| { - c.interleaving - .iter() - .enumerate() - .map(|(level, &n)| if level == 0 { n / 8 } else { 3 * n / 8 }) - .collect() - }), - ); - ps("LIG_TREE_DEPTH", flat(&|c| c.tree_depths.clone())); - ps("LIG_CAP_DEPTH", flat(&|c| c.cap_depths.clone())); - ps("LIG_CAP_OFF", flat(&|c| c.cap_offsets.clone())); - ps("LIG_CAP_LEN", scal(&|c| c.cap_depths.iter().map(|&d| 1 << d).sum())); - ps("LIG_SQUEEZES", flat(&|c| c.squeezes.clone())); - ps("LIG_POSITIONS_OFF", flat(&|c| c.positions_offsets.clone())); - ps("LIG_LOG_MSG_COLS", flat(&|c| c.log_message_columns.clone())); - ps("LIG_RESIDUAL_FOLD_OFF", flat(&|c| c.residual_fold_offsets.clone())); - ps( - "LIG_RESIDUAL_PREFIX_LEN", - flat(&|c| { - c.log_message_columns - .iter() - .map(|&columns| columns - c.yr_log_len) - .collect() - }), - ); - ps("LIG_FOLDS_OFF", flat(&|c| c.fold_offsets.clone())); - ps("LIG_VANISH_OFF", flat(&|c| c.vanish_offsets.clone())); - let mut svk2 = Vec::with_capacity(cands.len() * maxsvk); - let mut ivk2 = Vec::with_capacity(cands.len() * maxsvk); - for candidate in &cands { - let padded_len = svk2.len() + maxsvk; - svk2.extend_from_slice(&candidate.vanish_values); - ivk2.extend_from_slice(&candidate.vanish_inverses); - svk2.resize(padded_len, F192::ZERO); - ivk2.resize(padded_len, F192::ZERO); - } - ps("LIG_VANISH_VALS", flds(&svk2)); - ps("LIG_VANISH_INVS", flds(&ivk2)); - } - let n_log_sizes = maxm - minm + 1; - let n_rates = MAX_LOG_INV_RATE - MIN_LOG_INV_RATE + 1; - ps("LIG_N_LOG_SIZES", n_log_sizes.to_string()); - ps("LIG_N_RATES", n_rates.to_string()); - ps("LIG_N_CANDIDATES", (n_log_sizes * n_rates).to_string()); - ps( - "LIG_MIN_SHIFT_INV", - dsl_u128(F192::new(g_pow(minm).inv().0, 0, 0)).to_string(), - ); - ps("CLAIM_POINT_BUF", literals(claims.iter().map(|c| c.buffer))); - ps("CLAIM_COMMITTED_COL", literals(claims.iter().map(|c| c.column))); - let slot_stride_log = lean_vm::hash_flock::SLOT_STRIDE_LOG; - let cpqbits: Vec = claims - .iter() - .flat_map(|c| (0..slot_stride_log).map(move |k| (c.qflock_slot >> k) & 1)) - .collect(); - ps("CLAIM_QFLOCK_SLOT_BITS", literals(&cpqbits)); - ps("QFLOCK_COMMITTED_COL", qflock_compact.to_string()); - ps("QFLOCK_VARS_CAP", (33 + slot_stride_log).to_string()); - ps("BYTECODE_LOG", kbc.to_string()); - // The stacked bytecode: nbcv/2 encoding columns per side, aligned with the bus - // tuple, so their slots span the fingerprint's own bits. The defer region is - // 2*kbc points + sel bits + 2 reduced + alpha + z_skip + 2*lcrounds rounds - // + 64 z_partial + 1 matpart. - let bc_cols = nbcv / 2; - let log2_bc_cols = lean_vm::leaf::N_TUPLE_BITS; - ps("BYTECODE_COLS", bc_cols.to_string()); - ps("LOG2_BYTECODE_COLS", log2_bc_cols.to_string()); - ps("DEFER_SIZE", (kbc + log2_bc_cols + 2 * lcrounds + 68).to_string()); - ps("BYTECODE_VARS", (kbc + log2_bc_cols).to_string()); - let agg_state = pack_state(FiatShamirState::from_label(RECURSION_AGG_LABEL).state()); - ps("AGG_SEED_0", dsl_u128(agg_state[0]).to_string()); - ps("AGG_SEED_1", dsl_u128(agg_state[1]).to_string()); - - // ---- LeanDA (`doc/leanvm` §sec:leanda) ---- - let (pad_cell, pad_row) = lean_da::padding_digests(); - let (pad_cell, pad_row) = (pack_hash_state(&pad_cell), pack_hash_state(&pad_row)); - ps("DA_LOG_K", DA_LOG_K.to_string()); - ps("DA_LOG_CELL", DA_LOG_CELL.to_string()); - ps("DA_MAX_ROWS", DA_MAX_ROWS.to_string()); - ps("DA_LOG_MAX_ROWS", DA_MAX_ROWS.ilog2().to_string()); - ps("DA_PAD_CELL_0", f192_literal(pad_cell[0])); - ps("DA_PAD_CELL_1", f192_literal(pad_cell[1])); - ps("DA_PAD_ROW_0", f192_literal(pad_row[0])); - ps("DA_PAD_ROW_1", f192_literal(pad_row[1])); - let defer_cells = kbc + log2_bc_cols + 1 + 2 * flock::hash::K_LOG + 2; - ps("STMT_HEADER", STATEMENT_HEADER.to_string()); - let (off, pairs) = (STATEMENT_HEADER, defer_cells.div_ceil(2)); - let blocks = (off + 3 * pairs).div_ceil(4); - ps("STMT_ODD", (defer_cells % 2).to_string()); - ps("STMT_PAIRS", pairs.to_string()); - ps("STMT_PAD_CELLS", (4 * blocks - off - 3 * pairs).to_string()); - ps("STMT_BLOCKS", blocks.to_string()); - // A list is at most MAX_KEYS blocks (one a claim is the widest it gets), so it - // holds fewer than that many windows; a declared count is below MAX_KEYS, hence - // decomposes into that many bits. The first two bound a range check, which takes - // a COUNT, the third a bit decomposition. - ps("SIGNERS_WINDOW", SIGNERS_WINDOW.to_string()); - ps("SIGNERS_WINDOW_LOG", SIGNERS_WINDOW.ilog2().to_string()); - ps("SIGNERS_MAX_WINDOWS", SIGNERS_MAX_WINDOWS.to_string()); - ps("SIGNERS_COUNT_BITS", SIGNERS_COUNT_BITS.to_string()); - ps("BLAKE2S_IV_0", dsl_u128(lean_vm::hash_flock::IV_CELLS[0]).to_string()); - ps("BLAKE2S_IV_1", dsl_u128(lean_vm::hash_flock::IV_CELLS[1]).to_string()); - ps( - "MD_FINAL", - dsl_u128(lean_vm::hash_flock::metadata(0, lean_vm::hash_flock::FINAL_FLAG, 0)).to_string(), - ); - - // The XMSS instance, from which the guest derives every table width by - // compile-time integer arithmetic. - ps("V", xmss::V.to_string()); - ps("W", xmss::W.to_string()); - ps("TARGET_SUM", xmss::TARGET_SUM.to_string()); - ps("LOG_LIFETIME", xmss::LOG_LIFETIME.to_string()); - // Every XMSS tweak the guest builds is one of these constants plus the - // epoch's weighed bits, so the byte layout lives in `xmss::make_tweak` and - // nowhere else. The chain table is indexed `CHAIN_STEPS * i + s` and the - // Merkle one by level, exactly as `verify_sig` walks them. - ps( - "XM_ENC_TWEAK", - dsl_u128(tweak_cell(xmss::TWEAK_TYPE_ENCODING, 0)).to_string(), - ); - ps( - "XM_PK_TWEAK", - dsl_u128(tweak_cell(xmss::TWEAK_TYPE_WOTS_PK, 0)).to_string(), - ); - let chain_tweaks: Vec = (0..xmss::V) - .flat_map(|i| { - (0..xmss::CHAIN_LENGTH - 1) - .map(move |s| tweak_cell(xmss::TWEAK_TYPE_CHAIN, (i * xmss::CHAIN_LENGTH + s) as u32)) - }) - .collect(); - ps("XM_CHAIN_TWEAKS", flds(&chain_tweaks)); - let merkle_tweaks: Vec = (0..xmss::LOG_LIFETIME) - .map(|level| tweak_cell(xmss::TWEAK_TYPE_MERKLE, (level + 1) as u32)) - .collect(); - ps("XM_MERKLE_TWEAKS", flds(&merkle_tweaks)); - let index_weights: Vec = (0..xmss::LOG_LIFETIME).map(tweak_index_weight).collect(); - ps("XM_INDEX_WEIGHT", flds(&index_weights)); - ps("MAX_KEYS", MAX_KEYS.to_string()); - ps("MAX_DA_ROOTS", MAX_DA_ROOTS.to_string()); - ps("DA_ROOT_COUNTS", (MAX_DA_ROOTS + 1).to_string()); - ps("MAX_RECURSIONS", MAX_RECURSIONS.to_string()); - ps("MAX_EPOCHS", MAX_EPOCHS.to_string()); - - // The SPHINCS instance. Its tweaks are derived per signature from the index - // the message digest picks, where XMSS's come from one public epoch, so the - // guest receives the shape and the native tweak prefixes. - let dsl_list = |values: &[usize]| { - let inner: Vec = values.iter().map(usize::to_string).collect(); - format!("[{}]", inner.join(", ")) - }; - ps("SP_V", sphincs::V.to_string()); - ps("SP_W", sphincs::W.to_string()); - ps("SP_TARGET_SUM", sphincs::TARGET_SUM.to_string()); - ps("SP_D", sphincs::D.to_string()); - ps("SP_A", sphincs::A.to_string()); - ps("SP_K", sphincs::K.to_string()); - ps("SP_H", sphincs::H.to_string()); - ps("SP_HEIGHTS", dsl_list(&sphincs::HEIGHTS)); - ps("SP_SUFFIX", dsl_list(&sphincs::SUFFIX)); - for (name, tag) in [ - ("SP_TW_CHAIN", sphincs::TWEAK_CHAIN), - ("SP_TW_LEAF", sphincs::TWEAK_LEAF), - ("SP_TW_NODE", sphincs::TWEAK_NODE), - ("SP_TW_ENC", sphincs::TWEAK_ENC), - ("SP_TW_FTS_LEAF", sphincs::TWEAK_FTS_LEAF), - ("SP_TW_FTS_NODE", sphincs::TWEAK_FTS_NODE), - ("SP_TW_FTS_ROOTS", sphincs::TWEAK_FTS_ROOTS), - ("SP_TW_MSG", sphincs::TWEAK_MSG), - ] { - ps( - name, - dsl_u128(pack_16_bytes(&sphincs::tweak(tag, 0, 0, 0, 0))).to_string(), - ); - } - rep -} - -/// Build the guest bytecode and the stacked table its claims are about. Both are -/// cached, so this only moves the cost out of the first prove or verify. -pub fn warm_up() { - unified_guest(); - stacked_bytecode(); -} - -/// The aggregation bytecode, compiled to a fixed point on its own size. -/// -/// The recursion placeholders are a function of the inner bytecode's log size, -/// and here the inner bytecode is this one, so the size has to agree with -/// itself. Its *digest* needs no such loop: it rides the statement rather than -/// the code. The guess converges in one or two rounds because the map's only -/// size-dependent part is a handful of unrolled sumcheck rounds. -pub fn unified_guest() -> &'static Program { - static GUEST: std::sync::OnceLock = std::sync::OnceLock::new(); - GUEST.get_or_init(|| { - let mut guess = 20; - for _ in 0..8 { - let guest = compile_guest(guess); - let actual = guest.prog.len().trailing_zeros() as usize; - if actual == guess { - return guest; - } - guess = actual; - } - panic!("the aggregation bytecode's self-referential compile did not converge"); - }) -} - -fn compile_guest(kbc: usize) -> Program { - let replacements = placeholder_map(kbc); - // `DBG_PLACEHOLDERS=path`: dump the baked guest constants, to read alongside - // a `DBG_PROF_DUMP` profile (the guest's shape is entirely in these). - if let Ok(path) = std::env::var("DBG_PLACEHOLDERS") { - let dump: String = replacements.iter().map(|(k, v)| format!("{k} = {v}\n")).collect(); - std::fs::write(&path, dump).expect("write DBG_PLACEHOLDERS"); - } - let guest = compile( - &parse_with_replacements(include_str!("../guests/lean_ethereum.py"), &replacements) - .expect("the repository aggregation guest must parse"), - ); - // `DBG_DISASM=path`: dump the guest's disassembly, to read alongside a - // `DBG_PROF_DUMP` per-pc profile. Function boundaries lead the dump, so the - // pc a failed guest check reports can be resolved to a source function - // without re-deriving the layout by hand. - if let Ok(path) = std::env::var("DBG_DISASM") { - let mut ranges: Vec<_> = guest.fn_ranges.iter().collect(); - ranges.sort_by_key(|(_, entry, _)| *entry); - let mut dump = String::new(); - for (name, entry, len) in ranges { - dump += &format!("# fn {entry:>7}..{:<7} {name}\n", entry + len); - } - dump += &lean_compiler::disassemble(&guest.prog); - std::fs::write(&path, dump).expect("write DBG_DISASM"); - } - guest -} - -#[cfg(test)] -mod tests { - use super::*; - use rand::SeedableRng; - use rand::rngs::StdRng; - - use crate::signers_cache::{ - KEY_START, XMSS_EPOCH_A, get_signers, get_signers_at, get_sphincs_signers, message, message_for, - }; - - /// A second epoch inside the cached keys' window; signer `i` holds the same key at both. - const XMSS_EPOCH_B: xmss::Epoch = XMSS_EPOCH_A + 2; - const SMALL_LEAF_SIZE: usize = 6; - const LOG_INV_RATE: usize = lean_vm::pcs::TEST_LOG_INV_RATE; - - /// Cached `(key, signature)` pairs as the API takes them, every one at - /// `epoch` over the cache's message for it. - fn at_epoch( - signers: &[(XmssPublicKey, XmssSignature)], - epoch: xmss::Epoch, - ) -> Vec<(XmssPublicKey, xmss::Epoch, xmss::Message, XmssSignature)> { - signers - .iter() - .map(|(pk, sig)| (pk.clone(), epoch, message_for(epoch), sig.clone())) - .collect() - } - - fn xmss_claims(sig: &EthereumProof) -> usize { - sig.xmss_signers.iter().map(|group| group.keys.len()).sum() - } - - /// Distinct keys, strictly increasing, without generating any. - fn signer_set(len: usize) -> Vec { - (0..len) - .map(|i| XmssPublicKey { - merkle_root: (i as u128).to_be_bytes(), - public_param: [0; xmss::PUBLIC_PARAM_LEN], - }) - .collect() - } - - #[test] - fn signature_claims_keep_the_wire_layout() { - let claims = SignatureClaims { - xmss: vec![XmssClaimGroup { - epoch: XMSS_EPOCH_A, - message: message(), - keys: signer_set(2), - }], - sphincs: vec![( - SphincsPublicKey::from_bytes(&[0xa5; sphincs::PUB_KEY_SIZE]), - [0x3c; sphincs::MESSAGE_LEN], - )], - }; - let groups: Vec<_> = claims - .xmss - .iter() - .map(|group| (group.epoch, &group.message, &group.keys)) - .collect(); - let bytes = wire().serialize(&(groups, &claims.sphincs)).unwrap(); - assert_eq!(wire().serialize(&claims).unwrap(), bytes); - assert_eq!(wire().deserialize::(&bytes).unwrap(), claims); - } - - /// The guest compiles to one program, always. - /// - /// `unified_guest` finds a fixed point by compiling repeatedly and comparing - /// the result's log size, so a compiler that read a hash seed would not just - /// produce two incompatible transcripts, it could fail to converge at all. - /// The small programs in `lean_compiler`'s `determinism` suite pin the - /// digests; this pins the one program large enough to hit every path that - /// walks a map. A fixed `kbc` is enough: reproducibility does not depend on - /// the size being the fixed point. - #[test] - fn guest_compiles_reproducibly() { - let (one, two) = (compile_guest(20), compile_guest(20)); - assert_eq!( - format!("{:?}", one.prog), - format!("{:?}", two.prog), - "two compilations of the guest produced different bytecode, \ - so the compiler is reading a hash seed" - ); - } - - /// `MAX_KEYS` is exclusive at both host checks: one key short of it passes, - /// the cap itself is the documented error. The cap counts both schemes, so - /// one XMSS key short of it plus one SPHINCS claim is already over. No proof - /// involved, and the epoch cap has its own error alongside. - #[test] - fn max_keys_bound_is_exclusive() { - let full = signer_set(MAX_KEYS); - let group = |keys: &[XmssPublicKey]| { - vec![XmssClaimGroup { - epoch: XMSS_EPOCH_A, - message: message(), - keys: keys.to_vec(), - }] - }; - let claims = |keys: &[XmssPublicKey]| -> Vec<(XmssPublicKey, xmss::Epoch, xmss::Message)> { - keys.iter().map(|pk| (pk.clone(), XMSS_EPOCH_A, message())).collect() - }; - let claim = [( - SphincsPublicKey::from_bytes(&[0; sphincs::PUB_KEY_SIZE]), - [0; sphincs::MESSAGE_LEN], - )]; - check_signer_set(&group(&full[..MAX_KEYS - 1]), &[]).expect("one short of the cap"); - assert_eq!( - check_signer_set(&group(&full), &[]), - Err(AggregateVerifyError::MalformedSignerSet) - ); - assert_eq!( - check_signer_set(&group(&full[..MAX_KEYS - 1]), &claim), - Err(AggregateVerifyError::MalformedSignerSet) - ); - // One group per epoch: MAX_EPOCHS groups pass, one more is malformed. - let spread = |n: usize| -> Vec { - (0..n) - .map(|e| XmssClaimGroup { - epoch: e as u32, - message: message(), - keys: vec![full[e].clone()], - }) - .collect() - }; - check_signer_set(&spread(MAX_EPOCHS), &[]).expect("at the epoch cap"); - assert_eq!( - check_signer_set(&spread(MAX_EPOCHS + 1), &[]), - Err(AggregateVerifyError::MalformedSignerSet) - ); - plan_coverage(&claims(&full[..MAX_KEYS - 1]), &[], &[], None).expect("one short of the cap"); - assert_eq!( - plan_coverage(&claims(&full), &[], &[], None).err(), - Some(AggregationError::TooLarge) - ); - assert_eq!( - plan_coverage(&claims(&full[..MAX_KEYS - 1]), &claim, &[], None).err(), - Some(AggregationError::TooLarge) - ); - let spread_claims = |n: usize| -> Vec<(XmssPublicKey, xmss::Epoch, xmss::Message)> { - (0..n).map(|e| (full[e].clone(), e as u32, message())).collect() - }; - plan_coverage(&spread_claims(MAX_EPOCHS), &[], &[], None).expect("at the epoch cap"); - assert_eq!( - plan_coverage(&spread_claims(MAX_EPOCHS + 1), &[], &[], None).err(), - Some(AggregationError::TooManyEpochs) - ); - } - - fn prove_leaf(signers: &[(XmssPublicKey, XmssSignature)]) -> EthereumProof { - aggregate(&[], at_epoch(signers, XMSS_EPOCH_A), vec![], &[], None, LOG_INV_RATE).expect("leaf aggregates") - } - - #[test] - fn keygen_and_verification_hash_domains_are_disjoint() { - let xmss_tags = [ - xmss::TWEAK_TYPE_PRF, - xmss::TWEAK_TYPE_CHAIN, - xmss::TWEAK_TYPE_WOTS_PK, - xmss::TWEAK_TYPE_MERKLE, - xmss::TWEAK_TYPE_ENCODING, - xmss::TWEAK_TYPE_PARAMETER, - xmss::TWEAK_TYPE_FILLER, - ]; - let sphincs_tags = [ - sphincs::TWEAK_PRF, - sphincs::TWEAK_CHAIN, - sphincs::TWEAK_LEAF, - sphincs::TWEAK_NODE, - sphincs::TWEAK_ENC, - sphincs::TWEAK_FTS_PRF, - sphincs::TWEAK_FTS_LEAF, - sphincs::TWEAK_FTS_NODE, - sphincs::TWEAK_FTS_ROOTS, - sphincs::TWEAK_MSG, - sphincs::TWEAK_PARAMETER, - ]; - let domains: BTreeSet<_> = xmss_tags - .into_iter() - .map(|tag| xmss::make_tweak(tag, 0, 0)) - .chain(sphincs_tags.into_iter().map(|tag| sphincs::tweak(tag, 0, 0, 0, 0))) - .collect(); - assert_eq!(domains.len(), xmss_tags.len() + sphincs_tags.len()); - } - - #[test] - fn signature_tweaks_align_with_distinct_domains() { - for (xmss_tag, sphincs_tag) in [ - (xmss::TWEAK_TYPE_CHAIN, sphincs::TWEAK_CHAIN), - (xmss::TWEAK_TYPE_WOTS_PK, sphincs::TWEAK_LEAF), - (xmss::TWEAK_TYPE_MERKLE, sphincs::TWEAK_NODE), - (xmss::TWEAK_TYPE_ENCODING, sphincs::TWEAK_ENC), - ] { - for position in [0, 1, u32::MAX] { - for index in [0, 1, 3, 0xa0b0_c0d0, u32::MAX] { - let xmss_tweak = xmss::make_tweak(xmss_tag, position, index); - let sphincs_tweak = sphincs::tweak(sphincs_tag, 0, 0, position, index); - assert_eq!(&xmss_tweak[1..], &sphincs_tweak[1..]); - assert_ne!(xmss_tweak[0], sphincs_tweak[0]); - let mut guest_tweak = tweak_cell(xmss_tag, position); - for bit in 0..32 { - if index & (1 << bit) != 0 { - guest_tweak += tweak_index_weight(bit); - } - } - assert_eq!(guest_tweak, pack_16_bytes(&xmss_tweak)); - } - } - } - } - - type RawSphincs = (SphincsPublicKey, sphincs::Message, SphincsSignature); - - fn prove_sphincs_leaf(signers: &[RawSphincs]) -> EthereumProof { - aggregate(&[], vec![], signers.to_vec(), &[], None, LOG_INV_RATE).expect("leaf aggregates") - } - - #[test] - fn aggregate_one_sphincs_signer() { - lean_vm::init_prover_pool(); - let aggregate = prove_sphincs_leaf(&get_sphincs_signers(1)); - aggregate.verify().expect("verifies"); - assert!(aggregate.xmss_signers.is_empty()); - assert_eq!(aggregate.sphincs_signers.len(), 1); - } - - #[test] - fn aggregate_one_signer() { - lean_vm::init_prover_pool(); - let aggregate = prove_leaf(&get_signers(1)); - aggregate.verify().expect("verifies"); - assert_eq!(aggregate.xmss_signers[0].epoch, XMSS_EPOCH_A); - assert_eq!(aggregate.xmss_signers[0].message, message()); - } - - /// An odd XMSS count, so its digest chain takes its odd-key-out branch and - /// the `pubkeys` stream ends in a short entry; the SPHINCS list has no parity - /// case, absorbing one entry a frame. - #[test] - fn aggregate_mixed_leaf() { - lean_vm::init_prover_pool(); - let aggregate = aggregate( - &[], - at_epoch(&get_signers(3), XMSS_EPOCH_A), - get_sphincs_signers(3), - &[], - None, - LOG_INV_RATE, - ) - .expect("leaf aggregates"); - aggregate.verify().expect("verifies"); - assert_eq!((xmss_claims(&aggregate), aggregate.sphincs_signers.len()), (3, 3)); - } - - /// A node over children of both schemes, overlapping in one signer of each: - /// the coverage table then needs a duplicate slot in both regions, and each - /// child's two key lists have to land in their own. - #[test] - fn aggregate_mixed_two_to_one() { - lean_vm::init_prover_pool(); - let xmss = get_signers(6); - let sphincs = get_sphincs_signers(4); - let leaf = |x: &[(XmssPublicKey, XmssSignature)], s: &[RawSphincs]| { - aggregate(&[], at_epoch(x, XMSS_EPOCH_A), s.to_vec(), &[], None, LOG_INV_RATE).expect("leaf aggregates") - }; - let left = leaf(&xmss[..4], &sphincs[..3]); - let right = leaf(&xmss[3..], &sphincs[2..]); - let node = aggregate(&[left, right], vec![], vec![], &[], None, LOG_INV_RATE).expect("node aggregates"); - node.verify().expect("node verifies"); - assert_eq!((xmss_claims(&node), node.sphincs_signers.len()), (6, 4)); - assert!(node.xmss_signers[0].keys.windows(2).all(|w| w[0] < w[1])); - assert!(node.sphincs_signers.windows(2).all(|w| w[0] < w[1])); - } - - /// A node whose children are each of one scheme only: every key list it - /// rebuilds is empty on one side, which is the only way the guest's - /// key-absorbing loops run over an empty range and its bound `log(x) < - /// log(g^0)` (unsatisfiable, so nothing may be written there) is reached. - #[test] - fn aggregate_one_scheme_per_child() { - lean_vm::init_prover_pool(); - let xmss_child = prove_leaf(&get_signers(3)); - let sphincs_child = prove_sphincs_leaf(&get_sphincs_signers(2)); - let node = - aggregate(&[xmss_child, sphincs_child], vec![], vec![], &[], None, LOG_INV_RATE).expect("node aggregates"); - node.verify().expect("node verifies"); - assert_eq!((xmss_claims(&node), node.sphincs_signers.len()), (3, 2)); - } - - /// The repeat the statement allows: one key signing two messages is two - /// claims, ordered by the pair, each needing its own signature. Generated - /// here rather than cached, the cache holding one message per key. - #[test] - fn aggregate_one_key_two_messages() { - lean_vm::init_prover_pool(); - let mut rng = StdRng::seed_from_u64(77); - let (secret_key, public_key) = sphincs::key_gen(&mut rng); - let raw: Vec = [3u8, 9] - .into_iter() - .map(|tag| { - let signed: sphincs::Message = std::array::from_fn(|i| tag.wrapping_mul(i as u8 + 1)); - let signature = sphincs::sign(&secret_key, &signed).expect("signs"); - (public_key, signed, signature) - }) - .collect(); - let aggregate = prove_sphincs_leaf(&raw); - aggregate.verify().expect("verifies"); - assert_eq!(aggregate.sphincs_signers.len(), 2); - let (first, second) = (aggregate.sphincs_signers[0], aggregate.sphincs_signers[1]); - assert_eq!(first.0, second.0, "the same key, twice"); - assert!(first.1 < second.1, "ordered by the message"); - } - - #[test] - fn guest_column_selectors_match_native_eq() { - lean_vm::init_prover_pool(); - let (helpers, _) = include_str!("../guests/lean_ethereum.py") - .split_once("\ndef main():") - .unwrap(); - let source = format!( - r#"{helpers} -def main(): - point = HeapBuf(MAX_STACK_LOG) - hint_witness(point[0:MAX_STACK_LOG], "point") - offset = hint_witness("offset") - kappa_g = hint_witness("kappa") - weight = match(log(kappa_g), range(0, N_COLUMN_LOGS), lambda kappa: column_selector(offset, point, kappa)) - public = GEN ** 0 - assert public[1] == weight - return -"# - ); - let guest = compile(&parse_with_replacements(&source, &placeholder_map(18)).unwrap()); - let run = |point: &[F192], offset: usize, kappa: usize, expected: F192| { - let mut hints = Hints::default(); - hints.push("point", point.to_vec()); - hints.push("offset", vec![count(offset)]); - hints.push("kappa", vec![count(kappa)]); - let mut program = guest.clone(); - hints.install(&mut program); - program.execute([expected, F192::ZERO]) - }; - let mut rng = StdRng::seed_from_u64(813); - for mu in MU_MIN..=MU_MAX { - let mut point: Vec = (0..mu) - .map(|_| { - let (c0, c1, c2) = rand::Rng::random(&mut rng); - F192::new(c0, c1, c2) - }) - .collect(); - point.resize(MU_MAX, F192::ZERO); - for kappa in 0..=mu { - let mask = ((1usize << mu) - 1) & !((1usize << kappa) - 1); - for offset in [0, mask, 0x0555_5555 & mask] { - let selector: Vec<_> = (kappa..mu) - .map(|k| F192::from(F64(((offset >> k) & 1) as u64))) - .collect(); - let expected = primitives::multilinear::eq_eval(&selector, &point[kappa..mu]); - assert!( - run(&point, offset, kappa, expected) - .unwrap() - .unconstrained_reads - .is_empty() - ); - if kappa != 0 { - assert!(run(&point, offset + 1, kappa, expected).is_err()); - } - } - } - assert!(run(&point, 1usize << MU_MAX, 0, F192::ZERO).is_err()); - } - } - - #[test] - fn guest_merkle_children_bind_every_link() { - lean_vm::init_prover_pool(); - let (helpers, _) = include_str!("../guests/lean_ethereum.py") - .split_once("\ndef main():") - .unwrap(); - let source = format!( - r#"{helpers} -def main(): - bits = StackBuf(3) - hint_witness(bits[0:3], "bits") - for k in unroll(0, 3): - bits[k] = bits[k] * bits[k] - direction = addr(bits) - leaf = StackBuf(2) - hint_witness(leaf[0:2], "leaf") - a, b = verify_merkle_path(leaf[0], leaf[1], direction, 3) - public = GEN ** 0 - assert public[1] == a - assert public[GEN] == b - return -"# - ); - let guest = compile(&parse_with_replacements(&source, &placeholder_map(18)).unwrap()); - let mut tree = vec![[0u8; 32]; 16]; - for (i, leaf) in tree[8..].iter_mut().enumerate() { - *leaf = pcs::merkle::hash_leaf(&[i as u8; 64]); - } - for i in (1..8).rev() { - tree[i] = pcs::merkle::hash_pair(&tree[2 * i], &tree[2 * i + 1]); - } - let run = |index: usize, leaf: [u8; 32], pairs: &[[[u8; 32]; 2]], root: [u8; 32]| { - let mut hints = Hints::default(); - hints.push( - "bits", - (0..3).map(|k| F192::from(F64(((index >> k) & 1) as u64))).collect(), - ); - hints.push("leaf", pack_hash_state(&leaf).to_vec()); - hints.push( - "merkle_children", - pairs.iter().flatten().flat_map(pack_hash_state).collect(), - ); - let mut program = guest.clone(); - hints.install(&mut program); - program.execute(pack_hash_state(&root)) - }; - for index in 0..8 { - let pairs: Vec<_> = (0..3) - .map(|level| { - let left = ((8 + index) >> level) & !1; - [tree[left], tree[left + 1]] - }) - .collect(); - assert!( - run(index, tree[8 + index], &pairs, tree[1]) - .unwrap() - .unconstrained_reads - .is_empty() - ); - for level in 0..3 { - for side in 0..2 { - for byte in [0, 16] { - let mut forged = pairs.clone(); - forged[level][side][byte] ^= 1; - assert!(run(index, tree[8 + index], &forged, tree[1]).is_err()); - } - } - // Rehash a forged running child all the way to a matching public root. - // Root equality alone passes; the selected child must still bind to its predecessor. - for byte in [0, 16] { - let mut forged = pairs.clone(); - forged[level][(index >> level) & 1][byte] ^= 1; - let mut root = pcs::merkle::hash_pair(&forged[level][0], &forged[level][1]); - for (height, pair) in forged.iter_mut().enumerate().skip(level + 1) { - pair[(index >> height) & 1] = root; - root = pcs::merkle::hash_pair(&pair[0], &pair[1]); - } - assert!(run(index, tree[8 + index], &forged, root).is_err()); - } - } - assert!(run(index ^ 1, tree[8 + index], &pairs, tree[1]).is_err()); - } - } - - /// The right leaf holds more keys than one absorb window of its list hash - /// (`SIGNERS_WINDOW` blocks of two keys), so both that leaf and the parent - /// rebuilding it run the window loop and then a non-empty tail, while the left - /// leaf's list is a tail alone. Under a window everywhere, the loop would never - /// execute and neither would the byte counter's base. - #[test] - fn aggregate_two_to_one() { - lean_vm::init_prover_pool(); - let big = 2 * SIGNERS_WINDOW + 6; - let signers = get_signers(SMALL_LEAF_SIZE + big); - let left = prove_leaf(&signers[..SMALL_LEAF_SIZE]); - let right = prove_leaf(&signers[SMALL_LEAF_SIZE..]); - let node = aggregate(&[left, right], vec![], vec![], &[], None, LOG_INV_RATE).expect("node aggregates"); - node.verify().expect("node verifies"); - assert_eq!(xmss_claims(&node), SMALL_LEAF_SIZE + big); - } - - fn da_rows(n_rows: usize, seed: u64) -> Vec { - let mut rng = ::seed_from_u64(seed); - (0..n_rows * (1 << DA_LOG_K)) - .map(|_| rand::Rng::random(&mut rng)) - .collect() - } - - /// A LeanDA payload, proven without signatures and published in the - /// statement. The node has to reach the native committer's root, and the - /// aggregate has to verify against a statement that carries it. - #[test] - fn aggregate_with_a_da_payload() { - lean_vm::init_prover_pool(); - let rows = da_rows(3, 97); - - let node = aggregate(&[], vec![], vec![], &rows, None, LOG_INV_RATE).expect("node aggregates"); - node.verify().expect("node verifies"); - assert_eq!(node.num_signature_claims(), 0); - - let (commitment, _) = lean_da::commit(&rows); - assert_eq!( - node.da_roots, - vec![commitment.root], - "the guest committed to something else" - ); - let received = EthereumProof::from_bytes(&node.to_bytes()).unwrap(); - assert_eq!(received.da_commitments(), &[commitment.root]); - received.verify().unwrap(); - let keys = SignatureClaims { - xmss: node.xmss_signers.clone(), - sphincs: node.sphincs_signers.clone(), - }; - let mut core: WireCore = wire().deserialize(&node.to_bytes_without_pubkeys()).unwrap(); - core.0[0][0] ^= 1; - let bad = EthereumProof::from_bytes_without_pubkeys(&wire().serialize(&core).unwrap(), keys).unwrap(); - assert!(bad.verify().is_err(), "the DA root must bind the VM proof"); - } - - #[test] - fn invalid_da_selections_are_rejected_before_building_the_proof() { - let unknown = [[0xa5; 32]]; - assert_eq!( - aggregate( - &[], - vec![], - vec![], - &[], - Some(ClaimSelection { - signatures: &SignatureClaims::default(), - da_commitments: &unknown - }), - LOG_INV_RATE - ) - .unwrap_err(), - AggregationError::BlobNotCovered - ); - let too_many: Vec<_> = (0..=MAX_DA_ROOTS).map(|i| [i as u8; 32]).collect(); - assert_eq!( - aggregate( - &[], - vec![], - vec![], - &[], - Some(ClaimSelection { - signatures: &SignatureClaims::default(), - da_commitments: &too_many - }), - LOG_INV_RATE - ) - .unwrap_err(), - AggregationError::TooLarge - ); - } - - #[test] - fn da_roots_accumulate_and_can_be_selected_or_omitted() { - lean_vm::init_prover_pool(); - let signers = get_signers(SMALL_LEAF_SIZE); - let mut children = Vec::new(); - for seed in [509, 510] { - let rows = da_rows(1, seed); - children.push(aggregate(&[], at_epoch(&signers, XMSS_EPOCH_A), vec![], &rows, None, LOG_INV_RATE).unwrap()); - } - let signatures = SignatureClaims { - xmss: children[0].xmss_signers.clone(), - sphincs: children[0].sphincs_signers.clone(), - }; - let first = children[0].da_roots[0]; - let second = children[1].da_roots[0]; - assert_ne!(first, second); - let mut both = vec![first, second]; - both.sort(); - for selected in [ - None, - Some(vec![]), - Some(vec![first]), - Some(vec![second]), - Some(vec![second, first, second]), - ] { - let node = aggregate( - &children, - vec![], - vec![], - &[], - selected.as_deref().map(|roots| ClaimSelection { - signatures: &signatures, - da_commitments: roots, - }), - LOG_INV_RATE, - ) - .unwrap(); - node.verify().unwrap(); - let mut expected = selected.unwrap_or_else(|| both.clone()); - expected.sort(); - expected.dedup(); - assert_eq!(node.da_roots, expected); - assert_eq!(node.da_commitments_digest(), da_list_digest(&expected)); - let mut tampered = node.clone(); - tampered.da_roots = vec![[0xa5; 32]]; - assert!( - tampered.verify().is_err(), - "changing the root list requires a new proof" - ); - } - let repeated = aggregate( - &[children[0].clone(), children[0].clone()], - vec![], - vec![], - &[], - None, - LOG_INV_RATE, - ) - .unwrap(); - repeated.verify().unwrap(); - assert_eq!(repeated.da_roots, vec![first]); - let same_rows = da_rows(1, 509); - let repeated_direct = aggregate( - std::slice::from_ref(&repeated), - vec![], - vec![], - &same_rows, - None, - LOG_INV_RATE, - ) - .unwrap(); - repeated_direct.verify().unwrap(); - assert_eq!(repeated_direct.da_roots, vec![first]); - - let rows = da_rows(1, 511); - let new_root = lean_da::commit(&rows).0.root; - for selected in [vec![new_root], vec![first], vec![]] { - let node = aggregate( - &children, - vec![], - vec![], - &rows, - Some(ClaimSelection { - signatures: &signatures, - da_commitments: &selected, - }), - LOG_INV_RATE, - ) - .unwrap(); - node.verify().unwrap(); - assert_eq!(node.da_roots, selected); - } - assert!(matches!( - aggregate( - &children, - vec![], - vec![], - &rows, - Some(ClaimSelection { - signatures: &signatures, - da_commitments: &[[0xa5; 32]] - }), - LOG_INV_RATE - ), - Err(AggregationError::BlobNotCovered) - )); - let direct = aggregate(&children, vec![], vec![], &rows, None, LOG_INV_RATE).unwrap(); - direct.verify().unwrap(); - let mut three = both.clone(); - three.push(new_root); - three.sort(); - assert_eq!(direct.da_roots, three); - let received = EthereumProof::from_bytes(&direct.to_bytes()).unwrap(); - received.verify().unwrap(); - let nested = aggregate(&[received, repeated], vec![], vec![], &[], None, LOG_INV_RATE).unwrap(); - nested.verify().unwrap(); - assert_eq!(nested.da_roots, three); - let narrowed = aggregate( - &[nested], - vec![], - vec![], - &[], - Some(ClaimSelection { - signatures: &signatures, - da_commitments: &[second], - }), - LOG_INV_RATE, - ) - .unwrap(); - narrowed.verify().unwrap(); - assert_eq!(narrowed.da_roots, vec![second]); - let dropped = aggregate( - &[narrowed], - vec![], - vec![], - &[], - Some(ClaimSelection { - signatures: &signatures, - da_commitments: &[], - }), - LOG_INV_RATE, - ) - .unwrap(); - dropped.verify().unwrap(); - assert!(dropped.da_roots.is_empty()); - assert_eq!(dropped.da_commitments_digest(), primitives::hash::hash(&[])); - assert!(matches!( - aggregate( - &[dropped], - vec![], - vec![], - &[], - Some(ClaimSelection { - signatures: &signatures, - da_commitments: &[second] - }), - LOG_INV_RATE - ), - Err(AggregationError::BlobNotCovered) - )); - assert!(matches!( - aggregate( - &children, - vec![], - vec![], - &[], - Some(ClaimSelection { - signatures: &signatures, - da_commitments: &[[0xa5; 32]] - }), - LOG_INV_RATE - ), - Err(AggregationError::BlobNotCovered) - )); - - // The first child's omitted root occupies slot 1; it cannot cover slot 0 - // (the second child's declared root) or write outside the DA region. - for index in [count(0), count(2), count(MAX_KEYS - 1), F192::ZERO, F192::new(0, 1, 0)] { - let outcome = aggregate_tampered( - &children, - vec![], - vec![], - None, - DaInput { - rows: &[], - roots: Some(&[second]), - }, - LOG_INV_RATE, - |h| { - h.entries("da_index")[0] = vec![index]; - }, - ); - assert!(outcome.is_err(), "accepted false DA slot {index:?}"); - } - let outcome = aggregate_tampered( - &[children[0].clone(), children[0].clone()], - vec![], - vec![], - None, - DaInput::default(), - LOG_INV_RATE, - |h| { - h.entries("da_index")[1] = h.entries("da_index")[0].clone(); - }, - ); - assert!( - outcome.is_err(), - "identical roots still require distinct coverage writes" - ); - // An extra, unclaimed slot leaves the published digest and both child - // statements intact; only the final coverage count rejects it. - for (declared, duplicates) in [(1, 2), (MAX_DA_ROOTS + 1, 1), (1, MAX_RECURSIONS * MAX_DA_ROOTS + 2)] { - let outcome = aggregate_tampered( - &children, - vec![], - vec![], - None, - DaInput { - rows: &[], - roots: Some(&[second]), - }, - LOG_INV_RATE, - |h| { - h.entries("da_meta")[0] = vec![count(declared), count(duplicates)]; - h.entries("da_roots").push(da_claim_cells(&second)); - }, - ); - assert!(outcome.is_err(), "accepted invalid DA coverage shape"); - } - for selected in [None, Some([].as_slice()), Some([second].as_slice())] { - let outcome = aggregate_tampered( - &children, - vec![], - vec![], - None, - DaInput { - rows: &[], - roots: selected, - }, - LOG_INV_RATE, - |h| { - let root = h - .entries("da_roots") - .iter_mut() - .find(|r| **r == da_claim_cells(&second)) - .unwrap(); - *root = da_claim_cells(&first); - }, - ); - assert!(outcome.is_err(), "every child's complete DA list must be authenticated"); - } - for limb in [2, 3] { - let outcome = aggregate_tampered( - &children, - vec![], - vec![], - None, - DaInput { - rows: &[], - roots: Some(&[second]), - }, - LOG_INV_RATE, - |h| { - let omitted = h - .entries("da_roots") - .iter_mut() - .find(|claim| claim[..2] == pack_hash_state(&first)) - .unwrap(); - omitted[limb] += F192::ONE; - }, - ); - assert!( - outcome.is_err(), - "a child's vector hash cannot change, even for an omitted root" - ); - } - for roots in [ - vec![second, second], - vec![both[1], both[0]], - vec![[0; 32]; MAX_DA_ROOTS], - ] { - let mut bad = direct.clone(); - bad.da_roots = roots; - assert_eq!(bad.verify(), Err(AggregateVerifyError::MalformedDaCommitments)); - assert_eq!( - EthereumProof::from_bytes(&bad.to_bytes()).unwrap_err(), - AggregateVerifyError::MalformedDaCommitments - ); - } - } - - #[test] - fn da_root_lists_merge_two_and_three() { - lean_vm::init_prover_pool(); - let mut leaves = Vec::new(); - let mut expected = Vec::new(); - for seed in 600..605 { - let rows = da_rows(1, seed); - let leaf = aggregate(&[], vec![], vec![], &rows, None, LOG_INV_RATE).unwrap(); - expected.extend_from_slice(leaf.da_commitments()); - leaves.push(leaf); - } - let left = aggregate(&leaves[..2], vec![], vec![], &[], None, LOG_INV_RATE).unwrap(); - let right = aggregate(&leaves[2..], vec![], vec![], &[], None, LOG_INV_RATE).unwrap(); - assert_eq!(left.da_commitments().len(), 2); - assert_eq!(right.da_commitments().len(), 3); - let root = aggregate(&[left, right], vec![], vec![], &[], None, LOG_INV_RATE).unwrap(); - let root = EthereumProof::from_bytes(&root.to_bytes()).unwrap(); - root.verify().unwrap(); - expected.sort(); - assert_eq!(root.num_signature_claims(), 0); - assert_eq!(root.da_commitments().len(), 5); - assert_eq!(root.da_commitments(), expected); - assert_eq!(root.da_commitments_digest(), da_list_digest(&expected)); - - let boundary: Vec<_> = (0..=MAX_DA_ROOTS).map(|i| [i as u8; 32]).collect(); - check_da_roots(&boundary[..MAX_DA_ROOTS]).unwrap(); - let mut full = root.clone(); - full.da_roots = boundary[..MAX_DA_ROOTS].to_vec(); - let mut extra = root.clone(); - extra.da_roots = boundary[MAX_DA_ROOTS..].to_vec(); - assert_eq!( - aggregate(&[full, extra], vec![], vec![], &[], None, LOG_INV_RATE).unwrap_err(), - AggregationError::TooLarge, - "reject the oversized union before verifying the modified child statements" - ); - let mut too_many = root; - too_many.da_roots = boundary; - assert_eq!(too_many.verify(), Err(AggregateVerifyError::MalformedDaCommitments)); - assert_eq!( - EthereumProof::from_bytes(&too_many.to_bytes()).unwrap_err(), - AggregateVerifyError::MalformedDaCommitments - ); - } - - #[test] - fn da_guest_hashes_root_lists() { - lean_vm::init_prover_pool(); - let (helpers, _) = include_str!("../guests/lean_ethereum.py") - .split_once("\ndef main():") - .unwrap(); - let source = format!( - r#"{helpers} -def main(): - n_g = hint_witness("n") - assert log(n_g) < MAX_DA_ROOTS + 1 - roots = HeapBuf((n_g * GEN) ** 4) - for x in mul_range(1, n_g): - root = roots * (x ** 4) - hint_witness(root[0:4], "root") - a, b = da_list_digest(roots, n_g) - public = GEN ** 0 - assert public[1] == a - assert public[GEN] == b - return -"# - ); - let guest = compile(&parse_with_replacements(&source, &placeholder_map(20)).unwrap()); - for n in 0..=MAX_DA_ROOTS + 1 { - let roots: Vec<_> = (0..n) - .map(|i| primitives::hash::hash(&(i as u64).to_le_bytes())) - .collect(); - let mut hints = Hints::default(); - hints.push("n", vec![count(n)]); - for root in &roots { - hints.push("root", da_claim_cells(root)); - } - let mut program = guest.clone(); - hints.install(&mut program); - let public = pack_hash_state(&da_list_digest(&roots)); - if n <= MAX_DA_ROOTS { - let execution = program.execute(public).unwrap(); - assert!(execution.unconstrained_reads.is_empty(), "{n} roots"); - } else { - assert!(program.execute(public).is_err()); - } - } - } - - #[test] - fn da_guest_bounds_coverage_slots() { - lean_vm::init_prover_pool(); - let (helpers, _) = include_str!("../guests/lean_ethereum.py") - .split_once("\ndef main():") - .unwrap(); - let source = format!( - r#"{helpers} -def main(): - roots = HeapBuf(12) - hint_witness(roots[0:12], "roots") - n_slots = hint_witness("n_slots") - assert log(n_slots) < 3 - cover = HeapBuf(4) - # Adjacent signature slots must stay outside the DA writer's range. - cover[1] = 1 - a, b, v0, v1 = cover_da_root(roots, cover * GEN, n_slots, GEN) - digest = StackBuf(2) - blake2s([a, b], [v0, v1], digest) - public = GEN ** 0 - assert public[1] == digest[0] - assert public[GEN] == digest[1] - return -"# - ); - let guest = compile(&parse_with_replacements(&source, &placeholder_map(20)).unwrap()); - let first = [0x13; 32]; - let second = [0x27; 32]; - let outside = [0x39; 32]; - let run = |slots: usize, index: F192, claimed: [u8; 32]| { - let mut hints = Hints::default(); - hints.push( - "roots", - [ - da_claim_cells(&first), - da_claim_cells(&second), - da_claim_cells(&outside), - ] - .concat(), - ); - hints.push("n_slots", vec![count(slots)]); - hints.push("da_index", vec![index]); - let mut program = guest.clone(); - hints.install(&mut program); - program.execute(pack_hash_state(&da_list_digest(&[claimed]))) - }; - assert!(run(2, count(0), first).unwrap().unconstrained_reads.is_empty()); - assert!(run(2, count(1), second).unwrap().unconstrained_reads.is_empty()); - // Matching public roots must not bypass the region bound, even when the - // out-of-range root is present in memory. - for (slots, index, claimed) in [ - (0, count(0), first), - (2, count(2), outside), - (2, count(1).inv(), first), - (2, F192::ZERO, first), - (2, F192::new(0, 1, 0), first), - ] { - assert!(run(slots, index, claimed).is_err()); - } - } - - #[test] - fn da_guest_checks_commitment_and_codewords() { - lean_vm::init_prover_pool(); - let source = include_str!("../guests/lean_ethereum.py"); - let (helpers, _) = source.split_once("\ndef main():").unwrap(); - let source = format!( - "{helpers}\ndef main():\n _, squares = exponent_tables()\n a, b, v0, v1 = da_verify(squares)\n digest = StackBuf(2)\n blake2s([a, b], [v0, v1], digest)\n public = GEN ** 0\n assert public[1] == digest[0]\n assert public[GEN] == digest[1]\n return\n" - ); - let guest = compile(&parse_with_replacements(&source, &placeholder_map(20)).unwrap()); - let n_rows = 3usize; - let codewords = lean_da::encode_rows(&da_rows(3, 101)); - let run = |words: &[u64], tamper: &dyn Fn(&mut Hints, &mut [F192; 2])| { - let (commitment, _) = lean_da::commit_codewords(words.to_vec()); - let mut public = pack_hash_state(&da_list_digest(&[commitment.root])); - let mut hints = Hints::default(); - hints.push( - "da_shape", - vec![count(n_rows), count(n_rows.next_power_of_two().ilog2() as usize)], - ); - for j in 0..CELLS_PER_ROW { - for i in 0..n_rows { - let start = i * CODEWORD_SYMBOLS + j * CELL_SYMBOLS; - hints.push( - "da_symbols", - words[start..start + CELL_SYMBOLS] - .iter() - .map(|&w| F192::from(F64(w))) - .collect(), - ); - } - } - for block in lean_da::membership_vector(&commitment.root) - .as_chunks::() - .0 - { - hints.push("da_weights", block.to_vec()); - } - tamper(&mut hints, &mut public); - let mut program = guest.clone(); - hints.install(&mut program); - program.execute(public) - }; - let honest = run(&codewords, &|_, _| {}).unwrap(); - assert!(honest.unconstrained_reads.is_empty()); - - // Every vector is orthogonal to zero rows: only hashing can reject these changes. - let zeros = vec![0; codewords.len()]; - for limb in [F192::ONE, F192::new(0, 1, 0), F192::new(0, 0, 1)] { - assert!( - run(&zeros, &|h, _| { - h.entries("da_weights")[0][0] += limb; - }) - .is_err(), - "every limb of L must be bound by its hash" - ); - } - assert!( - run(&zeros, &|h, _| { - for block in h.entries("da_weights") { - block.fill(F192::ZERO); - } - }) - .is_err(), - "zero weights must not bypass the external vector hash" - ); - - // Recommit the corrupted matrix and use that root as the public input. - // Hashing and statement binding now pass; only membership can reject it. - for position in [ - 0, - BLOB_SYMBOLS, - CODEWORD_SYMBOLS - 1, - CODEWORD_SYMBOLS, - codewords.len() - 1, - ] { - let mut bad = codewords.clone(); - bad[position] ^= 1; - assert!(run(&bad, &|_, _| {}).is_err()); - } - // A zero vector with its own hash passes the guest even for bad data. - // The verifier must derive the expected hash from the root, never trust this hash. - let mut bad = codewords.clone(); - bad[0] ^= 1; - let bad_root = lean_da::commit_codewords(bad.clone()).0.root; - let zero_hash = lean_da::vector_digest(&vec![F192::ZERO; CODEWORD_SYMBOLS]); - let forged_digest = primitives::hash::hash([bad_root, zero_hash].as_flattened()); - assert_ne!(forged_digest, da_list_digest(&[bad_root])); - let unchecked = run(&bad, &|h, public| { - for block in h.entries("da_weights") { - block.fill(F192::ZERO); - } - *public = pack_hash_state(&forged_digest); - }); - assert!(unchecked.unwrap().unconstrained_reads.is_empty()); - assert!( - run(&codewords, &|_, public| { - public[0] += F192::ONE; - }) - .is_err() - ); - assert!( - run(&codewords, &|h, _| { - h.entries("da_symbols")[0][0] += F192::new(0, 1, 0); - }) - .is_err() - ); - // Give the inflated tree its own matching root, so rejection must come - // from the shape check rather than a mismatched public commitment. - let mut padded_words = codewords.clone(); - padded_words.resize(8 * CODEWORD_SYMBOLS, 0); - let (padded_commitment, _) = lean_da::commit_codewords(padded_words); - assert!( - run(&codewords, &|h, public| { - h.entries("da_shape")[0][1] = count(3); - *public = pack_hash_state(&da_list_digest(&[padded_commitment.root])); - }) - .is_err(), - "three rows must not use an eight-row tree, even with a matching root" - ); - assert!( - run(&zeros, &|h, _| { - h.entries("da_shape")[0][0] = count(0); - }) - .is_err(), - "an empty payload with a matching zero-padded root must be rejected" - ); - for (rows, log_pad) in [(DA_MAX_ROWS + 1, 10), (3, 1), (3, DA_MAX_ROWS.ilog2() as usize + 1)] { - assert!( - run(&codewords, &|h, _| { - h.entries("da_shape")[0] = vec![count(rows), count(log_pad)]; - }) - .is_err(), - "accepted shape ({rows}, {log_pad})" - ); - } - } - - #[test] - fn da_row_shape_checks_power_of_two_boundaries() { - lean_vm::init_prover_pool(); - let (helpers, _) = include_str!("../guests/lean_ethereum.py") - .split_once("\ndef main():") - .unwrap(); - let source = format!( - "{helpers}\ndef main():\n _, squares = exponent_tables()\n public = GEN ** 0\n _, _ = da_row_shape(public[1], public[GEN], squares)\n return\n" - ); - let guest = compile(&parse_with_replacements(&source, &placeholder_map(20)).unwrap()); - let max_depth = DA_MAX_ROWS.ilog2() as usize; - let mut counts = vec![0, DA_MAX_ROWS + 1]; - for depth in 0..=max_depth { - let power = 1usize << depth; - counts.extend([power - 1, power, power + 1]); - } - counts.sort_unstable(); - counts.dedup(); - for rows in counts { - for depth in 0..=max_depth + 1 { - let result = guest.execute([count(rows), count(depth)]); - let valid = (1..=DA_MAX_ROWS).contains(&rows) && rows.next_power_of_two() == 1 << depth; - assert_eq!(result.is_ok(), valid, "rows={rows}, depth={depth}"); - if let Ok(execution) = result { - assert!(execution.unconstrained_reads.is_empty()); - } - } - } - for bad in [F192::ZERO, F192::new(0, 1, 0), count(1).inv()] { - for public in [[bad, count(0)], [count(1), bad]] { - assert!(guest.execute(public).is_err()); - } - } - } - - #[test] - fn ceil_log_hints_enforce_rounding_and_floor() { - lean_vm::init_prover_pool(); - let (helpers, _) = include_str!("../guests/lean_ethereum.py") - .split_once("\ndef main():") - .unwrap(); - // Replace only advice generation, leaving every guest constraint intact. - assert!(helpers.contains("g_log = hint_log2_ceil(bits_buf, nbits, floor)")); - let helpers = helpers.replace( - "g_log = hint_log2_ceil(bits_buf, nbits, floor)", - "g_log = hint_witness(\"ceil_log\")", - ); - for floor in [0usize, 3] { - let source = format!( - "{helpers}\ndef main():\n powers, squares = exponent_tables()\n bits = HeapBuf(8)\n hint_witness(bits[0:8], \"bits\")\n depth, value = verify_log2_ceil(bits, powers, squares, {floor}, 8)\n public = GEN ** 0\n assert public[1] == value\n assert public[GEN] == depth\n return\n" - ); - let guest = compile(&parse_with_replacements(&source, &placeholder_map(20)).unwrap()); - for value in [0usize, 1, 2, 3, 4, 7, 8, 9, 15, 16, 17, 127, 128, 129, 255] { - let expected = value.max(1).next_power_of_two().ilog2() as usize; - for depth in 0..=9 { - let mut program = guest.clone(); - program.set_witness( - "bits", - vec![(0..8).map(|j| F192::from(F64(((value >> j) & 1) as u64))).collect()], - ); - program.set_witness("ceil_log", vec![vec![count(depth)]]); - let result = program.execute([count(value), count(depth)]); - assert_eq!( - result.is_ok(), - depth == expected.max(floor), - "value={value}, floor={floor}, depth={depth}" - ); - if let Ok(execution) = result { - assert!(execution.unconstrained_reads.is_empty()); - } - } - } - } - } - - #[test] - fn invalid_blob_sizes_are_rejected() { - for symbols in [1, BLOB_SYMBOLS - 1, BLOB_SYMBOLS + 1, (DA_MAX_ROWS + 1) * BLOB_SYMBOLS] { - let rows = vec![0; symbols]; - assert!(matches!( - aggregate(&[], vec![], vec![], &rows, None, LOG_INV_RATE), - Err(AggregationError::InvalidBlobSize { symbols: n }) if n == symbols - )); - } - } - - /// The row count is a run-time parameter, so payloads of different heights - /// must prove against the *same* bytecode and each reach its own committer's - /// root. Powers of two and the counts between them alike: 3 pads to 4 and 5 to - /// 8, exercising two different arms of the tree dispatch and a non-empty gap. - #[test] - fn da_row_count_is_a_run_time_parameter() { - lean_vm::init_prover_pool(); - let signers = get_signers(SMALL_LEAF_SIZE); - for n_rows in [1usize, 3, 4, 5] { - let rows = da_rows(n_rows, 200 + n_rows as u64); - let node = aggregate(&[], at_epoch(&signers, XMSS_EPOCH_A), vec![], &rows, None, LOG_INV_RATE) - .expect("node aggregates"); - node.verify().expect("node verifies"); - let (commitment, _) = lean_da::commit(&rows); - assert_eq!(node.da_roots, vec![commitment.root], "{n_rows} rows"); - } - } - - /// What one blob-carrying proof costs, at the shape EIP-4844 and EIP-7594 fix. - /// Reported, not asserted: run it by name when the shape or the sweep changes. - #[test] - #[ignore] - fn da_blob_proof() { - lean_vm::init_prover_pool(); - let signers = get_signers(SMALL_LEAF_SIZE); - // One discarded proof: the first pays the flock circuit build and the - // arena's page faults, which would otherwise land entirely on the first row - // count reported. - warm_up(); - println!( - "bytecode: {} instructions, DA_MAX_ROWS = {DA_MAX_ROWS}", - unified_guest().prog.len() - ); - let _ = aggregate_with_stats( - &[], - at_epoch(&signers, XMSS_EPOCH_A), - vec![], - None, - DaInput { - rows: &da_rows(1, 7), - roots: None, - }, - LOG_INV_RATE, - ); - for n_rows in [1usize, 6, 14, 32] { - let rows = da_rows(n_rows, 300 + n_rows as u64); - let started = std::time::Instant::now(); - let (node, stats) = aggregate_with_stats( - &[], - at_epoch(&signers, XMSS_EPOCH_A), - vec![], - None, - DaInput { - rows: &rows, - roots: None, - }, - LOG_INV_RATE, - ) - .expect("node aggregates"); - let elapsed = started.elapsed(); - let payload = n_rows * (1 << DA_LOG_K) * 8; - println!( - "{n_rows:>3} blobs ({:>5} KiB): {:>8.2?} {:>6.0} KiB/s cycles 2^{:.1} mem 2^{:.1} proof {:.0} KiB", - payload / 1024, - elapsed, - payload as f64 / 1024.0 / elapsed.as_secs_f64(), - (stats.cycles as f64).log2(), - (stats.mem_used as f64).log2(), - node.to_bytes().len() as f64 / 1024.0, - ); - } - } - - /// A leaf carrying no payload publishes the digest of an empty root list. - #[test] - fn no_payload_publishes_the_empty_root_list() { - lean_vm::init_prover_pool(); - let signers = get_signers(SMALL_LEAF_SIZE); - let node = prove_leaf(&signers); - assert!(node.da_roots.is_empty()); - assert_eq!(node.da_commitments_digest(), primitives::hash::hash(&[])); - } - - /// Two epochs in one tree. Signer `i` holds the same key at both epochs, so - /// group B repeats keys of group A as distinct claims; the left leaf holds - /// one epoch, the right both, and the node maps each child group onto its - /// own region, with a duplicate slot for the key both leaves cover at A. - /// Enough epoch groups that the set's own hash runs its window loop: its string - /// is two blocks a group plus a leading one, so it takes sixteen groups to fill - /// one window of SIGNERS_WINDOW blocks. Every other test stays inside the tail, - /// where `plain_window` never executes and neither does the byte counter's base. - #[test] - fn aggregate_many_epoch_groups() { - lean_vm::init_prover_pool(); - // Two blocks a group plus a leading one, so SIGNERS_WINDOW / 2 groups make - // SIGNERS_WINDOW + 1 blocks: one whole window and the final block. The cached - // keys are activated over exactly that many epochs, and one key may claim - // once per epoch, so a single signer covers them all. - let groups = SIGNERS_WINDOW / 2; - let raw: Vec<_> = (0..groups) - .map(|i| { - let epoch = KEY_START + i as xmss::Epoch; - let (public_key, signature) = get_signers_at(1, epoch).remove(0); - (public_key, epoch, message_for(epoch), signature) - }) - .collect(); - let leaf = aggregate(&[], raw, vec![], &[], None, LOG_INV_RATE).expect("many-group leaf aggregates"); - leaf.verify().expect("it verifies"); - assert_eq!(leaf.xmss_signers.len(), groups); - assert!(leaf.xmss_signers.iter().all(|group| group.keys.len() == 1)); - } - - /// A whole `(epoch, message)` needs the table's undeclared groups; keys of a group - /// that stays need only its duplicate slots. The first narrowing does one, the - /// second both. - #[test] - fn a_node_may_publish_less_than_it_covers() { - lean_vm::init_prover_pool(); - let at_a = get_signers(3); - let at_b = get_signers_at(2, XMSS_EPOCH_B); - let mut raw = at_epoch(&at_a, XMSS_EPOCH_A); - raw.extend(at_epoch(&at_b, XMSS_EPOCH_B)); - let wide = aggregate(&[], raw, vec![], &[], None, LOG_INV_RATE).expect("the wide leaf aggregates"); - wide.verify().expect("the wide leaf verifies"); - assert_eq!(wide.xmss_signers.len(), 2); - assert_eq!(xmss_claims(&wide), 5); - - let narrowed = |wide: &EthereumProof, declare: &SignatureClaims| { - aggregate( - std::slice::from_ref(wide), - vec![], - vec![], - &[], - Some(ClaimSelection { - signatures: declare, - da_commitments: &[], - }), - LOG_INV_RATE, - ) - }; - let (group_a, group_b) = (wide.xmss_signers[0].clone(), wide.xmss_signers[1].clone()); - - // One group declared: the other's epoch and message go with it. - let narrow = narrowed( - &wide, - &SignatureClaims { - xmss: vec![group_b.clone()], - sphincs: vec![], - }, - ) - .expect("narrows to one group"); - narrow.verify().expect("the one-group narrowing verifies"); - assert_eq!(narrow.xmss_signers, vec![group_b.clone()]); - - // One key of one group: B goes whole, A keeps one of three. - let one_of_a = XmssClaimGroup { - epoch: XMSS_EPOCH_A, - message: message(), - keys: vec![group_a.keys[0].clone()], - }; - let part = narrowed( - &wide, - &SignatureClaims { - xmss: vec![one_of_a.clone()], - sphincs: vec![], - }, - ) - .expect("narrows to one key"); - part.verify().expect("the one-key narrowing verifies"); - assert_eq!(part.xmss_signers, vec![one_of_a]); - - assert_eq!( - narrowed(&wide, &SignatureClaims::default()).err(), - Some(AggregationError::Empty), - "a declaration has to publish something" - ); - // A key A holds and B does not, declared at B: the cache reuses keys. - let only_at_a = group_a - .keys - .iter() - .find(|key| !group_b.keys.contains(key)) - .expect("A holds a key B does not") - .clone(); - assert_eq!( - narrowed( - &wide, - &SignatureClaims { - xmss: vec![XmssClaimGroup { - epoch: XMSS_EPOCH_B, - message: message_for(XMSS_EPOCH_B), - keys: vec![only_at_a], - }], - sphincs: vec![], - } - ) - .err(), - Some(AggregationError::NotCovered) - ); - // A covered key, against another message. - assert_eq!( - narrowed( - &wide, - &SignatureClaims { - xmss: vec![XmssClaimGroup { - epoch: XMSS_EPOCH_A, - message: message_for(XMSS_EPOCH_B), - keys: group_a.keys.clone(), - }], - sphincs: vec![], - } - ) - .err(), - Some(AggregationError::NotCovered) - ); - - // Putting the group back is a different signer set, whatever it covered. - let mut rewidened = narrow.clone(); - rewidened.xmss_signers.insert(0, wide.xmss_signers[0].clone()); - assert!( - rewidened.verify().is_err(), - "a split may not be re-widened after the fact" - ); - } - - /// The guest walks raw signatures group by group over the table, and a - /// declaration puts the undeclared groups last, so the table stops agreeing with - /// the epoch order `raw_xmss` is sorted by. Declaring only the HIGHER epoch is - /// what separates the two: the table becomes [B, A] while the raw stream starts - /// with A, and a signature verified against the wrong group's tweaks fails. - #[test] - fn raw_signatures_follow_the_table_not_the_epochs() { - lean_vm::init_prover_pool(); - const _: () = assert!(XMSS_EPOCH_A < XMSS_EPOCH_B, "A must sort first for this to bite"); - let a = get_signers(1); - let b = get_signers_at(1, XMSS_EPOCH_B); - let mut raw = at_epoch(&a, XMSS_EPOCH_A); - raw.extend(at_epoch(&b, XMSS_EPOCH_B)); - let group_b = XmssClaimGroup { - epoch: XMSS_EPOCH_B, - message: message_for(XMSS_EPOCH_B), - keys: vec![b[0].0.clone()], - }; - let sig = aggregate( - &[], - raw, - vec![], - &[], - Some(ClaimSelection { - signatures: &SignatureClaims { - xmss: vec![group_b.clone()], - sphincs: vec![], - }, - da_commitments: &[], - }), - LOG_INV_RATE, - ) - .expect("the narrowing leaf aggregates"); - sig.verify().expect("it verifies"); - assert_eq!(sig.xmss_signers, vec![group_b]); - } - - #[test] - fn aggregate_two_epochs() { - lean_vm::init_prover_pool(); - let at_a = get_signers(4); - let at_b = get_signers_at(2, XMSS_EPOCH_B); - assert_eq!(at_a[0].0, at_b[0].0, "the cache reuses keys across epochs"); - let left = aggregate(&[], at_epoch(&at_a[..3], XMSS_EPOCH_A), vec![], &[], None, LOG_INV_RATE).expect("left"); - let mut right_raw = at_epoch(&at_a[2..], XMSS_EPOCH_A); - right_raw.extend(at_epoch(&at_b, XMSS_EPOCH_B)); - let right = aggregate(&[], right_raw, vec![], &[], None, LOG_INV_RATE).expect("right"); - right.verify().expect("the two-epoch leaf verifies"); - // A second message at epoch A is its own group, beside `left`'s. - let (sk, pk) = xmss::key_gen_from_seed([42; 32], XMSS_EPOCH_A, XMSS_EPOCH_A).expect("keygen"); - let sig = xmss::sign(&sk, &message_for(XMSS_EPOCH_B), XMSS_EPOCH_A).expect("sign"); - let crossed = aggregate( - std::slice::from_ref(&left), - vec![(pk.clone(), XMSS_EPOCH_A, message_for(XMSS_EPOCH_B), sig)], - vec![], - &[], - None, - LOG_INV_RATE, - ) - .expect("two messages at one epoch aggregate"); - crossed.verify().expect("the two-message leaf verifies"); - assert_eq!( - crossed - .xmss_signers - .iter() - .map(|group| (group.epoch, group.message, group.keys.len())) - .collect::>(), - vec![ - (XMSS_EPOCH_A, message(), 3), - (XMSS_EPOCH_A, message_for(XMSS_EPOCH_B), 1) - ] - ); - assert_eq!(crossed.xmss_signers[1].keys, vec![pk]); - let node = aggregate(&[left, right], vec![], vec![], &[], None, LOG_INV_RATE).expect("node"); - node.verify().expect("the two-epoch node verifies"); - let messages: Vec = node.xmss_signers.iter().map(|group| group.message).collect(); - assert_eq!(messages, vec![message(), message_for(XMSS_EPOCH_B)]); - let epochs: Vec = node.xmss_signers.iter().map(|group| group.epoch).collect(); - assert_eq!(epochs, vec![XMSS_EPOCH_A, XMSS_EPOCH_B]); - assert_eq!(node.xmss_signers[0].keys.len(), 4); - let mut b_keys: Vec = at_b.iter().map(|(pk, _)| pk.clone()).collect(); - b_keys.sort(); - assert_eq!( - node.xmss_signers[1].keys, b_keys, - "the same keys, at B, are their own claims" - ); - // Statement tampers: no proving, the mutated aggregate just has to fail. - let tampered = |mutate: &dyn Fn(&mut EthereumProof)| { - let mut bad = node.clone(); - mutate(&mut bad); - assert!(bad.verify().is_err(), "a tampered aggregate must not verify"); - }; - tampered(&|s| s.xmss_signers.swap(0, 1)); - tampered(&|s| s.xmss_signers[1].epoch = XMSS_EPOCH_B + 1); - tampered(&|s| { - let moved = s.xmss_signers[1].keys.pop().expect("a key to move"); - s.xmss_signers[0].keys.push(moved); - s.xmss_signers[0].keys.sort(); - s.xmss_signers[0].keys.dedup(); - }); - tampered(&|s| { - // Relabel group B's claims as group A's: B's keys are already among - // A's, so this folds the two groups into one. - let XmssClaimGroup { keys, .. } = s.xmss_signers.remove(1); - s.xmss_signers[0].keys.extend(keys); - s.xmss_signers[0].keys.sort(); - s.xmss_signers[0].keys.dedup(); - }); - } - - #[test] - fn aggregate_overlapping_signers() { - lean_vm::init_prover_pool(); - let signers = get_signers(40); - let left = prove_leaf(&signers[..25]); - let right = prove_leaf(&signers[15..]); - let node = aggregate(&[left, right], vec![], vec![], &[], None, LOG_INV_RATE).expect("node aggregates"); - node.verify().expect("node verifies"); - assert_eq!(xmss_claims(&node), 40); - assert!(node.xmss_signers[0].keys.windows(2).all(|w| w[0] < w[1])); - } - - /// Three levels, both schemes. The SPHINCS claims are rebuilt twice over, once - /// into each node and again into the root, and the two nodes share one claim, - /// so the root needs a SPHINCS duplicate slot for a claim it never saw - /// directly. The root also adds a raw signature of each scheme alongside its - /// children. - #[test] - #[ignore] - fn aggregate_three_levels() { - lean_vm::init_prover_pool(); - let signers = get_signers(4 * SMALL_LEAF_SIZE); - let claims = get_sphincs_signers(5); - let leaf = |index: usize, sphincs: &[RawSphincs]| { - aggregate( - &[], - at_epoch( - &signers[index * SMALL_LEAF_SIZE..(index + 1) * SMALL_LEAF_SIZE], - XMSS_EPOCH_A, - ), - sphincs.to_vec(), - &[], - None, - LOG_INV_RATE, - ) - .expect("leaf aggregates") - }; - let node = |children: &[EthereumProof]| { - aggregate(children, vec![], vec![], &[], None, LOG_INV_RATE).expect("node aggregates") - }; - // Claim 1 is under both nodes; claim 4 arrives raw at the root, and so - // do two XMSS signatures at a second epoch, so the root holds a group - // its children never carried. - let left = node(&[leaf(0, &claims[..2]), leaf(1, &[])]); - let right = node(&[leaf(2, &claims[1..3]), leaf(3, &[])]); - let root = aggregate( - &[left, right], - at_epoch(&get_signers_at(2, XMSS_EPOCH_B), XMSS_EPOCH_B), - claims[4..].to_vec(), - &[], - None, - LOG_INV_RATE, - ) - .expect("root aggregates"); - root.verify().expect("root verifies"); - assert_eq!(xmss_claims(&root), 4 * SMALL_LEAF_SIZE + 2); - assert_eq!(root.xmss_signers.len(), 2, "the raw epoch-B group joins the children's"); - assert_eq!(root.sphincs_signers.len(), 4, "claims 0, 1, 2 and 4, the repeat merged"); - assert!( - root.xmss_signers - .iter() - .all(|group| group.keys.windows(2).all(|w| w[0] < w[1])) - ); - assert!(root.sphincs_signers.windows(2).all(|w| w[0] < w[1])); - } - - #[test] - #[ignore] - fn aggregate_statement_binds() { - lean_vm::init_prover_pool(); - let signers = get_signers(2 * SMALL_LEAF_SIZE); - let left = prove_leaf(&signers[..SMALL_LEAF_SIZE]); - let right = prove_leaf(&signers[SMALL_LEAF_SIZE..]); - // Mixed, so both published lists are non-empty and every tampering - // below has a SPHINCS counterpart. - let node = aggregate(&[left, right], vec![], get_sphincs_signers(3), &[], None, LOG_INV_RATE).expect("node"); - node.verify().expect("the honest node verifies"); - - assert_eq!( - EthereumProof::from_bytes(&node.to_bytes()) - .expect("round trip") - .to_bytes(), - node.to_bytes(), - "the wire format round-trips, recomputed claim values included" - ); - let without = EthereumProof::from_bytes_without_pubkeys( - &node.to_bytes_without_pubkeys(), - SignatureClaims { - xmss: node.xmss_signers.clone(), - sphincs: node.sphincs_signers.clone(), - }, - ) - .expect("round trip"); - without.verify().expect("a caller-supplied signer set verifies"); - - let tampered = |mutate: &dyn Fn(&mut EthereumProof)| { - let mut bad = node.clone(); - mutate(&mut bad); - assert!(bad.verify().is_err(), "a tampered aggregate must not verify"); - }; - tampered(&|s| s.xmss_signers[0].keys[0] = s.xmss_signers[0].keys[1].clone()); - tampered(&|s| { - s.xmss_signers[0].keys.swap(0, 1); - }); - tampered(&|s| { - s.xmss_signers[0].keys.pop(); - }); - tampered(&|s| s.sphincs_signers[0] = s.sphincs_signers[1]); - tampered(&|s| { - s.sphincs_signers.swap(0, 1); - }); - tampered(&|s| { - s.sphincs_signers.pop(); - }); - // Relabelling a signer's scheme: the same 32 bytes moved to the other - // list. Every count and the splits between them are in the statement, and - // the guest holds each region's writers to that region, so this is - // not a free relabelling of what the aggregate claims. - tampered(&|s| { - let moved = s.xmss_signers[0].keys.remove(0); - let claimed = ( - SphincsPublicKey::from_bytes(&moved.flatten()), - s.xmss_signers[0].message, - ); - s.sphincs_signers.push(claimed); - s.sphincs_signers.sort(); - }); - tampered(&|s| s.xmss_signers[0].epoch += 1); - tampered(&|s| s.xmss_signers[0].message[0] ^= 1); - // A signer's own message is in the statement too, so editing it is not a - // free re-attribution of that signature to another message. - tampered(&|s| s.sphincs_signers[0].1[0] ^= 1); - tampered(&|s| s.defer.bytecode_point[0] += F192::ONE); - tampered(&|s| s.defer.matrix_point[0] += F192::ONE); - tampered(&|s| s.xmss_signers[0].keys[0] = get_signers(2 * SMALL_LEAF_SIZE + 1)[2 * SMALL_LEAF_SIZE].0.clone()); - // Splitting one group's keys across two epochs: the same claims cannot - // be re-attributed to an epoch nothing signed at. - tampered(&|s| { - let moved = s.xmss_signers[0].keys.pop().expect("a key to move"); - let epoch = s.xmss_signers[0].epoch; - let message = s.xmss_signers[0].message; - s.xmss_signers.push(XmssClaimGroup { - epoch: epoch + 1, - message, - keys: vec![moved], - }); - }); - - // A claim off the wire carries only its points. Tampering with either - // half must be caught: a point by the recomputation, a value by the - // statement the proof is checked against. - let from_wire = |mutate: &dyn Fn(&mut EthereumProof)| { - let mut bad = EthereumProof::from_bytes(&node.to_bytes()).expect("round trip"); - mutate(&mut bad); - assert!(bad.verify().is_err(), "a tampered wire aggregate must not verify"); - }; - from_wire(&|s| s.defer.bytecode_point[0] += F192::ONE); - from_wire(&|s| s.defer.matrix_point[0] += F192::ONE); - from_wire(&|s| s.defer.bytecode_value += F192::ONE); - from_wire(&|s| s.defer.matrix_a_value += F192::ONE); - from_wire(&|s| s.defer.matrix_b_value += F192::ONE); - } - - /// The all-zeros fast path in `DeferredClaim::recompute` must agree with the - /// two full passes it replaces, or every leaf would verify against the wrong - /// statement. - #[test] - fn leaf_claim_matches_the_general_path() { - let klog = flock::hash::K_LOG; - let leaf = DeferredClaim::leaf(); - let general = { - let bytecode_value = mle_eval_par(stacked_bytecode(), &leaf.bytecode_point); - let eq_r = pcs::whir::build_eq_table_ext(&leaf.matrix_point[..klog]); - let eq_c = pcs::whir::build_eq_table_ext(&leaf.matrix_point[klog..]); - let (matrix_a_value, matrix_b_value) = flock::hash::bilinear_walk_pair(&eq_r, &eq_c); - (bytecode_value, matrix_a_value, matrix_b_value) - }; - assert_eq!((leaf.bytecode_value, leaf.matrix_a_value, leaf.matrix_b_value), general); - } - - type Tamper<'a> = (&'a str, &'a dyn Fn(&mut Hints)); - - /// Corrupt each security-critical hint and require rejection. - #[test] - #[ignore] - fn aggregate_hints_bind() { - lean_vm::init_prover_pool(); - let signers = get_signers(2 * SMALL_LEAF_SIZE); - - let rejects = |children: &[EthereumProof], - raw_signatures: Vec<(XmssPublicKey, xmss::Epoch, xmss::Message, XmssSignature)>, - raw_sphincs: Vec, - description: &str, - tamper: &dyn Fn(&mut Hints)| { - let outcome = aggregate_tampered( - children, - raw_signatures, - raw_sphincs, - None, - DaInput::default(), - LOG_INV_RATE, - |hints| tamper(hints), - ) - .map(|(signature, _)| signature.verify().is_ok()); - assert!(!matches!(outcome, Ok(true)), "tampering {description} must be rejected"); - }; - - let raw_signatures = at_epoch(&signers[..SMALL_LEAF_SIZE], XMSS_EPOCH_A); - prove_leaf(&signers[..SMALL_LEAF_SIZE]); - let leaf_cases: &[Tamper] = &[ - ("raw_index (duplicate slot)", &|h: &mut Hints| { - let entries = h.entries("raw_index"); - entries[1] = entries[0].clone(); - }), - ("raw_index (out of range)", &|h: &mut Hints| { - h.entries("raw_index")[0] = vec![count(SMALL_LEAF_SIZE)]; - }), - ("group (n_xmss inflated)", &|h: &mut Hints| { - h.entries("group")[0][3] = count(SMALL_LEAF_SIZE + 1); - }), - ("group (n_raw_xmss understated)", &|h: &mut Hints| { - h.entries("group")[0][5] = count(SMALL_LEAF_SIZE - 1); - }), - ("group (a spurious duplicate slot)", &|h: &mut Hints| { - h.entries("group")[0][4] = count(1); - }), - // One more group than the hint stream carries: witness generation - // has nothing to pop for it. - ("meta (n_epochs inflated)", &|h: &mut Hints| { - h.entries("meta")[0][0] = count(2); - }), - ("pubkeys (a key nobody signed for)", &|h: &mut Hints| { - h.entries("pubkeys")[0][0] += F192::ONE; - }), - // The window split of a list hash is advice, so both halves are pinned: - // the product identity ties them to the block count, and the tail's own - // range check keeps its `match` dispatch on a real arm. - ( - "signers_split (a window count the list does not have)", - &|h: &mut Hints| { - h.entries("signers_split")[0][0] = count(1); - }, - ), - ("signers_split (a tail past a whole window)", &|h: &mut Hints| { - h.entries("signers_split")[0][1] = count(SIGNERS_WINDOW); - }), - ("fs_seed", &|h: &mut Hints| { - h.entries("fs_seed")[0][0] += F192::ONE; - }), - ("leaf_defer", &|h: &mut Hints| { - h.entries("leaf_defer")[0][0] += F192::ONE; - }), - // A leaf derives its group's tweak table from this, so a wrong - // epoch is caught by the signatures long before the statement digest. - ("group (another epoch's tweak table)", &|h: &mut Hints| { - h.entries("group")[0][0] += F192::ONE; - }), - ("group (wider than the u32 the verifier holds)", &|h: &mut Hints| { - h.entries("group")[0][0] += F192::new(0, 1, 0); - }), - // The group's signatures were made over another message, so the - // encoding digests reject long before the statement digest. - ("group (another message under the signatures)", &|h: &mut Hints| { - h.entries("group")[0][1] += F192::ONE; - }), - ]; - for (description, tamper) in leaf_cases { - rejects(&[], raw_signatures.clone(), vec![], description, *tamper); - } - - // A leaf holding two epoch groups: the second group's region is one - // slot, so its writer cannot reach the first group's keys, and the two - // groups' tweak tables cannot be swapped. - let mut two_epoch_raw = at_epoch(&signers[..2], XMSS_EPOCH_A); - two_epoch_raw.extend(at_epoch(&get_signers_at(1, XMSS_EPOCH_B), XMSS_EPOCH_B)); - aggregate(&[], two_epoch_raw.clone(), vec![], &[], None, LOG_INV_RATE) - .expect("the honest two-epoch leaf aggregates"); - let two_epoch_cases: &[Tamper] = &[ - ( - "raw_index (an XMSS signature crossing into another epoch's region)", - &|h: &mut Hints| { - h.entries("raw_index")[2] = vec![count(1)]; - }, - ), - ("group (two epochs swapped)", &|h: &mut Hints| { - let entries = h.entries("group"); - let other = entries[1][0]; - entries[1][0] = entries[0][0]; - entries[0][0] = other; - }), - ("group (a group's count moved to the other)", &|h: &mut Hints| { - h.entries("group")[0][3] = count(1); - h.entries("group")[1][3] = count(2); - }), - // The declared count is advice: keep both table groups but hash only the - // first, while the statement still publishes both. - ("meta (a published group left out of the digest)", &|h: &mut Hints| { - h.entries("meta")[0][0] = count(1); - h.entries("meta")[0][1] = count(1); - }), - ]; - for (description, tamper) in two_epoch_cases { - rejects(&[], two_epoch_raw.clone(), vec![], description, *tamper); - } - - // A mixed leaf: three XMSS signers then two SPHINCS ones, so the XMSS - // region is slots 0..3 and the SPHINCS region 3..5. Each scheme's - // witness has to bind, and neither scheme's signature may cover the - // other's declared key, which is what the statement's split claims. - let mixed_xmss = at_epoch(&signers[..3], XMSS_EPOCH_A); - let mixed_sphincs = get_sphincs_signers(2); - aggregate(&[], mixed_xmss.clone(), mixed_sphincs.clone(), &[], None, LOG_INV_RATE) - .expect("the honest mixed leaf aggregates"); - let mixed_cases: &[Tamper] = &[ - ( - "raw_index (an XMSS signature reaching the SPHINCS region)", - &|h: &mut Hints| { - h.entries("raw_index")[0] = vec![count(3)]; - }, - ), - ("sp_raw_index (out of range)", &|h: &mut Hints| { - h.entries("sp_raw_index")[0] = vec![count(2)]; - }), - ("sp_raw_index (duplicate slot)", &|h: &mut Hints| { - let entries = h.entries("sp_raw_index"); - entries[1] = entries[0].clone(); - }), - ("meta (n_sphincs inflated)", &|h: &mut Hints| { - h.entries("meta")[0][1] = count(3); - }), - ("meta (n_raw_sphincs understated)", &|h: &mut Hints| { - h.entries("meta")[0][3] = count(1); - }), - ("sphincs_signers (a message nobody signed)", &|h: &mut Hints| { - h.entries("sphincs_signers")[0][2] += F192::ONE; - }), - ("sphincs_signers (a key nobody signed for)", &|h: &mut Hints| { - h.entries("sphincs_signers")[0][0] += F192::ONE; - }), - ("sp_rand (another randomizer, so another index)", &|h: &mut Hints| { - h.entries("sp_rand")[0][0] += F192::ONE; - }), - ("sp_counter", &|h: &mut Hints| { - h.entries("sp_counter")[0][0] += F192::ONE; - }), - ("sp_digits", &|h: &mut Hints| { - let entries = h.entries("sp_digits"); - entries[0][0] *= F192::from(primitives::field::G); - }), - ("sp_chain_starts", &|h: &mut Hints| { - h.entries("sp_chain_starts")[0][0] += F192::ONE; - }), - ("sp_fts_secrets", &|h: &mut Hints| { - h.entries("sp_fts_secrets")[0][0] += F192::ONE; - }), - ("sp_fts_paths", &|h: &mut Hints| { - h.entries("sp_fts_paths")[0][0] += F192::ONE; - }), - ("sp_siblings", &|h: &mut Hints| { - h.entries("sp_siblings")[0][0] += F192::ONE; - }), - ]; - for (description, tamper) in mixed_cases { - rejects(&[], mixed_xmss.clone(), mixed_sphincs.clone(), description, *tamper); - } - - let left = prove_leaf(&signers[..SMALL_LEAF_SIZE]); - let right = prove_leaf(&signers[SMALL_LEAF_SIZE..]); - let children = vec![left, right]; - aggregate(&children, vec![], vec![], &[], None, LOG_INV_RATE).expect("the honest node aggregates"); - let node_cases: &[Tamper] = &[ - ("column placement (swapped)", &|h: &mut Hints| { - h.entries("col_sort_order")[0].swap(0, 1); - }), - ("column placement (duplicate)", &|h: &mut Hints| { - let order = &mut h.entries("col_sort_order")[0]; - order[1] = order[0]; - }), - ("column placement (out of range)", &|h: &mut Hints| { - let order = &mut h.entries("col_sort_order")[0]; - order[0] = count(order.len()); - }), - ("merkle cap (all hashes skipped)", &|h: &mut Hints| { - h.entries("merkle_cap_active")[0].fill(F192::ZERO); - }), - ("merkle cap (one ancestor skipped)", &|h: &mut Hints| { - let flags = &mut h.entries("merkle_cap_active")[0]; - let active = flags.iter_mut().skip(2).find(|x| **x == F192::ONE).unwrap(); - *active = F192::ZERO; - }), - ("merkle cap (root)", &|h: &mut Hints| { - h.entries("merkle_caps")[0][2] += F192::ONE; - }), - ("merkle cap (subtree)", &|h: &mut Hints| { - h.entries("merkle_caps")[0][4] += F192::ONE; - }), - ("merkle children (reversed)", &|h: &mut Hints| { - let children = &mut h.entries("merkle_children")[0]; - children.swap(0, 2); - children.swap(1, 3); - }), - ("merkle path (below cap)", &|h: &mut Hints| { - h.entries("merkle_children")[0][0] += F192::ONE; - }), - ("merkle leaf", &|h: &mut Hints| { - h.entries("merkle_leaf_rows")[0][0] += F192::ONE; - }), - ("merkle leaf (extension limb outside K)", &|h: &mut Hints| { - let rows = h.entries("merkle_leaf_rows"); - let row = rows - .iter_mut() - .find(|row| row.len() == 3 << pcs::whir_config::SUBSEQUENT_FOLDING_FACTOR) - .unwrap(); - row[0] += F192::new(0, 1, 0); - }), - ("merkle path (last query)", &|h: &mut Hints| { - let paths = h.entries("merkle_children"); - *paths.last_mut().unwrap().last_mut().unwrap() += F192::ONE; - }), - ("child_index (duplicate slot)", &|h: &mut Hints| { - let entries = h.entries("child_index"); - entries[1] = entries[0].clone(); - }), - ("child_index (out of range)", &|h: &mut Hints| { - h.entries("child_index")[0] = vec![count(2 * SMALL_LEAF_SIZE)]; - }), - ("child_group (count understated)", &|h: &mut Hints| { - h.entries("child_group")[0][3] = count(SMALL_LEAF_SIZE - 1); - }), - // A child group claimed at an epoch or under a message the child - // never carried: the map equality or the rebuilt digest rejects. - ("child_group (epoch)", &|h: &mut Hints| { - h.entries("child_group")[0][0] += F192::ONE; - }), - ("child_group (message)", &|h: &mut Hints| { - h.entries("child_group")[0][1] += F192::ONE; - }), - ("child_defer (a forged carried claim)", &|h: &mut Hints| { - h.entries("child_defer")[0][0] += F192::ONE; - }), - ("bc_star_hint", &|h: &mut Hints| { - h.entries("bc_star_hint")[0][0] += F192::ONE; - }), - ("mat_stars_hint", &|h: &mut Hints| { - h.entries("mat_stars_hint")[0][0] += F192::ONE; - }), - // The one hint carrying flock's whole lincheck terminal. Pinned not - // by the guest's own assert (which merely defines it) but by the - // matrix batching, whose reduced claims the root discharges against - // the real A_0/B_0. - ("matpart", &|h: &mut Hints| { - h.entries("matpart")[0][0] += F192::ONE; - }), - // A node holding no raw XMSS signature builds no tweak tables, so - // the statement digest is all that pins its epochs. The children's - // epochs must then land on slots of this altered list, and the map - // equality has no target, so the node cannot be proven. - ( - "group (a node that derives nothing from the epoch)", - &|h: &mut Hints| { - h.entries("group")[0][0] += F192::ONE; - }, - ), - ]; - for (description, tamper) in node_cases { - rejects(&children, vec![], vec![], description, *tamper); - } - - // A node over children of two different epochs: the hinted group map is - // what ties each child's group to the parent region of the same epoch. - let epoch_children = vec![ - prove_leaf(&signers[..2]), - aggregate( - &[], - at_epoch(&get_signers_at(2, XMSS_EPOCH_B), XMSS_EPOCH_B), - vec![], - &[], - None, - LOG_INV_RATE, - ) - .expect("the honest epoch-B leaf aggregates"), - ]; - aggregate(&epoch_children, vec![], vec![], &[], None, LOG_INV_RATE) - .expect("the honest two-epoch node aggregates"); - let epoch_node_cases: &[Tamper] = &[ - // Pointing the second child's group at the parent's epoch-A region: - // the epochs disagree, so the map equality fails. - ("child_group_map (a group mapped across epochs)", &|h: &mut Hints| { - h.entries("child_group_map")[1][0] = count(0); - }), - ("child_group_map (out of range)", &|h: &mut Hints| { - h.entries("child_group_map")[0][0] = count(2); - }), - ]; - for (description, tamper) in epoch_node_cases { - rejects(&epoch_children, vec![], vec![], description, *tamper); - } - - // The same discipline over a child's SPHINCS claims, which are rebuilt by - // their own loop (`hash_child_sphincs`) rather than the XMSS helper, so - // the cases above do not reach them: these children carry claims. - let sphincs = get_sphincs_signers(4); - let mixed_child = |x: &[(XmssPublicKey, XmssSignature)], s: &[RawSphincs]| { - aggregate(&[], at_epoch(x, XMSS_EPOCH_A), s.to_vec(), &[], None, LOG_INV_RATE) - .expect("the honest mixed child aggregates") - }; - let mixed_children = vec![ - mixed_child(&signers[..2], &sphincs[..2]), - mixed_child(&signers[2..4], &sphincs[2..]), - ]; - let mixed_node_cases: &[Tamper] = &[ - ("child_sphincs_index (duplicate slot)", &|h: &mut Hints| { - let entries = h.entries("child_sphincs_index"); - entries[1] = entries[0].clone(); - }), - ("child_sphincs_index (out of range)", &|h: &mut Hints| { - h.entries("child_sphincs_index")[0] = vec![count(4)]; - }), - ("child_meta (a child's SPHINCS count understated)", &|h: &mut Hints| { - h.entries("child_meta")[0][1] = count(1); - }), - ]; - for (description, tamper) in mixed_node_cases { - rejects(&mixed_children, vec![], vec![], description, *tamper); - } - } - - #[test] - #[ignore] - fn aggregate_rejects_a_bad_signature() { - lean_vm::init_prover_pool(); - let mut raw_signatures = at_epoch(&get_signers(3), XMSS_EPOCH_A); - raw_signatures[1].3.wots_signature.chain_tips[0][0] ^= 1; - assert!( - aggregate(&[], raw_signatures, vec![], &[], None, LOG_INV_RATE).is_err(), - "a forged signature must be rejected" - ); - - let mut raw_sphincs = get_sphincs_signers(2); - raw_sphincs[1].2.ots[2][0][0] ^= 1; - assert!( - aggregate(&[], vec![], raw_sphincs, &[], None, LOG_INV_RATE).is_err(), - "a forged SPHINCS signature must be rejected" - ); - } - - /// Randomness that does not decode to a target-sum encoding used to panic - /// the hint builder; it is a typed error now, and the abandoned builder - /// leaves nothing behind. - #[test] - fn malformed_raw_signature_is_an_error() { - let message = message(); - let pk = XmssPublicKey { - merkle_root: [0; xmss::DIGEST_LEN], - public_param: [0; xmss::PUBLIC_PARAM_LEN], - }; - let randomness = (0..=u8::MAX) - .find_map(|byte| { - let mut randomness = [0; xmss::RANDOMNESS_LEN]; - randomness[0] = byte; - xmss::wots_encode(&message, XMSS_EPOCH_A, &pk.public_param, &randomness) - .is_none() - .then_some(randomness) - }) - .expect("some randomness fails the target sum"); - let sig = XmssSignature { - wots_signature: xmss::WotsSignature { - chain_tips: [[0; xmss::DIGEST_LEN]; xmss::V], - randomness, - }, - merkle_proof: [[0; xmss::DIGEST_LEN]; xmss::LOG_LIFETIME], - }; - - let mut hints = Hints::default(); - assert_eq!( - push_signature_hints(&mut hints, &pk, &sig, &message, XMSS_EPOCH_A), - Err(AggregationError::MalformedRawSignature) - ); - assert!(hints.is_empty()); - assert_eq!( - aggregate( - &[], - vec![(pk, XMSS_EPOCH_A, message, sig)], - vec![], - &[], - None, - LOG_INV_RATE - ) - .err(), - Some(AggregationError::MalformedRawSignature) - ); - } - - /// The same for a SPHINCS claim, whose witness walk is equally fallible: a - /// counter that does not encode has no witness, and that is an error rather - /// than a panic inside the prover. - #[test] - fn malformed_raw_sphincs_signature_is_an_error() { - let (public_key, signed, mut signature) = get_sphincs_signers(1).pop().expect("one signer"); - signature.counters[sphincs::D - 1] ^= 1; - assert!(sphincs::verify(&public_key, &signed, &signature).is_err()); - let raw = vec![(public_key, signed, signature)]; - assert_eq!( - aggregate(&[], vec![], raw, &[], None, LOG_INV_RATE).err(), - Some(AggregationError::MalformedRawSignature) - ); - } - - /// Every `BLAKE2s` the guest itself runs reads a metadata cell an earlier - /// instruction of its own function wrote: a `SET` for a compile-time counter, - /// an `XOR` for a window's base plus its offset. An unwritten cell is - /// prover-chosen (write-once memory constrains only what something writes), so - /// a compression whose metadata nothing writes would hand the prover that - /// hash's byte counter and both flags, and every guest digest rests on those - /// being the ones the scheme specifies. The fill blocks are the deliberate - /// exception: their dummy reads a cell nothing writes, and nothing reads what - /// they compress (`lean_vm::cpu::filler`). - /// - /// This is a scan by pc, not a dominance check: a writer sitting in a branch - /// nobody took would satisfy it. What makes naming such a cell impossible is - /// `FnLower::scoped` reverting the constant pool at every join, and the - /// `blake2s_default_iv_*` tests are what guard that, by proving both paths. - #[test] - fn every_guest_blake2s_metadata_cell_is_written_first() { - use lean_vm::cpu::{DerefMode, Op}; - - // Which frame cell an instruction writes, if any. A `DEREF` in cell mode is - // bidirectional under write-once, so its local operand counts as a write. - let written = |op: &Op| match *op { - Op::Set { o, .. } => vec![o], - Op::Xor { c, .. } | Op::Mul { c, .. } => vec![c], - Op::Deref { o3, mode, .. } => { - if mode == DerefMode::Cell { - vec![o3] - } else { - vec![] - } - } - Op::Blake2s { out, .. } => vec![out, out + 1], - Op::Jump { .. } => vec![], - }; - let program = unified_guest(); - let fill: Vec> = program - .filler - .iter() - .map(|b| b.pc as usize..(b.pc + b.size) as usize) - .collect(); - let mut unwritten = Vec::new(); - for (name, entry, len) in &program.fn_ranges { - let range = *entry as usize..(*entry + *len) as usize; - for pc in range.clone() { - let Op::Blake2s { md, .. } = program.prog[pc] else { - continue; - }; - if fill.iter().any(|f| f.contains(&pc)) { - continue; - } - if !program.prog[range.start..pc].iter().any(|op| written(op).contains(&md)) { - unwritten.push(format!("{name} pc {pc} md fp[{md}]")); - } - } - } - assert!(unwritten.is_empty(), "metadata cell never written: {unwritten:?}"); - } -} diff --git a/crates/rec_aggregation/src/benchmark.rs b/crates/rec_aggregation/src/benchmark.rs deleted file mode 100644 index 315564000..000000000 --- a/crates/rec_aggregation/src/benchmark.rs +++ /dev/null @@ -1,248 +0,0 @@ -//! The two benchmarks: one leaf of the aggregation tree (`aggregate`), and an -//! n→1 recursion step over leaves of that size (`recursion`). Aggregation accepts -//! signature counts and a blob count; recursion accepts these counts per leaf. - -use primitives::bench::Plan; -use primitives::{pretty_f64, pretty_integer}; -use rand::{Rng, SeedableRng, rngs::StdRng}; -use xmss::{XmssPublicKey, XmssSignature}; - -use crate::aggregation::{DaInput, EthereumProof, aggregate, aggregate_with_stats}; -use crate::signers_cache; - -fn blobs(n: usize, seed: u64) -> Vec { - let mut rng = StdRng::seed_from_u64(seed); - (0..n * lean_da::BLOB_SYMBOLS).map(|_| rng.random()).collect() -} - -/// Cached signers `[from, to)`, as the aggregation API takes them: each raw -/// signature carries its epoch and message, the benchmarks using one pair for -/// all. -fn signers(from: usize, to: usize) -> Vec<(XmssPublicKey, xmss::Epoch, xmss::Message, XmssSignature)> { - if to == 0 { - return Vec::new(); - } - signers_cache::get_signers(to)[from..to] - .iter() - .map(|(pk, sig)| { - ( - pk.clone(), - signers_cache::XMSS_EPOCH_A, - signers_cache::message(), - sig.clone(), - ) - }) - .collect() -} - -/// Each SPHINCS signer comes with the message it signed, as the XMSS ones do. -fn sphincs_signers( - from: usize, - to: usize, -) -> Vec<(sphincs::SphincsPublicKey, sphincs::Message, sphincs::SphincsSignature)> { - if to == 0 { - return Vec::new(); - } - signers_cache::get_sphincs_signers(to)[from..to].to_vec() -} - -/// Report the shape and cost of one aggregation node. -fn report(label: &str, stats: &lean_vm::cpu::Stats, sig: &EthereumProof, prove_time: &primitives::bench::Timing) { - let base_cycles: usize = stats.base_counts.iter().sum(); - println!("{label}"); - // The program's own work, then what gets proven: the fill blocks bring each table - // to a power of two so that none needs padding rows, so the proven total is the sum - // of those powers. - println!( - " cycles (VM steps) : {} = {}", - pretty_integer(base_cycles), - crate::report::pow(base_cycles) - ); - println!(" details : {}", stats.details()); - crate::report::print_proof_size(sig.proof()); - // The whole `aggregate` call, not just `cpu::prove`: for a node that also - // covers verifying each child and batching the deferred claims, which are - // real per-node costs. `--tracing` breaks it down. - println!( - " proving time : {} s{} peak memory {} GiB", - pretty_f64(prove_time.mean()), - prove_time.spread(), - crate::report::peak_gib() - ); -} - -/// How a benchmark names a leaf of either scheme or of both. -fn describe(n_xmss: usize, n_sphincs: usize) -> String { - match (n_xmss, n_sphincs) { - (x, 0) => format!("{} XMSS", pretty_integer(x)), - (0, s) => format!("{} SPHINCS", pretty_integer(s)), - (x, s) => format!("{} XMSS and {} SPHINCS", pretty_integer(x), pretty_integer(s)), - } -} - -/// Prove signatures and `n_blobs` blobs in one leaf, then verify it. -/// -/// Proving runs one discarded warmup pass followed by `plan.repeat` measured -/// passes; see [`primitives::bench`] for why the first pass is not -/// representative and why the cooldown matters. -pub fn run_aggregation(n_xmss: usize, n_sphincs: usize, n_blobs: usize, log_inv_rate: usize, plan: Plan) { - assert!( - n_xmss > 0 || n_sphincs > 0 || n_blobs > 0, - "a leaf needs signatures or blobs" - ); - assert!(n_blobs <= lean_da::DA_MAX_ROWS, "too many blobs in one commitment"); - let trace_span = tracing::info_span!("aggregation", n_xmss, n_sphincs, n_blobs, log_inv_rate).entered(); - // Spawn the worker pool before any timed work, so no kernel pays the spawn - // cost. Opting into the arena is the calling *process's* decision (one region, - // one proof at a time), so it stays in `main`, not here. - lean_vm::init_prover_pool(); - let raw_xmss = signers(0, n_xmss); - let raw_sphincs = sphincs_signers(0, n_sphincs); - let blobs = blobs(n_blobs, 0); - // Only the final measured pass of each stage is traced: the tree describes the - // proof the reported timings are about, instead of repeating itself per pass. - let ((sig, stats), prove_time) = plan.warm_then_measure(|last| { - let _quiet = (!last).then(primitives::suppress_tracing); - aggregate_with_stats( - &[], - raw_xmss.clone(), - raw_sphincs.clone(), - None, - DaInput { - rows: &blobs, - roots: None, - }, - log_inv_rate, - ) - .expect("leaf aggregates") - }); - let (_, verify_time) = Plan::new(plan.repeat, 0).measure_quiet(|last| { - let _quiet = (!last).then(primitives::suppress_tracing); - sig.verify().expect("the leaf aggregate verifies"); - }); - drop(trace_span); - - let mut inputs = Vec::new(); - if n_xmss > 0 || n_sphincs > 0 { - inputs.push(format!("{} signatures", describe(n_xmss, n_sphincs))); - } - if n_blobs > 0 { - inputs.push(format!("{} blobs", pretty_integer(n_blobs))); - } - report( - &format!("\naggregation, {}", inputs.join(", ")), - &stats, - &sig, - &prove_time, - ); - if n_xmss > 0 || n_sphincs > 0 { - println!( - " per signature : {} signatures/s", - pretty_f64((n_xmss + n_sphincs) as f64 / prove_time.mean()) - ); - } - if n_blobs > 0 { - let payload_mib = (blobs.len() * size_of::()) as f64 / (1 << 20) as f64; - println!( - " blob throughput : {} blobs/s, {} MiB/s", - pretty_f64(n_blobs as f64 / prove_time.mean()), - pretty_f64(payload_mib / prove_time.mean()) - ); - } - println!( - " verifying : {} ms", - pretty_f64(verify_time.mean() * 1000.0) - ); -} - -/// Prove `n` leaves with signatures and distinct blob payloads, then aggregate them in one -/// recursion step and verify the result. The leaves are built once; only the -/// recursion step is measured. -pub fn run_recursion( - n: usize, - per_leaf: usize, - sphincs_per_leaf: usize, - blobs_per_leaf: usize, - log_inv_rate: usize, - enable_tracing: bool, - plan: Plan, -) { - assert!((1..=crate::MAX_RECURSIONS).contains(&n), "invalid child count"); - assert!( - per_leaf > 0 || sphincs_per_leaf > 0 || blobs_per_leaf > 0, - "a leaf needs signatures or blobs" - ); - assert!( - blobs_per_leaf <= lean_da::DA_MAX_ROWS, - "too many blobs in one commitment" - ); - assert!( - blobs_per_leaf == 0 || n <= crate::MAX_DA_ROOTS, - "too many distinct DA roots" - ); - lean_vm::init_prover_pool(); - let all = signers(0, n * per_leaf); - let all_sphincs = sphincs_signers(0, n * sphincs_per_leaf); - let started = std::time::Instant::now(); - let guest_instructions = crate::aggregation::unified_guest().code_len(); - let compile_time = started.elapsed(); - - let children: Vec = (0..n) - .map(|k| { - aggregate( - &[], - all[k * per_leaf..(k + 1) * per_leaf].to_vec(), - all_sphincs[k * sphincs_per_leaf..(k + 1) * sphincs_per_leaf].to_vec(), - &blobs(blobs_per_leaf, k as u64), - None, - log_inv_rate, - ) - .expect("leaf aggregates") - }) - .collect(); - - if enable_tracing { - primitives::init_tracing(); - } - let ((sig, stats), prove_time) = plan.warm_then_measure(|last| { - let _quiet = (!last).then(primitives::suppress_tracing); - aggregate_with_stats(&children, vec![], vec![], None, DaInput::default(), log_inv_rate) - .expect("node aggregates") - }); - let (_, verify_time) = Plan::new(plan.repeat, 0).measure_quiet(|last| { - let _quiet = (!last).then(primitives::suppress_tracing); - sig.verify().expect("the recursive aggregate verifies"); - }); - - println!( - "aggregation bytecode: {} instructions (2^{} padded), compiled in {} s", - pretty_integer(guest_instructions), - crate::aggregation::unified_guest().prog.len().trailing_zeros(), - pretty_f64(compile_time.as_secs_f64()) - ); - let mut inputs = Vec::new(); - if per_leaf > 0 || sphincs_per_leaf > 0 { - inputs.push(format!("{} signatures", describe(per_leaf, sphincs_per_leaf))); - } - if blobs_per_leaf > 0 { - inputs.push(format!("{blobs_per_leaf} blobs")); - } - report( - &format!("\nrecursion {n}\u{2192}1, over leaves of {}", inputs.join(", ")), - &stats, - &sig, - &prove_time, - ); - if blobs_per_leaf > 0 { - assert_eq!( - sig.da_commitments().len(), - n, - "every child's distinct root must survive recursion" - ); - println!(" retained DA roots : {n}"); - } - println!( - " verifying : {} ms", - pretty_f64(verify_time.mean() * 1000.0) - ); -} diff --git a/crates/rec_aggregation/src/fibonacci.rs b/crates/rec_aggregation/src/fibonacci.rs deleted file mode 100644 index 432ea85a8..000000000 --- a/crates/rec_aggregation/src/fibonacci.rs +++ /dev/null @@ -1,116 +0,0 @@ -//! Fibonacci in the exponent: the demo benchmark (`buff[g^k] = g^{F(k)}`, -//! recurrence `buff[i·g²] = buff[i·g] · buff[i]`). - -use lean_compiler::{compile, parse}; -use lean_vm::cpu::{prove, verify}; -use primitives::{ - bench::Plan, - field::{F64, F192, g_pow}, - pretty_f64, pretty_integer, -}; - -/// Prove and verify Fibonacci-in-the-exponent over a `HeapBuf` (an unrolled -/// `mul_range` recurrence), binding `g^{F(n)}` as the public input. Prints the -/// benchmark report. Proving runs one discarded warmup pass followed by -/// `plan.repeat` measured passes (see [`primitives::bench`]). -pub fn run_fibonacci(n: usize, log_inv_rate: usize, plan: Plan) { - let trace_span = tracing::info_span!("Fibonacci", n, log_inv_rate).entered(); - - let (src, pi) = fibonacci_program(n); - let program = compile(&parse(&src).unwrap()); - - // Only the final measured pass of each stage is traced (see `run_recursion`). - let ((proof, stats), prove_time) = plan.warm_then_measure(|last| { - let _quiet = (!last).then(primitives::suppress_tracing); - prove(&program, pi, log_inv_rate).expect("the Fibonacci program proves") - }); - let (_, verify_time) = Plan::new(plan.repeat, 0).measure_quiet(|last| { - let _quiet = (!last).then(primitives::suppress_tracing); - verify(&program, &pi, &proof).unwrap() - }); - - drop(trace_span); - - println!( - "Fibonacci (in the exponent, i.e. modulo 2^64 - 1), N = {}", - pretty_integer(n) - ); - println!(" cycles (VM steps) : {}", pretty_integer(stats.cycles)); - println!(" details : {}", stats.details()); - crate::report::print_proof_size(&proof); - let cycles_per_second = (stats.cycles as f64 / prove_time.mean()).round() as u64; - println!( - " proving : {} s{} {} cycles/s peak memory {} GiB", - pretty_f64(prove_time.mean()), - prove_time.spread(), - pretty_integer(cycles_per_second), - crate::report::peak_gib() - ); - println!( - " verifying : {} ms", - pretty_f64(verify_time.mean() * 1000.0) - ); -} - -/// Build the demo program: Fibonacci in the exponent over `fib_n` steps (an -/// unrolled `mul_range` loop over a `HeapBuf`), with the result `g^{F(N)}` -/// published into cell `m[0]`. Returns the zkDSL source and the public input -/// `[g^{F(N)}, 0]`. -fn fibonacci_program(fib_n: usize) -> (String, [F192; 2]) { - const UNROLL: usize = 1000; - assert!( - fib_n >= UNROLL && fib_n.is_multiple_of(UNROLL), - "fib_n must be a positive multiple of {UNROLL}" - ); - let blocks = fib_n / UNROLL; - - let (mut previous, mut current) = (F64::ONE, g_pow(1)); - for _ in 1..=fib_n { - let next = previous * current; - previous = current; - current = next; - } - let public_input = [F192::from(previous), F192::ZERO]; - - // `K` blocks: each reads its boundary pair into locals, runs `UNROLL` - // Fibonacci `MUL`s in registers, and writes the next pair (4 DEREFs per - // block). The loop counter `x = gʲ` is the block index (×g each iteration); - // block `j`'s boundary pair lives at cells `g^{2j}, g^{2j+1}`, so its base is - // `b = x·x = g^{2j}`. - let mut body = String::from(" b = x * x\n f0 = buff[b]\n f1 = buff[b * GEN]\n"); - for j in 2..=UNROLL + 1 { - body.push_str(&format!(" f{j} = f{} * f{}\n", j - 2, j - 1)); - } - body.push_str(&format!(" buff[b * GEN ** 2] = f{UNROLL}\n")); - body.push_str(&format!(" buff[b * GEN ** 3] = f{}\n", UNROLL + 1)); - - // Publish the result g^{F(N)} = buff[GEN ** {2K}] into cell m[0]: a pointer - // whose value is g^0 (`p = 1`) addresses m[0] (`p[1] = m[1·g^0] = m[g^0]`), - // and write-once forces m[0] to equal the seeded public input pi[0]. - let publish = format!(" p = 1\n p[1] = buff[GEN ** {}]\n", 2 * blocks); - - let src = format!( - "def main():\n\ - \x20 buff = HeapBuf({size})\n\ - \x20 buff[1] = 1\n\ - \x20 buff[GEN] = GEN\n\ - \x20 for x in mul_range(1, GEN ** {blocks}):\n\ - {body}\ - {publish}\ - \x20 return\n", - size = 2 * blocks + 2, - ); - (src, public_input) -} - -#[cfg(test)] -mod tests { - #[test] - fn fibonacci() { - super::run_fibonacci( - 200_000, - lean_vm::pcs::TEST_LOG_INV_RATE, - primitives::bench::Plan::default(), - ); - } -} diff --git a/crates/rec_aggregation/src/hash_chain.rs b/crates/rec_aggregation/src/hash_chain.rs deleted file mode 100644 index e3d358075..000000000 --- a/crates/rec_aggregation/src/hash_chain.rs +++ /dev/null @@ -1,133 +0,0 @@ -//! End-to-end proof of an unrolled zkDSL BLAKE2s hash chain. - -use std::time::Instant; - -use lean_compiler::{compile, compile_without_filler, parse}; -use lean_vm::cpu::{prove, verify}; -use lean_vm::vmhash::compress; -use primitives::{ - field::{F64, F192}, - pretty_f64, pretty_integer, -}; - -fn instruction_counts(source: &str, public_input: [F192; 2]) -> [usize; lean_vm::cpu::Stats::TABLES.len()] { - compile_without_filler(&parse(source).expect("parse")) - .execute(public_input) - .unwrap() - .base_counts -} - -fn chain_source(steps: usize, unroll: usize) -> String { - assert!( - unroll >= 1 && steps.is_multiple_of(unroll), - "N must be a positive multiple of UNROLL" - ); - let blocks = steps / unroll; - let result_cell = 2 * blocks; - - let mut body = String::new(); - body.push_str(" b = i * i\n"); - body.push_str(" h0 = StackBuf(2)\n"); - body.push_str(" h0[0] = buff[b]\n"); - body.push_str(" h0[1] = buff[b * GEN]\n"); - for step in 1..=unroll { - body.push_str(&format!(" h{step} = StackBuf(2)\n")); - body.push_str(&format!( - " blake2s(h{previous}, h{previous}, h{step})\n", - previous = step - 1 - )); - } - for word in 0..2 { - body.push_str(&format!(" buff[b * GEN ** {}] = h{unroll}[{word}]\n", 2 + word)); - } - - format!( - "def main():\n\ - \x20 buff = HeapBuf({size})\n\ - \x20 buff[1] = 0\n\ - \x20 buff[GEN] = 0\n\ - \x20 for i in mul_range(1, GEN ** {blocks}):\n\ - {body}\ - \x20 output = 1\n\ - \x20 output[1] = buff[GEN ** {result_cell}]\n\ - \x20 output[GEN] = buff[GEN ** {next_cell}]\n\ - \x20 return\n", - size = result_cell + 2, - next_cell = result_cell + 1, - ) -} - -#[test] -fn blake2s_hash_chain() { - let env_usize = |key: &str, default: usize| { - std::env::var(key) - .ok() - .and_then(|value| value.parse().ok()) - .unwrap_or(default) - }; - let unroll = env_usize("LEANVM_HASH_UNROLL", 4); - let steps = env_usize("LEANVM_HASH_N", 8); - assert!( - steps.is_multiple_of(unroll), - "LEANVM_HASH_N must be a multiple of LEANVM_HASH_UNROLL" - ); - - let mut digest = [F64::ZERO; 4]; - for _ in 0..steps { - digest = compress(digest, digest); - } - let public_input = [ - F192::new(digest[0].0, digest[1].0, 0), - F192::new(digest[2].0, digest[3].0, 0), - ]; - - let source = chain_source(steps, unroll); - let program = compile(&parse(&source).expect("parse")); - - let started = Instant::now(); - let (proof, stats) = prove(&program, public_input, lean_vm::pcs::TEST_LOG_INV_RATE).expect("proves"); - let prove_time = started.elapsed(); - let started = Instant::now(); - verify(&program, &public_input, &proof).expect("hash-chain proof verifies"); - let verify_time = started.elapsed(); - - assert_eq!(instruction_counts(&source, public_input)[5], steps); - - println!( - "\nBLAKE2s hash chain, N = {}, unroll = {}", - pretty_integer(steps), - pretty_integer(unroll) - ); - println!(" cycles (VM steps) : {}", pretty_integer(stats.cycles)); - for (name, &c) in ["XOR", "MUL", "SET", "DEREF", "JUMP", "BLAKE2S"] - .iter() - .zip(&stats.counts) - { - let pow = if c == 0 { - "0".to_string() - } else { - format!("2^{}", pretty_f64((c as f64).log2())) - }; - println!(" {name:<6} instructions : {pow}"); - } - println!( - " committed witness size : 2^{:.3}", - (stats.committed as f64).log2() - ); - let proof_bytes = bincode::serialized_size(&proof).expect("proof is serializable"); - println!(" proof size : {:.1} KiB", proof_bytes as f64 / 1024.0); - println!(" proving : {prove_time:?}"); - println!( - " verifying : {} ms", - pretty_f64(verify_time.as_secs_f64() * 1000.0) - ); - let hashes_per_second = (steps as f64 / prove_time.as_secs_f64()).round() as u64; - println!( - " throughput : {} hashes/s", - pretty_integer(hashes_per_second) - ); - - let mut wrong_input = public_input; - wrong_input[0] += F192::ONE; - assert!(verify(&program, &wrong_input, &proof).is_err()); -} diff --git a/crates/rec_aggregation/src/lib.rs b/crates/rec_aggregation/src/lib.rs deleted file mode 100644 index b1221e4bc..000000000 --- a/crates/rec_aggregation/src/lib.rs +++ /dev/null @@ -1,45 +0,0 @@ -pub mod aggregation; -pub mod benchmark; -pub mod fibonacci; -/// The BLAKE2s hash chain, proven end to end. A `src` module rather than its own -/// test binary so it shares the process, and so the slow flock circuit build, -/// with the other workloads. -#[cfg(test)] -mod hash_chain; -pub mod signers_cache; - -pub use aggregation::{ - AggregateVerifyError, AggregationError, ClaimSelection, DA_LOG_CELL, DA_LOG_K, DA_MAX_ROWS, EthereumProof, - MAX_DA_ROOTS, MAX_EPOCHS, MAX_KEYS, MAX_RECURSIONS, SignatureClaims, SphincsClaim, XmssClaimGroup, aggregate, - warm_up, -}; -pub use benchmark::{run_aggregation, run_recursion}; -pub use fibonacci::run_fibonacci; - -/// The pieces every workload's benchmark report ends with. -/// -/// Each caller drops its root tracing span before printing: tracing-forest -/// renders its tree only when that span closes, so the complete trace has to be -/// flushed above the report. -mod report { - use primitives::pretty_f64; - - /// A count as a power of two, or a dash when the opcode never ran. - pub fn pow(x: usize) -> String { - if x == 0 { - " -".into() - } else { - format!("2^{}", pretty_f64((x as f64).log2())) - } - } - - /// Peak resident set size, in GiB. - pub fn peak_gib() -> String { - pretty_f64(primitives::bench::peak_rss_bytes() as f64 / (1u64 << 30) as f64) - } - - pub fn print_proof_size(proof: &T) { - let bytes = bincode::serialized_size(proof).expect("proof is serializable"); - println!(" proof size : {:.1} KiB", bytes as f64 / 1024.0); - } -} diff --git a/crates/rec_aggregation/src/signers_cache.rs b/crates/rec_aggregation/src/signers_cache.rs deleted file mode 100644 index aa86bdfb8..000000000 --- a/crates/rec_aggregation/src/signers_cache.rs +++ /dev/null @@ -1,323 +0,0 @@ -//! Persistent cache for deterministic benchmark signatures, one file per -//! scheme. -//! -//! The cache grows as needed and is memoized in-process. Its filename binds the -//! parameters, hash construction, encoding predicate, and key derivation. Loaded -//! signatures are also verified, so stale entries are regenerated from the first -//! invalid one. - -use std::collections::BTreeMap; -use std::collections::hash_map::DefaultHasher; -use std::fs; -use std::hash::{Hash, Hasher}; -use std::io::Write; -use std::path::PathBuf; -use std::sync::Mutex; -use std::time::Instant; - -use primitives::{pretty_f64, pretty_integer}; -use rand::SeedableRng; -use rand::rngs::StdRng; -use xmss::*; - -type CachedSignature = (XmssPublicKey, XmssSignature); - -const SCHEMA_VERSION: u32 = 3; - -/// The epoch `get_signers` signs at. SPHINCS has none. -pub const XMSS_EPOCH_A: Epoch = 3_000_000_007; -/// First epoch the cached keys are activated at, so a test that wants many epoch -/// groups can walk the whole window (`key_gen`'s range is inclusive). -pub const KEY_START: Epoch = 3_000_000_000; -const KEY_END: Epoch = 3_000_000_015; - -pub fn message() -> Message { - std::array::from_fn(|i| (i * 5 + 1) as u8) -} - -/// The message signed at `epoch`: distinct per epoch, and exactly [`message`] -/// at [`XMSS_EPOCH_A`], so pre-existing cache files stay valid. -pub fn message_for(epoch: Epoch) -> Message { - let mut msg = message(); - for (byte, delta) in msg.iter_mut().zip((epoch ^ XMSS_EPOCH_A).to_le_bytes()) { - *byte ^= delta; - } - msg -} - -fn compute_signer(index: usize, epoch: Epoch) -> CachedSignature { - // The index over its full width: a one-byte seed repeats every 256 signers, - // and a repeated signer is invisible until something deduplicates the set, - // at which point a batch of 900 quietly becomes one of 256. - let mut seed = [10u8; 32]; - seed[..8].copy_from_slice(&(index as u64).to_le_bytes()); - let (sk, pk) = xmss::key_gen_from_seed(seed, KEY_START, KEY_END).expect("keygen"); - let sig = xmss::sign(&sk, &message_for(epoch), epoch).expect("sign"); - (pk, sig) -} - -fn hash_fingerprint() -> [Digest; 2] { - let pp = [0xA5u8; PUBLIC_PARAM_LEN]; - [ - tweak_hash(&pp, TWEAK_TYPE_CHAIN, 1, 2, &[0x5Au8; DIGEST_LEN]), - tweak_hash(&pp, TWEAK_TYPE_ENCODING, 3, 4, &[0x3Cu8; 2 * STATE_LEN]), - ] -} - -fn encoding_fingerprint(epoch: Epoch) -> (u64, [u8; V]) { - let pp = [0xA5u8; PUBLIC_PARAM_LEN]; - let msg = message_for(epoch); - for counter in 0u64.. { - let mut randomness = [0u8; RANDOMNESS_LEN]; - randomness[..8].copy_from_slice(&counter.to_le_bytes()); - if let Some(digits) = wots_encode(&msg, epoch, &pp, &randomness) { - return (counter, digits); - } - } - unreachable!("some counter randomness encodes") -} - -fn footprint(epoch: Epoch) -> u64 { - let mut hasher = DefaultHasher::new(); - SCHEMA_VERSION.hash(&mut hasher); - epoch.hash(&mut hasher); - KEY_START.hash(&mut hasher); - KEY_END.hash(&mut hasher); - message_for(epoch).hash(&mut hasher); - (V, W, CHAIN_LENGTH, LOG_LIFETIME, TARGET_SUM, RANDOMNESS_LEN).hash(&mut hasher); - hash_fingerprint().hash(&mut hasher); - encoding_fingerprint(epoch).hash(&mut hasher); - // Key derivation and grinding, which verifying a cached signature does not exercise. - compute_signer(0, epoch).hash(&mut hasher); - hasher.finish() -} - -fn cache_dir() -> PathBuf { - PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../target/signers-cache") -} - -fn cache_path(epoch: Epoch) -> PathBuf { - cache_dir().join(format!("xmss_signers_{:016x}.bin", footprint(epoch))) -} - -fn try_load_cache(epoch: Epoch) -> Option> { - let bytes = fs::read(cache_path(epoch)).ok()?; - let (version, mut signers): (u32, Vec) = bincode::deserialize(&bytes).ok()?; - if version != SCHEMA_VERSION { - return None; - } - let msg = message_for(epoch); - let valid = signers - .iter() - .take_while(|(pk, sig)| xmss::verify(pk, &msg, sig, epoch).is_ok()) - .count(); - if valid < signers.len() { - eprintln!( - "warning: signers cache {} is stale (signer {valid} of {} no longer verifies); regenerating from there", - cache_path(epoch).display(), - signers.len() - ); - signers.truncate(valid); - } - Some(signers) -} - -fn save_cache(signers: &[CachedSignature], epoch: Epoch) { - let path = cache_path(epoch); - if let Some(parent) = path.parent() { - let _ = fs::create_dir_all(parent); - } - let bytes = bincode::serialize(&(SCHEMA_VERSION, signers)).expect("serialize signers cache"); - if let Err(error) = fs::write(&path, &bytes) { - eprintln!("warning: could not write signers cache to {}: {error}", path.display()); - } -} - -fn generate_range(start: usize, end: usize, epoch: Epoch) -> Vec { - let total = end - start; - let started = Instant::now(); - let mut signers = Vec::with_capacity(total); - for (done, index) in (start..end).enumerate() { - signers.push(compute_signer(index, epoch)); - print!( - "\r generating XMSS signers (one-time, then cached): {}/{}", - pretty_integer(done + 1), - pretty_integer(total) - ); - let _ = std::io::stdout().flush(); - } - println!( - "\r generated {} XMSS in {} s (cached to disk) ", - pretty_integer(total), - pretty_f64(started.elapsed().as_secs_f64()) - ); - signers -} - -static POOLS: Mutex>> = Mutex::new(BTreeMap::new()); - -pub fn get_signers(n: usize) -> Vec { - get_signers_at(n, XMSS_EPOCH_A) -} - -/// The first `n` cached signers, signing [`message_for`]`(epoch)` at `epoch`: -/// one cache file per epoch, the keys shared across them. -pub fn get_signers_at(n: usize, epoch: Epoch) -> Vec { - let mut pools = POOLS.lock().unwrap(); - let pool = pools.entry(epoch).or_default(); - if pool.len() < n { - if let Some(disk) = try_load_cache(epoch) - && disk.len() > pool.len() - { - *pool = disk; - } - if pool.len() < n { - let mut fresh = generate_range(pool.len(), n, epoch); - pool.append(&mut fresh); - save_cache(pool, epoch); - } - } - pool[..n].to_vec() -} - -/// A SPHINCS signer, generated the same way, with the message it signed. -/// Signing is stateless, so unlike XMSS there is no epoch and no key range: -/// one key answers for every index. -type CachedSphincsSignature = (sphincs::SphincsPublicKey, sphincs::Message, sphincs::SphincsSignature); - -/// Signer `index`'s own message, distinct from every other's and from the -/// XMSS ones, so a test that mixed them up would fail rather than pass. -pub fn sphincs_message(index: usize) -> sphincs::Message { - let mut msg = [0u8; sphincs::MESSAGE_LEN]; - msg[..8].copy_from_slice(&(index as u64).to_le_bytes()); - msg[8..].copy_from_slice(&[0xC5; sphincs::MESSAGE_LEN - 8]); - msg -} - -/// One cached SPHINCS signer, as fixed-size bytes: the scheme's own -/// serializations, so nothing here has to agree with a derived one. -const SPHINCS_RECORD: usize = sphincs::PUB_KEY_SIZE + sphincs::MESSAGE_LEN + sphincs::SIG_SIZE; - -fn compute_sphincs_signer(index: usize) -> CachedSphincsSignature { - let mut rng = StdRng::seed_from_u64(0x5F1A_C500 ^ index as u64); - let (secret_key, public_key) = sphincs::key_gen(&mut rng); - let message = sphincs_message(index); - let signature = sphincs::sign(&secret_key, &message).expect("sign"); - (public_key, message, signature) -} - -fn sphincs_footprint() -> u64 { - let mut hasher = DefaultHasher::new(); - SCHEMA_VERSION.hash(&mut hasher); - // The record layout and the per-signer messages, so a change to either - // invalidates the file rather than being read back as another scheme's. - SPHINCS_RECORD.hash(&mut hasher); - sphincs_message(0).hash(&mut hasher); - sphincs_message(1).hash(&mut hasher); - ( - sphincs::MASTER_SECRET_LEN, - sphincs::V, - sphincs::W, - sphincs::TARGET_SUM, - sphincs::D, - sphincs::HEIGHTS, - sphincs::A, - sphincs::K, - ) - .hash(&mut hasher); - // The tweakable hash itself, so a change to it invalidates the file. - sphincs::th( - &[0xA5; sphincs::PUBLIC_PARAM_LEN], - &sphincs::tweak(1, 2, 3, 4, 5), - &[0x3C; 16], - ) - .hash(&mut hasher); - hasher.finish() -} - -fn sphincs_cache_path() -> PathBuf { - cache_dir().join(format!("sphincs_signers_{:016x}.bin", sphincs_footprint())) -} - -fn try_load_sphincs_cache() -> Option> { - let bytes = fs::read(sphincs_cache_path()).ok()?; - let mut signers = Vec::with_capacity(bytes.len() / SPHINCS_RECORD); - for record in bytes.as_chunks::().0 { - let (key_bytes, rest) = record.split_at(sphincs::PUB_KEY_SIZE); - let (message_bytes, signature_bytes) = rest.split_at(sphincs::MESSAGE_LEN); - let public_key = sphincs::SphincsPublicKey::from_bytes(key_bytes.try_into().unwrap()); - let message: sphincs::Message = message_bytes.try_into().unwrap(); - let signature = sphincs::SphincsSignature::from_bytes(signature_bytes.try_into().unwrap()); - if sphincs::verify(&public_key, &message, &signature).is_err() { - eprintln!( - "warning: signers cache {} is stale (signer {} no longer verifies); regenerating from there", - sphincs_cache_path().display(), - signers.len() - ); - break; - } - signers.push((public_key, message, signature)); - } - Some(signers) -} - -fn save_sphincs_cache(signers: &[CachedSphincsSignature]) { - let path = sphincs_cache_path(); - if let Some(parent) = path.parent() { - let _ = fs::create_dir_all(parent); - } - let mut bytes = Vec::with_capacity(signers.len() * SPHINCS_RECORD); - for (public_key, message, signature) in signers { - bytes.extend_from_slice(&public_key.flatten()); - bytes.extend_from_slice(message); - bytes.extend_from_slice(&signature.to_bytes()); - } - if let Err(error) = fs::write(&path, &bytes) { - eprintln!("warning: could not write signers cache to {}: {error}", path.display()); - } -} - -static SPHINCS_POOL: Mutex> = Mutex::new(Vec::new()); - -pub fn get_sphincs_signers(n: usize) -> Vec { - let mut pool = SPHINCS_POOL.lock().unwrap(); - if pool.len() < n { - if let Some(disk) = try_load_sphincs_cache() - && disk.len() > pool.len() - { - *pool = disk; - } - // Key generation is one whole 2^12-leaf tree, which is the expensive - // part; it fans out internally, so this loop stays sequential. - let started = Instant::now(); - let missing = n.saturating_sub(pool.len()); - for index in pool.len()..n { - pool.push(compute_sphincs_signer(index)); - print!( - "\r generating SPHINCS signers (one-time, then cached): {}/{}", - pretty_integer(index + 1 - (n - missing)), - pretty_integer(missing) - ); - let _ = std::io::stdout().flush(); - } - if missing > 0 { - println!( - "\r generated {} SPHINCS in {} s (cached to disk) ", - pretty_integer(missing), - pretty_f64(started.elapsed().as_secs_f64()) - ); - save_sphincs_cache(&pool); - } - } - pool[..n].to_vec() -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn signer_seed_uses_more_than_one_byte() { - assert_ne!(compute_signer(0, XMSS_EPOCH_A).0, compute_signer(256, XMSS_EPOCH_A).0); - } -} diff --git a/crates/rec_aggregation/tests/arena_prove.rs b/crates/rec_aggregation/tests/arena_prove.rs deleted file mode 100644 index 0097859d8..000000000 --- a/crates/rec_aggregation/tests/arena_prove.rs +++ /dev/null @@ -1,27 +0,0 @@ -//! End-to-end proof verification across arena phase resets. -//! -//! This has its own binary because arena phases are process-global and cannot nest. - -use primitives::bench::Plan; - -#[test] -fn repeated_proofs_survive_phase_resets() { - lean_vm::init_prover(); - assert!( - zk_alloc::is_enabled(), - "this test is meaningless unless the arena is engaged" - ); - - rec_aggregation::run_aggregation(3, 1, 1, lean_vm::pcs::TEST_LOG_INV_RATE, Plan::new(2, 0)); - - let stats = zk_alloc::stats(); - assert!(stats.phases >= 3, "expected one phase per proof, got {stats:?}"); - assert!( - stats.peak_bytes > 0, - "no buffer reached the arena, so nothing was actually exercised: {stats:?}" - ); - assert_eq!( - stats.overflow, 0, - "a slab overflowed into the system allocator, so SLAB_SIZE is undersized: {stats:?}" - ); -} diff --git a/crates/sphincs/Cargo.toml b/crates/sphincs/Cargo.toml deleted file mode 100644 index 7ecb4abe6..000000000 --- a/crates/sphincs/Cargo.toml +++ /dev/null @@ -1,14 +0,0 @@ -[package] -name = "sphincs" -version.workspace = true -edition.workspace = true -publish = false - -[lints] -workspace = true - -[dependencies] -primitives.workspace = true -parallel.workspace = true -rand.workspace = true -serde.workspace = true diff --git a/crates/sphincs/src/fts.rs b/crates/sphincs/src/fts.rs deleted file mode 100644 index f139f754f..000000000 --- a/crates/sphincs/src/fts.rs +++ /dev/null @@ -1,86 +0,0 @@ -//! The few-time signature: a forest of `k-1` Merkle trees of `2^a` secret -//! leaves, one leaf opened per tree at an index the message digest picks -//! (FORS+C). -//! -//! Reuse leaks rather than breaks: after `r` signatures on one instance an -//! adversary holds `r` leaves per tree, and can sign a message only if it lands -//! on that instance, has last index zero, and has every other index on a leaf it -//! already holds. - -use crate::*; - -/// What a signature carries for the few-time key: the opened secret and the -/// Merkle path of each of the `k-1` trees. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct FtsOpening { - pub secrets: [Digest; NUM_FTS_TREES], - pub paths: [[Digest; A]; NUM_FTS_TREES], -} - -/// `s_{idx,kappa,j} = Th(P, tw_ftsprf(idx,kappa,j), S)`. -fn fts_secret(pp: &PublicParam, master: &MasterSecret, idx: u64, kappa: usize, j: usize) -> Digest { - th(pp, &tweak(TWEAK_FTS_PRF, kappa, idx as u32, 0, j as u32), master) -} - -fn fts_leaf(pp: &PublicParam, idx: u64, kappa: usize, j: usize, secret: &Digest) -> Digest { - th(pp, &tweak(TWEAK_FTS_LEAF, kappa, idx as u32, 0, j as u32), secret) -} - -fn fts_node(pp: &PublicParam, idx: u64, kappa: usize, level: usize, j: usize, left: &Digest, right: &Digest) -> Digest { - let tw = tweak(TWEAK_FTS_NODE, kappa, idx as u32, level as u32, j as u32); - th_digests(pp, &tw, &[*left, *right]) -} - -/// `Fts.key`: the few-time public key, `Th` over the `k-1` roots. -fn fts_key_of_roots(pp: &PublicParam, idx: u64, roots: &[Digest; NUM_FTS_TREES]) -> Digest { - th_digests(pp, &tweak(TWEAK_FTS_ROOTS, 0, idx as u32, 0, 0), roots) -} - -/// `Fts.key` and `Fts.open` together, the forest being built once. `u[k-1]` is -/// ignored: its tree is the dropped one. -pub fn fts_open(pp: &PublicParam, master: &MasterSecret, idx: u64, u: &[u32; K]) -> (Digest, FtsOpening) { - let mut opening = FtsOpening { - secrets: [[0; N]; NUM_FTS_TREES], - paths: [[[0; N]; A]; NUM_FTS_TREES], - }; - let mut roots = [[0; N]; NUM_FTS_TREES]; - for kappa in 0..NUM_FTS_TREES { - let opened = u[kappa] as usize; - let mut nodes = Vec::with_capacity(1 << A); - for j in 0..1 << A { - let secret = fts_secret(pp, master, idx, kappa, j); - if j == opened { - opening.secrets[kappa] = secret; - } - nodes.push(fts_leaf(pp, idx, kappa, j, &secret)); - } - for level in 0..A { - opening.paths[kappa][level] = nodes[(opened >> level) ^ 1]; - nodes = (0..nodes.len() / 2) - .map(|j| fts_node(pp, idx, kappa, level + 1, j, &nodes[2 * j], &nodes[2 * j + 1])) - .collect(); - } - roots[kappa] = nodes[0]; - } - (fts_key_of_roots(pp, idx, &roots), opening) -} - -/// `Fts.recover`: the few-time key an opening reaches, which is `Fts.key` on an -/// opening of the leaves `u` of that instance and nothing else short of a -/// collision. -pub fn fts_recover(pp: &PublicParam, idx: u64, u: &[u32; K], opening: &FtsOpening) -> Digest { - let roots = std::array::from_fn(|kappa| { - let opened = u[kappa] as usize; - let leaf = fts_leaf(pp, idx, kappa, opened, &opening.secrets[kappa]); - (0..A).fold(leaf, |node, level| { - let sibling = &opening.paths[kappa][level]; - let (left, right) = if (opened >> level) & 1 == 0 { - (node, *sibling) - } else { - (*sibling, node) - }; - fts_node(pp, idx, kappa, level + 1, opened >> (level + 1), &left, &right) - }) - }); - fts_key_of_roots(pp, idx, &roots) -} diff --git a/crates/sphincs/src/hash.rs b/crates/sphincs/src/hash.rs deleted file mode 100644 index 9da644caf..000000000 --- a/crates/sphincs/src/hash.rs +++ /dev/null @@ -1,60 +0,0 @@ -//! The tweakable hash `Th(P, tw, M) = Truncate_n(BLAKE2s(tw | P | M))`, and the -//! 16-byte tweak that names one hash call in the whole structure. -//! -//! Compressions per call, the input including the 32 bytes of tweak and public -//! parameter: 1 for a chain step, a Merkle node, a derived secret and an -//! encoding, 2 for the message digest, 4 for the few-time roots, and 11 for a -//! one-time leaf. - -use crate::*; - -pub const TWEAK_LEN: usize = 16; -pub type Tweak = [u8; TWEAK_LEN]; - -pub const PROTOCOL_DOMAIN_SEP: u8 = 1; - -// Tweak types (byte 1). -pub const TWEAK_PRF: u8 = 0; -pub const TWEAK_CHAIN: u8 = 1; -pub const TWEAK_LEAF: u8 = 2; -pub const TWEAK_NODE: u8 = 3; -pub const TWEAK_ENC: u8 = 4; -pub const TWEAK_PARAMETER: u8 = 5; -pub const TWEAK_RANDOMIZER: u8 = 7; -pub const TWEAK_FTS_PRF: u8 = 8; -pub const TWEAK_FTS_LEAF: u8 = 9; -pub const TWEAK_FTS_NODE: u8 = 10; -pub const TWEAK_FTS_ROOTS: u8 = 11; -pub const TWEAK_MSG: u8 = 12; - -/// `[protocol_domain_sep:1 | type:1 | layer:1 | zero:1 | p:4 | tree:4 | index:4]`, little endian. -/// `lay` identifies a hypertree layer or a tree of a few-time forest. -pub fn tweak(t: u8, lay: usize, tau: u32, p: u32, j: u32) -> Tweak { - debug_assert!(lay < 256); - let mut tw = [0u8; TWEAK_LEN]; - tw[0] = PROTOCOL_DOMAIN_SEP; - tw[1] = t; - tw[2] = lay as u8; - tw[4..8].copy_from_slice(&p.to_le_bytes()); - tw[8..12].copy_from_slice(&tau.to_le_bytes()); - tw[12..16].copy_from_slice(&j.to_le_bytes()); - tw -} - -/// `Th` over a byte payload: an encoding input or a derived secret. -pub fn th(pp: &PublicParam, tw: &Tweak, payload: &[u8]) -> Digest { - let mut hasher = primitives::hash::Hasher::new(); - hasher.update(tw).update(pp).update(payload); - hasher.finalize()[..N].try_into().unwrap() -} - -/// `Th` over a concatenation of digests: a Merkle node, a one-time leaf, or the -/// few-time roots. -pub fn th_digests(pp: &PublicParam, tw: &Tweak, values: &[Digest]) -> Digest { - let mut hasher = primitives::hash::Hasher::new(); - hasher.update(tw).update(pp); - for value in values { - hasher.update(value); - } - hasher.finalize()[..N].try_into().unwrap() -} diff --git a/crates/sphincs/src/lib.rs b/crates/sphincs/src/lib.rs deleted file mode 100644 index 5186138b5..000000000 --- a/crates/sphincs/src/lib.rs +++ /dev/null @@ -1,102 +0,0 @@ -//! SPHINCS+ over BLAKE2s: the stateless scheme specified in -//! `doc/sphincs/main.tex`, with WOTS+C and FORS+C, at `2^24` signatures per key -//! pair. A public key is 32 bytes, a signature 4924, and a verification 497 hash -//! calls. -//! -//! That specification is the reference and every symbol here carries its name: -//! `n`, `w`, `v`, `T`, `d`, `h_lay`, `a`, `k`. `Th` is standard BLAKE2s of -//! the exact byte string `tweak | P | payload` truncated to `n = 128` bits (the -//! `hash` module), and the tweak names one hash call in the whole structure. -//! -//! One 32-byte master seed derives the public parameter and all signing secrets. -//! The signer caches public nodes of the top tree. - -#![cfg_attr(not(test), warn(unused_crate_dependencies))] - -mod hash; -pub use hash::*; -mod ots; -pub use ots::*; -mod fts; -pub use fts::*; -mod sphincs; -pub use sphincs::*; - -/// `n`: hash value and Merkle node length, in bytes. -pub const N: usize = 16; -pub type Digest = [u8; N]; - -/// The public parameter derived from the master seed. -pub const PUBLIC_PARAM_LEN: usize = 16; -pub type PublicParam = [u8; PUBLIC_PARAM_LEN]; - -/// The master secret used to derive all WOTS and FORS secrets. -pub const MASTER_SECRET_LEN: usize = 32; -pub type MasterSecret = [u8; MASTER_SECRET_LEN]; - -/// The per-signature randomizer the message digest is computed under. -pub const RANDOMIZER_LEN: usize = 16; -pub type Randomizer = [u8; RANDOMIZER_LEN]; - -/// The message to sign (a 256-bit message hash). -pub const MESSAGE_LEN: usize = 32; -pub type Message = [u8; MESSAGE_LEN]; - -/// The serialized width of an encoding counter. -pub const COUNTER_LEN: usize = 4; - -// The one-time signature. -/// `w`: chunk size in bits. -pub const W: usize = 3; -/// `2^w`: one more than the steps of a hash chain. -pub const CHAIN_LEN: usize = 1 << W; -/// `v`: code length, one hash chain per chunk. -pub const V: usize = 42; -/// `T`: the sum every codeword has. Above the mean `v(2^w-1)/2 = 147`, so -/// verification walks fewer chain steps and the signer grinds a counter for it. -pub const TARGET_SUM: usize = 191; - -// The hypertree. -/// `d`: hypertree layers, numbered from the top. -pub const D: usize = 3; -/// `h_lay`: the Merkle tree height of each layer. -pub const HEIGHTS: [usize; D] = [12, 7, 7]; -/// `h`: total height, so `2^h` few-time keys. -pub const H: usize = 26; - -// The few-time signature. -/// `a`: log2 of the leaves in one few-time tree. -pub const A: usize = 10; -/// `k`: digest index groups. -pub const K: usize = 15; -/// The forest holds `k-1` trees: the tree of the last digest index carries no -/// information, that index being ground to zero (FORS$^+$C). -pub const NUM_FTS_TREES: usize = K - 1; - -/// `A_max`: digest attempts per signature. -pub const MAX_DIGEST_ATTEMPTS: u64 = 1 << 32; -/// `C_max`: encoding attempts per layer. -pub const MAX_ENCODING_ATTEMPTS: u64 = 1 << 32; - -/// `h + ka`: the message digest's width, all of it consumed by the index and the -/// `k` leaf indices. -pub const DIGEST_BITS: usize = H + K * A; -pub const DIGEST_BYTES: usize = DIGEST_BITS / 8; - -pub const PUB_KEY_SIZE: usize = N + PUBLIC_PARAM_LEN; -/// A secret key is its public parameter and its master secret; the rest is derived. -pub const SECRET_KEY_SIZE: usize = PUBLIC_PARAM_LEN + MASTER_SECRET_LEN; -pub const SIG_SIZE: usize = RANDOMIZER_LEN + NUM_FTS_TREES * (1 + A) * N + D * (COUNTER_LEN + V * N) + H * N; - -/// Calls to the hash function one verification makes: the digest, `Fts.recover`, -/// `d` times `Ots.leaf`, and `Tree.fold`. -pub const VERIFY_HASHES: usize = 1 + (NUM_FTS_TREES * (1 + A) + 1) + D * (V * (CHAIN_LEN - 1) - TARGET_SUM + 2) + H; - -const _: () = assert!(H == HEIGHTS[0] + HEIGHTS[1] + HEIGHTS[2]); -// Each half of an encoding digest holds `v/2` chunks and one pinned bit. -const _: () = assert!(W * V / 2 + 1 == 64); -const _: () = assert!(DIGEST_BITS == DIGEST_BYTES * 8); -const _: () = assert!(TARGET_SUM < V * (CHAIN_LEN - 1)); -const _: () = assert!(PUB_KEY_SIZE == 32); -const _: () = assert!(SIG_SIZE == 4924); -const _: () = assert!(VERIFY_HASHES == 497); diff --git a/crates/sphincs/src/ots.rs b/crates/sphincs/src/ots.rs deleted file mode 100644 index b9c345c48..000000000 --- a/crates/sphincs/src/ots.rs +++ /dev/null @@ -1,103 +0,0 @@ -//! The one-time signature: `v` hash chains of `2^w - 1` steps, and the -//! target-sum code that replaces the Winternitz checksum (WOTS+C). -//! -//! A codeword is `v` chunks summing to `T`. Two distinct words of equal sum -//! cannot be ordered componentwise, so revealing chain position `x_i` on every -//! chain gives a forger nothing: any other codeword needs a value above one of -//! the revealed ones. The price is that most messages do not encode into the -//! code at all, hence the counter the signer searches for and the signature -//! carries. - -use crate::*; - -/// One one-time key's position: the layer, the tree within it, the leaf within -/// that tree. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct Pos { - pub lay: usize, - pub tau: u32, - pub e: u32, -} - -impl Pos { - pub const fn new(lay: usize, tau: u32, e: u32) -> Self { - Self { lay, tau, e } - } -} - -/// `sk_{lay,tau,e,i} = Th(P, tw_prf(lay,tau,i,e), S)`. -pub fn ots_secret(pp: &PublicParam, master: &MasterSecret, pos: Pos, i: usize) -> Digest { - th(pp, &tweak(TWEAK_PRF, pos.lay, pos.tau, i as u32, pos.e), master) -} - -/// `Chain_{lay,tau,e,i}(P, start, steps, value)`: the step onto position `s` is -/// hashed under the tweak of the edge into it. -pub fn chain(pp: &PublicParam, pos: Pos, i: usize, start: usize, steps: usize, value: Digest) -> Digest { - debug_assert!(start + steps < CHAIN_LEN); - (1..=steps).fold(value, |current, step| { - let p = (CHAIN_LEN * i + start + step - 1) as u32; - th(pp, &tweak(TWEAK_CHAIN, pos.lay, pos.tau, p, pos.e), ¤t) - }) -} - -/// The Merkle leaf of a one-time key: `Th` over its `v` chain tips. -pub fn ots_leaf_hash(pp: &PublicParam, pos: Pos, tips: &[Digest; V]) -> Digest { - th_digests(pp, &tweak(TWEAK_LEAF, pos.lay, pos.tau, 0, pos.e), tips) -} - -/// `Enc(P, lay, tau, e, M, c)`: the codeword, or `None` if the digest of that -/// counter is not admissible. -pub fn encode(pp: &PublicParam, pos: Pos, m: &Digest, c: u32) -> Option<[u8; V]> { - let mut payload = [0u8; N + COUNTER_LEN]; - payload[..N].copy_from_slice(m); - payload[N..].copy_from_slice(&c.to_le_bytes()); - codeword(&th(pp, &tweak(TWEAK_ENC, pos.lay, pos.tau, 0, pos.e), &payload)) -} - -/// Each 64-bit half of the digest holds `v/2` chunks of `w` bits and one pinned -/// top bit; pinning it is what makes the codeword determine the digest. -fn codeword(digest: &Digest) -> Option<[u8; V]> { - let mut x = [0u8; V]; - let mut sum = 0; - for (q, half) in digest.as_chunks::<{ N / 2 }>().0.iter().enumerate() { - let d = u64::from_le_bytes(*half); - if d >> (W * V / 2) != 0 { - return None; - } - for r in 0..V / 2 { - let chunk = ((d >> (W * r)) & (CHAIN_LEN as u64 - 1)) as u8; - x[q * (V / 2) + r] = chunk; - sum += chunk as usize; - } - } - (sum == TARGET_SUM).then_some(x) -} - -/// `Ots.sign`: the LEAST admissible counter, and the chain value each chunk -/// opens. Deterministic in its inputs, which is what keeps one key to one -/// codeword: a resumed or randomized search would leak two incomparable -/// codewords and drop forgery to about `2^53`. -pub fn ots_sign(pp: &PublicParam, master: &MasterSecret, pos: Pos, m: &Digest) -> Option<(u32, [Digest; V])> { - let (c, x) = (0..MAX_ENCODING_ATTEMPTS).find_map(|c| encode(pp, pos, m, c as u32).map(|x| (c as u32, x)))?; - let signature = std::array::from_fn(|i| chain(pp, pos, i, 0, x[i] as usize, ots_secret(pp, master, pos, i))); - Some((c, signature)) -} - -/// `Ots.leaf`: the leaf a claimed signature recovers, or `None` if its counter -/// is not admissible for `m`. Does not touch the secrets, which is why it is the -/// verifier's half. -pub fn ots_leaf(pp: &PublicParam, pos: Pos, m: &Digest, c: u32, signature: &[Digest; V]) -> Option { - let x = encode(pp, pos, m, c)?; - let tips = std::array::from_fn(|i| { - let start = x[i] as usize; - chain(pp, pos, i, start, CHAIN_LEN - 1 - start, signature[i]) - }); - Some(ots_leaf_hash(pp, pos, &tips)) -} - -/// The leaf of the one-time key at `pos`, from the master secret: what key -/// generation and every tree rebuild spend their hashes on. -pub fn ots_public_leaf(pp: &PublicParam, master: &MasterSecret, pos: Pos) -> Digest { - let tips = std::array::from_fn(|i| chain(pp, pos, i, 0, CHAIN_LEN - 1, ots_secret(pp, master, pos, i))); - ots_leaf_hash(pp, pos, &tips) -} diff --git a/crates/sphincs/src/sphincs.rs b/crates/sphincs/src/sphincs.rs deleted file mode 100644 index 3b999e524..000000000 --- a/crates/sphincs/src/sphincs.rs +++ /dev/null @@ -1,468 +0,0 @@ -//! The hypertree and the three algorithms: `d` layers of Merkle trees over -//! one-time leaves, the bottom layer signing few-time keys, layer 0's root being -//! the public key. -//! -//! An index derived from the message digest says which few-time key signs, and -//! with it which tree and which leaf are used on every layer. Nothing is -//! reserved and nothing is spent: a key answers for all `2^h` indices, which is -//! what makes the scheme stateless. - -use rand::{CryptoRng, Rng}; -use serde::{Deserialize, Serialize}; - -use crate::*; - -/// `SUFFIX[lay] = sum_{j >= lay} h_j`, the height of everything at or below -/// layer `lay`: the divisors of the index decomposition. -const fn suffix_heights() -> [usize; D + 1] { - let mut suffix = [0; D + 1]; - let mut lay = D; - while lay > 0 { - lay -= 1; - suffix[lay] = suffix[lay + 1] + HEIGHTS[lay]; - } - suffix -} -pub const SUFFIX: [usize; D + 1] = suffix_heights(); -const _: () = assert!(SUFFIX[0] == H); - -/// The layer-0 depth whose nodes a signer caches, halfway up so that the subtree -/// to rebuild and the nodes to refold are both `2^(h_0/2)`. -pub const SPLIT_LEVEL: usize = HEIGHTS[0].div_ceil(2); -pub const CACHE_LEN: usize = 1 << (HEIGHTS[0] - SPLIT_LEVEL); -const _: () = assert!(CACHE_LEN * N == 1024); - -/// `tau_lay(idx)`: the tree used on layer `lay`. -pub fn tree_of(idx: u64, lay: usize) -> u32 { - (idx >> SUFFIX[lay]) as u32 -} - -/// `e_lay(idx)`: the leaf used within that tree. -pub fn leaf_of(idx: u64, lay: usize) -> u32 { - ((idx >> SUFFIX[lay + 1]) & ((1 << HEIGHTS[lay]) - 1)) as u32 -} - -/// Where layer `lay`'s siblings sit in a signature's flat path. -pub fn path_range(lay: usize) -> std::ops::Range { - let start: usize = HEIGHTS[..lay].iter().sum(); - start..start + HEIGHTS[lay] -} - -/// Ordered lexicographically on [`Self::flatten`], which is what an aggregate's -/// signer list is sorted and deduplicated by. -#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)] -pub struct SphincsPublicKey { - pub root: Digest, - pub public_param: PublicParam, -} - -impl SphincsPublicKey { - pub fn flatten(&self) -> [u8; PUB_KEY_SIZE] { - let mut out = [0; PUB_KEY_SIZE]; - out[..N].copy_from_slice(&self.root); - out[N..].copy_from_slice(&self.public_param); - out - } - - pub fn from_bytes(bytes: &[u8; PUB_KEY_SIZE]) -> Self { - Self { - root: bytes[..N].try_into().unwrap(), - public_param: bytes[N..].try_into().unwrap(), - } - } -} - -/// `P`, the root, and the master secret every secret is derived from, plus -/// layer 0's nodes at [`SPLIT_LEVEL`]. Those nodes are a cache and not state: a -/// deterministic function of the master secret, so losing them costs -/// recomputation and nothing else. -#[derive(Clone, Debug)] -pub struct SphincsSecretKey { - pub public_param: PublicParam, - pub root: Digest, - master: MasterSecret, - cache: [Digest; CACHE_LEN], -} - -impl SphincsSecretKey { - /// SECRET KEY MATERIAL: public parameter and master secret. - pub fn to_bytes(&self) -> [u8; SECRET_KEY_SIZE] { - let mut out = [0; SECRET_KEY_SIZE]; - out[..PUBLIC_PARAM_LEN].copy_from_slice(&self.public_param); - out[PUBLIC_PARAM_LEN..].copy_from_slice(&self.master); - out - } - - /// Inverse of [`Self::to_bytes`], costing what [`key_gen_from`] costs: the - /// layer-0 tree is rebuilt rather than stored. - pub fn from_bytes(bytes: &[u8; SECRET_KEY_SIZE]) -> Self { - let (public_param, master) = bytes.split_at(PUBLIC_PARAM_LEN); - key_gen_from(public_param.try_into().unwrap(), master.try_into().unwrap()).0 - } -} - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct SphincsSignature { - pub randomizer: Randomizer, - pub fts: FtsOpening, - pub counters: [u32; D], - pub ots: [[Digest; V]; D], - /// Layer 0's `h_0` siblings, then layer 1's, then layer 2's. - pub paths: [Digest; H], -} - -impl SphincsSignature { - /// The specification's serialization, exactly [`SIG_SIZE`] bytes. - pub fn to_bytes(&self) -> [u8; SIG_SIZE] { - let mut out = [0; SIG_SIZE]; - let mut at = 0; - let mut put = |bytes: &[u8]| { - out[at..at + bytes.len()].copy_from_slice(bytes); - at += bytes.len(); - }; - put(&self.randomizer); - for kappa in 0..NUM_FTS_TREES { - put(&self.fts.secrets[kappa]); - for sibling in &self.fts.paths[kappa] { - put(sibling); - } - } - for lay in 0..D { - put(&self.counters[lay].to_le_bytes()); - for value in &self.ots[lay] { - put(value); - } - for sibling in &self.paths[path_range(lay)] { - put(sibling); - } - } - debug_assert_eq!(at, SIG_SIZE); - out - } - - pub fn from_bytes(bytes: &[u8; SIG_SIZE]) -> Self { - let mut at = 0; - let mut take = |len: usize| { - at += len; - &bytes[at - len..at] - }; - let randomizer = take(RANDOMIZER_LEN).try_into().unwrap(); - let mut fts = FtsOpening { - secrets: [[0; N]; NUM_FTS_TREES], - paths: [[[0; N]; A]; NUM_FTS_TREES], - }; - for kappa in 0..NUM_FTS_TREES { - fts.secrets[kappa] = take(N).try_into().unwrap(); - for level in 0..A { - fts.paths[kappa][level] = take(N).try_into().unwrap(); - } - } - let mut counters = [0; D]; - let mut ots = [[[0; N]; V]; D]; - let mut paths = [[0; N]; H]; - for lay in 0..D { - counters[lay] = u32::from_le_bytes(take(COUNTER_LEN).try_into().unwrap()); - for i in 0..V { - ots[lay][i] = take(N).try_into().unwrap(); - } - for level in path_range(lay) { - paths[level] = take(N).try_into().unwrap(); - } - } - debug_assert_eq!(at, SIG_SIZE); - Self { - randomizer, - fts, - counters, - ots, - paths, - } - } -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] -pub enum SphincsSignError { - /// `A_max` digests in a row had a nonzero last index. - NoAdmissibleDigest, - /// `C_max` counters in a row failed to encode. - NoAdmissibleEncoding, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] -pub enum SphincsVerifyError { - /// The digest's last index is not zero. - InadmissibleDigest, - /// A layer's counter does not encode the message it signs. - InadmissibleEncoding, - RootMismatch, -} - -impl std::fmt::Display for SphincsSignError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::NoAdmissibleDigest => write!(f, "no admissible message digest within A_max attempts"), - Self::NoAdmissibleEncoding => write!(f, "no admissible encoding within C_max attempts"), - } - } -} - -impl std::error::Error for SphincsSignError {} - -impl std::fmt::Display for SphincsVerifyError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::InadmissibleDigest => write!(f, "the digest's last index is not zero"), - Self::InadmissibleEncoding => write!(f, "a layer's counter does not encode the message it signs"), - Self::RootMismatch => write!(f, "the hypertree walk does not reach the key's root"), - } - } -} - -impl std::error::Error for SphincsVerifyError {} - -/// The message digest, read as the index and the `k` leaf indices. `h + ka` bits -/// of a random oracle output, so the index and the last leaf index are disjoint -/// and grinding one does not bias the other. -pub fn message_digest(pp: &PublicParam, root: &Digest, rho: &Randomizer, m: &Message) -> (u64, [u32; K]) { - let mut hasher = primitives::hash::Hasher::new(); - hasher - .update(&tweak(TWEAK_MSG, 0, 0, 0, 0)) - .update(pp) - .update(rho) - .update(root) - .update(m); - let digest = &hasher.finalize()[..DIGEST_BYTES]; - let field = |offset: usize, len: usize| { - (0..len).fold(0u64, |value, bit| { - let position = offset + bit; - value | (u64::from(digest[position / 8] >> (position % 8) & 1) << bit) - }) - }; - (field(0, H), std::array::from_fn(|kappa| field(H + kappa * A, A) as u32)) -} - -fn node(pp: &PublicParam, lay: usize, tau: u32, level: usize, j: u64, left: &Digest, right: &Digest) -> Digest { - let tw = tweak(TWEAK_NODE, lay, tau, level as u32, j as u32); - th_digests(pp, &tw, &[*left, *right]) -} - -/// Merkle levels `from_level..=to_level` of layer `lay`'s tree `tau`, given -/// `bottom`, the complete band of level-`from_level` nodes starting at index -/// `first`. Level `l` is `layers[l - from_level]`. -fn build_up( - pp: &PublicParam, - lay: usize, - tau: u32, - bottom: Vec, - from_level: usize, - to_level: usize, - first: u64, -) -> Vec> { - let mut layers = vec![bottom]; - for level in from_level + 1..=to_level { - let base = first >> (level - from_level); - let children = layers.last().unwrap(); - layers.push( - (0..children.len() / 2) - .map(|j| { - node( - pp, - lay, - tau, - level, - base + j as u64, - &children[2 * j], - &children[2 * j + 1], - ) - }) - .collect(), - ); - } - layers -} - -/// `Gen`, on given `P` and master secret. Only layer 0 is built; the trees below -/// it are built when a signature needs them. -pub fn key_gen_from(public_param: PublicParam, master: MasterSecret) -> (SphincsSecretKey, SphincsPublicKey) { - let leaves = parallel::map_collect(1 << HEIGHTS[0], |e| { - ots_public_leaf(&public_param, &master, Pos::new(0, 0, e as u32)) - }); - let layers = build_up(&public_param, 0, 0, leaves, 0, HEIGHTS[0], 0); - let root = layers[HEIGHTS[0]][0]; - let cache = std::array::from_fn(|i| layers[SPLIT_LEVEL][i]); - ( - SphincsSecretKey { - public_param, - root, - master, - cache, - }, - SphincsPublicKey { root, public_param }, - ) -} - -/// `Gen`, on a fresh key: the seed comes from `rng`, so nothing can regenerate -/// the key. -pub fn key_gen(rng: &mut impl CryptoRng) -> (SphincsSecretKey, SphincsPublicKey) { - key_gen_from_seed(rng.random()) -} - -/// Deterministic [`key_gen`]: the seed is the master secret, and a dedicated -/// tweak derives the public parameter from it. -pub fn key_gen_from_seed(seed: MasterSecret) -> (SphincsSecretKey, SphincsPublicKey) { - let parameter = th(&[0; PUBLIC_PARAM_LEN], &tweak(TWEAK_PARAMETER, 0, 0, 0, 0), &seed); - key_gen_from(parameter, seed) -} - -impl SphincsSecretKey { - pub fn public_key(&self) -> SphincsPublicKey { - SphincsPublicKey { - root: self.root, - public_param: self.public_param, - } - } - - /// Layer `lay`'s tree `tau` rebuilt whole: the siblings at `e` into `path`, - /// and the root. - fn tree_path_and_root(&self, lay: usize, tau: u32, e: u32, path: &mut [Digest]) -> Digest { - debug_assert_eq!(path.len(), HEIGHTS[lay]); - let leaves = (0..1 << HEIGHTS[lay]) - .map(|leaf| ots_public_leaf(&self.public_param, &self.master, Pos::new(lay, tau, leaf))) - .collect(); - let layers = build_up(&self.public_param, lay, tau, leaves, 0, HEIGHTS[lay], 0); - for (level, sibling) in path.iter_mut().enumerate() { - *sibling = layers[level][((e >> level) ^ 1) as usize]; - } - layers[HEIGHTS[lay]][0] - } - - /// Layer 0's siblings at `e`, from the cache: one `2^SPLIT_LEVEL`-leaf - /// subtree rebuilt below it, the cached nodes refolded above it. Returns the - /// root, which the cache reproduces. - fn cached_path_and_root(&self, e: u32, path: &mut [Digest]) -> Digest { - debug_assert_eq!(path.len(), HEIGHTS[0]); - let first = u64::from(e >> SPLIT_LEVEL) << SPLIT_LEVEL; - let leaves = (first..first + (1 << SPLIT_LEVEL)) - .map(|leaf| ots_public_leaf(&self.public_param, &self.master, Pos::new(0, 0, leaf as u32))) - .collect(); - let below = build_up(&self.public_param, 0, 0, leaves, 0, SPLIT_LEVEL, first); - let above = build_up( - &self.public_param, - 0, - 0, - self.cache.to_vec(), - SPLIT_LEVEL, - HEIGHTS[0], - 0, - ); - debug_assert_eq!(below[SPLIT_LEVEL][0], self.cache[(first >> SPLIT_LEVEL) as usize]); - for (level, sibling) in path.iter_mut().enumerate() { - let index = u64::from(e >> level) ^ 1; - *sibling = if level < SPLIT_LEVEL { - below[level][(index - (first >> level)) as usize] - } else { - above[level - SPLIT_LEVEL][index as usize] - }; - } - above[HEIGHTS[0] - SPLIT_LEVEL][0] - } -} - -/// Sign at most `2^24` messages per key. -/// Signing is deterministic and stateless. -pub fn sign(sk: &SphincsSecretKey, message: &Message) -> Result { - // The digest is admissible when its last leaf index is zero, which is what - // drops that tree from the forest; it takes 2^a attempts on average. - let (randomizer, idx, u) = (0..MAX_DIGEST_ATTEMPTS) - .find_map(|trial| { - let mut hasher = primitives::hash::Hasher::new(); - hasher.update(&tweak(TWEAK_RANDOMIZER, 0, 0, trial as u32, 0)); - hasher.update(&sk.public_param).update(&sk.master).update(message); - let randomizer = hasher.finalize()[..RANDOMIZER_LEN].try_into().unwrap(); - let (idx, u) = message_digest(&sk.public_param, &sk.root, &randomizer, message); - (u[K - 1] == 0).then_some((randomizer, idx, u)) - }) - .ok_or(SphincsSignError::NoAdmissibleDigest)?; - - let (fts_key, fts) = fts_open(&sk.public_param, &sk.master, idx, &u); - - let mut message_of_layer = fts_key; - let mut counters = [0; D]; - let mut ots = [[[0; N]; V]; D]; - let mut paths = [[0; N]; H]; - for lay in (0..D).rev() { - let (tau, e) = (tree_of(idx, lay), leaf_of(idx, lay)); - let pos = Pos::new(lay, tau, e); - let (c, signature) = ots_sign(&sk.public_param, &sk.master, pos, &message_of_layer) - .ok_or(SphincsSignError::NoAdmissibleEncoding)?; - counters[lay] = c; - ots[lay] = signature; - let path = &mut paths[path_range(lay)]; - message_of_layer = if lay == 0 { - sk.cached_path_and_root(e, path) - } else { - sk.tree_path_and_root(lay, tau, e, path) - }; - } - // Layer 0's root is discarded: it is the public key's whenever the signer is - // honest, which is also the only check the cache gets. - debug_assert_eq!(message_of_layer, sk.root); - - Ok(SphincsSignature { - randomizer, - fts, - counters, - ots, - paths, - }) -} - -/// `Tree.fold`: the other half of a Merkle opening. -pub fn tree_fold(pp: &PublicParam, pos: Pos, leaf: Digest, path: &[Digest]) -> Digest { - path.iter().enumerate().fold(leaf, |current, (level, sibling)| { - let (left, right) = if (pos.e >> level) & 1 == 0 { - (current, *sibling) - } else { - (*sibling, current) - }; - node( - pp, - pos.lay, - pos.tau, - level + 1, - u64::from(pos.e >> (level + 1)), - &left, - &right, - ) - }) -} - -/// `Ver`. -pub fn verify( - pk: &SphincsPublicKey, - message: &Message, - signature: &SphincsSignature, -) -> Result<(), SphincsVerifyError> { - let (idx, u) = message_digest(&pk.public_param, &pk.root, &signature.randomizer, message); - if u[K - 1] != 0 { - return Err(SphincsVerifyError::InadmissibleDigest); - } - let mut message_of_layer = fts_recover(&pk.public_param, idx, &u, &signature.fts); - for lay in (0..D).rev() { - let pos = Pos::new(lay, tree_of(idx, lay), leaf_of(idx, lay)); - let leaf = ots_leaf( - &pk.public_param, - pos, - &message_of_layer, - signature.counters[lay], - &signature.ots[lay], - ) - .ok_or(SphincsVerifyError::InadmissibleEncoding)?; - message_of_layer = tree_fold(&pk.public_param, pos, leaf, &signature.paths[path_range(lay)]); - } - if message_of_layer == pk.root { - Ok(()) - } else { - Err(SphincsVerifyError::RootMismatch) - } -} diff --git a/crates/sphincs/tests/sphincs_tests.rs b/crates/sphincs/tests/sphincs_tests.rs deleted file mode 100644 index 4317c96b4..000000000 --- a/crates/sphincs/tests/sphincs_tests.rs +++ /dev/null @@ -1,223 +0,0 @@ -use rand::{Rng, SeedableRng, rngs::StdRng}; -use sphincs::*; - -fn test_message() -> Message { - std::array::from_fn(|i| (i * 5 + 3) as u8) -} - -fn test_key(seed: u64) -> (SphincsSecretKey, SphincsPublicKey) { - key_gen(&mut StdRng::seed_from_u64(seed)) -} - -#[test] -fn keygen_sign_verify() { - let (sk, pk) = test_key(0); - assert_eq!(sk.public_key(), pk); - let message = test_message(); - let signature = sign(&sk, &message).unwrap(); - verify(&pk, &message, &signature).unwrap(); - assert_eq!(sign(&sk, &message).unwrap(), signature); -} - -#[test] -fn serialized_sizes_and_roundtrip() { - let (sk, pk) = test_key(1); - let message = test_message(); - let signature = sign(&sk, &message).unwrap(); - - let public_key_bytes = pk.flatten(); - assert_eq!(public_key_bytes.len(), 32); - assert_eq!(SphincsPublicKey::from_bytes(&public_key_bytes), pk); - - let signature_bytes = signature.to_bytes(); - assert_eq!(signature_bytes.len(), 4924); - let decoded = SphincsSignature::from_bytes(&signature_bytes); - assert_eq!(decoded, signature); - verify(&pk, &message, &decoded).unwrap(); -} - -#[test] -fn tampered_signatures_rejected() { - let (sk, pk) = test_key(2); - let message = test_message(); - let signature = sign(&sk, &message).unwrap(); - verify(&pk, &message, &signature).unwrap(); - - let mut other_message = message; - other_message[0] ^= 1; - assert!(verify(&pk, &other_message, &signature).is_err()); - - let mut other_key = pk; - other_key.root[0] ^= 1; - assert!(verify(&other_key, &message, &signature).is_err()); - - // Verification recomputes the digest, so a tampered randomizer asks for - // another index, and asks it of a digest that is admissible only one time in - // 2^a. - let mut tampered = signature.clone(); - tampered.randomizer[0] ^= 1; - assert_eq!( - verify(&pk, &message, &tampered), - Err(SphincsVerifyError::InadmissibleDigest) - ); - - // Everything the bottom layers carry feeds the message a layer above signs, - // and a counter is admissible for one message in 2^13.6, so tampering - // surfaces as an inadmissible encoding rather than as a wrong root. - for tamper in [ - (|s: &mut SphincsSignature| s.fts.secrets[5][0] ^= 1) as fn(&mut SphincsSignature), - |s: &mut SphincsSignature| s.fts.paths[9][4][0] ^= 1, - |s: &mut SphincsSignature| s.counters[2] ^= 1, - |s: &mut SphincsSignature| s.ots[1][17][0] ^= 1, - |s: &mut SphincsSignature| s.paths[H - 1][0] ^= 1, - ] { - let mut tampered = signature.clone(); - tamper(&mut tampered); - assert_eq!( - verify(&pk, &message, &tampered), - Err(SphincsVerifyError::InadmissibleEncoding) - ); - } - - // Layer 0's path is the exception: nothing is signed above it, so it can - // only fail the root comparison. - let mut tampered = signature.clone(); - tampered.paths[0][0] ^= 1; - assert_eq!(verify(&pk, &message, &tampered), Err(SphincsVerifyError::RootMismatch)); - - let mut tampered = signature.clone(); - tampered.ots[0][17][0] ^= 1; - assert_eq!(verify(&pk, &message, &tampered), Err(SphincsVerifyError::RootMismatch)); -} - -/// One key signs one codeword, on which the whole one-time argument rests: the -/// counter is the least admissible one, not any admissible one. -#[test] -fn ots_counter_is_the_least_admissible() { - let mut rng = StdRng::seed_from_u64(4); - let public_param: PublicParam = rng.random(); - let master: MasterSecret = rng.random(); - let pos = Pos::new(2, 1234, 56); - let message: Digest = rng.random(); - - let (counter, signature) = ots_sign(&public_param, &master, pos, &message).unwrap(); - assert!((0..counter).all(|c| encode(&public_param, pos, &message, c).is_none())); - assert_eq!( - ots_leaf(&public_param, pos, &message, counter, &signature), - Some(ots_public_leaf(&public_param, &master, pos)) - ); -} - -#[test] -fn index_decomposition_is_a_bijection_onto_the_bottom_layer() { - let mut rng = StdRng::seed_from_u64(5); - for _ in 0..1000 { - let idx = rng.random::() % (1 << H); - // Every layer's tree is the one whose root sits at the leaf its parent - // layer uses. - for lay in 1..D { - let expected = - u64::from(tree_of(idx, lay - 1)) * (1 << HEIGHTS[lay - 1]) + u64::from(leaf_of(idx, lay - 1)); - assert_eq!(u64::from(tree_of(idx, lay)), expected); - } - assert_eq!(tree_of(idx, 0), 0); - assert_eq!( - u64::from(tree_of(idx, D - 1)) * (1 << HEIGHTS[D - 1]) + u64::from(leaf_of(idx, D - 1)), - idx - ); - } -} - -/// The counter search and the digest resampling are the signer's two grinding -/// loops; both costs are a property of the predicates, so a drift here is a -/// change of scheme. -#[test] -#[ignore] -fn grinding_bits() { - let mut rng = StdRng::seed_from_u64(6); - let public_param: PublicParam = rng.random(); - let master: MasterSecret = rng.random(); - - let samples = 200; - let counters: u64 = (0..samples) - .map(|i| { - let message: Digest = rng.random(); - let pos = Pos::new(i % D, i as u32, i as u32); - u64::from(ots_sign(&public_param, &master, pos, &message).unwrap().0) - }) - .sum(); - // A codeword is one admissible digest, so 1/p is the number of them over - // 2^128: 2^13.60 for T = 191. - let encoding_bits = ((counters as f64 / samples as f64) + 1.0).log2(); - println!("counter search: 2^{encoding_bits:.2} attempts"); - assert!( - (12.6..14.6).contains(&encoding_bits), - "encoding cost moved: {encoding_bits:.2} bits" - ); - - let root: Digest = rng.random(); - let message = test_message(); - let mut attempts = 0u64; - for _ in 0..samples { - loop { - attempts += 1; - let randomizer: Randomizer = rng.random(); - if message_digest(&public_param, &root, &randomizer, &message).1[K - 1] == 0 { - break; - } - } - } - let digest_bits = (attempts as f64 / samples as f64).log2(); - println!("digest resampling: 2^{digest_bits:.2} attempts"); - assert!( - ((A as f64 - 1.0)..(A as f64 + 1.0)).contains(&digest_bits), - "digest cost moved: {digest_bits:.2} bits" - ); -} - -/// A reloaded secret key must sign exactly as the original does: `key_gen` -/// samples the master secret itself, so these bytes are the only way back to a -/// key it produced. -#[test] -fn secret_key_survives_a_round_trip() { - let (sk, pk) = test_key(7); - let reloaded = SphincsSecretKey::from_bytes(&sk.to_bytes()); - - assert_eq!(reloaded.public_key(), pk); - let message = test_message(); - let sig = sign(&reloaded, &message).unwrap(); - verify(&pk, &message, &sig).unwrap(); -} - -#[test] -fn secret_derivation_uses_full_master() { - let pp = [3; PUBLIC_PARAM_LEN]; - let master = [7; MASTER_SECRET_LEN]; - let pos = Pos::new(2, 5, 6); - let ots = ots_secret(&pp, &master, pos, 4); - let (fts, _) = fts_open(&pp, &master, 5, &[0; K]); - for byte in 0..MASTER_SECRET_LEN { - let mut changed = master; - changed[byte] ^= 1; - assert_ne!(ots_secret(&pp, &changed, pos, 4), ots); - assert_ne!(fts_open(&pp, &changed, 5, &[0; K]).0, fts); - } -} - -/// The split between the two entry points: the seed alone determines the key, -/// and the rng one draws a fresh seed per call rather than a fixed one. -#[test] -fn key_gen_entry_points() { - let seed = [17u8; 32]; - let (_, from_seed) = key_gen_from_seed(seed); - assert_eq!(key_gen_from_seed(seed).1, from_seed); - - let mut rng = StdRng::seed_from_u64(31); - let (_, first) = key_gen(&mut rng); - let (_, second) = key_gen(&mut rng); - assert_ne!(first, second); - - // The rng path is exactly a drawn seed, so it reproduces under a fixed rng. - let drawn: [u8; 32] = StdRng::seed_from_u64(31).random(); - assert_eq!(key_gen_from_seed(drawn).1, first); -} diff --git a/crates/xmss/Cargo.toml b/crates/xmss/Cargo.toml deleted file mode 100644 index d0d15a71c..000000000 --- a/crates/xmss/Cargo.toml +++ /dev/null @@ -1,18 +0,0 @@ -[package] -name = "xmss" -version.workspace = true -edition.workspace = true -publish = false - -[lints] -workspace = true - -[dependencies] -primitives.workspace = true -parallel.workspace = true -ethereum_ssz.workspace = true -rand.workspace = true -serde.workspace = true - -[dev-dependencies] -bincode.workspace = true diff --git a/crates/xmss/src/hash.rs b/crates/xmss/src/hash.rs deleted file mode 100644 index f09243115..000000000 --- a/crates/xmss/src/hash.rs +++ /dev/null @@ -1,56 +0,0 @@ -//! The XMSS hash layer: [`tweak_hash`] is standard BLAKE2s of the exact byte -//! string `tweak | pp | payload`, for chain steps, Merkle nodes, WOTS public -//! keys, and message encodings alike. -//! -//! The 16-byte tweak makes every call site a distinct hash function -//! (multi-target separation, as in leanVM) and the public parameter separates -//! users. Standard BLAKE2s binds the exact payload length. -//! -//! Compression counts per call: chain step 1, Merkle node 1, message encoding -//! 2, WOTS public key 11. A full XMSS verification is a constant 144 -//! compressions: 2 (encoding) + 99 (chains, fixed by the target sum) + 11 -//! (tips) + 32 (Merkle path). - -use crate::*; - -pub const PROTOCOL_DOMAIN_SEP: u8 = 0; - -// Tweak types (byte 1). -pub const TWEAK_TYPE_PRF: u8 = 0; -pub const TWEAK_TYPE_CHAIN: u8 = 1; -pub const TWEAK_TYPE_WOTS_PK: u8 = 2; -pub const TWEAK_TYPE_MERKLE: u8 = 3; -pub const TWEAK_TYPE_ENCODING: u8 = 4; -pub const TWEAK_TYPE_PARAMETER: u8 = 5; -pub const TWEAK_TYPE_FILLER: u8 = 6; -pub const TWEAK_TYPE_RANDOMIZER: u8 = 7; - -pub const TWEAK_LEN: usize = 16; -pub type Tweak = [u8; TWEAK_LEN]; - -/// A full 32-byte BLAKE2s chaining value/output. -pub const STATE_LEN: usize = 32; - -/// `[protocol_domain_sep:1 | type:1 | layer:1 | zero:1 | p:4 | tree:4 | index:4]`, little endian. -/// XMSS sets `layer` and `tree` to zero. -/// `index` is the epoch (chain / wots_pk / encoding) or the Merkle node index; -/// `sub_position` is the chain position or the Merkle level. -pub fn make_tweak(tweak_type: u8, sub_position: u32, index: u32) -> Tweak { - let mut tweak = [0u8; TWEAK_LEN]; - tweak[0] = PROTOCOL_DOMAIN_SEP; - tweak[1] = tweak_type; - tweak[4..8].copy_from_slice(&sub_position.to_le_bytes()); - tweak[12..16].copy_from_slice(&index.to_le_bytes()); - tweak -} - -/// Standard BLAKE2s of the exact-length `tweak | pp | payload` byte string. One -/// compression for chain steps (48 bytes total) and Merkle nodes (64 bytes -/// total), more for the multi-block WOTS public-key and encoding inputs. -pub fn tweak_hash(pp: &PublicParam, tweak_type: u8, sub_position: u32, index: u32, payload: &[u8]) -> Digest { - let mut hasher = primitives::hash::Hasher::new(); - hasher.update(&make_tweak(tweak_type, sub_position, index)); - hasher.update(pp); - hasher.update(payload); - hasher.finalize()[..DIGEST_LEN].try_into().unwrap() -} diff --git a/crates/xmss/src/lib.rs b/crates/xmss/src/lib.rs deleted file mode 100644 index 4ca40dbb5..000000000 --- a/crates/xmss/src/lib.rs +++ /dev/null @@ -1,106 +0,0 @@ -//! XMSS over BLAKE2s, with byte-oriented keys and signatures. -//! The concrete scheme is defined in the [XMSS specification]. -//! -//! Every hash is standard BLAKE2s of the exact byte string -//! `tweak | pp | payload`. Randomizer derivation retains 192 bits; other calls -//! retain 128 bits. See the `hash` module for the constructions. -//! -//! [XMSS specification]: https://github.com/leanEthereum/leanVM/releases/download/doc-latest/XMSS.pdf - -#![cfg_attr(not(test), warn(unused_crate_dependencies))] - -mod hash; -pub use hash::*; -mod ssz_serialization; -pub use ssz_serialization::*; -mod wots; -pub use wots::*; -mod xmss; -pub use xmss::*; - -/// n = 128 bits. -pub const DIGEST_LEN: usize = 16; -pub type Digest = [u8; DIGEST_LEN]; -pub type PublicParam = [u8; PUBLIC_PARAM_LEN]; -pub type Randomness = [u8; RANDOMNESS_LEN]; -/// The message to sign (a 256-bit message hash). -pub type Message = [u8; MESSAGE_LEN]; - -// WOTS -pub const V: usize = 42; // number of hash chains -pub const W: usize = 3; -pub const CHAIN_LENGTH: usize = 1 << W; // 8 -/// Chain hashes the VERIFIER walks, summed over all chains: `sum(chain_length - -/// 1 - e_i)`. Constant because the encoding sum is fixed to [`TARGET_SUM`]. -pub const NUM_CHAIN_HASHES: usize = 99; -/// A WOTS encoding `(e_0, .., e_{v-1})` is valid iff every `e_i < CHAIN_LENGTH`, -/// `sum(e_i) = TARGET_SUM`, and the 2 leftover digest bits are zero (see -/// [`wots_encode`]). The signer grinds the randomness until the encoding is -/// valid (no checksum chains). 195 sits above the mean (147) so verification -/// walks fewer chain steps; grinding takes fewer than 2^15 encode attempts on -/// average. -pub const TARGET_SUM: usize = V * (CHAIN_LENGTH - 1) - NUM_CHAIN_HASHES; // 195 -/// Maximum randomizer trials per signature. -pub const MAX_RANDOMIZER_TRIALS: u64 = 1 << 32; -pub const RANDOMNESS_LEN: usize = 24; -pub const MESSAGE_LEN: usize = 32; -pub const PUBLIC_PARAM_LEN: usize = 16; - -// XMSS -/// Merkle tree height: a key is valid for up to `2^32` epochs. -pub const LOG_LIFETIME: usize = 32; - -/// When a signature was made. Each epoch in the key's range may sign only one message. -pub type Epoch = u32; - -/// Serialized sizes (exact under bincode: fixed arrays, no length prefixes). -pub const WOTS_SIG_SIZE: usize = RANDOMNESS_LEN + V * DIGEST_LEN; // 696 -pub const SIG_SIZE: usize = WOTS_SIG_SIZE + LOG_LIFETIME * DIGEST_LEN; // 1208 -pub const PUB_KEY_SIZE: usize = DIGEST_LEN + PUBLIC_PARAM_LEN; // 32 - -// The encoding uses v*w = 126 of the digest's 128 bits; the 2 leftover top -// bits are ground to zero, so the digest decomposes exactly into the chunks. -const _: () = assert!(V * W + 2 == DIGEST_LEN * 8); - -/// Serde for `[T; N]` with N > 32 (serde only derives arrays up to 32): -/// serialized as a fixed-length tuple, exactly like the native array impls. -pub mod array_serialization { - use serde::de::{Error, SeqAccess, Visitor}; - use serde::ser::SerializeTuple; - use serde::{Deserialize, Deserializer, Serialize, Serializer}; - use std::marker::PhantomData; - - pub fn serialize(data: &[T; N], ser: S) -> Result { - let mut tup = ser.serialize_tuple(N)?; - for elem in data { - tup.serialize_element(elem)?; - } - tup.end() - } - - struct ArrayVisitor(PhantomData); - - impl<'de, T: Deserialize<'de> + Copy + Default, const N: usize> Visitor<'de> for ArrayVisitor { - type Value = [T; N]; - - fn expecting(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result { - write!(f, "an array of length {N}") - } - - fn visit_seq>(self, mut seq: A) -> Result<[T; N], A::Error> { - let mut out = [T::default(); N]; - for (i, slot) in out.iter_mut().enumerate() { - *slot = seq.next_element()?.ok_or_else(|| Error::invalid_length(i, &self))?; - } - Ok(out) - } - } - - pub fn deserialize<'de, D, T, const N: usize>(de: D) -> Result<[T; N], D::Error> - where - D: Deserializer<'de>, - T: Deserialize<'de> + Copy + Default, - { - de.deserialize_tuple(N, ArrayVisitor::(PhantomData)) - } -} diff --git a/crates/xmss/src/ssz_serialization.rs b/crates/xmss/src/ssz_serialization.rs deleted file mode 100644 index a51affa29..000000000 --- a/crates/xmss/src/ssz_serialization.rs +++ /dev/null @@ -1,158 +0,0 @@ -//! SSZ encodings (Ethereum consensus-layer compatibility) for the public key and -//! the signature, the two objects that appear in consensus data. The secret key -//! never does, so it keeps only its serde persistence. -//! -//! Both are SSZ containers of fixed-size byte vectors, so each encoding is the -//! plain concatenation of its fields and every byte string of the right length -//! decodes: -//! -//! - [`XmssPublicKey`]: `merkle_root | public_param`, [`PUB_KEY_SSZ_LEN`] bytes. -//! - [`XmssSignature`]: `chain_tips | randomness | merkle_proof`, -//! [`SIGNATURE_SSZ_LEN`] bytes. - -pub use ssz::{Decode, DecodeError, Encode}; - -use crate::*; - -/// SSZ length of an encoded public key. Equal to [`PUB_KEY_SIZE`]: both encodings -/// concatenate the same fixed-size fields. -pub const PUB_KEY_SSZ_LEN: usize = PUB_KEY_SIZE; -/// SSZ length of an encoded signature. Equal to [`SIG_SIZE`], for the same reason. -pub const SIGNATURE_SSZ_LEN: usize = SIG_SIZE; - -/// Peels fixed-size fields off a buffer whose length was checked up front. -struct Reader<'a>(&'a [u8]); - -impl Reader<'_> { - fn take(&mut self) -> [u8; N] { - let (head, tail) = self.0.split_at(N); - self.0 = tail; - head.try_into().unwrap() - } -} - -fn check_len(bytes: &[u8], expected: usize) -> Result, DecodeError> { - if bytes.len() == expected { - Ok(Reader(bytes)) - } else { - Err(DecodeError::InvalidByteLength { - len: bytes.len(), - expected, - }) - } -} - -impl Encode for WotsSignature { - fn is_ssz_fixed_len() -> bool { - true - } - - fn ssz_fixed_len() -> usize { - WOTS_SIG_SIZE - } - - fn ssz_bytes_len(&self) -> usize { - WOTS_SIG_SIZE - } - - fn ssz_append(&self, buf: &mut Vec) { - for chain_tip in &self.chain_tips { - buf.extend_from_slice(chain_tip); - } - buf.extend_from_slice(&self.randomness); - } -} - -impl Decode for WotsSignature { - fn is_ssz_fixed_len() -> bool { - true - } - - fn ssz_fixed_len() -> usize { - WOTS_SIG_SIZE - } - - fn from_ssz_bytes(bytes: &[u8]) -> Result { - let mut reader = check_len(bytes, WOTS_SIG_SIZE)?; - Ok(Self { - chain_tips: std::array::from_fn(|_| reader.take()), - randomness: reader.take(), - }) - } -} - -impl Encode for XmssPublicKey { - fn is_ssz_fixed_len() -> bool { - true - } - - fn ssz_fixed_len() -> usize { - PUB_KEY_SSZ_LEN - } - - fn ssz_bytes_len(&self) -> usize { - PUB_KEY_SSZ_LEN - } - - fn ssz_append(&self, buf: &mut Vec) { - buf.extend_from_slice(&self.merkle_root); - buf.extend_from_slice(&self.public_param); - } -} - -impl Decode for XmssPublicKey { - fn is_ssz_fixed_len() -> bool { - true - } - - fn ssz_fixed_len() -> usize { - PUB_KEY_SSZ_LEN - } - - fn from_ssz_bytes(bytes: &[u8]) -> Result { - let mut reader = check_len(bytes, PUB_KEY_SSZ_LEN)?; - Ok(Self { - merkle_root: reader.take(), - public_param: reader.take(), - }) - } -} - -impl Encode for XmssSignature { - fn is_ssz_fixed_len() -> bool { - true - } - - fn ssz_fixed_len() -> usize { - SIGNATURE_SSZ_LEN - } - - fn ssz_bytes_len(&self) -> usize { - SIGNATURE_SSZ_LEN - } - - fn ssz_append(&self, buf: &mut Vec) { - self.wots_signature.ssz_append(buf); - for neighbour in &self.merkle_proof { - buf.extend_from_slice(neighbour); - } - } -} - -impl Decode for XmssSignature { - fn is_ssz_fixed_len() -> bool { - true - } - - fn ssz_fixed_len() -> usize { - SIGNATURE_SSZ_LEN - } - - fn from_ssz_bytes(bytes: &[u8]) -> Result { - let mut reader = check_len(bytes, SIGNATURE_SSZ_LEN)?; - Ok(Self { - wots_signature: WotsSignature::from_ssz_bytes(&reader.take::())?, - merkle_proof: std::array::from_fn(|_| reader.take()), - }) - } -} diff --git a/crates/xmss/src/wots.rs b/crates/xmss/src/wots.rs deleted file mode 100644 index a3d8cc6af..000000000 --- a/crates/xmss/src/wots.rs +++ /dev/null @@ -1,156 +0,0 @@ -//! WOTS (Winternitz one-time signature) with target-sum encoding. - -use serde::{Deserialize, Serialize}; - -use crate::*; - -#[derive(Debug)] -pub struct WotsSecretKey { - pre_images: [Digest; V], -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct WotsPublicKey(pub [Digest; V]); - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Hash)] -pub struct WotsSignature { - #[serde(with = "crate::array_serialization")] - pub chain_tips: [Digest; V], - pub randomness: Randomness, -} - -impl WotsSecretKey { - pub const fn new(pre_images: [Digest; V]) -> Self { - Self { pre_images } - } - - /// Walk every chain to its tip. Only key generation needs this; signing - /// stops each chain at its encoding digit. - pub fn public_key(&self, public_param: &PublicParam, epoch: Epoch) -> WotsPublicKey { - WotsPublicKey(std::array::from_fn(|i| { - iterate_hash(&self.pre_images[i], CHAIN_LENGTH - 1, public_param, epoch, i, 0) - })) - } - - /// `encoding` and `randomness` come as a pair from - /// [`find_randomness_for_wots_encoding`], which only returns a randomness - /// whose encoding is valid. - pub fn sign( - &self, - encoding: &[u8; V], - randomness: Randomness, - epoch: Epoch, - public_param: &PublicParam, - ) -> WotsSignature { - WotsSignature { - chain_tips: std::array::from_fn(|i| { - iterate_hash(&self.pre_images[i], encoding[i] as usize, public_param, epoch, i, 0) - }), - randomness, - } - } -} - -impl WotsSignature { - pub fn recover_public_key( - &self, - message: &Message, - epoch: Epoch, - public_param: &PublicParam, - ) -> Option { - let encoding = wots_encode(message, epoch, public_param, &self.randomness)?; - Some(WotsPublicKey(std::array::from_fn(|i| { - iterate_hash( - &self.chain_tips[i], - CHAIN_LENGTH - 1 - encoding[i] as usize, - public_param, - epoch, - i, - encoding[i] as usize, - ) - }))) - } -} - -impl WotsPublicKey { - /// The Merkle leaf: standard BLAKE2s over the tweak, public parameter, and - /// 42 concatenated chain tips (704 bytes, 11 compressions). - pub fn hash(&self, public_param: &PublicParam, epoch: Epoch) -> Digest { - tweak_hash(public_param, TWEAK_TYPE_WOTS_PK, 0, epoch, self.0.as_flattened()) - } -} - -/// One chain step (1 compression). The position `chain_index * CHAIN_LENGTH + -/// step` identifies the edge from chain value `step` to `step + 1`. -fn chain_step(public_param: &PublicParam, epoch: Epoch, chain_index: usize, step: usize, value: &Digest) -> Digest { - let position = (chain_index * CHAIN_LENGTH + step) as u32; - tweak_hash(public_param, TWEAK_TYPE_CHAIN, position, epoch, value) -} - -/// Walk chain `chain_index` for `n` steps starting at chain value `start_step`. -fn iterate_hash( - start: &Digest, - steps: usize, - public_param: &PublicParam, - epoch: Epoch, - chain_index: usize, - start_step: usize, -) -> Digest { - (0..steps).fold(*start, |value, offset| { - chain_step(public_param, epoch, chain_index, start_step + offset, &value) - }) -} - -pub fn find_randomness_for_wots_encoding( - message: &Message, - epoch: Epoch, - public_param: &PublicParam, - seed: &[u8; 32], -) -> Option<(Randomness, [u8; V], u64)> { - (0..MAX_RANDOMIZER_TRIALS).find_map(|trial| { - let mut hasher = primitives::hash::Hasher::new(); - hasher.update(&make_tweak(TWEAK_TYPE_RANDOMIZER, trial as u32, epoch)); - hasher.update(public_param).update(seed).update(message); - let randomness = hasher.finalize()[..RANDOMNESS_LEN].try_into().unwrap(); - wots_encode(message, epoch, public_param, &randomness).map(|encoding| (randomness, encoding, trial + 1)) - }) -} - -/// The target-sum encoding. `D = MD(msg | randomness | zeros)` under the -/// encoding tweak, truncated to 16 bytes: 2 standard BLAKE2s compressions over -/// the 96-byte exact input. `D`'s two little-endian 64-bit words each hold 21 -/// chunks of 3 bits (the VM's word width budgets the monomial encoding at 64 -/// bits per word: `g^k = x^k` only for `k < 64`): digit `i < 21` sits at bits -/// `3i` of word 0, digit `i >= 21` at bits `3(i-21)` of word 1. The encoding is -/// valid iff the leftover top bit of EACH word (bits 63 and 127) is zero AND -/// the chunks sum to [`TARGET_SUM`]. Grinding the top bits to zero makes each -/// digest word exactly `sum(e_i * 2^{3i})` of its 21 digits, so both words -/// decompose into the chunks with no slack term. In-circuit this is checked -/// over GF(2^64) per word by accumulating the dispatched digit literals against -/// `8^i` monomial weights (see `verify_sig` in `rec_aggregation`'s `guests/lean_ethereum.py`). -pub fn wots_encode( - message: &Message, - epoch: Epoch, - public_param: &PublicParam, - randomness: &Randomness, -) -> Option<[u8; V]> { - let mut data = [0u8; 2 * STATE_LEN]; - data[..MESSAGE_LEN].copy_from_slice(message); - data[MESSAGE_LEN..][..RANDOMNESS_LEN].copy_from_slice(randomness); - let digest = tweak_hash(public_param, TWEAK_TYPE_ENCODING, 0, epoch, &data); - - let mut encoding = [0u8; V]; - let mut sum = 0; - for (half, bytes) in digest.as_chunks::<8>().0.iter().enumerate() { - let word = u64::from_le_bytes(*bytes); - if word >> (W * V / 2) != 0 { - return None; - } - for i in 0..V / 2 { - let digit = ((word >> (W * i)) & (CHAIN_LENGTH as u64 - 1)) as u8; - encoding[half * (V / 2) + i] = digit; - sum += digit as usize; - } - } - (sum == TARGET_SUM).then_some(encoding) -} diff --git a/crates/xmss/src/xmss.rs b/crates/xmss/src/xmss.rs deleted file mode 100644 index a75e4e90d..000000000 --- a/crates/xmss/src/xmss.rs +++ /dev/null @@ -1,383 +0,0 @@ -//! XMSS: a Merkle tree of `2^LOG_LIFETIME` WOTS public-key hashes. -//! -//! For a range of R = epoch_end - -//! epoch_start + 1 epochs, storage is O(sqrt(R) + LOG_LIFETIME) instead of O(R). -//! The key stores the top tree (in-range band plus a thin spine) and one cached -//! bottom subtree, cut at `split_level = ceil(log2(R)) / 2`. Out-of-range nodes -//! are deterministic `gen_random_node` fillers. - -use std::sync::{Mutex, MutexGuard}; - -use rand::{CryptoRng, Rng}; -use serde::{Deserialize, Serialize}; - -use crate::*; - -/// The encoding is SECRET KEY MATERIAL: it carries the seed. -#[derive(Debug, Serialize, Deserialize)] -pub struct XmssSecretKey { - pub(crate) epoch_start: Epoch, - pub(crate) epoch_end: Epoch, - pub(crate) public_param: PublicParam, - pub(crate) seed: [u8; 32], - pub(crate) split_level: usize, - pub(crate) top: Vec>, - #[serde(skip)] - pub(crate) cache: Mutex>, -} - -#[derive(Debug)] -pub(crate) struct BottomSubtree { - subtree_index: u64, - layers: Vec>, -} - -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Hash)] -pub struct XmssSignature { - pub wots_signature: WotsSignature, - pub merkle_proof: [Digest; LOG_LIFETIME], -} - -/// Ordered lexicographically on `flatten()`, which is what an aggregate's signer -/// set is sorted and deduplicated by. -#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, Hash, PartialOrd, Ord)] -pub struct XmssPublicKey { - pub merkle_root: Digest, - pub public_param: PublicParam, -} - -impl XmssPublicKey { - pub fn flatten(&self) -> [u8; PUB_KEY_SIZE] { - let mut out = [0u8; PUB_KEY_SIZE]; - out[..DIGEST_LEN].copy_from_slice(&self.merkle_root); - out[DIGEST_LEN..].copy_from_slice(&self.public_param); - out - } -} - -fn gen_wots_secret_key(seed: &[u8; 32], public_param: &PublicParam, epoch: Epoch) -> WotsSecretKey { - let pre_images = std::array::from_fn(|i| tweak_hash(public_param, TWEAK_TYPE_PRF, i as u32, epoch, seed)); - WotsSecretKey::new(pre_images) -} - -fn gen_public_param(seed: &[u8; 32]) -> PublicParam { - tweak_hash(&[0; PUBLIC_PARAM_LEN], TWEAK_TYPE_PARAMETER, 0, 0, seed) -} - -fn gen_random_node(seed: &[u8; 32], public_param: &PublicParam, level: usize, index: u64) -> Digest { - tweak_hash(public_param, TWEAK_TYPE_FILLER, level as u32, index as u32, seed) -} - -/// Merkle parent at `level` (1 compression: both children fill one block). -fn merkle_node(public_param: &PublicParam, level: usize, index: u64, left: &Digest, right: &Digest) -> Digest { - let mut data = [0u8; 2 * DIGEST_LEN]; - data[..DIGEST_LEN].copy_from_slice(left); - data[DIGEST_LEN..].copy_from_slice(right); - tweak_hash(public_param, TWEAK_TYPE_MERKLE, level as u32, index as u32, &data) -} - -/// Level-0 layer: WOTS public-key hashes for the in-range leaves `[lo, hi]`. -/// -/// Sequential: this runs once per bottom subtree, and the subtrees are what -/// [`key_gen`] fans out over. -fn leaf_layer(seed: &[u8; 32], public_param: &PublicParam, first_epoch: u64, last_epoch: u64) -> Vec { - (first_epoch..=last_epoch) - .map(|epoch| { - gen_wots_secret_key(seed, public_param, epoch as Epoch) - .public_key(public_param, epoch as Epoch) - .hash(public_param, epoch as Epoch) - }) - .collect() -} - -/// Build levels `(from_level+1)..=to_level` onto `layers`; out-of-range children -/// use `gen_random_node`. -/// -/// Sequential: each level depends on the one below it, and every level here is at -/// most `O(sqrt(R))` wide, whether it is a bottom subtree's levels or the top part -/// built from the subtree roots. -fn build_up( - seed: &[u8; 32], - public_param: &PublicParam, - layers: &mut Vec>, - first_epoch: u64, - last_epoch: u64, - from_level: usize, - to_level: usize, -) { - for level in (from_level + 1)..=to_level { - let (first_node, last_node) = (first_epoch >> level, last_epoch >> level); - let (first_child, last_child) = (first_epoch >> (level - 1), last_epoch >> (level - 1)); - let children = layers.last().unwrap(); - let nodes: Vec = (first_node..=last_node) - .map(|index| { - let child = |child_index: u64| { - if child_index >= first_child && child_index <= last_child { - children[(child_index - first_child) as usize] - } else { - gen_random_node(seed, public_param, level - 1, child_index) - } - }; - merkle_node(public_param, level, index, &child(2 * index), &child(2 * index + 1)) - }) - .collect(); - layers.push(nodes); - } -} - -fn subtree_bounds(epoch_start: u64, epoch_end: u64, split_level: usize, subtree_index: u64) -> (u64, u64) { - ( - epoch_start.max(subtree_index << split_level), - epoch_end.min(((subtree_index + 1) << split_level) - 1), - ) -} - -fn build_subtree_layers( - seed: &[u8; 32], - public_param: &PublicParam, - first_epoch: u64, - last_epoch: u64, - to_level: usize, -) -> Vec> { - let mut layers = vec![leaf_layer(seed, public_param, first_epoch, last_epoch)]; - build_up(seed, public_param, &mut layers, first_epoch, last_epoch, 0, to_level); - layers -} - -#[derive(Debug, PartialEq, Eq, Clone, Copy, Hash)] -pub enum XmssKeyGenError { - InvalidRange, -} - -impl std::fmt::Display for XmssKeyGenError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::InvalidRange => write!(f, "epoch_start is past epoch_end"), - } - } -} - -impl std::error::Error for XmssKeyGenError {} - -/// A fresh key pair for `epoch_start..=epoch_end`, with its seed sampled from `rng`. -pub fn key_gen( - rng: &mut impl CryptoRng, - epoch_start: Epoch, - epoch_end: Epoch, -) -> Result<(XmssSecretKey, XmssPublicKey), XmssKeyGenError> { - key_gen_from_seed(rng.random(), epoch_start, epoch_end) -} - -/// Deterministic [`key_gen`]: one `(seed, epoch range)` always regenerates the -/// same key pair, the seed being the key's entire secret material. -pub fn key_gen_from_seed( - seed: [u8; 32], - epoch_start: Epoch, - epoch_end: Epoch, -) -> Result<(XmssSecretKey, XmssPublicKey), XmssKeyGenError> { - if epoch_start > epoch_end { - return Err(XmssKeyGenError::InvalidRange); - } - let public_param = gen_public_param(&seed); - let (first_epoch, last_epoch) = (epoch_start as u64, epoch_end as u64); - - let split_level = primitives::log2_ceil_usize((last_epoch - first_epoch + 1) as usize).div_ceil(2); - - // The bottom subtrees are the only fan-out in key generation: `O(sqrt(R))` of - // them, each independent, each holding only its own `O(sqrt(R))` layers, so - // peak memory stays `O(sqrt(R))` and one flat dispatch covers the whole tree. - // Everything below this is sequential by construction. - let first_subtree = first_epoch >> split_level; - let last_subtree = last_epoch >> split_level; - let root_layer: Vec = parallel::map_collect((last_subtree - first_subtree + 1) as usize, |i| { - let subtree = first_subtree + i as u64; - let (first_leaf, last_leaf) = subtree_bounds(first_epoch, last_epoch, split_level, subtree); - build_subtree_layers(&seed, &public_param, first_leaf, last_leaf, split_level)[split_level][0] - }); - - let mut top = vec![root_layer]; - build_up( - &seed, - &public_param, - &mut top, - first_epoch, - last_epoch, - split_level, - LOG_LIFETIME, - ); - - let pub_key = XmssPublicKey { - merkle_root: top.last().unwrap()[0], - public_param, - }; - let secret_key = XmssSecretKey { - epoch_start, - epoch_end, - public_param, - seed, - split_level, - top, - cache: Mutex::new(None), - }; - Ok((secret_key, pub_key)) -} - -#[derive(Debug, PartialEq, Eq, Clone, Copy, Hash)] -pub enum XmssSignError { - EpochOutOfRange, - NoAdmissibleEncoding, -} - -impl std::fmt::Display for XmssSignError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::EpochOutOfRange => write!(f, "the epoch is outside the key's range"), - Self::NoAdmissibleEncoding => write!(f, "no admissible encoding within the randomizer trial limit"), - } - } -} - -impl std::error::Error for XmssSignError {} - -/// Never use the same key and epoch to sign two different messages. -/// Signing is deterministic. -pub fn sign(secret_key: &XmssSecretKey, message: &Message, epoch: Epoch) -> Result { - if epoch < secret_key.epoch_start || epoch > secret_key.epoch_end { - return Err(XmssSignError::EpochOutOfRange); - } - let (randomness, encoding, _) = - find_randomness_for_wots_encoding(message, epoch, &secret_key.public_param, &secret_key.seed) - .ok_or(XmssSignError::NoAdmissibleEncoding)?; - let wots_secret_key = gen_wots_secret_key(&secret_key.seed, &secret_key.public_param, epoch); - let wots_signature = wots_secret_key.sign(&encoding, randomness, epoch, &secret_key.public_param); - - let cache = secret_key.cached_bottom_subtree(epoch); - let subtree = cache.as_ref().unwrap(); - let merkle_proof = std::array::from_fn(|level| { - let neighbour_index = ((epoch as u64) >> level) ^ 1; - secret_key.merkle_sibling(level, neighbour_index, subtree) - }); - drop(cache); - Ok(XmssSignature { - wots_signature, - merkle_proof, - }) -} - -impl XmssSecretKey { - /// The epochs this key can sign at. The caller must ensure each epoch signs only one message. - pub fn epoch_range(&self) -> std::ops::RangeInclusive { - self.epoch_start..=self.epoch_end - } - - pub fn public_key(&self) -> XmssPublicKey { - XmssPublicKey { - merkle_root: self.top.last().unwrap()[0], - public_param: self.public_param, - } - } - - /// Warms the signing cache for `epoch`: when the next epoch to sign at is - /// known in advance, calling this ahead of time makes the [`sign`] faster. - pub fn prepare(&self, epoch: Epoch) -> Result<(), XmssSignError> { - if epoch < self.epoch_start || epoch > self.epoch_end { - return Err(XmssSignError::EpochOutOfRange); - } - drop(self.cached_bottom_subtree(epoch)); - Ok(()) - } - - /// The bottom subtree covering `epoch`, rebuilt only on a miss: one subtree - /// serves all `2^split_level` epochs under it. - fn cached_bottom_subtree(&self, epoch: Epoch) -> MutexGuard<'_, Option> { - let subtree_index = (epoch as u64) >> self.split_level; - let mut cache = self.cache.lock().unwrap(); - if cache.as_ref().is_none_or(|s| s.subtree_index != subtree_index) { - *cache = Some(self.build_bottom_subtree(subtree_index)); - } - cache - } - - fn build_bottom_subtree(&self, subtree_index: u64) -> BottomSubtree { - let (lo, hi) = subtree_bounds( - self.epoch_start as u64, - self.epoch_end as u64, - self.split_level, - subtree_index, - ); - let layers = build_subtree_layers(&self.seed, &self.public_param, lo, hi, self.split_level); - BottomSubtree { subtree_index, layers } - } - - /// Authentication-path sibling at `level`: from the top part, the cached - /// subtree, or `gen_random_node`. - fn merkle_sibling(&self, level: usize, neighbour_index: u64, subtree: &BottomSubtree) -> Digest { - let (first_epoch, last_epoch, level_base, layers) = if level >= self.split_level { - ( - self.epoch_start as u64, - self.epoch_end as u64, - self.split_level, - &self.top, - ) - } else { - let (first_epoch, last_epoch) = subtree_bounds( - self.epoch_start as u64, - self.epoch_end as u64, - self.split_level, - subtree.subtree_index, - ); - (first_epoch, last_epoch, 0, &subtree.layers) - }; - let first_node = first_epoch >> level; - if neighbour_index >= first_node && neighbour_index <= (last_epoch >> level) { - layers[level - level_base][(neighbour_index - first_node) as usize] - } else { - gen_random_node(&self.seed, &self.public_param, level, neighbour_index) - } - } -} - -#[derive(Debug, PartialEq, Eq, Clone, Copy, Hash)] -pub enum XmssVerifyError { - InvalidWots, - InvalidMerklePath, -} - -impl std::fmt::Display for XmssVerifyError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::InvalidWots => write!(f, "the WOTS signature does not recover a public key"), - Self::InvalidMerklePath => write!(f, "the authentication path does not reach the key's root"), - } - } -} - -impl std::error::Error for XmssVerifyError {} - -pub fn verify( - pub_key: &XmssPublicKey, - message: &Message, - signature: &XmssSignature, - epoch: Epoch, -) -> Result<(), XmssVerifyError> { - let wots_public_key = signature - .wots_signature - .recover_public_key(message, epoch, &pub_key.public_param) - .ok_or(XmssVerifyError::InvalidWots)?; - let mut current = wots_public_key.hash(&pub_key.public_param, epoch); - for (level, neighbour) in signature.merkle_proof.iter().enumerate() { - let is_left = ((epoch as u64 >> level) & 1) == 0; - let parent_index = (epoch as u64) >> (level + 1); - let (left, right) = if is_left { - (current, *neighbour) - } else { - (*neighbour, current) - }; - current = merkle_node(&pub_key.public_param, level + 1, parent_index, &left, &right); - } - if current == pub_key.merkle_root { - Ok(()) - } else { - Err(XmssVerifyError::InvalidMerklePath) - } -} diff --git a/crates/xmss/tests/xmss_tests.rs b/crates/xmss/tests/xmss_tests.rs deleted file mode 100644 index da12fbcc4..000000000 --- a/crates/xmss/tests/xmss_tests.rs +++ /dev/null @@ -1,259 +0,0 @@ -use rand::{Rng, SeedableRng, rngs::StdRng}; -use xmss::*; - -fn test_message() -> Message { - std::array::from_fn(|i| (i * 3 + 7) as u8) -} - -#[test] -fn keygen_sign_verify() { - let seed: [u8; 32] = std::array::from_fn(|i| i as u8); - let message = test_message(); - - for epoch in [0u32, 1234, u32::MAX] { - let (sk, pk) = key_gen_from_seed(seed, epoch.saturating_sub(1), epoch.saturating_add(2)).unwrap(); - let sig = sign(&sk, &message, epoch).unwrap(); - verify(&pk, &message, &sig, epoch).unwrap(); - assert_eq!(sign(&sk, &message, epoch).unwrap(), sig); - } -} - -#[test] -fn serialize_deserialize_and_size() { - let seed: [u8; 32] = std::array::from_fn(|i| i as u8); - let message = test_message(); - let epoch = 110; - - let (sk, pk) = key_gen_from_seed(seed, 100, 115).unwrap(); - let sig = sign(&sk, &message, epoch).unwrap(); - - let public_key_bytes = bincode::serialize(&pk).unwrap(); - assert_eq!(public_key_bytes.len(), PUB_KEY_SIZE); - let decoded_public_key: XmssPublicKey = bincode::deserialize(&public_key_bytes).unwrap(); - assert_eq!(pk, decoded_public_key); - - let signature_bytes = bincode::serialize(&sig).unwrap(); - assert_eq!(signature_bytes.len(), SIG_SIZE); - let decoded_signature: XmssSignature = bincode::deserialize(&signature_bytes).unwrap(); - assert_eq!(sig, decoded_signature); - - verify(&decoded_public_key, &message, &decoded_signature, epoch).unwrap(); -} - -#[test] -fn deterministic_keygen_and_range_separation() { - let seed = [3u8; 32]; - let (_, pk) = key_gen_from_seed(seed, 50, 60).unwrap(); - let (_, same_seed_and_range) = key_gen_from_seed(seed, 50, 60).unwrap(); - assert_eq!(pk, same_seed_and_range); - // A different range changes the filler/real split, hence the root. - let (_, longer_range) = key_gen_from_seed(seed, 50, 61).unwrap(); - assert_ne!(pk.merkle_root, longer_range.merkle_root); -} - -#[test] -fn tweak_separates_hash_domains() { - let pp = [7u8; PUBLIC_PARAM_LEN]; - let x = [1u8; DIGEST_LEN]; - let base = tweak_hash(&pp, TWEAK_TYPE_CHAIN, 3, 5, &x); - // Different type, position, index, or public parameter: different hash. - assert_ne!(base, tweak_hash(&pp, TWEAK_TYPE_MERKLE, 3, 5, &x)); - assert_ne!(base, tweak_hash(&pp, TWEAK_TYPE_CHAIN, 4, 5, &x)); - assert_ne!(base, tweak_hash(&pp, TWEAK_TYPE_CHAIN, 3, 6, &x)); - assert_ne!(base, tweak_hash(&[8u8; PUBLIC_PARAM_LEN], TWEAK_TYPE_CHAIN, 3, 5, &x)); - // Standard BLAKE2s binds the exact payload length. - let mut extended = [0u8; STATE_LEN]; - extended[..DIGEST_LEN].copy_from_slice(&x); - assert_ne!(base, tweak_hash(&pp, TWEAK_TYPE_CHAIN, 3, 5, &extended)); -} - -/// The multi-block WOTS public-key hash is standard BLAKE2s of `tweak | pp | -/// payload`, assembled independently here and streamed in unrelated chunks. -#[test] -fn multi_block_tweak_hash_is_standard_blake2s() { - let pp = [9u8; PUBLIC_PARAM_LEN]; - let payload = [5u8; V * DIGEST_LEN]; - let mut input = Vec::new(); - input.extend_from_slice(&make_tweak(TWEAK_TYPE_WOTS_PK, 0, 42)); - input.extend_from_slice(&pp); - input.extend_from_slice(&payload); - - let mut hasher = primitives::hash::Hasher::new(); - for chunk in input.chunks(37) { - hasher.update(chunk); - } - let expected = hasher.finalize(); - assert_eq!(expected, primitives::hash::hash(&input)); - assert_eq!( - tweak_hash(&pp, TWEAK_TYPE_WOTS_PK, 0, 42, &payload), - expected[..DIGEST_LEN] - ); -} - -#[test] -fn tampered_signatures_rejected() { - let seed = [9u8; 32]; - let message = test_message(); - let epoch = 7; - let (sk, pk) = key_gen_from_seed(seed, 0, 15).unwrap(); - let sig = sign(&sk, &message, epoch).unwrap(); - verify(&pk, &message, &sig, epoch).unwrap(); - - let mut bad_message = message; - bad_message[0] ^= 1; - assert!(verify(&pk, &bad_message, &sig, epoch).is_err()); - - assert!(verify(&pk, &message, &sig, epoch + 1).is_err()); - - let mut bad_chain_tip = sig.clone(); - bad_chain_tip.wots_signature.chain_tips[5][0] ^= 1; - assert!(verify(&pk, &message, &bad_chain_tip, epoch).is_err()); - - let mut bad_randomness = sig.clone(); - bad_randomness.wots_signature.randomness[0] ^= 1; - assert!(verify(&pk, &message, &bad_randomness, epoch).is_err()); - - let mut bad_merkle_path = sig.clone(); - bad_merkle_path.merkle_proof[10][3] ^= 1; - assert_eq!( - verify(&pk, &message, &bad_merkle_path, epoch), - Err(XmssVerifyError::InvalidMerklePath) - ); - - assert_eq!(sign(&sk, &message, 16), Err(XmssSignError::EpochOutOfRange)); -} - -/// Detect changes to the encoding predicate through its grinding cost. -#[test] -#[ignore] -fn encoding_grinding_bits() { - let n = 200; - let pp = [0u8; PUBLIC_PARAM_LEN]; - let mut total_iters = 0u64; - for i in 0..n { - let mut rng = StdRng::seed_from_u64(i as u64); - let message: Message = rng.random(); - let (_, _, num_iters) = find_randomness_for_wots_encoding(&message, i as u32, &pp, &rng.random()).unwrap(); - total_iters += num_iters; - } - let bits = (total_iters as f64 / n as f64).log2(); - println!("Average grinding bits: {bits:.1}"); - assert!((13.5..15.5).contains(&bits), "grinding cost moved: {bits:.2} bits"); -} - -/// A reloaded secret key must sign exactly as the original does, so that -/// persisting one is a real alternative to regenerating it. -#[test] -fn secret_key_survives_a_round_trip() { - let seed: [u8; 32] = std::array::from_fn(|i| (i * 11) as u8); - let (sk, pk) = key_gen_from_seed(seed, 40, 45).unwrap(); - let reloaded: XmssSecretKey = bincode::deserialize(&bincode::serialize(&sk).unwrap()).unwrap(); - - assert_eq!(reloaded.public_key(), pk); - assert_eq!(reloaded.epoch_range(), 40..=45); - let message = test_message(); - for epoch in [40, 43, 45] { - let sig = sign(&reloaded, &message, epoch as u32).unwrap(); - verify(&pk, &message, &sig, epoch as u32).unwrap(); - } - assert_eq!(sign(&reloaded, &message, 46), Err(XmssSignError::EpochOutOfRange)); -} - -/// The SSZ encoding is the container's fields concatenated, in declaration -/// order. Assembled independently here, so a reordered or resized field cannot -/// be mirrored by the impl. -#[test] -fn ssz_layout_is_exact() { - let seed: [u8; 32] = std::array::from_fn(|i| (i * 5 + 1) as u8); - let message = test_message(); - let epoch = 300; - let (sk, pk) = key_gen_from_seed(seed, 290, 310).unwrap(); - let sig = sign(&sk, &message, epoch).unwrap(); - - let mut expected_pk = Vec::new(); - expected_pk.extend_from_slice(&pk.merkle_root); - expected_pk.extend_from_slice(&pk.public_param); - assert_eq!(expected_pk.len(), PUB_KEY_SSZ_LEN); - assert_eq!(pk.as_ssz_bytes(), expected_pk); - assert_eq!(expected_pk, pk.flatten()); - - let mut expected_sig = Vec::new(); - for chain_tip in &sig.wots_signature.chain_tips { - expected_sig.extend_from_slice(chain_tip); - } - expected_sig.extend_from_slice(&sig.wots_signature.randomness); - for neighbour in &sig.merkle_proof { - expected_sig.extend_from_slice(neighbour); - } - assert_eq!(expected_sig.len(), SIGNATURE_SSZ_LEN); - assert_eq!(sig.as_ssz_bytes(), expected_sig); - - // Round trip, and the decoded pair still verifies. - let decoded_pk = XmssPublicKey::from_ssz_bytes(&expected_pk).unwrap(); - let decoded_sig = XmssSignature::from_ssz_bytes(&expected_sig).unwrap(); - assert_eq!(decoded_pk, pk); - assert_eq!(decoded_sig, sig); - verify(&decoded_pk, &message, &decoded_sig, epoch).unwrap(); -} - -/// A fixed-length container rejects any other length, and only that. -#[test] -fn ssz_rejects_wrong_lengths() { - for len in [0, PUB_KEY_SSZ_LEN - 1, PUB_KEY_SSZ_LEN + 1] { - assert!(matches!( - XmssPublicKey::from_ssz_bytes(&vec![0u8; len]), - Err(DecodeError::InvalidByteLength { .. }) - )); - } - for len in [0, SIGNATURE_SSZ_LEN - 1, SIGNATURE_SSZ_LEN + 1] { - assert!(matches!( - XmssSignature::from_ssz_bytes(&vec![0u8; len]), - Err(DecodeError::InvalidByteLength { .. }) - )); - } - // Every byte string of the right length is a well-formed encoding. - XmssPublicKey::from_ssz_bytes(&[0xab; PUB_KEY_SSZ_LEN]).unwrap(); - XmssSignature::from_ssz_bytes(&[0xcd; SIGNATURE_SSZ_LEN]).unwrap(); -} - -/// `prepare` only warms a cache, so it must change no signature, and a later -/// epoch in a different bottom subtree must still evict what it left behind. -#[test] -fn prepare_warms_without_changing_signatures() { - let seed = [21u8; 32]; - let message = test_message(); - let (sk, pk) = key_gen_from_seed(seed, 0, 255).unwrap(); - - // 0 and 200 are far enough apart to land in different bottom subtrees. - sk.prepare(200).unwrap(); - let after_prepare = sign(&sk, &message, 200).unwrap(); - verify(&pk, &message, &after_prepare, 200).unwrap(); - - let fresh = sign(&sk, &message, 200).unwrap(); - assert_eq!(after_prepare, fresh); - - // A miss on the warmed subtree rebuilds rather than reusing it. - sk.prepare(0).unwrap(); - let other = sign(&sk, &message, 0).unwrap(); - verify(&pk, &message, &other, 0).unwrap(); - - assert_eq!(sk.prepare(256), Err(XmssSignError::EpochOutOfRange)); -} - -/// The rng entry point must forward `(seed, epoch_start, epoch_end)` in that -/// order, and draw a fresh seed on every call. -#[test] -fn key_gen_draws_a_usable_seed() { - let mut rng = StdRng::seed_from_u64(99); - let message = test_message(); - let (sk, pk) = key_gen(&mut rng, 70, 80).unwrap(); - assert_eq!(sk.epoch_range(), 70..=80); - let sig = sign(&sk, &message, 75).unwrap(); - verify(&pk, &message, &sig, 75).unwrap(); - - // A fresh draw is a different key. - let (_, other) = key_gen(&mut rng, 70, 80).unwrap(); - assert_ne!(pk, other); - - assert_eq!(key_gen(&mut rng, 80, 70).unwrap_err(), XmssKeyGenError::InvalidRange); -} diff --git a/doc/bench_blake3_vs_sha2_vs_sha3.sh b/doc/bench_blake3_vs_sha2_vs_sha3.sh deleted file mode 100755 index 498c8b950..000000000 --- a/doc/bench_blake3_vs_sha2_vs_sha3.sh +++ /dev/null @@ -1,161 +0,0 @@ -#!/usr/bin/env bash -# -# ./doc/bench_blake3_vs_sha2_vs_sha3.sh -# -# Compares the three candidate hashes on the branch that implements each: -# BLAKE2s (main), SHA-256 (sha2), Keccak (sha3). Every branch is fetched and -# fast-forwarded first, then measured four ways: -# -# raw hashing, from `multithreaded_throughput`, which reports one 64-byte -# input per compression (per permutation, for Keccak). Run HASH_RUNS times, -# best kept: the test already medians its own passes, so a low run is the -# machine being busy rather than the hash being slow. From it comes the -# compression rate, and the time to generate one XMSS key over a -# 2^LOG_LIFETIME lifetime at COMPRESSIONS compressions per epoch. Comparing -# the three rates, mind that a 64-byte hash is one compression for BLAKE2s -# and one permutation for Keccak, but two compressions for SHA-256, whose -# padding spills a 64-byte input into a second block. -# -# proving that hashing, from flock's `hash_batch_prove_verify`, whose -# throughput line counts compressions (permutations, for Keccak) proven per -# second, with none of the VM around it. -# -# XMSS aggregation, SPHINCS aggregation, and REC_N-to-1 recursion over leaves -# of XMSS aggregates. None of the three counts is the same on every branch: -# each is whatever fills a proof of the same proven size, so a costlier hash -# fits fewer signatures, as does flock's batch, and the counts below are the -# ones each branch's -# README quotes. Comparing them means comparing signatures per second, not -# proving times, the recursion rate counting every signature the node covers -# (REC_N leaves of XMSS_PER_LEAF). -# -# Wrapped in braces so bash parses the whole file before running it: checking -# out a branch where this script does not exist would otherwise pull the rest of -# it out from under the interpreter. -{ -set -euo pipefail - -HASH_RUNS=5 # hash-throughput runs per branch, best kept -HASH_COOLDOWN=10 # seconds between them, to let the machine settle -LOG_LIFETIME=30 # XMSS lifetime, as log2 of the number of epochs -COMPRESSIONS=390 # compressions per epoch of an XMSS keygen - -REPEAT=5 # measured proving passes per benchmark, after a warmup -COOLDOWN=5 # idle seconds before each of them -LOG_INV_RATE=1 # for the two aggregations -REC_LOG_INV_RATE=2 # for recursion -REC_N=2 # child aggregates per recursion node - -BRANCHES=(main sha2 sha3) -HASHES=( BLAKE2s SHA-256 Keccak) -XMSS=( 900 450 205) # signatures per leaf, per that branch's README -SPHINCS=( 245 122 55) # likewise, a leaf of SPHINCS signatures -PER_LEAF=(900 900 450) # likewise, the XMSS leaves a recursion node covers -FLOCK_LOG=(18 17 16) # likewise, log2 of flock's batch of compressions - -TEST=multithreaded_throughput -PACKAGE=primitives -BIN=hash_bench - -cd "$(dirname "$0")/.." - -if [ -n "$(git status --porcelain --untracked-files=no)" ]; then - echo "working tree has changes; commit or stash them first" >&2 - exit 1 -fi -start=$(git symbolic-ref --quiet --short HEAD || git rev-parse HEAD) - -# Two of these at once check out branches under each other, so the runs measure -# whatever branch the other one left in the tree. Take the lock before the first -# checkout, and before arming the trap that undoes it. -LOCK="$(git rev-parse --git-dir)/bench.lock" -if ! mkdir "$LOCK" 2>/dev/null; then - echo "another ./doc/bench.sh is running; wait for it, or remove $LOCK" >&2 - exit 1 -fi -trap 'rmdir "$LOCK"; git checkout --quiet "$start"' EXIT INT TERM - -git fetch --quiet --all - -# The number following $1 in $2, commas stripped. -num() { - local line - line=$(grep -m1 "$1" <<<"$2") || { echo "no \"$1\" line in the output" >&2; exit 1; } - sed -E "s#.*$1[^0-9]*([0-9,]+(\.[0-9]+)?).*#\1#" <<<"$line" | tr -d ',' -} - -# Best of HASH_RUNS `multithreaded_throughput` runs, in Mhash/s. -hash_rate() { - local r out line - local results=() - for ((r = 1; r <= HASH_RUNS; r++)); do - out=$(cargo test --release -p "$PACKAGE" --test "$BIN" "$TEST" -- --exact --ignored --nocapture) - line=$(grep -m1 'Mhash/s' <<<"$out") || { echo "no measurement in the output" >&2; exit 1; } - results+=("$(sed -E 's#.*[^0-9.]([0-9]+(\.[0-9]+)?) Mhash/s.*#\1#' <<<"$line")") - printf 'run %d/%d: %s\n' "$r" "$HASH_RUNS" "$line" >&2 - if ((r < HASH_RUNS)); then sleep "$HASH_COOLDOWN"; fi - done - printf 'samples: %s\n' "${results[*]}" >&2 - printf '%s\n' "${results[@]}" | sort -g | tail -1 -} - -# `aggregate` of $1 signatures of scheme $2, in signatures per second. -aggregate() { - local out - out=$(cargo run --release -- aggregate "--$2" "$1" --log-inv-rate "$LOG_INV_RATE" \ - --repeat "$REPEAT" --cooldown "$COOLDOWN") - echo "$out" >&2 - num 'per signature' "$out" -} - -# Flock's batch proving of 2^$1 compressions, in compressions per second. -flock_rate() { - local out - out=$(BENCH_REPEAT="$REPEAT" BENCH_COOLDOWN="$COOLDOWN" FLOCK_N_LOG="$1" \ - cargo test --release --package flock --test batch_proving_hashes -- \ - hash_batch_prove_verify --exact --nocapture --include-ignored) - echo "$out" >&2 - num 'throughput' "$out" -} - -# `recursion` over REC_N leaves of $1 XMSS signatures, in seconds. -recursion() { - local out - out=$(cargo run --release -- recursion --n "$REC_N" --xmss-per-leaf "$1" \ - --log-inv-rate "$REC_LOG_INV_RATE" --repeat "$REPEAT" --cooldown "$COOLDOWN") - echo "$out" >&2 - num 'proving time' "$out" -} - -rows="" -for i in "${!BRANCHES[@]}"; do - b=${BRANCHES[$i]} - git checkout --quiet "$b" - git merge --quiet --ff-only "origin/$b" - echo "=== $b ===" - rows+=$(printf '%s\t%s\t%s\t%s\t%s\t%s\t%s %s\n' \ - "$b" "${HASHES[$i]}" "$(hash_rate)" "$(flock_rate "${FLOCK_LOG[$i]}")" \ - "$(aggregate "${XMSS[$i]}" xmss)" "$(aggregate "${SPHINCS[$i]}" sphincs)" \ - "${PER_LEAF[$i]}" "$(recursion "${PER_LEAF[$i]}")")$'\n' - echo -done - -awk -F'\t' -v l="$LOG_LIFETIME" -v c="$COMPRESSIONS" -v n="$REC_N" ' -function dur(t) { - if (t < 60) return sprintf("%.1f s", t) - else if (t < 3600) return sprintf("%.1f min", t / 60) - else if (t < 86400) return sprintf("%.1f h", t / 3600) - else return sprintf("%.1f days", t / 86400) -} -BEGIN { printf "%-6s %-8s %10s %10s %20s %9s %11s\n", "branch", "hash", "compr/s", "proven/s", - sprintf("keygen (2^%d)", l), "XMSS/s", "SPHINCS/s" } -{ - split($7, r, " ") - t = 2 ^ l * c / ($3 * 1e6) - printf "%-6s %-8s %8.0f M %8.0f K %20s %9.0f %11.0f\n", $1, $2, $3, $4 / 1e3, - sprintf("%.0f s (%s)", t, dur(t)), $5, $6 - recs = recs sprintf("%-6s %-8s %10d %9.3f s\n", $1, $2, r[1], r[2]) -} -END { printf "\n%-6s %-8s %10s %11s\n%s", "branch", "hash", "XMSS/leaf", sprintf("%d->1", n), recs } -' <<<"${rows%$'\n'}" -} diff --git a/doc/leanvm/body/01-introduction.tex b/doc/leanvm/body/01-introduction.tex index d484c4af8..5efc5e95f 100644 --- a/doc/leanvm/body/01-introduction.tex +++ b/doc/leanvm/body/01-introduction.tex @@ -1,13 +1,4 @@ % !TeX root = ../drafts/01-introduction.tex \section{Introduction}\label{sec:intro} -Ethereum's consensus layer currently uses BLS signatures, its execution layer ECDSA, and its data availability layer KZG. All three are based on elliptic curves, which do not survive a quantum computer~\cite{shor}. Hash-based cryptography, both signatures and snarks (the hash-based ones are sometimes called STARKs~\cite{stark}), is a promising candidate for Ethereum's post-quantum transition: -\begin{itemize}[itemsep=1pt,topsep=3pt] -\item BLS $\to$ XMSS~\cite{rfc8391} (using each Ethereum slot as the nonce, statefulness is no longer an issue when attesting: double signing is already a slashable offense). But XMSS has no native aggregation, unlike BLS, so we need a snark to handle it. -\item ECDSA $\to$ SPHINCS+~\cite{sphincs}. Drawback: dozens of times bigger. Again, snark-based aggregation is a natural solution to mitigate signature size increase. -\item KZG $\to$ Reed--Solomon encode each blob, and provide the Merkle root of the codeword. A snark guarantees that the root commits to a correct encoding. -\end{itemize} - -\vspace{4mm} - -leanVM is a candidate as the snark machinery required above. It's a verifiable virtual machine (commonly a zkVM, for "zero-knowledge virtual machine", though leanVM is not (yet) zk), with a minimalistic ISA (inspired by Cairo \cite{cairo}). It is based on binary fields, inspired by \cite{DP24} and \cite{Flock26}, currently using BLAKE2s \cite{blake2} as hash function. +leanVM is a verifiable virtual machine (commonly a zkVM, for "zero-knowledge virtual machine", though leanVM is not (yet) zk) for RISC-V, the rv64im instruction set \cite{riscv} with one custom instruction, the BLAKE2s compression. It is based on binary fields, inspired by \cite{DP24} and \cite{Flock26}, currently using BLAKE2s \cite{blake2} as hash function. diff --git a/doc/leanvm/body/02-vm-specification.tex b/doc/leanvm/body/02-vm-specification.tex index 6217b8ccc..d35f03900 100644 --- a/doc/leanvm/body/02-vm-specification.tex +++ b/doc/leanvm/body/02-vm-specification.tex @@ -13,85 +13,34 @@ \section{VM specification}\label{sec:vm} An element of $\K$ is a polynomial in $x$ of degree below $64$, written in hex with the coefficient of $x^{i}$ in bit $i$, so $\mathtt{2}$ is $x$ and $\mathtt{3}$ is $x+1$. An element of $\E$ is three of those, its coefficients on $1,y,y^{2}$. -\vspace{5mm} - -LeanVM ISA, also known as leanISA, is inspired by \cite{cairo}. The Virtual Machine takes as input a bytecode (padded to a power of two, $\nprog = 2^{\kbc}$) and some public input $\textsf{input} \in \cube{256} \simeq (\textsf{input}_0, \dots, \textsf{input}_3) \in (\cube{64})^4$. The Virtual Machine then initializes its two $\K$-valued registers: -\begin{itemize} - \item the program counter $\pc \xleftarrow{} 1_\K$ - \item the frame pointer $\fp \xleftarrow{} 1_\K$ -\end{itemize} - -Then, initialize the write-once memory: a buffer, of size $\nmem = 2^{\kmem}$, containing $\E$-valued \textit{memory words} (192 bits). The memory is indexed \textit{in the exponent} of $\gen$ (64 bits). The first memory word is $\mem[\gen^0] \in \E$, the second is $\mem[\gen^1]$, the third is $\mem[\gen^2]$, etc. The bytecode addressing follows the same convention. Each memory access beyond $\gen^{\nmem - 1}$ (resp. $\gen^{\nprog - 1}$ for the bytecode) is invalid. - -The public input fixes the first two memory words $\mem[\gen^0] = \textsf{input}_0 + \textsf{input}_1 \cdot y \in \E$ and $\mem[\gen^1] = \textsf{input}_2 + \textsf{input}_3 \cdot y \in \E$. - -The Virtual Machine execution loop is the following: -\begin{enumerate} - \item fetch the instruction $\mathsf{inst}$ at address $\pc$. - \item execute $\mathsf{inst}$: - \begin{itemize} - \item assert the memory constraints (e.g. $\loc{o_C}=\loc{o_A}+\loc{o_B}$ for XOR) - \item update $\pc$ and $\fp$ (all instructions except \texttt{JUMP} set $\pc \xleftarrow{} \pc \cdot \gen$, advancing to the next instruction, and leave $\fp \xleftarrow{} \fp$ unchanged) - \end{itemize} - \item if $\pc$ equals $\pc_{\textsf{final}} := \gen^{\nprog - 1}$, exit the loop (successful execution). Otherwise go back to $1.$ -\end{enumerate} - -\paragraph{Write-once memory} +An integer $i<2^{64}$ names the element $\intk{i}\in\K$ with that hex writing, the coefficient of $x^{k}$ being bit $k$ of $i$. This is how a register or a memory word holds an integer. It is a bijection and nothing more: $\intk{i+j}$ is not $\intk{i}+\intk{j}$ in general, the field sum being the XOR of the integers. -The first time a memory word is set defines its value for the rest of the execution, i.e. it cannot be modified again in the future. - -When proving the VM execution via a snark, the memory is committed (by an untrusted prover) at the beginning of the execution (with all the values already filled): every instruction then simply consists in reading a bunch of memory cells, and asserting constraints on them (\textit{write-once memory} is also sometimes called \textit{read-only memory}). - -\paragraph{Non determinism} - -For a given pair of bytecode and public input, there is more than one valid execution. +\vspace{5mm} -In particular, the memory log-size $\kmem$ is a non deterministic parameter, freely chosen at the beginning of execution, with the constraint $16\le\kmem\le 32$. +The machine is RISC-V: the RV64I base integer instruction set with the M extension~\cite{riscv}, plus one custom instruction, the BLAKE2s compression (below). We do not restate the instruction semantics, which are the specification's; this section fixes what the specification leaves to the execution environment, and nothing else. A program is a text of $32$-bit instruction words and an entry point, compiled by an ordinary RISC-V toolchain (the repository's guests are Rust, built for \texttt{riscv64im-unknown-none-elf}). -\paragraph{Operands and addressing.} An instruction is an opcode followed by a list of $\K$-valued operands. An operand is either an \emph{fp-relative reference}, an offset $j$ into the current frame encoded as the $\gen$-power $o=\gen^{\,j}$, or one $\K$-lane of an \emph{immediate} ($k_0,k_1,k_2$ in \texttt{SET\_CONSTANT}). With the frame based at $\fp=\gen^{\mathrm{base}}$, a reference names the cell -\[ - \mem[\fp\cdot o],\qquad \fp\cdot o\;=\;\gen^{\mathrm{base}}\cdot\gen^{\,j}\;=\;\gen^{\,\mathrm{base}+j}. -\] - -\paragraph{Instruction set.} The machine implements six instructions. +\paragraph{Memory map.} Memory holds $64$-bit words, word $z$ of a region being the bytes $8z,\dots,8z+7$ past its base, little-endian. Three regions exist, each at a base that is a multiple of the region's largest size, so that word $z$ sits at the byte address $\mathrm{base}\oplus 8z$ as well as $\mathrm{base}+8z$: \begin{center} -\begin{tabularx}{\textwidth}{@{}llX@{}} +\begin{tabularx}{\textwidth}{@{}lllX@{}} \hline -Instruction & Operands & Semantics\\ +Region & Base & Size & Holds\\ \hline -\texttt{XOR} & $[o_A,o_B,o_C]$ & $\loc{o_C}=\loc{o_A}+\loc{o_B}$ \quad(addition in $\E$)\\ -\texttt{MUL\_NATIVE} & $[o_A,o_B,o_C]$ & $\loc{o_C}=\loc{o_A}\cdot\loc{o_B}$ (multiplication in $\E$)\\ -\texttt{SET\_CONSTANT} & $[o,k_0,k_1,k_2]$ & $\loc{o}=k_0+k_1y+k_2y^{2}$\\ -\texttt{DEREF} & $[o_1,o_2,o_3;\,\dmode]$ & $\mem[\loc{o_1}\cdot o_2]=\srcsel{\dmode}$ (see below)\\ -\texttt{JUMP} & $[o_c,o_d,o_f]$ & conditional jump (see below)\\ -\texttt{BLAKE2S} & $[o_{m_0},o_{m_1},o_{m_2},o_{m_3},o_{\mathit{cv}},o_{\mathit{out}},o_{\md}]$ & BLAKE2s compression (see below)\\ +text & $\tbase=\mathtt{0x1000\_0000}$ & $\nprog=2^{\kbc}\le2^{26}$ instructions & the program, readable by fetches only\\ +advice & $\abase=\mathtt{0x2000\_0000}$ & $2^{\kadv}\le2^{26}$ words & what the prover supplies (below)\\ +RAM & $\rbase=\mathtt{0x4000\_0000}$ & $2^{\kmem}$ words, $2\le\kmem\le27$ & data, the stack, the public input and the image\\ \hline \end{tabularx} \end{center} +The three sizes are constants of the program. The whole map lies in one $2$\,GiB window, which is what the \texttt{medany} code model reaches from any $\pc$, and all of it but the last $2$\,KiB of a maximal RAM lies below $\mathtt{0x7FFF\_F800}$, the most \texttt{lui} can form, since \texttt{lui} sign-extends bit $31$ on RV64: \texttt{medlow} code and \texttt{li} address that part. Loads and stores must be naturally aligned, each touching one word; the address of a load or a store is computed modulo $2^{64}$ as the specification says, and it is that address that has to fall in RAM or the advice. The text is not data: a load from it is an access outside memory. +\paragraph{State.} The $32$ integer registers $\regs[0],\dots,\regs[31]$, each one word, and the program counter $\pc$, a byte address. Both parties know the initial state: every register is zero and $\pc$ is the program's entry point. A program sets its own stack pointer. -\texttt{DEREF} stores, at the dereferenced address $\loc{o_1}\cdot o_2$, a value chosen by a \emph{store mode} $\dmode\in\{\texttt{deref\_cell}[o_3],\texttt{deref\_pc},\texttt{deref\_fp}\}$: -\[ - \srcsel{\dmode}\;=\; - \begin{cases} - \loc{o_3} & \dmode=\texttt{deref\_cell}[o_3] \quad(\text{the local cell at offset }o_3 \in \K),\\ - \gen^{2}\cdot\pc & \dmode=\texttt{deref\_pc} \quad(\text{the current program counter advanced by two}),\\ - \fp & \dmode=\texttt{deref\_fp} \quad(\text{the current frame pointer}). - \end{cases} -\] - -Additionally, it asserts $\loc{o_1} \in \K$. - -\vspace{5mm} +\paragraph{Input and output.} RAM's first four words are the run's public input, $\mem[i]=\textsf{input}_i$ for $i<4$; the words after them hold the program's \emph{image}, its initialized data, and the rest of RAM is zero. The run ends when it executes \texttt{ecall} with $\regs[17]=93$ (\texttt{a7} holding the number of the \texttt{exit} system call): its public output is then $\regs[10],\dots,\regs[13]$ (\texttt{a0} to \texttt{a3}), four words. An \texttt{ecall} with another number is a trap. +\paragraph{Advice.} The advice region holds what the prover chooses, $2^{\kadv}$ words read and written like RAM, and the statement says nothing about them: for one program and one input there is a valid execution per advice. A program therefore has to check whatever it reads there, and this is the one channel through which a witness reaches it. -\texttt{JUMP} reads a condition $c=\loc{o_c}$, and branches on whether it is zero. When $c\neq0$ it transfers control, $\pc\gets\loc{o_d}$ and $\fp\gets\loc{o_f}$; when $c=0$ it defaults to $\pc\gets\gen\cdot\pc$ and $\fp\gets\fp$. Additionally, it asserts $\loc{o_c},\loc{o_f}, \loc{o_d} \in \K$ (unconditionally). - -\vspace{5mm} - -\texttt{BLAKE2S} is the standard BLAKE2s compression, consuming a $64$-byte message block and a $32$-byte chaining value and producing $32$ bytes. Each $128$-bit chunk occupies one full $\E$ memory cell through the canonical embedding $a_0+a_1y\mapsto a_0+a_1y+0y^2$; using a cell as a BLAKE2s operand therefore constrains its top limb to zero. The four message chunks are addressed \emph{independently} by references $o_{m_0},\dots,o_{m_3}$ (chunk $i$ at address $\fp\cdot o_{m_i}$), while $o_{\mathit{cv}}$ names two \emph{consecutive} chaining-value cells $\loc{o_{\mathit{cv}}},\loc{\gen\cdot o_{\mathit{cv}}}$. The $256$-bit result is written to the consecutive cells $\loc{o_{\mathit{out}}},\loc{\gen\cdot o_{\mathit{out}}}$, and $o_{\md}$ names one more cell, so each row accesses nine. That cell $\md=\loc{o_{\md}}=\md_0+\md_1y$ packs the standard compression metadata in little-endian order, -\[ - \md=\mathit{counter}_{64}\;\Vert\;\mathit{final}_{32}\;\Vert\;\mathit{last\_node}_{32}. -\] +\paragraph{Traps.} An execution that traps is no execution at all, and has no proof: the prover is refused, not the verifier. The traps are an instruction the machine does not define (a reserved encoding, \texttt{ebreak}, the CSR and \texttt{fence.i} instructions, an unaligned or out-of-text $\pc$, or any $\pc$ in a slot the program left empty), a misaligned load or store, an access outside memory, and an \texttt{ecall} that is not \texttt{exit}. \texttt{fence} is a no-op. +\paragraph{The execution loop.} Fetch the instruction at $\pc$; read its source registers; make its memory access if it has one; write its destination register and the next $\pc$; stop if the instruction was \texttt{exit}. Every instruction reads two registers and writes one, an instruction with fewer reading $\regs[0]$, and one with no destination or with $\regs[0]$ as its destination writing a $33$rd cell, the $\sink$, which nothing reads: this is how $\regs[0]$ is hardwired to zero (\S\ref{sec:e2e-bc}). Reads come before the write, so \texttt{jalr ra, ra} returns to the address it linked from. +\paragraph{The BLAKE2s compression.} The custom instruction \texttt{blake2s rs1, rs2} (opcode \texttt{custom-0}, \texttt{0x0b}, R-type, with \texttt{funct7} and \texttt{rd} zero) is the compression function of BLAKE2s~\cite{blake2}, on the $128$-byte \emph{block} at the address $\regs[\mathit{rs1}]$: its words $0$ to $3$ hold the $32$-byte chaining value $h$, its words $8$ to $15$ the $64$-byte message $m$, and the instruction writes the $32$-byte result to its words $4$ to $7$, leaving the rest as it found it. The byte counter $t$ is $\regs[\mathit{rs2}]$, and the finalization word $f_0$ is all ones when \texttt{funct3} is $1$ (the final block) and zero when it is $0$; the last-node flag $f_1$ is always zero. Word $k$ of the block is the cell at the address $\regs[\mathit{rs1}]\oplus 8k$, which is $\regs[\mathit{rs1}]+8k$ when the block is aligned to $128$ bytes, as the guest library ensures; an unaligned block permutes the words, deterministically, which is a guest bug rather than a soundness problem, and a base that is no word address traps like a misaligned load. A guest hashes a message by keeping $h$ and $t$ across the blocks and copying each result into $h$; the instruction reads $h$ and writes the result in different words because a proof's padding rows can rewrite a cell but not update it (\S\ref{sec:e2e-pad}). diff --git a/doc/leanvm/body/04-committing-the-witness.tex b/doc/leanvm/body/04-committing-the-witness.tex index 68e3308be..e08c20d77 100644 --- a/doc/leanvm/body/04-committing-the-witness.tex +++ b/doc/leanvm/body/04-committing-the-witness.tex @@ -29,4 +29,4 @@ \subsection{Multilinear stacking}\label{sec:stacking} \subsection{Committing to booleans via Ring-Switching}\label{sec:ringswitch} -Flock's BLAKE2s argument (\S\ref{sec:tab-blake2s}) requires a commitment to a Boolean multilinear in $\nflock+\kskip$ variables. Ring switching~\cite{DP24} lets the prover pack its low $\kskip=6$ coordinates into the $\K$-valued multilinear $\qflock$ in $\nflock$ variables (Annex~\ref{annex:rs}), which occupies one region of the stack in \S\ref{sec:stacking}. +Flock's argument for an instruction class (\S\ref{sec:tab-class}) requires a commitment to a Boolean multilinear in $\nflock+\kskip$ variables. Ring switching~\cite{DP24} lets the prover pack its low $\kskip=6$ coordinates into the $\K$-valued multilinear $\qflock$ in $\nflock$ variables (Annex~\ref{annex:rs}), which occupies one region of the stack in \S\ref{sec:stacking}, one per class. diff --git a/doc/leanvm/body/05-arithmetization.tex b/doc/leanvm/body/05-arithmetization.tex index d326f7562..6fae25d7e 100644 --- a/doc/leanvm/body/05-arithmetization.tex +++ b/doc/leanvm/body/05-arithmetization.tex @@ -9,11 +9,11 @@ \subsection{M3 model}\label{sec:m3} \paragraph{Constraints.} The constraints of $T_j$ are $C_{j,1},\dots,C_{j,\ncons{j}}$, each a polynomial over $\K$ of degree at most $d$ ($d=2$ throughout) in the $\ncol{j}$ columns, required to vanish on every row in $\cube{\tau_j}$. -\paragraph{The bus.} All interactions (between tables) share a single \emph{bus}, a channel carrying tuples of up to $m=16$ $\K$-elements (shorter ones are zero-padded). Table $T_j$ has $\nfl{j}$ flushes $\btup_{j,1},\dots,\btup_{j,\nfl{j}}$, each is either a \emph{push} or a \emph{pull}. Each one of the $m$ coordinates in each tuple is a fixed polynomial of degree at most $d$ in the table's columns, so every row $x\in\cube{\tau_j}$ emits one tuple $\btup_{j,f}(x)$ per flush. A few \emph{boundary} tuples are pushed (resp. pulled) by default to the bus (for instance for the VM state $(\dsep{ST}, \pc, \fp)$, the initial $(\dsep{ST}, \gen^0, \gen^0)$ is pushed and the final $(\dsep{ST}, \gen^{\,\nprog-1}, \gen^{0})$ is pulled). +\paragraph{The bus.} All interactions (between tables) share a single \emph{bus}, a channel carrying tuples of up to $m=16$ $\K$-elements (shorter ones are zero-padded). Table $T_j$ has $\nfl{j}$ flushes $\btup_{j,1},\dots,\btup_{j,\nfl{j}}$, each is either a \emph{push} or a \emph{pull}. Each one of the $m$ coordinates in each tuple is a fixed polynomial of degree at most $d$ in the table's columns, so every row $x\in\cube{\tau_j}$ emits one tuple $\btup_{j,f}(x)$ per flush. A few \emph{boundary} tuples are pushed (resp. pulled) by default to the bus (for instance for the VM state $(\dsep{ST}, \pc, \ts)$, the initial $(\dsep{ST}, \intk{\pc_{\mathrm{entry}}}, \gen^{4})$ is pushed and the final $(\dsep{ST}, \intk{\tbase+4(\nprog-1)}, \tsfinal)$ is pulled, \S\ref{sec:state}). \paragraph{Balance.} The bus balances when its pushed tuples and its pulled tuples form the same multiset (\ref{def:multiset}). -\paragraph{Domain separation.} When several interactions share the bus, the first coordinate of every tuple is a \emph{domain separator}, a fixed constant naming its interaction: $\dsep{ST}=\gen^{0}$ for VM state, $\dsep{MEM}=\gen^{1}$ for memory, $\dsep{BC}=\gen^{2}$ for bytecode. (They are needed since all interactions share a common bus.) +\paragraph{Domain separation.} When several interactions share the bus, the first coordinate of every tuple is a \emph{domain separator}, a fixed constant naming its interaction: $\dsep{ST}=\gen^{0}$ for VM state, $\dsep{MEM}=\gen^{1}$ for memory, $\dsep{BC}=\gen^{2}$ for bytecode, $\dsep{RLO}=\gen^{3}$ and $\dsep{RHI}=\gen^{4}$ for the two range arrays, $\dsep{REG}=\gen^{5}$ for the registers. (They are needed since all interactions share a common bus.) \subsection{Balancing the bus: grand product}\label{sec:gp} @@ -48,7 +48,15 @@ \subsection{Balancing the bus: grand product}\label{sec:gp} \end{lemma} \begin{proof}[Proof of Lemma~\ref{lem:gp}] -TODO +One direction is immediate. For the other, work in $R=\K[A_0,\dots,A_3]$, an integral domain, so that $\Pi_P$ lies in $R[X]$ and is monic of degree $|P|$ in $X$: an identity $\Pi_P=\Pi_Q$ therefore forces $|P|=|Q|$, and we induct on that common size, the empty case being $1=1$. + +A tuple is recoverable from its fingerprint. By Fact~\ref{fact:eq}, $\pi_A(\btup)$ is the multilinear extension of $\iota\mapsto\btup_\iota$, so substituting a Boolean point $A=\iota\in\cube{4}$ returns $\btup_\iota$; two tuples with the same fingerprint polynomial therefore agree slot by slot, and $\btup\mapsto\pi_A(\btup)$ is injective into $R$. + +Now take any $\btup\in P$ and substitute $X:=\pi_A(\btup)$, an element of $R$. The factor of $\Pi_P$ at $\btup$ vanishes, so $\Pi_P(A,\pi_A(\btup))=0$, hence +\[ + \Pi_Q(A,\pi_A(\btup))=\prod_{\btup'\in Q}\bigl(\pi_A(\btup)-\pi_A(\btup')\bigr)=0 . +\] +A product vanishes in a domain only if one of its factors does, so $\pi_A(\btup)=\pi_A(\btup')$ for some $\btup'\in Q$, and $\btup=\btup'$ by injectivity. Both products now carry the same monic factor $X-\pi_A(\btup)$, which cancels in the domain $R[X]$, leaving $\Pi_{P\setminus\btup}=\Pi_{Q\setminus\btup}$ over multisets one smaller. The induction hypothesis gives $P\setminus\btup=Q\setminus\btup$, hence $P=Q$. \end{proof} @@ -86,7 +94,7 @@ \subsubsection{Two layers at a time} \subsubsection{Batching several grand-products} -For recursion friendliness, the $\nside$ grand products, numbered $s=1,\dots,\nside$, are proven in one pass, at one sumcheck per layer instead of $\nside$. +The $\nside$ grand products, numbered $s=1,\dots,\nside$, are proven in one pass, at one sumcheck per layer instead of $\nside$. Pad every tree with $1$-leaves up to the depth $\mu$ of the deepest. The verifier then samples at every layer a combiner $\lambda \in \E$, and each layer runs a single sumcheck on the random linear combination of the $\nside$ identities, \[ @@ -107,7 +115,7 @@ \subsection{Stacking the leaves}\label{sec:leafstack} \paragraph{Who settles what.} Three kinds of terms: \begin{itemize}[itemsep=1pt,topsep=3pt] \item \emph{Padding.} Public: the verifier evaluates it. -\item \emph{Boundary blocks.} Every slot of their tuples is public (a constant, a public input, or the index column of \S\ref{sec:idxcol}) or a single committed column, so $\widetilde L_b(\zeta_{<\kappa_b})$ is a linear combination of their evaluations at $\zeta_{<\kappa_b}$. The verifier computes the public ones; the prover sends each committed one, as a claim deferred to the PCS (\S\ref{sec:stacking}). +\item \emph{Boundary blocks.} Every slot of their tuples is public (a constant, a public component, or an index column of \S\ref{sec:idxcol}) or a single committed column, so $\widetilde L_b(\zeta_{<\kappa_b})$ is a linear combination of their evaluations at $\zeta_{<\kappa_b}$. The verifier computes the public ones; the prover sends each committed one, as a claim deferred to the PCS (\S\ref{sec:stacking}). \item \emph{Table blocks.} Nothing is sent for them: the table sumcheck proves them. \end{itemize} The verifier subtracts the first two kinds from the claimed value; the difference, $\mathsf{rem}_{s}$, is what the tables owe. diff --git a/doc/leanvm/body/06-bus-interactions.tex b/doc/leanvm/body/06-bus-interactions.tex index e38b29ee6..e285c0b37 100644 --- a/doc/leanvm/body/06-bus-interactions.tex +++ b/doc/leanvm/body/06-bus-interactions.tex @@ -1,52 +1,68 @@ % !TeX root = ../drafts/06-bus-interactions.tex \section{Bus interactions}\label{sec:omc} -The machine runs three interactions on the bus, differentiated by their domain separators (\S\ref{sec:m3}): the VM state, the read-only memory, and the public bytecode. The last two are lookups using offline memory checking~\cite{BEGKN94}. +The machine runs six interactions on the bus, differentiated by their domain separators (\S\ref{sec:m3}): the VM state, the registers, the memory (RAM and the advice, which share a separator and never share an address), the public bytecode, and two range arrays. All but the first use offline memory checking~\cite{BEGKN94}: the three read-write arrays ordered by a clock (\S\ref{sec:memchan}), and the three read-only arrays in the simpler form of \S\ref{sec:lookup}. + +Two encodings of an integer $i$ coexist on the bus. What the machine indexes (a register, a memory cell, an instruction) and what it holds (a register's value, an address) is the integer itself, $\intk{i}$ (\S\ref{sec:vm}). What the proof system only ever steps (a clock, a read count) is the power $\gen^{i}$, whose successor is a free multiplication by $\gen$. Nothing adds two integers on the bus: every address a row names is a field of its bytecode entry, a word its circuit computed, or an XOR (\S\ref{sec:exp}). \subsection{VM State}\label{sec:state} -The state interaction uses domain separator $\dsep{ST} = \gen^{0}$, and carries the 2 registers, program counter and frame pointer: $\tup{\dsep{ST},\pc,\fp}$. Each instruction row pulls its current state and pushes its successor $\tup{\dsep{ST},\nxt{\pc},\nxt{\fp}}$; the bus is preseeded by pushing the initial state $\tup{\dsep{ST},\pc_{\mathrm{initial}},\fp_{\mathrm{initial}}} = \tup{\dsep{ST},\gen^0,\gen^0}$ and pulling the final state $\tup{\dsep{ST},\pc_{\mathrm{final}},\fp_{\mathrm{final}}} = \tup{\dsep{ST},\gen^{\nprog - 1},\gen^0}$ (boundary conditions). +The state interaction uses domain separator $\dsep{ST} = \gen^{0}$, and carries the program counter and a \emph{clock} $\ts\in\K$: $\tup{\dsep{ST},\pc,\ts}$. Each instruction row pulls its current state and pushes its successor $\tup{\dsep{ST},\pc',\gen^{s}\cdot\ts}$, where the \emph{stride} $s$ is $4$ for every class but \texttt{blake2s}, whose stride is $18$. An honest clock is therefore $\gen^{4c}$ at cycle $c$ of a run without compressions, and a compression counts for four and a half cycles; either way each of a row's accesses gets a timestamp of its own, $\gen^{k}\cdot\ts$ for the access in \emph{slot} $k={Stealth[length=1.5mm]}, - edge/.style={semithick,draw=black!65}, - rowedge/.style={semithick,draw=blue!65!black}, - digest/.style={circle,fill=black!65,inner sep=1.5pt}, - rowdigest/.style={circle,fill=blue!65!black,inner sep=1.5pt} -] -\node[anchor=west,font=\small\bfseries] at (0,1.3) {Encoded blobs: one row per blob, $128$ cells per row}; -\node[blue!65!black] at (2,0.75) {First $64$ cells}; -\node at (6,0.75) {Remaining $64$ cells}; -\foreach \j/\label in {0/0,1/1,2/\cdots,3/63,4/64,5/\cdots,6/126,7/127} { - \node at (\j+0.5,0.25) {$\label$}; -} -\foreach \i in {0,1,2,3} { - \fill[blue!8] (0,-\i*0.65) rectangle (4,-\i*0.65-0.65); - \foreach \j/\label in {0/0,1/1,2/\cdots,3/63,4/64,5/\cdots,6/126,7/127} { - \draw[black!35] (\j,-\i*0.65) rectangle (\j+1,-\i*0.65-0.65); - \ifnum\j=2 - \node at (\j+0.5,-\i*0.65-0.325) {$\cdots$}; - \else\ifnum\j=5 - \node at (\j+0.5,-\i*0.65-0.325) {$\cdots$}; - \else - \node at (\j+0.5,-\i*0.65-0.325) {$e_{\i,\label}$}; - \fi\fi - } - \node[rowdigest,label=above:{$r_\i$}] (row\i) at (-0.9,-\i*0.65-0.325) {}; - \draw[rowedge,->] (0,-\i*0.65-0.325) -- (row\i); -} -\draw[blue!65!black,thick] (0,0) rectangle (4,-2.6); -\draw[orange!85!black,thick] (1,0) rectangle (2,-2.6); -\node[align=center,blue!65!black] at (-1.8,0.75) {$r_i=H(e_{i,0},\ldots,e_{i,63})$}; -\node[rowdigest] (row01) at (-1.8,-0.65) {}; -\node[rowdigest] (row23) at (-1.8,-1.95) {}; -\node[rowdigest,label=above:{$\darow$}] (rowroot) at (-2.9,-1.3) {}; -\draw[rowedge] (row0) -- (row01) -- (row1); -\draw[rowedge] (row2) -- (row23) -- (row3); -\draw[rowedge] (row01) -- (rowroot) -- (row23); -\node[align=center] at (5,-3.05) {Each rectangle: one $2$ KiB cell, labelled by\\its digest $e_{i,j}=H(\text{cell}_{i,j})$.}; - -\draw[orange!85!black,dashed,->] (1.5,-2.6) -- (1.5,-3.65); -\node[anchor=west,font=\small\bfseries] at (-0.3,-4) {Expand column $1$}; -\node[anchor=west] at (3.1,-4) {The same tree is built for every column.}; -\foreach \i in {0,1,2,3} { - \node[draw=orange!85!black,fill=orange!8,minimum width=1.05cm,minimum height=0.45cm] (colleaf\i) at (\i*1.6,-4.65) {$e_{\i,1}$}; -} -\node[digest] (col01) at (0.8,-5.4) {}; -\node[digest] (col23) at (4,-5.4) {}; -\node[digest,label=below right:{$C_1$}] (col1) at (2.4,-6.15) {}; -\draw[edge] (colleaf0) -- (col01) -- (colleaf1); -\draw[edge] (colleaf2) -- (col23) -- (colleaf3); -\draw[edge] (col01) -- (col1) -- (col23); - -\node[digest,label=above:{$C_0$}] (col0) at (0.4,-6.15) {}; -\node[digest,label=above:{$C_{126}$}] (col126) at (6.4,-6.15) {}; -\node[digest,label=above:{$C_{127}$}] (col127) at (8.4,-6.15) {}; -\node at (4.4,-6.15) {$\cdots$}; -\node[digest] (cols01) at (1.4,-6.9) {}; -\node[digest] (cols126127) at (7.4,-6.9) {}; -\draw[edge] (col0) -- (cols01) -- (col1); -\draw[edge] (col126) -- (cols126127) -- (col127); -\node[digest,label=below:{$\dacol$}] (colroot) at (4.4,-8.2) {}; -\draw[edge,dashed] (cols01) -- (colroot) -- (cols126127); -\node[fill=white,inner sep=3pt] at (4.4,-7.35) {Merkle tree over all $128$ column roots}; - -\node[draw,rounded corners=2pt,fill=black!4,inner sep=5pt] (root) at (0.5,-9.35) {$\daroot=H(\darow,\dacol)$}; -\draw[rowedge,->] (rowroot) -- (-2.9,-9.35) -- (root.west); -\draw[edge,->] (colroot) -- (4.9,-8.2) |- (root.east); -\end{tikzpicture} -\caption{Example of encoding of 4 blobs.} -\label{fig:da-hashing} -\end{figure} - -\subsubsection{Codeword membership}\label{sec:da-membership}\label{sec:da-circuit} - -A row is valid exactly when it consists of evaluations of a polynomial of degree less than $\kdim$. On our additive domain at rate $1/2$, such rows are orthogonal to every codeword, and this condition also characterizes them: the code is self-dual (Corollary~\ref{cor:rs-dual}). This gives an efficient membership test. - -The Fiat-Shamir challenges $\log_2\kdim$ in $\E$ are derived from the matrix root $\daroot$. These define the polynomial $\dual$ of \eqref{eq:dual-product} and its evaluation vector $\davec=(\dual(x))_{x\in\dom}$. Let $\davechash=H(\davec)$, serializing each entry as its three little-endian $64$-bit limbs, in domain order. - -The guest recomputes $\daroot$ from the encoded rows, receives $\davec$ as a hinted vector, checks its hash against $\davechash$, and uses it to check membership: -\[ - \inner{\davec}{w_i} - =\sum_{x\in\dom}\dual(x)w_i(x) - \;\qeq\;0. -\] -for every row $w_i$. The vector is shared across all rows and hashed once. It contains $32768$ entries, or $768$ KiB. - -Recursive nodes preserve the pair $(\daroot,\davechash)$ for each retained claim. The final native verifier recomputes $\davechash$ from every retained root (outside of the snark). - -Every valid row passes. Under uniform challenges, a fixed invalid matrix passes with probability at most $\log_2\kdim/|\E|=14/2^{192}$ (Proposition~\ref{prop:rs-membership}). diff --git a/doc/leanvm/body/c-flock-protocol.tex b/doc/leanvm/body/c-flock-protocol.tex index 79bb2ea20..32dbbdb26 100644 --- a/doc/leanvm/body/c-flock-protocol.tex +++ b/doc/leanvm/body/c-flock-protocol.tex @@ -1,13 +1,13 @@ % !TeX root = ../drafts/c-flock-protocol.tex \section{The Flock protocol}\label{annex:flock} -Flock~\cite{Flock26} proves many executions of one Boolean circuit at once, in our case the BLAKE2s compression. Since we do not follow their protocol exactly, we describe our version here. +Flock~\cite{Flock26} proves many executions of one Boolean circuit at once. We use it for the eight instruction classes' circuits (\S\ref{sec:tables}, \S\ref{flock:classes}), each getting its own run of the protocol over its own witness. Since we do not follow their protocol exactly, we describe our version here, for one circuit of block size $\kblock$ with its constant wire at position $\jconst$; the eight differ in those two numbers alone, which \S\ref{flock:classes} gives, and the BLAKE2s compression, the largest at $\kblock=14$, is the running example. \subsection{Statement and commitment}\label{flock:statement} -One BLAKE2s compression has a Boolean R1CS witness block of $2^{\kblock}$ bits, where $\kblock=14$. Its fixed matrices are $A_0,B_0,C_0\in\Ftwo^{2^{\kblock}\times 2^{\kblock}}$, with $C_0=I$. The witness includes the circuit inputs, outputs, AND-gate values, addition carries and one constant-one position; linear wires are substituted away. Within-block coordinates are the low-order coordinates and batch coordinates are the high-order coordinates throughout this annex, and a Boolean tuple is also read as an integer, lowest coordinate first. +One instance of the circuit has a Boolean R1CS witness block of $2^{\kblock}$ bits. Its fixed matrices are $A_0,B_0,C_0\in\Ftwo^{2^{\kblock}\times 2^{\kblock}}$, with $C_0=I$. The witness includes the circuit inputs, outputs, AND-gate values, addition carries and one constant-one position; linear wires are substituted away. Within-block coordinates are the low-order coordinates and batch coordinates are the high-order coordinates throughout this annex, and a Boolean tuple is also read as an integer, lowest coordinate first. -For $2^{\kbatch}$ independent compressions, the witness is: +For $2^{\kbatch}$ independent instances, the witness is: \[ z:\cube{\kbool}\longrightarrow\Ftwo,\qquad \kbool=\kblock+\kbatch, \] @@ -20,12 +20,12 @@ \subsection{Statement and commitment}\label{flock:statement} a(u)b(u)+c(u)=0\qquad\text{for every }u\in\cube{\kbool} \end{equation} -We also enforce the position $512$ of every block to be $1$ (preventing the all-zero witness): +We also enforce the position $\jconst$ of every block to be $1$ (preventing the all-zero witness): \begin{equation}\label{flock:eq:const-one} -z(0,0,0,0,0,0,0,0,0,1,0,0,0,0,\,t)=1\qquad\text{for every }t\in\cube{\kbatch} +z(\jconst,\,t)=1\qquad\text{for every }t\in\cube{\kbatch} \end{equation} -This second condition is enforced by lincheck (\S\ref{flock:lincheck}), and position $512=\jconst$ is where every circuit below puts its constant wire. Since it is asked of every block, the prover pads the batch out to $2^{\kbatch}$ with copies of a genuine witness block rather than with zeros. +This second condition is enforced by lincheck (\S\ref{flock:lincheck}), and $\jconst$ is where the circuit puts its constant wire, the position after its ports (\S\ref{flock:classes}). Since it is asked of every block, the prover pads the batch out to $2^{\kbatch}$ with genuine witness blocks, the instances of its class's no-op on zero inputs, rather than with zeros. Set $\kskip=6$ and $\nflock=\kbool-\kskip$. The prover commits to $z$ by packing its low $\kskip$ coordinates, $64$ bits, to a $\K$-element, following Ring-Switching (Annex~\ref{annex:rs}). The resulting multilinear $\qflock:\cube{\nflock}\to\K$ is committed as one region of the stack (\S\ref{sec:stacking}). @@ -78,7 +78,7 @@ \subsection{Zerocheck}\label{flock:zerocheck} \subsection{Lincheck}\label{flock:lincheck} -Lincheck reduces the three zerocheck claims to claims on $z$. The block structure is what keeps it cheap: the matrices repeat one BLAKE2s circuit, so lincheck runs over a single block, not over the batch. Split $\fc=(\fc_{\mathrm{in}},\fc_{\mathrm{out}})$, where $\fc_{\mathrm{in}}\in\E^{\kblock-\kskip}$ selects the remaining within-block coordinates and $\fc_{\mathrm{out}}\in\E^{\kbatch}$ selects a batch instance. +Lincheck reduces the three zerocheck claims to claims on $z$. The block structure is what keeps it cheap: the matrices repeat one circuit, so lincheck runs over a single block, not over the batch. Split $\fc=(\fc_{\mathrm{in}},\fc_{\mathrm{out}})$, where $\fc_{\mathrm{in}}\in\E^{\kblock-\kskip}$ selects the remaining within-block coordinates and $\fc_{\mathrm{out}}\in\E^{\kbatch}$ selects a batch instance. \begin{lemma}[Block diagonality folds the batch into $z$]\label{lem:flock-block-diagonal} For $a=Az$ with $A=I_{2^{\kbatch}}\otimes A_0$, at every point $(\zskip,\fc_{\mathrm{in}},\fc_{\mathrm{out}})\in\E\times\E^{\kblock-\kskip}\times\E^{\kbatch}$, @@ -113,7 +113,7 @@ \subsection{Lincheck}\label{flock:lincheck} \begin{equation}\label{flock:eq:lincheck} \begin{split} v_a+\lcalpha v_b+\lcalpha^2v_c+\lcalpha^3\qeq\sum_{j\in\cube{\kblock}}\Bigl(&\qext A_0(\zskip,\fc_{\mathrm{in}},j)+\lcalpha\qext B_0(\zskip,\fc_{\mathrm{in}},j)\\ -&+\lcalpha^2\lag_{\jsk}(\zskip)\eq(\fc_{\mathrm{in}},\jin)+\lcalpha^3[j=512]\Bigr)\mle z(j,\fc_{\mathrm{out}}). +&+\lcalpha^2\lag_{\jsk}(\zskip)\eq(\fc_{\mathrm{in}},\jin)+\lcalpha^3[j=\jconst]\Bigr)\mle z(j,\fc_{\mathrm{out}}). \end{split} \end{equation} @@ -132,7 +132,7 @@ \subsection{Lincheck}\label{flock:lincheck} \end{split} \end{equation} -\subsubsection{Concluding Lincheck for a native verifier}\label{flock:native} +\subsubsection{Concluding Lincheck}\label{flock:native} With the row split $k=(\ksk,\kin) \in\cube{\kskip} \times \cube{\kblock-\kskip}$ of the proof above, @@ -147,15 +147,7 @@ \subsubsection{Concluding Lincheck for a native verifier}\label{flock:native} \sum_{i=0}^{63}\qext A_0(\zskip,\fc_{\mathrm{in}},i,\fc'_{\mathrm{in}})\,s_i=\sum_{k,j\in\cube{\kblock}}A_0(k,j)\,\erow(k)\,\wcol(j). \end{equation} -One forward walk of the BLAKE2s circuit returns \eqref{flock:eq:matrix-form} for both matrices (\S\ref{flock:matrices}), never touching the roughly $89$ million nonzero entries they hold. Its committed wires are $896$ free inputs, the constant, $256$ committed $\oplus$ wires carrying the output chaining value, and $14{,}720$ $\wedge$ wires, one per product bit of the $320$ modular additions of the ten rounds; at one multiplication per committed wire on the $A$ side, one per $\wedge$ wire on the $B$ side, and one for the constant factor the remaining $B$ rows share, that is $30{,}594$ field multiplications in $O(\text{circuit})$ work. The $c$ term of \eqref{flock:eq:lincheck-terminal} is $\lcalpha^2\inner{\erow}{\wcol}$ in the same notation, and needs no walk: both sides are tensors, so it contracts to $\lcalpha^2\eq(\fc_{\mathrm{in}},\fc'_{\mathrm{in}})\sum_i\lag_i(\zskip)s_i$. - -\subsubsection{Concluding Lincheck for a recursive verifier}\label{flock:defer} - -To avoid the costly circuit-walk in recursion, the recursion program receives as hint: -\begin{equation}\label{flock:eq:matpart} -M_{\mathrm{lc}}=\sum_{k,j\in\cube{\kblock}}\bigl(A_0(k,j)+\lcalpha B_0(k,j)\bigr)\erow(k)\,\wcol(j). -\end{equation} -This makes it possible to cheaply verify \eqref{flock:eq:lincheck-terminal}. Ensuring validity of the hint is deferred outside of the recursion program, via sumcheck reduction, as described in \S\ref{sec:deferred-claims}. +One forward walk of the circuit returns \eqref{flock:eq:matrix-form} for both matrices (\S\ref{flock:matrices}), never touching the millions of nonzero entries they hold: one multiplication per committed wire on the $A$ side, one per $\wedge$ wire on the $B$ side, and one for the constant factor the remaining $B$ rows share, in $O(\text{circuit})$ work. For the BLAKE2s compression the committed wires are $864$ free input bits (over $896$ positions, the $32$-bit finalization word leaving half of its word empty), the constant, $256$ committed $\oplus$ wires carrying the result, and $14{,}878$ $\wedge$ wires: one per carry of the $480$ $32$-bit additions of the ten rounds, less the two that the first round's constant lanes make structurally zero (\S\ref{flock:classes}). The $c$ term of \eqref{flock:eq:lincheck-terminal} is $\lcalpha^2\inner{\erow}{\wcol}$ in the same notation, and needs no walk: both sides are tensors, so it contracts to $\lcalpha^2\eq(\fc_{\mathrm{in}},\fc'_{\mathrm{in}})\sum_i\lag_i(\zskip)s_i$. \subsubsection{Ring switching}\label{flock:ringswitch} @@ -173,7 +165,7 @@ \subsection{Protocol summary}\label{flock:summary} \item The verifier forms the zerocheck point $r$: seven fixed coordinates, the rest sampled. The $\kskip$ variables the univariate skip replaced consume no equality challenge. \item The prover sends $P|_{\skipcoset}$, $64$ values. The verifier samples $\zskip$ and reconstructs $v_P$. \item The parties run $\nflock$ quadratic zerocheck rounds on \eqref{flock:eq:pskip}, which the prover ends by sending $v_a,v_b$; the terminal identity \eqref{flock:eq:zc-terminal} then fixes $v_c$. -\item The verifier samples $\lcalpha$ and the parties run the $8$ quadratic lincheck rounds on \eqref{flock:eq:lincheck}. The prover sends $(s_0,\ldots,s_{63})$, and the verifier checks \eqref{flock:eq:lincheck-terminal}: one circuit walk natively, the hint \eqref{flock:eq:matpart} in recursion. +\item The verifier samples $\lcalpha$ and the parties run the $\kblock-\kskip$ quadratic lincheck rounds on \eqref{flock:eq:lincheck}. The prover sends $(s_0,\ldots,s_{63})$, and the verifier checks \eqref{flock:eq:lincheck-terminal} with one circuit walk. \end{enumerate} What Flock leaves is one family of $64$ claims for Ring-Switching (\S\ref{flock:ringswitch}): $(s_i \qeq \mle z(i, \fc'_{\mathrm{in}},\fc_{\mathrm{out}}))_{0 \leq i \leq 63}$. \end{protocol} @@ -218,7 +210,7 @@ \subsubsection{The R1CS matrices}\label{flock:rows} The R1CS constraints given by $A_0$, $B_0$ and $C_0=I$ faithfully encode the circuit. \subsubsection{Forward algorithm}\label{flock:forward} -Given a column vector $W:\cube{\kblock}\to\E$, the forward algorithm computes the two row-indexed vectors $A_0W$ and $B_0W$ without constructing either matrix. The native lincheck verifier uses $W=\wcol$ and retains only the scalar $\inner{\erow}{A_0W}+\lcalpha\inner{\erow}{B_0W}$. The prover of the deferred matrix sumcheck in \S\ref{sec:deferred-claims} instead stores both vectors and folds them through its first $\kblock$ rounds. +Given a column vector $W:\cube{\kblock}\to\E$, the forward algorithm computes the two row-indexed vectors $A_0W$ and $B_0W$ without constructing either matrix. The native lincheck verifier uses $W=\wcol$ and retains only the scalar $\inner{\erow}{A_0W}+\lcalpha\inner{\erow}{B_0W}$. To compute them, visit the wires from $w_1$ to $w_N$ and evaluate every expansion against $W$: \[ @@ -258,6 +250,45 @@ \subsubsection{Transpose algorithm}\label{flock:backward} so add $\chg{w}^{X}$ to the existing coefficient $\chg{u}^{X}$ of every predecessor $u\in S_w$. \end{itemize} +\subsection{The class circuits}\label{flock:classes} + +Each instruction class's circuit (\S\ref{sec:tables}) is a circuit in the sense of Definition~\ref{def:circuit}, built from a small vocabulary of gates over $64$-bit words: XOR, AND, the multiplexer $s\,(x\oplus y)\oplus y$, the ripple-carry adder and the carry-save multiplier below. One block layout serves all eight: the ports first, $64$ positions per word, the inputs then the outputs, bit $i$ of a word at position $i$ of its range; the constant wire at $\jconst$, the position right after them; then the $\wedge$ wires in the order the circuit makes them. A port's bits are free wires when it is an input, and committed $\oplus$ wires when it is an output, so that the result leaves the circuit; a port bit with no wire, such as the $63$ high bits of a one-bit output, is an empty position, which \eqref{flock:eq:r1cs} forces to zero. Each port is therefore one packed $\K$-element, equal to the register, bytecode or memory word it stands for (\S\ref{sec:tab-class}). A hint (\texttt{DIV}'s quotient and remainder) is an input port that no tuple carries. +\begin{center} +\begin{tabular}{lrrrrr} +\hline +Circuit & ports (in $\to$ out) & $\kblock$ & $\jconst$ & $\wedge$ wires & positions used\\ +\hline +\texttt{ALU} & $4\to2$ & $10$ & $384$ & $424$ & $809$\\ +\texttt{LOAD} & $4\to2$ & $10$ & $384$ & $320$ & $705$\\ +\texttt{STORE} & $5\to3$ & $10$ & $512$ & $346$ & $859$\\ +\texttt{SHIFT} & $4\to1$ & $10$ & $320$ & $579$ & $900$\\ +\texttt{MUL} & $3\to1$ & $12$ & $256$ & $2{,}175$ & $2{,}432$\\ +\texttt{MULH} & $3\to1$ & $13$ & $256$ & $4{,}545$ & $4{,}802$\\ +\texttt{DIV} & $5\to2$ & $13$ & $448$ & $5{,}156$ & $5{,}605$\\ +\texttt{HASH} & $14\to4$ & $14$ & $1{,}152$ & $14{,}878$ & $16{,}031$\\ +\hline +\end{tabular} +\end{center} +What the two verifiers and the prover have to agree on is this layout and the order the $\wedge$ wires are made in, which fixes every position; the order of the $\oplus$ wires is free, since they are substituted away. + +\paragraph{Addition.} A ripple-carry adder over $n$ bits ($n=64$, or $32$ in the compression). With $c_0=0$, the carry into position $i+1$ is the majority of $a_i,b_i,c_i$, +\[ +c_{i+1}=(a_i\oplus c_i)(b_i\oplus c_i)\oplus c_i , +\] +one $\wedge$ wire, and the sum bit $a_i\oplus b_i\oplus c_i$ is affine in the wires before it. The carry out of the top position falls off the modulus, which leaves $n-1$ products. A subtraction is the addition of the complement with a carry in of $1$, and a comparison is its carry out. + +\paragraph{Multiplication.} A schoolbook multiplier would pay one product per partial product $a_ib_j$ before summing them. Over $\Ftwo$ the partial products are free: with $e_{ij}=1\oplus a_i\oplus b_j$, one has $2a_ib_j=a_i+b_j-1+e_{ij}$ over the integers. Summing over $i,j$, with $\bar a=2^{64}-1-a$ the complement, +\[ +2ab=\sum_{i<64}r_i\,2^{i}+\bar a+\bar b+(a+b)\,2^{64}+1-2^{128},\qquad r_i=\begin{cases}b&a_i=1\\ \bar b&a_i=0,\end{cases} +\] +whose right side is a sum of rows of bits each affine in the inputs. Its column $0$ is $e_{00}+\bar a_0+\bar b_0+1=2+2g$ with $g=\bar a_0\bar b_0$, so after that one product the identity halves: $ab$ is $1+g$ plus the other columns shifted down one place, which is $66$ rows, the $64$ rows $r_i$ and one each for $a$ and $b$, with $1$ and $g$ in the empty low positions of two of them. + +Carry-save steps then compress the rows. A step replaces three rows by their sum row, the position-wise $\oplus$, and their carry row, the position-wise majority shifted up one place: a majority costs one product where at least two of the three rows have a bit, except that where exactly two do and the carry row is still free at that position, one of the two bits moves into it for nothing. Each step takes the three rows that end lowest. After $64$ steps two rows remain, and the adder above finishes. The low word of the product keeps $64$ positions of every row and drops the carry out of the top one, which is \texttt{MUL}'s $2{,}175$ products; the high word keeps all $128$, which is \texttt{MULH}'s, its signed forms correcting the product of the operands' magnitudes. \texttt{DIV} multiplies the divisor by the hinted quotient the same way and adds the remainder. + +\paragraph{The compression.} \texttt{HASH} is the BLAKE2s compression itself: the ten rounds' $80$ G functions, each six $32$-bit additions on the halves of the ports' words, a three-operand addition being two of them chained, and XORs and rotations, which are free. Only the carries are products, $31$ per addition. + +\paragraph{Batch size.} The zerocheck of \S\ref{flock:zerocheck} fixes seven coordinates of its point beyond the $\kskip$ skipped ones, so it needs $\kbool\ge13$, and lincheck works on batches of at least eight blocks, $\kbatch\ge3$. For the circuits of $\kblock\ge10$ the second bound is the binding one; a batch of the $\kblock=10$ circuits therefore has at least eight instances, and every table at least eight rows (\S\ref{sec:tables}). Lincheck runs $\kblock-\kskip$ rounds, from $4$ to $8$. + \subsection{Soundness analysis}\label{flock:soundness} @@ -267,11 +298,11 @@ \subsection{Soundness analysis}\label{flock:soundness} \] up to the PCS error and the $2^{-160}$ of Annex~\ref{annex:rs}, and let us show a $z$ failing \eqref{flock:eq:r1cs} or \eqref{flock:eq:const-one} is rejected. -Lincheck comes first, and it is the only thing that checks anything: the zerocheck's own terminal identity defines $v_c$ instead of testing it, so every lie upstream reaches lincheck as a wrong triple $(v_a,v_b,v_c)$. Its terminal check \eqref{flock:eq:lincheck-terminal} is evaluated from the true $s$ and the matrices themselves, by the walk of \S\ref{flock:native} natively and, in recursion, once \eqref{flock:eq:matpart} is discharged (\S\ref{sec:deferred-claims}), so a false \eqref{flock:eq:lincheck} survives the $8$ rounds with probability at most $16/|\E|$ (Fact~\ref{fact:sumcheck}, degree two in each round). And by Lemma~\ref{lem:flock-block-diagonal} and \eqref{flock:eq:c-as-column}, \eqref{flock:eq:lincheck} says +Lincheck comes first, and it is the only thing that checks anything: the zerocheck's own terminal identity defines $v_c$ instead of testing it, so every lie upstream reaches lincheck as a wrong triple $(v_a,v_b,v_c)$. Its terminal check \eqref{flock:eq:lincheck-terminal} is evaluated from the true $s$ and the matrices themselves, by the walk of \S\ref{flock:native}, so a false \eqref{flock:eq:lincheck} survives the $\kblock-\kskip$ rounds with probability at most $2(\kblock-\kskip)/|\E|$ (Fact~\ref{fact:sumcheck}, degree two in each round). And by Lemma~\ref{lem:flock-block-diagonal} and \eqref{flock:eq:c-as-column}, \eqref{flock:eq:lincheck} says \[ -\bigl(v_a+\qext a(\zskip,\fc)\bigr)+\lcalpha\bigl(v_b+\qext b(\zskip,\fc)\bigr)+\lcalpha^2\bigl(v_c+\qext z(\zskip,\fc)\bigr)+\lcalpha^3\bigl(1+\mle z(512,\fc_{\mathrm{out}})\bigr)=0, +\bigl(v_a+\qext a(\zskip,\fc)\bigr)+\lcalpha\bigl(v_b+\qext b(\zskip,\fc)\bigr)+\lcalpha^2\bigl(v_c+\qext z(\zskip,\fc)\bigr)+\lcalpha^3\bigl(1+\mle z(\jconst,\fc_{\mathrm{out}})\bigr)=0, \] -degree three in $\lcalpha$, all four coefficients fixed before $\lcalpha$ is drawn. Acceptance therefore pins them all, except with $3/|\E|$ (Lemma~\ref{lem:sz}): $v_a$, $v_b$ and $v_c$ are the true evaluations, and $\mle z(512,\fc_{\mathrm{out}})=1$, which is \eqref{flock:eq:const-one} except with $\kbatch/|\E|$ more, $\fc_{\mathrm{out}}$ being random. +degree three in $\lcalpha$, all four coefficients fixed before $\lcalpha$ is drawn. Acceptance therefore pins them all, except with $3/|\E|$ (Lemma~\ref{lem:sz}): $v_a$, $v_b$ and $v_c$ are the true evaluations, and $\mle z(\jconst,\fc_{\mathrm{out}})=1$, which is \eqref{flock:eq:const-one} except with $\kbatch/|\E|$ more, $\fc_{\mathrm{out}}$ being random. With all three pinned, the zerocheck's terminal identity \eqref{flock:eq:zc-terminal} becomes a real check again: it fixes $R_{\mathrm{zc}}$ to the true value of the summand at $\fc$. @@ -279,6 +310,6 @@ \subsection{Soundness analysis}\label{flock:soundness} Summing those terms, such a $z$ is accepted with probability at most \[ -\frac{16+3+\kbatch+(\nflock-7)+127+2\nflock}{|\E|}=\frac{4\kbatch+163}{|\E|}<2^{-183}, +\frac{2(\kblock-\kskip)+3+\kbatch+(\nflock-7)+127+2\nflock}{|\E|}=\frac{5\kblock+4\kbatch+93}{|\E|}<2^{-183}, \] on top of the two errors granted above. diff --git a/doc/leanvm/body/d-novel-basis.tex b/doc/leanvm/body/d-novel-basis.tex index 914d45d4d..956bb4bbc 100644 --- a/doc/leanvm/body/d-novel-basis.tex +++ b/doc/leanvm/body/d-novel-basis.tex @@ -1,7 +1,7 @@ % !TeX root = ../drafts/d-novel-basis.tex \section{Novel polynomial basis and additive NTT}\label{app:ntt} -This annex collects the algebra shared by the PCS (Annex~\ref{annex:pcs}) and LeanDA (\S\ref{sec:leanda}): additive domains, the novel polynomial basis, its additive FFT (called an NTT over a finite field), and codeword membership. The basis and transform follow~\cite{LCH14}, as presented in~\cite{DP24}. +This annex collects the algebra the PCS (Annex~\ref{annex:pcs}) builds on: additive domains, the novel polynomial basis, and its additive FFT (called an NTT over a finite field). The basis and transform follow~\cite{LCH14}, as presented in~\cite{DP24}. Fix the $\Ftwo$-basis $e_c:=x^c$, $0\le c<64$, of $\K$, the first $64$ vectors of the basis of $\E$ in Annex~\ref{annex:rs}. In particular $e_0=1$. For a bit vector $u$, write $\langle u\rangle:=\sum_i u_i2^i$ for its integer index, with bit $0$ first. @@ -115,61 +115,3 @@ \subsection{Column weights}\label{sec:ntt-weights} \] using $\eq(0, r_i) = 1 + r_i$ and $\eq(1, r_i) = r_i$. \end{proof} - -\subsection{Duality and codeword membership}\label{sec:ntt-duality} - -\begin{lemma}[Summation on an additive domain]\label{lem:domain-sum} -Let $\dom=\subsp_{\log_2\blen}$, with $\blen\ge2$ a power of two. The derivative $\vansub_{\log_2\blen}'$ is a nonzero constant. For every polynomial $h$ of degree less than $\blen$, -\begin{equation}\label{eq:da-sum} - \sum_{x\in\dom} h(x) - = \vansub_{\log_2\blen}'\,[X^{\blen-1}]\,h, -\end{equation} -where $[X^{\blen-1}]\,h$ denotes the coefficient of $X^{\blen-1}$ in $h$. In particular, the sum is zero if $\deg h\le\blen-2$. -\end{lemma} - -\begin{proof} -The vanishing polynomial $\vansub_{\log_2\blen}(X)=\prod_{x\in\dom}(X-x)$ is linearized (Lemma~\ref{lem:subspace}): its only powers of $X$ are $X,X^2,X^4,\ldots$. In characteristic two, the derivative of every term except the $X$ term is zero. Its derivative is therefore constant, equal to the coefficient of $X$. This constant is nonzero: at any root $x\in\dom$, differentiating the product gives $\vansub_{\log_2\blen}'(x)=\prod_{y\in\dom\setminus\{x\}}(x-y)$, a product of nonzero factors. - -Now let $h$ have degree less than $\blen$. Its values at the $\blen$ domain points determine it by Lagrange interpolation: -\[ - h(X)=\sum_{x\in\dom} - \frac{h(x)}{\vansub_{\log_2\blen}'(x)} - \frac{\vansub_{\log_2\blen}(X)}{X-x}. -\] -Each quotient $\vansub_{\log_2\blen}(X)/(X-x)$ is monic of degree $\blen-1$, so its coefficient of $X^{\blen-1}$ is $1$. Taking that coefficient on both sides, and using the fact that the derivative is the same constant at every $x$, gives -\[ - [X^{\blen-1}]\,h - =\sum_{x\in\dom}\frac{h(x)}{\vansub_{\log_2\blen}'(x)} - =\frac{1}{\vansub_{\log_2\blen}'}\sum_{x\in\dom}h(x). -\] -Multiplying by the nonzero derivative gives \eqref{eq:da-sum}. -\end{proof} - -\begin{corollary}[Dual code]\label{cor:rs-dual} -For $1\le\kdim<\blen$ on the domain of Lemma~\ref{lem:domain-sum}, -\[ - \RS[\E,\dom,\kdim]^\perp=\RS[\E,\dom,\blen-\kdim]. -\] -At rate $1/2$, the code equals its dual. -\end{corollary} - -\begin{proof} -If $f$ and $g$ have degrees less than $\kdim$ and $\blen-\kdim$, then $\deg(fg)\le\blen-2$. Lemma~\ref{lem:domain-sum} gives $\sum_{x\in\dom}f(x)g(x)=0$, so the two codes are orthogonal. Their dimensions sum to $\blen$, giving equality with the dual. -\end{proof} - -\begin{proposition}[Membership with tensor weights]\label{prop:rs-membership} -Let $\kdim=2^\kappa$, $\blen=2\kdim$, and $\dom=\subsp_{\kappa+1}$. For $z=(z_0,\ldots,z_{\kappa-1})\in\E^\kappa$, define -\begin{equation}\label{eq:dual-product} - \dual(X):=\prod_{j<\kappa}\bigl(1+z_j\Wn{j}(X)\bigr). -\end{equation} -Its evaluation vector is a codeword. For any fixed $w\in\E^\dom$, the inner product $\inner{\dual}{w}:=\sum_{x\in\dom}\dual(x)w(x)$ vanishes identically as a polynomial in $z$ if and only if $w$ is a codeword. If $w$ is not a codeword and the $z_j$ are independent and uniform, then -\[ - \Pr_z[\inner{\dual}{w}=0]\le\frac{\kappa}{|\E|}. -\] -\end{proposition} - -\begin{proof} -Since $\Wn{j}=\novel_{2^j}$, expanding \eqref{eq:dual-product} gives the novel-basis polynomials of index less than $\kdim$, with coefficient tensor $\bigotimes_j(1,z_j)$. Thus $\deg\dual<\kdim$, and Corollary~\ref{cor:rs-dual} makes it orthogonal to every codeword. - -Conversely, the coefficient of $\prod_j z_j^{u_j}$ in $\inner{\dual}{w}$ is $\inner{\novel_{\langle u\rangle}}{w}$. If the inner product vanished identically, $w$ would be orthogonal to every basis codeword, hence would itself belong to the code. For an invalid $w$, it is therefore a nonzero multilinear polynomial of total degree at most $\kappa$. The probability bound follows from Lemma~\ref{lem:sz}. -\end{proof} diff --git a/doc/leanvm/drafts/09-isa-programming.tex b/doc/leanvm/drafts/09-isa-programming.tex deleted file mode 100644 index 85fa863c6..000000000 --- a/doc/leanvm/drafts/09-isa-programming.tex +++ /dev/null @@ -1,19 +0,0 @@ -% GENERATED by make-drafts.sh. Do not edit; edit the section itself. -\documentclass[11pt]{article} - -\input{../preamble/packages} -\input{../preamble/macros} -\input{../preamble/theorems} - -\usepackage{xr} -\externaldocument{../.build/main} - -\pagestyle{empty} - -\begin{document} -\setcounter{section}{8} -\input{../body/09-isa-programming} - -\bibliographystyle{alphaurl} -\bibliography{refs} -\end{document} diff --git a/doc/leanvm/drafts/10-ethereum.tex b/doc/leanvm/drafts/10-ethereum.tex deleted file mode 100644 index 106e11b39..000000000 --- a/doc/leanvm/drafts/10-ethereum.tex +++ /dev/null @@ -1,19 +0,0 @@ -% GENERATED by make-drafts.sh. Do not edit; edit the section itself. -\documentclass[11pt]{article} - -\input{../preamble/packages} -\input{../preamble/macros} -\input{../preamble/theorems} - -\usepackage{xr} -\externaldocument{../.build/main} - -\pagestyle{empty} - -\begin{document} -\setcounter{section}{9} -\input{../body/10-ethereum} - -\bibliographystyle{alphaurl} -\bibliography{refs} -\end{document} diff --git a/doc/leanvm/main.tex b/doc/leanvm/main.tex index 87a758fbc..997a42164 100644 --- a/doc/leanvm/main.tex +++ b/doc/leanvm/main.tex @@ -28,8 +28,6 @@ \input{body/06-bus-interactions} \input{body/07-instruction-tables} \input{body/08-end-to-end-protocol} -\input{body/09-isa-programming} -\input{body/10-ethereum} \appendix \input{body/a-ring-switching} diff --git a/doc/leanvm/preamble/macros.tex b/doc/leanvm/preamble/macros.tex index 915bdf041..f0140c5a1 100644 --- a/doc/leanvm/preamble/macros.tex +++ b/doc/leanvm/preamble/macros.tex @@ -15,8 +15,8 @@ % ---------- fields ---------- \newcommand{\F}{\mathbb{F}} \newcommand{\Ftwo}{\mathbb{F}_2} -\newcommand{\K}{K} % F_{2^64}: addresses, counters, committed lanes -\newcommand{\E}{E} % F_{2^192} = K[y]/(y^3+y+1): words and challenges +\newcommand{\K}{K} % F_{2^64}: memory words, addresses, counters, committed columns +\newcommand{\E}{E} % F_{2^192} = K[y]/(y^3+y+1): challenges \newcommand{\gen}{\mathsf{g}} % generator of K^*, order 2^64 - 1 % ---------- checked relations ---------- @@ -55,17 +55,35 @@ \newcommand{\colmar}{M_{\mathrm{col}}} % column marginal of one matrix, A_0 or B_0 % ---------- VM ---------- -\newcommand{\loc}[1]{\mem[\fp\cdot #1]} -\newcommand{\dmode}{\mathsf{mode}} % a DEREF store mode +\newcommand{\intk}[1]{\underline{#1}} % the integer #1 as the element of K with those bits \newcommand{\pc}{\mathbf{pc}} -\newcommand{\fp}{\mathbf{fp}} -\newcommand{\nxt}[1]{\mathrm{next}(#1)} % the successor of a register: next(pc), next(fp) -\newcommand{\mem}{\mathbf{m}} +\newcommand{\mem}{\mathbf{m}} % RAM +\newcommand{\regs}{\mathbf{x}} % the register array: x0..x31, then the sink +\newcommand{\adv}{\mathbf{a}} % the advice region +\newcommand{\tbase}{\mathsf{TEXT}} % where the text sits +\newcommand{\rbase}{\mathsf{RAM}} % where RAM sits +\newcommand{\abase}{\mathsf{ADVICE}} % where the advice sits +\newcommand{\sink}{\mathsf{sink}} % the register cell an instruction with no destination writes +\newcommand{\kadv}{\kappa_{\mathrm{adv}}} % log advice length +\newcommand{\qcls}[1]{q_{\mathsf{#1}}} % the packed flock witness of one instruction class \newcommand{\addr}{\mathit{addr}} \newcommand{\val}{\mathit{val}} \newcommand{\cnt}{\mathit{count}} -\newcommand{\md}{\mathrm{md}} % BLAKE2s metadata: byte counter, final and last-node flags \newcommand{\cntfin}{\mathit{count}^{\mathrm{fin}}} % committed finalize-count column +\newcommand{\ts}{\mathbf{ts}} % the clock: g^{4 cycle}, the last component of the VM state +\newcommand{\tsfinal}{\ts_{\mathrm{final}}} % the clock the run ends on, announced by the prover +\newcommand{\tsprev}{\mathit{prev}} % the timestamp of a cell's previous access, a committed column +\newcommand{\tsfin}{\mathit{ts}^{\mathrm{fin}}} % committed column: each cell's last timestamp +\newcommand{\meminit}{\mem^{\mathrm{init}}} % RAM before the run, a public column +\newcommand{\memfin}{\mem^{\mathrm{fin}}} % committed RAM after the run +\newcommand{\regfin}{\regs^{\mathrm{fin}}} % committed registers after the run +\newcommand{\advinit}{\adv^{\mathrm{init}}} % committed advice before the run +\newcommand{\advfin}{\adv^{\mathrm{fin}}} % committed advice after the run +\newcommand{\genhi}{\mathsf{h}} % g^{2^16}: the step of the high range array +\newcommand{\glo}{\mathit{lo}} % low chunk of an access's gap, as the range-array address g^{d_lo + 1} +\newcommand{\ghi}{\mathit{hi}} % high chunk of the gap, as the range-array address h^{-d_hi} +\newcommand{\rlo}{\mathbf{R}_{\mathrm{lo}}} % the low range array +\newcommand{\rhi}{\mathbf{R}_{\mathrm{hi}}} % the high range array \newcommand{\rcnts}{\mathcal{N}} % multiset of read counts at one (address, value) \newcommand{\mset}[1]{\{\!\{#1\}\!\}} % a multiset (plain braces stay sets) \newcommand{\lut}{\mathbf{L}} % a lookup array (memory, or the program) @@ -75,9 +93,9 @@ \newcommand{\pull}{\textsc{pull}} \newcommand{\tup}[1]{( #1 )} \newcommand{\rdf}[1]{\dsep{read}\,\tup{#1}} % a pull/push pair differing only in the count +\newcommand{\acc}[1]{\dsep{access}_{#1}} % a memory access in clock slot #1: pull, push and the gap's two range reads \newcommand{\dsep}[1]{\mathsf{#1}} \newcommand{\opc}[1]{\mathsf{op}_{\mathsf{#1}}} -\newcommand{\srcsel}[1]{\mathsf{src}(#1)} \newcommand{\nprog}{N_{\mathrm{prog}}} % program length \newcommand{\nmem}{N_{\mathrm{mem}}} % memory length \newcommand{\kbc}{\kappa_{\mathrm{bc}}} % log program length: nprog = 2^kbc @@ -134,18 +152,6 @@ \newcommand{\bfarr}{\mathsf{B}} % NTT butterfly array (was B) \newcommand{\mpoly}{\mathcal{P}} % message polynomial (was P_u) -% ---------- data availability (\S\ref{sec:leanda}) ---------- -% The code itself reuses the PCS letters: k = message dimension, n = block -% length, rho = rate, D = domain, X_j = novel basis. -\newcommand{\nblob}{N_{\mathrm{blob}}} % rows of the payload matrix, one per blob -\newcommand{\cellsz}{N_{\mathrm{cell}}} % symbols in one cell, the sampling unit -\newcommand{\daroot}{\mathsf{root}} % the commitment -\newcommand{\darow}{\mathsf{root}_{\mathrm{row}}} % its repairability branch -\newcommand{\dacol}{\mathsf{root}_{\mathrm{col}}} % its sampling branch -\newcommand{\dual}{L} % the random dual codeword membership tests against -\newcommand{\davec}{\mathbf{L}} % membership vector in domain order -\newcommand{\davechash}{\mathsf{h}_L} % hash of the serialized membership vector - % ---------- ring switching (Annex A) ---------- \newcommand{\pair}{\Omega} % error pairing values (was V_k) \newcommand{\supp}{\mathcal{S}} % support of \Phimap (was S) diff --git a/doc/leanvm/refs.bib b/doc/leanvm/refs.bib index 60aca68be..77177fa98 100644 --- a/doc/leanvm/refs.bib +++ b/doc/leanvm/refs.bib @@ -129,43 +129,6 @@ @misc{GRS23 note = {Draft textbook}, } -@misc{stark, - author = {Eli Ben-Sasson and Iddo Bentov and Yinon Horesh and Michael Riabzev}, - title = {Scalable, Transparent, and Post-Quantum Secure Computational Integrity}, - year = {2018}, - howpublished = {Cryptology ePrint Archive, Paper 2018/046}, - url = {https://eprint.iacr.org/2018/046}, -} - -@misc{rfc8391, - author = {Andreas H{\"u}lsing and Denis Butin and Stefan-Lukas Gazdag and - Joost Rijneveld and Aziz Mohaisen}, - title = {{XMSS}: e{X}tended {M}erkle {S}ignature {S}cheme}, - year = {2018}, - howpublished = {RFC 8391}, - url = {https://datatracker.ietf.org/doc/html/rfc8391}, -} - -@inproceedings{sphincs, - author = {Daniel J. Bernstein and Andreas H{\"u}lsing and Stefan K{\"o}lbl and - Ruben Niederhagen and Joost Rijneveld and Peter Schwabe}, - title = {The {SPHINCS+} Signature Framework}, - booktitle = {ACM Conference on Computer and Communications Security (CCS 2019)}, - year = {2019}, - url = {https://sphincs.org/}, -} - -@article{shor, - author = {Peter W. Shor}, - title = {Polynomial-Time Algorithms for Prime Factorization and Discrete - Logarithms on a Quantum Computer}, - journal = {SIAM Journal on Computing}, - volume = {26}, - number = {5}, - pages = {1484--1509}, - year = {1997}, -} - @misc{m3, author = {Irreducible}, title = {Multi-Multiset Matching ({M3})}, @@ -200,19 +163,10 @@ @misc{cairo url = {https://eprint.iacr.org/2021/1063} } -@misc{LeanDA, - author = {Long Meng and Benedikt Wagner and George Kadianakis and Francesco Risitano}, - title = {{LeanDA}: Design and Benchmark}, - year = {2026}, - howpublished = {Ethereum Research}, - url = {https://ethresear.ch/t/leanda-design-and-benchmark/25642}, -} - -@misc{LeanDASecurity, - author = {Benedikt Wagner}, - title = {Formally Verified Security for {PQ-DAS} / {leanDA}}, - year = {2026}, - month = aug, - howpublished = {Ethereum Research}, - url = {https://ethresear.ch/t/formally-verified-security-for-pq-das-leanda/25746}, +@misc{riscv, + author = {{RISC-V International}}, + title = {The {RISC-V} Instruction Set Manual, Volume {I}: Unprivileged {ISA}}, + year = {2024}, + note = {Version 20240411}, + url = {https://riscv.org/specifications/} } diff --git a/doc/sphincs/latexmkrc b/doc/sphincs/latexmkrc deleted file mode 100644 index ec7c0e475..000000000 --- a/doc/sphincs/latexmkrc +++ /dev/null @@ -1,4 +0,0 @@ -$pdf_mode = 1; -$out_dir = '.build'; -$bibtex_use = 2; -$clean_ext = 'bbl synctex.gz'; diff --git a/doc/sphincs/main.tex b/doc/sphincs/main.tex deleted file mode 100644 index a44c305d1..000000000 --- a/doc/sphincs/main.tex +++ /dev/null @@ -1,471 +0,0 @@ -% Build with: latexmk -pdf main.tex -\documentclass[11pt]{article} - -\usepackage[T1]{fontenc} -\usepackage{lmodern} -\usepackage[margin=1in]{geometry} -\usepackage{microtype} -\usepackage{amsmath,amssymb,amsthm,mathtools} -\usepackage{booktabs} -\usepackage{enumitem} -\usepackage{xcolor} -\usepackage[colorlinks=true,linkcolor=blue!50!black,citecolor=blue!50!black,urlcolor=blue!50!black]{hyperref} - -\theoremstyle{plain} -\newtheorem{theorem}{Theorem}[section] -\theoremstyle{definition} -\newtheorem{definition}[theorem]{Definition} - -\newcommand{\bits}[1]{\{0,1\}^{#1}} -\newcommand{\getsr}{\stackrel{\$}{\gets}} -\newcommand{\Th}{\mathsf{Th}} -\newcommand{\Enc}{\mathsf{Enc}} -\newcommand{\Digest}{\mathsf{Digest}} -\newcommand{\Gen}{\mathsf{Gen}} -\newcommand{\Sig}{\mathsf{Sig}} -\newcommand{\Ver}{\mathsf{Ver}} -\newcommand{\hash}{\mathsf{H}} -\newcommand{\LE}{\mathsf{LE}} -\newcommand{\Truncate}{\mathsf{Truncate}} -\newcommand{\concat}{\mathbin\Vert} -\newcommand{\sk}{\mathit{sk}} -\newcommand{\pk}{\mathit{pk}} -\newcommand{\rootnode}{\mathit{root}} -\newcommand{\tw}{\mathit{tw}} -\newcommand{\qs}{q_{\mathrm{s}}} -\newcommand{\amax}{A_{\max}} -\newcommand{\cmax}{C_{\max}} -\newcommand{\idx}{\mathit{idx}} -\newcommand{\lay}{\mathit{lay}} -\newcommand{\OtsSign}{\mathsf{Ots.sign}} -\newcommand{\OtsLeaf}{\mathsf{Ots.leaf}} -\newcommand{\TreeRoot}{\mathsf{Tree.root}} -\newcommand{\TreePath}{\mathsf{Tree.path}} -\newcommand{\TreeFold}{\mathsf{Tree.fold}} -\newcommand{\FtsKey}{\mathsf{Fts.key}} -\newcommand{\FtsOpen}{\mathsf{Fts.open}} -\newcommand{\FtsRec}{\mathsf{Fts.recover}} - -\emergencystretch=1.5em -\setlist[enumerate]{leftmargin=2em, itemsep=4pt} - -\title{A SPHINCS$^+$ variant} -\author{} -\date{} - -\begin{document} -\maketitle - -\begin{abstract} - -This specification defines a SPHINCS$^+$ variant with a lifetime of $2^{24}$ signatures per key pair: - -\begin{itemize} - \item \textbf{NIST security level~1}~\cite{NISTPQC}, $\approx 128$-bit (resp. $\approx 64$-bit) of classical (resp. quantum) security, in the Random Oracle model, ROM (resp. Quantum Random Oracle model, QROM).\footnote{The classical bound is proved (Section~\ref{sec:security}); the quantum one is a target, not a proved statement (Section~\ref{sec:quantum}).} - \item \textbf{Public key: 32 bytes.} - \item \textbf{Signature: 4924 bytes.} - \item \textbf{Verification: 497 hashes.} - \item \textbf{Key generation: 1.38M hashes.} - \item \textbf{Signing: approximately 190K hashes on average}, with a 1024-byte public cache. -\end{itemize} -\end{abstract} - -The construction uses compressed variants of the Winternitz one-time signature (WOTS) and forest of random subsets (FORS), called WOTS$^+$C and FORS$^+$C~\cite{HK22C,KN25}. They are combined within the SPHINCS$^+$ framework~\cite{SPHINCSPLUS}. The definitions below specify the concrete variant completely. - -\section{Hashing and parameters} -\label{sec:parameters} - -Write $\bits r$ for $r$-bit strings, $\concat$ for concatenation, and $\LE_r(z)$ for the unsigned $r$-bit little-endian encoding of $z$. Write $x\getsr A$ for a uniform sample from $A$. Indices and bit positions start at zero; $\bot$ denotes failure. - -Let $\hash:\bits{*}\to\bits{256}$ be the hash function. Most operations use its first $n=128$ output bits: -\[ - \Th(P,\tw,M)=\Truncate_n\!\left(\hash(\tw\concat P\concat M)\right). -\] -Here $P$ is a 128-bit public parameter and $\tw$ is a 128-bit \emph{tweak}: an address identifying the operation and its position in the construction. Appendix~\ref{sec:tweaks} gives every tweak's exact bytes. $\Truncate_r$ always keeps the first $r$ bits, with bits read least significant first within each byte. - -The message $m$ and master secret $S$ are each 256 bits. WOTS signs 128-bit values $M$. The signature contains a 128-bit randomizer $\rho$ and one 32-bit counter $c$ per WOTS signature. Derived signing secrets, chain values and Merkle nodes are 128 bits; write $\mathcal H=\bits n$. - -\begin{center} -\begin{tabular}{@{}lll@{}} -\toprule -Symbol & Value & Meaning\\ -\midrule -$w$ & $3$ & bit-size of WOTS chain positions\\ -$v$ & $42$ & chains per WOTS key\\ -$T$ & $191$ & sum of the signed chain positions\\ -$d$ & $3$ & layers, numbered from the top\\ -$(h_0,h_1,h_2)$ & $(12,7,7)$ & tree heights at those layers\\ -$h$ & $26$ & total height, $h=h_0+h_1+h_2$\\ -$a$ & $10$ & height of each FORS tree\\ -$k$ & $15$ & digest indices, of which $k-1$ open trees\\ -$\qs$ & $2^{24}$ & signing requests allowed per key pair\\ -$\amax$ & $2^{32}$ & maximum randomizer trials per signature\\ -$\cmax$ & $2^{32}$ & maximum encoding attempts per WOTS signature\\ -\bottomrule -\end{tabular} -\end{center} - -The public parameter and all signing secrets are derived from $S$. In particular, -\[ - P=\Th\!\left(0^{128},\mathsf{tw}_{\mathrm{parameter}},S\right). -\] -The component definitions below derive signing secrets as needed from this fixed $S$. Functions that read secrets use $S$ implicitly. - -\section{WOTS: signing with hash chains} -\label{sec:ots} - -A WOTS key is identified by a layer $\lay$, a tree $\tau$ and a leaf $e$. It contains $v=42$ chains, each with eight positions, numbered $0$ through $7$. Fix such a key. For $0\leq i\lay}h_j}\right\rfloor\bmod 2^{h_\lay}. -\] -Concretely, $e_0$ is the high 12 bits of $\idx$, $e_1$ the next 7 bits, and $e_2$ the low 7 bits. Then -\[ - \tau_0=0,\qquad \tau_1=e_0,\qquad \tau_2=2^7e_0+e_1. -\] -Thus the key at $(0,0,e_0)$ signs tree $(1,\tau_1)$'s root, the key at $(1,\tau_1,e_1)$ signs tree $(2,\tau_2)$'s root, and the key at $(2,\tau_2,e_2)$ signs $\FtsKey(P,\idx)$. - -\paragraph{Selecting the route and FORS leaves.} -\label{rem:rho} -For public root $\rootnode$, message $m$ and randomizer $\rho$, define -\[ - \Digest(P,\rootnode,m,\rho)=\Truncate_{h+ka}\!\left( - \hash(\mathsf{tw}_{\mathrm{msg}}\concat P\concat\rho\concat\rootnode\concat m)\right). -\] -Interpret these $h+ka=176$ bits as a little-endian integer $N$ and extract -\[ - \idx=N\bmod2^h,\qquad - u_\kappa=\left\lfloor N/2^{h+\kappa a}\right\rfloor\bmod2^a, - \quad 0\leq\kappa