From f557c9b45aa76d4e75d410fb645e294a3aaf1e03 Mon Sep 17 00:00:00 2001 From: Yordan Boev Date: Sun, 13 Sep 2026 12:33:43 +0200 Subject: [PATCH] Make SIMD work by default in the interpreter The runtime gets a scalar v128 implementation (V128Ops), so every v128 instruction runs on every supported JDK without the incubator Vector API. All simd_*.wast spec tests run in the default build. The simd module is removed; the docs carry a migration note. Refs #181 --- AGENT.md | 1 - bom/pom.xml | 5 - docs/docs/advanced/simd.md | 52 +- docs/docs/index.md | 2 +- pom.xml | 9 - runtime-tests/pom.xml | 611 +--- .../testing/InterpreterMachineFactory.java | 13 - .../testing/InterpreterMachineFactory.java | 13 - .../run/endive/testing/BasicSimdTest.java | 168 + .../java/run/endive/testing/TestModule.java | 3 +- .../endive/runtime/InterpreterMachine.java | 20 +- .../run/endive/runtime/internal/V128Ops.java | 2655 +++++++++++++++ simd/pom.xml | 125 - .../run/endive/simd/VectorOperators.java | 28 - .../run/endive/simd/VectorOperators.java | 28 - simd/src/main/java/module-info.java | 5 - .../endive/simd/SimdInterpreterMachine.java | 2986 ----------------- .../java/run/endive/simd/BasicSimdTest.java | 68 - .../compiled/simd-memory-bounds.wat.wasm | Bin 0 -> 590 bytes .../compiled/simd-mixed-lanes.wat.wasm | Bin 0 -> 454 bytes .../main/resources/wat/simd-memory-bounds.wat | 26 + .../main/resources/wat/simd-mixed-lanes.wat | 27 + 22 files changed, 2957 insertions(+), 3888 deletions(-) delete mode 100644 runtime-tests/src/main/java-templates-21/run/endive/testing/InterpreterMachineFactory.java delete mode 100644 runtime-tests/src/main/java-templates/run/endive/testing/InterpreterMachineFactory.java create mode 100644 runtime-tests/src/test/java/run/endive/testing/BasicSimdTest.java create mode 100644 runtime/src/main/java/run/endive/runtime/internal/V128Ops.java delete mode 100644 simd/pom.xml delete mode 100644 simd/src/main/java-templates-21/run/endive/simd/VectorOperators.java delete mode 100644 simd/src/main/java-templates/run/endive/simd/VectorOperators.java delete mode 100644 simd/src/main/java/module-info.java delete mode 100644 simd/src/main/java/run/endive/simd/SimdInterpreterMachine.java delete mode 100644 simd/src/test/java/run/endive/simd/BasicSimdTest.java create mode 100644 wasm-corpus/src/main/resources/compiled/simd-memory-bounds.wat.wasm create mode 100644 wasm-corpus/src/main/resources/compiled/simd-mixed-lanes.wat.wasm create mode 100644 wasm-corpus/src/main/resources/wat/simd-memory-bounds.wat create mode 100644 wasm-corpus/src/main/resources/wat/simd-mixed-lanes.wat diff --git a/AGENT.md b/AGENT.md index f0557e221..ca458e57a 100644 --- a/AGENT.md +++ b/AGENT.md @@ -37,7 +37,6 @@ wasm (parser, validator, types) ├── wasi (WASI preview1) │ └── wasm-tools (wat2wasm, wast2json via WASI) ├── compiler (JVM bytecode compiler) - ├── simd (SIMD opcodes, pluggable machine) └── log ``` diff --git a/bom/pom.xml b/bom/pom.xml index afe9217ed..163dd44e3 100644 --- a/bom/pom.xml +++ b/bom/pom.xml @@ -103,11 +103,6 @@ runtime ${project.version} - - run.endive - simd - ${project.version} - run.endive wabt diff --git a/docs/docs/advanced/simd.md b/docs/docs/advanced/simd.md index 8a78fcdb6..c3387e4f3 100644 --- a/docs/docs/advanced/simd.md +++ b/docs/docs/advanced/simd.md @@ -4,53 +4,22 @@ sidebar_label: SIMD title: SIMD support --- -:::info[Availability] -SIMD support is available only for Java 21+ and interpreter mode. -::: - -If you are using a version of Java that supports [JEP 448 - Vector API](https://openjdk.org/jeps/448) you can leverage [Vector instructions](https://webassembly.github.io/spec/core/syntax/instructions.html#vector-instructions). - - - - - -After adding the dependency: - -```xml - - run.endive - simd - -``` - -You can instantiate a module with SIMD support by explicitly providing a `MachineFactory`: - -```java -import run.endive.simd.SimdInterpreterMachine; - -var module = Parser.parse(new File("your.wasm")); -var instance = Instance.builder(module).withMachineFactory(SimdInterpreterMachine::new).build(); -``` +All WebAssembly `v128` instructions use the scalar interpreter implementation on every supported +JDK. No extra dependency, JVM flag, or machine-factory configuration is required. :::warning SIMD support **REQUIRES** validation. Disabling validation (`WasmModule.builder().withValidation(false)`) is likely to produce incorrect results. ::: +### Migration + +The `run.endive:simd` module is gone. If your application declares it and uses +`SimdInterpreterMachine`, remove that dependency and the explicit +`withMachineFactory(SimdInterpreterMachine::new)` call; the default machine from +`run.endive:runtime` now executes `v128` instructions. + - diff --git a/docs/docs/index.md b/docs/docs/index.md index cee002ed6..388014756 100644 --- a/docs/docs/index.md +++ b/docs/docs/index.md @@ -5,7 +5,7 @@ title: Quick start --- :::info[Requirements] -Endive requires **Java 11** or later. SIMD support requires Java 21+. +Endive requires **Java 11** or later. SIMD instructions are supported on all supported Java versions. ::: ### Install the dependency diff --git a/pom.xml b/pom.xml index 0548c71da..736adb340 100644 --- a/pom.xml +++ b/pom.xml @@ -303,11 +303,6 @@ runtime ${project.version} - - run.endive - simd - ${project.version} - run.endive test-gen-lib @@ -678,7 +673,6 @@ ${maven.dependency.failOnWarning} true - run.endive:simd run.endive:wasm-corpus org.junit.jupiter:junit-jupiter-engine @@ -972,9 +966,6 @@ [21,) - - simd - diff --git a/runtime-tests/pom.xml b/runtime-tests/pom.xml index ebe2e10c3..07b36c269 100644 --- a/runtime-tests/pom.xml +++ b/runtime-tests/pom.xml @@ -18,10 +18,6 @@ - - run.endive - runtime - org.junit.jupiter junit-jupiter-api @@ -32,6 +28,11 @@ junit-jupiter-engine test + + run.endive + runtime + test + run.endive wasm @@ -52,18 +53,6 @@ - - org.codehaus.mojo - templating-maven-plugin - - - filter-src - - filter-sources - - - - run.endive test-gen-plugin @@ -274,6 +263,64 @@ ref_null.wast return.wast select.wast + simd_address.wast + simd_align.wast + simd_bit_shift.wast + simd_bitwise.wast + simd_boolean.wast + simd_const.wast + simd_conversions.wast + simd_f32x4.wast + simd_f32x4_arith.wast + simd_f32x4_cmp.wast + simd_f32x4_pmin_pmax.wast + simd_f32x4_rounding.wast + simd_f64x2.wast + simd_f64x2_arith.wast + simd_f64x2_cmp.wast + simd_f64x2_pmin_pmax.wast + simd_f64x2_rounding.wast + simd_i16x8_arith.wast + simd_i16x8_arith2.wast + simd_i16x8_cmp.wast + simd_i16x8_extadd_pairwise_i8x16.wast + simd_i16x8_extmul_i8x16.wast + simd_i16x8_q15mulr_sat_s.wast + simd_i16x8_sat_arith.wast + simd_i32x4_arith.wast + simd_i32x4_arith2.wast + simd_i32x4_cmp.wast + simd_i32x4_dot_i16x8.wast + simd_i32x4_extadd_pairwise_i16x8.wast + simd_i32x4_extmul_i16x8.wast + simd_i32x4_trunc_sat_f32x4.wast + simd_i32x4_trunc_sat_f64x2.wast + simd_i64x2_arith.wast + simd_i64x2_arith2.wast + simd_i64x2_cmp.wast + simd_i64x2_extmul_i32x4.wast + simd_i8x16_arith.wast + simd_i8x16_arith2.wast + simd_i8x16_cmp.wast + simd_i8x16_sat_arith.wast + simd_int_to_int_extend.wast + simd_lane.wast + simd_linking.wast + simd_load.wast + simd_load16_lane.wast + simd_load32_lane.wast + simd_load64_lane.wast + simd_load8_lane.wast + simd_load_extend.wast + simd_load_splat.wast + simd_load_zero.wast + simd_select.wast + simd_splat.wast + simd_store.wast + simd_store16_lane.wast + simd_store32_lane.wast + simd_store64_lane.wast + simd_store8_lane.wast skip-stack-guard-page.wast stack.wast start.wast @@ -401,64 +448,6 @@ obsolete-keywords.wast - simd_address.wast - simd_align.wast - simd_bit_shift.wast - simd_bitwise.wast - simd_boolean.wast - simd_const.wast - simd_conversions.wast - simd_f32x4.wast - simd_f32x4_arith.wast - simd_f32x4_cmp.wast - simd_f32x4_pmin_pmax.wast - simd_f32x4_rounding.wast - simd_f64x2.wast - simd_f64x2_arith.wast - simd_f64x2_cmp.wast - simd_f64x2_pmin_pmax.wast - simd_f64x2_rounding.wast - simd_i16x8_arith.wast - simd_i16x8_arith2.wast - simd_i16x8_cmp.wast - simd_i16x8_extadd_pairwise_i8x16.wast - simd_i16x8_extmul_i8x16.wast - simd_i16x8_q15mulr_sat_s.wast - simd_i16x8_sat_arith.wast - simd_i32x4_arith.wast - simd_i32x4_arith2.wast - simd_i32x4_cmp.wast - simd_i32x4_dot_i16x8.wast - simd_i32x4_extadd_pairwise_i16x8.wast - simd_i32x4_extmul_i16x8.wast - simd_i32x4_trunc_sat_f32x4.wast - simd_i32x4_trunc_sat_f64x2.wast - simd_i64x2_arith.wast - simd_i64x2_arith2.wast - simd_i64x2_cmp.wast - simd_i64x2_extmul_i32x4.wast - simd_i8x16_arith.wast - simd_i8x16_arith2.wast - simd_i8x16_cmp.wast - simd_i8x16_sat_arith.wast - simd_int_to_int_extend.wast - simd_lane.wast - simd_linking.wast - simd_load.wast - simd_load16_lane.wast - simd_load32_lane.wast - simd_load64_lane.wast - simd_load8_lane.wast - simd_load_extend.wast - simd_load_splat.wast - simd_load_zero.wast - simd_select.wast - simd_splat.wast - simd_store.wast - simd_store16_lane.wast - simd_store32_lane.wast - simd_store64_lane.wast - simd_store8_lane.wast @@ -472,478 +461,4 @@ - - - java21 - - [21,) - - - 21 - - - false - - - - run.endive - simd - - - - - - org.apache.maven.plugins - maven-compiler-plugin - - - --add-modules - jdk.incubator.vector - - - - - org.apache.maven.plugins - maven-javadoc-plugin - - - - false - false - -Xdoclint:none - - - - org.apache.maven.plugins - maven-surefire-plugin - - --add-modules=jdk.incubator.vector - - - - - org.codehaus.mojo - templating-maven-plugin - - - filter-src - - filter-sources - - - ${basedir}/src/main/java-templates-21 - - - - - - run.endive - test-gen-plugin - ${project.version} - - - address.wast - align.wast - binary-leb128.wast - binary.wast - block.wast - br.wast - br_if.wast - br_table.wast - bulk.wast - call.wast - call_indirect.wast - comments.wast - const.wast - conversions.wast - custom.wast - data.wast - elem.wast - endianness.wast - exports.wast - f32.wast - f32_bitwise.wast - f32_cmp.wast - f64.wast - f64_bitwise.wast - f64_cmp.wast - fac.wast - float_exprs.wast - float_literals.wast - float_memory.wast - float_misc.wast - forward.wast - func.wast - func_ptrs.wast - global.wast - i32.wast - i64.wast - if.wast - imports.wast - inline-module.wast - int_exprs.wast - int_literals.wast - labels.wast - left-to-right.wast - linking.wast - load.wast - local_get.wast - local_set.wast - local_tee.wast - loop.wast - memory.wast - memory_copy.wast - memory_fill.wast - memory_grow.wast - memory_init.wast - memory_redundancy.wast - memory_size.wast - memory_trap.wast - names.wast - nop.wast - proposals/exception-handling/binary.wast - proposals/exception-handling/exports.wast - proposals/exception-handling/imports.wast - proposals/exception-handling/ref_null.wast - proposals/exception-handling/tag.wast - proposals/exception-handling/throw.wast - proposals/exception-handling/throw_ref.wast - proposals/exception-handling/try_table.wast - proposals/extended-const/data.wast - proposals/extended-const/elem.wast - proposals/extended-const/global.wast - proposals/function-references/binary.wast - proposals/function-references/br_on_non_null.wast - proposals/function-references/br_on_null.wast - proposals/function-references/br_table.wast - proposals/function-references/call_ref.wast - proposals/function-references/data.wast - proposals/function-references/elem.wast - proposals/function-references/func.wast - proposals/function-references/global.wast - proposals/function-references/if.wast - proposals/function-references/linking.wast - proposals/function-references/local_get.wast - proposals/function-references/local_init.wast - proposals/function-references/ref.wast - proposals/function-references/ref_as_non_null.wast - proposals/function-references/ref_is_null.wast - proposals/function-references/ref_null.wast - proposals/function-references/return_call.wast - proposals/function-references/return_call_indirect.wast - proposals/function-references/return_call_ref.wast - proposals/function-references/select.wast - proposals/function-references/table-sub.wast - proposals/function-references/table.wast - proposals/function-references/type-equivalence.wast - proposals/function-references/unreached-invalid.wast - proposals/function-references/unreached-valid.wast - proposals/gc/array.wast - proposals/gc/array_copy.wast - proposals/gc/array_fill.wast - proposals/gc/array_init_data.wast - proposals/gc/array_init_elem.wast - proposals/gc/array_new_data.wast - proposals/gc/array_new_elem.wast - proposals/gc/binary-gc.wast - proposals/gc/binary.wast - proposals/gc/br_if.wast - proposals/gc/br_on_cast.wast - proposals/gc/br_on_cast_fail.wast - proposals/gc/br_on_non_null.wast - proposals/gc/br_on_null.wast - proposals/gc/br_table.wast - proposals/gc/call_ref.wast - proposals/gc/data.wast - proposals/gc/elem.wast - proposals/gc/extern.wast - proposals/gc/func.wast - proposals/gc/global.wast - proposals/gc/i31.wast - proposals/gc/if.wast - proposals/gc/linking.wast - proposals/gc/local_get.wast - proposals/gc/local_init.wast - proposals/gc/local_tee.wast - proposals/gc/ref.wast - proposals/gc/ref_as_non_null.wast - proposals/gc/ref_cast.wast - proposals/gc/ref_eq.wast - proposals/gc/ref_is_null.wast - proposals/gc/ref_null.wast - proposals/gc/ref_test.wast - proposals/gc/return_call.wast - proposals/gc/return_call_indirect.wast - proposals/gc/return_call_ref.wast - proposals/gc/select.wast - proposals/gc/struct.wast - proposals/gc/table-sub.wast - proposals/gc/table.wast - proposals/gc/type-canon.wast - proposals/gc/type-equivalence.wast - proposals/gc/type-rec.wast - proposals/gc/type-subtyping-invalid.wast - proposals/gc/type-subtyping.wast - proposals/gc/unreached-invalid.wast - proposals/gc/unreached-valid.wast - proposals/multi-memory/address0.wast - proposals/multi-memory/address1.wast - proposals/multi-memory/align.wast - proposals/multi-memory/align0.wast - proposals/multi-memory/binary.wast - proposals/multi-memory/binary0.wast - proposals/multi-memory/data.wast - proposals/multi-memory/data0.wast - proposals/multi-memory/data1.wast - proposals/multi-memory/data_drop0.wast - proposals/multi-memory/exports0.wast - proposals/multi-memory/float_exprs0.wast - proposals/multi-memory/float_exprs1.wast - proposals/multi-memory/float_memory0.wast - proposals/multi-memory/imports.wast - proposals/multi-memory/imports0.wast - proposals/multi-memory/imports1.wast - proposals/multi-memory/imports2.wast - proposals/multi-memory/imports3.wast - proposals/multi-memory/imports4.wast - proposals/multi-memory/linking0.wast - proposals/multi-memory/linking1.wast - proposals/multi-memory/linking2.wast - proposals/multi-memory/linking3.wast - proposals/multi-memory/load.wast - proposals/multi-memory/load0.wast - proposals/multi-memory/load1.wast - proposals/multi-memory/load2.wast - proposals/multi-memory/memory-multi.wast - proposals/multi-memory/memory.wast - proposals/multi-memory/memory_copy0.wast - proposals/multi-memory/memory_copy1.wast - proposals/multi-memory/memory_fill0.wast - proposals/multi-memory/memory_grow.wast - proposals/multi-memory/memory_init0.wast - proposals/multi-memory/memory_size.wast - proposals/multi-memory/memory_size0.wast - proposals/multi-memory/memory_size1.wast - proposals/multi-memory/memory_size2.wast - proposals/multi-memory/memory_size3.wast - proposals/multi-memory/memory_trap0.wast - proposals/multi-memory/memory_trap1.wast - proposals/multi-memory/start0.wast - proposals/multi-memory/store.wast - proposals/multi-memory/store0.wast - proposals/multi-memory/store1.wast - proposals/multi-memory/traps0.wast - proposals/tail-call/return_call.wast - proposals/tail-call/return_call_indirect.wast - proposals/threads/atomic.wast - proposals/threads/exports.wast - proposals/threads/imports.wast - proposals/threads/memory.wast - proposals/wasm-3.0/ref_null.wast - proposals/wasm-3.0/try_table.wast - ref_func.wast - ref_is_null.wast - ref_null.wast - return.wast - select.wast - simd_address.wast - simd_align.wast - simd_bit_shift.wast - simd_bitwise.wast - simd_boolean.wast - simd_const.wast - simd_conversions.wast - simd_f32x4.wast - simd_f32x4_arith.wast - simd_f32x4_cmp.wast - simd_f32x4_pmin_pmax.wast - simd_f32x4_rounding.wast - simd_f64x2.wast - simd_f64x2_arith.wast - simd_f64x2_cmp.wast - simd_f64x2_pmin_pmax.wast - simd_f64x2_rounding.wast - simd_i16x8_arith.wast - simd_i16x8_arith2.wast - simd_i16x8_cmp.wast - simd_i16x8_extadd_pairwise_i8x16.wast - simd_i16x8_extmul_i8x16.wast - simd_i16x8_q15mulr_sat_s.wast - simd_i16x8_sat_arith.wast - simd_i32x4_arith.wast - simd_i32x4_arith2.wast - simd_i32x4_cmp.wast - simd_i32x4_dot_i16x8.wast - simd_i32x4_extadd_pairwise_i16x8.wast - simd_i32x4_extmul_i16x8.wast - simd_i32x4_trunc_sat_f32x4.wast - simd_i32x4_trunc_sat_f64x2.wast - simd_i64x2_arith.wast - simd_i64x2_arith2.wast - simd_i64x2_cmp.wast - simd_i64x2_extmul_i32x4.wast - simd_i8x16_arith.wast - simd_i8x16_arith2.wast - simd_i8x16_cmp.wast - simd_i8x16_sat_arith.wast - simd_int_to_int_extend.wast - simd_lane.wast - simd_linking.wast - simd_load.wast - simd_load16_lane.wast - simd_load32_lane.wast - simd_load64_lane.wast - simd_load8_lane.wast - simd_load_extend.wast - simd_load_splat.wast - simd_load_zero.wast - simd_select.wast - simd_splat.wast - simd_store.wast - simd_store16_lane.wast - simd_store32_lane.wast - simd_store64_lane.wast - simd_store8_lane.wast - skip-stack-guard-page.wast - stack.wast - start.wast - store.wast - switch.wast - table-sub.wast - table.wast - table_copy.wast - table_fill.wast - table_get.wast - table_grow.wast - table_init.wast - table_set.wast - table_size.wast - token.wast - traps.wast - type.wast - unreachable.wast - unreached-invalid.wast - unreached-valid.wast - unwind.wast - utf8-custom-section-id.wast - utf8-import-field.wast - utf8-import-module.wast - utf8-invalid-encoding.wast - - - SpecV1BinaryTest.test40, - SpecV1EhBinaryTest.test40, - SpecV1ThreadsImportsTest.test64, - SpecV1ThreadsImportsTest.test65, - SpecV1ThreadsImportsTest.test66, - SpecV1FunctionReferencesTypeEquivalenceTest.test2, - SpecV1DataTest.test9, - SpecV1DataTest.test10, - SpecV1GlobalTest.test76, - SpecV1GlobalTest.test77, - SpecV1ExtendedConstDataTest.test9, - SpecV1ExtendedConstDataTest.test10, - SpecV1ExtendedConstGlobalTest.test80, - SpecV1ExtendedConstGlobalTest.test81, - SpecV1ElemTest.test14, - SpecV1ElemTest.test15, - SpecV1ElemTest.test67, - SpecV1ElemTest.test68, - SpecV1ExtendedConstElemTest.test14, - SpecV1ExtendedConstElemTest.test15, - SpecV1ExtendedConstElemTest.test76, - SpecV1ExtendedConstElemTest.test77, - SpecV1FunctionReferencesElemTest.test93, - SpecV1FunctionReferencesElemTest.test101, - SpecV1GcElemTest.test103, - - SpecV1MemoryTest.test6, - SpecV1MemoryTest.test7, - SpecV1ImportsTest.test127, - SpecV1ImportsTest.test128, - SpecV1ImportsTest.test129, - SpecV1ExceptionHandlingImportsTest.test133, - SpecV1ExceptionHandlingImportsTest.test134, - SpecV1ExceptionHandlingImportsTest.test135, - - SpecV1BinaryTest.test41, - SpecV1BinaryTest.test42, - SpecV1BinaryTest.test43, - SpecV1BinaryTest.test44, - SpecV1BinaryTest.test45, - SpecV1BinaryTest.test46, - SpecV1BinaryTest.test47, - SpecV1BinaryTest.test48, - SpecV1BinaryTest.test49, - SpecV1BinaryTest.test50, - SpecV1ExceptionHandlingBinaryTest.test41, - SpecV1ExceptionHandlingBinaryTest.test42, - SpecV1ExceptionHandlingBinaryTest.test43, - SpecV1ExceptionHandlingBinaryTest.test44, - SpecV1ExceptionHandlingBinaryTest.test45, - SpecV1ExceptionHandlingBinaryTest.test46, - SpecV1ExceptionHandlingBinaryTest.test47, - SpecV1ExceptionHandlingBinaryTest.test48, - SpecV1ExceptionHandlingBinaryTest.test49, - SpecV1ExceptionHandlingBinaryTest.test50, - SpecV1GcBinaryTest.test41, - SpecV1GcBinaryTest.test42, - SpecV1GcBinaryTest.test43, - SpecV1GcBinaryTest.test44, - SpecV1GcBinaryTest.test45, - SpecV1GcBinaryTest.test46, - SpecV1GcBinaryTest.test47, - SpecV1GcBinaryTest.test48, - SpecV1GcBinaryTest.test49, - SpecV1GcBinaryTest.test50, - SpecV1FunctionReferencesBinaryTest.test41, - SpecV1FunctionReferencesBinaryTest.test42, - SpecV1FunctionReferencesBinaryTest.test43, - SpecV1FunctionReferencesBinaryTest.test44, - SpecV1FunctionReferencesBinaryTest.test45, - SpecV1FunctionReferencesBinaryTest.test46, - SpecV1FunctionReferencesBinaryTest.test47, - SpecV1FunctionReferencesBinaryTest.test48, - SpecV1FunctionReferencesBinaryTest.test49, - SpecV1FunctionReferencesBinaryTest.test50, - - SpecV1AlignTest.test157, - SpecV1AlignTest.test158, - SpecV1AlignTest.test159, - SpecV1AlignTest.test160, - SpecV1AlignTest.test161, - - SpecV1ThreadsImportsTest.test98, - SpecV1ThreadsImportsTest.test99, - SpecV1ThreadsImportsTest.test100, - SpecV1ThreadsMemoryTest.test9, - SpecV1ThreadsMemoryTest.test10, - - SpecV1MultiMemoryBinaryTest.test82, - - SpecV1MultiMemoryDataTest.test9, - SpecV1MultiMemoryDataTest.test10 - - - - - - - - obsolete-keywords.wast - simd_select.wast - - - - - - wasm-test-gen - - - - - - - - diff --git a/runtime-tests/src/main/java-templates-21/run/endive/testing/InterpreterMachineFactory.java b/runtime-tests/src/main/java-templates-21/run/endive/testing/InterpreterMachineFactory.java deleted file mode 100644 index e7f92d83c..000000000 --- a/runtime-tests/src/main/java-templates-21/run/endive/testing/InterpreterMachineFactory.java +++ /dev/null @@ -1,13 +0,0 @@ -package run.endive.testing; - -import run.endive.runtime.Instance; -import run.endive.runtime.Machine; -import run.endive.simd.SimdInterpreterMachine; - -public class InterpreterMachineFactory { - - public static Machine create(Instance instance) { - return new SimdInterpreterMachine(instance); - } - -} diff --git a/runtime-tests/src/main/java-templates/run/endive/testing/InterpreterMachineFactory.java b/runtime-tests/src/main/java-templates/run/endive/testing/InterpreterMachineFactory.java deleted file mode 100644 index 05502b715..000000000 --- a/runtime-tests/src/main/java-templates/run/endive/testing/InterpreterMachineFactory.java +++ /dev/null @@ -1,13 +0,0 @@ -package run.endive.testing; - -import run.endive.runtime.Instance; -import run.endive.runtime.InterpreterMachine; -import run.endive.runtime.Machine; - -public class InterpreterMachineFactory { - - public static InterpreterMachine create(Instance instance) { - return new InterpreterMachine(instance); - } - -} diff --git a/runtime-tests/src/test/java/run/endive/testing/BasicSimdTest.java b/runtime-tests/src/test/java/run/endive/testing/BasicSimdTest.java new file mode 100644 index 000000000..2563f40b7 --- /dev/null +++ b/runtime-tests/src/test/java/run/endive/testing/BasicSimdTest.java @@ -0,0 +1,168 @@ +package run.endive.testing; + +import static org.junit.jupiter.api.Assertions.assertArrayEquals; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static run.endive.wasm.types.Value.i16ToVec; +import static run.endive.wasm.types.Value.i32ToVec; +import static run.endive.wasm.types.Value.i8ToVec; + +import java.util.Arrays; +import java.util.List; +import java.util.Map; +import org.junit.jupiter.api.Test; +import run.endive.corpus.CorpusResources; +import run.endive.runtime.Instance; +import run.endive.runtime.InterpreterMachine; +import run.endive.runtime.WasmRuntimeException; +import run.endive.wasm.Parser; + +public class BasicSimdTest { + + private static Instance instance(String wasm) { + return Instance.builder(Parser.parse(CorpusResources.getResource(wasm))) + .withMachineFactory(InterpreterMachine::new) + .build(); + } + + @Test + public void shouldRunBasicExample() { + // from: https://blog.dkwr.de/development/wasm-simd-operations/ + var instance = instance("compiled/simd-example.wat.wasm"); + assertEquals(6L, instance.export("main").apply()[0]); + } + + @Test + public void shouldRoundTripV128Locals() { + var instance = instance("compiled/simd-locals.wat.wasm"); + assertEquals(10L, instance.export("local_roundtrip").apply()[0]); + assertEquals(7L, instance.export("local_roundtrip_lane0").apply()[0]); + assertEquals(10L, instance.export("local_tee").apply()[0]); + assertEquals(7L, instance.export("local_tee_get").apply()[0]); + } + + @Test + public void shouldTrapOnStoreEffectiveAddressOverflow() { + var instance = instance("compiled/simd-store-offset-wrap.wat.wasm"); + var memory = instance.memory(); + + for (var name : List.of("v128_store", "v128_store8_lane")) { + var store = instance.export(name); + assertThrows(WasmRuntimeException.class, () -> store.apply(-1L), name); + assertThrows(WasmRuntimeException.class, () -> store.apply(0xFFFFFFFFL), name); + } + for (var name : List.of("v128_store_max_offset", "v128_store8_lane_max_offset")) { + var store = instance.export(name); + assertThrows(WasmRuntimeException.class, () -> store.apply(1L), name); + } + assertArrayEquals(new byte[16], memory.readBytes(0, 16)); + } + + @Test + public void shouldPairLanesWithDistinctValues() { + var instance = instance("compiled/simd-mixed-lanes.wat.wasm"); + + assertArrayEquals( + i32ToVec(new long[] {50, 250, 610, -150}), + apply( + instance, + "i32x4.dot_i16x8_s", + i16ToVec(new long[] {1, 2, 3, 4, 5, 6, 7, -8}), + i16ToVec(new long[] {10, 20, 30, 40, 50, 60, 70, 80}))); + + var bytes = i8ToVec(new long[] {1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, -1, -128}); + assertArrayEquals( + i16ToVec(new long[] {3, 7, 11, 15, 19, 23, 27, -129}), + apply(instance, "i16x8.extadd_pairwise_i8x16_s", bytes)); + assertArrayEquals( + i16ToVec(new long[] {3, 7, 11, 15, 19, 23, 27, 383}), + apply(instance, "i16x8.extadd_pairwise_i8x16_u", bytes)); + + var shorts = i16ToVec(new long[] {1, 2, 3, 4, 5, 6, -1, -32768}); + assertArrayEquals( + i32ToVec(new long[] {3, 7, 11, -32769}), + apply(instance, "i32x4.extadd_pairwise_i16x8_s", shorts)); + assertArrayEquals( + i32ToVec(new long[] {3, 7, 11, 98303}), + apply(instance, "i32x4.extadd_pairwise_i16x8_u", shorts)); + } + + @Test + public void shouldSelectExtmulHalvesWithDistinctValues() { + var instance = instance("compiled/simd-mixed-lanes.wat.wasm"); + + var bytesA = i8ToVec(new long[] {1, 2, 3, 4, 5, 6, 7, -8, 9, 10, 11, 12, 13, 14, 15, -16}); + var bytesB = i8ToVec(new long[] {2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17}); + assertArrayEquals( + i16ToVec(new long[] {2, 6, 12, 20, 30, 42, 56, -72}), + apply(instance, "i16x8.extmul_low_i8x16_s", bytesA, bytesB)); + assertArrayEquals( + i16ToVec(new long[] {90, 110, 132, 156, 182, 210, 240, 4080}), + apply(instance, "i16x8.extmul_high_i8x16_u", bytesA, bytesB)); + + var shortsA = i16ToVec(new long[] {1, 2, 3, -4, 5, 6, 7, -8}); + var shortsB = i16ToVec(new long[] {10, 20, 30, 40, 50, 60, 70, 80}); + assertArrayEquals( + i32ToVec(new long[] {10, 40, 90, 2621280}), + apply(instance, "i32x4.extmul_low_i16x8_u", shortsA, shortsB)); + assertArrayEquals( + i32ToVec(new long[] {250, 360, 490, -640}), + apply(instance, "i32x4.extmul_high_i16x8_s", shortsA, shortsB)); + + var intsA = i32ToVec(new long[] {3, -4, 5, -6}); + var intsB = i32ToVec(new long[] {7, 8, 9, 10}); + assertArrayEquals( + new long[] {21, -32}, apply(instance, "i64x2.extmul_low_i32x4_s", intsA, intsB)); + assertArrayEquals( + new long[] {45, 42949672900L}, + apply(instance, "i64x2.extmul_high_i32x4_u", intsA, intsB)); + } + + @Test + public void shouldBoundsCheckWholeAccessAtEndOfMemory() { + var instance = instance("compiled/simd-memory-bounds.wat.wasm"); + var memory = instance.memory(); + int memorySize = 65536; + // export suffix -> access width in bytes, narrowest first + var accesses = + List.of( + Map.entry("8_lane", 1), + Map.entry("16_lane", 2), + Map.entry("32_lane", 4), + Map.entry("64_lane", 8), + Map.entry("", 16)); + + for (var access : accesses) { + long pastEnd = memorySize - access.getValue() + 1L; + for (var name : + List.of("v128.load" + access.getKey(), "v128.store" + access.getKey())) { + var function = instance.export(name); + assertThrows(WasmRuntimeException.class, () -> function.apply(pastEnd), name); + } + } + assertArrayEquals(new byte[16], memory.readBytes(memorySize - 16, 16)); + + for (var access : accesses) { + int width = access.getValue(); + int lastValid = memorySize - width; + instance.export("v128.store" + access.getKey()).apply(lastValid); + + var tail = new byte[16]; + Arrays.fill(tail, 16 - width, 16, (byte) -1); + assertArrayEquals(tail, memory.readBytes(memorySize - 16, 16), access.getKey()); + long lane0 = width >= 8 ? -1L : (1L << (8 * width)) - 1; + assertEquals( + lane0, + instance.export("v128.load" + access.getKey()).apply(lastValid)[0], + access.getKey()); + } + } + + private static long[] apply(Instance instance, String name, long[]... vectors) { + var args = ArgsAdapter.builder(); + for (var vector : vectors) { + args.add(vector); + } + return instance.export(name).apply(args.build()); + } +} diff --git a/runtime-tests/src/test/java/run/endive/testing/TestModule.java b/runtime-tests/src/test/java/run/endive/testing/TestModule.java index a77a0b071..07b4550e6 100644 --- a/runtime-tests/src/test/java/run/endive/testing/TestModule.java +++ b/runtime-tests/src/test/java/run/endive/testing/TestModule.java @@ -4,6 +4,7 @@ import run.endive.runtime.ByteArrayMemory; import run.endive.runtime.ImportValues; import run.endive.runtime.Instance; +import run.endive.runtime.InterpreterMachine; import run.endive.runtime.Store; import run.endive.tools.wasm.Wat2Wasm; import run.endive.wasm.MalformedException; @@ -73,7 +74,7 @@ public Instance instantiate(Store s) { ImportValues importValues = s.toImportValues(); return Instance.builder(module) .withImportValues(importValues) - .withMachineFactory(InterpreterMachineFactory::create) + .withMachineFactory(InterpreterMachine::new) .withMemoryFactory(ByteArrayMemory::new) .build(); } diff --git a/runtime/src/main/java/run/endive/runtime/InterpreterMachine.java b/runtime/src/main/java/run/endive/runtime/InterpreterMachine.java index 3244e8496..62d19fc77 100644 --- a/runtime/src/main/java/run/endive/runtime/InterpreterMachine.java +++ b/runtime/src/main/java/run/endive/runtime/InterpreterMachine.java @@ -7,13 +7,13 @@ import java.util.ArrayDeque; import java.util.Deque; import java.util.List; +import run.endive.runtime.internal.V128Ops; import run.endive.wasm.InvalidException; import run.endive.wasm.WasmEngineException; import run.endive.wasm.types.AnnotatedInstruction; import run.endive.wasm.types.BlockType; import run.endive.wasm.types.CatchOpCode; import run.endive.wasm.types.FunctionType; -import run.endive.wasm.types.Instruction; import run.endive.wasm.types.OpCode; import run.endive.wasm.types.TypeSection; import run.endive.wasm.types.ValType; @@ -50,17 +50,6 @@ protected interface Operands { long get(int index); } - @SuppressWarnings("DoNotCallSuggester") - protected void evalDefault( - MStack stack, - Instance instance, - Deque callStack, - Instruction instruction, - Operands operands) - throws WasmEngineException { - throw new WasmEngineException("Machine doesn't recognize Instruction " + instruction); - } - @Override public long[] call(int funcId, long[] args) throws WasmEngineException { return call(stack, instance, callStack, funcId, args, null, null, true); @@ -1195,10 +1184,11 @@ protected void eval(MStack stack, Instance instance, Deque callStack EXTERN_CONVERT_ANY(stack); break; default: - { - evalDefault(stack, instance, callStack, instruction, operands); - break; + if (!V128Ops.eval(stack, instance, instruction)) { + throw new WasmEngineException( + "Machine doesn't recognize Instruction " + instruction); } + break; } } } diff --git a/runtime/src/main/java/run/endive/runtime/internal/V128Ops.java b/runtime/src/main/java/run/endive/runtime/internal/V128Ops.java new file mode 100644 index 000000000..a8295cf95 --- /dev/null +++ b/runtime/src/main/java/run/endive/runtime/internal/V128Ops.java @@ -0,0 +1,2655 @@ +package run.endive.runtime.internal; + +import run.endive.runtime.BitOps; +import run.endive.runtime.Instance; +import run.endive.runtime.MStack; +import run.endive.runtime.OpcodeImpl; +import run.endive.runtime.WasmRuntimeException; +import run.endive.wasm.types.Instruction; + +/** + * Scalar implementation of the WebAssembly SIMD instruction set. + * + *

Vectors use two stack longs: the low word is pushed first, and lane 0 occupies the low bits + * of the low word. Lane memory operands are offset = operand(1), memory index = operand(2), and + * lane index = operand(3); extract/replace use operand(0). Float abs/neg/pmin/pmax preserve + * operand bits, while float arithmetic canonicalizes NaN results. + */ +public final class V128Ops { + // binaryInt / binaryFloat + private static final int ADD = 1; + private static final int SUB = 2; + private static final int MUL = 3; + private static final int MIN_S = 4; + private static final int MIN_U = 5; + private static final int MAX_S = 6; + private static final int MAX_U = 7; + private static final int EQ = 8; + private static final int NE = 9; + private static final int LT_S = 10; + private static final int LT_U = 11; + private static final int GT_S = 12; + private static final int GT_U = 13; + private static final int LE_S = 14; + private static final int LE_U = 15; + private static final int GE_S = 16; + private static final int GE_U = 17; + private static final int ADD_SAT_S = 18; + private static final int ADD_SAT_U = 19; + private static final int SUB_SAT_S = 20; + private static final int SUB_SAT_U = 21; + private static final int AVGR_U = 22; + private static final int Q15MULR_SAT_S = 23; + private static final int DIV = 24; + private static final int PMIN = 25; + private static final int PMAX = 26; + private static final int MIN = 27; + private static final int MAX = 28; + private static final int LT = 29; + private static final int LE = 30; + private static final int GT = 31; + private static final int GE = 32; + + // bitwise + private static final int AND = 100; + private static final int ANDNOT = 101; + private static final int OR = 102; + private static final int XOR = 103; + + // shift + private static final int SHL = 200; + private static final int SHR_S = 201; + private static final int SHR_U = 202; + + // unaryInt / unaryFloat + private static final int ABS = 300; + private static final int NEG = 301; + private static final int POPCNT = 302; + private static final int SQRT = 303; + private static final int CEIL = 304; + private static final int FLOOR = 305; + private static final int TRUNC = 306; + private static final int NEAREST = 307; + + private V128Ops() {} + + public static boolean eval(MStack stack, Instance instance, Instruction instruction) { + switch (instruction.opcode()) { + case V128_LOAD: + V128_LOAD(stack, instance, instruction); + break; + case V128_LOAD8x8_S: + V128_LOAD8X8_S(stack, instance, instruction); + break; + case V128_LOAD8x8_U: + V128_LOAD8X8_U(stack, instance, instruction); + break; + case V128_LOAD16x4_S: + V128_LOAD16X4_S(stack, instance, instruction); + break; + case V128_LOAD16x4_U: + V128_LOAD16X4_U(stack, instance, instruction); + break; + case V128_LOAD32x2_S: + V128_LOAD32X2_S(stack, instance, instruction); + break; + case V128_LOAD32x2_U: + V128_LOAD32X2_U(stack, instance, instruction); + break; + case V128_LOAD8_SPLAT: + V128_LOAD8_SPLAT(stack, instance, instruction); + break; + case V128_LOAD16_SPLAT: + V128_LOAD16_SPLAT(stack, instance, instruction); + break; + case V128_LOAD32_SPLAT: + V128_LOAD32_SPLAT(stack, instance, instruction); + break; + case V128_LOAD64_SPLAT: + V128_LOAD64_SPLAT(stack, instance, instruction); + break; + case V128_STORE: + V128_STORE(stack, instance, instruction); + break; + case V128_CONST: + V128_CONST(stack, instruction); + break; + case I8x16_SHUFFLE: + I8X16_SHUFFLE(stack, instruction); + break; + case I8x16_SWIZZLE: + I8X16_SWIZZLE(stack); + break; + case I8x16_SPLAT: + I8X16_SPLAT(stack); + break; + case I16x8_SPLAT: + I16X8_SPLAT(stack); + break; + case I32x4_SPLAT: + I32X4_SPLAT(stack); + break; + case I64x2_SPLAT: + I64X2_SPLAT(stack); + break; + case F32x4_SPLAT: + F32X4_SPLAT(stack); + break; + case F64x2_SPLAT: + F64X2_SPLAT(stack); + break; + case I8x16_EXTRACT_LANE_S: + I8X16_EXTRACT_LANE_S(stack, instruction); + break; + case I8x16_EXTRACT_LANE_U: + I8X16_EXTRACT_LANE_U(stack, instruction); + break; + case I8x16_REPLACE_LANE: + I8X16_REPLACE_LANE(stack, instruction); + break; + case I16x8_EXTRACT_LANE_S: + I16X8_EXTRACT_LANE_S(stack, instruction); + break; + case I16x8_EXTRACT_LANE_U: + I16X8_EXTRACT_LANE_U(stack, instruction); + break; + case I16x8_REPLACE_LANE: + I16X8_REPLACE_LANE(stack, instruction); + break; + case I32x4_EXTRACT_LANE: + I32X4_EXTRACT_LANE(stack, instruction); + break; + case I32x4_REPLACE_LANE: + I32X4_REPLACE_LANE(stack, instruction); + break; + case I64x2_EXTRACT_LANE: + I64X2_EXTRACT_LANE(stack, instruction); + break; + case I64x2_REPLACE_LANE: + I64X2_REPLACE_LANE(stack, instruction); + break; + case F32x4_EXTRACT_LANE: + F32X4_EXTRACT_LANE(stack, instruction); + break; + case F32x4_REPLACE_LANE: + F32X4_REPLACE_LANE(stack, instruction); + break; + case F64x2_EXTRACT_LANE: + F64X2_EXTRACT_LANE(stack, instruction); + break; + case F64x2_REPLACE_LANE: + F64X2_REPLACE_LANE(stack, instruction); + break; + case I8x16_EQ: + I8X16_EQ(stack); + break; + case I8x16_NE: + I8X16_NE(stack); + break; + case I8x16_LT_S: + I8X16_LT_S(stack); + break; + case I8x16_LT_U: + I8X16_LT_U(stack); + break; + case I8x16_GT_S: + I8X16_GT_S(stack); + break; + case I8x16_GT_U: + I8X16_GT_U(stack); + break; + case I8x16_LE_S: + I8X16_LE_S(stack); + break; + case I8x16_LE_U: + I8X16_LE_U(stack); + break; + case I8x16_GE_S: + I8X16_GE_S(stack); + break; + case I8x16_GE_U: + I8X16_GE_U(stack); + break; + case I16x8_EQ: + I16X8_EQ(stack); + break; + case I16x8_NE: + I16X8_NE(stack); + break; + case I16x8_LT_S: + I16X8_LT_S(stack); + break; + case I16x8_LT_U: + I16X8_LT_U(stack); + break; + case I16x8_GT_S: + I16X8_GT_S(stack); + break; + case I16x8_GT_U: + I16X8_GT_U(stack); + break; + case I16x8_LE_S: + I16X8_LE_S(stack); + break; + case I16x8_LE_U: + I16X8_LE_U(stack); + break; + case I16x8_GE_S: + I16X8_GE_S(stack); + break; + case I16x8_GE_U: + I16X8_GE_U(stack); + break; + case I32x4_EQ: + I32X4_EQ(stack); + break; + case I32x4_NE: + I32X4_NE(stack); + break; + case I32x4_LT_S: + I32X4_LT_S(stack); + break; + case I32x4_LT_U: + I32X4_LT_U(stack); + break; + case I32x4_GT_S: + I32X4_GT_S(stack); + break; + case I32x4_GT_U: + I32X4_GT_U(stack); + break; + case I32x4_LE_S: + I32X4_LE_S(stack); + break; + case I32x4_LE_U: + I32X4_LE_U(stack); + break; + case I32x4_GE_S: + I32X4_GE_S(stack); + break; + case I32x4_GE_U: + I32X4_GE_U(stack); + break; + case F32x4_EQ: + F32X4_EQ(stack); + break; + case F32x4_NE: + F32X4_NE(stack); + break; + case F32x4_LT: + F32X4_LT(stack); + break; + case F32x4_GT: + F32X4_GT(stack); + break; + case F32x4_LE: + F32X4_LE(stack); + break; + case F32x4_GE: + F32X4_GE(stack); + break; + case F64x2_EQ: + F64X2_EQ(stack); + break; + case F64x2_NE: + F64X2_NE(stack); + break; + case F64x2_LT: + F64X2_LT(stack); + break; + case F64x2_GT: + F64X2_GT(stack); + break; + case F64x2_LE: + F64X2_LE(stack); + break; + case F64x2_GE: + F64X2_GE(stack); + break; + case V128_NOT: + V128_NOT(stack); + break; + case V128_AND: + V128_AND(stack); + break; + case V128_ANDNOT: + V128_ANDNOT(stack); + break; + case V128_OR: + V128_OR(stack); + break; + case V128_XOR: + V128_XOR(stack); + break; + case V128_BITSELECT: + V128_BITSELECT(stack); + break; + case V128_ANY_TRUE: + V128_ANY_TRUE(stack); + break; + case V128_LOAD8_LANE: + V128_LOAD8_LANE(stack, instance, instruction); + break; + case V128_LOAD16_LANE: + V128_LOAD16_LANE(stack, instance, instruction); + break; + case V128_LOAD32_LANE: + V128_LOAD32_LANE(stack, instance, instruction); + break; + case V128_LOAD64_LANE: + V128_LOAD64_LANE(stack, instance, instruction); + break; + case V128_STORE8_LANE: + V128_STORE8_LANE(stack, instance, instruction); + break; + case V128_STORE16_LANE: + V128_STORE16_LANE(stack, instance, instruction); + break; + case V128_STORE32_LANE: + V128_STORE32_LANE(stack, instance, instruction); + break; + case V128_STORE64_LANE: + V128_STORE64_LANE(stack, instance, instruction); + break; + case V128_LOAD32_ZERO: + V128_LOAD32_ZERO(stack, instance, instruction); + break; + case V128_LOAD64_ZERO: + V128_LOAD64_ZERO(stack, instance, instruction); + break; + case F32x4_DEMOTE_LOW_F64x2_ZERO: + F32X4_DEMOTE_LOW_F64X2_ZERO(stack); + break; + case F64x2_PROMOTE_LOW_F32x4: + F64X2_PROMOTE_LOW_F32X4(stack); + break; + case I8x16_ABS: + I8X16_ABS(stack); + break; + case I8x16_NEG: + I8X16_NEG(stack); + break; + case I8x16_POPCNT: + I8X16_POPCNT(stack); + break; + case I8x16_ALL_TRUE: + I8X16_ALL_TRUE(stack); + break; + case I8x16_BITMASK: + I8X16_BITMASK(stack); + break; + case I8x16_NARROW_I16x8_S: + I8X16_NARROW_I16X8_S(stack); + break; + case I8x16_NARROW_I16x8_U: + I8X16_NARROW_I16X8_U(stack); + break; + case F32x4_CEIL: + F32X4_CEIL(stack); + break; + case F32x4_FLOOR: + F32X4_FLOOR(stack); + break; + case F32x4_TRUNC: + F32X4_TRUNC(stack); + break; + case F32x4_NEAREST: + F32X4_NEAREST(stack); + break; + case I8x16_SHL: + I8X16_SHL(stack); + break; + case I8x16_SHR_S: + I8X16_SHR_S(stack); + break; + case I8x16_SHR_U: + I8X16_SHR_U(stack); + break; + case I8x16_ADD: + I8X16_ADD(stack); + break; + case I8x16_ADD_SAT_S: + I8X16_ADD_SAT_S(stack); + break; + case I8x16_ADD_SAT_U: + I8X16_ADD_SAT_U(stack); + break; + case I8x16_SUB: + I8X16_SUB(stack); + break; + case I8x16_SUB_SAT_S: + I8X16_SUB_SAT_S(stack); + break; + case I8x16_SUB_SAT_U: + I8X16_SUB_SAT_U(stack); + break; + case F64x2_CEIL: + F64X2_CEIL(stack); + break; + case F64x2_FLOOR: + F64X2_FLOOR(stack); + break; + case I8x16_MIN_S: + I8X16_MIN_S(stack); + break; + case I8x16_MIN_U: + I8X16_MIN_U(stack); + break; + case I8x16_MAX_S: + I8X16_MAX_S(stack); + break; + case I8x16_MAX_U: + I8X16_MAX_U(stack); + break; + case F64x2_TRUNC: + F64X2_TRUNC(stack); + break; + case I8x16_AVGR_U: + I8X16_AVGR_U(stack); + break; + case I16x8_EXTADD_PAIRWISE_I8x16_S: + I16X8_EXTADD_PAIRWISE_I8X16_S(stack); + break; + case I16x8_EXTADD_PAIRWISE_I8x16_U: + I16X8_EXTADD_PAIRWISE_I8X16_U(stack); + break; + case I32x4_EXTADD_PAIRWISE_I16x8_S: + I32X4_EXTADD_PAIRWISE_I16X8_S(stack); + break; + case I32x4_EXTADD_PAIRWISE_I16x8_U: + I32X4_EXTADD_PAIRWISE_I16X8_U(stack); + break; + case I16x8_ABS: + I16X8_ABS(stack); + break; + case I16x8_NEG: + I16X8_NEG(stack); + break; + case I16x8_Q15MULR_SAT_S: + I16X8_Q15MULR_SAT_S(stack); + break; + case I16x8_ALL_TRUE: + I16X8_ALL_TRUE(stack); + break; + case I16x8_BITMASK: + I16X8_BITMASK(stack); + break; + case I16x8_NARROW_I32x4_S: + I16X8_NARROW_I32X4_S(stack); + break; + case I16x8_NARROW_I32x4_U: + I16X8_NARROW_I32X4_U(stack); + break; + case I16x8_EXTEND_LOW_I8x16_S: + I16X8_EXTEND_LOW_I8X16_S(stack); + break; + case I16x8_EXTEND_HIGH_I8x16_S: + I16X8_EXTEND_HIGH_I8X16_S(stack); + break; + case I16x8_EXTEND_LOW_I8x16_U: + I16X8_EXTEND_LOW_I8X16_U(stack); + break; + case I16x8_EXTEND_HIGH_I8x16_U: + I16X8_EXTEND_HIGH_I8X16_U(stack); + break; + case I16x8_SHL: + I16X8_SHL(stack); + break; + case I16x8_SHR_S: + I16X8_SHR_S(stack); + break; + case I16x8_SHR_U: + I16X8_SHR_U(stack); + break; + case I16x8_ADD: + I16X8_ADD(stack); + break; + case I16x8_ADD_SAT_S: + I16X8_ADD_SAT_S(stack); + break; + case I16x8_ADD_SAT_U: + I16X8_ADD_SAT_U(stack); + break; + case I16x8_SUB: + I16X8_SUB(stack); + break; + case I16x8_SUB_SAT_S: + I16X8_SUB_SAT_S(stack); + break; + case I16x8_SUB_SAT_U: + I16X8_SUB_SAT_U(stack); + break; + case F64x2_NEAREST: + F64X2_NEAREST(stack); + break; + case I16x8_MUL: + I16X8_MUL(stack); + break; + case I16x8_MIN_S: + I16X8_MIN_S(stack); + break; + case I16x8_MIN_U: + I16X8_MIN_U(stack); + break; + case I16x8_MAX_S: + I16X8_MAX_S(stack); + break; + case I16x8_MAX_U: + I16X8_MAX_U(stack); + break; + case I16x8_AVGR_U: + I16X8_AVGR_U(stack); + break; + case I16x8_EXTMUL_LOW_I8x16_S: + I16X8_EXTMUL_LOW_I8X16_S(stack); + break; + case I16x8_EXTMUL_HIGH_I8x16_S: + I16X8_EXTMUL_HIGH_I8X16_S(stack); + break; + case I16x8_EXTMUL_LOW_I8x16_U: + I16X8_EXTMUL_LOW_I8X16_U(stack); + break; + case I16x8_EXTMUL_HIGH_I8x16_U: + I16X8_EXTMUL_HIGH_I8X16_U(stack); + break; + case I32x4_ABS: + I32X4_ABS(stack); + break; + case I32x4_NEG: + I32X4_NEG(stack); + break; + case I32x4_ALL_TRUE: + I32X4_ALL_TRUE(stack); + break; + case I32x4_BITMASK: + I32X4_BITMASK(stack); + break; + case I32x4_EXTEND_LOW_I16x8_S: + I32X4_EXTEND_LOW_I16X8_S(stack); + break; + case I32x4_EXTEND_HIGH_I16x8_S: + I32X4_EXTEND_HIGH_I16X8_S(stack); + break; + case I32x4_EXTEND_LOW_I16x8_U: + I32X4_EXTEND_LOW_I16X8_U(stack); + break; + case I32x4_EXTEND_HIGH_I16x8_U: + I32X4_EXTEND_HIGH_I16X8_U(stack); + break; + case I32x4_SHL: + I32X4_SHL(stack); + break; + case I32x4_SHR_S: + I32X4_SHR_S(stack); + break; + case I32x4_SHR_U: + I32X4_SHR_U(stack); + break; + case I32x4_ADD: + I32X4_ADD(stack); + break; + case I32x4_SUB: + I32X4_SUB(stack); + break; + case I32x4_MUL: + I32X4_MUL(stack); + break; + case I32x4_MIN_S: + I32X4_MIN_S(stack); + break; + case I32x4_MIN_U: + I32X4_MIN_U(stack); + break; + case I32x4_MAX_S: + I32X4_MAX_S(stack); + break; + case I32x4_MAX_U: + I32X4_MAX_U(stack); + break; + case I32x4_DOT_I16x8_S: + I32X4_DOT_I16X8_S(stack); + break; + case I32x4_EXTMUL_LOW_I16x8_S: + I32X4_EXTMUL_LOW_I16X8_S(stack); + break; + case I32x4_EXTMUL_HIGH_I16x8_S: + I32X4_EXTMUL_HIGH_I16X8_S(stack); + break; + case I32x4_EXTMUL_LOW_I16x8_U: + I32X4_EXTMUL_LOW_I16X8_U(stack); + break; + case I32x4_EXTMUL_HIGH_I16x8_U: + I32X4_EXTMUL_HIGH_I16X8_U(stack); + break; + case I64x2_ABS: + I64X2_ABS(stack); + break; + case I64x2_NEG: + I64X2_NEG(stack); + break; + case I64x2_ALL_TRUE: + I64X2_ALL_TRUE(stack); + break; + case I64x2_BITMASK: + I64X2_BITMASK(stack); + break; + case I64x2_EXTEND_LOW_I32x4_S: + I64X2_EXTEND_LOW_I32X4_S(stack); + break; + case I64x2_EXTEND_HIGH_I32x4_S: + I64X2_EXTEND_HIGH_I32X4_S(stack); + break; + case I64x2_EXTEND_LOW_I32x4_U: + I64X2_EXTEND_LOW_I32X4_U(stack); + break; + case I64x2_EXTEND_HIGH_I32x4_U: + I64X2_EXTEND_HIGH_I32X4_U(stack); + break; + case I64x2_SHL: + I64X2_SHL(stack); + break; + case I64x2_SHR_S: + I64X2_SHR_S(stack); + break; + case I64x2_SHR_U: + I64X2_SHR_U(stack); + break; + case I64x2_ADD: + I64X2_ADD(stack); + break; + case I64x2_SUB: + I64X2_SUB(stack); + break; + case I64x2_MUL: + I64X2_MUL(stack); + break; + case I64x2_EQ: + I64X2_EQ(stack); + break; + case I64x2_NE: + I64X2_NE(stack); + break; + case I64x2_LT_S: + I64X2_LT_S(stack); + break; + case I64x2_GT_S: + I64X2_GT_S(stack); + break; + case I64x2_LE_S: + I64X2_LE_S(stack); + break; + case I64x2_GE_S: + I64X2_GE_S(stack); + break; + case I64x2_EXTMUL_LOW_I32x4_S: + I64X2_EXTMUL_LOW_I32X4_S(stack); + break; + case I64x2_EXTMUL_HIGH_I32x4_S: + I64X2_EXTMUL_HIGH_I32X4_S(stack); + break; + case I64x2_EXTMUL_LOW_I32x4_U: + I64X2_EXTMUL_LOW_I32X4_U(stack); + break; + case I64x2_EXTMUL_HIGH_I32x4_U: + I64X2_EXTMUL_HIGH_I32X4_U(stack); + break; + case F32x4_ABS: + F32X4_ABS(stack); + break; + case F32x4_NEG: + F32X4_NEG(stack); + break; + case F32x4_SQRT: + F32X4_SQRT(stack); + break; + case F32x4_ADD: + F32X4_ADD(stack); + break; + case F32x4_SUB: + F32X4_SUB(stack); + break; + case F32x4_MUL: + F32X4_MUL(stack); + break; + case F32x4_DIV: + F32X4_DIV(stack); + break; + case F32x4_MIN: + F32X4_MIN(stack); + break; + case F32x4_MAX: + F32X4_MAX(stack); + break; + case F32x4_PMIN: + F32X4_PMIN(stack); + break; + case F32x4_PMAX: + F32X4_PMAX(stack); + break; + case F64x2_ABS: + F64X2_ABS(stack); + break; + case F64x2_NEG: + F64X2_NEG(stack); + break; + case F64x2_SQRT: + F64X2_SQRT(stack); + break; + case F64x2_ADD: + F64X2_ADD(stack); + break; + case F64x2_SUB: + F64X2_SUB(stack); + break; + case F64x2_MUL: + F64X2_MUL(stack); + break; + case F64x2_DIV: + F64X2_DIV(stack); + break; + case F64x2_MIN: + F64X2_MIN(stack); + break; + case F64x2_MAX: + F64X2_MAX(stack); + break; + case F64x2_PMIN: + F64X2_PMIN(stack); + break; + case F64x2_PMAX: + F64X2_PMAX(stack); + break; + case I32x4_TRUNC_SAT_F32X4_S: + I32X4_TRUNC_SAT_F32X4_S(stack); + break; + case I32x4_TRUNC_SAT_F32X4_U: + I32X4_TRUNC_SAT_F32X4_U(stack); + break; + case F32x4_CONVERT_I32x4_S: + F32X4_CONVERT_I32X4_S(stack); + break; + case F32x4_CONVERT_I32x4_U: + F32X4_CONVERT_I32X4_U(stack); + break; + case I32x4_TRUNC_SAT_F64x2_S_ZERO: + I32X4_TRUNC_SAT_F64X2_S_ZERO(stack); + break; + case I32x4_TRUNC_SAT_F64x2_U_ZERO: + I32X4_TRUNC_SAT_F64X2_U_ZERO(stack); + break; + case F64x2_CONVERT_LOW_I32x4_S: + F64X2_CONVERT_LOW_I32X4_S(stack); + break; + case F64x2_CONVERT_LOW_I32x4_U: + F64X2_CONVERT_LOW_I32X4_U(stack); + break; + default: + return false; + } + return true; + } + + private static void V128_LOAD(MStack stack, Instance instance, Instruction instruction) { + load(stack, instance, instruction); + } + + private static void V128_LOAD8X8_S(MStack stack, Instance instance, Instruction instruction) { + loadExtend(stack, instance, instruction, 8, true); + } + + private static void V128_LOAD8X8_U(MStack stack, Instance instance, Instruction instruction) { + loadExtend(stack, instance, instruction, 8, false); + } + + private static void V128_LOAD16X4_S(MStack stack, Instance instance, Instruction instruction) { + loadExtend(stack, instance, instruction, 16, true); + } + + private static void V128_LOAD16X4_U(MStack stack, Instance instance, Instruction instruction) { + loadExtend(stack, instance, instruction, 16, false); + } + + private static void V128_LOAD32X2_S(MStack stack, Instance instance, Instruction instruction) { + loadExtend(stack, instance, instruction, 32, true); + } + + private static void V128_LOAD32X2_U(MStack stack, Instance instance, Instruction instruction) { + loadExtend(stack, instance, instruction, 32, false); + } + + private static void V128_LOAD8_SPLAT(MStack stack, Instance instance, Instruction instruction) { + loadSplat(stack, instance, instruction, 8); + } + + private static void V128_LOAD16_SPLAT( + MStack stack, Instance instance, Instruction instruction) { + loadSplat(stack, instance, instruction, 16); + } + + private static void V128_LOAD32_SPLAT( + MStack stack, Instance instance, Instruction instruction) { + loadSplat(stack, instance, instruction, 32); + } + + private static void V128_LOAD64_SPLAT( + MStack stack, Instance instance, Instruction instruction) { + loadSplat(stack, instance, instruction, 64); + } + + private static void V128_STORE(MStack stack, Instance instance, Instruction instruction) { + store(stack, instance, instruction); + } + + private static void V128_CONST(MStack stack, Instruction instruction) { + stack.push(instruction.operand(0)); + stack.push(instruction.operand(1)); + } + + private static void I8X16_SHUFFLE(MStack stack, Instruction instruction) { + shuffle(stack, instruction); + } + + private static void I8X16_SWIZZLE(MStack stack) { + swizzle(stack); + } + + private static void I8X16_SPLAT(MStack stack) { + splat(stack, 8); + } + + private static void I16X8_SPLAT(MStack stack) { + splat(stack, 16); + } + + private static void I32X4_SPLAT(MStack stack) { + splat(stack, 32); + } + + private static void I64X2_SPLAT(MStack stack) { + splat(stack, 64); + } + + private static void F32X4_SPLAT(MStack stack) { + splat(stack, 32); + } + + private static void F64X2_SPLAT(MStack stack) { + splat(stack, 64); + } + + private static void I8X16_EXTRACT_LANE_S(MStack stack, Instruction instruction) { + extract(stack, instruction, 8, true); + } + + private static void I8X16_EXTRACT_LANE_U(MStack stack, Instruction instruction) { + extract(stack, instruction, 8, false); + } + + private static void I8X16_REPLACE_LANE(MStack stack, Instruction instruction) { + replace(stack, instruction, 8); + } + + private static void I16X8_EXTRACT_LANE_S(MStack stack, Instruction instruction) { + extract(stack, instruction, 16, true); + } + + private static void I16X8_EXTRACT_LANE_U(MStack stack, Instruction instruction) { + extract(stack, instruction, 16, false); + } + + private static void I16X8_REPLACE_LANE(MStack stack, Instruction instruction) { + replace(stack, instruction, 16); + } + + private static void I32X4_EXTRACT_LANE(MStack stack, Instruction instruction) { + extract(stack, instruction, 32, true); + } + + private static void I32X4_REPLACE_LANE(MStack stack, Instruction instruction) { + replace(stack, instruction, 32); + } + + private static void I64X2_EXTRACT_LANE(MStack stack, Instruction instruction) { + extract(stack, instruction, 64, false); + } + + private static void I64X2_REPLACE_LANE(MStack stack, Instruction instruction) { + replace(stack, instruction, 64); + } + + private static void F32X4_EXTRACT_LANE(MStack stack, Instruction instruction) { + extract(stack, instruction, 32, true); + } + + private static void F32X4_REPLACE_LANE(MStack stack, Instruction instruction) { + replace(stack, instruction, 32); + } + + private static void F64X2_EXTRACT_LANE(MStack stack, Instruction instruction) { + extract(stack, instruction, 64, false); + } + + private static void F64X2_REPLACE_LANE(MStack stack, Instruction instruction) { + replace(stack, instruction, 64); + } + + private static void I8X16_EQ(MStack stack) { + binaryInt(stack, 8, EQ); + } + + private static void I8X16_NE(MStack stack) { + binaryInt(stack, 8, NE); + } + + private static void I8X16_LT_S(MStack stack) { + binaryInt(stack, 8, LT_S); + } + + private static void I8X16_LT_U(MStack stack) { + binaryInt(stack, 8, LT_U); + } + + private static void I8X16_GT_S(MStack stack) { + binaryInt(stack, 8, GT_S); + } + + private static void I8X16_GT_U(MStack stack) { + binaryInt(stack, 8, GT_U); + } + + private static void I8X16_LE_S(MStack stack) { + binaryInt(stack, 8, LE_S); + } + + private static void I8X16_LE_U(MStack stack) { + binaryInt(stack, 8, LE_U); + } + + private static void I8X16_GE_S(MStack stack) { + binaryInt(stack, 8, GE_S); + } + + private static void I8X16_GE_U(MStack stack) { + binaryInt(stack, 8, GE_U); + } + + private static void I16X8_EQ(MStack stack) { + binaryInt(stack, 16, EQ); + } + + private static void I16X8_NE(MStack stack) { + binaryInt(stack, 16, NE); + } + + private static void I16X8_LT_S(MStack stack) { + binaryInt(stack, 16, LT_S); + } + + private static void I16X8_LT_U(MStack stack) { + binaryInt(stack, 16, LT_U); + } + + private static void I16X8_GT_S(MStack stack) { + binaryInt(stack, 16, GT_S); + } + + private static void I16X8_GT_U(MStack stack) { + binaryInt(stack, 16, GT_U); + } + + private static void I16X8_LE_S(MStack stack) { + binaryInt(stack, 16, LE_S); + } + + private static void I16X8_LE_U(MStack stack) { + binaryInt(stack, 16, LE_U); + } + + private static void I16X8_GE_S(MStack stack) { + binaryInt(stack, 16, GE_S); + } + + private static void I16X8_GE_U(MStack stack) { + binaryInt(stack, 16, GE_U); + } + + private static void I32X4_EQ(MStack stack) { + binaryInt(stack, 32, EQ); + } + + private static void I32X4_NE(MStack stack) { + binaryInt(stack, 32, NE); + } + + private static void I32X4_LT_S(MStack stack) { + binaryInt(stack, 32, LT_S); + } + + private static void I32X4_LT_U(MStack stack) { + binaryInt(stack, 32, LT_U); + } + + private static void I32X4_GT_S(MStack stack) { + binaryInt(stack, 32, GT_S); + } + + private static void I32X4_GT_U(MStack stack) { + binaryInt(stack, 32, GT_U); + } + + private static void I32X4_LE_S(MStack stack) { + binaryInt(stack, 32, LE_S); + } + + private static void I32X4_LE_U(MStack stack) { + binaryInt(stack, 32, LE_U); + } + + private static void I32X4_GE_S(MStack stack) { + binaryInt(stack, 32, GE_S); + } + + private static void I32X4_GE_U(MStack stack) { + binaryInt(stack, 32, GE_U); + } + + private static void F32X4_EQ(MStack stack) { + binaryFloat(stack, 32, EQ); + } + + private static void F32X4_NE(MStack stack) { + binaryFloat(stack, 32, NE); + } + + private static void F32X4_LT(MStack stack) { + binaryFloat(stack, 32, LT); + } + + private static void F32X4_GT(MStack stack) { + binaryFloat(stack, 32, GT); + } + + private static void F32X4_LE(MStack stack) { + binaryFloat(stack, 32, LE); + } + + private static void F32X4_GE(MStack stack) { + binaryFloat(stack, 32, GE); + } + + private static void F64X2_EQ(MStack stack) { + binaryFloat(stack, 64, EQ); + } + + private static void F64X2_NE(MStack stack) { + binaryFloat(stack, 64, NE); + } + + private static void F64X2_LT(MStack stack) { + binaryFloat(stack, 64, LT); + } + + private static void F64X2_GT(MStack stack) { + binaryFloat(stack, 64, GT); + } + + private static void F64X2_LE(MStack stack) { + binaryFloat(stack, 64, LE); + } + + private static void F64X2_GE(MStack stack) { + binaryFloat(stack, 64, GE); + } + + private static void V128_NOT(MStack stack) { + int offset = stack.size() - 2; + stack.array()[offset] = ~stack.array()[offset]; + stack.array()[offset + 1] = ~stack.array()[offset + 1]; + } + + private static void V128_AND(MStack stack) { + bitwise(stack, AND); + } + + private static void V128_ANDNOT(MStack stack) { + bitwise(stack, ANDNOT); + } + + private static void V128_OR(MStack stack) { + bitwise(stack, OR); + } + + private static void V128_XOR(MStack stack) { + bitwise(stack, XOR); + } + + private static void V128_BITSELECT(MStack stack) { + bitselect(stack); + } + + private static void V128_ANY_TRUE(MStack stack) { + anyTrue(stack); + } + + private static void V128_LOAD8_LANE(MStack stack, Instance instance, Instruction instruction) { + loadLane(stack, instance, instruction, 8); + } + + private static void V128_LOAD16_LANE(MStack stack, Instance instance, Instruction instruction) { + loadLane(stack, instance, instruction, 16); + } + + private static void V128_LOAD32_LANE(MStack stack, Instance instance, Instruction instruction) { + loadLane(stack, instance, instruction, 32); + } + + private static void V128_LOAD64_LANE(MStack stack, Instance instance, Instruction instruction) { + loadLane(stack, instance, instruction, 64); + } + + private static void V128_STORE8_LANE(MStack stack, Instance instance, Instruction instruction) { + storeLane(stack, instance, instruction, 8); + } + + private static void V128_STORE16_LANE( + MStack stack, Instance instance, Instruction instruction) { + storeLane(stack, instance, instruction, 16); + } + + private static void V128_STORE32_LANE( + MStack stack, Instance instance, Instruction instruction) { + storeLane(stack, instance, instruction, 32); + } + + private static void V128_STORE64_LANE( + MStack stack, Instance instance, Instruction instruction) { + storeLane(stack, instance, instruction, 64); + } + + private static void V128_LOAD32_ZERO(MStack stack, Instance instance, Instruction instruction) { + loadZero(stack, instance, instruction, 32); + } + + private static void V128_LOAD64_ZERO(MStack stack, Instance instance, Instruction instruction) { + loadZero(stack, instance, instruction, 64); + } + + private static void F32X4_DEMOTE_LOW_F64X2_ZERO(MStack stack) { + demote(stack); + } + + private static void F64X2_PROMOTE_LOW_F32X4(MStack stack) { + promote(stack); + } + + private static void I8X16_ABS(MStack stack) { + unaryInt(stack, 8, ABS); + } + + private static void I8X16_NEG(MStack stack) { + unaryInt(stack, 8, NEG); + } + + private static void I8X16_POPCNT(MStack stack) { + unaryInt(stack, 8, POPCNT); + } + + private static void I8X16_ALL_TRUE(MStack stack) { + allTrue(stack, 8); + } + + private static void I8X16_BITMASK(MStack stack) { + bitmask(stack, 8); + } + + private static void I8X16_NARROW_I16X8_S(MStack stack) { + narrow(stack, 16, true); + } + + private static void I8X16_NARROW_I16X8_U(MStack stack) { + narrow(stack, 16, false); + } + + private static void F32X4_CEIL(MStack stack) { + unaryFloat(stack, 32, CEIL); + } + + private static void F32X4_FLOOR(MStack stack) { + unaryFloat(stack, 32, FLOOR); + } + + private static void F32X4_TRUNC(MStack stack) { + unaryFloat(stack, 32, TRUNC); + } + + private static void F32X4_NEAREST(MStack stack) { + unaryFloat(stack, 32, NEAREST); + } + + private static void I8X16_SHL(MStack stack) { + shift(stack, 8, SHL); + } + + private static void I8X16_SHR_S(MStack stack) { + shift(stack, 8, SHR_S); + } + + private static void I8X16_SHR_U(MStack stack) { + shift(stack, 8, SHR_U); + } + + private static void I8X16_ADD(MStack stack) { + binaryInt(stack, 8, ADD); + } + + private static void I8X16_ADD_SAT_S(MStack stack) { + binaryInt(stack, 8, ADD_SAT_S); + } + + private static void I8X16_ADD_SAT_U(MStack stack) { + binaryInt(stack, 8, ADD_SAT_U); + } + + private static void I8X16_SUB(MStack stack) { + binaryInt(stack, 8, SUB); + } + + private static void I8X16_SUB_SAT_S(MStack stack) { + binaryInt(stack, 8, SUB_SAT_S); + } + + private static void I8X16_SUB_SAT_U(MStack stack) { + binaryInt(stack, 8, SUB_SAT_U); + } + + private static void F64X2_CEIL(MStack stack) { + unaryFloat(stack, 64, CEIL); + } + + private static void F64X2_FLOOR(MStack stack) { + unaryFloat(stack, 64, FLOOR); + } + + private static void I8X16_MIN_S(MStack stack) { + binaryInt(stack, 8, MIN_S); + } + + private static void I8X16_MIN_U(MStack stack) { + binaryInt(stack, 8, MIN_U); + } + + private static void I8X16_MAX_S(MStack stack) { + binaryInt(stack, 8, MAX_S); + } + + private static void I8X16_MAX_U(MStack stack) { + binaryInt(stack, 8, MAX_U); + } + + private static void F64X2_TRUNC(MStack stack) { + unaryFloat(stack, 64, TRUNC); + } + + private static void I8X16_AVGR_U(MStack stack) { + binaryInt(stack, 8, AVGR_U); + } + + private static void I16X8_EXTADD_PAIRWISE_I8X16_S(MStack stack) { + pairwise(stack, 8, true); + } + + private static void I16X8_EXTADD_PAIRWISE_I8X16_U(MStack stack) { + pairwise(stack, 8, false); + } + + private static void I32X4_EXTADD_PAIRWISE_I16X8_S(MStack stack) { + pairwise(stack, 16, true); + } + + private static void I32X4_EXTADD_PAIRWISE_I16X8_U(MStack stack) { + pairwise(stack, 16, false); + } + + private static void I16X8_ABS(MStack stack) { + unaryInt(stack, 16, ABS); + } + + private static void I16X8_NEG(MStack stack) { + unaryInt(stack, 16, NEG); + } + + private static void I16X8_Q15MULR_SAT_S(MStack stack) { + q15(stack); + } + + private static void I16X8_ALL_TRUE(MStack stack) { + allTrue(stack, 16); + } + + private static void I16X8_BITMASK(MStack stack) { + bitmask(stack, 16); + } + + private static void I16X8_NARROW_I32X4_S(MStack stack) { + narrow(stack, 32, true); + } + + private static void I16X8_NARROW_I32X4_U(MStack stack) { + narrow(stack, 32, false); + } + + private static void I16X8_EXTEND_LOW_I8X16_S(MStack stack) { + extend(stack, 8, true, false); + } + + private static void I16X8_EXTEND_HIGH_I8X16_S(MStack stack) { + extend(stack, 8, true, true); + } + + private static void I16X8_EXTEND_LOW_I8X16_U(MStack stack) { + extend(stack, 8, false, false); + } + + private static void I16X8_EXTEND_HIGH_I8X16_U(MStack stack) { + extend(stack, 8, false, true); + } + + private static void I16X8_SHL(MStack stack) { + shift(stack, 16, SHL); + } + + private static void I16X8_SHR_S(MStack stack) { + shift(stack, 16, SHR_S); + } + + private static void I16X8_SHR_U(MStack stack) { + shift(stack, 16, SHR_U); + } + + private static void I16X8_ADD(MStack stack) { + binaryInt(stack, 16, ADD); + } + + private static void I16X8_ADD_SAT_S(MStack stack) { + binaryInt(stack, 16, ADD_SAT_S); + } + + private static void I16X8_ADD_SAT_U(MStack stack) { + binaryInt(stack, 16, ADD_SAT_U); + } + + private static void I16X8_SUB(MStack stack) { + binaryInt(stack, 16, SUB); + } + + private static void I16X8_SUB_SAT_S(MStack stack) { + binaryInt(stack, 16, SUB_SAT_S); + } + + private static void I16X8_SUB_SAT_U(MStack stack) { + binaryInt(stack, 16, SUB_SAT_U); + } + + private static void F64X2_NEAREST(MStack stack) { + unaryFloat(stack, 64, NEAREST); + } + + private static void I16X8_MUL(MStack stack) { + binaryInt(stack, 16, MUL); + } + + private static void I16X8_MIN_S(MStack stack) { + binaryInt(stack, 16, MIN_S); + } + + private static void I16X8_MIN_U(MStack stack) { + binaryInt(stack, 16, MIN_U); + } + + private static void I16X8_MAX_S(MStack stack) { + binaryInt(stack, 16, MAX_S); + } + + private static void I16X8_MAX_U(MStack stack) { + binaryInt(stack, 16, MAX_U); + } + + private static void I16X8_AVGR_U(MStack stack) { + binaryInt(stack, 16, AVGR_U); + } + + private static void I16X8_EXTMUL_LOW_I8X16_S(MStack stack) { + extmul(stack, 8, true, false); + } + + private static void I16X8_EXTMUL_HIGH_I8X16_S(MStack stack) { + extmul(stack, 8, true, true); + } + + private static void I16X8_EXTMUL_LOW_I8X16_U(MStack stack) { + extmul(stack, 8, false, false); + } + + private static void I16X8_EXTMUL_HIGH_I8X16_U(MStack stack) { + extmul(stack, 8, false, true); + } + + private static void I32X4_ABS(MStack stack) { + unaryInt(stack, 32, ABS); + } + + private static void I32X4_NEG(MStack stack) { + unaryInt(stack, 32, NEG); + } + + private static void I32X4_ALL_TRUE(MStack stack) { + allTrue(stack, 32); + } + + private static void I32X4_BITMASK(MStack stack) { + bitmask(stack, 32); + } + + private static void I32X4_EXTEND_LOW_I16X8_S(MStack stack) { + extend(stack, 16, true, false); + } + + private static void I32X4_EXTEND_HIGH_I16X8_S(MStack stack) { + extend(stack, 16, true, true); + } + + private static void I32X4_EXTEND_LOW_I16X8_U(MStack stack) { + extend(stack, 16, false, false); + } + + private static void I32X4_EXTEND_HIGH_I16X8_U(MStack stack) { + extend(stack, 16, false, true); + } + + private static void I32X4_SHL(MStack stack) { + shift(stack, 32, SHL); + } + + private static void I32X4_SHR_S(MStack stack) { + shift(stack, 32, SHR_S); + } + + private static void I32X4_SHR_U(MStack stack) { + shift(stack, 32, SHR_U); + } + + private static void I32X4_ADD(MStack stack) { + binaryInt(stack, 32, ADD); + } + + private static void I32X4_SUB(MStack stack) { + binaryInt(stack, 32, SUB); + } + + private static void I32X4_MUL(MStack stack) { + binaryInt(stack, 32, MUL); + } + + private static void I32X4_MIN_S(MStack stack) { + binaryInt(stack, 32, MIN_S); + } + + private static void I32X4_MIN_U(MStack stack) { + binaryInt(stack, 32, MIN_U); + } + + private static void I32X4_MAX_S(MStack stack) { + binaryInt(stack, 32, MAX_S); + } + + private static void I32X4_MAX_U(MStack stack) { + binaryInt(stack, 32, MAX_U); + } + + private static void I32X4_DOT_I16X8_S(MStack stack) { + dot(stack); + } + + private static void I32X4_EXTMUL_LOW_I16X8_S(MStack stack) { + extmul(stack, 16, true, false); + } + + private static void I32X4_EXTMUL_HIGH_I16X8_S(MStack stack) { + extmul(stack, 16, true, true); + } + + private static void I32X4_EXTMUL_LOW_I16X8_U(MStack stack) { + extmul(stack, 16, false, false); + } + + private static void I32X4_EXTMUL_HIGH_I16X8_U(MStack stack) { + extmul(stack, 16, false, true); + } + + private static void I64X2_ABS(MStack stack) { + unaryInt(stack, 64, ABS); + } + + private static void I64X2_NEG(MStack stack) { + unaryInt(stack, 64, NEG); + } + + private static void I64X2_ALL_TRUE(MStack stack) { + allTrue(stack, 64); + } + + private static void I64X2_BITMASK(MStack stack) { + bitmask(stack, 64); + } + + private static void I64X2_EXTEND_LOW_I32X4_S(MStack stack) { + extend(stack, 32, true, false); + } + + private static void I64X2_EXTEND_HIGH_I32X4_S(MStack stack) { + extend(stack, 32, true, true); + } + + private static void I64X2_EXTEND_LOW_I32X4_U(MStack stack) { + extend(stack, 32, false, false); + } + + private static void I64X2_EXTEND_HIGH_I32X4_U(MStack stack) { + extend(stack, 32, false, true); + } + + private static void I64X2_SHL(MStack stack) { + shift(stack, 64, SHL); + } + + private static void I64X2_SHR_S(MStack stack) { + shift(stack, 64, SHR_S); + } + + private static void I64X2_SHR_U(MStack stack) { + shift(stack, 64, SHR_U); + } + + private static void I64X2_ADD(MStack stack) { + binaryInt(stack, 64, ADD); + } + + private static void I64X2_SUB(MStack stack) { + binaryInt(stack, 64, SUB); + } + + private static void I64X2_MUL(MStack stack) { + binaryInt(stack, 64, MUL); + } + + private static void I64X2_EQ(MStack stack) { + binaryInt(stack, 64, EQ); + } + + private static void I64X2_NE(MStack stack) { + binaryInt(stack, 64, NE); + } + + private static void I64X2_LT_S(MStack stack) { + binaryInt(stack, 64, LT_S); + } + + private static void I64X2_GT_S(MStack stack) { + binaryInt(stack, 64, GT_S); + } + + private static void I64X2_LE_S(MStack stack) { + binaryInt(stack, 64, LE_S); + } + + private static void I64X2_GE_S(MStack stack) { + binaryInt(stack, 64, GE_S); + } + + private static void I64X2_EXTMUL_LOW_I32X4_S(MStack stack) { + extmul(stack, 32, true, false); + } + + private static void I64X2_EXTMUL_HIGH_I32X4_S(MStack stack) { + extmul(stack, 32, true, true); + } + + private static void I64X2_EXTMUL_LOW_I32X4_U(MStack stack) { + extmul(stack, 32, false, false); + } + + private static void I64X2_EXTMUL_HIGH_I32X4_U(MStack stack) { + extmul(stack, 32, false, true); + } + + private static void F32X4_ABS(MStack stack) { + unaryFloat(stack, 32, ABS); + } + + private static void F32X4_NEG(MStack stack) { + unaryFloat(stack, 32, NEG); + } + + private static void F32X4_SQRT(MStack stack) { + unaryFloat(stack, 32, SQRT); + } + + private static void F32X4_ADD(MStack stack) { + binaryFloat(stack, 32, ADD); + } + + private static void F32X4_SUB(MStack stack) { + binaryFloat(stack, 32, SUB); + } + + private static void F32X4_MUL(MStack stack) { + binaryFloat(stack, 32, MUL); + } + + private static void F32X4_DIV(MStack stack) { + binaryFloat(stack, 32, DIV); + } + + private static void F32X4_MIN(MStack stack) { + binaryFloat(stack, 32, MIN); + } + + private static void F32X4_MAX(MStack stack) { + binaryFloat(stack, 32, MAX); + } + + private static void F32X4_PMIN(MStack stack) { + binaryFloat(stack, 32, PMIN); + } + + private static void F32X4_PMAX(MStack stack) { + binaryFloat(stack, 32, PMAX); + } + + private static void F64X2_ABS(MStack stack) { + unaryFloat(stack, 64, ABS); + } + + private static void F64X2_NEG(MStack stack) { + unaryFloat(stack, 64, NEG); + } + + private static void F64X2_SQRT(MStack stack) { + unaryFloat(stack, 64, SQRT); + } + + private static void F64X2_ADD(MStack stack) { + binaryFloat(stack, 64, ADD); + } + + private static void F64X2_SUB(MStack stack) { + binaryFloat(stack, 64, SUB); + } + + private static void F64X2_MUL(MStack stack) { + binaryFloat(stack, 64, MUL); + } + + private static void F64X2_DIV(MStack stack) { + binaryFloat(stack, 64, DIV); + } + + private static void F64X2_MIN(MStack stack) { + binaryFloat(stack, 64, MIN); + } + + private static void F64X2_MAX(MStack stack) { + binaryFloat(stack, 64, MAX); + } + + private static void F64X2_PMIN(MStack stack) { + binaryFloat(stack, 64, PMIN); + } + + private static void F64X2_PMAX(MStack stack) { + binaryFloat(stack, 64, PMAX); + } + + private static void I32X4_TRUNC_SAT_F32X4_S(MStack stack) { + truncSatF32(stack, true); + } + + private static void I32X4_TRUNC_SAT_F32X4_U(MStack stack) { + truncSatF32(stack, false); + } + + private static void F32X4_CONVERT_I32X4_S(MStack stack) { + convertI32ToF32(stack, true); + } + + private static void F32X4_CONVERT_I32X4_U(MStack stack) { + convertI32ToF32(stack, false); + } + + private static void I32X4_TRUNC_SAT_F64X2_S_ZERO(MStack stack) { + truncSatF64Zero(stack, true); + } + + private static void I32X4_TRUNC_SAT_F64X2_U_ZERO(MStack stack) { + truncSatF64Zero(stack, false); + } + + private static void F64X2_CONVERT_LOW_I32X4_S(MStack stack) { + convertLowI32ToF64(stack, true); + } + + private static void F64X2_CONVERT_LOW_I32X4_U(MStack stack) { + convertLowI32ToF64(stack, false); + } + + private static void load(MStack stack, Instance instance, Instruction ins) { + int ptr = readMemPtr(stack, ins); + var memory = instance.memory((int) ins.operand(2)); + stack.push(memory.readLong(ptr)); + stack.push(memory.readLong(ptr + 8)); + } + + private static void loadExtend( + MStack stack, Instance instance, Instruction ins, int width, boolean signed) { + int ptr = readMemPtr(stack, ins); + var memory = instance.memory((int) ins.operand(2)); + long low = 0; + long high = 0; + int count = 64 / width; + for (int i = 0; i < count; i++) { + long value; + int at = ptr + i * (width / 8); + if (width == 8) { + value = signed ? memory.read(at) : memory.readU8(at); + } else if (width == 16) { + value = signed ? memory.readShort(at) : memory.readU16(at); + } else { + value = signed ? memory.readInt(at) : memory.readU32(at); + } + if (i < 64 / (width * 2)) { + low = put(low, i, width * 2, value); + } else { + high = put(high, i - 64 / (width * 2), width * 2, value); + } + } + stack.push(low); + stack.push(high); + } + + private static void loadSplat(MStack stack, Instance instance, Instruction ins, int width) { + int ptr = readMemPtr(stack, ins); + var memory = instance.memory((int) ins.operand(2)); + long value = + width == 8 + ? memory.read(ptr) + : width == 16 + ? memory.readShort(ptr) + : width == 32 ? memory.readInt(ptr) : memory.readLong(ptr); + stack.push(repeat(value, width)); + stack.push(repeat(value, width)); + } + + private static void loadZero(MStack stack, Instance instance, Instruction ins, int width) { + int ptr = readMemPtr(stack, ins); + var memory = instance.memory((int) ins.operand(2)); + stack.push(width == 32 ? memory.readU32(ptr) : memory.readLong(ptr)); + stack.push(0); + } + + private static void store(MStack stack, Instance instance, Instruction ins) { + long high = stack.pop(); + long low = stack.pop(); + int ptr = readMemPtr(stack, ins); + var memory = instance.memory((int) ins.operand(2)); + // high half first: if the store ends out of bounds it traps before writing anything + memory.writeLong(ptr + 8, high); + memory.writeLong(ptr, low); + } + + private static void loadLane(MStack stack, Instance instance, Instruction ins, int width) { + long high = stack.pop(); + long low = stack.pop(); + int ptr = readMemPtr(stack, ins); + int lane = (int) ins.operand(3); + var memory = instance.memory((int) ins.operand(2)); + long value = + width == 8 + ? memory.read(ptr) + : width == 16 + ? memory.readShort(ptr) + : width == 32 ? memory.readInt(ptr) : memory.readLong(ptr); + if (lane < 64 / width) { + low = put(low, lane, width, value); + } else { + high = put(high, lane - 64 / width, width, value); + } + stack.push(low); + stack.push(high); + } + + private static void storeLane(MStack stack, Instance instance, Instruction ins, int width) { + long high = stack.pop(); + long low = stack.pop(); + int ptr = readMemPtr(stack, ins); + int lane = (int) ins.operand(3); + long value = get(lane < 64 / width ? low : high, lane % (64 / width), width); + var memory = instance.memory((int) ins.operand(2)); + if (width == 8) { + memory.writeByte(ptr, (byte) value); + } else if (width == 16) { + memory.writeShort(ptr, (short) value); + } else if (width == 32) { + memory.writeI32(ptr, (int) value); + } else { + memory.writeLong(ptr, value); + } + } + + static int readMemPtr(MStack stack, Instruction ins) { + int address = (int) stack.pop(); + if (ins.operand(1) < 0 || ins.operand(1) >= Integer.MAX_VALUE || address < 0) { + throw new WasmRuntimeException("out of bounds memory access"); + } + return (int) (ins.operand(1) + address); + } + + private static void splat(MStack stack, int width) { + long value = stack.pop(); + long repeated = repeat(value, width); + stack.push(repeated); + stack.push(repeated); + } + + private static long repeat(long value, int width) { + if (width == 8) { + return (value & 0xffL) * 0x0101010101010101L; + } + if (width == 16) { + return (value & 0xffffL) * 0x0001000100010001L; + } + if (width == 32) { + return (value & 0xffffffffL) | ((value & 0xffffffffL) << 32); + } + return value; + } + + private static void shuffle(MStack stack, Instruction ins) { + long rightHigh = stack.pop(); + long rightLow = stack.pop(); + long leftHigh = stack.pop(); + long leftLow = stack.pop(); + long low = 0; + long high = 0; + for (int i = 0; i < 16; i++) { + int lane = + (int) (i < 8 ? ins.operand(0) >>> (i * 8) : ins.operand(1) >>> ((i - 8) * 8)) + & 0xff; + long value = + lane < 8 + ? get(leftLow, lane, 8) + : lane < 16 + ? get(leftHigh, lane - 8, 8) + : lane < 24 + ? get(rightLow, lane - 16, 8) + : get(rightHigh, lane - 24, 8); + if (i < 8) { + low = put(low, i, 8, value); + } else { + high = put(high, i - 8, 8, value); + } + } + stack.push(low); + stack.push(high); + } + + private static void swizzle(MStack stack) { + long indexHigh = stack.pop(); + long indexLow = stack.pop(); + long baseHigh = stack.pop(); + long baseLow = stack.pop(); + long low = 0; + long high = 0; + for (int i = 0; i < 16; i++) { + long index = get(i < 8 ? indexLow : indexHigh, i & 7, 8); + long value = + index < 8 + ? get(baseLow, (int) index, 8) + : index < 16 ? get(baseHigh, (int) index - 8, 8) : 0; + if (i < 8) { + low = put(low, i, 8, value); + } else { + high = put(high, i - 8, 8, value); + } + } + stack.push(low); + stack.push(high); + } + + private static void extract(MStack stack, Instruction ins, int width, boolean signed) { + int lane = (int) ins.operand(0); + int offset = stack.size() - 2; + long value = + get( + stack.array()[lane < 64 / width ? offset : offset + 1], + lane % (64 / width), + width); + if (signed) { + value = signExtend(value, width); + } + stack.pop(); + stack.array()[stack.size() - 1] = value; + } + + private static void replace(MStack stack, Instruction ins, int width) { + long value = stack.pop(); + int lane = (int) ins.operand(0); + int offset = stack.size() - 2; + if (lane < 64 / width) { + stack.array()[offset] = put(stack.array()[offset], lane, width, value); + } else { + stack.array()[offset + 1] = + put(stack.array()[offset + 1], lane - 64 / width, width, value); + } + } + + private static void bitwise(MStack stack, int operation) { + long bHigh = stack.pop(); + long bLow = stack.pop(); + int offset = stack.size() - 2; + long aLow = stack.array()[offset]; + long aHigh = stack.array()[offset + 1]; + stack.array()[offset] = bitwise(aLow, bLow, operation); + stack.array()[offset + 1] = bitwise(aHigh, bHigh, operation); + } + + private static long bitwise(long a, long b, int operation) { + switch (operation) { + case AND: + return a & b; + case ANDNOT: + return a & ~b; + case OR: + return a | b; + case XOR: + return a ^ b; + default: + throw new AssertionError(operation); + } + } + + private static void bitselect(MStack stack) { + long maskHigh = stack.pop(); + long maskLow = stack.pop(); + long secondHigh = stack.pop(); + long secondLow = stack.pop(); + int offset = stack.size() - 2; + long firstLow = stack.array()[offset]; + long firstHigh = stack.array()[offset + 1]; + stack.array()[offset] = (firstLow & maskLow) | (secondLow & ~maskLow); + stack.array()[offset + 1] = (firstHigh & maskHigh) | (secondHigh & ~maskHigh); + } + + private static void anyTrue(MStack stack) { + long high = stack.pop(); + long low = stack.pop(); + stack.push((low | high) == 0 ? BitOps.FALSE : BitOps.TRUE); + } + + private static void binaryInt(MStack stack, int width, int operation) { + long bHigh = stack.pop(); + long bLow = stack.pop(); + int offset = stack.size() - 2; + long aLow = stack.array()[offset]; + long aHigh = stack.array()[offset + 1]; + long low = 0; + long high = 0; + int lanes = 64 / width; + for (int i = 0; i < lanes; i++) { + low = put(low, i, width, intLane(aLow, bLow, i, width, operation)); + high = put(high, i, width, intLane(aHigh, bHigh, i, width, operation)); + } + stack.array()[offset] = low; + stack.array()[offset + 1] = high; + } + + private static long intLane(long aWord, long bWord, int lane, int width, int operation) { + long aBits = get(aWord, lane, width); + long bBits = get(bWord, lane, width); + long a = signExtend(aBits, width); + long b = signExtend(bBits, width); + long au = unsigned(aBits, width); + long bu = unsigned(bBits, width); + switch (operation) { + case ADD: + return a + b; + case SUB: + return a - b; + case MUL: + return a * b; + case MIN_S: + return Math.min(a, b); + case MIN_U: + return Math.min(au, bu); + case MAX_S: + return Math.max(a, b); + case MAX_U: + return Math.max(au, bu); + case EQ: + return aBits == bBits ? mask(width) : 0; + case NE: + return aBits != bBits ? mask(width) : 0; + case LT_S: + return a < b ? mask(width) : 0; + case LT_U: + return compareUnsigned(au, bu, width) < 0 ? mask(width) : 0; + case GT_S: + return a > b ? mask(width) : 0; + case GT_U: + return compareUnsigned(au, bu, width) > 0 ? mask(width) : 0; + case LE_S: + return a <= b ? mask(width) : 0; + case LE_U: + return compareUnsigned(au, bu, width) <= 0 ? mask(width) : 0; + case GE_S: + return a >= b ? mask(width) : 0; + case GE_U: + return compareUnsigned(au, bu, width) >= 0 ? mask(width) : 0; + case ADD_SAT_S: + return saturate(a + b, width); + case ADD_SAT_U: + return saturateUnsigned(au + bu, width); + case SUB_SAT_S: + return saturate(a - b, width); + case SUB_SAT_U: + return Math.max(0, au - bu); + case AVGR_U: + return (au + bu + 1) >>> 1; + case Q15MULR_SAT_S: + return saturate((a * b + 16384) >> 15, 16); + default: + throw new AssertionError(operation); + } + } + + private static void shift(MStack stack, int width, int operation) { + int shift = (int) stack.pop() & (width - 1); + int offset = stack.size() - 2; + for (int word = 0; word < 2; word++) { + long value = stack.array()[offset + word]; + long result = 0; + for (int lane = 0; lane < 64 / width; lane++) { + long bits = get(value, lane, width); + long signed = signExtend(bits, width); + long shifted; + switch (operation) { + case SHL: + shifted = bits << shift; + break; + case SHR_S: + shifted = signed >> shift; + break; + case SHR_U: + shifted = unsigned(bits, width) >>> shift; + break; + default: + throw new AssertionError(operation); + } + result = put(result, lane, width, shifted); + } + stack.array()[offset + word] = result; + } + } + + private static void unaryInt(MStack stack, int width, int operation) { + int offset = stack.size() - 2; + for (int word = 0; word < 2; word++) { + long value = stack.array()[offset + word]; + long result = 0; + for (int lane = 0; lane < 64 / width; lane++) { + long bits = get(value, lane, width); + long signed = signExtend(bits, width); + long resultValue; + switch (operation) { + case ABS: + resultValue = Math.abs(signed); + break; + case NEG: + resultValue = -signed; + break; + case POPCNT: + resultValue = Integer.bitCount((int) bits); + break; + default: + throw new AssertionError(operation); + } + result = put(result, lane, width, resultValue); + } + stack.array()[offset + word] = result; + } + } + + private static void allTrue(MStack stack, int width) { + long high = stack.pop(); + long low = stack.pop(); + boolean result = true; + for (int i = 0; i < 64 / width; i++) { + result &= get(low, i, width) != 0; + result &= get(high, i, width) != 0; + } + stack.push(result ? BitOps.TRUE : BitOps.FALSE); + } + + private static void bitmask(MStack stack, int width) { + long high = stack.pop(); + long low = stack.pop(); + long result = 0; + int lanes = 128 / width; + for (int i = 0; i < lanes; i++) { + long value = get(i < 64 / width ? low : high, i % (64 / width), width); + if ((value & (1L << (width - 1))) != 0) { + result |= 1L << i; + } + } + stack.push(result); + } + + private static void binaryFloat(MStack stack, int width, int operation) { + long bHigh = stack.pop(); + long bLow = stack.pop(); + int offset = stack.size() - 2; + long aLow = stack.array()[offset]; + long aHigh = stack.array()[offset + 1]; + long low = 0; + long high = 0; + int lanes = 64 / width; + for (int i = 0; i < lanes; i++) { + low = put(low, i, width, floatLane(aLow, bLow, i, width, operation)); + high = put(high, i, width, floatLane(aHigh, bHigh, i, width, operation)); + } + stack.array()[offset] = low; + stack.array()[offset + 1] = high; + } + + private static long floatLane(long aWord, long bWord, int lane, int width, int operation) { + long aBits = get(aWord, lane, width); + long bBits = get(bWord, lane, width); + if (width == 32) { + float a = Float.intBitsToFloat((int) aBits); + float b = Float.intBitsToFloat((int) bBits); + switch (operation) { + case ADD: + return Float.floatToIntBits(a + b); + case SUB: + return Float.floatToIntBits(a - b); + case MUL: + return Float.floatToIntBits(a * b); + case DIV: + return Float.floatToIntBits(a / b); + case MIN: + return Float.floatToIntBits(Math.min(a, b)); + case MAX: + return Float.floatToIntBits(Math.max(a, b)); + case PMIN: + return Float.floatToRawIntBits(pmin(a, b)); + case PMAX: + return Float.floatToRawIntBits(pmax(a, b)); + default: + return floatCompare(a, b, operation) ? 0xffffffffL : 0; + } + } + double a = Double.longBitsToDouble(aBits); + double b = Double.longBitsToDouble(bBits); + switch (operation) { + case ADD: + return Double.doubleToLongBits(a + b); + case SUB: + return Double.doubleToLongBits(a - b); + case MUL: + return Double.doubleToLongBits(a * b); + case DIV: + return Double.doubleToLongBits(a / b); + case MIN: + return Double.doubleToLongBits(Math.min(a, b)); + case MAX: + return Double.doubleToLongBits(Math.max(a, b)); + case PMIN: + return Double.doubleToRawLongBits(pmin(a, b)); + case PMAX: + return Double.doubleToRawLongBits(pmax(a, b)); + default: + return floatCompare(a, b, operation) ? -1L : 0; + } + } + + private static float pmin(float a, float b) { + return b < a ? b : a; + } + + private static double pmin(double a, double b) { + return b < a ? b : a; + } + + private static float pmax(float a, float b) { + return a < b ? b : a; + } + + private static double pmax(double a, double b) { + return a < b ? b : a; + } + + private static boolean floatCompare(float a, float b, int operation) { + switch (operation) { + case EQ: + return a == b; + case NE: + return a != b; + case LT: + return a < b; + case GT: + return a > b; + case LE: + return a <= b; + case GE: + return a >= b; + default: + throw new AssertionError(operation); + } + } + + private static boolean floatCompare(double a, double b, int operation) { + switch (operation) { + case EQ: + return a == b; + case NE: + return a != b; + case LT: + return a < b; + case GT: + return a > b; + case LE: + return a <= b; + case GE: + return a >= b; + default: + throw new AssertionError(operation); + } + } + + // Bit-preserving abs/neg/pmin/pmax; arithmetic operations canonicalize NaN. + private static void unaryFloat(MStack stack, int width, int operation) { + int offset = stack.size() - 2; + for (int word = 0; word < 2; word++) { + long value = stack.array()[offset + word]; + long result = 0; + for (int lane = 0; lane < 64 / width; lane++) { + long bits = get(value, lane, width); + if (width == 32) { + long laneResult; + switch (operation) { + case ABS: + laneResult = bits & 0x7fffffffL; + break; + case NEG: + laneResult = bits ^ 0x80000000L; + break; + case SQRT: + laneResult = + Float.floatToIntBits( + (float) Math.sqrt(Float.intBitsToFloat((int) bits))); + break; + case CEIL: + laneResult = + Float.floatToIntBits( + (float) Math.ceil(Float.intBitsToFloat((int) bits))); + break; + case FLOOR: + laneResult = + Float.floatToIntBits( + (float) Math.floor(Float.intBitsToFloat((int) bits))); + break; + case TRUNC: + float x = Float.intBitsToFloat((int) bits); + laneResult = + Float.floatToIntBits( + (float) (x < 0 ? Math.ceil(x) : Math.floor(x))); + break; + case NEAREST: + laneResult = + Float.floatToIntBits( + (float) Math.rint(Float.intBitsToFloat((int) bits))); + break; + default: + throw new AssertionError(operation); + } + result = put(result, lane, width, laneResult); + } else { + long laneResult; + switch (operation) { + case ABS: + laneResult = bits & 0x7fffffffffffffffL; + break; + case NEG: + laneResult = bits ^ 0x8000000000000000L; + break; + case SQRT: + laneResult = + Double.doubleToLongBits( + Math.sqrt(Double.longBitsToDouble(bits))); + break; + case CEIL: + laneResult = + Double.doubleToLongBits( + Math.ceil(Double.longBitsToDouble(bits))); + break; + case FLOOR: + laneResult = + Double.doubleToLongBits( + Math.floor(Double.longBitsToDouble(bits))); + break; + case TRUNC: + double x = Double.longBitsToDouble(bits); + laneResult = + Double.doubleToLongBits(x < 0 ? Math.ceil(x) : Math.floor(x)); + break; + case NEAREST: + laneResult = + Double.doubleToLongBits( + Math.rint(Double.longBitsToDouble(bits))); + break; + default: + throw new AssertionError(operation); + } + result = put(result, lane, width, laneResult); + } + } + stack.array()[offset + word] = result; + } + } + + private static void narrow(MStack stack, int inputWidth, boolean signed) { + long secondHigh = stack.pop(); + long secondLow = stack.pop(); + int offset = stack.size() - 2; + long firstLow = stack.array()[offset]; + long firstHigh = stack.array()[offset + 1]; + int outputWidth = inputWidth / 2; + long low = 0; + long high = 0; + int inputLanes = 128 / inputWidth; + for (int i = 0; i < inputLanes; i++) { + int secondLane = i + inputLanes; + long firstValue = + narrowValue(vectorGet(firstLow, firstHigh, i, inputWidth), inputWidth, signed); + long secondValue = + narrowValue( + vectorGet(secondLow, secondHigh, i, inputWidth), inputWidth, signed); + if (i < 64 / outputWidth) { + low = put(low, i, outputWidth, firstValue); + } else { + high = put(high, i - 64 / outputWidth, outputWidth, firstValue); + } + if (secondLane < 64 / outputWidth) { + low = put(low, secondLane, outputWidth, secondValue); + } else { + high = put(high, secondLane - 64 / outputWidth, outputWidth, secondValue); + } + } + stack.array()[offset] = low; + stack.array()[offset + 1] = high; + } + + private static long narrowValue(long bits, int width, boolean signed) { + long value = signExtend(bits, width); + if (!signed) { + return Math.max(0, Math.min(mask(width / 2), value)); + } + long min = -(1L << (width / 2 - 1)); + long max = (1L << (width / 2 - 1)) - 1; + return Math.max(min, Math.min(max, value)); + } + + private static void extend(MStack stack, int inputWidth, boolean signed, boolean high) { + long inputHigh = stack.pop(); + long inputLow = stack.pop(); + long source = high ? inputHigh : inputLow; + int outputWidth = inputWidth * 2; + long low = 0; + long resultHigh = 0; + int outputLanes = 128 / outputWidth; + for (int i = 0; i < outputLanes; i++) { + long bits = get(source, i, inputWidth); + long value = signed ? signExtend(bits, inputWidth) : unsigned(bits, inputWidth); + if (i < 64 / outputWidth) { + low = put(low, i, outputWidth, value); + } else { + resultHigh = put(resultHigh, i - 64 / outputWidth, outputWidth, value); + } + } + stack.push(low); + stack.push(resultHigh); + } + + private static void pairwise(MStack stack, int inputWidth, boolean signed) { + long high = stack.pop(); + long low = stack.pop(); + int outputWidth = inputWidth * 2; + long resultLow = 0; + long resultHigh = 0; + int count = 128 / inputWidth; + for (int i = 0; i < count / 2; i++) { + long first = + get( + i * 2 < 64 / inputWidth ? low : high, + i * 2 % (64 / inputWidth), + inputWidth); + long second = + get( + i * 2 + 1 < 64 / inputWidth ? low : high, + (i * 2 + 1) % (64 / inputWidth), + inputWidth); + long value = + signed + ? signExtend(first, inputWidth) + signExtend(second, inputWidth) + : unsigned(first, inputWidth) + unsigned(second, inputWidth); + if (i < 64 / outputWidth) { + resultLow = put(resultLow, i, outputWidth, value); + } else { + resultHigh = put(resultHigh, i - 64 / outputWidth, outputWidth, value); + } + } + stack.push(resultLow); + stack.push(resultHigh); + } + + private static void extmul(MStack stack, int inputWidth, boolean signed, boolean high) { + long secondHigh = stack.pop(); + long secondLow = stack.pop(); + int offset = stack.size() - 2; + long firstLow = stack.array()[offset]; + long firstHigh = stack.array()[offset + 1]; + int outputWidth = inputWidth * 2; + int lanes = 64 / inputWidth; + int start = high ? 64 / inputWidth : 0; + long low = 0; + long resultHigh = 0; + for (int i = 0; i < lanes; i++) { + int lane = start + i; + long aBits = vectorGet(firstLow, firstHigh, lane, inputWidth); + long bBits = vectorGet(secondLow, secondHigh, lane, inputWidth); + long value = + signed + ? signExtend(aBits, inputWidth) * signExtend(bBits, inputWidth) + : unsigned(aBits, inputWidth) * unsigned(bBits, inputWidth); + if (i < 64 / outputWidth) { + low = put(low, i, outputWidth, value); + } else { + resultHigh = put(resultHigh, i - 64 / outputWidth, outputWidth, value); + } + } + stack.array()[offset] = low; + stack.array()[offset + 1] = resultHigh; + } + + private static void dot(MStack stack) { + long secondHigh = stack.pop(); + long secondLow = stack.pop(); + int offset = stack.size() - 2; + long firstLow = stack.array()[offset]; + long firstHigh = stack.array()[offset + 1]; + long low = 0; + long high = 0; + for (int i = 0; i < 4; i++) { + int lane = 2 * i; + long first = + signExtend(vectorGet(firstLow, firstHigh, lane, 16), 16) + * signExtend(vectorGet(secondLow, secondHigh, lane, 16), 16); + long second = + signExtend(vectorGet(firstLow, firstHigh, lane + 1, 16), 16) + * signExtend(vectorGet(secondLow, secondHigh, lane + 1, 16), 16); + long value = first + second; + if (i < 2) { + low = put(low, i, 32, value); + } else { + high = put(high, i - 2, 32, value); + } + } + stack.array()[offset] = low; + stack.array()[offset + 1] = high; + } + + private static void q15(MStack stack) { + binaryInt(stack, 16, Q15MULR_SAT_S); + } + + private static void truncSatF32(MStack stack, boolean signed) { + int offset = stack.size() - 2; + for (int word = 0; word < 2; word++) { + long input = stack.array()[offset + word]; + long result = 0; + for (int i = 0; i < 2; i++) { + float value = Float.intBitsToFloat((int) get(input, i, 32)); + long converted = + signed + ? OpcodeImpl.I32_TRUNC_SAT_F32_S(value) + : OpcodeImpl.I32_TRUNC_SAT_F32_U(value); + result = put(result, i, 32, converted); + } + stack.array()[offset + word] = result; + } + } + + private static void truncSatF64Zero(MStack stack, boolean signed) { + int offset = stack.size() - 2; + long low = stack.array()[offset]; + long high = stack.array()[offset + 1]; + double first = Double.longBitsToDouble(low); + double second = Double.longBitsToDouble(high); + long resultLow = + signed + ? OpcodeImpl.I32_TRUNC_SAT_F64_S(first) + : OpcodeImpl.I32_TRUNC_SAT_F64_U(first); + long resultHigh = + signed + ? OpcodeImpl.I32_TRUNC_SAT_F64_S(second) + : OpcodeImpl.I32_TRUNC_SAT_F64_U(second); + stack.array()[offset] = (resultLow & 0xffffffffL) | ((resultHigh & 0xffffffffL) << 32); + stack.array()[offset + 1] = 0; + } + + private static void convertI32ToF32(MStack stack, boolean signed) { + int offset = stack.size() - 2; + for (int word = 0; word < 2; word++) { + long input = stack.array()[offset + word]; + long result = 0; + for (int i = 0; i < 2; i++) { + int value = (int) get(input, i, 32); + float converted = + signed + ? OpcodeImpl.F32_CONVERT_I32_S(value) + : OpcodeImpl.F32_CONVERT_I32_U(value); + result = put(result, i, 32, Float.floatToIntBits(converted)); + } + stack.array()[offset + word] = result; + } + } + + private static void convertLowI32ToF64(MStack stack, boolean signed) { + int offset = stack.size() - 2; + long input = stack.array()[offset]; + double first = + signed ? (int) get(input, 0, 32) : Integer.toUnsignedLong((int) get(input, 0, 32)); + double second = + signed ? (int) get(input, 1, 32) : Integer.toUnsignedLong((int) get(input, 1, 32)); + stack.array()[offset] = Double.doubleToLongBits(first); + stack.array()[offset + 1] = Double.doubleToLongBits(second); + } + + private static void demote(MStack stack) { + long high = stack.pop(); + long low = stack.pop(); + stack.push( + (Float.floatToIntBits((float) Double.longBitsToDouble(low)) & 0xffffffffL) + | ((long) Float.floatToIntBits((float) Double.longBitsToDouble(high)) + << 32)); + stack.push(0); + } + + private static void promote(MStack stack) { + stack.pop(); + long low = stack.pop(); + stack.push(Double.doubleToLongBits(Float.intBitsToFloat((int) low))); + stack.push(Double.doubleToLongBits(Float.intBitsToFloat((int) (low >>> 32)))); + } + + private static long get(long word, int lane, int width) { + if (width == 64) { + return word; + } + return (word >>> (lane * width)) & mask(width); + } + + private static long vectorGet(long low, long high, int lane, int width) { + return lane < 64 / width ? get(low, lane, width) : get(high, lane - 64 / width, width); + } + + private static long put(long word, int lane, int width, long value) { + if (width == 64) { + return value; + } + long shift = (long) lane * width; + long laneMask = mask(width) << shift; + return (word & ~laneMask) | ((value & mask(width)) << shift); + } + + private static long mask(int width) { + return width == 64 ? -1L : (1L << width) - 1; + } + + private static long unsigned(long value, int width) { + return width == 64 ? value : value & mask(width); + } + + private static long signExtend(long value, int width) { + if (width == 64) { + return value; + } + long mask = mask(width); + long sign = 1L << (width - 1); + value &= mask; + return (value ^ sign) - sign; + } + + private static int compareUnsigned(long a, long b, int width) { + return width == 64 ? Long.compareUnsigned(a, b) : Long.compare(a, b); + } + + private static long saturate(long value, int width) { + long min = -(1L << (width - 1)); + long max = (1L << (width - 1)) - 1; + return Math.max(min, Math.min(max, value)); + } + + private static long saturateUnsigned(long value, int width) { + return Math.min(mask(width), value); + } +} diff --git a/simd/pom.xml b/simd/pom.xml deleted file mode 100644 index f63818d90..000000000 --- a/simd/pom.xml +++ /dev/null @@ -1,125 +0,0 @@ - - - 4.0.0 - - - run.endive - endive - 999-SNAPSHOT - ../pom.xml - - simd - jar - Endive - SIMD - SIMD instructions support for Endive - - - 25 - - - false - - - - - run.endive - runtime - - - run.endive - wasm - - - org.junit.jupiter - junit-jupiter-api - test - - - org.junit.jupiter - junit-jupiter-engine - test - - - run.endive - wasm-corpus - test - - - - - - - org.apache.maven.plugins - maven-compiler-plugin - - - --add-modules - jdk.incubator.vector - - - - - org.apache.maven.plugins - maven-javadoc-plugin - - - - false - false - -Xdoclint:none - - - - org.apache.maven.plugins - maven-surefire-plugin - - --add-modules=jdk.incubator.vector - - - - org.codehaus.mojo - templating-maven-plugin - - - filter-src - - filter-sources - - - - - - - - - - java21 - - 21 - - - 21 - - - - - org.codehaus.mojo - templating-maven-plugin - - - filter-src - - filter-sources - - - ${basedir}/src/main/java-templates-21 - - - - - - - - - - diff --git a/simd/src/main/java-templates-21/run/endive/simd/VectorOperators.java b/simd/src/main/java-templates-21/run/endive/simd/VectorOperators.java deleted file mode 100644 index 951bab3ac..000000000 --- a/simd/src/main/java-templates-21/run/endive/simd/VectorOperators.java +++ /dev/null @@ -1,28 +0,0 @@ -package run.endive.simd; - -/** - * Generated compatibility shim for cross-version compatibility with jdk.incubator.vector.VectorOperators - * - */ -final class VectorOperators { - - private VectorOperators() {} - - static final jdk.incubator.vector.VectorOperators.Comparison NE = jdk.incubator.vector.VectorOperators.NE; - static final jdk.incubator.vector.VectorOperators.Binary LSHL = jdk.incubator.vector.VectorOperators.LSHL; - static final jdk.incubator.vector.VectorOperators.Binary LSHR = jdk.incubator.vector.VectorOperators.LSHR; - static final jdk.incubator.vector.VectorOperators.Binary ASHR = jdk.incubator.vector.VectorOperators.ASHR; - static final jdk.incubator.vector.VectorOperators.Comparison UNSIGNED_LT = jdk.incubator.vector.VectorOperators.UNSIGNED_LT; - static final jdk.incubator.vector.VectorOperators.Comparison LE = jdk.incubator.vector.VectorOperators.LE; - static final jdk.incubator.vector.VectorOperators.Comparison UNSIGNED_LE = jdk.incubator.vector.VectorOperators.UNSIGNED_LE; - static final jdk.incubator.vector.VectorOperators.Comparison GT = jdk.incubator.vector.VectorOperators.GT; - static final jdk.incubator.vector.VectorOperators.Comparison UNSIGNED_GT = jdk.incubator.vector.VectorOperators.UNSIGNED_GT; - static final jdk.incubator.vector.VectorOperators.Comparison GE = jdk.incubator.vector.VectorOperators.GE; - static final jdk.incubator.vector.VectorOperators.Comparison UNSIGNED_GE = jdk.incubator.vector.VectorOperators.UNSIGNED_GE; - static final jdk.incubator.vector.VectorOperators.Unary BIT_COUNT = jdk.incubator.vector.VectorOperators.BIT_COUNT; - static final jdk.incubator.vector.VectorOperators.Comparison LT = jdk.incubator.vector.VectorOperators.LT; - static final jdk.incubator.vector.VectorOperators.Unary SQRT = jdk.incubator.vector.VectorOperators.SQRT; - static final jdk.incubator.vector.VectorOperators.Unary ABS = jdk.incubator.vector.VectorOperators.ABS; - static final jdk.incubator.vector.VectorOperators.Unary NEG = jdk.incubator.vector.VectorOperators.NEG; - -} diff --git a/simd/src/main/java-templates/run/endive/simd/VectorOperators.java b/simd/src/main/java-templates/run/endive/simd/VectorOperators.java deleted file mode 100644 index 784581d9b..000000000 --- a/simd/src/main/java-templates/run/endive/simd/VectorOperators.java +++ /dev/null @@ -1,28 +0,0 @@ -package run.endive.simd; - -/** - * Generated compatibility shim for cross-version compatibility with jdk.incubator.vector.VectorOperators - * - */ -final class VectorOperators { - - private VectorOperators() {} - - static final jdk.incubator.vector.VectorOperators.Comparison NE = jdk.incubator.vector.VectorOperators.NE; - static final jdk.incubator.vector.VectorOperators.Binary LSHL = jdk.incubator.vector.VectorOperators.LSHL; - static final jdk.incubator.vector.VectorOperators.Binary LSHR = jdk.incubator.vector.VectorOperators.LSHR; - static final jdk.incubator.vector.VectorOperators.Binary ASHR = jdk.incubator.vector.VectorOperators.ASHR; - static final jdk.incubator.vector.VectorOperators.Comparison UNSIGNED_LT = jdk.incubator.vector.VectorOperators.ULT; - static final jdk.incubator.vector.VectorOperators.Comparison LE = jdk.incubator.vector.VectorOperators.LE; - static final jdk.incubator.vector.VectorOperators.Comparison UNSIGNED_LE = jdk.incubator.vector.VectorOperators.ULE; - static final jdk.incubator.vector.VectorOperators.Comparison GT = jdk.incubator.vector.VectorOperators.GT; - static final jdk.incubator.vector.VectorOperators.Comparison UNSIGNED_GT = jdk.incubator.vector.VectorOperators.UGT; - static final jdk.incubator.vector.VectorOperators.Comparison GE = jdk.incubator.vector.VectorOperators.GE; - static final jdk.incubator.vector.VectorOperators.Comparison UNSIGNED_GE = jdk.incubator.vector.VectorOperators.UGE; - static final jdk.incubator.vector.VectorOperators.Unary BIT_COUNT = jdk.incubator.vector.VectorOperators.BIT_COUNT; - static final jdk.incubator.vector.VectorOperators.Comparison LT = jdk.incubator.vector.VectorOperators.LT; - static final jdk.incubator.vector.VectorOperators.Unary SQRT = jdk.incubator.vector.VectorOperators.SQRT; - static final jdk.incubator.vector.VectorOperators.Unary ABS = jdk.incubator.vector.VectorOperators.ABS; - static final jdk.incubator.vector.VectorOperators.Unary NEG = jdk.incubator.vector.VectorOperators.NEG; - -} diff --git a/simd/src/main/java/module-info.java b/simd/src/main/java/module-info.java deleted file mode 100644 index 5e8096f4f..000000000 --- a/simd/src/main/java/module-info.java +++ /dev/null @@ -1,5 +0,0 @@ -module run.endive.simd { - requires transitive run.endive.runtime; - requires run.endive.wasm; - requires jdk.incubator.vector; -} diff --git a/simd/src/main/java/run/endive/simd/SimdInterpreterMachine.java b/simd/src/main/java/run/endive/simd/SimdInterpreterMachine.java deleted file mode 100644 index 8b8a81d3b..000000000 --- a/simd/src/main/java/run/endive/simd/SimdInterpreterMachine.java +++ /dev/null @@ -1,2986 +0,0 @@ -package run.endive.simd; - -import java.util.Arrays; -import java.util.Deque; -import java.util.function.BiConsumer; -import java.util.function.BiFunction; -import java.util.function.Function; -import jdk.incubator.vector.LongVector; -import jdk.incubator.vector.Vector; -import run.endive.runtime.BitOps; -import run.endive.runtime.Instance; -import run.endive.runtime.InterpreterMachine; -import run.endive.runtime.MStack; -import run.endive.runtime.OpcodeImpl; -import run.endive.runtime.StackFrame; -import run.endive.wasm.WasmEngineException; -import run.endive.wasm.types.Instruction; -import run.endive.wasm.types.OpCode; -import run.endive.wasm.types.Value; - -public final class SimdInterpreterMachine extends InterpreterMachine { - - public SimdInterpreterMachine(Instance instance) { - super(instance); - } - - @Override - protected void evalDefault( - MStack stack, - Instance instance, - Deque callStack, - Instruction instruction, - Operands operands) - throws WasmEngineException { - switch (instruction.opcode()) { - case OpCode.V128_CONST: - V128_CONST(stack, operands); - break; - case OpCode.V128_LOAD: - V128_LOAD(stack, instance, operands); - break; - case OpCode.V128_LOAD32_ZERO: - V128_LOAD32_ZERO(stack, instance, operands); - break; - case OpCode.V128_LOAD64_ZERO: - V128_LOAD64_ZERO(stack, instance, operands); - break; - case OpCode.V128_LOAD8_LANE: - LOAD_LANE( - stack, - operands, - (v, ptr) -> - v.reinterpretAsBytes() - .withLane( - (int) operands.get(3), - instance.memory((int) operands.get(2)).read(ptr)) - .reinterpretAsLongs()); - break; - case OpCode.V128_LOAD16_LANE: - LOAD_LANE( - stack, - operands, - (v, ptr) -> - v.reinterpretAsShorts() - .withLane( - (int) operands.get(3), - instance.memory((int) operands.get(2)) - .readShort(ptr)) - .reinterpretAsLongs()); - break; - case OpCode.V128_LOAD32_LANE: - LOAD_LANE( - stack, - operands, - (v, ptr) -> - v.reinterpretAsInts() - .withLane( - (int) operands.get(3), - instance.memory((int) operands.get(2)).readInt(ptr)) - .reinterpretAsLongs()); - break; - case OpCode.V128_LOAD64_LANE: - LOAD_LANE( - stack, - operands, - (v, ptr) -> - v.withLane( - (int) operands.get(3), - instance.memory((int) operands.get(2)).readLong(ptr))); - break; - case OpCode.V128_LOAD8x8_S: - V128_LOAD8x8_S(stack, instance, operands); - break; - case OpCode.V128_LOAD8x8_U: - V128_LOAD8x8_U(stack, instance, operands); - break; - case OpCode.V128_LOAD16x4_S: - V128_LOAD16x4_S(stack, instance, operands); - break; - case OpCode.V128_LOAD16x4_U: - V128_LOAD16x4_U(stack, instance, operands); - break; - case OpCode.V128_LOAD32x2_S: - V128_LOAD32x2_S(stack, instance, operands); - break; - case OpCode.V128_LOAD32x2_U: - V128_LOAD32x2_U(stack, instance, operands); - break; - case OpCode.V128_STORE: - V128_STORE(stack, instance, operands); - break; - case OpCode.V128_LOAD8_SPLAT: - V128_LOAD8_SPLAT(stack, instance, operands); - break; - case OpCode.V128_LOAD16_SPLAT: - V128_LOAD16_SPLAT(stack, instance, operands); - break; - case OpCode.V128_LOAD32_SPLAT: - V128_LOAD32_SPLAT(stack, instance, operands); - break; - case OpCode.V128_LOAD64_SPLAT: - V128_LOAD64_SPLAT(stack, instance, operands); - break; - case OpCode.I8x16_SHUFFLE: - I8x16_SHUFFLE(stack, operands); - break; - case OpCode.I8x16_SPLAT: - I8x16_SPLAT(stack); - break; - case OpCode.I16x8_SPLAT: - I16x8_SPLAT(stack); - break; - case OpCode.I32x4_SPLAT: - I32x4_SPLAT(stack); - break; - case OpCode.F32x4_SPLAT: - F32x4_SPLAT(stack); - break; - case OpCode.I64x2_SPLAT: - I64x2_SPLAT(stack); - break; - case OpCode.F64x2_SPLAT: - F64x2_SPLAT(stack); - break; - case OpCode.V128_STORE8_LANE: - STORE_LANE( - stack, - operands, - (v, ptr) -> - instance.memory((int) operands.get(2)) - .writeByte( - ptr, - v.reinterpretAsBytes() - .lane((int) operands.get(3)))); - break; - case OpCode.V128_STORE16_LANE: - STORE_LANE( - stack, - operands, - (v, ptr) -> - instance.memory((int) operands.get(2)) - .writeShort( - ptr, - v.reinterpretAsShorts() - .lane((int) operands.get(3)))); - break; - case OpCode.V128_STORE32_LANE: - STORE_LANE( - stack, - operands, - (v, ptr) -> - instance.memory((int) operands.get(2)) - .writeI32( - ptr, - v.reinterpretAsInts().lane((int) operands.get(3)))); - break; - case OpCode.V128_STORE64_LANE: - STORE_LANE( - stack, - operands, - (v, ptr) -> - instance.memory((int) operands.get(2)) - .writeLong( - ptr, - v.reinterpretAsLongs() - .lane((int) operands.get(3)))); - break; - case OpCode.I8x16_REPLACE_LANE: - REPLACE_LANE( - stack, - (v, val) -> - v.reinterpretAsBytes() - .withLane((int) operands.get(0), val.byteValue()) - .reinterpretAsLongs()); - break; - case OpCode.I16x8_REPLACE_LANE: - REPLACE_LANE( - stack, - (v, val) -> - v.reinterpretAsShorts() - .withLane((int) operands.get(0), val.shortValue()) - .reinterpretAsLongs()); - break; - case OpCode.I32x4_REPLACE_LANE: - REPLACE_LANE( - stack, - (v, val) -> - v.reinterpretAsInts() - .withLane((int) operands.get(0), val.intValue()) - .reinterpretAsLongs()); - break; - case OpCode.F32x4_REPLACE_LANE: - REPLACE_LANE( - stack, - (v, val) -> - v.reinterpretAsFloats() - .withLane((int) operands.get(0), Value.longToFloat(val)) - .reinterpretAsLongs()); - break; - case OpCode.I64x2_REPLACE_LANE: - REPLACE_LANE(stack, (v, val) -> v.withLane((int) operands.get(0), val)); - break; - case OpCode.F64x2_REPLACE_LANE: - REPLACE_LANE( - stack, - (v, val) -> - v.reinterpretAsDoubles() - .withLane((int) operands.get(0), Value.longToDouble(val)) - .reinterpretAsLongs()); - break; - case OpCode.I8x16_EXTRACT_LANE_U: - I8x16_EXTRACT_LANE_U(stack, operands); - break; - case OpCode.I16x8_EXTRACT_LANE_U: - I16x8_EXTRACT_LANE_U(stack, operands); - break; - case OpCode.I8x16_EXTRACT_LANE_S: - EXTRACT_LANE( - stack, - operands, - v -> (long) v.reinterpretAsBytes().lane((int) operands.get(0))); - break; - case OpCode.I16x8_EXTRACT_LANE_S: - EXTRACT_LANE( - stack, - operands, - v -> (long) v.reinterpretAsShorts().lane((int) operands.get(0))); - break; - case OpCode.I32x4_EXTRACT_LANE: - EXTRACT_LANE( - stack, - operands, - v -> (long) v.reinterpretAsInts().lane((int) operands.get(0))); - break; - case OpCode.F32x4_EXTRACT_LANE: - EXTRACT_LANE( - stack, - operands, - v -> - Value.floatToLong( - v.reinterpretAsFloats().lane((int) operands.get(0)))); - break; - case OpCode.I64x2_EXTRACT_LANE: - EXTRACT_LANE( - stack, operands, v -> v.reinterpretAsLongs().lane((int) operands.get(0))); - break; - case OpCode.F64x2_EXTRACT_LANE: - EXTRACT_LANE( - stack, - operands, - v -> - Value.doubleToLong( - v.reinterpretAsDoubles().lane((int) operands.get(0)))); - break; - case OpCode.V128_NOT: - V128_NOT(stack); - break; - case OpCode.V128_AND: - V128_BINOP(stack, (v1, v2) -> v1.and(v2)); - break; - case OpCode.V128_ANDNOT: - V128_BINOP(stack, (v1, v2) -> v1.not().and(v2)); - break; - case OpCode.V128_OR: - V128_BINOP(stack, (v1, v2) -> v1.or(v2)); - break; - case OpCode.V128_XOR: - V128_BINOP(stack, (v1, v2) -> v1.and(v2.not()).or(v1.not().and(v2))); - break; - case OpCode.V128_BITSELECT: - V128_BITSELECT(stack); - break; - case OpCode.V128_ANY_TRUE: - V128_ANY_TRUE(stack); - break; - case OpCode.I8x16_EQ: - BINOP(stack, LongVector::reinterpretAsBytes, (v1, v2) -> v1.eq(v2).toVector()); - break; - case OpCode.I16x8_EQ: - BINOP(stack, LongVector::reinterpretAsShorts, (v1, v2) -> v1.eq(v2).toVector()); - break; - case OpCode.I32x4_EQ: - BINOP(stack, LongVector::reinterpretAsInts, (v1, v2) -> v1.eq(v2).toVector()); - break; - case OpCode.F64x2_EQ: - BINOP(stack, LongVector::reinterpretAsDoubles, (v1, v2) -> v1.eq(v2).toVector()); - break; - case OpCode.I8x16_SUB: - I8x16_SUB(stack); - break; - case OpCode.I8x16_SWIZZLE: - I8x16_SWIZZLE(stack); - break; - case OpCode.I8x16_ALL_TRUE: - BOOL_OP( - stack, - v -> v.reinterpretAsBytes().compare(VectorOperators.NE, 0).allTrue()); - break; - case OpCode.I16x8_ALL_TRUE: - BOOL_OP( - stack, - v -> v.reinterpretAsShorts().compare(VectorOperators.NE, 0).allTrue()); - break; - case OpCode.I32x4_ALL_TRUE: - BOOL_OP(stack, v -> v.reinterpretAsInts().compare(VectorOperators.NE, 0).allTrue()); - break; - case OpCode.I64x2_ALL_TRUE: - BOOL_OP(stack, v -> v.compare(VectorOperators.NE, 0).allTrue()); - break; - case OpCode.I8x16_BITMASK: - BITMASK(stack, v -> v.reinterpretAsBytes().toLongArray()); - break; - case OpCode.I16x8_BITMASK: - BITMASK(stack, v -> v.reinterpretAsShorts().toLongArray()); - break; - case OpCode.I32x4_BITMASK: - BITMASK(stack, v -> v.reinterpretAsInts().toLongArray()); - break; - case OpCode.I64x2_BITMASK: - BITMASK(stack, v -> v.toLongArray()); - break; - case OpCode.I8x16_SHL: - SH( - stack, - (v, s) -> - v.reinterpretAsBytes() - .lanewise(VectorOperators.LSHL, s.byteValue()) - .reinterpretAsLongs()); - break; - case OpCode.I8x16_SHR_U: - SH( - stack, - (v, s) -> - v.reinterpretAsBytes() - .lanewise(VectorOperators.LSHR, s.byteValue()) - .reinterpretAsLongs()); - break; - case OpCode.I8x16_SHR_S: - SH( - stack, - (v, s) -> - v.reinterpretAsBytes() - .lanewise(VectorOperators.ASHR, s.byteValue()) - .reinterpretAsLongs()); - break; - case OpCode.I8x16_ADD: - BINOP(stack, LongVector::reinterpretAsBytes, (v1, v2) -> v1.add(v2)); - break; - case OpCode.I8x16_ADD_SAT_S: - I8x16_ADD_SAT_S(stack); - break; - case OpCode.I8x16_ADD_SAT_U: - I8x16_ADD_SAT_U(stack); - break; - case OpCode.I8x16_SUB_SAT_U: - I8x16_SUB_SAT_U(stack); - break; - case OpCode.I8x16_SUB_SAT_S: - I8x16_SUB_SAT_S(stack); - break; - case OpCode.I8x16_MIN_S: - BINOP(stack, LongVector::reinterpretAsBytes, (v1, v2) -> v1.min(v2)); - break; - case OpCode.I8x16_MAX_S: - BINOP(stack, LongVector::reinterpretAsBytes, (v1, v2) -> v1.max(v2)); - break; - case OpCode.I8x16_MAX_U: - I8x16( - stack, - (a, b) -> (long) Math.max(Byte.toUnsignedInt(a), Byte.toUnsignedInt(b))); - break; - case OpCode.I8x16_MIN_U: - I8x16( - stack, - (a, b) -> (long) Math.min(Byte.toUnsignedInt(a), Byte.toUnsignedInt(b))); - break; - case OpCode.I8x16_AVGR_U: - I8x16( - stack, - (a, b) -> (long) ((Byte.toUnsignedInt(a) + Byte.toUnsignedInt(b) + 1) / 2)); - break; - case OpCode.I8x16_ABS: - UNARY(stack, LongVector::reinterpretAsBytes, Vector::abs); - break; - case OpCode.I8x16_NEG: - UNARY(stack, LongVector::reinterpretAsBytes, Vector::neg); - break; - case OpCode.I8x16_NE: - BINOP( - stack, - LongVector::reinterpretAsBytes, - (v1, v2) -> v1.eq(v2).not().toVector()); - break; - case OpCode.I8x16_LT_S: - BINOP(stack, LongVector::reinterpretAsBytes, (v1, v2) -> v2.lt(v1).toVector()); - break; - case OpCode.I8x16_LT_U: - BINOP( - stack, - LongVector::reinterpretAsBytes, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_LT, v1).toVector()); - break; - case OpCode.I8x16_LE_S: - BINOP( - stack, - LongVector::reinterpretAsBytes, - (v1, v2) -> v2.compare(VectorOperators.LE, v1).toVector()); - break; - case OpCode.I8x16_LE_U: - BINOP( - stack, - LongVector::reinterpretAsBytes, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_LE, v1).toVector()); - break; - case OpCode.I8x16_GT_S: - BINOP( - stack, - LongVector::reinterpretAsBytes, - (v1, v2) -> v2.compare(VectorOperators.GT, v1).toVector()); - break; - case OpCode.I8x16_GT_U: - BINOP( - stack, - LongVector::reinterpretAsBytes, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_GT, v1).toVector()); - break; - case OpCode.I8x16_GE_S: - BINOP( - stack, - LongVector::reinterpretAsBytes, - (v1, v2) -> v2.compare(VectorOperators.GE, v1).toVector()); - break; - case OpCode.I8x16_GE_U: - BINOP( - stack, - LongVector::reinterpretAsBytes, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_GE, v1).toVector()); - break; - case OpCode.I8x16_POPCNT: - UNARY( - stack, - LongVector::reinterpretAsBytes, - v -> v.lanewise(VectorOperators.BIT_COUNT)); - break; - case OpCode.I16x8_NEG: - UNARY(stack, LongVector::reinterpretAsShorts, Vector::neg); - break; - case OpCode.I16x8_NE: - BINOP( - stack, - LongVector::reinterpretAsShorts, - (v1, v2) -> v1.eq(v2).not().toVector()); - break; - case OpCode.I16x8_LT_S: - BINOP(stack, LongVector::reinterpretAsShorts, (v1, v2) -> v2.lt(v1).toVector()); - break; - case OpCode.I16x8_LT_U: - BINOP( - stack, - LongVector::reinterpretAsShorts, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_LT, v1).toVector()); - break; - case OpCode.I16x8_LE_S: - BINOP( - stack, - LongVector::reinterpretAsShorts, - (v1, v2) -> v2.compare(VectorOperators.LE, v1).toVector()); - break; - case OpCode.I16x8_LE_U: - BINOP( - stack, - LongVector::reinterpretAsShorts, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_LE, v1).toVector()); - break; - case OpCode.I16x8_GT_S: - BINOP( - stack, - LongVector::reinterpretAsShorts, - (v1, v2) -> v2.compare(VectorOperators.GT, v1).toVector()); - break; - case OpCode.I16x8_GT_U: - BINOP( - stack, - LongVector::reinterpretAsShorts, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_GT, v1).toVector()); - break; - case OpCode.I16x8_GE_S: - BINOP( - stack, - LongVector::reinterpretAsShorts, - (v1, v2) -> v2.compare(VectorOperators.GE, v1).toVector()); - break; - case OpCode.I16x8_GE_U: - BINOP( - stack, - LongVector::reinterpretAsShorts, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_GE, v1).toVector()); - break; - case OpCode.I16x8_MIN_S: - BINOP(stack, LongVector::reinterpretAsShorts, (v1, v2) -> v1.min(v2)); - break; - case OpCode.I16x8_MAX_S: - BINOP(stack, LongVector::reinterpretAsShorts, (v1, v2) -> v1.max(v2)); - break; - case OpCode.I16x8_MAX_U: - I16x8( - stack, - (a, b) -> (long) Math.max(Short.toUnsignedInt(a), Short.toUnsignedInt(b))); - break; - case OpCode.I16x8_MIN_U: - I16x8( - stack, - (a, b) -> (long) Math.min(Short.toUnsignedInt(a), Short.toUnsignedInt(b))); - break; - case OpCode.I16x8_AVGR_U: - I16x8( - stack, - (a, b) -> - (long) ((Short.toUnsignedInt(a) + Short.toUnsignedInt(b) + 1) / 2)); - break; - case OpCode.I16x8_ADD: - BINOP(stack, LongVector::reinterpretAsShorts, (v1, v2) -> v1.add(v2)); - break; - case OpCode.I16x8_ADD_SAT_S: - I16x8_ADD_SAT_S(stack); - break; - case OpCode.I16x8_ADD_SAT_U: - I16x8_ADD_SAT_U(stack); - break; - case OpCode.I16x8_SUB_SAT_U: - I16x8_SUB_SAT_U(stack); - break; - case OpCode.I16x8_SUB_SAT_S: - I16x8_SUB_SAT_S(stack); - break; - case OpCode.I16x8_SUB: - BINOP(stack, LongVector::reinterpretAsShorts, (v1, v2) -> v2.sub(v1)); - break; - case OpCode.I16x8_ABS: - UNARY(stack, LongVector::reinterpretAsShorts, Vector::abs); - break; - case OpCode.I32x4_ABS: - UNARY(stack, LongVector::reinterpretAsInts, Vector::abs); - break; - case OpCode.I32x4_NEG: - UNARY(stack, LongVector::reinterpretAsInts, Vector::neg); - break; - case OpCode.I32x4_NE: - BINOP(stack, LongVector::reinterpretAsInts, (v1, v2) -> v1.eq(v2).not().toVector()); - break; - case OpCode.I32x4_LT_S: - BINOP(stack, LongVector::reinterpretAsInts, (v1, v2) -> v2.lt(v1).toVector()); - break; - case OpCode.I32x4_LT_U: - BINOP( - stack, - LongVector::reinterpretAsInts, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_LT, v1).toVector()); - break; - case OpCode.I32x4_LE_S: - BINOP( - stack, - LongVector::reinterpretAsInts, - (v1, v2) -> v2.compare(VectorOperators.LE, v1).toVector()); - break; - case OpCode.I32x4_LE_U: - BINOP( - stack, - LongVector::reinterpretAsInts, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_LE, v1).toVector()); - break; - case OpCode.I32x4_GT_S: - BINOP( - stack, - LongVector::reinterpretAsInts, - (v1, v2) -> v2.compare(VectorOperators.GT, v1).toVector()); - break; - case OpCode.I32x4_GT_U: - BINOP( - stack, - LongVector::reinterpretAsInts, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_GT, v1).toVector()); - break; - case OpCode.I32x4_GE_S: - BINOP( - stack, - LongVector::reinterpretAsInts, - (v1, v2) -> v2.compare(VectorOperators.GE, v1).toVector()); - break; - case OpCode.I32x4_GE_U: - BINOP( - stack, - LongVector::reinterpretAsInts, - (v1, v2) -> v2.compare(VectorOperators.UNSIGNED_GE, v1).toVector()); - break; - case OpCode.I32x4_MAX_S: - BINOP(stack, LongVector::reinterpretAsInts, (v1, v2) -> v1.max(v2)); - break; - case OpCode.I32x4_MAX_U: - I32x4( - stack, - (a, b) -> Math.max(Integer.toUnsignedLong(a), Integer.toUnsignedLong(b))); - break; - case OpCode.I32x4_MIN_S: - BINOP(stack, LongVector::reinterpretAsInts, (v1, v2) -> v1.min(v2)); - break; - case OpCode.I32x4_MIN_U: - I32x4( - stack, - (a, b) -> Math.min(Integer.toUnsignedLong(a), Integer.toUnsignedLong(b))); - break; - case I32x4_DOT_I16x8_S: - I32x4_DOT_I16x8_S(stack); - break; - case OpCode.I32x4_EXTMUL_LOW_I16x8_S: - I32x4_EXTMUL_LOW_I16x8_S(stack); - break; - case OpCode.I32x4_EXTMUL_HIGH_I16x8_S: - I32x4_EXTMUL_HIGH_I16x8_S(stack); - break; - case OpCode.I32x4_EXTMUL_LOW_I16x8_U: - I32x4_EXTMUL_LOW_I16x8_U(stack); - break; - case OpCode.I32x4_EXTMUL_HIGH_I16x8_U: - I32x4_EXTMUL_HIGH_I16x8_U(stack); - break; - case OpCode.I64x2_ABS: - UNARY(stack, LongVector::reinterpretAsLongs, Vector::abs); - break; - case OpCode.I64x2_NEG: - UNARY(stack, LongVector::reinterpretAsLongs, Vector::neg); - break; - case OpCode.I64x2_MUL: - BINOP(stack, LongVector::reinterpretAsLongs, (v1, v2) -> v1.mul(v2)); - break; - case OpCode.I64x2_EQ: - BINOP(stack, LongVector::reinterpretAsLongs, (v1, v2) -> v1.eq(v2).toVector()); - break; - case OpCode.I64x2_NE: - BINOP( - stack, - LongVector::reinterpretAsLongs, - (v1, v2) -> v1.compare(VectorOperators.NE, v2).toVector()); - break; - case OpCode.I64x2_LT_S: - BINOP( - stack, - LongVector::reinterpretAsLongs, - (v1, v2) -> v1.compare(VectorOperators.LT, v2).toVector()); - break; - case OpCode.I64x2_LE_S: - BINOP( - stack, - LongVector::reinterpretAsLongs, - (v1, v2) -> v1.compare(VectorOperators.LE, v2).toVector()); - break; - case OpCode.I64x2_GT_S: - BINOP( - stack, - LongVector::reinterpretAsLongs, - (v1, v2) -> v1.compare(VectorOperators.GT, v2).toVector()); - break; - case OpCode.I64x2_GE_S: - BINOP( - stack, - LongVector::reinterpretAsLongs, - (v1, v2) -> v1.compare(VectorOperators.GE, v2).toVector()); - break; - case OpCode.I64x2_SUB: - BINOP(stack, LongVector::reinterpretAsLongs, (v1, v2) -> v2.sub(v1)); - break; - case OpCode.I64x2_EXTMUL_LOW_I32x4_S: - I64x2_EXTMUL_LOW_I32x4_S(stack); - break; - case OpCode.I64x2_EXTMUL_HIGH_I32x4_S: - I64x2_EXTMUL_HIGH_I32x4_S(stack); - break; - case OpCode.I64x2_EXTMUL_LOW_I32x4_U: - I64x2_EXTMUL_LOW_I32x4_U(stack); - break; - case OpCode.I64x2_EXTMUL_HIGH_I32x4_U: - I64x2_EXTMUL_HIGH_I32x4_U(stack); - break; - case OpCode.F32x4_EQ: - F32x4(stack, (a, b) -> (OpcodeImpl.F32_EQ(a, b) > 0 ? 0xFFFFFFFFL : 0x0L)); - break; - case OpCode.F32x4_ABS: - UNARY(stack, LongVector::reinterpretAsFloats, v -> v.abs()); - break; - case OpCode.F32x4_MIN: - BINOP(stack, LongVector::reinterpretAsFloats, (v1, v2) -> v1.min(v2)); - break; - case OpCode.F32x4_MAX: - BINOP(stack, LongVector::reinterpretAsFloats, (v1, v2) -> v1.max(v2)); - break; - case OpCode.F32x4_ADD: - BINOP(stack, LongVector::reinterpretAsFloats, (v1, v2) -> v1.add(v2)); - break; - case OpCode.F32x4_SUB: - BINOP(stack, LongVector::reinterpretAsFloats, (v1, v2) -> v2.sub(v1)); - break; - case OpCode.F32x4_MUL: - BINOP(stack, LongVector::reinterpretAsFloats, (v1, v2) -> v1.mul(v2)); - break; - case OpCode.F32x4_DIV: - BINOP(stack, LongVector::reinterpretAsFloats, (v1, v2) -> v2.div(v1)); - break; - case OpCode.F32x4_NEG: - UNARY(stack, LongVector::reinterpretAsFloats, v -> v.neg()); - break; - case OpCode.F32x4_SQRT: - UNARY( - stack, - LongVector::reinterpretAsFloats, - v -> v.lanewise(VectorOperators.SQRT)); - break; - case OpCode.F32x4_LE: - F32x4(stack, (a, b) -> (le(b, a) ? 0xFFFFFFFFL : 0x0L)); - break; - case OpCode.F32x4_LT: - F32x4(stack, (a, b) -> (lt(b, a) ? 0xFFFFFFFFL : 0x0L)); - break; - case OpCode.F32x4_GE: - F32x4(stack, (a, b) -> (ge(b, a) ? 0xFFFFFFFFL : 0x0L)); - break; - case OpCode.F32x4_GT: - F32x4(stack, (a, b) -> (gt(b, a) ? 0xFFFFFFFFL : 0x0L)); - break; - case OpCode.F32x4_NE: - F32x4(stack, (a, b) -> (!equals(a, b) ? 0xFFFFFFFFL : 0x0L)); - break; - case OpCode.F32x4_PMIN: - F32x4(stack, (a, b) -> Value.floatToLong((a < b) ? a : b)); - break; - case OpCode.F32x4_PMAX: - F32x4(stack, (a, b) -> Value.floatToLong((a > b) ? a : b)); - break; - case OpCode.F32x4_CEIL: - F32x4(stack, v -> Value.floatToLong(OpcodeImpl.F32_CEIL(v))); - break; - case OpCode.F32x4_TRUNC: - F32x4(stack, v -> Value.floatToLong(OpcodeImpl.F32_TRUNC(v))); - break; - case OpCode.F32x4_FLOOR: - F32x4(stack, v -> Value.floatToLong(OpcodeImpl.F32_FLOOR(v))); - break; - case OpCode.F32x4_NEAREST: - F32x4(stack, v -> Value.floatToLong(OpcodeImpl.F32_NEAREST(v))); - break; - case OpCode.F64x2_ADD: - BINOP(stack, LongVector::reinterpretAsLongs, (v1, v2) -> v1.add(v2)); - break; - case OpCode.F64x2_MUL: - BINOP(stack, LongVector::reinterpretAsDoubles, (v1, v2) -> v1.mul(v2)); - break; - case OpCode.F64x2_DIV: - BINOP(stack, LongVector::reinterpretAsDoubles, (v1, v2) -> v1.div(v2)); - break; - case OpCode.F64x2_SUB: - BINOP(stack, LongVector::reinterpretAsDoubles, (v1, v2) -> v2.sub(v1)); - break; - case OpCode.F64x2_MIN: - BINOP(stack, LongVector::reinterpretAsDoubles, (v1, v2) -> v2.min(v1)); - break; - case OpCode.F64x2_MAX: - BINOP(stack, LongVector::reinterpretAsDoubles, (v1, v2) -> v2.max(v1)); - break; - case OpCode.F64x2_ABS: - UNARY( - stack, - LongVector::reinterpretAsDoubles, - v -> v.lanewise(VectorOperators.ABS)); - break; - case OpCode.F64x2_SQRT: - UNARY( - stack, - LongVector::reinterpretAsDoubles, - v -> v.lanewise(VectorOperators.SQRT)); - break; - case OpCode.F64x2_NEG: - UNARY( - stack, - LongVector::reinterpretAsDoubles, - v -> v.lanewise(VectorOperators.NEG)); - break; - case OpCode.F64x2_NE: - BINOP( - stack, - LongVector::reinterpretAsDoubles, - (v1, v2) -> v2.compare(VectorOperators.NE, v1).toVector()); - break; - case OpCode.F64x2_GE: - BINOP( - stack, - LongVector::reinterpretAsDoubles, - (v1, v2) -> v2.compare(VectorOperators.GE, v1).toVector()); - break; - case OpCode.F64x2_LT: - BINOP( - stack, - LongVector::reinterpretAsDoubles, - (v1, v2) -> v2.compare(VectorOperators.LT, v1).toVector()); - break; - case OpCode.F64x2_LE: - BINOP( - stack, - LongVector::reinterpretAsDoubles, - (v1, v2) -> v2.compare(VectorOperators.LE, v1).toVector()); - break; - case OpCode.F64x2_GT: - BINOP( - stack, - LongVector::reinterpretAsDoubles, - (v1, v2) -> v2.compare(VectorOperators.GT, v1).toVector()); - break; - case OpCode.F64x2_PMIN: - F64x2(stack, (a, b) -> Value.doubleToLong((a < b) ? a : b)); - break; - case OpCode.F64x2_PMAX: - F64x2(stack, (a, b) -> Value.doubleToLong((a > b) ? a : b)); - break; - case OpCode.F64x2_CEIL: - F64x2(stack, v -> Value.doubleToLong(OpcodeImpl.F64_CEIL(v))); - break; - case OpCode.F64x2_TRUNC: - F64x2(stack, v -> Value.doubleToLong(OpcodeImpl.F64_TRUNC(v))); - break; - case OpCode.F64x2_FLOOR: - F64x2(stack, v -> Value.doubleToLong(OpcodeImpl.F64_FLOOR(v))); - break; - case OpCode.F64x2_NEAREST: - F64x2(stack, v -> Value.doubleToLong(OpcodeImpl.F64_NEAREST(v))); - break; - case OpCode.I16x8_SHL: - SH( - stack, - (v, s) -> - v.reinterpretAsShorts() - .lanewise(VectorOperators.LSHL, s.shortValue()) - .reinterpretAsLongs()); - break; - case OpCode.I16x8_SHR_U: - SH( - stack, - (v, s) -> - v.reinterpretAsShorts() - .lanewise(VectorOperators.LSHR, s.shortValue()) - .reinterpretAsLongs()); - break; - case OpCode.I16x8_SHR_S: - SH( - stack, - (v, s) -> - v.reinterpretAsShorts() - .lanewise(VectorOperators.ASHR, s.shortValue()) - .reinterpretAsLongs()); - break; - case OpCode.I32x4_SHL: - SH( - stack, - (v, s) -> - v.reinterpretAsInts() - .lanewise(VectorOperators.LSHL, s.intValue()) - .reinterpretAsLongs()); - break; - case OpCode.I32x4_SHR_U: - SH( - stack, - (v, s) -> - v.reinterpretAsInts() - .lanewise(VectorOperators.LSHR, s.intValue()) - .reinterpretAsLongs()); - break; - case OpCode.I32x4_SHR_S: - SH( - stack, - (v, s) -> - v.reinterpretAsInts() - .lanewise(VectorOperators.ASHR, s.intValue()) - .reinterpretAsLongs()); - break; - case OpCode.I16x8_MUL: - BINOP(stack, LongVector::reinterpretAsShorts, (v1, v2) -> v2.mul(v1)); - break; - case OpCode.I32x4_ADD: - BINOP(stack, LongVector::reinterpretAsInts, (v1, v2) -> v2.add(v1)); - break; - case OpCode.I32x4_SUB: - BINOP(stack, LongVector::reinterpretAsInts, (v1, v2) -> v2.sub(v1)); - break; - case OpCode.I32x4_MUL: - BINOP(stack, LongVector::reinterpretAsInts, (v1, v2) -> v2.mul(v1)); - break; - case OpCode.I64x2_SHL: - SH(stack, (v, s) -> v.lanewise(VectorOperators.LSHL, s)); - break; - case OpCode.I64x2_SHR_U: - SH(stack, (v, s) -> v.lanewise(VectorOperators.LSHR, s)); - break; - case OpCode.I64x2_SHR_S: - SH(stack, (v, s) -> v.lanewise(VectorOperators.ASHR, s)); - break; - case OpCode.I64x2_ADD: - BINOP(stack, LongVector::reinterpretAsLongs, (v1, v2) -> v2.add(v1)); - break; - case OpCode.I32x4_TRUNC_SAT_F32X4_S: - I32x4_TRUNC_SAT_F32x4_S(stack); - break; - case OpCode.I32x4_TRUNC_SAT_F32X4_U: - I32x4_TRUNC_SAT_F32x4_U(stack); - break; - case OpCode.I32x4_TRUNC_SAT_F64x2_S_ZERO: - I32x4_TRUNC_SAT_F64x2_S_ZERO(stack); - break; - case OpCode.I32x4_TRUNC_SAT_F64x2_U_ZERO: - I32x4_TRUNC_SAT_F64x2_U_ZERO(stack); - break; - case OpCode.F32x4_CONVERT_I32x4_S: - F32x4_CONVERT_I32x4_S(stack); - break; - case OpCode.F32x4_CONVERT_I32x4_U: - F32x4_CONVERT_I32x4_U(stack); - break; - case OpCode.F64x2_PROMOTE_LOW_F32x4: - case OpCode.F64x2_CONVERT_LOW_I32x4_S: - case OpCode.F64x2_CONVERT_LOW_I32x4_U: - F64x2_PROMOTE_LOW_F32x4(stack); - break; - case OpCode.F32x4_DEMOTE_LOW_F64x2_ZERO: - F32x4_DEMOTE_LOW_F64x2_ZERO(stack); - break; - case OpCode.I8x16_NARROW_I16x8_S: - I8x16_NARROW_I16x8(stack, SimdInterpreterMachine::narrowS); - break; - case OpCode.I8x16_NARROW_I16x8_U: - I8x16_NARROW_I16x8(stack, SimdInterpreterMachine::narrowU); - break; - case OpCode.I16x8_EXTADD_PAIRWISE_I8x16_S: - I16x8_EXTADD_PAIRWISE_I8x16_S(stack); - break; - case OpCode.I16x8_EXTADD_PAIRWISE_I8x16_U: - I16x8_EXTADD_PAIRWISE_I8x16_U(stack); - break; - case OpCode.I16x8_EXTMUL_LOW_I8x16_S: - I16x8_EXTMUL_LOW_I8x16_S(stack); - break; - case OpCode.I16x8_EXTMUL_HIGH_I8x16_S: - I16x8_EXTMUL_HIGH_I8x16_S(stack); - break; - case OpCode.I16x8_EXTMUL_LOW_I8x16_U: - I16x8_EXTMUL_LOW_I8x16_U(stack); - break; - case OpCode.I16x8_EXTMUL_HIGH_I8x16_U: - I16x8_EXTMUL_HIGH_I8x16_U(stack); - break; - case OpCode.I16x8_Q15MULR_SAT_S: - I16x8_Q15MULR_SAT_S(stack); - break; - case OpCode.I16x8_NARROW_I32x4_S: - I16x8_NARROW_I32x4(stack, SimdInterpreterMachine::narrowS); - break; - case OpCode.I16x8_NARROW_I32x4_U: - I16x8_NARROW_I32x4(stack, SimdInterpreterMachine::narrowU); - break; - case OpCode.I16x8_EXTEND_LOW_I8x16_S: - I16x8_EXTEND_LOW_I8x16_S(stack); - break; - case OpCode.I16x8_EXTEND_HIGH_I8x16_S: - I16x8_EXTEND_HIGH_I8x16_S(stack); - break; - case OpCode.I16x8_EXTEND_LOW_I8x16_U: - I16x8_EXTEND_LOW_I8x16_U(stack); - break; - case OpCode.I16x8_EXTEND_HIGH_I8x16_U: - I16x8_EXTEND_HIGH_I8x16_U(stack); - break; - case OpCode.I32x4_EXTEND_LOW_I16x8_S: - I32x4_EXTEND_LOW_I16x8_S(stack); - break; - case OpCode.I32x4_EXTEND_HIGH_I16x8_S: - I32x4_EXTEND_HIGH_I16x8_S(stack); - break; - case OpCode.I32x4_EXTEND_LOW_I16x8_U: - I32x4_EXTEND_LOW_I16x8_U(stack); - break; - case OpCode.I32x4_EXTEND_HIGH_I16x8_U: - I32x4_EXTEND_HIGH_I16x8_U(stack); - break; - case OpCode.I32x4_EXTADD_PAIRWISE_I16x8_S: - I32x4_EXTADD_PAIRWISE_I16x8_S(stack); - break; - case OpCode.I32x4_EXTADD_PAIRWISE_I16x8_U: - I32x4_EXTADD_PAIRWISE_I16x8_U(stack); - break; - case OpCode.I64x2_EXTEND_LOW_I32x4_S: - I64x2_EXTEND_LOW_I32x4_S(stack); - break; - case OpCode.I64x2_EXTEND_HIGH_I32x4_S: - I64x2_EXTEND_HIGH_I32x4_S(stack); - break; - case OpCode.I64x2_EXTEND_LOW_I32x4_U: - I64x2_EXTEND_LOW_I32x4_U(stack); - break; - case OpCode.I64x2_EXTEND_HIGH_I32x4_U: - I64x2_EXTEND_HIGH_I32x4_U(stack); - break; - default: - super.evalDefault(stack, instance, callStack, instruction, operands); - break; - } - } - - private static void V128_CONST(MStack stack, Operands operands) { - stack.push(operands.get(0)); - stack.push(operands.get(1)); - } - - private static void V128_LOAD(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var valHigh = instance.memory((int) operands.get(2)).readLong(ptr); - var valLow = instance.memory((int) operands.get(2)).readLong(ptr + 8); - stack.push(valHigh); - stack.push(valLow); - } - - private static void V128_LOAD8_SPLAT(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var value = instance.memory((int) operands.get(2)).read(ptr); - var bytes = new long[16]; - Arrays.fill(bytes, value); - var vals = Value.i8ToVec(bytes); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD16_SPLAT(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var value = instance.memory((int) operands.get(2)).readShort(ptr); - var shorts = new long[8]; - Arrays.fill(shorts, value); - var vals = Value.i16ToVec(shorts); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD32_SPLAT(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var value = instance.memory((int) operands.get(2)).readInt(ptr); - var ints = new long[4]; - Arrays.fill(ints, value); - var vals = Value.i32ToVec(ints); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD64_SPLAT(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var value = instance.memory((int) operands.get(2)).readLong(ptr); - stack.push(value); - stack.push(value); - } - - private static void V128_LOAD8x8_S(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var bytes = new long[8]; - for (int i = 0; i < 8; i++) { - bytes[i] = instance.memory((int) operands.get(2)).read(ptr + i); - } - var vals = Value.i16ToVec(bytes); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD8x8_U(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var bytes = new long[8]; - for (int i = 0; i < 8; i++) { - bytes[i] = Byte.toUnsignedLong(instance.memory((int) operands.get(2)).read(ptr + i)); - } - var vals = Value.i16ToVec(bytes); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD16x4_S(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var bytes = new long[4]; - for (int i = 0; i < 4; i++) { - bytes[i] = instance.memory((int) operands.get(2)).readShort(ptr + (i * 2)); - } - var vals = Value.i32ToVec(bytes); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD16x4_U(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var bytes = new long[4]; - for (int i = 0; i < 4; i++) { - bytes[i] = - Short.toUnsignedLong( - instance.memory((int) operands.get(2)).readShort(ptr + (i * 2))); - } - var vals = Value.i32ToVec(bytes); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD32x2_S(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var bytes = new long[8]; - for (int i = 0; i < 2; i++) { - bytes[i] = instance.memory((int) operands.get(2)).readInt(ptr + (i * 4)); - } - var vals = Value.i64ToVec(bytes); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD32x2_U(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var bytes = new long[8]; - for (int i = 0; i < 2; i++) { - bytes[i] = - Integer.toUnsignedLong( - instance.memory((int) operands.get(2)).readInt(ptr + (i * 4))); - } - var vals = Value.i64ToVec(bytes); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD32_ZERO(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var val = instance.memory((int) operands.get(2)).readInt(ptr); - var vals = Value.i32ToVec(new long[] {val, 0, 0, 0}); - for (var v : vals) { - stack.push(v); - } - } - - private static void V128_LOAD64_ZERO(MStack stack, Instance instance, Operands operands) { - var ptr = readMemPtr(stack, operands); - var val = instance.memory((int) operands.get(2)).readLong(ptr); - var vals = Value.i64ToVec(new long[] {val, 0}); - for (var v : vals) { - stack.push(v); - } - } - - private static void LOAD_LANE( - MStack stack, Operands operands, BiFunction loadLane) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - var ptr = readMemPtr(stack, operands); - - var result = - loadLane.apply( - LongVector.fromArray( - LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0), - ptr) - .toArray(); - - for (var v : result) { - stack.push(v); - } - } - - private static void V128_STORE(MStack stack, Instance instance, Operands operands) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - var ptr = readMemPtr(stack, operands); - - instance.memory((int) operands.get(2)).writeLong(ptr, valLow); - instance.memory((int) operands.get(2)).writeLong(ptr + 8, valHigh); - } - - private static void I8x16_SHUFFLE(MStack stack, Operands operands) { - var v2High = stack.pop(); - var v2Low = stack.pop(); - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var select = - LongVector.fromArray( - LongVector.SPECIES_128, - new long[] {operands.get(0), operands.get(1)}, - 0) - .reinterpretAsBytes() - .toArray(); - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsBytes() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v2Low, v2High}, 0) - .reinterpretAsBytes() - .toArray(); - - var result = new byte[16]; - for (int i = 0; i < 16; i++) { - var s = select[i]; - if (s >= 16) { - result[i] = v2[s - 16]; - } else { - result[i] = v1[s]; - } - } - - var res = Value.bytesToVec(result); - for (var v : res) { - stack.push(v); - } - } - - private static void I8x16_SPLAT(MStack stack) { - var val = stack.pop(); - var vals = - Value.i8ToVec( - new long[] { - val, val, val, val, val, val, val, val, val, val, val, val, val, val, - val, val - }); - for (var v : vals) { - stack.push(v); - } - } - - private static void I16x8_SPLAT(MStack stack) { - var val = stack.pop(); - var vals = Value.i16ToVec(new long[] {val, val, val, val, val, val, val, val}); - for (var v : vals) { - stack.push(v); - } - } - - private static void I32x4_SPLAT(MStack stack) { - var val = stack.pop(); - var vals = Value.i32ToVec(new long[] {val, val, val, val}); - for (var v : vals) { - stack.push(v); - } - } - - private static void F32x4_SPLAT(MStack stack) { - var val = stack.pop(); - var vals = Value.f32ToVec(new long[] {val, val, val, val}); - for (var v : vals) { - stack.push(v); - } - } - - private static void I64x2_SPLAT(MStack stack) { - var val = stack.pop(); - var vals = Value.i64ToVec(new long[] {val, val}); - for (var v : vals) { - stack.push(v); - } - } - - private static void F64x2_SPLAT(MStack stack) { - var val = stack.pop(); - var vals = Value.f64ToVec(new long[] {val, val}); - for (var v : vals) { - stack.push(v); - } - } - - private static void STORE_LANE( - MStack stack, Operands operands, BiConsumer store) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - var ptr = readMemPtr(stack, operands); - - var result = LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0); - - store.accept(result, ptr); - } - - private static void EXTRACT_LANE( - MStack stack, Operands operands, Function extract) { - var offset = stack.size() - 2; - var result = - extract.apply(LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset)); - - // consume one element - stack.pop(); - stack.array()[stack.size() - 1] = result; - } - - private static void REPLACE_LANE( - MStack stack, BiFunction replace) { - var val = stack.pop(); - var offset = stack.size() - 2; - - var result = - replace.apply( - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset), - val) - .toArray(); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I8x16_EXTRACT_LANE_U(MStack stack, Operands operands) { - var offset = stack.size() - 2; - var result = - Byte.toUnsignedLong( - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .lane((int) operands.get(0))); - - // consume one element - stack.pop(); - stack.array()[stack.size() - 1] = result; - } - - private static void I16x8_EXTRACT_LANE_U(MStack stack, Operands operands) { - var offset = stack.size() - 2; - var result = - Short.toUnsignedLong( - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .lane((int) operands.get(0))); - - // consume one element - stack.pop(); - stack.array()[stack.size() - 1] = result; - } - - private static byte addSatU(byte a, byte b) { - int result = Byte.toUnsignedInt(a) + Byte.toUnsignedInt(b); - if (result >= 0xFF) { - return (byte) 0xFF; - } else { - return (byte) result; - } - } - - private static long addSatU(short a, short b) { - int result = Short.toUnsignedInt(a) + Short.toUnsignedInt(b); - return Math.min(result, 0xFFFF); - } - - private static byte subSatS(byte a, byte b) { - int result = a - b; - if (result > Byte.MAX_VALUE) { - return Byte.MAX_VALUE; - } else if (result < Byte.MIN_VALUE) { - return Byte.MIN_VALUE; - } else { - return (byte) result; - } - } - - private static long subSatS(short a, short b) { - int result = a - b; - if (result > Short.MAX_VALUE) { - return Short.MAX_VALUE; - } else if (result < Short.MIN_VALUE) { - return Short.MIN_VALUE; - } else { - return result; - } - } - - private static byte subSatU(byte a, byte b) { - int result = Byte.toUnsignedInt(a) - Byte.toUnsignedInt(b); - if (result < 0) { - return 0; - } else { - return (byte) result; - } - } - - private static short subSatU(short a, short b) { - int result = Short.toUnsignedInt(a) - Short.toUnsignedInt(b); - if (result < 0) { - return 0; - } else { - return (short) result; - } - } - - private static void I16x8_SUB_SAT_U(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsShorts() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var result = - Value.i16ToVec( - new long[] { - subSatU(v2[0], v1[0]), - subSatU(v2[1], v1[1]), - subSatU(v2[2], v1[2]), - subSatU(v2[3], v1[3]), - subSatU(v2[4], v1[4]), - subSatU(v2[5], v1[5]), - subSatU(v2[6], v1[6]), - subSatU(v2[7], v1[7]) - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I16x8_SUB_SAT_S(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsShorts() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var result = - Value.i16ToVec( - new long[] { - subSatS(v2[0], v1[0]), - subSatS(v2[1], v1[1]), - subSatS(v2[2], v1[2]), - subSatS(v2[3], v1[3]), - subSatS(v2[4], v1[4]), - subSatS(v2[5], v1[5]), - subSatS(v2[6], v1[6]), - subSatS(v2[7], v1[7]) - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I8x16_SUB_SAT_U(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsBytes() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var result = - Value.i8ToVec( - new long[] { - subSatU(v2[0], v1[0]), - subSatU(v2[1], v1[1]), - subSatU(v2[2], v1[2]), - subSatU(v2[3], v1[3]), - subSatU(v2[4], v1[4]), - subSatU(v2[5], v1[5]), - subSatU(v2[6], v1[6]), - subSatU(v2[7], v1[7]), - subSatU(v2[8], v1[8]), - subSatU(v2[9], v1[9]), - subSatU(v2[10], v1[10]), - subSatU(v2[11], v1[11]), - subSatU(v2[12], v1[12]), - subSatU(v2[13], v1[13]), - subSatU(v2[14], v1[14]), - subSatU(v2[15], v1[15]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I8x16_SUB_SAT_S(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsBytes() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var result = - Value.i8ToVec( - new long[] { - subSatS(v2[0], v1[0]), - subSatS(v2[1], v1[1]), - subSatS(v2[2], v1[2]), - subSatS(v2[3], v1[3]), - subSatS(v2[4], v1[4]), - subSatS(v2[5], v1[5]), - subSatS(v2[6], v1[6]), - subSatS(v2[7], v1[7]), - subSatS(v2[8], v1[8]), - subSatS(v2[9], v1[9]), - subSatS(v2[10], v1[10]), - subSatS(v2[11], v1[11]), - subSatS(v2[12], v1[12]), - subSatS(v2[13], v1[13]), - subSatS(v2[14], v1[14]), - subSatS(v2[15], v1[15]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I8x16_ADD_SAT_S(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsBytes() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var result = - Value.i8ToVec( - new long[] { - narrowS((short) (v1[0] + v2[0])), - narrowS((short) (v1[1] + v2[1])), - narrowS((short) (v1[2] + v2[2])), - narrowS((short) (v1[3] + v2[3])), - narrowS((short) (v1[4] + v2[4])), - narrowS((short) (v1[5] + v2[5])), - narrowS((short) (v1[6] + v2[6])), - narrowS((short) (v1[7] + v2[7])), - narrowS((short) (v1[8] + v2[8])), - narrowS((short) (v1[9] + v2[9])), - narrowS((short) (v1[10] + v2[10])), - narrowS((short) (v1[11] + v2[11])), - narrowS((short) (v1[12] + v2[12])), - narrowS((short) (v1[13] + v2[13])), - narrowS((short) (v1[14] + v2[14])), - narrowS((short) (v1[15] + v2[15])), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I8x16_ADD_SAT_U(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsBytes() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var result = - Value.i8ToVec( - new long[] { - addSatU(v1[0], v2[0]), - addSatU(v1[1], v2[1]), - addSatU(v1[2], v2[2]), - addSatU(v1[3], v2[3]), - addSatU(v1[4], v2[4]), - addSatU(v1[5], v2[5]), - addSatU(v1[6], v2[6]), - addSatU(v1[7], v2[7]), - addSatU(v1[8], v2[8]), - addSatU(v1[9], v2[9]), - addSatU(v1[10], v2[10]), - addSatU(v1[11], v2[11]), - addSatU(v1[12], v2[12]), - addSatU(v1[13], v2[13]), - addSatU(v1[14], v2[14]), - addSatU(v1[15], v2[15]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I16x8_ADD_SAT_S(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsShorts() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var result = - Value.i16ToVec( - new long[] { - narrowS(v1[0] + v2[0]), - narrowS(v1[1] + v2[1]), - narrowS(v1[2] + v2[2]), - narrowS(v1[3] + v2[3]), - narrowS(v1[4] + v2[4]), - narrowS(v1[5] + v2[5]), - narrowS(v1[6] + v2[6]), - narrowS(v1[7] + v2[7]) - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I16x8_ADD_SAT_U(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsShorts() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var result = - Value.i16ToVec( - new long[] { - addSatU(v1[0], v2[0]), - addSatU(v1[1], v2[1]), - addSatU(v1[2], v2[2]), - addSatU(v1[3], v2[3]), - addSatU(v1[4], v2[4]), - addSatU(v1[5], v2[5]), - addSatU(v1[6], v2[6]), - addSatU(v1[7], v2[7]) - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void UNARY( - MStack stack, Function reinterpret, Function fun) { - var offset = stack.size() - 2; - var v = - reinterpret.apply( - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset)); - - var result = fun.apply(v).reinterpretAsLongs().toArray(); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void BINOP( - MStack stack, - Function reinterpret, - BiFunction fun) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - reinterpret.apply( - LongVector.fromArray( - LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0)); - var v2 = - reinterpret.apply( - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset)); - - var result = fun.apply(v1, v2).reinterpretAsLongs().toArray(); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static boolean lt(float a, float b) { - return a < b; - } - - private static boolean le(float a, float b) { - return a <= b; - } - - private static boolean gt(float a, float b) { - return a > b; - } - - private static boolean ge(float a, float b) { - return a >= b; - } - - private static boolean equals(float a, float b) { - return a == b; - } - - private static void F32x4(MStack stack, BiFunction fn) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsFloats() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsFloats() - .toArray(); - - var result = - Value.i32ToVec( - new long[] { - fn.apply(v1[0], v2[0]), - fn.apply(v1[1], v2[1]), - fn.apply(v1[2], v2[2]), - fn.apply(v1[3], v2[3]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void F32x4(MStack stack, Function fn) { - var offset = stack.size() - 2; - - var v = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsFloats() - .toArray(); - - var result = - Value.i32ToVec( - new long[] { - fn.apply(v[0]), fn.apply(v[1]), fn.apply(v[2]), fn.apply(v[3]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void F64x2(MStack stack, BiFunction fn) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsDoubles() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsDoubles() - .toArray(); - - var result = - Value.i64ToVec( - new long[] { - fn.apply(v1[0], v2[0]), fn.apply(v1[1], v2[1]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void F64x2(MStack stack, Function fn) { - var offset = stack.size() - 2; - - var v = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsDoubles() - .toArray(); - - var result = - Value.i64ToVec( - new long[] { - fn.apply(v[0]), fn.apply(v[1]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I8x16_SUB(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsBytes(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes(); - - var result = v2.sub(v1).reinterpretAsLongs().toArray(); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void SH(MStack stack, BiFunction shl) { - var s = stack.pop(); - var offset = stack.size() - 2; - - var result = - shl.apply(LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset), s) - .toArray(); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void BOOL_OP(MStack stack, Function condition) { - var vHigh = stack.pop(); - var vLow = stack.pop(); - - var result = - condition.apply( - LongVector.fromArray(LongVector.SPECIES_128, new long[] {vLow, vHigh}, 0)); - - if (result) { - stack.push(BitOps.TRUE); - } else { - stack.push(BitOps.FALSE); - } - } - - private static void BITMASK(MStack stack, Function reduce) { - var vHigh = stack.pop(); - var vLow = stack.pop(); - - var vals = - reduce.apply( - LongVector.fromArray(LongVector.SPECIES_128, new long[] {vLow, vHigh}, 0)); - - var result = 0L; - for (int i = 0; i < vals.length; i++) { - if (vals[i] < 0) { - result |= 1L << i; - } - } - - stack.push(result); - } - - private static void V128_NOT(MStack stack) { - var offset = stack.size() - 2; - var not = LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset).not(); - var res = not.toArray(); - - System.arraycopy(res, 0, stack.array(), offset, 2); - } - - private static void V128_BINOP( - MStack stack, BiFunction binop) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - var offset = stack.size() - 2; - var v1 = LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0); - var v2 = LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset); - var res = binop.apply(v1, v2).toArray(); - - System.arraycopy(res, 0, stack.array(), offset, 2); - } - - private static void V128_ANY_TRUE(MStack stack) { - var vHigh = stack.pop(); - var vLow = stack.pop(); - - // TODO: check the spec! - if (vLow != 0L || vHigh != 0L) { - stack.push(BitOps.TRUE); - } else { - stack.push(BitOps.FALSE); - } - } - - private static void V128_BITSELECT(MStack stack) { - var cHi = stack.pop(); - var cLo = stack.pop(); - var x1Hi = stack.pop(); - var x1Lo = stack.pop(); - - var m = LongVector.fromArray(LongVector.SPECIES_128, new long[] {cLo, cHi}, 0); - var v1 = LongVector.fromArray(LongVector.SPECIES_128, new long[] {x1Lo, x1Hi}, 0); - var v2 = LongVector.fromArray(LongVector.SPECIES_128, stack.array(), stack.size() - 2); - - var result = v1.bitwiseBlend(v2, m).toArray(); - - System.arraycopy(result, 0, stack.array(), stack.size() - 2, 2); - } - - private static void I32x4_DOT_I16x8_S(MStack stack) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - int offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsShorts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var result = - Value.i32ToVec( - new long[] { - (v1[0] * v2[0]) + (v1[4] * v2[4]), - (v1[1] * v2[1]) + (v1[5] * v2[5]), - (v1[2] * v2[2]) + (v1[6] * v2[6]), - (v1[3] * v2[3]) + (v1[7] * v2[7]), - }); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I8x16(MStack stack, BiFunction op) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - int offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsBytes() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - long[] result = new long[16]; - for (int i = 0; i < 16; i++) { - result[i] = op.apply(v1[i], v2[i]); - } - System.arraycopy(Value.i8ToVec(result), 0, stack.array(), offset, 2); - } - - private static void I16x8(MStack stack, BiFunction op) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - int offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsShorts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - long[] result = new long[8]; - for (int i = 0; i < 8; i++) { - result[i] = op.apply(v1[i], v2[i]); - } - System.arraycopy(Value.i16ToVec(result), 0, stack.array(), offset, 2); - } - - private static void I32x4(MStack stack, BiFunction op) { - var v1High = stack.pop(); - var v1Low = stack.pop(); - - int offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {v1Low, v1High}, 0) - .reinterpretAsInts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsInts() - .toArray(); - - long[] result = new long[4]; - for (int i = 0; i < 4; i++) { - result[i] = op.apply(v1[i], v2[i]); - } - System.arraycopy(Value.i32ToVec(result), 0, stack.array(), offset, 2); - } - - private static void I32x4_TRUNC_SAT_F32x4_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - long resultLow = 0L; - long resultHigh = 0L; - - for (int i = 0; i < 2; i++) { - var shift = i * 32L; - resultHigh |= - (OpcodeImpl.I32_TRUNC_SAT_F32_S( - Float.intBitsToFloat( - (int) ((valHigh >> shift) & 0xFFFFFFFFL))) - & 0xFFFFFFFFL) - << shift; - resultLow |= - (OpcodeImpl.I32_TRUNC_SAT_F32_S( - Float.intBitsToFloat( - (int) ((valLow >> shift) & 0xFFFFFFFFL))) - & 0xFFFFFFFFL) - << shift; - } - - stack.push(resultLow); - stack.push(resultHigh); - } - - private static void I32x4_TRUNC_SAT_F32x4_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - long resultLow = 0L; - long resultHigh = 0L; - - for (int i = 0; i < 2; i++) { - var shift = i * 32L; - resultHigh |= - (OpcodeImpl.I32_TRUNC_SAT_F32_U( - Float.intBitsToFloat( - (int) ((valHigh >> shift) & 0xFFFFFFFFL))) - & 0xFFFFFFFFL) - << shift; - resultLow |= - (OpcodeImpl.I32_TRUNC_SAT_F32_U( - Float.intBitsToFloat( - (int) ((valLow >> shift) & 0xFFFFFFFFL))) - & 0xFFFFFFFFL) - << shift; - } - - stack.push(resultLow); - stack.push(resultHigh); - } - - private static void I32x4_TRUNC_SAT_F64x2_S_ZERO(MStack stack) { - var offset = stack.size() - 2; - - var v = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsDoubles() - .toArray(); - - var result = - Value.i32ToVec( - new long[] { - OpcodeImpl.I32_TRUNC_SAT_F64_S(v[0]), - OpcodeImpl.I32_TRUNC_SAT_F64_S(v[1]), - 0L, - 0L - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I32x4_TRUNC_SAT_F64x2_U_ZERO(MStack stack) { - var offset = stack.size() - 2; - - var v = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsDoubles() - .toArray(); - - var result = - Value.i32ToVec( - new long[] { - OpcodeImpl.I32_TRUNC_SAT_F64_U(v[0]), - OpcodeImpl.I32_TRUNC_SAT_F64_U(v[1]), - 0L, - 0L - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void F32x4_CONVERT_I32x4_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - long resultLow = 0L; - long resultHigh = 0L; - - for (int i = 0; i < 2; i++) { - var shift = i * 32L; - resultHigh |= - (Float.floatToIntBits( - OpcodeImpl.F32_CONVERT_I32_U( - (int) ((valHigh >> shift) & 0xFFFFFFFFL))) - & 0xFFFFFFFFL) - << shift; - resultLow |= - (Float.floatToIntBits( - OpcodeImpl.F32_CONVERT_I32_U( - (int) ((valLow >> shift) & 0xFFFFFFFFL))) - & 0xFFFFFFFFL) - << shift; - } - - stack.push(resultLow); - stack.push(resultHigh); - } - - private static void F32x4_CONVERT_I32x4_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - long resultLow = 0L; - long resultHigh = 0L; - - for (int i = 0; i < 2; i++) { - var shift = i * 32L; - resultHigh |= - (Float.floatToIntBits( - OpcodeImpl.F32_CONVERT_I32_S( - (int) ((valHigh >> shift) & 0xFFFFFFFFL))) - & 0xFFFFFFFFL) - << shift; - resultLow |= - (Float.floatToIntBits( - OpcodeImpl.F32_CONVERT_I32_S( - (int) ((valLow >> shift) & 0xFFFFFFFFL))) - & 0xFFFFFFFFL) - << shift; - } - - stack.push(resultLow); - stack.push(resultHigh); - } - - private static void F64x2_PROMOTE_LOW_F32x4(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsFloats() - .toArray(); - - stack.push(Value.floatToLong(v[0])); - stack.push(Value.floatToLong(v[1])); - } - - private static byte narrowS(short a) { - if (a < Byte.MIN_VALUE) { - return Byte.MIN_VALUE; - } else if (a > Byte.MAX_VALUE) { - return Byte.MAX_VALUE; - } else { - return (byte) a; - } - } - - private static short narrowS(int a) { - if (a < Short.MIN_VALUE) { - return Short.MIN_VALUE; - } else if (a > Short.MAX_VALUE) { - return Short.MAX_VALUE; - } else { - return (short) a; - } - } - - private static byte narrowU(short a) { - if (a < 0) { - return 0; - } else if (a > 255) { - return -1; - } else { - return (byte) a; - } - } - - private static short narrowU(int a) { - if (a < 0) { - return 0; - } else if (a > 65535) { - return -1; - } else { - return (short) a; - } - } - - private static void I8x16_NARROW_I16x8(MStack stack, Function narrow) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsShorts() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var result = - Value.i8ToVec( - new long[] { - narrow.apply(v2[0]), - narrow.apply(v2[1]), - narrow.apply(v2[2]), - narrow.apply(v2[3]), - narrow.apply(v2[4]), - narrow.apply(v2[5]), - narrow.apply(v2[6]), - narrow.apply(v2[7]), - narrow.apply(v1[0]), - narrow.apply(v1[1]), - narrow.apply(v1[2]), - narrow.apply(v1[3]), - narrow.apply(v1[4]), - narrow.apply(v1[5]), - narrow.apply(v1[6]), - narrow.apply(v1[7]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I16x8_NARROW_I32x4(MStack stack, Function narrow) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsInts() - .toArray(); - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsInts() - .toArray(); - - var result = - Value.i16ToVec( - new long[] { - narrow.apply(v2[0]), - narrow.apply(v2[1]), - narrow.apply(v2[2]), - narrow.apply(v2[3]), - narrow.apply(v1[0]), - narrow.apply(v1[1]), - narrow.apply(v1[2]), - narrow.apply(v1[3]), - }); - - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void F32x4_DEMOTE_LOW_F64x2_ZERO(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsDoubles() - .toArray(); - - var vals = - Value.f32ToVec( - new long[] { - Value.floatToLong((float) v[0]), Value.floatToLong((float) v[1]), 0, 0 - }); - for (var val : vals) { - stack.push(val); - } - } - - private static void I16x8_EXTMUL_LOW_I8x16_S(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsBytes() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var res = new long[8]; - for (int i = 0; i < 8; i++) { - res[i] = v1[i] * v2[i]; - } - var result = Value.i16ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I16x8_EXTMUL_HIGH_I8x16_S(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsBytes() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var res = new long[8]; - for (int i = 0; i < 8; i++) { - res[i] = v1[8 + i] * v2[8 + i]; - } - var result = Value.i16ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I16x8_EXTMUL_LOW_I8x16_U(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsBytes() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var res = new long[8]; - for (int i = 0; i < 8; i++) { - res[i] = Byte.toUnsignedLong(v1[i]) * Byte.toUnsignedLong(v2[i]); - } - var result = Value.i16ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I16x8_EXTMUL_HIGH_I8x16_U(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsBytes() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var res = new long[8]; - for (int i = 0; i < 8; i++) { - res[i] = Byte.toUnsignedLong(v1[8 + i]) * Byte.toUnsignedLong(v2[8 + i]); - } - var result = Value.i16ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I32x4_EXTMUL_LOW_I16x8_S(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsShorts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var res = new long[4]; - for (int i = 0; i < 4; i++) { - res[i] = v1[i] * v2[i]; - } - var result = Value.i32ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I32x4_EXTMUL_HIGH_I16x8_S(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsShorts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var res = new long[4]; - for (int i = 0; i < 4; i++) { - res[i] = v1[4 + i] * v2[4 + i]; - } - var result = Value.i32ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I32x4_EXTMUL_LOW_I16x8_U(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsShorts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var res = new long[4]; - for (int i = 0; i < 4; i++) { - res[i] = Short.toUnsignedLong(v1[i]) * Short.toUnsignedLong(v2[i]); - } - var result = Value.i32ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I32x4_EXTMUL_HIGH_I16x8_U(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsShorts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var res = new long[4]; - for (int i = 0; i < 4; i++) { - res[i] = Short.toUnsignedLong(v1[4 + i]) * Short.toUnsignedLong(v2[4 + i]); - } - var result = Value.i32ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I64x2_EXTMUL_LOW_I32x4_S(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsInts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsInts() - .toArray(); - - var res = - new long[] { - ((long) v1[0]) * v2[0], ((long) v1[1]) * v2[1], - }; - var result = Value.i64ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I64x2_EXTMUL_HIGH_I32x4_S(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsInts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsInts() - .toArray(); - - var res = - new long[] { - ((long) v1[2]) * v2[2], ((long) v1[3]) * v2[3], - }; - var result = Value.i64ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I64x2_EXTMUL_LOW_I32x4_U(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsInts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsInts() - .toArray(); - - var res = - new long[] { - Integer.toUnsignedLong(v1[0]) * Integer.toUnsignedLong(v2[0]), - Integer.toUnsignedLong(v1[1]) * Integer.toUnsignedLong(v2[1]), - }; - var result = Value.i64ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I64x2_EXTMUL_HIGH_I32x4_U(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsBytes() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsBytes() - .toArray(); - - var res = - new long[] { - Integer.toUnsignedLong(v1[2]) * Integer.toUnsignedLong(v2[2]), - Integer.toUnsignedLong(v1[3]) * Integer.toUnsignedLong(v2[3]), - }; - - var result = Value.i64ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static short roundQ15(int a) { - if (a < Short.MIN_VALUE) { - return Short.MIN_VALUE; - } else if (a > Short.MAX_VALUE) { - return Short.MAX_VALUE; - } else { - return (short) a; - } - } - - private static void I16x8_Q15MULR_SAT_S(MStack stack) { - var val1High = stack.pop(); - var val1Low = stack.pop(); - var offset = stack.size() - 2; - - var v1 = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {val1Low, val1High}, 0) - .reinterpretAsShorts() - .toArray(); - - var v2 = - LongVector.fromArray(LongVector.SPECIES_128, stack.array(), offset) - .reinterpretAsShorts() - .toArray(); - - var res = new long[8]; - for (int i = 0; i < 8; i++) { - // https://github.com/argon-lang/jawawasm/blob/0193bc0bbd1157d0de4d83c8139d317e638d44ed/engine/src/main/java/dev/argon/jawawasm/engine/StackFrame.java#L903-L910 - res[i] = roundQ15((v1[i] * v2[i] + (1 << 14)) >> 15); - } - var result = Value.i16ToVec(res); - System.arraycopy(result, 0, stack.array(), offset, 2); - } - - private static void I32x4_EXTADD_PAIRWISE_I16x8_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsShorts() - .toArray(); - - var vals = Value.i32ToVec(new long[] {v[0] + v[1], v[2] + v[3], v[4] + v[5], v[6] + v[7]}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I32x4_EXTADD_PAIRWISE_I16x8_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsShorts() - .toArray(); - - var vals = - Value.i32ToVec( - new long[] { - Short.toUnsignedLong(v[0]) + Short.toUnsignedLong(v[1]), - Short.toUnsignedLong(v[2]) + Short.toUnsignedLong(v[3]), - Short.toUnsignedLong(v[4]) + Short.toUnsignedLong(v[5]), - Short.toUnsignedLong(v[6]) + Short.toUnsignedLong(v[7]) - }); - for (var val : vals) { - stack.push(val); - } - } - - private static void I16x8_EXTADD_PAIRWISE_I8x16_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsBytes() - .toArray(); - - var vals = - Value.i16ToVec( - new long[] { - v[0] + v[1], - v[2] + v[3], - v[4] + v[5], - v[6] + v[7], - v[8] + v[9], - v[10] + v[11], - v[12] + v[13], - v[14] + v[15], - }); - for (var val : vals) { - stack.push(val); - } - } - - private static void I16x8_EXTADD_PAIRWISE_I8x16_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsBytes() - .toArray(); - - var vals = - Value.i16ToVec( - new long[] { - Byte.toUnsignedLong(v[0]) + Byte.toUnsignedLong(v[1]), - Byte.toUnsignedLong(v[2]) + Byte.toUnsignedLong(v[3]), - Byte.toUnsignedLong(v[4]) + Byte.toUnsignedLong(v[5]), - Byte.toUnsignedLong(v[6]) + Byte.toUnsignedLong(v[7]), - Byte.toUnsignedLong(v[8]) + Byte.toUnsignedLong(v[9]), - Byte.toUnsignedLong(v[10]) + Byte.toUnsignedLong(v[11]), - Byte.toUnsignedLong(v[12]) + Byte.toUnsignedLong(v[13]), - Byte.toUnsignedLong(v[14]) + Byte.toUnsignedLong(v[15]), - }); - for (var val : vals) { - stack.push(val); - } - } - - private static void I16x8_EXTEND_LOW_I8x16_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsBytes() - .toArray(); - - var vals = Value.i16ToVec(new long[] {v[0], v[1], v[2], v[3], v[4], v[5], v[6], v[7]}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I16x8_EXTEND_HIGH_I8x16_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsBytes() - .toArray(); - - var vals = - Value.i16ToVec(new long[] {v[8], v[9], v[10], v[11], v[12], v[13], v[14], v[15]}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I32x4_EXTEND_LOW_I16x8_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsShorts() - .toArray(); - - var vals = Value.i32ToVec(new long[] {v[0], v[1], v[2], v[3]}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I32x4_EXTEND_HIGH_I16x8_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsShorts() - .toArray(); - - var vals = Value.i32ToVec(new long[] {v[4], v[5], v[6], v[7]}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I32x4_EXTEND_LOW_I16x8_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsShorts() - .toArray(); - - var vals = - Value.i32ToVec( - new long[] { - Short.toUnsignedLong(v[0]), - Short.toUnsignedLong(v[1]), - Short.toUnsignedLong(v[2]), - Short.toUnsignedLong(v[3]) - }); - for (var val : vals) { - stack.push(val); - } - } - - private static void I32x4_EXTEND_HIGH_I16x8_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsShorts() - .toArray(); - - var vals = - Value.i32ToVec( - new long[] { - Short.toUnsignedLong(v[4]), - Short.toUnsignedLong(v[5]), - Short.toUnsignedLong(v[6]), - Short.toUnsignedLong(v[7]) - }); - for (var val : vals) { - stack.push(val); - } - } - - private static void I64x2_EXTEND_HIGH_I32x4_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsInts() - .toArray(); - - var vals = Value.i64ToVec(new long[] {v[2], v[3]}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I64x2_EXTEND_LOW_I32x4_S(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsInts() - .toArray(); - - var vals = Value.i64ToVec(new long[] {v[0], v[1]}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I64x2_EXTEND_HIGH_I32x4_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsInts() - .toArray(); - - var vals = - Value.i64ToVec( - new long[] {Integer.toUnsignedLong(v[2]), Integer.toUnsignedLong(v[3])}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I64x2_EXTEND_LOW_I32x4_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsInts() - .toArray(); - - var vals = - Value.i64ToVec( - new long[] {Integer.toUnsignedLong(v[0]), Integer.toUnsignedLong(v[1])}); - for (var val : vals) { - stack.push(val); - } - } - - private static void I16x8_EXTEND_LOW_I8x16_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsBytes() - .toArray(); - - var vals = - Value.i16ToVec( - new long[] { - Byte.toUnsignedLong(v[0]), - Byte.toUnsignedLong(v[1]), - Byte.toUnsignedLong(v[2]), - Byte.toUnsignedLong(v[3]), - Byte.toUnsignedLong(v[4]), - Byte.toUnsignedLong(v[5]), - Byte.toUnsignedLong(v[6]), - Byte.toUnsignedLong(v[7]) - }); - for (var val : vals) { - stack.push(val); - } - } - - private static void I16x8_EXTEND_HIGH_I8x16_U(MStack stack) { - var valHigh = stack.pop(); - var valLow = stack.pop(); - - var v = - LongVector.fromArray(LongVector.SPECIES_128, new long[] {valLow, valHigh}, 0) - .reinterpretAsBytes() - .toArray(); - - var vals = - Value.i16ToVec( - new long[] { - Byte.toUnsignedLong(v[8]), - Byte.toUnsignedLong(v[9]), - Byte.toUnsignedLong(v[10]), - Byte.toUnsignedLong(v[11]), - Byte.toUnsignedLong(v[12]), - Byte.toUnsignedLong(v[13]), - Byte.toUnsignedLong(v[14]), - Byte.toUnsignedLong(v[15]) - }); - for (var val : vals) { - stack.push(val); - } - } - - private static void I8x16_SWIZZLE(MStack stack) { - var idxHigh = stack.pop(); - var idxLow = stack.pop(); - var baseHigh = stack.pop(); - var baseLow = stack.pop(); - - long resultLow = 0L; - long resultHigh = 0L; - - for (int i = 0; i < 16; i++) { - long id; - if (i < 8) { - id = (idxLow >> (i * 8)) & 0xFFL; - } else { - id = (idxHigh >> ((i - 8) * 8)) & 0xFFL; - } - - long base; - if (id < 8) { - base = (baseLow >> (id * 8)) & 0xFFL; - } else if (id < 16) { - base = (baseHigh >> ((id - 8) * 8)) & 0xFFL; - } else { - base = 0x00L; - } - - if (i < 8) { - resultLow |= base << (i * 8); - } else { - resultHigh |= base << ((i - 8) * 8); - } - } - - stack.push(resultLow); - stack.push(resultHigh); - } -} diff --git a/simd/src/test/java/run/endive/simd/BasicSimdTest.java b/simd/src/test/java/run/endive/simd/BasicSimdTest.java deleted file mode 100644 index f5a533461..000000000 --- a/simd/src/test/java/run/endive/simd/BasicSimdTest.java +++ /dev/null @@ -1,68 +0,0 @@ -package run.endive.simd; - -import static org.junit.jupiter.api.Assertions.assertArrayEquals; -import static org.junit.jupiter.api.Assertions.assertEquals; -import static org.junit.jupiter.api.Assertions.assertThrows; - -import java.util.List; -import org.junit.jupiter.api.Test; -import run.endive.corpus.CorpusResources; -import run.endive.runtime.Instance; -import run.endive.runtime.WasmRuntimeException; -import run.endive.wasm.Parser; - -public class BasicSimdTest { - - @Test - public void shouldRunBasicExample() { - // from: https://blog.dkwr.de/development/wasm-simd-operations/ - var instance = - Instance.builder( - Parser.parse( - CorpusResources.getResource( - "compiled/simd-example.wat.wasm"))) - .withMachineFactory(SimdInterpreterMachine::new) - .build(); - var main = instance.export("main"); - var result = main.apply()[0]; - assertEquals(6L, result); - } - - @Test - public void shouldRoundTripV128Locals() { - var instance = - Instance.builder( - Parser.parse( - CorpusResources.getResource( - "compiled/simd-locals.wat.wasm"))) - .withMachineFactory(SimdInterpreterMachine::new) - .build(); - assertEquals(10L, instance.export("local_roundtrip").apply()[0]); - assertEquals(7L, instance.export("local_roundtrip_lane0").apply()[0]); - assertEquals(10L, instance.export("local_tee").apply()[0]); - assertEquals(7L, instance.export("local_tee_get").apply()[0]); - } - - @Test - public void shouldTrapOnStoreEffectiveAddressOverflow() { - var instance = - Instance.builder( - Parser.parse( - CorpusResources.getResource( - "compiled/simd-store-offset-wrap.wat.wasm"))) - .withMachineFactory(SimdInterpreterMachine::new) - .build(); - var memory = instance.memory(); - - for (var name : List.of("v128_store", "v128_store8_lane")) { - var store = instance.export(name); - assertThrows(WasmRuntimeException.class, () -> store.apply(-1L), name); - assertThrows(WasmRuntimeException.class, () -> store.apply(0xFFFFFFFFL), name); - } - for (var name : List.of("v128_store_max_offset", "v128_store8_lane_max_offset")) { - var store = instance.export(name); - assertThrows(WasmRuntimeException.class, () -> store.apply(1L), name); - } - assertArrayEquals(new byte[16], memory.readBytes(0, 16)); - } -} diff --git a/wasm-corpus/src/main/resources/compiled/simd-memory-bounds.wat.wasm b/wasm-corpus/src/main/resources/compiled/simd-memory-bounds.wat.wasm new file mode 100644 index 0000000000000000000000000000000000000000..5ed912ac37a6d091778fcf09b2ab495f8a104d48 GIT binary patch literal 590 zcmaKpOHRWu5QhJ8oF=i6fGr5b36NTKDgZCRL+`F!QRXJ&}s&v|5Rqf2|m9cn0361g&_WroO z@Na8OlZmZ3Ga8GS+$_=%-8Fynb(Mz`| d)UyFWyF6c$q>|)+?zxtXC3}*IWGb0S{sE#ClNe}U~K9@5u8Ql z!Iw8L6Ns_|0RVf}%qj{jW~h%wj{)k^K$m^BSYM0-H5tX}jtf6jQ=S0~N;O)jT@=TB z6NzmuvIQ4Y<%gQ%UoOEKn3+HRYJTG{m(X-6Kap#`PIHQGc{Q{W%T>iBv|WlNTbU@q z_G-x3s@HT}5l&R_*_LddFVTxSUJV&r$s}~UbLPMhhWNT7Yisq|n|3hmXxiy}>x^G@ OA$2EpBlRHlMD_-n+<`*? literal 0 HcmV?d00001 diff --git a/wasm-corpus/src/main/resources/wat/simd-memory-bounds.wat b/wasm-corpus/src/main/resources/wat/simd-memory-bounds.wat new file mode 100644 index 000000000..d023d5e55 --- /dev/null +++ b/wasm-corpus/src/main/resources/wat/simd-memory-bounds.wat @@ -0,0 +1,26 @@ +(module + (memory 1) + + ;; an access that ends past the last byte of memory must trap, without partial writes + (func (export "v128.load") (param $addr i32) (result i64) + (i64x2.extract_lane 0 (v128.load (local.get $addr)))) + (func (export "v128.load8_lane") (param $addr i32) (result i64) + (i64x2.extract_lane 0 (v128.load8_lane 0 (local.get $addr) (v128.const i64x2 0 0)))) + (func (export "v128.load16_lane") (param $addr i32) (result i64) + (i64x2.extract_lane 0 (v128.load16_lane 0 (local.get $addr) (v128.const i64x2 0 0)))) + (func (export "v128.load32_lane") (param $addr i32) (result i64) + (i64x2.extract_lane 0 (v128.load32_lane 0 (local.get $addr) (v128.const i64x2 0 0)))) + (func (export "v128.load64_lane") (param $addr i32) (result i64) + (i64x2.extract_lane 0 (v128.load64_lane 0 (local.get $addr) (v128.const i64x2 0 0)))) + + (func (export "v128.store") (param $addr i32) + (v128.store (local.get $addr) (v128.const i64x2 -1 -1))) + (func (export "v128.store8_lane") (param $addr i32) + (v128.store8_lane 0 (local.get $addr) (v128.const i64x2 -1 -1))) + (func (export "v128.store16_lane") (param $addr i32) + (v128.store16_lane 0 (local.get $addr) (v128.const i64x2 -1 -1))) + (func (export "v128.store32_lane") (param $addr i32) + (v128.store32_lane 0 (local.get $addr) (v128.const i64x2 -1 -1))) + (func (export "v128.store64_lane") (param $addr i32) + (v128.store64_lane 0 (local.get $addr) (v128.const i64x2 -1 -1))) +) diff --git a/wasm-corpus/src/main/resources/wat/simd-mixed-lanes.wat b/wasm-corpus/src/main/resources/wat/simd-mixed-lanes.wat new file mode 100644 index 000000000..3d3fda5f9 --- /dev/null +++ b/wasm-corpus/src/main/resources/wat/simd-mixed-lanes.wat @@ -0,0 +1,27 @@ +(module + ;; ops that combine or select lanes; the spec tests feed them uniform lanes only + (func (export "i32x4.dot_i16x8_s") (param v128 v128) (result v128) + (i32x4.dot_i16x8_s (local.get 0) (local.get 1))) + + (func (export "i16x8.extadd_pairwise_i8x16_s") (param v128) (result v128) + (i16x8.extadd_pairwise_i8x16_s (local.get 0))) + (func (export "i16x8.extadd_pairwise_i8x16_u") (param v128) (result v128) + (i16x8.extadd_pairwise_i8x16_u (local.get 0))) + (func (export "i32x4.extadd_pairwise_i16x8_s") (param v128) (result v128) + (i32x4.extadd_pairwise_i16x8_s (local.get 0))) + (func (export "i32x4.extadd_pairwise_i16x8_u") (param v128) (result v128) + (i32x4.extadd_pairwise_i16x8_u (local.get 0))) + + (func (export "i16x8.extmul_low_i8x16_s") (param v128 v128) (result v128) + (i16x8.extmul_low_i8x16_s (local.get 0) (local.get 1))) + (func (export "i16x8.extmul_high_i8x16_u") (param v128 v128) (result v128) + (i16x8.extmul_high_i8x16_u (local.get 0) (local.get 1))) + (func (export "i32x4.extmul_low_i16x8_u") (param v128 v128) (result v128) + (i32x4.extmul_low_i16x8_u (local.get 0) (local.get 1))) + (func (export "i32x4.extmul_high_i16x8_s") (param v128 v128) (result v128) + (i32x4.extmul_high_i16x8_s (local.get 0) (local.get 1))) + (func (export "i64x2.extmul_low_i32x4_s") (param v128 v128) (result v128) + (i64x2.extmul_low_i32x4_s (local.get 0) (local.get 1))) + (func (export "i64x2.extmul_high_i32x4_u") (param v128 v128) (result v128) + (i64x2.extmul_high_i32x4_u (local.get 0) (local.get 1))) +)