Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
268 changes: 257 additions & 11 deletions Cargo.lock

Large diffs are not rendered by default.

7 changes: 7 additions & 0 deletions cranelift/codegen/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,9 @@ wast = { version = "35.0.0", optional = true }
# machine code. Integration tests that need external dependencies can be
# accomodated in `tests`.

[dev-dependencies]
criterion = "0.3"

[build-dependencies]
cranelift-codegen-meta = { path = "meta", version = "0.73.0" }

Expand Down Expand Up @@ -103,3 +106,7 @@ souper-harvest = ["souper-ir", "souper-ir/stringify"]

[badges]
maintenance = { status = "experimental" }

[[bench]]
name = "x64-evex-encoding"
harness = false
138 changes: 138 additions & 0 deletions cranelift/codegen/benches/x64-evex-encoding.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,138 @@
//! Measure instruction encoding latency using various approaches; the
//! benchmarking is feature-gated on `x86` since it only measures the encoding
//! mechanism of that backend.

#[cfg(feature = "x86")]
mod x86 {
use cranelift_codegen::isa::x64::encoding::{
evex::{EvexContext, EvexInstruction, EvexMasking, EvexVectorLength, Register},
rex::OpcodeMap,
rex::{encode_modrm, LegacyPrefixes},
ByteSink,
};
use cranelift_codegen_shared::isa::x86::EncodingBits;
use criterion::{criterion_group, Criterion};

// Define the benchmarks.
fn x64_evex_encoding_benchmarks(c: &mut Criterion) {
let mut group = c.benchmark_group("x64 EVEX encoding");
let rax = Register::from(0);
let rdx = Register::from(2);

group.bench_function("EvexInstruction (builder pattern)", |b| {
let mut sink = vec![];
b.iter(|| {
sink.clear();
EvexInstruction::new()
.prefix(LegacyPrefixes::_66)
.map(OpcodeMap::_0F38)
.w(true)
.opcode(0x1F)
.reg(rax)
.rm(rdx)
.length(EvexVectorLength::V128)
.encode(&mut sink);
});
});

group.bench_function("encode_evex (function pattern)", |b| {
let mut sink = vec![];
let bits = EncodingBits::new(&[0x66, 0x0f, 0x38, 0x1f], 0, 1);
let vvvvv = Register::from(0);
b.iter(|| {
sink.clear();
encode_evex(
bits,
rax,
vvvvv,
rdx,
EvexContext::Other {
length: EvexVectorLength::V128,
},
EvexMasking::default(),
&mut sink,
);
})
});
}
criterion_group!(benches, x64_evex_encoding_benchmarks);

/// Using an inner module to feature-gate the benchmarks means that we must
/// manually specify how to run the benchmarks (see `criterion_main!`).
pub fn run_benchmarks() {
criterion::__warn_about_html_reports_feature();
criterion::__warn_about_cargo_bench_support_feature();
benches();
Criterion::default().configure_from_args().final_summary();
}

/// From the legacy x86 backend: a mechanism for encoding an EVEX
/// instruction, including the prefixes, the instruction opcode, and the
/// ModRM byte. This EVEX encoding function only encodes the `reg` (operand
/// 1), `vvvv` (operand 2), `rm` (operand 3) form; other forms are possible
/// (see section 2.6.2, Intel Software Development Manual, volume 2A),
/// requiring refactoring of this function or separate functions for each
/// form (e.g. as for the REX prefix).
#[inline(always)]
pub fn encode_evex<CS: ByteSink + ?Sized>(
enc: EncodingBits,
reg: Register,
vvvvv: Register,
rm: Register,
context: EvexContext,
masking: EvexMasking,
sink: &mut CS,
) {
let reg: u8 = reg.into();
let rm: u8 = rm.into();
let vvvvv: u8 = vvvvv.into();

// EVEX prefix.
sink.put1(0x62);

debug_assert!(enc.mm() < 0b100);
let mut p0 = enc.mm() & 0b11;
p0 |= evex2(rm, reg) << 4; // bits 3:2 are always unset
sink.put1(p0);

let mut p1 = enc.pp() | 0b100; // bit 2 is always set
p1 |= (!(vvvvv) & 0b1111) << 3;
p1 |= (enc.rex_w() & 0b1) << 7;
sink.put1(p1);

let mut p2 = masking.aaa_bits();
p2 |= (!(vvvvv >> 4) & 0b1) << 3;
p2 |= context.bits() << 4;
p2 |= masking.z_bit() << 7;
sink.put1(p2);

// Opcode.
sink.put1(enc.opcode_byte());

// ModR/M byte.
sink.put1(encode_modrm(3, reg & 7, rm & 7))
}

/// From the legacy x86 backend: encode the RXBR' bits of the EVEX P0 byte.
/// For an explanation of these bits, see section 2.6.1 in the Intel
/// Software Development Manual, volume 2A. These bits can be used by
/// different addressing modes (see section 2.6.2), requiring different
/// `vex*` functions than this one.
fn evex2(rm: u8, reg: u8) -> u8 {
let b = !(rm >> 3) & 1;
let x = !(rm >> 4) & 1;
let r = !(reg >> 3) & 1;
let r_ = !(reg >> 4) & 1;
0x00 | r_ | (b << 1) | (x << 2) | (r << 3)
}
}

fn main() {
#[cfg(feature = "x86")]
x86::run_benchmarks();

#[cfg(not(feature = "x86"))]
println!(
"Unable to run the x64-evex-encoding benchmark; the `x86` feature must be enabled in Cargo.",
);
}
4 changes: 3 additions & 1 deletion cranelift/codegen/src/isa/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -81,8 +81,10 @@ mod riscv;
#[cfg(feature = "x86")]
mod x86;

// This module is made public here for benchmarking purposes. No guarantees are
// made regarding API stability.
#[cfg(feature = "x86")]
mod x64;
pub mod x64;

#[cfg(feature = "arm32")]
mod arm32;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -208,6 +208,8 @@ impl EvexInstruction {
}
}

/// Describe the register index to use. This wrapper is a type-safe way to pass
/// around the registers defined in `inst/regs.rs`.
#[derive(Copy, Clone, Default)]
pub struct Register(u8);
impl From<u8> for Register {
Expand All @@ -216,13 +218,18 @@ impl From<u8> for Register {
Self(reg)
}
}
impl Into<u8> for Register {
fn into(self) -> u8 {
self.0
}
}

/// Defines the EVEX context for the `L'`, `L`, and `b` bits (bits 6:4 of EVEX P2 byte). Table 2-36 in
/// section 2.6.10 (Intel Software Development Manual, volume 2A) describes how these bits can be
/// used together for certain classes of instructions; i.e., special care should be taken to ensure
/// that instructions use an applicable correct `EvexContext`. Table 2-39 contains cases where
/// opcodes can result in an #UD.
#[allow(dead_code)] // Rounding and broadcast modes are not yet used.
#[allow(dead_code, missing_docs)] // Rounding and broadcast modes are not yet used.
pub enum EvexContext {
RoundingRegToRegFP {
rc: EvexRoundingControl,
Expand Down Expand Up @@ -250,7 +257,7 @@ impl Default for EvexContext {

impl EvexContext {
/// Encode the `L'`, `L`, and `b` bits (bits 6:4 of EVEX P2 byte) for merging with the P2 byte.
fn bits(&self) -> u8 {
pub fn bits(&self) -> u8 {
match self {
Self::RoundingRegToRegFP { rc } => 0b001 | rc.bits() << 1,
Self::NoRoundingFP { sae, length } => (*sae as u8) | length.bits() << 1,
Expand All @@ -261,7 +268,7 @@ impl EvexContext {
}

/// The EVEX format allows choosing a vector length in the `L'` and `L` bits; see `EvexContext`.
#[allow(dead_code)] // Wider-length vectors are not yet used.
#[allow(dead_code, missing_docs)] // Wider-length vectors are not yet used.
pub enum EvexVectorLength {
V128,
V256,
Expand All @@ -287,7 +294,7 @@ impl Default for EvexVectorLength {
}

/// The EVEX format allows defining rounding control in the `L'` and `L` bits; see `EvexContext`.
#[allow(dead_code)] // Rounding controls are not yet used.
#[allow(dead_code, missing_docs)] // Rounding controls are not yet used.
pub enum EvexRoundingControl {
RNE,
RD,
Expand All @@ -309,7 +316,7 @@ impl EvexRoundingControl {

/// Defines the EVEX masking behavior; masking support is described in section 2.6.4 of the Intel
/// Software Development Manual, volume 2A.
#[allow(dead_code)] // Masking is not yet used.
#[allow(dead_code, missing_docs)] // Masking is not yet used.
pub enum EvexMasking {
None,
Merging { k: u8 },
Expand All @@ -324,15 +331,15 @@ impl Default for EvexMasking {

impl EvexMasking {
/// Encode the `z` bit for merging with the P2 byte.
fn z_bit(&self) -> u8 {
pub fn z_bit(&self) -> u8 {
match self {
Self::None | Self::Merging { .. } => 0,
Self::Zeroing { .. } => 1,
}
}

/// Encode the `aaa` bits for merging with the P2 byte.
fn aaa_bits(&self) -> u8 {
pub fn aaa_bits(&self) -> u8 {
match self {
Self::None => 0b000,
Self::Merging { k } | Self::Zeroing { k } => {
Expand Down
Original file line number Diff line number Diff line change
@@ -1,10 +1,13 @@
//! Contains the encoding machinery for the various x64 instruction formats.
use crate::{isa::x64, machinst::MachBuffer};
use std::vec::Vec;

pub mod evex;
pub mod rex;
pub mod vex;

/// The encoding formats in this module all require a way of placing bytes into
/// a buffer.
pub trait ByteSink {
/// Add 1 byte to the code section.
fn put1(&mut self, _: u8);
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -28,8 +28,9 @@ pub(crate) fn low8_will_sign_extend_to_32(x: u32) -> bool {
xs == ((xs << 24) >> 24)
}

/// Encode the ModR/M byte.
#[inline(always)]
pub(crate) fn encode_modrm(m0d: u8, enc_reg_g: u8, rm_e: u8) -> u8 {
pub fn encode_modrm(m0d: u8, enc_reg_g: u8, rm_e: u8) -> u8 {
debug_assert!(m0d < 4);
debug_assert!(enc_reg_g < 8);
debug_assert!(rm_e < 8);
Expand Down Expand Up @@ -155,6 +156,7 @@ impl From<(OperandSize, Reg)> for RexFlags {

/// Allows using the same opcode byte in different "opcode maps" to allow for more instruction
/// encodings. See appendix A in the Intel Software Developer's Manual, volume 2A, for more details.
#[allow(missing_docs)]
pub enum OpcodeMap {
None,
_0F,
Expand Down
12 changes: 6 additions & 6 deletions cranelift/codegen/src/isa/x64/inst/emit.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2,16 +2,16 @@ use crate::binemit::{Addend, Reloc};
use crate::ir::immediates::{Ieee32, Ieee64};
use crate::ir::LibCall;
use crate::ir::TrapCode;
use crate::isa::x64::inst::args::*;
use crate::isa::x64::inst::*;
use crate::machinst::{inst_common, MachBuffer, MachInstEmit, MachLabel};
use core::convert::TryInto;
use encoding::evex::{EvexInstruction, EvexVectorLength};
use encoding::rex::{
use crate::isa::x64::encoding::evex::{EvexInstruction, EvexVectorLength};
use crate::isa::x64::encoding::rex::{
emit_simm, emit_std_enc_enc, emit_std_enc_mem, emit_std_reg_mem, emit_std_reg_reg, int_reg_enc,
low8_will_sign_extend_to_32, low8_will_sign_extend_to_64, reg_enc, LegacyPrefixes, OpcodeMap,
RexFlags,
};
use crate::isa::x64::inst::args::*;
use crate::isa::x64::inst::*;
use crate::machinst::{inst_common, MachBuffer, MachInstEmit, MachLabel};
use core::convert::TryInto;
use log::debug;
use regalloc::{Reg, Writable};

Expand Down
3 changes: 1 addition & 2 deletions cranelift/codegen/src/isa/x64/inst/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,6 @@ pub mod args;
mod emit;
#[cfg(test)]
mod emit_tests;
pub(crate) mod encoding;
pub mod regs;
pub mod unwind;

Expand Down Expand Up @@ -2856,7 +2855,7 @@ impl EmitState {
self.stack_map = None;
}

fn cur_srcloc(&self) -> SourceLoc {
pub(crate) fn cur_srcloc(&self) -> SourceLoc {
self.cur_srcloc
}
}
Expand Down
1 change: 1 addition & 0 deletions cranelift/codegen/src/isa/x64/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@ use target_lexicon::Triple;
use crate::isa::unwind::systemv;

mod abi;
pub mod encoding;
mod inst;
mod lower;
mod settings;
Expand Down
3 changes: 3 additions & 0 deletions deny.toml
Original file line number Diff line number Diff line change
Expand Up @@ -45,4 +45,7 @@ skip = [
{ name = "wast" }, # old one pulled in by witx
{ name = "itertools" }, # 0.9 pulled in by zstd-sys
{ name = "quick-error" }, # transitive dependencies
{ name = "rustc_version" }, # transitive dependencies of criterion's build script (see https://github.com/japaric/cast.rs/pull/26)
{ name = "semver" }, # transitive dependencies of criterion's build script (see https://github.com/japaric/cast.rs/pull/26)
{ name = "semver-parser" }, # transitive dependencies of criterion's build script (see https://github.com/japaric/cast.rs/pull/26)
]