Compare commits
4 Commits
1b7d22b87e
...
45072ca743
| Author | SHA1 | Date | |
|---|---|---|---|
| 45072ca743 | |||
| c55d9f3dad | |||
| 910701dfaf | |||
| c69bec6636 |
@@ -4,7 +4,7 @@ use mimalloc::MiMalloc;
|
||||
static GLOBAL: MiMalloc = MiMalloc;
|
||||
|
||||
use anyhow::{Context, Result, bail};
|
||||
use clap::Parser;
|
||||
use clap::{Parser, ValueEnum};
|
||||
use glob::glob;
|
||||
use pimcore::binary_to_instruction::binary_to_executor;
|
||||
use pimcore::cpu::crossbar::Crossbar;
|
||||
@@ -14,7 +14,7 @@ use pimcore::tracing::TRACER;
|
||||
use serde_json::Value;
|
||||
use std::collections::HashMap;
|
||||
use std::fs::{self, File};
|
||||
use std::io::{BufReader, Write};
|
||||
use std::io::BufReader;
|
||||
use std::path::PathBuf;
|
||||
|
||||
/// Program to simulate core execution configuration
|
||||
@@ -44,14 +44,40 @@ struct Args {
|
||||
/// Comma separated list of (address,size) for memory output dump
|
||||
#[arg(short, long, value_delimiter = ',', num_args = 1.., value_name = "ADDR,SIZE")]
|
||||
dump: Vec<usize>,
|
||||
|
||||
/// Simulator execution mode
|
||||
#[arg(long, value_enum, default_value_t = ExecutionMode::Latency)]
|
||||
mode: ExecutionMode,
|
||||
|
||||
/// Number of inputs to execute (required in throughput mode)
|
||||
#[arg(long)]
|
||||
batch_size: Option<u32>,
|
||||
|
||||
/// Input binary for one iteration; repeat once per batch entry
|
||||
#[arg(long = "input")]
|
||||
inputs: Vec<PathBuf>,
|
||||
|
||||
/// Optional directory for per-iteration output dumps
|
||||
#[arg(long)]
|
||||
batch_output_dir: Option<PathBuf>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, ValueEnum)]
|
||||
enum ExecutionMode {
|
||||
Latency,
|
||||
Throughput,
|
||||
}
|
||||
|
||||
fn main() -> Result<()> {
|
||||
let args = Args::parse();
|
||||
|
||||
let config_json = retrive_config(&args)?;
|
||||
let mut core_inputs = retrive_cores(&args)?;
|
||||
let memory = retrive_memory(&args)?;
|
||||
let config_json = retrieve_config(&args)?;
|
||||
let batch_size = batch_size(&args)?;
|
||||
let input_regions = input_regions(&config_json)?;
|
||||
let input_data = retrieve_inputs(&args, batch_size)?;
|
||||
let inputs: Vec<&[u8]> = input_data.iter().map(Vec::as_slice).collect();
|
||||
let mut core_inputs = retrieve_cores(&args)?;
|
||||
let memory = retrieve_memory(&args)?;
|
||||
let global_crossbars = get_crossbars(&config_json, &args).unwrap();
|
||||
let crossbars = map_crossbars_to_cores(&config_json, &args, &global_crossbars);
|
||||
let mut executor = match &mut core_inputs {
|
||||
@@ -67,11 +93,68 @@ fn main() -> Result<()> {
|
||||
.lock()
|
||||
.unwrap()
|
||||
.init(executor.cpu().num_core(), args.output.clone());
|
||||
executor.execute()?;
|
||||
dump_memory(executor, &args)?;
|
||||
let dumps = dump_ranges(&args.dump)?;
|
||||
let batch_outputs = executor.execute_batch(&inputs, &input_regions, &dumps)?;
|
||||
fs::write(
|
||||
&args.output,
|
||||
batch_outputs
|
||||
.last()
|
||||
.context("simulation produced no output")?,
|
||||
)?;
|
||||
if let Some(batch_output_dir) = args.batch_output_dir {
|
||||
write_batch_outputs(batch_output_dir, batch_outputs)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn batch_size(args: &Args) -> Result<u32> {
|
||||
match (&args.mode, args.batch_size) {
|
||||
(ExecutionMode::Latency, None | Some(1)) => Ok(1),
|
||||
(ExecutionMode::Latency, Some(_)) => bail!("latency mode requires batch size 1"),
|
||||
(ExecutionMode::Throughput, Some(0)) => bail!("batch size must be positive"),
|
||||
(ExecutionMode::Throughput, Some(batch_size)) => Ok(batch_size),
|
||||
(ExecutionMode::Throughput, None) => bail!("throughput mode requires --batch-size"),
|
||||
}
|
||||
}
|
||||
|
||||
fn input_regions(config: &Value) -> Result<Vec<(usize, usize)>> {
|
||||
let addresses = config
|
||||
.get("inputs_addresses")
|
||||
.and_then(Value::as_array)
|
||||
.context("config.json has no inputs_addresses array")?;
|
||||
let sizes = config
|
||||
.get("inputs_sizes")
|
||||
.and_then(Value::as_array)
|
||||
.context("config.json has no inputs_sizes array")?;
|
||||
if addresses.len() != sizes.len() {
|
||||
bail!("config.json input address/size count mismatch");
|
||||
}
|
||||
addresses
|
||||
.iter()
|
||||
.zip(sizes)
|
||||
.map(|(address, size)| {
|
||||
Ok((
|
||||
usize::try_from(address.as_u64().context("invalid input address")?)?,
|
||||
usize::try_from(size.as_u64().context("invalid input size")?)?,
|
||||
))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn retrieve_inputs(args: &Args, batch_size: u32) -> Result<Vec<Vec<u8>>> {
|
||||
if args.inputs.len() != batch_size as usize {
|
||||
bail!(
|
||||
"batch size {batch_size} requires {} inputs, got {}",
|
||||
batch_size,
|
||||
args.inputs.len()
|
||||
);
|
||||
}
|
||||
args.inputs
|
||||
.iter()
|
||||
.map(|path| fs::read(path).with_context(|| format!("Failed to read input file: {path:?}")))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn map_crossbars_to_cores<'c>(
|
||||
config: &Value,
|
||||
args: &Args,
|
||||
@@ -114,7 +197,7 @@ fn map_crossbars_to_cores<'c>(
|
||||
let path_as_str = real_path.to_str().unwrap();
|
||||
assert!(
|
||||
global_crossbars.contains_key(path_as_str),
|
||||
"symlink point to {:?}\n a not stored crossbar",
|
||||
"symlink points to {:?}\n a crossbar that was not stored",
|
||||
real_path
|
||||
);
|
||||
|
||||
@@ -131,7 +214,7 @@ fn map_crossbars_to_cores<'c>(
|
||||
fn get_crossbars(config: &Value, args: &Args) -> anyhow::Result<HashMap<String, Crossbar>> {
|
||||
let xbar_size = config.get("xbar_size").unwrap().as_array().unwrap();
|
||||
let rows_crossbar = xbar_size[0].as_i64().unwrap() as usize;
|
||||
let column_corssbar = xbar_size[1].as_i64().unwrap() as usize;
|
||||
let column_crossbar = xbar_size[1].as_i64().unwrap() as usize;
|
||||
let mut res = HashMap::new();
|
||||
|
||||
if let Some(folder) = args.folder.as_ref() {
|
||||
@@ -154,7 +237,7 @@ fn get_crossbars(config: &Value, args: &Args) -> anyhow::Result<HashMap<String,
|
||||
let bytes = std::fs::read(weight_file.path()).expect("Failed to read binary file");
|
||||
let stored_row_bytes = bytes.len() / rows_crossbar;
|
||||
let mut crossbar = Crossbar::new(
|
||||
std::cmp::max(column_corssbar * 4, stored_row_bytes),
|
||||
std::cmp::max(column_crossbar * 4, stored_row_bytes),
|
||||
rows_crossbar,
|
||||
CoreMemory::new(),
|
||||
);
|
||||
@@ -174,21 +257,22 @@ fn get_crossbars(config: &Value, args: &Args) -> anyhow::Result<HashMap<String,
|
||||
Ok(res)
|
||||
}
|
||||
|
||||
fn dump_memory(mut executor: pimcore::Executable, args: &Args) -> Result<()> {
|
||||
let dumps: Vec<(usize, usize)> = args
|
||||
.dump
|
||||
fn dump_ranges(values: &[usize]) -> Result<Vec<(usize, usize)>> {
|
||||
if !values.len().is_multiple_of(2) {
|
||||
bail!("memory dump requires address,size pairs");
|
||||
}
|
||||
Ok(values
|
||||
.chunks_exact(2)
|
||||
.map(|chunk| (chunk[0], chunk[1]))
|
||||
.collect();
|
||||
let mut out_file = fs::OpenOptions::new()
|
||||
.create(true)
|
||||
.write(true)
|
||||
.truncate(true)
|
||||
.open(&args.output)
|
||||
.with_context(|| format!("cannot open file {:?} for writing", args.output))?;
|
||||
.collect())
|
||||
}
|
||||
|
||||
for (address, size) in dumps {
|
||||
out_file.write_all(executor.cpu_mut().host().load::<u8>(address, size).unwrap()[0])?;
|
||||
fn write_batch_outputs(output_dir: PathBuf, outputs: Vec<Vec<u8>>) -> Result<()> {
|
||||
fs::create_dir_all(&output_dir)
|
||||
.with_context(|| format!("cannot create batch output directory {output_dir:?}"))?;
|
||||
for (iteration, output) in outputs.into_iter().enumerate() {
|
||||
let path = output_dir.join(format!("output_{iteration:06}.bin"));
|
||||
fs::write(&path, output).with_context(|| format!("cannot write batch output {path:?}"))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -197,7 +281,7 @@ fn set_memory(executor: &mut pimcore::Executable, memory: Vec<u8>) {
|
||||
executor.cpu_mut().host().execute_store(0, &memory).unwrap();
|
||||
}
|
||||
|
||||
fn retrive_memory(args: &Args) -> Result<Vec<u8>> {
|
||||
fn retrieve_memory(args: &Args) -> Result<Vec<u8>> {
|
||||
let memory_path = if let Some(mem_override) = &args.memory {
|
||||
mem_override.clone()
|
||||
} else if let Some(folder) = &args.folder.as_ref() {
|
||||
@@ -237,7 +321,7 @@ enum CoreInputs {
|
||||
Binary(Vec<Vec<u8>>),
|
||||
}
|
||||
|
||||
fn retrive_cores(args: &Args) -> Result<CoreInputs, anyhow::Error> {
|
||||
fn retrieve_cores(args: &Args) -> Result<CoreInputs, anyhow::Error> {
|
||||
if let Some(cores_override) = &args.cores {
|
||||
let first_extension = cores_override
|
||||
.first()
|
||||
@@ -310,7 +394,7 @@ fn core_sort_key(path: &PathBuf) -> i32 {
|
||||
stem.parse::<i32>().unwrap()
|
||||
}
|
||||
|
||||
fn retrive_config(args: &Args) -> Result<Value, anyhow::Error> {
|
||||
fn retrieve_config(args: &Args) -> Result<Value, anyhow::Error> {
|
||||
let config_path: PathBuf = {
|
||||
let override_path = args.config.as_ref();
|
||||
let folder = args.folder.as_ref();
|
||||
|
||||
@@ -13,6 +13,33 @@ pub mod crossbar;
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct CPU<'a> {
|
||||
cores: Box<[Core<'a>]>,
|
||||
batch_outputs: Option<BatchOutputs>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
struct BatchOutputs {
|
||||
iteration: usize,
|
||||
ranges: Vec<(usize, usize)>,
|
||||
outputs: Vec<Vec<u8>>,
|
||||
}
|
||||
|
||||
impl BatchOutputs {
|
||||
fn record(&mut self, address: usize, bytes: &[u8]) {
|
||||
let output = &mut self.outputs[self.iteration];
|
||||
let store_end = address + bytes.len();
|
||||
let mut output_offset = 0;
|
||||
for &(range_address, range_size) in &self.ranges {
|
||||
let start = address.max(range_address);
|
||||
let end = store_end.min(range_address + range_size);
|
||||
if start < end {
|
||||
let size = end - start;
|
||||
output[output_offset + start - range_address
|
||||
..output_offset + start - range_address + size]
|
||||
.copy_from_slice(&bytes[start - address..start - address + size]);
|
||||
}
|
||||
output_offset += range_size;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<'a> CPU<'a> {
|
||||
@@ -25,9 +52,63 @@ impl<'a> CPU<'a> {
|
||||
}
|
||||
Self {
|
||||
cores: cores.into(),
|
||||
batch_outputs: None,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn set_current_iteration(&mut self, iteration: u32) {
|
||||
if let Some(batch_outputs) = &mut self.batch_outputs {
|
||||
batch_outputs.iteration = iteration as usize;
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn begin_host_store_recording(
|
||||
&mut self,
|
||||
batch_size: usize,
|
||||
dump_ranges: &[(usize, usize)],
|
||||
) -> Result<()> {
|
||||
let mut initial = Vec::new();
|
||||
for &(address, size) in dump_ranges {
|
||||
initial.extend_from_slice(self.host().load::<u8>(address, size)?[0]);
|
||||
}
|
||||
self.batch_outputs = Some(BatchOutputs {
|
||||
iteration: 0,
|
||||
ranges: dump_ranges.to_vec(),
|
||||
outputs: vec![initial; batch_size],
|
||||
});
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub(crate) fn store_to_host(
|
||||
&mut self,
|
||||
core: impl TryToUsize,
|
||||
host_address: impl AddressArg,
|
||||
core_address: impl AddressArg,
|
||||
size: impl TryToUsize,
|
||||
) -> Result<()> {
|
||||
let core = core.try_into().expect("core can not be negative");
|
||||
let host_address = host_address.to_address_usize()?;
|
||||
let core_address = core_address.to_address_usize()?;
|
||||
let size = size.try_into().context("size can not be negative")?;
|
||||
let Self {
|
||||
cores,
|
||||
batch_outputs,
|
||||
} = self;
|
||||
let (host, cores) = cores.split_at_mut(1);
|
||||
let bytes = cores[core - 1].load::<u8>(core_address, size)?[0];
|
||||
host[0].execute_store(host_address, bytes)?;
|
||||
if let Some(batch_outputs) = batch_outputs {
|
||||
batch_outputs.record(host_address, bytes);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub(crate) fn finish_host_store_recording(&mut self) -> Vec<Vec<u8>> {
|
||||
self.batch_outputs
|
||||
.take()
|
||||
.map_or_else(Vec::new, |batch_outputs| batch_outputs.outputs)
|
||||
}
|
||||
|
||||
pub fn host<'b>(&'b mut self) -> &'b mut Core<'a>
|
||||
where
|
||||
'a: 'b,
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use crate::{
|
||||
cpu::{CPU, crossbar},
|
||||
instruction_set::{
|
||||
Instruction, InstructionData, InstructionStatus, InstructionType, VectorBitWith,
|
||||
Instruction, InstructionData, InstructionStatus, InstructionType, VectorBitWidth,
|
||||
helper::add_all,
|
||||
},
|
||||
memory_manager::{
|
||||
@@ -200,20 +200,20 @@ pub fn isa_simd(functor: InstructionType) -> bool {
|
||||
|
||||
pub fn dispatch_simd(
|
||||
functor: InstructionType,
|
||||
vector_bit_with: VectorBitWith,
|
||||
vector_bit_width: VectorBitWidth,
|
||||
) -> Result<InstructionType> {
|
||||
let VectorBitWith {
|
||||
vector_input_bitwith,
|
||||
vector_output_bitwith,
|
||||
} = vector_bit_with;
|
||||
let VectorBitWidth {
|
||||
vector_input_bitwidth,
|
||||
vector_output_bitwidth,
|
||||
} = vector_bit_width;
|
||||
let res = SIMD
|
||||
.get(&(functor as usize))
|
||||
.context("Request a non present simd")?
|
||||
.get(&(vector_input_bitwith, vector_output_bitwith))
|
||||
.get(&(vector_input_bitwidth, vector_output_bitwidth))
|
||||
.with_context(|| {
|
||||
format!(
|
||||
"Function not found for the requested size input:{} output:{}",
|
||||
vector_input_bitwith, vector_output_bitwith
|
||||
vector_input_bitwidth, vector_output_bitwidth
|
||||
)
|
||||
})?;
|
||||
Ok(*res)
|
||||
@@ -819,13 +819,15 @@ pub fn st(cores: &mut CPU, data: InstructionData) -> Result<InstructionStatus> {
|
||||
let (core, rd, r1, _, imm_len, offset_select, offset_value) =
|
||||
data.get_core_rd_r1_r2_immlen_offset();
|
||||
ensure!(core != 0, "ST cannot be used to move from host to host");
|
||||
let (host, core) = cores.host_and_cores(core);
|
||||
let r1_val = core.register(r1);
|
||||
let rd_val = core.register(rd);
|
||||
let r1_val = add_offset_r1(r1_val, offset_select, offset_value);
|
||||
let rd_val = add_offset_rd(rd_val, offset_select, offset_value);
|
||||
let local_memory = core.load::<u8>(r1_val, imm_len)?;
|
||||
host.execute_store(rd_val, local_memory[0]);
|
||||
let (rd_val, r1_val) = {
|
||||
let core = cores.core(core);
|
||||
let r1_val = core.register(r1);
|
||||
let rd_val = core.register(rd);
|
||||
let r1_val = add_offset_r1(r1_val, offset_select, offset_value);
|
||||
let rd_val = add_offset_rd(rd_val, offset_select, offset_value);
|
||||
(rd_val, r1_val)
|
||||
};
|
||||
cores.store_to_host(core, rd_val, r1_val, imm_len)?;
|
||||
TRACER.lock().unwrap().post_st(cores, data);
|
||||
Ok(InstructionStatus::Completed)
|
||||
}
|
||||
@@ -881,7 +883,7 @@ pub fn isa_recv(functor: usize) -> bool {
|
||||
|
||||
#[inline(never)]
|
||||
pub fn recv(cores: &mut CPU, data: InstructionData) -> Result<InstructionStatus> {
|
||||
Ok(InstructionStatus::Reciving(data))
|
||||
Ok(InstructionStatus::Receiving(data))
|
||||
}
|
||||
|
||||
#[inline(never)]
|
||||
|
||||
@@ -22,7 +22,7 @@ pub enum InstructionStatus {
|
||||
Completed,
|
||||
Waiting(InstructionData),
|
||||
Sending(InstructionData),
|
||||
Reciving(InstructionData),
|
||||
Receiving(InstructionData),
|
||||
Sync(InstructionData),
|
||||
#[default]
|
||||
NotExecuted,
|
||||
@@ -59,21 +59,21 @@ pub type Instructions = Vec<Instruction>;
|
||||
pub type InstructionType = fn(&mut CPU, InstructionData) -> Result<InstructionStatus>;
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default)]
|
||||
pub struct VectorBitWith {
|
||||
pub vector_input_bitwith: usize,
|
||||
pub vector_output_bitwith: usize,
|
||||
pub struct VectorBitWidth {
|
||||
pub vector_input_bitwidth: usize,
|
||||
pub vector_output_bitwidth: usize,
|
||||
}
|
||||
|
||||
/// Support for the
|
||||
/// setbw ibiw, obiw
|
||||
/// Set the bit-widths of each element for input vectors and output vectors. Related vector instructions
|
||||
/// use the configured bit-widths. Once setbw is caled, all subsequent related vector instructions will
|
||||
/// use the configured bit-widths. Once setbw is called, all subsequent related vector instructions will
|
||||
/// use the configured bit-widths, until a new setbw is called. Once ibiw and obiw are set, ibyw and
|
||||
/// obyw are also set accordingly by the hardware.
|
||||
/// If the hardware does not support variable bit-width, this instruction is invalid and the matrix/vector
|
||||
/// instructions use the fixed bit-width of the hardware.
|
||||
pub struct InstructionsBuilder {
|
||||
vector_bit_with: VectorBitWith,
|
||||
vector_bit_width: VectorBitWidth,
|
||||
instructions: Instructions,
|
||||
}
|
||||
|
||||
@@ -86,9 +86,9 @@ impl Default for InstructionsBuilder {
|
||||
impl InstructionsBuilder {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
vector_bit_with: VectorBitWith {
|
||||
vector_input_bitwith: 32,
|
||||
vector_output_bitwith: 32,
|
||||
vector_bit_width: VectorBitWidth {
|
||||
vector_input_bitwidth: 32,
|
||||
vector_output_bitwidth: 32,
|
||||
},
|
||||
instructions: Instructions::new(),
|
||||
}
|
||||
@@ -97,9 +97,9 @@ impl InstructionsBuilder {
|
||||
pub fn make_inst(&mut self, functor: InstructionType, data: InstructionData) {
|
||||
if is_setbw(functor) {
|
||||
let (ibiw, obiw) = data.get_ibiw_obiw();
|
||||
self.vector_bit_with.vector_input_bitwith =
|
||||
self.vector_bit_width.vector_input_bitwidth =
|
||||
ibiw.try_into().expect("ibiw can not be negative");
|
||||
self.vector_bit_with.vector_output_bitwith =
|
||||
self.vector_bit_width.vector_output_bitwidth =
|
||||
obiw.try_into().expect("obiw can not be negative");
|
||||
return;
|
||||
}
|
||||
@@ -107,7 +107,7 @@ impl InstructionsBuilder {
|
||||
if (isa_simd(functor)) {
|
||||
self.instructions.push(Instruction::new(
|
||||
data,
|
||||
dispatch_simd(functor, self.vector_bit_with).unwrap(),
|
||||
dispatch_simd(functor, self.vector_bit_width).unwrap(),
|
||||
))
|
||||
} else {
|
||||
self.instructions.push(Instruction::new(data, functor))
|
||||
|
||||
@@ -1,8 +1,12 @@
|
||||
#![allow(unused)]
|
||||
|
||||
use anyhow::{Result, bail};
|
||||
use anyhow::{Context, Result, bail};
|
||||
use std::{
|
||||
collections::{HashMap, HashSet},
|
||||
sync::{
|
||||
Mutex,
|
||||
atomic::{AtomicU32, Ordering},
|
||||
},
|
||||
time::{Duration, SystemTime},
|
||||
};
|
||||
|
||||
@@ -25,6 +29,9 @@ pub mod send_recv;
|
||||
pub mod tracing;
|
||||
pub mod utility;
|
||||
|
||||
static GLOBAL_ITERATION: AtomicU32 = AtomicU32::new(0);
|
||||
static EXECUTION_LOCK: Mutex<()> = Mutex::new(());
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct CoreInstructionsBuilder {
|
||||
core_instructions: Vec<CoreInstructions>,
|
||||
@@ -54,6 +61,7 @@ impl CoreInstructionsBuilder {
|
||||
pub struct CoreInstructions {
|
||||
instructions: Instructions,
|
||||
program_counter: usize,
|
||||
current_iteration: u32,
|
||||
}
|
||||
|
||||
impl CoreInstructions {
|
||||
@@ -61,6 +69,7 @@ impl CoreInstructions {
|
||||
Self {
|
||||
instructions,
|
||||
program_counter,
|
||||
current_iteration: 0,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,6 +77,7 @@ impl CoreInstructions {
|
||||
Self {
|
||||
instructions: Vec::new(),
|
||||
program_counter: 0,
|
||||
current_iteration: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -77,6 +87,7 @@ impl From<Instructions> for CoreInstructions {
|
||||
CoreInstructions {
|
||||
instructions: value,
|
||||
program_counter: 0,
|
||||
current_iteration: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -130,6 +141,40 @@ impl<'a> Executable<'a> {
|
||||
where
|
||||
'a: 'b,
|
||||
{
|
||||
self.execute_batch(&[&[]], &[], &[]).map(|_| ())
|
||||
}
|
||||
|
||||
pub fn execute_batch<'b>(
|
||||
&'b mut self,
|
||||
inputs: &[&[u8]],
|
||||
input_regions: &[(usize, usize)],
|
||||
dump_ranges: &[(usize, usize)],
|
||||
) -> Result<Vec<Vec<u8>>>
|
||||
where
|
||||
'a: 'b,
|
||||
{
|
||||
validate_inputs(inputs, input_regions)?;
|
||||
self.execute_iterations(inputs, input_regions, dump_ranges)
|
||||
}
|
||||
|
||||
fn execute_iterations<'b>(
|
||||
&'b mut self,
|
||||
inputs: &[&[u8]],
|
||||
input_regions: &[(usize, usize)],
|
||||
dump_ranges: &[(usize, usize)],
|
||||
) -> Result<Vec<Vec<u8>>>
|
||||
where
|
||||
'a: 'b,
|
||||
{
|
||||
let _execution_lock = EXECUTION_LOCK.lock().unwrap();
|
||||
let batch_size = u32::try_from(inputs.len().max(1)).context("batch size exceeds u32")?;
|
||||
GLOBAL_ITERATION.store(0, Ordering::SeqCst);
|
||||
if let Some(input) = inputs.first() {
|
||||
store_input(&mut self.cpu, input, input_regions)?;
|
||||
}
|
||||
self.cpu
|
||||
.begin_host_store_recording(batch_size as usize, dump_ranges)?;
|
||||
|
||||
let Self {
|
||||
cpu,
|
||||
core_instructions: cores_instructions,
|
||||
@@ -147,9 +192,24 @@ impl<'a> Executable<'a> {
|
||||
&& let Some(core_instruction) = cores_instructions.get_mut(cpu_index)
|
||||
{
|
||||
core_result = InstructionStatus::NotExecuted;
|
||||
if core_instruction.program_counter == core_instruction.instructions.len() {
|
||||
if core_instruction.instructions.is_empty()
|
||||
|| core_instruction.current_iteration + 1 >= batch_size
|
||||
{
|
||||
break;
|
||||
}
|
||||
core_instruction.current_iteration += 1;
|
||||
core_instruction.program_counter = 0;
|
||||
let iteration = core_instruction.current_iteration;
|
||||
if iteration > GLOBAL_ITERATION.fetch_max(iteration, Ordering::SeqCst) {
|
||||
store_input(cpu, inputs[iteration as usize], input_regions)?;
|
||||
}
|
||||
}
|
||||
cpu.set_current_iteration(core_instruction.current_iteration);
|
||||
let CoreInstructions {
|
||||
instructions,
|
||||
program_counter,
|
||||
..
|
||||
} = core_instruction;
|
||||
core_result = instructions
|
||||
.get(*program_counter)
|
||||
@@ -211,7 +271,7 @@ impl<'a> Executable<'a> {
|
||||
|
||||
#[cfg(feature = "profile_time")]
|
||||
TRACER.lock().unwrap().report();
|
||||
Ok(())
|
||||
Ok(cpu.finish_host_store_recording())
|
||||
}
|
||||
|
||||
pub fn cpu(&self) -> &CPU<'a> {
|
||||
@@ -233,6 +293,29 @@ impl<'a> Executable<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_inputs(inputs: &[&[u8]], input_regions: &[(usize, usize)]) -> Result<()> {
|
||||
let input_size = input_regions.iter().try_fold(0usize, |total, (_, size)| {
|
||||
total.checked_add(*size).context("input size overflow")
|
||||
})?;
|
||||
if inputs.is_empty() {
|
||||
bail!("at least one input is required");
|
||||
}
|
||||
if inputs.iter().any(|input| input.len() != input_size) {
|
||||
bail!("each input must contain exactly {input_size} bytes");
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn store_input(cpu: &mut CPU, input: &[u8], input_regions: &[(usize, usize)]) -> Result<()> {
|
||||
let mut offset = 0;
|
||||
for &(address, size) in input_regions {
|
||||
cpu.host()
|
||||
.execute_store(address, &input[offset..offset + size])?;
|
||||
offset += size;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn detect_deadlock(cores_instructions: &[CoreInstructions]) -> Option<DeadlockInfo> {
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
enum CoreState {
|
||||
|
||||
@@ -33,10 +33,10 @@ pub struct SendRecv {
|
||||
impl SendRecv {
|
||||
pub fn new(num_core: usize) -> Self {
|
||||
let sending = [Option::None].repeat(num_core);
|
||||
let reciving = [Option::None].repeat(num_core);
|
||||
let receiving = [Option::None].repeat(num_core);
|
||||
Self {
|
||||
sending: sending.into(),
|
||||
receiving: reciving.into(),
|
||||
receiving: receiving.into(),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -73,18 +73,18 @@ where
|
||||
let data = inst.data;
|
||||
TRACER.lock().unwrap().pre_recv(cpu, data);
|
||||
}
|
||||
let [sender_core, reciver_core] =
|
||||
let [sender_core, receiver_core] =
|
||||
cpu.get_multiple_cores([sender.internal_core, receiver.internal_core]);
|
||||
let memory = sender_core
|
||||
.load::<u8>(sender.address, sender.size)
|
||||
.with_context(|| {
|
||||
format!(
|
||||
"Sender crash tranfering memroy from {} with size {}",
|
||||
"Sender crashed while transferring memory from {} with size {}",
|
||||
sender.address, sender.size
|
||||
)
|
||||
})
|
||||
.unwrap();
|
||||
reciver_core.execute_store(receiver.address, memory[0]);
|
||||
receiver_core.execute_store(receiver.address, memory[0]);
|
||||
{
|
||||
let sender = &mut core_instructions[sender.internal_core];
|
||||
let pc = sender.program_counter;
|
||||
@@ -124,19 +124,19 @@ where
|
||||
let receiver: usize = imm_core.try_into().expect("imm_core can not be negative");
|
||||
assert_ne!(receiver, 0, "Host can not use receive");
|
||||
send_recv.sending[sender] = Some(SendRecvInfo::new(sender, receiver, address, imm_len));
|
||||
let transfered = transfer_memory(
|
||||
let transferred = transfer_memory(
|
||||
cpu,
|
||||
core_instructions,
|
||||
send_recv.sending[sender],
|
||||
send_recv.receiving[receiver],
|
||||
);
|
||||
if transfered {
|
||||
if transferred {
|
||||
send_recv.sending[sender] = None;
|
||||
send_recv.receiving[receiver] = None;
|
||||
}
|
||||
(transfered, if transfered { receiver } else { 0 })
|
||||
(transferred, if transferred { receiver } else { 0 })
|
||||
}
|
||||
InstructionStatus::Reciving(instruction_data) => {
|
||||
InstructionStatus::Receiving(instruction_data) => {
|
||||
let (core_idx, imm_core) = instruction_data.get_core_immcore();
|
||||
let rd = instruction_data.rd();
|
||||
let imm_len = instruction_data
|
||||
@@ -153,17 +153,17 @@ where
|
||||
assert_ne!(sender, 0, "Host can not use send");
|
||||
send_recv.receiving[receiver] =
|
||||
Some(SendRecvInfo::new(receiver, sender, address, imm_len));
|
||||
let transfered = transfer_memory(
|
||||
let transferred = transfer_memory(
|
||||
cpu,
|
||||
core_instructions,
|
||||
send_recv.sending[sender],
|
||||
send_recv.receiving[receiver],
|
||||
);
|
||||
if transfered {
|
||||
if transferred {
|
||||
send_recv.sending[sender] = None;
|
||||
send_recv.receiving[receiver] = None;
|
||||
}
|
||||
(transfered, if transfered { sender } else { 0 })
|
||||
(transferred, if transferred { sender } else { 0 })
|
||||
}
|
||||
_ => (false, 0),
|
||||
}
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
mod common;
|
||||
|
||||
use pimcore::{
|
||||
CoreInstructionsBuilder, Executable,
|
||||
instruction_set::{InstructionsBuilder, instruction_data::InstructionDataBuilder, isa::*},
|
||||
};
|
||||
|
||||
#[test]
|
||||
fn restarts_cores_and_loads_each_input() {
|
||||
let cpu = common::empty_cpu(1);
|
||||
let mut cores = CoreInstructionsBuilder::new(1);
|
||||
let mut instructions = InstructionsBuilder::new();
|
||||
let mut data = InstructionDataBuilder::new();
|
||||
data.set_core_indx(1).fix_core_indx();
|
||||
instructions.make_inst(sldi, data.set_rdimm(1, 0).build());
|
||||
instructions.make_inst(sldi, data.set_rdimm(2, 0).build());
|
||||
instructions.make_inst(ld, data.set_rdr1(2, 1).set_imm_len(4).build());
|
||||
instructions.make_inst(sldi, data.set_rdimm(3, 4).build());
|
||||
instructions.make_inst(st, data.set_rdr1(3, 2).set_imm_len(4).build());
|
||||
cores.set_core(1, instructions.build());
|
||||
|
||||
let mut executable = Executable::new(cpu, cores.build());
|
||||
let first = 1.0f32.to_ne_bytes();
|
||||
let second = 2.0f32.to_ne_bytes();
|
||||
assert!(
|
||||
executable
|
||||
.execute_batch(&[&first[..3]], &[(0, 4)], &[])
|
||||
.is_err()
|
||||
);
|
||||
executable
|
||||
.execute_batch(&[&first, &second], &[(0, 4)], &[])
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
executable.cpu_mut().host().load::<f32>(4, 4).unwrap()[0],
|
||||
[2.0]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn records_each_iteration_output() {
|
||||
let cpu = common::empty_cpu(1);
|
||||
let mut cores = CoreInstructionsBuilder::new(1);
|
||||
let mut instructions = InstructionsBuilder::new();
|
||||
let mut data = InstructionDataBuilder::new();
|
||||
data.set_core_indx(1).fix_core_indx();
|
||||
instructions.make_inst(sldi, data.set_rdimm(1, 0).build());
|
||||
instructions.make_inst(sldi, data.set_rdimm(2, 0).build());
|
||||
instructions.make_inst(ld, data.set_rdr1(2, 1).set_imm_len(4).build());
|
||||
instructions.make_inst(sldi, data.set_rdimm(3, 4).build());
|
||||
instructions.make_inst(st, data.set_rdr1(3, 2).set_imm_len(4).build());
|
||||
cores.set_core(1, instructions.build());
|
||||
|
||||
let mut executable = Executable::new(cpu, cores.build());
|
||||
let first = 1.0f32.to_ne_bytes();
|
||||
let second = 2.0f32.to_ne_bytes();
|
||||
let outputs = executable
|
||||
.execute_batch(&[&first, &second], &[(0, 4)], &[(4, 2), (6, 2)])
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(outputs.len(), 2);
|
||||
assert_eq!(
|
||||
f32::from_ne_bytes(outputs[0].as_slice().try_into().unwrap()),
|
||||
1.0
|
||||
);
|
||||
assert_eq!(
|
||||
f32::from_ne_bytes(outputs[1].as_slice().try_into().unwrap()),
|
||||
2.0
|
||||
);
|
||||
}
|
||||
@@ -140,9 +140,13 @@ OnnxMlirCompilerErrorCodes writeConfigJson(func::FuncOp funcOp,
|
||||
configJson["array_group_map"] = std::move(xbarsPerArrayGroup);
|
||||
|
||||
json::Array inputsAddresses;
|
||||
for (BlockArgument input : funcOp.getArguments())
|
||||
json::Array inputsSizes;
|
||||
for (BlockArgument input : funcOp.getArguments()) {
|
||||
inputsAddresses.push_back(memory.getValueAddress(input));
|
||||
inputsSizes.push_back(memory.hostMem.getMemEntry({input, std::nullopt}).size);
|
||||
}
|
||||
configJson["inputs_addresses"] = std::move(inputsAddresses);
|
||||
configJson["inputs_sizes"] = std::move(inputsSizes);
|
||||
|
||||
json::Array outputsAddresses;
|
||||
for (func::ReturnOp returnOp : funcOp.getOps<func::ReturnOp>())
|
||||
|
||||
@@ -695,13 +695,16 @@ void PimCodeGen::codeGenSendOp(pim::PimSendOp sendOp, const StaticValueKnowledge
|
||||
void PimCodeGen::codeGenWaitOp(
|
||||
pim::PimWaitOp waitOp, const StaticValueKnowledge& knowledge) const {
|
||||
auto eventRegister = indexOf(waitOp.getEventRegister(), knowledge);
|
||||
assert(succeeded(eventRegister)
|
||||
&& "pim.wait event register must be statically resolvable during codegen");
|
||||
auto waitValue = indexOf(waitOp.getWaitValue(), knowledge);
|
||||
assert(succeeded(eventRegister) && succeeded(waitValue)
|
||||
&& "pim.wait operands must be statically resolvable during codegen");
|
||||
if (*waitValue == 0)
|
||||
return;
|
||||
pim_binary::InstructionRecord instruction;
|
||||
instruction.opcode = pim_binary::Opcode::wait;
|
||||
instruction.generic1 = pim::checkedI32OrCrash(
|
||||
*eventRegister, "wait event register");
|
||||
instruction.generic2 = waitOp.getWaitValue();
|
||||
instruction.generic2 = pim::checkedI32OrCrash(*waitValue, "wait value");
|
||||
emitInstruction(instruction);
|
||||
}
|
||||
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#include <limits>
|
||||
#include <tuple>
|
||||
|
||||
#include "src/Accelerators/PIM/Common/PimCommon.hpp"
|
||||
#include "src/Accelerators/PIM/Compiler/PimCompilerOptions.hpp"
|
||||
#include "src/Accelerators/PIM/Compiler/PimCompilerUtils.hpp"
|
||||
#include "src/Accelerators/PIM/Conversion/ONNXToSpatial/ONNXToSpatialOptions.hpp"
|
||||
@@ -78,6 +79,7 @@ spatial::SchedulingTarget getDefaultPimSchedulingTarget() {
|
||||
target.residentWeightCapacity = crossbarCountInCore.getValue();
|
||||
target.matrixRows = crossbarSize.getValue();
|
||||
target.matrixColumns = crossbarSize.getValue();
|
||||
target.synchronizationRegisterCount = kPimEventRegisterCount;
|
||||
|
||||
setDefaultPimInterProcessorLatencies(target);
|
||||
return target;
|
||||
|
||||
@@ -368,11 +368,14 @@ LogicalResult raptor::SpatialToPimPass::lowerComputeOp(spatial::SpatScheduledCom
|
||||
return failure();
|
||||
PimWaitOp::create(
|
||||
rewriter, receiveOp->getLoc(), hostWaitLoad.getEventRegister(),
|
||||
rewriter.getI32IntegerAttr(1));
|
||||
hostWaitLoad.getWaitValue());
|
||||
received = PimMemCopyHostToDevOp::create(
|
||||
rewriter, receiveOp->getLoc(), outputBuffer.getType(), zero,
|
||||
hostWaitLoad.getHostOffset(), outputBuffer, *hostBuffer, *sizeAttr)
|
||||
.getOutput();
|
||||
PimSyncOp::create(
|
||||
rewriter, receiveOp->getLoc(), hostWaitLoad.getSourceCoreId(),
|
||||
hostWaitLoad.getAcknowledgementEventRegister());
|
||||
} else {
|
||||
received = PimReceiveOp::create(
|
||||
rewriter, receiveOp->getLoc(), outputBuffer.getType(), outputBuffer,
|
||||
|
||||
@@ -147,15 +147,42 @@ struct HostWaitLoadLowering : OpRewritePattern<spatial::SpatHostWaitLoadOp> {
|
||||
return failure();
|
||||
auto wait = pim::PimWaitOp::create(
|
||||
rewriter, op.getLoc(), op.getEventRegister(),
|
||||
rewriter.getI32IntegerAttr(1));
|
||||
op.getWaitValue());
|
||||
copyRaptorDebugAttrs(op.getOperation(), wait.getOperation());
|
||||
return pim::PimMemCopyHostToDevOp::create(
|
||||
Value output = pim::PimMemCopyHostToDevOp::create(
|
||||
rewriter, op.getLoc(), outputBuffer.getType(), zero,
|
||||
op.getHostOffset(), outputBuffer, *hostBuffer, sizeAttr).getOutput();
|
||||
auto sync = pim::PimSyncOp::create(
|
||||
rewriter, op.getLoc(), op.getSourceCoreId(),
|
||||
op.getAcknowledgementEventRegister());
|
||||
copyRaptorDebugAttrs(op.getOperation(), sync.getOperation());
|
||||
return output;
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
struct SyncLowering : OpRewritePattern<spatial::SpatSyncOp> {
|
||||
using OpRewritePattern::OpRewritePattern;
|
||||
|
||||
LogicalResult matchAndRewrite(spatial::SpatSyncOp op,
|
||||
PatternRewriter& rewriter) const override {
|
||||
rewriter.replaceOpWithNewOp<pim::PimSyncOp>(
|
||||
op, op.getTargetCoreId(), op.getEventRegister());
|
||||
return success();
|
||||
}
|
||||
};
|
||||
|
||||
struct WaitLowering : OpRewritePattern<spatial::SpatWaitOp> {
|
||||
using OpRewritePattern::OpRewritePattern;
|
||||
|
||||
LogicalResult matchAndRewrite(spatial::SpatWaitOp op,
|
||||
PatternRewriter& rewriter) const override {
|
||||
rewriter.replaceOpWithNewOp<pim::PimWaitOp>(
|
||||
op, op.getEventRegister(), op.getWaitValue());
|
||||
return success();
|
||||
}
|
||||
};
|
||||
|
||||
struct ExtractRowsLowering : OpRewritePattern<spatial::SpatExtractRowsOp> {
|
||||
using OpRewritePattern::OpRewritePattern;
|
||||
|
||||
@@ -200,7 +227,8 @@ struct ConcatLowering : OpRewritePattern<spatial::SpatConcatOp> {
|
||||
void populateChannelLoweringPatterns(RewritePatternSet& patterns) {
|
||||
patterns.add<ChannelSendLowering, ChannelReceiveLowering,
|
||||
HostStoreSyncLowering, HostWaitLoadLowering,
|
||||
ExtractRowsLowering, ConcatLowering>(patterns.getContext());
|
||||
SyncLowering, WaitLowering, ExtractRowsLowering,
|
||||
ConcatLowering>(patterns.getContext());
|
||||
}
|
||||
|
||||
} // namespace onnx_mlir
|
||||
|
||||
@@ -128,6 +128,8 @@ void onnx_mlir::raptor::SpatialToPimPass::runOnOperation() {
|
||||
spatial::SpatChannelSendOp,
|
||||
spatial::SpatHostStoreSyncOp,
|
||||
spatial::SpatHostWaitLoadOp,
|
||||
spatial::SpatSyncOp,
|
||||
spatial::SpatWaitOp,
|
||||
spatial::SpatExtractRowsOp>();
|
||||
|
||||
RewritePatternSet initialPatterns(ctx);
|
||||
@@ -223,6 +225,8 @@ void onnx_mlir::raptor::SpatialToPimPass::runOnOperation() {
|
||||
spatial::SpatChannelSendOp,
|
||||
spatial::SpatHostStoreSyncOp,
|
||||
spatial::SpatHostWaitLoadOp,
|
||||
spatial::SpatSyncOp,
|
||||
spatial::SpatWaitOp,
|
||||
spatial::SpatExtractRowsOp>();
|
||||
|
||||
SmallVector<pim::PimCoreOp> coreOps;
|
||||
@@ -274,6 +278,8 @@ void onnx_mlir::raptor::SpatialToPimPass::runOnOperation() {
|
||||
spatial::SpatChannelSendOp,
|
||||
spatial::SpatHostStoreSyncOp,
|
||||
spatial::SpatHostWaitLoadOp,
|
||||
spatial::SpatSyncOp,
|
||||
spatial::SpatWaitOp,
|
||||
spatial::SpatExtractRowsOp>();
|
||||
|
||||
RewritePatternSet communicationPatterns(ctx);
|
||||
|
||||
@@ -302,12 +302,19 @@ static FailureOr<int64_t> getShapedByteSize(MemRefType type) {
|
||||
return static_cast<int64_t>(*byteSize);
|
||||
}
|
||||
|
||||
static FailureOr<SmallVector<int64_t>>
|
||||
struct LogicalCopyShape {
|
||||
SmallVector<int64_t> dimensions;
|
||||
Type elementType;
|
||||
};
|
||||
|
||||
static bool isPackedByteBuffer(MemRefType type) {
|
||||
return type.getRank() == 1 && type.getElementType().isInteger(8);
|
||||
}
|
||||
|
||||
static FailureOr<LogicalCopyShape>
|
||||
inferLogicalCopyShape(MemRefType targetType, MemRefType sourceType, int64_t size) {
|
||||
if (!targetType.hasStaticShape() || !sourceType.hasStaticShape())
|
||||
return failure();
|
||||
if (targetType.getElementType() != sourceType.getElementType() || targetType.getRank() != sourceType.getRank())
|
||||
return failure();
|
||||
|
||||
auto targetBytes = getShapedByteSize(targetType);
|
||||
auto sourceBytes = getShapedByteSize(sourceType);
|
||||
@@ -316,18 +323,37 @@ inferLogicalCopyShape(MemRefType targetType, MemRefType sourceType, int64_t size
|
||||
|
||||
bool targetMatches = *targetBytes == size;
|
||||
bool sourceMatches = *sourceBytes == size;
|
||||
if (targetMatches && sourceMatches && targetType.getShape() != sourceType.getShape())
|
||||
bool matchingTypes = targetType.getElementType() == sourceType.getElementType()
|
||||
&& targetType.getRank() == sourceType.getRank();
|
||||
if (matchingTypes) {
|
||||
if (targetMatches && sourceMatches
|
||||
&& targetType.getShape() != sourceType.getShape())
|
||||
return failure();
|
||||
MemRefType logicalType = targetMatches ? targetType : sourceType;
|
||||
if (targetMatches || sourceMatches)
|
||||
return LogicalCopyShape {
|
||||
SmallVector<int64_t>(logicalType.getShape()),
|
||||
logicalType.getElementType()};
|
||||
return failure();
|
||||
if (targetMatches)
|
||||
return SmallVector<int64_t>(targetType.getShape().begin(), targetType.getShape().end());
|
||||
if (sourceMatches)
|
||||
return SmallVector<int64_t>(sourceType.getShape().begin(), sourceType.getShape().end());
|
||||
}
|
||||
if (targetMatches && isPackedByteBuffer(sourceType))
|
||||
return LogicalCopyShape {
|
||||
SmallVector<int64_t>(targetType.getShape()),
|
||||
targetType.getElementType()};
|
||||
if (sourceMatches && isPackedByteBuffer(targetType))
|
||||
return LogicalCopyShape {
|
||||
SmallVector<int64_t>(sourceType.getShape()),
|
||||
sourceType.getElementType()};
|
||||
return failure();
|
||||
}
|
||||
|
||||
static FailureOr<int64_t> getContiguousSuffixRank(Value value, ArrayRef<int64_t> copyShape) {
|
||||
static FailureOr<int64_t> getContiguousSuffixRank(
|
||||
Value value, ArrayRef<int64_t> copyShape, Type elementType = {}) {
|
||||
auto type = dyn_cast<MemRefType>(value.getType());
|
||||
if (type && elementType && isPackedByteBuffer(type))
|
||||
return copyShape.size();
|
||||
if (!type || !type.hasStaticShape() || !hasByteSizedElementType(type.getElementType())
|
||||
|| (elementType && type.getElementType() != elementType)
|
||||
|| type.getRank() != static_cast<int64_t>(copyShape.size()))
|
||||
return failure();
|
||||
if (llvm::any_of(copyShape, [](int64_t dim) { return dim <= 0; }))
|
||||
@@ -351,6 +377,30 @@ static FailureOr<int64_t> getContiguousSuffixRank(Value value, ArrayRef<int64_t>
|
||||
return contiguousSuffixRank;
|
||||
}
|
||||
|
||||
static FailureOr<SmallVector<int64_t>> getOuterByteStrides(
|
||||
Value value, const LogicalCopyShape ©Shape, size_t outerRank) {
|
||||
auto type = cast<MemRefType>(value.getType());
|
||||
SmallVector<int64_t> strides;
|
||||
if (isPackedByteBuffer(type))
|
||||
strides = computeRowMajorStrides(copyShape.dimensions);
|
||||
else {
|
||||
auto proven = getProvenMemRefStrides(value);
|
||||
if (failed(proven))
|
||||
return failure();
|
||||
strides = std::move(*proven);
|
||||
}
|
||||
int64_t elementByteWidth = static_cast<int64_t>(
|
||||
getElementTypeSizeInBytes(copyShape.elementType));
|
||||
SmallVector<int64_t> result;
|
||||
for (int64_t stride : ArrayRef<int64_t>(strides).take_front(outerRank)) {
|
||||
auto byteStride = checkedPositiveMul(stride, elementByteWidth);
|
||||
if (failed(byteStride))
|
||||
return failure();
|
||||
result.push_back(*byteStride);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
static FailureOr<CopyEndpointPlan> analyzeCopyEndpoint(Value value, Value initialByteOffset, MemRefType logicalType) {
|
||||
if (!logicalType.hasStaticShape() || !hasByteSizedElementType(logicalType.getElementType()))
|
||||
return failure();
|
||||
@@ -448,8 +498,10 @@ analyzeCopyRewrite(Value target, Value source, Value targetOffset, Value sourceO
|
||||
if (failed(logicalCopyShape))
|
||||
return failure();
|
||||
|
||||
auto targetSuffixRank = getContiguousSuffixRank(target, *logicalCopyShape);
|
||||
auto sourceSuffixRank = getContiguousSuffixRank(source, *logicalCopyShape);
|
||||
auto targetSuffixRank = getContiguousSuffixRank(
|
||||
target, logicalCopyShape->dimensions, logicalCopyShape->elementType);
|
||||
auto sourceSuffixRank = getContiguousSuffixRank(
|
||||
source, logicalCopyShape->dimensions, logicalCopyShape->elementType);
|
||||
if (failed(targetSuffixRank) || failed(sourceSuffixRank))
|
||||
return failure();
|
||||
|
||||
@@ -458,23 +510,24 @@ analyzeCopyRewrite(Value target, Value source, Value targetOffset, Value sourceO
|
||||
plan.source = *sourcePlan;
|
||||
|
||||
int64_t contiguousSuffixRank = std::min(*targetSuffixRank, *sourceSuffixRank);
|
||||
if (contiguousSuffixRank == static_cast<int64_t>(logicalCopyShape->size())) {
|
||||
if (contiguousSuffixRank
|
||||
== static_cast<int64_t>(logicalCopyShape->dimensions.size())) {
|
||||
plan.kind = CopyRewritePlan::Kind::Direct;
|
||||
plan.directBytes = size;
|
||||
return plan;
|
||||
}
|
||||
|
||||
auto targetStrides = getProvenMemRefStrides(target);
|
||||
auto sourceStrides = getProvenMemRefStrides(source);
|
||||
if (failed(targetStrides) || failed(sourceStrides))
|
||||
return failure();
|
||||
|
||||
int64_t elementByteWidth = static_cast<int64_t>(getElementTypeSizeInBytes(targetType.getElementType()));
|
||||
int64_t elementByteWidth = static_cast<int64_t>(
|
||||
getElementTypeSizeInBytes(logicalCopyShape->elementType));
|
||||
plan.kind = CopyRewritePlan::Kind::Loop;
|
||||
plan.loop.targetBaseOffset = plan.target.offset;
|
||||
plan.loop.sourceBaseOffset = plan.source.offset;
|
||||
plan.loop.outerShape.assign(logicalCopyShape->begin(), logicalCopyShape->end() - contiguousSuffixRank);
|
||||
SmallVector<int64_t> chunkShape(logicalCopyShape->end() - contiguousSuffixRank, logicalCopyShape->end());
|
||||
plan.loop.outerShape.assign(
|
||||
logicalCopyShape->dimensions.begin(),
|
||||
logicalCopyShape->dimensions.end() - contiguousSuffixRank);
|
||||
SmallVector<int64_t> chunkShape(
|
||||
logicalCopyShape->dimensions.end() - contiguousSuffixRank,
|
||||
logicalCopyShape->dimensions.end());
|
||||
auto outerElements = checkedPositiveProduct(plan.loop.outerShape);
|
||||
auto chunkElements = checkedPositiveProduct(chunkShape);
|
||||
auto chunkBytes = failed(chunkElements)
|
||||
@@ -484,18 +537,14 @@ analyzeCopyRewrite(Value target, Value source, Value targetOffset, Value sourceO
|
||||
return failure();
|
||||
plan.loop.outerElements = *outerElements;
|
||||
plan.loop.chunkBytes = *chunkBytes;
|
||||
for (int64_t stride : ArrayRef<int64_t>(*targetStrides).take_front(plan.loop.outerShape.size())) {
|
||||
auto byteStride = checkedPositiveMul(stride, elementByteWidth);
|
||||
if (failed(byteStride))
|
||||
return failure();
|
||||
plan.loop.targetOuterByteStrides.push_back(*byteStride);
|
||||
}
|
||||
for (int64_t stride : ArrayRef<int64_t>(*sourceStrides).take_front(plan.loop.outerShape.size())) {
|
||||
auto byteStride = checkedPositiveMul(stride, elementByteWidth);
|
||||
if (failed(byteStride))
|
||||
return failure();
|
||||
plan.loop.sourceOuterByteStrides.push_back(*byteStride);
|
||||
}
|
||||
auto targetStrides = getOuterByteStrides(
|
||||
target, *logicalCopyShape, plan.loop.outerShape.size());
|
||||
auto sourceStrides = getOuterByteStrides(
|
||||
source, *logicalCopyShape, plan.loop.outerShape.size());
|
||||
if (failed(targetStrides) || failed(sourceStrides))
|
||||
return failure();
|
||||
plan.loop.targetOuterByteStrides = std::move(*targetStrides);
|
||||
plan.loop.sourceOuterByteStrides = std::move(*sourceStrides);
|
||||
if (plan.loop.chunkBytes <= 0)
|
||||
return failure();
|
||||
return plan;
|
||||
|
||||
@@ -602,18 +602,30 @@ static LogicalResult normalizePimMemory(ModuleOp moduleOp, func::FuncOp funcOp)
|
||||
PatternRewriter rewriter(ctx);
|
||||
|
||||
SmallVector<MemRefCopyWorkItem> copyWorklist;
|
||||
SmallVector<PimMemCopyDevToHostOp> hostToHostCopies;
|
||||
llvm::SmallPtrSet<Operation*, 16> seenCopyOps;
|
||||
llvm::SmallPtrSet<Operation*, 4> seenHostToHostCopies;
|
||||
auto addCopyOp = [&](memref::CopyOp copyOp, const StaticValueKnowledge& knowledge) {
|
||||
if (seenCopyOps.insert(copyOp.getOperation()).second)
|
||||
copyWorklist.push_back({copyOp, knowledge});
|
||||
};
|
||||
auto collectCopy = [&](Operation &op,
|
||||
const StaticValueKnowledge &knowledge) {
|
||||
if (auto copyOp = dyn_cast<memref::CopyOp>(&op))
|
||||
addCopyOp(copyOp, knowledge);
|
||||
if (auto copyOp = dyn_cast<PimMemCopyDevToHostOp>(&op);
|
||||
copyOp
|
||||
&& isHostBackedPimAddress(copyOp.getDeviceSource(), knowledge)
|
||||
&& isHostBackedPimAddress(copyOp.getHostTarget(), knowledge)
|
||||
&& seenHostToHostCopies.insert(copyOp).second)
|
||||
hostToHostCopies.push_back(copyOp);
|
||||
};
|
||||
|
||||
moduleOp.walk([&](pim::PimCoreOp coreOp) {
|
||||
StaticValueKnowledge knowledge = seedCoreKnowledge(coreOp);
|
||||
(void) walkPimCoreBlockStructurally(
|
||||
coreOp.getBody().front(), knowledge, [&](Operation& op, const StaticValueKnowledge& opKnowledge) {
|
||||
if (auto copyOp = dyn_cast<memref::CopyOp>(&op))
|
||||
addCopyOp(copyOp, opKnowledge);
|
||||
collectCopy(op, opKnowledge);
|
||||
return success();
|
||||
});
|
||||
});
|
||||
@@ -622,8 +634,7 @@ static LogicalResult normalizePimMemory(ModuleOp moduleOp, func::FuncOp funcOp)
|
||||
StaticValueKnowledge knowledge = seedCoreBatchKnowledge(coreBatchOp, lane);
|
||||
(void) walkPimCoreBlockStructurally(
|
||||
coreBatchOp.getBody().front(), knowledge, [&](Operation& op, const StaticValueKnowledge& opKnowledge) {
|
||||
if (auto copyOp = dyn_cast<memref::CopyOp>(&op))
|
||||
addCopyOp(copyOp, opKnowledge);
|
||||
collectCopy(op, opKnowledge);
|
||||
return success();
|
||||
});
|
||||
}
|
||||
@@ -631,6 +642,22 @@ static LogicalResult normalizePimMemory(ModuleOp moduleOp, func::FuncOp funcOp)
|
||||
|
||||
bool hasFailed = false;
|
||||
Value zeroOffset = getOrCreateIndexConstant(rewriter, funcOp, 0);
|
||||
for (PimMemCopyDevToHostOp copyOp : hostToHostCopies) {
|
||||
rewriter.setInsertionPoint(copyOp);
|
||||
auto scratchType = MemRefType::get(
|
||||
{copyOp.getSize()}, rewriter.getI8Type());
|
||||
Value scratch = memref::AllocOp::create(
|
||||
rewriter, copyOp.getLoc(), scratchType);
|
||||
auto load = PimMemCopyHostToDevOp::create(
|
||||
rewriter, copyOp.getLoc(), scratchType, zeroOffset,
|
||||
copyOp.getDeviceSourceOffset(), scratch, copyOp.getDeviceSource(),
|
||||
copyOp.getSizeAttr());
|
||||
auto store = PimMemCopyDevToHostOp::create(
|
||||
rewriter, copyOp.getLoc(), copyOp.getHostTarget().getType(),
|
||||
copyOp.getHostTargetOffset(), zeroOffset, copyOp.getHostTarget(),
|
||||
load.getOutput(), copyOp.getSizeAttr());
|
||||
rewriter.replaceOp(copyOp, store.getOutput());
|
||||
}
|
||||
for (const MemRefCopyWorkItem& workItem : copyWorklist) {
|
||||
memref::CopyOp copyOp = workItem.copyOp;
|
||||
rewriter.setInsertionPoint(copyOp);
|
||||
|
||||
@@ -136,7 +136,7 @@ def PimWaitOp : PimOp<"wait", []> {
|
||||
|
||||
let arguments = (ins
|
||||
Index:$eventRegister,
|
||||
I32Attr:$waitValue
|
||||
Index:$waitValue
|
||||
);
|
||||
|
||||
let assemblyFormat = [{
|
||||
|
||||
+139
-1
@@ -242,11 +242,146 @@ static void appendReceive(BoundaryProgram &boundary,
|
||||
target.collection, {slice}, {0, 1}, {target.position}, {lanes}, lanes});
|
||||
}
|
||||
|
||||
struct HostTransferRef {
|
||||
ExternalTransferFamily *family = nullptr;
|
||||
size_t index = 0;
|
||||
};
|
||||
|
||||
static unsigned getBarrierRoundCount(size_t coreCount) {
|
||||
unsigned rounds = 0;
|
||||
for (size_t distance = 1; distance < coreCount; distance *= 2)
|
||||
++rounds;
|
||||
return rounds;
|
||||
}
|
||||
|
||||
static LogicalResult assignPipelineSynchronization(
|
||||
DeferredTransferPlan &transfers,
|
||||
ArrayRef<BoundaryProgram> boundaries,
|
||||
size_t synchronizationRegisterCount) {
|
||||
bool pipelined = false;
|
||||
for (ScheduledInfo &scheduled : transfers.scheduled) {
|
||||
if (scheduled.pipelineStages.empty())
|
||||
continue;
|
||||
pipelined = true;
|
||||
llvm::append_range(transfers.downstreamCores, scheduled.cores);
|
||||
for (auto [core, stage] :
|
||||
llvm::zip_equal(scheduled.cores, scheduled.pipelineStages))
|
||||
if (stage == 0)
|
||||
transfers.stageZeroCores.push_back(core);
|
||||
}
|
||||
if (!pipelined)
|
||||
return success();
|
||||
transfers.synchronizationRegisterCount = synchronizationRegisterCount;
|
||||
llvm::sort(transfers.stageZeroCores);
|
||||
transfers.stageZeroCores.erase(
|
||||
llvm::unique(transfers.stageZeroCores), transfers.stageZeroCores.end());
|
||||
llvm::sort(transfers.downstreamCores);
|
||||
transfers.downstreamCores.erase(
|
||||
llvm::unique(transfers.downstreamCores),
|
||||
transfers.downstreamCores.end());
|
||||
llvm::erase_if(transfers.downstreamCores, [&](int64_t core) {
|
||||
return llvm::is_contained(transfers.stageZeroCores, core);
|
||||
});
|
||||
|
||||
DenseMap<int64_t, SmallVector<HostTransferRef>> incomingByCore;
|
||||
DenseMap<ExternalTransferFamily *, SmallVector<int64_t>> eventRegisters;
|
||||
DenseMap<ExternalTransferFamily *, SmallVector<int64_t>> waitValues;
|
||||
DenseMap<ExternalTransferFamily *, SmallVector<int64_t>> acknowledgementRegisters;
|
||||
auto initialize = [&](ExternalTransferFamily &family) {
|
||||
size_t count = family.targetCores.size();
|
||||
eventRegisters.try_emplace(&family, count, 0);
|
||||
waitValues.try_emplace(&family, count, 0);
|
||||
acknowledgementRegisters.try_emplace(&family, count, 0);
|
||||
};
|
||||
for (const BoundaryProgram &boundary : boundaries)
|
||||
for (const BoundaryInstruction &instruction : boundary.instructions) {
|
||||
auto *receive = std::get_if<EmitReceiveAssemblyRun>(&instruction);
|
||||
if (!receive || receive->slices.empty()
|
||||
|| !receive->slices.front().family->hostRouted)
|
||||
continue;
|
||||
for (const ScheduledTransferSlice &slice : receive->slices) {
|
||||
ExternalTransferFamily &family = *slice.family;
|
||||
initialize(family);
|
||||
for (size_t offset = 0; offset < slice.transferCount; ++offset) {
|
||||
size_t index = slice.familyOffset + offset;
|
||||
int64_t source = family.sourceCores.valueAt(index);
|
||||
int64_t target = family.targetCores.valueAt(index);
|
||||
incomingByCore[target].push_back({&family, index});
|
||||
++transfers.hostAcknowledgementCounts[source];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
unsigned barrierRounds = getBarrierRoundCount(
|
||||
transfers.stageZeroCores.size());
|
||||
bool stageZeroNeedsAcknowledgements = llvm::any_of(
|
||||
transfers.stageZeroCores, [&](int64_t core) {
|
||||
return transfers.hostAcknowledgementCounts.contains(core);
|
||||
});
|
||||
for (auto &[target, incoming] : incomingByCore) {
|
||||
bool needsAcknowledgementRegister =
|
||||
transfers.hostAcknowledgementCounts.contains(target);
|
||||
bool stageZero = llvm::is_contained(transfers.stageZeroCores, target);
|
||||
size_t reserved = stageZero
|
||||
? barrierRounds + (stageZeroNeedsAcknowledgements ? 1 : 0)
|
||||
: 1 + (needsAcknowledgementRegister ? 1 : 0);
|
||||
if (reserved >= synchronizationRegisterCount) {
|
||||
incoming.front().family->requirement->exchange->deferred.emitOpError(
|
||||
"pipeline synchronization leaves no event register for incoming host transfers");
|
||||
return failure();
|
||||
}
|
||||
size_t groupCount = std::min(
|
||||
incoming.size(), synchronizationRegisterCount - reserved);
|
||||
// One wait consumes a complete consecutive group of producer signals.
|
||||
SmallVector<size_t> groupSizes(groupCount);
|
||||
for (size_t ordinal = 0; ordinal < incoming.size(); ++ordinal)
|
||||
++groupSizes[ordinal * groupCount / incoming.size()];
|
||||
SmallVector<bool> first(groupCount, true);
|
||||
for (size_t ordinal = 0; ordinal < incoming.size(); ++ordinal) {
|
||||
size_t group = ordinal * groupCount / incoming.size();
|
||||
HostTransferRef transfer = incoming[ordinal];
|
||||
eventRegisters[transfer.family][transfer.index] = group;
|
||||
acknowledgementRegisters[transfer.family][transfer.index] =
|
||||
synchronizationRegisterCount - 1;
|
||||
if (first[group]) {
|
||||
waitValues[transfer.family][transfer.index] = groupSizes[group];
|
||||
first[group] = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (auto &[family, values] : eventRegisters) {
|
||||
family->eventRegisters = StaticIntSequence::fromValues(values);
|
||||
family->waitValues = StaticIntSequence::fromValues(waitValues[family]);
|
||||
family->acknowledgementEventRegisters =
|
||||
StaticIntSequence::fromValues(acknowledgementRegisters[family]);
|
||||
}
|
||||
|
||||
if (!transfers.stageZeroCores.empty()) {
|
||||
size_t reserved = barrierRounds
|
||||
+ (stageZeroNeedsAcknowledgements ? 1 : 0);
|
||||
if (reserved > synchronizationRegisterCount)
|
||||
return transfers.scheduled.front().op->emitOpError(
|
||||
"pipeline stage-zero barrier requires more synchronization registers than the target provides");
|
||||
}
|
||||
if (!transfers.downstreamCores.empty()) {
|
||||
bool needsAcknowledgements = llvm::any_of(
|
||||
transfers.downstreamCores, [&](int64_t core) {
|
||||
return transfers.hostAcknowledgementCounts.contains(core);
|
||||
});
|
||||
if (1 + (needsAcknowledgements ? 1 : 0)
|
||||
> synchronizationRegisterCount)
|
||||
return transfers.scheduled.front().op->emitOpError(
|
||||
"pipeline stage-zero release requires more synchronization registers than the target provides");
|
||||
}
|
||||
return success();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
FailureOr<DeferredBoundaryPlan> buildDeferredBoundaryPlan(
|
||||
DeferredTransferPlan &transfers,
|
||||
const ScheduledCommunicationPlan &schedule) {
|
||||
const ScheduledCommunicationPlan &schedule,
|
||||
size_t synchronizationRegisterCount) {
|
||||
DeferredBoundaryPlan result;
|
||||
SmallVector<BoundaryProgram> boundaries;
|
||||
DenseMap<BoundaryKey, unsigned> indices;
|
||||
@@ -373,6 +508,9 @@ FailureOr<DeferredBoundaryPlan> buildDeferredBoundaryPlan(
|
||||
return std::tie(scheduledOrder[lhs.key.first], lhs.key.second)
|
||||
< std::tie(scheduledOrder[rhs.key.first], rhs.key.second);
|
||||
});
|
||||
if (failed(assignPipelineSynchronization(
|
||||
transfers, boundaries, synchronizationRegisterCount)))
|
||||
return failure();
|
||||
result.boundaries = std::move(boundaries);
|
||||
return result;
|
||||
}
|
||||
|
||||
+2
-1
@@ -53,6 +53,7 @@ struct DeferredBoundaryPlan {
|
||||
};
|
||||
|
||||
mlir::FailureOr<DeferredBoundaryPlan> buildDeferredBoundaryPlan(DeferredTransferPlan& transfers,
|
||||
const ScheduledCommunicationPlan& schedule);
|
||||
const ScheduledCommunicationPlan& schedule,
|
||||
size_t synchronizationRegisterCount);
|
||||
|
||||
} // namespace onnx_mlir::spatial
|
||||
|
||||
+233
-5
@@ -4,6 +4,7 @@
|
||||
#include "DeferredBoundaryRealization.hpp"
|
||||
#include "DeferredProjectionAnalysis.hpp"
|
||||
#include "DeferredResultRealization.hpp"
|
||||
#include "DeferredTransferPlanning.hpp"
|
||||
#include "src/Accelerators/PIM/Common/IR/LoopUtils.hpp"
|
||||
#include "src/Accelerators/PIM/Common/IR/StaticIntGrid.hpp"
|
||||
#include "src/Accelerators/PIM/Common/IR/StaticIntSequence.hpp"
|
||||
@@ -21,6 +22,8 @@ struct LogicalTransferMetadataView {
|
||||
StaticIntSequenceChain targetCores;
|
||||
StaticIntSequenceChain hostOffsets;
|
||||
StaticIntSequenceChain eventRegisters;
|
||||
StaticIntSequenceChain waitValues;
|
||||
StaticIntSequenceChain acknowledgementEventRegisters;
|
||||
StaticIntSequenceChain targetLanes;
|
||||
StaticIntSequenceChain localOffsets;
|
||||
SmallVector<StaticIntSequenceChain> projectionOffsets;
|
||||
@@ -91,6 +94,9 @@ static void appendMetadata(const ScheduledTransferSlice &slice, LogicalTransferM
|
||||
metadata.hostOffsets.append(family.hostOffsets, familyIndex, count);
|
||||
metadata.eventRegisters.append(
|
||||
family.eventRegisters, familyIndex, count);
|
||||
metadata.waitValues.append(family.waitValues, familyIndex, count);
|
||||
metadata.acknowledgementEventRegisters.append(
|
||||
family.acknowledgementEventRegisters, familyIndex, count);
|
||||
}
|
||||
metadata.targetLanes.append(StaticIntSequence::affine(targetLane, 1, count));
|
||||
if (family.requirement->producerLocalOffsets)
|
||||
@@ -296,13 +302,21 @@ static FailureOr<Value> emitReceiveValue(ArrayRef<ScheduledTransferSlice> slices
|
||||
if (failed(grids)) return failure();
|
||||
std::optional<StaticIntGrid> hostOffsets;
|
||||
std::optional<StaticIntGrid> eventRegisters;
|
||||
std::optional<StaticIntGrid> waitValues;
|
||||
std::optional<StaticIntGrid> acknowledgementEventRegisters;
|
||||
if (slices.front().family->hostRouted) {
|
||||
auto offsets = buildGrid(metadata.hostOffsets);
|
||||
auto events = buildGrid(metadata.eventRegisters);
|
||||
if (failed(offsets) || failed(events))
|
||||
auto waits = buildGrid(metadata.waitValues);
|
||||
auto acknowledgements = buildGrid(
|
||||
metadata.acknowledgementEventRegisters);
|
||||
if (failed(offsets) || failed(events) || failed(waits)
|
||||
|| failed(acknowledgements))
|
||||
return failure();
|
||||
hostOffsets = std::move(*offsets);
|
||||
eventRegisters = std::move(*events);
|
||||
waitValues = std::move(*waits);
|
||||
acknowledgementEventRegisters = std::move(*acknowledgements);
|
||||
}
|
||||
Value position = lane ? lane : context.constants.getIndex(0);
|
||||
Value row = context.constants.getIndex(0);
|
||||
@@ -319,6 +333,10 @@ static FailureOr<Value> emitReceiveValue(ArrayRef<ScheduledTransferSlice> slices
|
||||
hostOffsets->emitLookup(
|
||||
row, position, anchor, context.constants, context.rewriter, anchor->getLoc()),
|
||||
eventRegisters->emitLookup(
|
||||
row, position, anchor, context.constants, context.rewriter, anchor->getLoc()),
|
||||
waitValues->emitLookup(
|
||||
row, position, anchor, context.constants, context.rewriter, anchor->getLoc()),
|
||||
acknowledgementEventRegisters->emitLookup(
|
||||
row, position, anchor, context.constants, context.rewriter, anchor->getLoc()));
|
||||
receive = op;
|
||||
output = op.getOutput();
|
||||
@@ -387,6 +405,8 @@ static FailureOr<Value> emitReceiveAssembly(const EmitReceiveAssemblyRun &run, V
|
||||
std::optional<StaticIntGrid> positions;
|
||||
std::optional<StaticIntGrid> hostOffsets;
|
||||
std::optional<StaticIntGrid> eventRegisters;
|
||||
std::optional<StaticIntGrid> waitValues;
|
||||
std::optional<StaticIntGrid> acknowledgementEventRegisters;
|
||||
bool hostRouted = run.slices.front().family->hostRouted;
|
||||
auto metadataByEntry = buildRectangularReceiveMetadata(run, laneCount);
|
||||
if (succeeded(metadataByEntry)) {
|
||||
@@ -402,10 +422,17 @@ static FailureOr<Value> emitReceiveAssembly(const EmitReceiveAssemblyRun &run, V
|
||||
&LogicalTransferMetadataView::hostOffsets);
|
||||
auto events = buildRows(
|
||||
&LogicalTransferMetadataView::eventRegisters);
|
||||
if (failed(offsets) || failed(events))
|
||||
auto waits = buildRows(
|
||||
&LogicalTransferMetadataView::waitValues);
|
||||
auto acknowledgements = buildRows(
|
||||
&LogicalTransferMetadataView::acknowledgementEventRegisters);
|
||||
if (failed(offsets) || failed(events) || failed(waits)
|
||||
|| failed(acknowledgements))
|
||||
return failure();
|
||||
hostOffsets = std::move(*offsets);
|
||||
eventRegisters = std::move(*events);
|
||||
waitValues = std::move(*waits);
|
||||
acknowledgementEventRegisters = std::move(*acknowledgements);
|
||||
}
|
||||
SmallVector<StaticIntSequence> positionRows;
|
||||
for (unsigned position : run.positions)
|
||||
@@ -456,10 +483,17 @@ static FailureOr<Value> emitReceiveAssembly(const EmitReceiveAssemblyRun &run, V
|
||||
&LogicalTransferMetadataView::hostOffsets);
|
||||
auto events = buildGrid(
|
||||
&LogicalTransferMetadataView::eventRegisters);
|
||||
if (failed(offsets) || failed(events))
|
||||
auto waits = buildGrid(
|
||||
&LogicalTransferMetadataView::waitValues);
|
||||
auto acknowledgements = buildGrid(
|
||||
&LogicalTransferMetadataView::acknowledgementEventRegisters);
|
||||
if (failed(offsets) || failed(events) || failed(waits)
|
||||
|| failed(acknowledgements))
|
||||
return failure();
|
||||
hostOffsets = std::move(*offsets);
|
||||
eventRegisters = std::move(*events);
|
||||
waitValues = std::move(*waits);
|
||||
acknowledgementEventRegisters = std::move(*acknowledgements);
|
||||
}
|
||||
SmallVector<StaticIntSequence> positionColumns;
|
||||
for (const StaticIntSequenceChain &values : positionsByLane)
|
||||
@@ -491,6 +525,10 @@ static FailureOr<Value> emitReceiveAssembly(const EmitReceiveAssemblyRun &run, V
|
||||
hostOffsets->emitLookup(
|
||||
entry, runtimeLane, anchor, context.constants, context.rewriter, loc),
|
||||
eventRegisters->emitLookup(
|
||||
entry, runtimeLane, anchor, context.constants, context.rewriter, loc),
|
||||
waitValues->emitLookup(
|
||||
entry, runtimeLane, anchor, context.constants, context.rewriter, loc),
|
||||
acknowledgementEventRegisters->emitLookup(
|
||||
entry, runtimeLane, anchor, context.constants, context.rewriter, loc));
|
||||
receive = op;
|
||||
output = op.getOutput();
|
||||
@@ -1158,9 +1196,199 @@ static LogicalResult emitBoundary(const BoundaryProgram &boundary, ArrayRef<Defe
|
||||
return failed(values) ? failure() : replaceResults(exchanges, *values, replacements);
|
||||
}
|
||||
|
||||
static unsigned getBarrierRoundCount(size_t coreCount) {
|
||||
unsigned rounds = 0;
|
||||
for (size_t distance = 1; distance < coreCount; distance *= 2)
|
||||
++rounds;
|
||||
return rounds;
|
||||
}
|
||||
|
||||
static LogicalResult emitCompletionSynchronization(
|
||||
DeferredTransferPlan &transfers, DeferredEmissionContext &context) {
|
||||
if (transfers.synchronizationRegisterCount == 0)
|
||||
return success();
|
||||
size_t acknowledgementRegister =
|
||||
transfers.synchronizationRegisterCount - 1;
|
||||
unsigned barrierRounds = getBarrierRoundCount(
|
||||
transfers.stageZeroCores.size());
|
||||
bool stageZeroNeedsAcknowledgements = llvm::any_of(
|
||||
transfers.stageZeroCores, [&](int64_t core) {
|
||||
return transfers.hostAcknowledgementCounts.contains(core);
|
||||
});
|
||||
size_t firstBarrierRegister = acknowledgementRegister
|
||||
- (stageZeroNeedsAcknowledgements ? 1 : 0);
|
||||
DenseMap<int64_t, unsigned> stageZeroRank;
|
||||
for (auto [rank, core] : llvm::enumerate(transfers.stageZeroCores))
|
||||
stageZeroRank[core] = rank;
|
||||
DenseMap<int64_t, unsigned> downstreamRank;
|
||||
for (auto [rank, core] : llvm::enumerate(transfers.downstreamCores))
|
||||
downstreamRank[core] = rank;
|
||||
auto getReleaseRegister = [&](int64_t core) {
|
||||
return acknowledgementRegister
|
||||
- (transfers.hostAcknowledgementCounts.contains(core) ? 1 : 0);
|
||||
};
|
||||
|
||||
for (ScheduledInfo &scheduled : transfers.scheduled) {
|
||||
Block *block = scheduled.blocks.front();
|
||||
context.rewriter.setInsertionPoint(block->getTerminator());
|
||||
Location loc = scheduled.op->getLoc();
|
||||
Value lane;
|
||||
if (auto batch = dyn_cast<SpatScheduledComputeBatch>(scheduled.op))
|
||||
lane = *batch.getLaneArgument();
|
||||
|
||||
SmallVector<int64_t> acknowledgementCounts, releaseRegisters;
|
||||
SmallVector<int64_t> releaseWaitValues, leftTargets, leftRegisters;
|
||||
SmallVector<int64_t> rightTargets, rightRegisters;
|
||||
LaneSet barrierLanes, leaderLanes, leftLanes, rightLanes;
|
||||
for (auto [index, core] : llvm::enumerate(scheduled.cores)) {
|
||||
acknowledgementCounts.push_back(
|
||||
transfers.hostAcknowledgementCounts.lookup(core));
|
||||
if (stageZeroRank.contains(core))
|
||||
barrierLanes = barrierLanes.unite(
|
||||
LaneSet::range(index, index + 1));
|
||||
if (!transfers.stageZeroCores.empty()
|
||||
&& core == transfers.stageZeroCores.front())
|
||||
leaderLanes = leaderLanes.unite(LaneSet::range(index, index + 1));
|
||||
|
||||
auto rank = downstreamRank.find(core);
|
||||
if (rank == downstreamRank.end()) {
|
||||
releaseRegisters.push_back(0);
|
||||
releaseWaitValues.push_back(0);
|
||||
leftTargets.push_back(core);
|
||||
leftRegisters.push_back(0);
|
||||
rightTargets.push_back(core);
|
||||
rightRegisters.push_back(0);
|
||||
continue;
|
||||
}
|
||||
releaseRegisters.push_back(getReleaseRegister(core));
|
||||
releaseWaitValues.push_back(1);
|
||||
size_t left = 2 * rank->second + 1;
|
||||
size_t right = left + 1;
|
||||
if (left < transfers.downstreamCores.size()) {
|
||||
int64_t child = transfers.downstreamCores[left];
|
||||
leftTargets.push_back(child);
|
||||
leftRegisters.push_back(getReleaseRegister(child));
|
||||
leftLanes = leftLanes.unite(LaneSet::range(index, index + 1));
|
||||
} else {
|
||||
leftTargets.push_back(core);
|
||||
leftRegisters.push_back(0);
|
||||
}
|
||||
if (right < transfers.downstreamCores.size()) {
|
||||
int64_t child = transfers.downstreamCores[right];
|
||||
rightTargets.push_back(child);
|
||||
rightRegisters.push_back(getReleaseRegister(child));
|
||||
rightLanes = rightLanes.unite(LaneSet::range(index, index + 1));
|
||||
} else {
|
||||
rightTargets.push_back(core);
|
||||
rightRegisters.push_back(0);
|
||||
}
|
||||
}
|
||||
Value runtimeLane = lane ? lane : context.constants.getIndex(0);
|
||||
auto emitForLanes = [&](const LaneSet &active, auto emit) -> LogicalResult {
|
||||
if (active.empty())
|
||||
return success();
|
||||
if (!lane) {
|
||||
if (active.contains(0))
|
||||
emit();
|
||||
return success();
|
||||
}
|
||||
auto condition = emitLaneCondition(
|
||||
active, lane, scheduled.cores.size(), scheduled.op, context, loc);
|
||||
if (failed(condition))
|
||||
return failure();
|
||||
auto conditional = scf::IfOp::create(
|
||||
context.rewriter, loc, TypeRange {}, *condition, false);
|
||||
OpBuilder::InsertionGuard guard(context.rewriter);
|
||||
context.rewriter.setInsertionPoint(
|
||||
conditional.getThenRegion().front().getTerminator());
|
||||
emit();
|
||||
return success();
|
||||
};
|
||||
Value acknowledgementCount = emitStaticIntLookup(
|
||||
StaticIntSequence::fromValues(acknowledgementCounts),
|
||||
runtimeLane, scheduled.op,
|
||||
context.constants, context.rewriter, loc);
|
||||
SpatWaitOp::create(
|
||||
context.rewriter, loc,
|
||||
context.constants.getIndex(acknowledgementRegister),
|
||||
acknowledgementCount);
|
||||
|
||||
// Dissemination barrier: every round doubles the covered stage-zero peers.
|
||||
auto emitBarrier = [&]() {
|
||||
for (unsigned round = 0; round < barrierRounds; ++round) {
|
||||
SmallVector<int64_t> targets;
|
||||
targets.reserve(scheduled.cores.size());
|
||||
size_t distance = size_t {1} << round;
|
||||
for (int64_t core : scheduled.cores) {
|
||||
auto rank = stageZeroRank.find(core);
|
||||
targets.push_back(rank == stageZeroRank.end()
|
||||
? core
|
||||
: transfers.stageZeroCores[
|
||||
(rank->second + distance)
|
||||
% transfers.stageZeroCores.size()]);
|
||||
}
|
||||
Value target = emitStaticIntLookup(
|
||||
StaticIntSequence::fromValues(targets),
|
||||
runtimeLane, scheduled.op,
|
||||
context.constants, context.rewriter, loc);
|
||||
Value eventRegister = context.constants.getIndex(
|
||||
firstBarrierRegister - round);
|
||||
SpatSyncOp::create(
|
||||
context.rewriter, loc, target, eventRegister);
|
||||
SpatWaitOp::create(
|
||||
context.rewriter, loc, eventRegister,
|
||||
context.constants.getIndex(1));
|
||||
}
|
||||
};
|
||||
if (barrierRounds > 0
|
||||
&& failed(emitForLanes(barrierLanes, emitBarrier)))
|
||||
return failure();
|
||||
|
||||
// Gate downstream restarts so no core advances the simulator input
|
||||
// iteration ahead of stage zero.
|
||||
if (!transfers.downstreamCores.empty()
|
||||
&& failed(emitForLanes(leaderLanes, [&]() {
|
||||
int64_t root = transfers.downstreamCores.front();
|
||||
SpatSyncOp::create(
|
||||
context.rewriter, loc, context.constants.getIndex(root),
|
||||
context.constants.getIndex(getReleaseRegister(root)));
|
||||
})))
|
||||
return failure();
|
||||
|
||||
Value releaseRegister = emitStaticIntLookup(
|
||||
StaticIntSequence::fromValues(releaseRegisters), runtimeLane,
|
||||
scheduled.op, context.constants, context.rewriter, loc);
|
||||
Value releaseWaitValue = emitStaticIntLookup(
|
||||
StaticIntSequence::fromValues(releaseWaitValues), runtimeLane,
|
||||
scheduled.op, context.constants, context.rewriter, loc);
|
||||
SpatWaitOp::create(
|
||||
context.rewriter, loc, releaseRegister, releaseWaitValue);
|
||||
|
||||
auto emitChild = [&](ArrayRef<int64_t> targets,
|
||||
ArrayRef<int64_t> registers) {
|
||||
Value target = emitStaticIntLookup(
|
||||
StaticIntSequence::fromValues(targets), runtimeLane, scheduled.op,
|
||||
context.constants, context.rewriter, loc);
|
||||
Value eventRegister = emitStaticIntLookup(
|
||||
StaticIntSequence::fromValues(registers), runtimeLane, scheduled.op,
|
||||
context.constants, context.rewriter, loc);
|
||||
SpatSyncOp::create(context.rewriter, loc, target, eventRegister);
|
||||
};
|
||||
if (failed(emitForLanes(leftLanes, [&]() {
|
||||
emitChild(leftTargets, leftRegisters);
|
||||
}))
|
||||
|| failed(emitForLanes(rightLanes, [&]() {
|
||||
emitChild(rightTargets, rightRegisters);
|
||||
})))
|
||||
return failure();
|
||||
}
|
||||
return success();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
LogicalResult realizeDeferredBoundaries(ArrayRef<BoundaryProgram> boundaries, ArrayRef<DeferredResultPlan> results, DeferredEmissionContext &context,
|
||||
LogicalResult realizeDeferredBoundaries(ArrayRef<BoundaryProgram> boundaries, ArrayRef<DeferredResultPlan> results,
|
||||
DeferredTransferPlan &transfers, DeferredEmissionContext &context,
|
||||
DeferredReplacementMap &replacements) {
|
||||
ScheduledInfo *scheduled = nullptr;
|
||||
for (const BoundaryProgram &boundary : boundaries) {
|
||||
@@ -1171,7 +1399,7 @@ LogicalResult realizeDeferredBoundaries(ArrayRef<BoundaryProgram> boundaries, Ar
|
||||
if (failed(emitBoundary(boundary, results, context, replacements)))
|
||||
return boundary.key.first->op->emitOpError("phase 2 failed to realize a communication boundary");
|
||||
}
|
||||
return success();
|
||||
return emitCompletionSynchronization(transfers, context);
|
||||
}
|
||||
|
||||
} // namespace onnx_mlir::spatial
|
||||
|
||||
+1
@@ -42,6 +42,7 @@ using DeferredReplacementMap =
|
||||
|
||||
mlir::LogicalResult realizeDeferredBoundaries(mlir::ArrayRef<BoundaryProgram> boundaries,
|
||||
mlir::ArrayRef<DeferredResultPlan> results,
|
||||
DeferredTransferPlan& transfers,
|
||||
DeferredEmissionContext& context,
|
||||
DeferredReplacementMap& replacements);
|
||||
|
||||
|
||||
+3
@@ -236,6 +236,9 @@ struct ExternalTransferFamily {
|
||||
StaticIntSequence channelIds = StaticIntSequence::uniform(0, 1);
|
||||
StaticIntSequence hostOffsets = StaticIntSequence::uniform(0, 1);
|
||||
StaticIntSequence eventRegisters = StaticIntSequence::uniform(0, 1);
|
||||
StaticIntSequence waitValues = StaticIntSequence::uniform(1, 1);
|
||||
StaticIntSequence acknowledgementEventRegisters =
|
||||
StaticIntSequence::uniform(0, 1);
|
||||
bool hostRouted = false;
|
||||
};
|
||||
|
||||
|
||||
+5
-2
@@ -232,7 +232,8 @@ LogicalResult realizeDeferredCommunication(func::FuncOp funcOp,
|
||||
auto schedule = scheduleDeferredCommunication(funcOp, *transfers);
|
||||
if (failed(schedule) || failed(verifyPlannedCommunicationDeadlockFree(funcOp, transfers->stepCounts, *schedule)))
|
||||
return funcOp.emitOpError("phase 2 failed to schedule symbolic communication");
|
||||
auto boundaries = buildDeferredBoundaryPlan(*transfers, *schedule);
|
||||
auto boundaries = buildDeferredBoundaryPlan(
|
||||
*transfers, *schedule, target.synchronizationRegisterCount);
|
||||
if (failed(boundaries))
|
||||
return funcOp.emitOpError("phase 2 failed to build sparse boundary programs");
|
||||
|
||||
@@ -242,7 +243,9 @@ LogicalResult realizeDeferredCommunication(func::FuncOp funcOp,
|
||||
ConstantPool constants(funcOp, rewriter);
|
||||
DeferredEmissionContext context(rewriter, constants);
|
||||
DeferredReplacementMap replacements;
|
||||
if (failed(realizeDeferredBoundaries(boundaries->boundaries, boundaries->results, context, replacements)))
|
||||
if (failed(realizeDeferredBoundaries(
|
||||
boundaries->boundaries, boundaries->results, *transfers,
|
||||
context, replacements)))
|
||||
return failure();
|
||||
for (auto [op, replacement] : replacements) {
|
||||
if (op->getResult(0) == replacement)
|
||||
|
||||
+2
-17
@@ -322,8 +322,7 @@ static LogicalResult buildRequirementFamilies(DeferredTransferPlan& plan,
|
||||
static LogicalResult buildAvailabilityFamilies(
|
||||
DeferredTransferPlan &plan,
|
||||
DeferredExchangePlan& exchange,
|
||||
uint64_t& nextChannel,
|
||||
DenseMap<int64_t, DenseMap<int64_t, unsigned>>& eventRegistersByTarget) {
|
||||
uint64_t& nextChannel) {
|
||||
enum class Availability { Local, Direct, Host };
|
||||
for (RequirementFamily& requirement : exchange.requirements) {
|
||||
for (LaneInterval interval : requirement.targetLanes.intervals()) {
|
||||
@@ -357,18 +356,6 @@ static LogicalResult buildAvailabilityFamilies(
|
||||
family.channelIds = StaticIntSequence::affine(nextChannel, 1, count);
|
||||
family.hostRouted = runAvailability == Availability::Host;
|
||||
if (family.hostRouted) {
|
||||
SmallVector<int64_t> eventRegisters;
|
||||
for (int64_t targetCore : targetCores) {
|
||||
auto ®isters = eventRegistersByTarget[targetCore];
|
||||
auto it = registers.try_emplace(
|
||||
requirement.producer->core, registers.size()).first;
|
||||
if (it->second >= kPimEventRegisterCount)
|
||||
return exchange.deferred.emitOpError(
|
||||
"pipeline host transfer requires more event registers than the target core provides");
|
||||
eventRegisters.push_back(it->second);
|
||||
}
|
||||
family.eventRegisters = StaticIntSequence::fromValues(
|
||||
eventRegisters);
|
||||
auto fragmentType = dyn_cast<ShapedType>(
|
||||
requirement.publicationFragmentType);
|
||||
auto fragmentBytes = fragmentType
|
||||
@@ -431,7 +418,6 @@ static LogicalResult buildExchanges(func::FuncOp funcOp, DeferredTransferPlan& p
|
||||
funcOp.walk([&](SpatDeferredCommunicationOp op) { deferredOps.push_back(op); });
|
||||
GraphBatchPublicationCache publicationCache;
|
||||
uint64_t nextChannel = 0;
|
||||
DenseMap<int64_t, DenseMap<int64_t, unsigned>> eventRegistersByTarget;
|
||||
for (SpatDeferredCommunicationOp deferred : deferredOps) {
|
||||
Operation* targetOp = deferred->getParentOfType<SpatScheduledCompute>();
|
||||
if (!targetOp)
|
||||
@@ -451,8 +437,7 @@ static LogicalResult buildExchanges(func::FuncOp funcOp, DeferredTransferPlan& p
|
||||
exchange->program = std::move(*program);
|
||||
if (failed(buildRequirementFamilies(plan, *exchange, publicationCache)))
|
||||
return failure();
|
||||
if (failed(buildAvailabilityFamilies(
|
||||
plan, *exchange, nextChannel, eventRegistersByTarget)))
|
||||
if (failed(buildAvailabilityFamilies(plan, *exchange, nextChannel)))
|
||||
return failure();
|
||||
plan.exchanges.push_back(std::move(exchange));
|
||||
}
|
||||
|
||||
+4
@@ -13,6 +13,10 @@ struct DeferredTransferPlan {
|
||||
llvm::DenseMap<int64_t, llvm::SmallVector<ProducedValue*>> producedByGraph;
|
||||
llvm::SmallVector<std::unique_ptr<DeferredExchangePlan>> exchanges;
|
||||
llvm::SmallVector<unsigned> stepCounts;
|
||||
llvm::DenseMap<int64_t, unsigned> hostAcknowledgementCounts;
|
||||
llvm::SmallVector<int64_t> stageZeroCores;
|
||||
llvm::SmallVector<int64_t> downstreamCores;
|
||||
size_t synchronizationRegisterCount = 0;
|
||||
size_t pipelineHostBufferBytes = 0;
|
||||
};
|
||||
|
||||
|
||||
@@ -89,6 +89,8 @@ struct ScheduleAndRealizeSpatialPass final
|
||||
return;
|
||||
}
|
||||
if (pipelineStages == 0 || target.processorCount % pipelineStages != 0
|
||||
|| (pipelineStages > 1
|
||||
&& target.synchronizationRegisterCount == 0)
|
||||
|| target.residentWeightCapacity
|
||||
> std::numeric_limits<size_t>::max() / pipelineStages) {
|
||||
moduleOp.emitError("ScheduleAndRealizeSpatial requires valid pipeline stages and resource counts");
|
||||
|
||||
+1
@@ -23,6 +23,7 @@ struct SchedulingTarget {
|
||||
Cost transferWidthBytes = 8;
|
||||
Cost vectorWidth = 16;
|
||||
Cost vectorLatencyCycles = 4;
|
||||
size_t synchronizationRegisterCount = 0;
|
||||
|
||||
Cost matrixRows = 128;
|
||||
Cost matrixColumns = 128;
|
||||
|
||||
@@ -592,13 +592,15 @@ def SpatHostStoreSyncOp : SpatOp<"host_store_sync", []> {
|
||||
}
|
||||
|
||||
def SpatHostWaitLoadOp : SpatOp<"host_wait_load", []> {
|
||||
let summary = "Wait for a producer and load its tensor from host memory";
|
||||
let summary = "Wait for producers, load from host memory, and acknowledge consumption";
|
||||
|
||||
let arguments = (ins
|
||||
Index:$sourceCoreId,
|
||||
Index:$targetCoreId,
|
||||
Index:$hostOffset,
|
||||
Index:$eventRegister
|
||||
Index:$eventRegister,
|
||||
Index:$waitValue,
|
||||
Index:$acknowledgementEventRegister
|
||||
);
|
||||
|
||||
let results = (outs
|
||||
@@ -607,7 +609,34 @@ def SpatHostWaitLoadOp : SpatOp<"host_wait_load", []> {
|
||||
|
||||
let assemblyFormat = [{
|
||||
`from` $sourceCoreId `to` $targetCoreId
|
||||
`host_offset` $hostOffset `event` $eventRegister attr-dict `:` type($output)
|
||||
`host_offset` $hostOffset `event` $eventRegister `count` $waitValue
|
||||
`ack` $acknowledgementEventRegister attr-dict `:` type($output)
|
||||
}];
|
||||
}
|
||||
|
||||
def SpatSyncOp : SpatOp<"sync", []> {
|
||||
let summary = "Signal a synchronization register on another processor";
|
||||
|
||||
let arguments = (ins
|
||||
Index:$targetCoreId,
|
||||
Index:$eventRegister
|
||||
);
|
||||
|
||||
let assemblyFormat = [{
|
||||
$targetCoreId `event` $eventRegister attr-dict
|
||||
}];
|
||||
}
|
||||
|
||||
def SpatWaitOp : SpatOp<"wait", []> {
|
||||
let summary = "Wait for a synchronization register value";
|
||||
|
||||
let arguments = (ins
|
||||
Index:$eventRegister,
|
||||
Index:$waitValue
|
||||
);
|
||||
|
||||
let assemblyFormat = [{
|
||||
$eventRegister `value` $waitValue attr-dict
|
||||
}];
|
||||
}
|
||||
|
||||
|
||||
+31
-9
@@ -8,7 +8,7 @@ model it can:
|
||||
3. compile PIM artifacts with Raptor;
|
||||
4. run the reference implementation and functional PIM simulator;
|
||||
5. compare their outputs;
|
||||
6. run `pimsim-nn` to report latency, power, and energy.
|
||||
6. run `pimsim-nn` to report latency, throughput, power, and energy.
|
||||
|
||||
Run the script from the repository root with the repository Python environment.
|
||||
|
||||
@@ -59,7 +59,7 @@ Validate a network or network slice:
|
||||
|
||||
`--operations-dir` may point to any directory tree containing `.onnx` files.
|
||||
The script discovers them recursively and writes `validation_results.csv` in
|
||||
that directory while retaining the terminal table.
|
||||
that directory while retaining separate latency and throughput terminal tables.
|
||||
|
||||
## Raptor vs PIMCOMP comparison
|
||||
|
||||
@@ -87,7 +87,10 @@ profiles, model provenance, limitations, and remote execution.
|
||||
|
||||
## Validation modes
|
||||
|
||||
The default mode performs the complete workflow.
|
||||
The default mode runs latency and throughput in one validation job. Latency
|
||||
uses one input, while throughput uses `--pipeline=4` with four distinct inputs.
|
||||
Both modes reuse the generated input batch, native runner, and reference
|
||||
outputs, and every throughput output is compared with its own reference.
|
||||
|
||||
Use `--compile-only` to build the reference runner and PIM artifacts without
|
||||
executing either implementation:
|
||||
@@ -138,7 +141,8 @@ count with `-j` or `--jobs`:
|
||||
| `--simulator-dir PATH` | Functional `pim-simulator` crate directory. Defaults to the in-tree simulator. |
|
||||
| `--non-functional-simulator-build-dir PATH` | `pimsim-nn` build directory. Defaults to the in-tree build. |
|
||||
| `--pimcomp-config {arch-a,arch-b,arch-c}` | Non-functional hardware/timing profile. Defaults to `arch-a`. |
|
||||
| `--skip-non-functional-simulation` | Skip `pimsim-nn` latency, power, and energy measurement. |
|
||||
| `--skip-non-functional-simulation` | Skip `pimsim-nn` latency, throughput, power, and energy measurement. |
|
||||
| `--no-fast` | Disable fast throughput convergence for authoritative full-duration `pimsim-nn` experiments. |
|
||||
| `--threshold FLOAT` | Absolute output-comparison tolerance. Defaults to `1e-3`. |
|
||||
| `--relative-threshold FLOAT` | Relative output-comparison tolerance. Defaults to `1e-5`. |
|
||||
| `--seed INT` | Seed for generated inputs. Defaults to `0`. |
|
||||
@@ -171,13 +175,19 @@ The checked-in profiles are under
|
||||
`latency_config.json`, while throughput uses
|
||||
`throughput_config_<time>ms.json` and the mesh beside it.
|
||||
|
||||
Use `--skip-non-functional-simulation` when latency, power, and energy are not required.
|
||||
Throughput measurement defaults to `pimsim-nn --fast` with a 1000 ms
|
||||
convergence deadline. Fast mode compares consecutive two-round windows with a
|
||||
fixed 1% tolerance and falls back to the legacy full-duration result if it
|
||||
does not converge. Use `--no-fast` for authoritative experiments.
|
||||
|
||||
Use `--skip-non-functional-simulation` when latency, throughput, power, and energy are not required.
|
||||
The summary reports non-functional results as measured, failed, unsupported, or
|
||||
skipped.
|
||||
|
||||
Overall PASS/FAIL is determined by compilation and functional output
|
||||
comparison. A non-functional simulation failure remains visible as `ERROR` in
|
||||
the latency, power, and energy columns but does not change a functional PASS.
|
||||
the corresponding latency, throughput, power, or energy columns but does not
|
||||
change a functional PASS.
|
||||
|
||||
`pimsim-nn` does not currently implement the `vsoftmax` instruction. When its
|
||||
explicit unsupported-op diagnostic is encountered, Softmax validations retain
|
||||
@@ -190,6 +200,10 @@ Artifacts are written beside each model:
|
||||
|
||||
| Path | Contents |
|
||||
|---|---|
|
||||
| `inputs.csv` | Generated inputs, one batch entry per line. |
|
||||
| `inputs/`, `outputs/`, `runner/` | Inputs, reference outputs, and the runner shared by latency and throughput validation. |
|
||||
| `raptor/pim/`, `simulation/latency/` | Latency PIM artifacts and functional simulator outputs. |
|
||||
| `raptor/throughput/pim/`, `simulation/throughput/` | Pipeline-4, batch-4 throughput PIM artifacts and functional simulator outputs. |
|
||||
| `common/inputs/` | Shared generated input CSV files. |
|
||||
| `common/outputs/` | Shared ONNX-MLIR reference output CSV files. |
|
||||
| `common/runner/` | Shared reference runner source, build tree, and library. |
|
||||
@@ -235,17 +249,25 @@ cargo run --no-default-features --features tracing --release \
|
||||
--package pim-simulator --bin pim-simulator -- \
|
||||
-f /path/to/workspace/raptor/pim \
|
||||
-o /path/to/workspace/simulation/out.bin \
|
||||
-d <addr0>,<size0>,<addr1>,<size1>,...
|
||||
-d <addr0>,<size0>,<addr1>,<size1>,... \
|
||||
--mode latency \
|
||||
--input /path/to/workspace/simulation/inputs/input_0.bin
|
||||
```
|
||||
|
||||
Throughput mode additionally requires `--batch-size N` and exactly `N`
|
||||
`--input` arguments. Each input binary concatenates the model tensors in graph
|
||||
input order. The comparison validator also writes one native reference and one
|
||||
`simulation/*_iterations/output_*.bin` dump per batch entry, and checks every
|
||||
entry rather than only the final output.
|
||||
|
||||
Tracing writes `TraceCore0`, `TraceCore1`, and so on beside `out.bin`. The
|
||||
validator normally derives the `-d` address and byte ranges from
|
||||
`raptor/pim/config.json` and the model output shapes.
|
||||
|
||||
## Results and exit status
|
||||
|
||||
The final table reports functional pass/fail state and non-functional latency
|
||||
and power. The summary includes pass/fail totals, non-functional simulation
|
||||
The final table reports latency and throughput functional pass/fail state plus
|
||||
non-functional latency, throughput, power, and energy. The summary includes pass/fail totals, non-functional simulation
|
||||
counts, total measured latency, and average PIM pass timings when `--verbose`
|
||||
is enabled.
|
||||
|
||||
|
||||
@@ -250,6 +250,8 @@ recomputes the aggregate `results.csv`; missing shared inputs, outputs, or the
|
||||
reference runner are generated even for an isolated run. Use `--jobs 4` to cap
|
||||
parallel comparisons. The per-stage timeout is unlimited by default; pass a
|
||||
positive `--timeout-seconds` value to impose one.
|
||||
Throughput comparisons default to `pimsim-nn --fast` with a 1000 ms
|
||||
convergence deadline. Add `--no-fast` for authoritative full-duration runs.
|
||||
PIMCOMP receives the original ONNX model, and its frontend applies native
|
||||
BatchNormalization fusion when the graph matches its supported Conv/Gemm pattern.
|
||||
The runner continues after a failed model so all reports are produced.
|
||||
|
||||
@@ -38,12 +38,12 @@ Run the complete suite with deadlock detection:
|
||||
|
||||
Use `--compile-only` for compiler and deadlock checks, then `--run-only` to
|
||||
reuse those artifacts for reference execution, simulation, and comparison.
|
||||
The validator prints the complete operation results table before its summary
|
||||
and writes the same rows to `validation_results.csv`.
|
||||
The validator prints separate latency and throughput operation tables before
|
||||
its summary and writes all of their rows to `validation_results.csv`.
|
||||
|
||||
## Complete inventory
|
||||
|
||||
The suite contains 168 models. Tensor shapes, attributes, and constants are
|
||||
The suite contains 177 models. Tensor shapes, attributes, and constants are
|
||||
defined in `gen_tests.py` and in the checked-in ONNX models.
|
||||
|
||||
### Add (5)
|
||||
@@ -64,12 +64,13 @@ defined in `gen_tests.py` and in the checked-in ONNX models.
|
||||
| `negative_axis` | Concatenates tensors using a negative axis. |
|
||||
| `three_inputs_channel_axis` | Concatenates three runtime NCHW tensors along the channel axis. |
|
||||
|
||||
### Conv (34)
|
||||
### Conv (42)
|
||||
|
||||
| Case | Description |
|
||||
|---|---|
|
||||
| `batch_2` | Batched Conv with SAME_UPPER padding and bias. |
|
||||
| `batch_4_pointwise` | Pointwise Conv with batch size four. |
|
||||
| `input_224_7x7_stride2` | 224x224 RGB Conv with 64 output channels, a 7x7 kernel, stride two, and bias. |
|
||||
| `depthwise_1024_channels` | Depthwise pointwise Conv with 1024 groups. |
|
||||
| `depthwise_grouped` | Depthwise-style grouped Conv with one input channel per group. |
|
||||
| `dilated_3x3` | Conv with a dilated 3x3 kernel. |
|
||||
@@ -88,14 +89,21 @@ defined in `gen_tests.py` and in the checked-in ONNX models.
|
||||
| `non_square_kernel_1x3` | Conv with a non-square 1x3 kernel. |
|
||||
| `non_square_kernel_3x1` | Conv with a non-square 3x1 kernel. |
|
||||
| `non_uniform_stride` | Conv with different height and width strides. |
|
||||
| `output_channel_grouping_minimal` | Minimal 64-to-256 pointwise Conv for output-channel grouping. |
|
||||
| `pointwise_1x1` | Basic pointwise channel-mixing Conv. |
|
||||
| `pointwise_tiled_chain` | Relu and chained pointwise Convs with a tiled intermediate. |
|
||||
| `real_asymmetric_padding` | Conv with asymmetric explicit padding. |
|
||||
| `relu_conv_store` | Conv, Relu, and a second Conv to validate an intermediate stored result. |
|
||||
| `same_lower_3x3` | 3x3 Conv with SAME_LOWER padding. |
|
||||
| `same_padding_3x3` | 3x3 Conv with SAME_UPPER padding. |
|
||||
| `simple` | Hand-authored basic 2x2 Conv. |
|
||||
| `kernel_2x2` | Hand-authored Conv with a 2x2 kernel. |
|
||||
| `stride_2` | 3x3 Conv with stride two. |
|
||||
| `strategy_depthwise_16` | 16-channel depthwise 3x3 Conv for depthwise lowering. |
|
||||
| `strategy_input_k_tiled` | 32-channel 3x3 Conv sized to exercise input-K tiling. |
|
||||
| `strategy_output_channel_tiled` | 8-to-192 3x3 Conv sized to exercise output-channel tiling. |
|
||||
| `strategy_streamed_packed` | 3-to-16 3x3 Conv on 128x128 input for streamed packed lowering. |
|
||||
| `strategy_streamed_patch` | 3-to-16 3x3 Conv on 64x64 input for streamed patch lowering. |
|
||||
| `strategy_tiled_2d` | 32-to-192 3x3 Conv sized to exercise 2D tiling. |
|
||||
| `with_bias_3x3` | Multi-channel 3x3 Conv with bias. |
|
||||
| `with_constant` | Hand-authored SAME_UPPER Conv with constant weight and bias. |
|
||||
| `without_kernel_shape_attr` | Conv whose kernel shape is inferred from its weight tensor. |
|
||||
@@ -135,30 +143,30 @@ defined in `gen_tests.py` and in the checked-in ONNX models.
|
||||
| `dynamic_beta` | Uses runtime operands and bias with non-default beta scaling. |
|
||||
| `dynamic_bias` | Uses runtime matrix operands and runtime bias. |
|
||||
| `dynamic_bias_alpha_beta` | Combines runtime operands and bias with alpha and beta scaling. |
|
||||
| `dynamic_transB` | Transposes a runtime right-hand matrix. |
|
||||
| `dynamic_transpose_b` | Transposes a runtime right-hand matrix. |
|
||||
| `huge_1024` | Uses 1024-wide inner and output dimensions. |
|
||||
| `large` | Exercises larger rectangular matrices. |
|
||||
| `large_k_small_n` | Uses a large reduction dimension and narrow output. |
|
||||
| `non_square` | Uses different reduction and output widths. |
|
||||
| `scalar_bias` | Broadcasts a scalar bias to the full output. |
|
||||
| `simple` | Basic Gemm with square weights. |
|
||||
| `square_weights` | Basic Gemm with square weights. |
|
||||
| `small` | Tiny Gemm for fast focused validation. |
|
||||
| `small_k_large_n` | Uses a modest reduction dimension and wide output. |
|
||||
| `transA` | Transposes the left-hand matrix. |
|
||||
| `transA_transB` | Transposes both matrix operands. |
|
||||
| `transB` | Transposes the right-hand weight matrix. |
|
||||
| `transB_with_bias` | Combines a transposed weight matrix with bias. |
|
||||
| `transpose_a` | Transposes the left-hand matrix. |
|
||||
| `transpose_a_and_b` | Transposes both matrix operands. |
|
||||
| `transpose_b` | Transposes the right-hand weight matrix. |
|
||||
| `transpose_b_with_bias` | Combines a transposed weight matrix with bias. |
|
||||
| `with_bias` | Basic matrix product with vector bias. |
|
||||
|
||||
### Gemv (5)
|
||||
|
||||
| Case | Description |
|
||||
|---|---|
|
||||
| `constant` | Vector-matrix product with all inputs constant. |
|
||||
| `simple` | Basic single-row vector-matrix product. |
|
||||
| `with_heterogeneous_constant` | Adds a non-uniform constant bias pattern. |
|
||||
| `with_homogeneous_constant` | Adds a constant bias matching the output shape. |
|
||||
| `with_scalar_constant` | Adds a scalar broadcast bias. |
|
||||
| `all_constant` | Vector-matrix product with all inputs constant. |
|
||||
| `constant_weight` | Basic single-row vector-matrix product with constant weights. |
|
||||
| `non_uniform_bias` | Adds a non-uniform constant bias pattern. |
|
||||
| `uniform_bias` | Adds a uniform constant bias pattern. |
|
||||
| `scalar_bias` | Adds a scalar broadcast bias. |
|
||||
|
||||
### MatMul (12)
|
||||
|
||||
@@ -177,11 +185,12 @@ defined in `gen_tests.py` and in the checked-in ONNX models.
|
||||
| `vector_matrix` | Vector-matrix multiplication producing a 1D output. |
|
||||
| `yolo_attention` | YOLO11n rank-4 dynamic MatMul-scale-transpose-MatMul attention chain. |
|
||||
|
||||
### Mul (5)
|
||||
### Mul (6)
|
||||
|
||||
| Case | Description |
|
||||
|---|---|
|
||||
| `after_conv` | Conv followed by per-channel scaling. |
|
||||
| `after_conv_scalar_constant` | Conv followed by Mul with a scalar constant. |
|
||||
| `basic` | Elementwise Mul on two inputs with identical shapes. |
|
||||
| `channel_broadcast_1024` | Mul with NCHW per-channel broadcasting over 1024 channels. |
|
||||
| `leading_dimension_broadcast` | Mul with trailing-dimension broadcasting. |
|
||||
@@ -254,7 +263,7 @@ defined in `gen_tests.py` and in the checked-in ONNX models.
|
||||
| `height_only` | Nearest-neighbor resize of only the height dimension. |
|
||||
| `nearest_2x` | Nearest-neighbor upsampling by a factor of two. |
|
||||
| `nearest_downsample` | Nearest-neighbor downsampling. |
|
||||
| `non_uniform` | Nearest-neighbor resize with different spatial scales. |
|
||||
| `non_uniform_scales` | Nearest-neighbor resize with different spatial scales. |
|
||||
| `width_only` | Nearest-neighbor resize of only the width dimension. |
|
||||
| `with_sizes` | Resize using explicit output sizes instead of scales. |
|
||||
|
||||
|
||||
BIN
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
@@ -485,16 +485,16 @@ def conv_without_kernel_shape_attr():
|
||||
# GEMM tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def gemm_simple():
|
||||
def gemm_square_weights():
|
||||
"""Simple GEMM with square weights: [10, 132] @ [132, 132]."""
|
||||
B, K, N = 10, 132, 132
|
||||
W = numpy_helper.from_array(np.random.default_rng(41).uniform(-1, 1, (K, N)).astype(np.float32), name="W")
|
||||
A = helper.make_tensor_value_info("A", TensorProto.FLOAT, [B, K])
|
||||
Y = helper.make_tensor_value_info("Y", TensorProto.FLOAT, [B, N])
|
||||
node = helper.make_node("Gemm", ["A", "W"], ["Y"])
|
||||
graph = helper.make_graph([node], "gemm_simple", [A], [Y], initializer=[W])
|
||||
graph = helper.make_graph([node], "gemm_square_weights", [A], [Y], initializer=[W])
|
||||
model = helper.make_model(graph, opset_imports=[helper.make_opsetid("", 13)])
|
||||
save_model(model, "gemm/simple", "gemm_simple.onnx")
|
||||
save_model(model, "gemm/square_weights", "gemm_square_weights.onnx")
|
||||
|
||||
|
||||
def gemm_non_square():
|
||||
@@ -523,7 +523,7 @@ def gemm_with_bias():
|
||||
save_model(model, "gemm/with_bias", "gemm_with_bias.onnx")
|
||||
|
||||
|
||||
def gemm_transB():
|
||||
def gemm_transpose_b():
|
||||
"""GEMM with transB=1: Y = A @ W^T."""
|
||||
B, K, N = 4, 128, 64
|
||||
rng = np.random.default_rng(44)
|
||||
@@ -532,9 +532,9 @@ def gemm_transB():
|
||||
A = helper.make_tensor_value_info("A", TensorProto.FLOAT, [B, K])
|
||||
Y = helper.make_tensor_value_info("Y", TensorProto.FLOAT, [B, N])
|
||||
node = helper.make_node("Gemm", ["A", "W"], ["Y"], transB=1)
|
||||
graph = helper.make_graph([node], "gemm_transB", [A], [Y], initializer=[W])
|
||||
graph = helper.make_graph([node], "gemm_transpose_b", [A], [Y], initializer=[W])
|
||||
model = helper.make_model(graph, opset_imports=[helper.make_opsetid("", 13)])
|
||||
save_model(model, "gemm/transB", "gemm_transB.onnx")
|
||||
save_model(model, "gemm/transpose_b", "gemm_transpose_b.onnx")
|
||||
|
||||
|
||||
def gemm_alpha_beta():
|
||||
@@ -577,7 +577,7 @@ def gemm_large():
|
||||
save_model(model, "gemm/large", "gemm_large.onnx")
|
||||
|
||||
|
||||
def gemm_transB_with_bias():
|
||||
def gemm_transpose_b_with_bias():
|
||||
"""GEMM with transB and bias: Y = A @ W^T + C."""
|
||||
B, K, N = 4, 128, 64
|
||||
rng = np.random.default_rng(48)
|
||||
@@ -586,9 +586,9 @@ def gemm_transB_with_bias():
|
||||
A = helper.make_tensor_value_info("A", TensorProto.FLOAT, [B, K])
|
||||
Y = helper.make_tensor_value_info("Y", TensorProto.FLOAT, [B, N])
|
||||
node = helper.make_node("Gemm", ["A", "W", "C"], ["Y"], transB=1)
|
||||
graph = helper.make_graph([node], "gemm_transB_with_bias", [A], [Y], initializer=[W, C])
|
||||
graph = helper.make_graph([node], "gemm_transpose_b_with_bias", [A], [Y], initializer=[W, C])
|
||||
model = helper.make_model(graph, opset_imports=[helper.make_opsetid("", 13)])
|
||||
save_model(model, "gemm/transB_with_bias", "gemm_transB_with_bias.onnx")
|
||||
save_model(model, "gemm/transpose_b_with_bias", "gemm_transpose_b_with_bias.onnx")
|
||||
|
||||
|
||||
def gemm_dynamic():
|
||||
@@ -602,15 +602,15 @@ def gemm_dynamic():
|
||||
save_model(model, "gemm/dynamic", "gemm_dynamic.onnx")
|
||||
|
||||
|
||||
def gemm_dynamic_transB():
|
||||
def gemm_dynamic_transpose_b():
|
||||
"""GEMM with runtime matrix operands and transposed runtime B."""
|
||||
A = helper.make_tensor_value_info("A", TensorProto.FLOAT, [2, 8])
|
||||
B = helper.make_tensor_value_info("B", TensorProto.FLOAT, [4, 8])
|
||||
Y = helper.make_tensor_value_info("Y", TensorProto.FLOAT, [2, 4])
|
||||
node = helper.make_node("Gemm", ["A", "B"], ["Y"], transB=1)
|
||||
graph = helper.make_graph([node], "gemm_dynamic_transB", [A, B], [Y])
|
||||
graph = helper.make_graph([node], "gemm_dynamic_transpose_b", [A, B], [Y])
|
||||
model = helper.make_model(graph, opset_imports=[helper.make_opsetid("", 13)])
|
||||
save_model(model, "gemm/dynamic_transB", "gemm_dynamic_transB.onnx")
|
||||
save_model(model, "gemm/dynamic_transpose_b", "gemm_dynamic_transpose_b.onnx")
|
||||
|
||||
|
||||
def gemm_dynamic_bias():
|
||||
@@ -696,26 +696,26 @@ def gemm_small_k_large_n():
|
||||
save_model(model, "gemm/small_k_large_n", "gemm_small_k_large_n.onnx")
|
||||
|
||||
|
||||
def gemm_transA():
|
||||
def gemm_transpose_a():
|
||||
"""GEMM with transA=1: A is stored as [K, M] and used as [M, K]."""
|
||||
A = helper.make_tensor_value_info("A", TensorProto.FLOAT, [8, 4])
|
||||
Y = helper.make_tensor_value_info("Y", TensorProto.FLOAT, [4, 6])
|
||||
W = numpy_helper.from_array(np.random.default_rng(90).uniform(-1, 1, (8, 6)).astype(np.float32), name="W")
|
||||
node = helper.make_node("Gemm", ["A", "W"], ["Y"], transA=1)
|
||||
graph = helper.make_graph([node], "gemm_transA", [A], [Y], initializer=[W])
|
||||
graph = helper.make_graph([node], "gemm_transpose_a", [A], [Y], initializer=[W])
|
||||
model = helper.make_model(graph, opset_imports=[helper.make_opsetid("", 13)])
|
||||
save_model(model, "gemm/transA", "gemm_transA.onnx")
|
||||
save_model(model, "gemm/transpose_a", "gemm_transpose_a.onnx")
|
||||
|
||||
|
||||
def gemm_transA_transB():
|
||||
def gemm_transpose_a_and_b():
|
||||
"""GEMM with transA=1 and transB=1."""
|
||||
A = helper.make_tensor_value_info("A", TensorProto.FLOAT, [8, 4])
|
||||
Y = helper.make_tensor_value_info("Y", TensorProto.FLOAT, [4, 6])
|
||||
W = numpy_helper.from_array(np.random.default_rng(91).uniform(-1, 1, (6, 8)).astype(np.float32), name="W")
|
||||
node = helper.make_node("Gemm", ["A", "W"], ["Y"], transA=1, transB=1)
|
||||
graph = helper.make_graph([node], "gemm_transA_transB", [A], [Y], initializer=[W])
|
||||
graph = helper.make_graph([node], "gemm_transpose_a_and_b", [A], [Y], initializer=[W])
|
||||
model = helper.make_model(graph, opset_imports=[helper.make_opsetid("", 13)])
|
||||
save_model(model, "gemm/transA_transB", "gemm_transA_transB.onnx")
|
||||
save_model(model, "gemm/transpose_a_and_b", "gemm_transpose_a_and_b.onnx")
|
||||
|
||||
|
||||
def gemm_bias_rank2_broadcast():
|
||||
@@ -1415,7 +1415,7 @@ def resize_nearest_2x():
|
||||
save_model(model, "resize/nearest_2x", "resize_nearest_2x.onnx")
|
||||
|
||||
|
||||
def resize_nearest_non_uniform():
|
||||
def resize_nearest_non_uniform_scales():
|
||||
"""Resize an NCHW tensor with non-uniform nearest-neighbor scales."""
|
||||
X = helper.make_tensor_value_info("X", TensorProto.FLOAT, [1, 1, 2, 3])
|
||||
Y = helper.make_tensor_value_info("Y", TensorProto.FLOAT, [1, 1, 6, 6])
|
||||
@@ -1424,9 +1424,9 @@ def resize_nearest_non_uniform():
|
||||
node = helper.make_node(
|
||||
"Resize", ["X", "roi", "scales"], ["Y"],
|
||||
mode="nearest", coordinate_transformation_mode="asymmetric", nearest_mode="floor")
|
||||
graph = helper.make_graph([node], "resize_nearest_non_uniform", [X], [Y], initializer=[roi, scales])
|
||||
graph = helper.make_graph([node], "resize_nearest_non_uniform_scales", [X], [Y], initializer=[roi, scales])
|
||||
model = helper.make_model(graph, opset_imports=[helper.make_opsetid("", 13)])
|
||||
save_model(model, "resize/non_uniform", "resize_non_uniform.onnx")
|
||||
save_model(model, "resize/non_uniform_scales", "resize_non_uniform_scales.onnx")
|
||||
|
||||
|
||||
def resize_with_sizes():
|
||||
@@ -2079,16 +2079,16 @@ def div_leading_dimension_broadcast():
|
||||
|
||||
if __name__ == "__main__":
|
||||
print("Generating GEMM tests:")
|
||||
gemm_simple()
|
||||
gemm_square_weights()
|
||||
gemm_non_square()
|
||||
gemm_with_bias()
|
||||
gemm_transB()
|
||||
gemm_transpose_b()
|
||||
gemm_alpha_beta()
|
||||
gemm_small()
|
||||
gemm_large()
|
||||
gemm_transB_with_bias()
|
||||
gemm_transpose_b_with_bias()
|
||||
gemm_dynamic()
|
||||
gemm_dynamic_transB()
|
||||
gemm_dynamic_transpose_b()
|
||||
gemm_dynamic_bias()
|
||||
gemm_dynamic_alpha()
|
||||
gemm_dynamic_beta()
|
||||
@@ -2096,8 +2096,8 @@ if __name__ == "__main__":
|
||||
gemm_huge_1024()
|
||||
gemm_large_k_small_n()
|
||||
gemm_small_k_large_n()
|
||||
gemm_transA()
|
||||
gemm_transA_transB()
|
||||
gemm_transpose_a()
|
||||
gemm_transpose_a_and_b()
|
||||
gemm_bias_rank2_broadcast()
|
||||
gemm_scalar_bias()
|
||||
|
||||
@@ -2220,7 +2220,7 @@ if __name__ == "__main__":
|
||||
|
||||
print("\nGenerating Resize tests:")
|
||||
resize_nearest_2x()
|
||||
resize_nearest_non_uniform()
|
||||
resize_nearest_non_uniform_scales()
|
||||
resize_with_sizes()
|
||||
resize_nearest_downsample()
|
||||
resize_height_only()
|
||||
|
||||
BIN
Binary file not shown.
@@ -1,169 +1,178 @@
|
||||
Operation,Result,Compile,Host mem,Cores mem,Cores,Xbars,Latency,Power,Energy
|
||||
add/after_gemm,PASS,0.063 s,0.01 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
add/basic,PASS,0.057 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
add/broadcast_row,PASS,0.057 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
add/channel_broadcast_1024,PASS,0.061 s,0.02 MiB,0.01 MiB,1,0,SKIP,SKIP,SKIP
|
||||
add/leading_dimension_broadcast,PASS,0.057 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
concat/channel_axis,PASS,0.069 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
concat/negative_axis,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
concat/three_inputs_channel_axis,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
conv/batch_2,PASS,0.066 s,0.00 MiB,0.00 MiB,2,2,SKIP,SKIP,SKIP
|
||||
conv/batch_4_pointwise,PASS,0.065 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
conv/depthwise_1024_channels,PASS,0.083 s,0.19 MiB,0.38 MiB,129,128,SKIP,SKIP,SKIP
|
||||
conv/depthwise_grouped,PASS,0.071 s,0.01 MiB,0.00 MiB,5,4,SKIP,SKIP,SKIP
|
||||
conv/dilated_3x3,PASS,0.068 s,0.01 MiB,0.01 MiB,10,9,SKIP,SKIP,SKIP
|
||||
conv/dynamic,PASS,0.065 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
conv/explicit_padding,PASS,0.062 s,0.01 MiB,0.02 MiB,17,16,SKIP,SKIP,SKIP
|
||||
conv/grouped_many_groups,PASS,0.442 s,0.05 MiB,0.09 MiB,65,64,SKIP,SKIP,SKIP
|
||||
conv/grouped_two_groups,PASS,0.060 s,0.00 MiB,0.00 MiB,3,2,SKIP,SKIP,SKIP
|
||||
conv/huge_pointwise_1024,PASS,0.156 s,0.01 MiB,0.11 MiB,73,64,SKIP,SKIP,SKIP
|
||||
conv/huge_pointwise_1024_dynamic,PASS,0.079 s,8.04 MiB,12.61 MiB,168,0,SKIP,SKIP,SKIP
|
||||
conv/kernel_3x3,PASS,0.062 s,0.01 MiB,0.01 MiB,10,9,SKIP,SKIP,SKIP
|
||||
conv/kernel_equals_input_spatial,PASS,0.064 s,0.00 MiB,0.00 MiB,2,2,SKIP,SKIP,SKIP
|
||||
conv/large_input_channels_1x1,PASS,0.089 s,0.01 MiB,0.02 MiB,9,8,SKIP,SKIP,SKIP
|
||||
conv/large_output_channels_1x1,PASS,0.098 s,0.01 MiB,0.02 MiB,17,8,SKIP,SKIP,SKIP
|
||||
conv/large_spatial,PASS,0.069 s,0.01 MiB,0.04 MiB,37,36,SKIP,SKIP,SKIP
|
||||
conv/multi_channel,PASS,0.066 s,0.00 MiB,0.00 MiB,4,3,SKIP,SKIP,SKIP
|
||||
conv/non_square_kernel_1x3,PASS,0.061 s,0.00 MiB,0.00 MiB,3,2,SKIP,SKIP,SKIP
|
||||
conv/non_square_kernel_3x1,PASS,0.064 s,0.00 MiB,0.00 MiB,3,2,SKIP,SKIP,SKIP
|
||||
conv/non_uniform_stride,PASS,0.062 s,0.00 MiB,0.00 MiB,4,3,SKIP,SKIP,SKIP
|
||||
conv/pointwise_1x1,PASS,0.059 s,0.00 MiB,0.00 MiB,1,1,SKIP,SKIP,SKIP
|
||||
conv/pointwise_tiled_chain,PASS,0.604 s,0.01 MiB,0.04 MiB,20,80,SKIP,SKIP,SKIP
|
||||
conv/real_asymmetric_padding,PASS,0.060 s,0.01 MiB,0.03 MiB,29,28,SKIP,SKIP,SKIP
|
||||
conv/relu_conv_store,PASS,0.091 s,0.16 MiB,0.67 MiB,168,184,SKIP,SKIP,SKIP
|
||||
conv/same_lower_3x3,PASS,0.078 s,0.01 MiB,0.02 MiB,26,25,SKIP,SKIP,SKIP
|
||||
conv/same_padding_3x3,PASS,0.070 s,0.01 MiB,0.02 MiB,26,25,SKIP,SKIP,SKIP
|
||||
conv/simple,PASS,0.064 s,0.00 MiB,0.00 MiB,1,1,SKIP,SKIP,SKIP
|
||||
conv/stride_2,PASS,0.064 s,0.01 MiB,0.00 MiB,5,4,SKIP,SKIP,SKIP
|
||||
conv/with_bias_3x3,PASS,0.067 s,0.00 MiB,0.01 MiB,4,3,SKIP,SKIP,SKIP
|
||||
conv/with_constant,PASS,0.070 s,0.00 MiB,0.00 MiB,1,1,SKIP,SKIP,SKIP
|
||||
conv/without_kernel_shape_attr,PASS,0.069 s,0.01 MiB,0.01 MiB,10,9,SKIP,SKIP,SKIP
|
||||
conv/yolo11n_depthwise_head,PASS,1.482 s,8.66 MiB,34.24 MiB,168,255,SKIP,SKIP,SKIP
|
||||
conv/yolo11n_heavy,PASS,0.431 s,4.82 MiB,19.10 MiB,161,800,SKIP,SKIP,SKIP
|
||||
conv/yolo11n_stem,PASS,0.783 s,12.86 MiB,37.59 MiB,168,488,SKIP,SKIP,SKIP
|
||||
div/after_gemm,PASS,0.072 s,0.01 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
div/basic,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
div/channel_broadcast_1024,PASS,0.066 s,0.02 MiB,0.01 MiB,1,0,SKIP,SKIP,SKIP
|
||||
div/leading_dimension_broadcast,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
div/runtime_scalar_rhs,PASS,0.056 s,0.02 MiB,0.01 MiB,1,0,SKIP,SKIP,SKIP
|
||||
div/scalar_constant,PASS,0.073 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
gather/3d_input_axis1,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
gather/axis0_matrix_indices,PASS,0.056 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
gather/axis1,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
gather/negative_axis,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
gather/negative_indices,PASS,0.063 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
gemm/alpha_beta,PASS,0.067 s,0.01 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemm/bias_rank2_broadcast,PASS,0.060 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemm/dynamic,PASS,0.064 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
gemm/dynamic_alpha,PASS,0.062 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
gemm/dynamic_beta,PASS,0.060 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
gemm/dynamic_bias,PASS,0.058 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
gemm/dynamic_bias_alpha_beta,PASS,0.067 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
gemm/dynamic_transB,PASS,0.062 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
gemm/huge_1024,PASS,0.147 s,0.01 MiB,0.10 MiB,73,64,SKIP,SKIP,SKIP
|
||||
gemm/large,PASS,0.068 s,0.02 MiB,0.03 MiB,17,16,SKIP,SKIP,SKIP
|
||||
gemm/large_k_small_n,PASS,0.095 s,0.01 MiB,0.01 MiB,9,8,SKIP,SKIP,SKIP
|
||||
gemm/non_square,PASS,0.063 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemm/scalar_bias,PASS,0.060 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemm/simple,PASS,0.072 s,0.03 MiB,0.08 MiB,42,40,SKIP,SKIP,SKIP
|
||||
gemm/small,PASS,0.065 s,0.00 MiB,0.00 MiB,2,2,SKIP,SKIP,SKIP
|
||||
gemm/small_k_large_n,PASS,0.097 s,0.01 MiB,0.02 MiB,17,8,SKIP,SKIP,SKIP
|
||||
gemm/transA,PASS,0.064 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemm/transA_transB,PASS,0.069 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemm/transB,PASS,0.062 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemm/transB_with_bias,PASS,0.055 s,0.01 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemm/with_bias,PASS,0.067 s,0.01 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
gemv/constant,PASS,0.064 s,0.00 MiB,0.00 MiB,0,0,SKIP,SKIP,SKIP
|
||||
gemv/simple,PASS,0.069 s,0.00 MiB,0.01 MiB,6,4,SKIP,SKIP,SKIP
|
||||
gemv/with_heterogeneous_constant,PASS,0.066 s,0.00 MiB,0.01 MiB,6,4,SKIP,SKIP,SKIP
|
||||
gemv/with_homogeneous_constant,PASS,0.070 s,0.00 MiB,0.01 MiB,6,4,SKIP,SKIP,SKIP
|
||||
gemv/with_scalar_constant,PASS,0.070 s,0.00 MiB,0.01 MiB,6,4,SKIP,SKIP,SKIP
|
||||
matmul/basic,PASS,0.062 s,0.00 MiB,0.00 MiB,2,2,SKIP,SKIP,SKIP
|
||||
matmul/batched_3d,PASS,0.066 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
matmul/batched_3d_dynamic,PASS,0.057 s,0.00 MiB,0.00 MiB,4,0,SKIP,SKIP,SKIP
|
||||
matmul/batched_left_constant,PASS,0.070 s,0.00 MiB,0.02 MiB,9,8,SKIP,SKIP,SKIP
|
||||
matmul/batched_lhs_broadcast,PASS,0.063 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
matmul/batched_rhs_broadcast,PASS,0.062 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
matmul/dynamic,PASS,0.060 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
matmul/huge_1024,PASS,0.145 s,0.01 MiB,0.10 MiB,73,64,SKIP,SKIP,SKIP
|
||||
matmul/left_constant,PASS,0.051 s,0.00 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
matmul/matrix_vector,PASS,0.095 s,0.52 MiB,0.78 MiB,168,173,SKIP,SKIP,SKIP
|
||||
matmul/vector_matrix,PASS,0.087 s,0.01 MiB,0.01 MiB,9,8,SKIP,SKIP,SKIP
|
||||
matmul/yolo_attention,PASS,0.385 s,1.02 MiB,43.44 MiB,168,0,SKIP,SKIP,SKIP
|
||||
mul/after_conv,PASS,0.055 s,0.00 MiB,0.00 MiB,4,3,SKIP,SKIP,SKIP
|
||||
mul/basic,PASS,0.055 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
mul/channel_broadcast_1024,PASS,0.058 s,0.02 MiB,0.01 MiB,1,0,SKIP,SKIP,SKIP
|
||||
mul/leading_dimension_broadcast,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
mul/scalar_constant,PASS,0.056 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/avg_basic,PASS,0.063 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/avg_ceil_mode,PASS,0.063 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/avg_explicit_padding,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/avg_include_pad,PASS,0.049 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/avg_large_channels,PASS,0.059 s,0.04 MiB,0.02 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/avg_non_uniform_stride,PASS,0.066 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/avg_real_asymmetric_padding,PASS,0.055 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/max_after_conv,PASS,0.069 s,0.00 MiB,0.00 MiB,5,4,SKIP,SKIP,SKIP
|
||||
pool/max_basic,PASS,0.064 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/max_ceil_mode,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/max_global_style_kernel_equals_input,PASS,0.063 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/max_non_square_kernel,PASS,0.067 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/max_real_asymmetric_padding,PASS,0.063 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/max_same_upper,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
pool/max_stride2_multichannel,PASS,0.076 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/4d_spatial,PASS,0.058 s,0.00 MiB,0.00 MiB,3,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/4d_spatial_keepdims_0,PASS,0.068 s,0.00 MiB,0.00 MiB,4,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/after_conv,PASS,0.067 s,0.00 MiB,0.00 MiB,5,3,SKIP,SKIP,SKIP
|
||||
reduce_mean/all_axes_keepdims_0,PASS,0.057 s,0.00 MiB,0.00 MiB,2,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/all_axes_keepdims_1,PASS,0.053 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/basic,PASS,0.058 s,0.00 MiB,0.00 MiB,4,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/channel_axis_nchw,PASS,0.063 s,0.03 MiB,0.02 MiB,4,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/keepdims_0,PASS,0.064 s,0.00 MiB,0.00 MiB,5,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/large_dimension_1024,PASS,0.066 s,0.01 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/legacy_axes_1_2_keepdims_1,PASS,0.069 s,0.00 MiB,0.00 MiB,2,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/legacy_axis1_keepdims_0,PASS,0.067 s,0.00 MiB,0.00 MiB,9,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/legacy_axis1_keepdims_1,PASS,0.057 s,0.00 MiB,0.00 MiB,8,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/legacy_empty_axes_noop,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/legacy_nchw_spatial,PASS,0.059 s,0.00 MiB,0.00 MiB,3,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/legacy_negative_axis,PASS,0.052 s,0.00 MiB,0.00 MiB,6,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/legacy_reduce_all_keepdims_1,PASS,0.057 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reduce_mean/negative_axis,PASS,0.055 s,0.00 MiB,0.00 MiB,6,0,SKIP,SKIP,SKIP
|
||||
relu/4d,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
relu/after_conv,PASS,0.062 s,0.00 MiB,0.00 MiB,4,3,SKIP,SKIP,SKIP
|
||||
relu/after_gemm,PASS,0.062 s,0.01 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
relu/basic,PASS,0.062 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reshape/4d_to_2d_flatten,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reshape/infer_dim_minus_one,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reshape/same_rank,PASS,0.056 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
reshape/zero_copies_input_dim,PASS,0.077 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
resize/height_only,PASS,0.059 s,0.00 MiB,0.00 MiB,4,0,SKIP,SKIP,SKIP
|
||||
resize/nearest_2x,PASS,0.066 s,0.00 MiB,0.00 MiB,4,0,SKIP,SKIP,SKIP
|
||||
resize/nearest_downsample,PASS,0.062 s,0.00 MiB,0.00 MiB,2,0,SKIP,SKIP,SKIP
|
||||
resize/non_uniform,PASS,0.069 s,0.00 MiB,0.00 MiB,6,0,SKIP,SKIP,SKIP
|
||||
resize/width_only,PASS,0.055 s,0.00 MiB,0.00 MiB,2,0,SKIP,SKIP,SKIP
|
||||
resize/with_sizes,PASS,0.060 s,0.00 MiB,0.00 MiB,3,0,SKIP,SKIP,SKIP
|
||||
sigmoid/4d,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
sigmoid/after_gemm,PASS,0.059 s,0.01 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
sigmoid/basic,PASS,0.053 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
slice/2d_basic,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
slice/after_conv,PASS,0.070 s,0.00 MiB,0.01 MiB,7,6,SKIP,SKIP,SKIP
|
||||
slice/default_axes,PASS,0.066 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
slice/large_channel_1024,PASS,0.064 s,0.01 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
slice/nchw_spatial_crop,PASS,0.055 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
slice/negative_axis,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
slice/negative_indices,PASS,0.053 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
slice/step2,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
softmax/3d_last_axis,PASS,0.050 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
softmax/basic,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
softmax/channel_axis,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
softmax/large_dimension_1024,PASS,0.061 s,0.01 MiB,0.01 MiB,1,0,SKIP,SKIP,SKIP
|
||||
softmax/negative_axis,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
split/basic,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
split/equal_three_way,PASS,0.063 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
split/negative_axis,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
split/uneven_channel_axis_4d,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
sub/after_gemm,PASS,0.064 s,0.01 MiB,0.01 MiB,5,4,SKIP,SKIP,SKIP
|
||||
sub/basic,PASS,0.054 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
sub/broadcast_row,PASS,0.064 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
sub/channel_broadcast_1024,PASS,0.063 s,0.02 MiB,0.01 MiB,1,0,SKIP,SKIP,SKIP
|
||||
sub/constant_lhs_broadcast,PASS,0.054 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
sub/leading_dimension_broadcast,PASS,0.054 s,0.00 MiB,0.00 MiB,1,0,SKIP,SKIP,SKIP
|
||||
Operation,Arch,Result (l),Result (t),Compile (l),Host mem (l),Cores mem (l),Cores (l),Xbars (l),Latency (l),Power (l),Energy (l),Compile (t),Host mem (t),Cores mem (t),Cores (t),Xbars (t),Avg latency (t),Throughput (t),Avg power (t),Avg energy (t)
|
||||
add/after_gemm,arch-a,PASS,PASS,0.058 s,0.01 MiB,0.01 MiB,5,4,0.01 ms,104.70 mW,815012.96 pJ,0.057 s,0.01 MiB,0.01 MiB,6,4,145000.00 samples/s,0.01 ms,31.45 mW,216167.21 pJ/it
|
||||
add/basic,arch-a,PASS,PASS,0.048 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.050 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
add/broadcast_row,arch-a,PASS,PASS,0.048 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.051 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
add/channel_broadcast_1024,arch-a,PASS,PASS,0.049 s,0.02 MiB,0.01 MiB,1,0,0.01 ms,78.12 mW,540030.00 pJ,0.051 s,0.02 MiB,0.01 MiB,1,0,145000.00 samples/s,0.01 ms,2.11 mW,13388.67 pJ/it
|
||||
add/leading_dimension_broadcast,arch-a,PASS,PASS,0.051 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.049 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
concat/channel_axis,arch-a,PASS,PASS,0.048 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.16 mW,35718.00 pJ,0.050 s,0.00 MiB,0.00 MiB,1,0,2200000.00 samples/s,0.00 ms,2.16 mW,934.67 pJ/it
|
||||
concat/negative_axis,arch-a,PASS,PASS,0.050 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.09 mW,81450.00 pJ,0.051 s,0.00 MiB,0.00 MiB,1,0,961000.00 samples/s,0.00 ms,2.09 mW,2108.00 pJ/it
|
||||
concat/three_inputs_channel_axis,arch-a,PASS,PASS,0.048 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.15 mW,50328.00 pJ,0.047 s,0.00 MiB,0.00 MiB,1,0,1560000.00 samples/s,0.00 ms,2.15 mW,1332.67 pJ/it
|
||||
conv/batch_2,arch-a,PASS,PASS,0.059 s,0.00 MiB,0.00 MiB,2,2,0.01 ms,82.62 mW,1131451.48 pJ,0.062 s,0.00 MiB,0.01 MiB,4,2,129000.00 samples/s,0.01 ms,51.25 mW,406238.48 pJ/it
|
||||
conv/batch_4_pointwise,arch-a,PASS,PASS,0.058 s,0.00 MiB,0.01 MiB,5,4,0.00 ms,116.08 mW,456420.96 pJ,0.061 s,0.00 MiB,0.01 MiB,5,4,243000.00 samples/s,0.00 ms,44.13 mW,180813.46 pJ/it
|
||||
conv/depthwise_1024_channels,arch-a,PASS,PASS,0.080 s,0.19 MiB,0.38 MiB,129,128,0.22 ms,178.45 mW,39393966.72 pJ,0.141 s,0.36 MiB,0.48 MiB,87,128,3620.00 samples/s,0.28 ms,131.43 mW,37256350.26 pJ/it
|
||||
conv/depthwise_grouped,arch-a,PASS,PASS,0.056 s,0.01 MiB,0.00 MiB,5,4,0.01 ms,107.78 mW,671878.96 pJ,0.061 s,0.01 MiB,0.00 MiB,7,4,235000.00 samples/s,0.00 ms,53.10 mW,227356.96 pJ/it
|
||||
conv/dilated_3x3,arch-a,PASS,PASS,0.061 s,0.01 MiB,0.01 MiB,10,9,0.01 ms,118.77 mW,1034819.16 pJ,0.071 s,0.01 MiB,0.01 MiB,12,9,119000.00 samples/s,0.01 ms,61.00 mW,511357.16 pJ/it
|
||||
conv/dynamic,arch-a,PASS,PASS,0.056 s,0.00 MiB,0.00 MiB,5,0,0.00 ms,92.28 mW,169336.00 pJ,0.057 s,0.00 MiB,0.00 MiB,6,0,784000.00 samples/s,0.00 ms,18.61 mW,26517.00 pJ/it
|
||||
conv/explicit_padding,arch-a,PASS,PASS,0.060 s,0.01 MiB,0.02 MiB,17,16,0.01 ms,145.34 mW,1454397.84 pJ,0.064 s,0.01 MiB,0.02 MiB,19,16,153000.00 samples/s,0.01 ms,109.61 mW,715669.59 pJ/it
|
||||
conv/grouped_many_groups,arch-a,PASS,PASS,0.498 s,0.05 MiB,0.09 MiB,65,64,0.18 ms,142.21 mW,25867112.36 pJ,0.547 s,0.11 MiB,0.79 MiB,127,64,3750.00 samples/s,0.27 ms,141.11 mW,43353235.67 pJ/it
|
||||
conv/grouped_two_groups,arch-a,PASS,PASS,0.064 s,0.00 MiB,0.00 MiB,3,2,0.01 ms,101.46 mW,543914.48 pJ,0.066 s,0.00 MiB,0.01 MiB,9,2,146000.00 samples/s,0.01 ms,108.34 mW,741101.98 pJ/it
|
||||
conv/huge_pointwise_1024,arch-a,PASS,PASS,0.166 s,0.01 MiB,0.11 MiB,73,64,0.02 ms,249.55 mW,3896647.36 pJ,0.182 s,0.04 MiB,0.11 MiB,74,64,33300.00 samples/s,0.03 ms,133.89 mW,4052259.07 pJ/it
|
||||
conv/huge_pointwise_1024_dynamic,arch-a,PASS,PASS,0.084 s,8.04 MiB,12.61 MiB,168,0,2.63 ms,169.52 mW,445489032.00 pJ,0.263 s,11.49 MiB,10.61 MiB,127,0,213.00 samples/s,4.70 ms,164.24 mW,811591564.70 pJ/it
|
||||
conv/input_224_7x7_stride2,arch-a,PASS,PASS,0.775 s,24.14 MiB,61.87 MiB,168,169,38.41 ms,185.26 mW,7116544212.12 pJ,1.142 s,46.43 MiB,73.41 MiB,126,153,27.30 samples/s,36.66 ms,177.05 mW,6915042527.00 pJ/it
|
||||
conv/kernel_2x2,arch-a,PASS,PASS,0.056 s,0.00 MiB,0.00 MiB,1,1,0.00 ms,83.83 mW,360568.24 pJ,0.055 s,0.00 MiB,0.00 MiB,3,1,334000.00 samples/s,0.00 ms,51.45 mW,171905.91 pJ/it
|
||||
conv/kernel_3x3,arch-a,PASS,PASS,0.060 s,0.01 MiB,0.01 MiB,10,9,0.01 ms,123.80 mW,889640.16 pJ,0.063 s,0.01 MiB,0.01 MiB,12,9,219000.00 samples/s,0.00 ms,83.71 mW,382318.91 pJ/it
|
||||
conv/kernel_equals_input_spatial,arch-a,PASS,PASS,0.054 s,0.00 MiB,0.00 MiB,2,2,0.00 ms,89.61 mW,415689.48 pJ,0.057 s,0.00 MiB,0.00 MiB,4,2,293000.00 samples/s,0.00 ms,59.39 mW,204713.48 pJ/it
|
||||
conv/large_input_channels_1x1,arch-a,PASS,PASS,0.096 s,0.01 MiB,0.02 MiB,9,8,0.01 ms,117.82 mW,901121.92 pJ,0.092 s,0.01 MiB,0.02 MiB,10,8,132000.00 samples/s,0.01 ms,59.24 mW,447909.92 pJ/it
|
||||
conv/large_output_channels_1x1,arch-a,PASS,PASS,0.089 s,0.01 MiB,0.02 MiB,17,8,0.01 ms,128.44 mW,1139415.92 pJ,0.095 s,0.01 MiB,0.02 MiB,18,8,123000.00 samples/s,0.01 ms,43.92 mW,355735.17 pJ/it
|
||||
conv/large_spatial,arch-a,PASS,PASS,0.059 s,0.01 MiB,0.04 MiB,37,36,0.02 ms,172.07 mW,2928344.64 pJ,0.078 s,0.01 MiB,0.04 MiB,39,36,88500.00 samples/s,0.01 ms,169.91 mW,1920027.89 pJ/it
|
||||
conv/multi_channel,arch-a,PASS,PASS,0.057 s,0.00 MiB,0.00 MiB,4,3,0.01 ms,105.68 mW,685040.72 pJ,0.060 s,0.00 MiB,0.00 MiB,4,3,146000.00 samples/s,0.01 ms,30.09 mW,205787.97 pJ/it
|
||||
conv/non_square_kernel_1x3,arch-a,PASS,PASS,0.055 s,0.00 MiB,0.00 MiB,3,2,0.01 ms,99.35 mW,679752.48 pJ,0.059 s,0.00 MiB,0.00 MiB,3,2,141000.00 samples/s,0.01 ms,12.12 mW,85739.48 pJ/it
|
||||
conv/non_square_kernel_3x1,arch-a,PASS,PASS,0.058 s,0.00 MiB,0.00 MiB,3,2,0.01 ms,95.89 mW,1292976.48 pJ,0.061 s,0.00 MiB,0.00 MiB,3,2,72900.00 samples/s,0.01 ms,8.83 mW,121109.48 pJ/it
|
||||
conv/non_uniform_stride,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,4,3,0.01 ms,104.05 mW,790874.72 pJ,0.059 s,0.00 MiB,0.00 MiB,4,3,131000.00 samples/s,0.01 ms,29.05 mW,221084.97 pJ/it
|
||||
conv/output_channel_grouping_minimal,arch-a,PASS,PASS,0.089 s,0.10 MiB,0.34 MiB,131,128,0.26 ms,170.73 mW,44125916.72 pJ,0.181 s,0.18 MiB,0.33 MiB,131,128,3910.00 samples/s,0.26 ms,181.50 mW,48146979.72 pJ/it
|
||||
conv/pointwise_1x1,arch-a,PASS,PASS,0.071 s,0.00 MiB,0.00 MiB,1,1,0.01 ms,80.24 mW,987244.24 pJ,0.084 s,0.00 MiB,0.00 MiB,3,1,131000.00 samples/s,0.01 ms,47.08 mW,380210.74 pJ/it
|
||||
conv/pointwise_tiled_chain,arch-a,PASS,PASS,0.819 s,0.01 MiB,0.04 MiB,20,80,0.04 ms,153.88 mW,6445455.20 pJ,0.777 s,0.05 MiB,0.08 MiB,22,80,12500.00 samples/s,0.08 ms,69.67 mW,5573378.45 pJ/it
|
||||
conv/real_asymmetric_padding,arch-a,PASS,PASS,0.074 s,0.01 MiB,0.03 MiB,29,28,0.01 ms,153.67 mW,2221606.72 pJ,0.087 s,0.00 MiB,0.03 MiB,31,28,104000.00 samples/s,0.01 ms,135.38 mW,1295814.97 pJ/it
|
||||
conv/relu_conv_store,arch-a,PASS,PASS,0.102 s,0.16 MiB,0.67 MiB,168,184,0.56 ms,183.08 mW,103057892.80 pJ,0.291 s,0.32 MiB,0.67 MiB,168,166,1640.00 samples/s,0.61 ms,182.39 mW,113644022.20 pJ/it
|
||||
conv/same_lower_3x3,arch-a,PASS,PASS,0.069 s,0.01 MiB,0.02 MiB,26,25,0.01 ms,166.15 mW,2215009.00 pJ,0.088 s,0.01 MiB,0.03 MiB,28,25,114000.00 samples/s,0.01 ms,134.46 mW,1180460.00 pJ/it
|
||||
conv/same_padding_3x3,arch-a,PASS,PASS,0.062 s,0.01 MiB,0.02 MiB,26,25,0.01 ms,166.15 mW,2215009.00 pJ,0.083 s,0.01 MiB,0.03 MiB,28,25,114000.00 samples/s,0.01 ms,134.46 mW,1180460.00 pJ/it
|
||||
conv/strategy_depthwise_16,arch-a,PASS,PASS,0.093 s,0.06 MiB,0.35 MiB,168,168,0.34 ms,197.94 mW,66331479.08 pJ,0.298 s,0.15 MiB,0.37 MiB,168,168,2890.00 samples/s,0.35 ms,196.87 mW,70672344.81 pJ/it
|
||||
conv/strategy_input_k_tiled,arch-a,PASS,PASS,0.079 s,0.08 MiB,0.27 MiB,109,108,0.35 ms,170.81 mW,60422605.92 pJ,0.120 s,0.16 MiB,0.30 MiB,85,101,3520.00 samples/s,0.28 ms,138.29 mW,40167697.42 pJ/it
|
||||
conv/strategy_output_channel_tiled,arch-a,PASS,PASS,0.079 s,0.03 MiB,0.16 MiB,74,72,0.09 ms,155.74 mW,14244739.28 pJ,0.146 s,0.08 MiB,0.25 MiB,111,72,12000.00 samples/s,0.08 ms,137.73 mW,12695085.91 pJ/it
|
||||
conv/strategy_streamed_packed,arch-a,PASS,PASS,0.168 s,3.34 MiB,7.89 MiB,168,168,9.35 ms,179.86 mW,1682364509.56 pJ,0.453 s,5.38 MiB,7.87 MiB,127,126,119.00 samples/s,8.39 ms,175.52 mW,1616768905.00 pJ/it
|
||||
conv/strategy_streamed_patch,arch-a,PASS,PASS,0.110 s,0.34 MiB,1.32 MiB,168,168,1.90 ms,181.91 mW,346476645.64 pJ,0.416 s,0.84 MiB,1.29 MiB,127,126,525.00 samples/s,1.90 ms,176.18 mW,359355537.30 pJ/it
|
||||
conv/strategy_tiled_2d,arch-a,PASS,PASS,0.170 s,0.11 MiB,0.44 MiB,168,168,0.42 ms,182.13 mW,75690907.84 pJ,0.235 s,0.28 MiB,0.45 MiB,130,168,3010.00 samples/s,0.33 ms,178.45 mW,62153061.01 pJ/it
|
||||
conv/stride_2,arch-a,PASS,PASS,0.060 s,0.01 MiB,0.00 MiB,5,4,0.01 ms,110.78 mW,580154.96 pJ,0.061 s,0.01 MiB,0.00 MiB,7,4,297000.00 samples/s,0.00 ms,48.26 mW,163092.63 pJ/it
|
||||
conv/with_bias_3x3,arch-a,PASS,PASS,0.069 s,0.00 MiB,0.01 MiB,4,3,0.01 ms,104.16 mW,776220.72 pJ,0.066 s,0.00 MiB,0.01 MiB,4,3,128000.00 samples/s,0.01 ms,28.71 mW,224217.97 pJ/it
|
||||
conv/with_constant,arch-a,PASS,PASS,0.058 s,0.00 MiB,0.00 MiB,1,1,0.01 ms,81.74 mW,541270.24 pJ,0.067 s,0.00 MiB,0.00 MiB,4,1,138000.00 samples/s,0.01 ms,90.41 mW,664255.74 pJ/it
|
||||
conv/without_kernel_shape_attr,arch-a,PASS,PASS,0.064 s,0.01 MiB,0.01 MiB,10,9,0.01 ms,123.80 mW,889640.16 pJ,0.066 s,0.01 MiB,0.01 MiB,12,9,219000.00 samples/s,0.00 ms,83.71 mW,382318.91 pJ/it
|
||||
conv/yolo11n_depthwise_head,arch-a,PASS,PASS,2.447 s,8.66 MiB,34.24 MiB,168,255,42.70 ms,200.52 mW,8562449708.00 pJ,3.011 s,22.90 MiB,34.20 MiB,168,216,19.40 samples/s,51.59 ms,195.15 mW,10205156420.00 pJ/it
|
||||
conv/yolo11n_heavy,arch-a,PASS,PASS,0.585 s,4.82 MiB,19.10 MiB,161,800,8.54 ms,350.86 mW,2994764012.00 pJ,1.897 s,10.40 MiB,20.59 MiB,161,800,83.80 samples/s,11.93 ms,299.23 mW,3739084612.00 pJ/it
|
||||
conv/yolo11n_stem,arch-a,PASS,PASS,0.996 s,12.86 MiB,37.59 MiB,168,488,14.24 ms,301.23 mW,4289558753.00 pJ,1.726 s,22.34 MiB,32.79 MiB,168,362,23.80 samples/s,42.04 ms,214.78 mW,9030156087.00 pJ/it
|
||||
div/after_gemm,arch-a,PASS,PASS,0.065 s,0.01 MiB,0.01 MiB,5,4,0.01 ms,104.70 mW,815012.96 pJ,0.071 s,0.01 MiB,0.01 MiB,6,4,145000.00 samples/s,0.01 ms,31.45 mW,216167.21 pJ/it
|
||||
div/basic,arch-a,PASS,PASS,0.057 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.057 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
div/channel_broadcast_1024,arch-a,PASS,PASS,0.060 s,0.02 MiB,0.01 MiB,1,0,0.01 ms,78.12 mW,540030.00 pJ,0.056 s,0.02 MiB,0.01 MiB,1,0,145000.00 samples/s,0.01 ms,2.11 mW,13388.67 pJ/it
|
||||
div/leading_dimension_broadcast,arch-a,PASS,PASS,0.067 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.060 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
div/runtime_scalar_rhs,arch-a,PASS,PASS,0.057 s,0.02 MiB,0.01 MiB,1,0,0.01 ms,78.12 mW,540030.00 pJ,0.055 s,0.02 MiB,0.01 MiB,1,0,145000.00 samples/s,0.01 ms,2.11 mW,13388.67 pJ/it
|
||||
div/scalar_constant,arch-a,PASS,PASS,0.054 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.055 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
gather/3d_input_axis1,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.08 mW,45990.00 pJ,0.056 s,0.00 MiB,0.00 MiB,1,0,1700000.00 samples/s,0.00 ms,2.08 mW,1174.67 pJ/it
|
||||
gather/axis0_matrix_indices,arch-a,PASS,PASS,0.083 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.07 mW,54414.00 pJ,0.072 s,0.00 MiB,0.00 MiB,1,0,1440000.00 samples/s,0.00 ms,2.07 mW,1390.67 pJ/it
|
||||
gather/axis1,arch-a,PASS,PASS,0.064 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.06 mW,62526.00 pJ,0.066 s,0.00 MiB,0.00 MiB,1,0,1250000.00 samples/s,0.00 ms,2.06 mW,1598.67 pJ/it
|
||||
gather/negative_axis,arch-a,PASS,PASS,0.078 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.03 mW,112134.00 pJ,0.064 s,0.00 MiB,0.00 MiB,1,0,697000.00 samples/s,0.00 ms,2.03 mW,2870.67 pJ/it
|
||||
gather/negative_indices,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.13 mW,29376.00 pJ,0.062 s,0.00 MiB,0.00 MiB,1,0,2670000.00 samples/s,0.00 ms,2.12 mW,748.67 pJ/it
|
||||
gemm/alpha_beta,arch-a,PASS,PASS,0.068 s,0.01 MiB,0.01 MiB,5,4,0.01 ms,105.27 mW,784908.96 pJ,0.070 s,0.01 MiB,0.01 MiB,6,4,153000.00 samples/s,0.01 ms,32.18 mW,210663.21 pJ/it
|
||||
gemm/bias_rank2_broadcast,arch-a,PASS,PASS,0.063 s,0.00 MiB,0.01 MiB,5,4,0.01 ms,105.98 mW,749484.96 pJ,0.063 s,0.01 MiB,0.01 MiB,6,4,168000.00 samples/s,0.01 ms,33.68 mW,200469.21 pJ/it
|
||||
gemm/dynamic,arch-a,PASS,PASS,0.065 s,0.00 MiB,0.00 MiB,5,0,0.00 ms,91.48 mW,221475.00 pJ,0.062 s,0.00 MiB,0.00 MiB,5,0,471000.00 samples/s,0.00 ms,20.30 mW,43105.75 pJ/it
|
||||
gemm/dynamic_alpha,arch-a,PASS,PASS,0.064 s,0.00 MiB,0.00 MiB,5,0,0.00 ms,91.42 mW,298198.00 pJ,0.063 s,0.00 MiB,0.00 MiB,5,0,337000.00 samples/s,0.00 ms,20.28 mW,60117.75 pJ/it
|
||||
gemm/dynamic_beta,arch-a,PASS,PASS,0.070 s,0.00 MiB,0.00 MiB,5,0,0.00 ms,91.32 mW,398595.00 pJ,0.059 s,0.00 MiB,0.00 MiB,5,0,246000.00 samples/s,0.00 ms,20.21 mW,82201.75 pJ/it
|
||||
gemm/dynamic_bias,arch-a,PASS,PASS,0.066 s,0.00 MiB,0.00 MiB,5,0,0.00 ms,91.45 mW,243703.00 pJ,0.063 s,0.00 MiB,0.00 MiB,5,0,422000.00 samples/s,0.00 ms,20.28 mW,48009.75 pJ/it
|
||||
gemm/dynamic_bias_alpha_beta,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,5,0,0.01 ms,91.28 mW,513811.00 pJ,0.077 s,0.00 MiB,0.00 MiB,5,0,188000.00 samples/s,0.01 ms,20.20 mW,107673.75 pJ/it
|
||||
gemm/dynamic_transpose_b,arch-a,PASS,PASS,0.064 s,0.00 MiB,0.00 MiB,5,0,0.00 ms,91.38 mW,118883.00 pJ,0.065 s,0.00 MiB,0.00 MiB,5,0,781000.00 samples/s,0.00 ms,20.51 mW,26151.50 pJ/it
|
||||
gemm/huge_1024,arch-a,PASS,PASS,0.182 s,0.01 MiB,0.10 MiB,73,64,0.02 ms,215.04 mW,3767885.36 pJ,0.220 s,0.03 MiB,0.10 MiB,73,64,36900.00 samples/s,0.03 ms,148.63 mW,4053069.50 pJ/it
|
||||
gemm/large,arch-a,PASS,PASS,0.074 s,0.02 MiB,0.03 MiB,17,16,0.01 ms,140.15 mW,1573768.84 pJ,0.082 s,0.02 MiB,0.03 MiB,17,16,88800.00 samples/s,0.01 ms,84.59 mW,942235.51 pJ/it
|
||||
gemm/large_k_small_n,arch-a,PASS,PASS,0.142 s,0.01 MiB,0.01 MiB,9,8,0.00 ms,133.48 mW,633769.92 pJ,0.119 s,0.01 MiB,0.01 MiB,9,8,194000.00 samples/s,0.01 ms,76.91 mW,390598.09 pJ/it
|
||||
gemm/non_square,arch-a,PASS,PASS,0.069 s,0.00 MiB,0.01 MiB,5,4,0.00 ms,118.96 mW,419565.96 pJ,0.069 s,0.00 MiB,0.01 MiB,5,4,270000.00 samples/s,0.00 ms,46.78 mW,172713.46 pJ/it
|
||||
gemm/scalar_bias,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.01 MiB,5,4,0.01 ms,105.98 mW,749484.96 pJ,0.066 s,0.01 MiB,0.01 MiB,6,4,168000.00 samples/s,0.01 ms,33.68 mW,200469.21 pJ/it
|
||||
gemm/small,arch-a,PASS,PASS,0.058 s,0.00 MiB,0.00 MiB,2,2,0.00 ms,90.14 mW,398436.48 pJ,0.059 s,0.00 MiB,0.00 MiB,4,2,327000.00 samples/s,0.00 ms,61.13 mW,188023.48 pJ/it
|
||||
gemm/small_k_large_n,arch-a,PASS,PASS,0.112 s,0.01 MiB,0.02 MiB,17,8,0.01 ms,131.01 mW,1043061.92 pJ,0.100 s,0.01 MiB,0.02 MiB,18,8,141000.00 samples/s,0.01 ms,47.48 mW,336507.17 pJ/it
|
||||
gemm/square_weights,arch-a,PASS,PASS,0.080 s,0.03 MiB,0.08 MiB,42,40,0.02 ms,151.77 mW,3284393.60 pJ,0.100 s,0.03 MiB,0.09 MiB,44,40,51800.00 samples/s,0.02 ms,115.71 mW,2278356.60 pJ/it
|
||||
gemm/transpose_a,arch-a,PASS,PASS,0.063 s,0.00 MiB,0.01 MiB,5,4,0.01 ms,109.14 mW,628868.96 pJ,0.062 s,0.00 MiB,0.01 MiB,6,4,212000.00 samples/s,0.00 ms,38.03 mW,179501.21 pJ/it
|
||||
gemm/transpose_a_and_b,arch-a,PASS,PASS,0.070 s,0.00 MiB,0.01 MiB,5,4,0.01 ms,109.14 mW,628868.96 pJ,0.080 s,0.00 MiB,0.01 MiB,6,4,212000.00 samples/s,0.00 ms,38.03 mW,179501.21 pJ/it
|
||||
gemm/transpose_b,arch-a,PASS,PASS,0.065 s,0.00 MiB,0.01 MiB,5,4,0.00 ms,118.96 mW,419565.96 pJ,0.069 s,0.00 MiB,0.01 MiB,5,4,270000.00 samples/s,0.00 ms,46.78 mW,172713.46 pJ/it
|
||||
gemm/transpose_b_with_bias,arch-a,PASS,PASS,0.064 s,0.01 MiB,0.01 MiB,5,4,0.01 ms,110.55 mW,557818.96 pJ,0.071 s,0.01 MiB,0.01 MiB,5,4,191000.00 samples/s,0.01 ms,38.98 mW,203117.46 pJ/it
|
||||
gemm/with_bias,arch-a,PASS,PASS,0.064 s,0.01 MiB,0.01 MiB,5,4,0.01 ms,108.77 mW,604966.96 pJ,0.062 s,0.01 MiB,0.01 MiB,5,4,175000.00 samples/s,0.01 ms,37.33 mW,213443.71 pJ/it
|
||||
gemv/all_constant,arch-a,PASS,PASS,0.071 s,0.00 MiB,0.00 MiB,0,0,0.00 ms,2.00 mW,0.00 pJ,0.061 s,0.00 MiB,0.00 MiB,0,0,0.00 samples/s,0.00 ms,2.00 mW,0.00 pJ/it
|
||||
gemv/constant_weight,arch-a,PASS,PASS,0.100 s,0.00 MiB,0.01 MiB,6,4,0.01 ms,111.15 mW,573535.96 pJ,0.079 s,0.00 MiB,0.01 MiB,8,4,235000.00 samples/s,0.00 ms,68.14 mW,293181.96 pJ/it
|
||||
gemv/non_uniform_bias,arch-a,PASS,PASS,0.080 s,0.00 MiB,0.01 MiB,6,4,0.01 ms,109.82 mW,609371.96 pJ,0.081 s,0.00 MiB,0.01 MiB,8,4,215000.00 samples/s,0.00 ms,66.23 mW,310779.96 pJ/it
|
||||
gemv/scalar_bias,arch-a,PASS,PASS,0.092 s,0.00 MiB,0.01 MiB,6,4,0.01 ms,109.82 mW,609371.96 pJ,0.095 s,0.00 MiB,0.01 MiB,8,4,215000.00 samples/s,0.00 ms,66.23 mW,310779.96 pJ/it
|
||||
gemv/uniform_bias,arch-a,PASS,PASS,0.090 s,0.00 MiB,0.01 MiB,6,4,0.01 ms,109.82 mW,609371.96 pJ,0.149 s,0.00 MiB,0.01 MiB,8,4,215000.00 samples/s,0.00 ms,66.23 mW,310779.96 pJ/it
|
||||
matmul/basic,arch-a,PASS,PASS,0.089 s,0.00 MiB,0.00 MiB,2,2,0.00 ms,90.14 mW,398436.48 pJ,0.071 s,0.00 MiB,0.00 MiB,4,2,327000.00 samples/s,0.00 ms,61.13 mW,188023.48 pJ/it
|
||||
matmul/batched_3d,arch-a,PASS,PASS,0.099 s,0.00 MiB,0.01 MiB,5,4,0.01 ms,108.59 mW,646972.96 pJ,0.086 s,0.00 MiB,0.01 MiB,6,4,207000.00 samples/s,0.00 ms,37.52 mW,181507.21 pJ/it
|
||||
matmul/batched_3d_dynamic,arch-a,PASS,PASS,0.065 s,0.00 MiB,0.00 MiB,4,0,0.00 ms,92.19 mW,167975.00 pJ,0.065 s,0.00 MiB,0.00 MiB,5,0,736000.00 samples/s,0.00 ms,17.42 mW,23971.67 pJ/it
|
||||
matmul/batched_left_constant,arch-a,PASS,PASS,0.069 s,0.00 MiB,0.02 MiB,9,8,0.01 ms,114.39 mW,1009105.92 pJ,0.070 s,0.01 MiB,0.02 MiB,11,8,133000.00 samples/s,0.01 ms,58.19 mW,441494.75 pJ/it
|
||||
matmul/batched_lhs_broadcast,arch-a,PASS,PASS,0.065 s,0.00 MiB,0.01 MiB,5,4,0.01 ms,109.39 mW,621440.96 pJ,0.071 s,0.00 MiB,0.01 MiB,6,4,217000.00 samples/s,0.00 ms,38.52 mW,177665.21 pJ/it
|
||||
matmul/batched_rhs_broadcast,arch-a,PASS,PASS,0.065 s,0.00 MiB,0.01 MiB,5,4,0.01 ms,108.59 mW,646972.96 pJ,0.065 s,0.00 MiB,0.01 MiB,6,4,207000.00 samples/s,0.00 ms,37.52 mW,181507.21 pJ/it
|
||||
matmul/dynamic,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,5,0,0.00 ms,91.42 mW,148195.00 pJ,0.080 s,0.00 MiB,0.00 MiB,5,0,628000.00 samples/s,0.00 ms,20.41 mW,32505.75 pJ/it
|
||||
matmul/huge_1024,arch-a,PASS,PASS,0.188 s,0.01 MiB,0.10 MiB,73,64,0.02 ms,215.04 mW,3767885.36 pJ,0.224 s,0.03 MiB,0.10 MiB,73,64,36900.00 samples/s,0.03 ms,148.63 mW,4053069.50 pJ/it
|
||||
matmul/left_constant,arch-a,PASS,PASS,0.076 s,0.00 MiB,0.01 MiB,5,4,0.01 ms,108.86 mW,637168.96 pJ,0.068 s,0.00 MiB,0.01 MiB,6,4,208000.00 samples/s,0.00 ms,37.62 mW,180976.21 pJ/it
|
||||
matmul/matrix_vector,arch-a,PASS,PASS,0.120 s,0.52 MiB,0.78 MiB,168,173,0.38 ms,202.13 mW,77751814.88 pJ,0.384 s,0.97 MiB,0.72 MiB,127,173,2250.00 samples/s,0.44 ms,193.72 mW,92630594.79 pJ/it
|
||||
matmul/vector_matrix,arch-a,PASS,PASS,0.099 s,0.01 MiB,0.01 MiB,9,8,0.01 ms,118.68 mW,879301.92 pJ,0.104 s,0.01 MiB,0.01 MiB,9,8,132000.00 samples/s,0.01 ms,45.10 mW,342617.42 pJ/it
|
||||
matmul/yolo_attention,arch-a,PASS,PASS,0.526 s,1.02 MiB,43.44 MiB,168,0,8.15 ms,170.00 mW,1385775865.00 pJ,0.796 s,13.76 MiB,43.56 MiB,136,0,65.40 samples/s,15.29 ms,166.46 mW,2545338467.00 pJ/it
|
||||
mul/after_conv,arch-a,PASS,PASS,0.072 s,0.00 MiB,0.00 MiB,4,3,0.01 ms,107.64 mW,586955.72 pJ,0.072 s,0.00 MiB,0.00 MiB,4,3,183000.00 samples/s,0.01 ms,32.66 mW,178132.97 pJ/it
|
||||
mul/after_conv_scalar_constant,arch-a,PASS,PASS,0.120 s,0.00 MiB,0.00 MiB,4,3,0.01 ms,107.64 mW,586955.72 pJ,0.160 s,0.00 MiB,0.00 MiB,4,3,183000.00 samples/s,0.01 ms,32.66 mW,178132.97 pJ/it
|
||||
mul/basic,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.086 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
mul/channel_broadcast_1024,arch-a,PASS,PASS,0.063 s,0.02 MiB,0.01 MiB,1,0,0.01 ms,78.12 mW,540030.00 pJ,0.061 s,0.02 MiB,0.01 MiB,1,0,145000.00 samples/s,0.01 ms,2.11 mW,13388.67 pJ/it
|
||||
mul/leading_dimension_broadcast,arch-a,PASS,PASS,0.064 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.059 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
mul/scalar_constant,arch-a,PASS,PASS,0.065 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.066 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
pool/avg_basic,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.00 MiB,1,0,0.01 ms,78.02 mW,931506.00 pJ,0.063 s,0.00 MiB,0.00 MiB,1,0,84000.00 samples/s,0.01 ms,2.02 mW,24067.00 pJ/it
|
||||
pool/avg_ceil_mode,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.03 mW,340146.00 pJ,0.057 s,0.00 MiB,0.00 MiB,1,0,230000.00 samples/s,0.00 ms,2.03 mW,8810.67 pJ/it
|
||||
pool/avg_explicit_padding,arch-a,PASS,PASS,0.125 s,0.00 MiB,0.00 MiB,1,0,0.01 ms,78.03 mW,688356.00 pJ,0.077 s,0.00 MiB,0.00 MiB,1,0,114000.00 samples/s,0.01 ms,2.03 mW,17809.00 pJ/it
|
||||
pool/avg_include_pad,arch-a,PASS,PASS,0.064 s,0.00 MiB,0.00 MiB,1,0,0.01 ms,78.02 mW,663612.00 pJ,0.059 s,0.00 MiB,0.00 MiB,1,0,118000.00 samples/s,0.01 ms,2.02 mW,17081.00 pJ/it
|
||||
pool/avg_large_channels,arch-a,PASS,PASS,0.069 s,0.04 MiB,0.02 MiB,1,0,0.24 ms,78.00 mW,18399156.00 pJ,0.067 s,0.04 MiB,0.02 MiB,1,0,4250.00 samples/s,0.24 ms,2.00 mW,471428.00 pJ/it
|
||||
pool/avg_non_uniform_stride,arch-a,PASS,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,0.01 ms,78.02 mW,1132254.00 pJ,0.060 s,0.00 MiB,0.00 MiB,1,0,69100.00 samples/s,0.01 ms,2.02 mW,29191.00 pJ/it
|
||||
pool/avg_real_asymmetric_padding,arch-a,PASS,PASS,0.070 s,0.00 MiB,0.00 MiB,1,0,0.03 ms,78.02 mW,1966692.00 pJ,0.073 s,0.00 MiB,0.00 MiB,1,0,39700.00 samples/s,0.03 ms,2.02 mW,50961.00 pJ/it
|
||||
pool/max_after_conv,arch-a,PASS,PASS,0.069 s,0.00 MiB,0.00 MiB,5,4,0.01 ms,99.12 mW,1210689.96 pJ,0.074 s,0.00 MiB,0.00 MiB,5,4,81600.00 samples/s,0.01 ms,28.10 mW,344619.71 pJ/it
|
||||
pool/max_basic,arch-a,PASS,PASS,0.057 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.06 mW,324744.00 pJ,0.055 s,0.00 MiB,0.00 MiB,1,0,241000.00 samples/s,0.00 ms,2.06 mW,8532.67 pJ/it
|
||||
pool/max_ceil_mode,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.07 mW,151464.00 pJ,0.059 s,0.00 MiB,0.00 MiB,1,0,516000.00 samples/s,0.00 ms,2.07 mW,3972.67 pJ/it
|
||||
pool/max_global_style_kernel_equals_input,arch-a,PASS,PASS,0.066 s,0.00 MiB,0.00 MiB,1,0,0.01 ms,78.01 mW,658626.00 pJ,0.063 s,0.00 MiB,0.00 MiB,1,0,119000.00 samples/s,0.01 ms,2.01 mW,16871.00 pJ/it
|
||||
pool/max_non_square_kernel,arch-a,PASS,PASS,0.070 s,0.00 MiB,0.00 MiB,1,0,0.01 ms,78.02 mW,1063068.00 pJ,0.063 s,0.00 MiB,0.00 MiB,1,0,73600.00 samples/s,0.01 ms,2.02 mW,27417.00 pJ/it
|
||||
pool/max_real_asymmetric_padding,arch-a,PASS,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,0.01 ms,78.03 mW,814992.00 pJ,0.060 s,0.00 MiB,0.00 MiB,1,0,96100.00 samples/s,0.01 ms,2.03 mW,21173.00 pJ/it
|
||||
pool/max_same_upper,arch-a,PASS,PASS,0.066 s,0.00 MiB,0.00 MiB,1,0,0.01 ms,78.04 mW,625068.00 pJ,0.067 s,0.00 MiB,0.00 MiB,1,0,125000.00 samples/s,0.01 ms,2.04 mW,16233.00 pJ/it
|
||||
pool/max_stride2_multichannel,arch-a,PASS,PASS,0.067 s,0.00 MiB,0.00 MiB,1,0,0.02 ms,78.02 mW,1247274.00 pJ,0.074 s,0.00 MiB,0.00 MiB,1,0,62700.00 samples/s,0.02 ms,2.02 mW,32153.00 pJ/it
|
||||
reduce_mean/4d_spatial,arch-a,PASS,PASS,0.068 s,0.00 MiB,0.00 MiB,3,0,0.00 ms,92.45 mW,29676.00 pJ,0.068 s,0.00 MiB,0.00 MiB,3,0,2310000.00 samples/s,0.00 ms,4.54 mW,1959.17 pJ/it
|
||||
reduce_mean/4d_spatial_keepdims_0,arch-a,PASS,PASS,0.072 s,0.00 MiB,0.00 MiB,4,0,0.00 ms,94.35 mW,61801.00 pJ,0.070 s,0.00 MiB,0.00 MiB,4,0,1210000.00 samples/s,0.00 ms,19.43 mW,16020.25 pJ/it
|
||||
reduce_mean/after_conv,arch-a,PASS,PASS,0.071 s,0.00 MiB,0.00 MiB,5,3,0.01 ms,106.95 mW,571332.72 pJ,0.075 s,0.00 MiB,0.00 MiB,5,3,183000.00 samples/s,0.01 ms,19.71 mW,107526.72 pJ/it
|
||||
reduce_mean/all_axes_keepdims_0,arch-a,PASS,PASS,0.059 s,0.00 MiB,0.00 MiB,2,0,0.00 ms,79.24 mW,30982.00 pJ,0.063 s,0.00 MiB,0.00 MiB,2,0,2530000.00 samples/s,0.00 ms,3.31 mW,1260.00 pJ/it
|
||||
reduce_mean/all_axes_keepdims_1,arch-a,PASS,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,17286.00 pJ,0.067 s,0.00 MiB,0.00 MiB,1,0,4570000.00 samples/s,0.00 ms,2.22 mW,437.33 pJ/it
|
||||
reduce_mean/basic,arch-a,PASS,PASS,0.068 s,0.00 MiB,0.00 MiB,4,0,0.00 ms,93.51 mW,34881.00 pJ,0.061 s,0.00 MiB,0.00 MiB,4,0,2600000.00 samples/s,0.00 ms,5.85 mW,2235.67 pJ/it
|
||||
reduce_mean/channel_axis_nchw,arch-a,PASS,PASS,0.061 s,0.03 MiB,0.02 MiB,4,0,0.16 ms,93.60 mW,15436518.00 pJ,0.062 s,0.03 MiB,0.08 MiB,4,0,12900.00 samples/s,0.08 ms,5.00 mW,388853.50 pJ/it
|
||||
reduce_mean/keepdims_0,arch-a,PASS,PASS,0.067 s,0.00 MiB,0.00 MiB,5,0,0.00 ms,91.40 mW,68368.00 pJ,0.056 s,0.00 MiB,0.00 MiB,5,0,1300000.00 samples/s,0.00 ms,20.71 mW,16115.50 pJ/it
|
||||
reduce_mean/large_dimension_1024,arch-a,PASS,PASS,0.057 s,0.01 MiB,0.00 MiB,1,0,0.00 ms,78.02 mW,217278.00 pJ,0.056 s,0.01 MiB,0.00 MiB,1,0,359000.00 samples/s,0.00 ms,2.02 mW,5274.00 pJ/it
|
||||
reduce_mean/legacy_axes_1_2_keepdims_1,arch-a,PASS,PASS,0.055 s,0.00 MiB,0.00 MiB,2,0,0.00 ms,79.35 mW,21505.00 pJ,0.058 s,0.00 MiB,0.00 MiB,2,0,3620000.00 samples/s,0.00 ms,3.45 mW,898.00 pJ/it
|
||||
reduce_mean/legacy_axis1_keepdims_0,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.00 MiB,9,0,0.00 ms,92.50 mW,183708.00 pJ,0.065 s,0.00 MiB,0.00 MiB,9,0,679000.00 samples/s,0.00 ms,38.84 mW,57998.17 pJ/it
|
||||
reduce_mean/legacy_axis1_keepdims_1,arch-a,PASS,PASS,0.060 s,0.00 MiB,0.00 MiB,8,0,0.00 ms,94.56 mW,129830.00 pJ,0.066 s,0.00 MiB,0.00 MiB,8,0,1340000.00 samples/s,0.00 ms,10.15 mW,7594.50 pJ/it
|
||||
reduce_mean/legacy_empty_axes_noop,arch-a,PASS,PASS,0.051 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,17286.00 pJ,0.055 s,0.00 MiB,0.00 MiB,1,0,4570000.00 samples/s,0.00 ms,2.22 mW,437.33 pJ/it
|
||||
reduce_mean/legacy_nchw_spatial,arch-a,PASS,PASS,0.058 s,0.00 MiB,0.00 MiB,3,0,0.00 ms,92.45 mW,29676.00 pJ,0.058 s,0.00 MiB,0.00 MiB,3,0,1720000.00 samples/s,0.00 ms,4.40 mW,2552.75 pJ/it
|
||||
reduce_mean/legacy_negative_axis,arch-a,PASS,PASS,0.057 s,0.00 MiB,0.00 MiB,6,0,0.00 ms,93.52 mW,51717.00 pJ,0.060 s,0.00 MiB,0.00 MiB,6,0,1760000.00 samples/s,0.00 ms,8.07 mW,4588.50 pJ/it
|
||||
reduce_mean/legacy_reduce_all_keepdims_1,arch-a,PASS,PASS,0.055 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,17286.00 pJ,0.054 s,0.00 MiB,0.00 MiB,1,0,4570000.00 samples/s,0.00 ms,2.22 mW,437.33 pJ/it
|
||||
reduce_mean/negative_axis,arch-a,PASS,PASS,0.059 s,0.00 MiB,0.00 MiB,6,0,0.00 ms,93.52 mW,51717.00 pJ,0.058 s,0.00 MiB,0.00 MiB,6,0,1760000.00 samples/s,0.00 ms,8.07 mW,4588.50 pJ/it
|
||||
relu/4d,arch-a,PASS,PASS,0.056 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.18 mW,40734.00 pJ,0.053 s,0.00 MiB,0.00 MiB,1,0,1930000.00 samples/s,0.00 ms,2.18 mW,1014.00 pJ/it
|
||||
relu/after_conv,arch-a,PASS,PASS,0.063 s,0.00 MiB,0.00 MiB,4,3,0.01 ms,107.89 mW,577437.72 pJ,0.067 s,0.00 MiB,0.00 MiB,4,3,187000.00 samples/s,0.01 ms,32.91 mW,176189.97 pJ/it
|
||||
relu/after_gemm,arch-a,PASS,PASS,0.069 s,0.01 MiB,0.01 MiB,5,4,0.01 ms,105.16 mW,790056.96 pJ,0.074 s,0.01 MiB,0.01 MiB,6,4,151000.00 samples/s,0.01 ms,32.04 mW,211536.21 pJ/it
|
||||
relu/basic,arch-a,PASS,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,17286.00 pJ,0.057 s,0.00 MiB,0.00 MiB,1,0,4570000.00 samples/s,0.00 ms,2.22 mW,437.33 pJ/it
|
||||
reshape/4d_to_2d_flatten,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.28 mW,20196.00 pJ,0.061 s,0.00 MiB,0.00 MiB,1,0,3910000.00 samples/s,0.00 ms,2.28 mW,488.00 pJ/it
|
||||
reshape/infer_dim_minus_one,arch-a,PASS,PASS,0.055 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,12684.00 pJ,0.058 s,0.00 MiB,0.00 MiB,1,0,6250000.00 samples/s,0.00 ms,2.30 mW,308.00 pJ/it
|
||||
reshape/same_rank,arch-a,PASS,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,12684.00 pJ,0.058 s,0.00 MiB,0.00 MiB,1,0,6250000.00 samples/s,0.00 ms,2.30 mW,308.00 pJ/it
|
||||
reshape/zero_copies_input_dim,arch-a,PASS,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,12684.00 pJ,0.056 s,0.00 MiB,0.00 MiB,1,0,6250000.00 samples/s,0.00 ms,2.30 mW,308.00 pJ/it
|
||||
resize/height_only,arch-a,PASS,PASS,0.060 s,0.00 MiB,0.00 MiB,4,0,0.00 ms,93.55 mW,64833.00 pJ,0.063 s,0.00 MiB,0.00 MiB,4,0,1880000.00 samples/s,0.00 ms,5.60 mW,2986.00 pJ/it
|
||||
resize/nearest_2x,arch-a,PASS,PASS,0.058 s,0.00 MiB,0.00 MiB,4,0,0.00 ms,93.57 mW,109761.00 pJ,0.059 s,0.00 MiB,0.00 MiB,4,0,1450000.00 samples/s,0.00 ms,5.46 mW,3776.00 pJ/it
|
||||
resize/nearest_downsample,arch-a,PASS,PASS,0.059 s,0.00 MiB,0.00 MiB,2,0,0.00 ms,79.45 mW,33925.00 pJ,0.059 s,0.00 MiB,0.00 MiB,2,0,2330000.00 samples/s,0.00 ms,3.28 mW,1360.50 pJ/it
|
||||
resize/non_uniform_scales,arch-a,PASS,PASS,0.063 s,0.00 MiB,0.00 MiB,6,0,0.00 ms,93.58 mW,164037.00 pJ,0.069 s,0.00 MiB,0.00 MiB,6,0,1250000.00 samples/s,0.00 ms,7.76 mW,6207.25 pJ/it
|
||||
resize/width_only,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.00 MiB,2,0,0.00 ms,79.50 mW,53029.00 pJ,0.059 s,0.00 MiB,0.00 MiB,2,0,1700000.00 samples/s,0.00 ms,3.20 mW,1833.50 pJ/it
|
||||
resize/with_sizes,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,3,0,0.00 ms,92.54 mW,73756.00 pJ,0.068 s,0.00 MiB,0.00 MiB,3,0,1700000.00 samples/s,0.00 ms,4.39 mW,2586.75 pJ/it
|
||||
sigmoid/4d,arch-a,PASS,PASS,0.057 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.18 mW,40734.00 pJ,0.060 s,0.00 MiB,0.00 MiB,1,0,1930000.00 samples/s,0.00 ms,2.18 mW,1014.00 pJ/it
|
||||
sigmoid/after_gemm,arch-a,PASS,PASS,0.063 s,0.01 MiB,0.01 MiB,5,4,0.01 ms,105.16 mW,790056.96 pJ,0.062 s,0.01 MiB,0.01 MiB,6,4,151000.00 samples/s,0.01 ms,32.04 mW,211536.21 pJ/it
|
||||
sigmoid/basic,arch-a,PASS,PASS,0.055 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,17286.00 pJ,0.060 s,0.00 MiB,0.00 MiB,1,0,4570000.00 samples/s,0.00 ms,2.22 mW,437.33 pJ/it
|
||||
slice/2d_basic,arch-a,PASS,PASS,0.056 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,18948.00 pJ,0.055 s,0.00 MiB,0.00 MiB,1,0,4170000.00 samples/s,0.00 ms,2.30 mW,491.67 pJ/it
|
||||
slice/after_conv,arch-a,PASS,PASS,0.067 s,0.00 MiB,0.01 MiB,7,6,0.01 ms,118.19 mW,1335082.88 pJ,0.074 s,0.00 MiB,0.01 MiB,7,6,87400.00 samples/s,0.01 ms,47.90 mW,547806.13 pJ/it
|
||||
slice/default_axes,arch-a,PASS,PASS,0.055 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,18948.00 pJ,0.054 s,0.00 MiB,0.00 MiB,1,0,4170000.00 samples/s,0.00 ms,2.30 mW,491.67 pJ/it
|
||||
slice/large_channel_1024,arch-a,PASS,PASS,0.059 s,0.01 MiB,0.00 MiB,1,0,0.00 ms,78.14 mW,221304.00 pJ,0.052 s,0.01 MiB,0.00 MiB,1,0,353000.00 samples/s,0.00 ms,2.14 mW,5058.00 pJ/it
|
||||
slice/nchw_spatial_crop,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.24 mW,101868.00 pJ,0.060 s,0.00 MiB,0.00 MiB,1,0,769000.00 samples/s,0.00 ms,2.24 mW,2851.67 pJ/it
|
||||
slice/negative_axis,arch-a,PASS,PASS,0.054 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,44004.00 pJ,0.055 s,0.00 MiB,0.00 MiB,1,0,1790000.00 samples/s,0.00 ms,2.30 mW,1227.67 pJ/it
|
||||
slice/negative_indices,arch-a,PASS,PASS,0.056 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,25212.00 pJ,0.054 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.30 mW,675.67 pJ/it
|
||||
slice/step2,arch-a,PASS,PASS,0.054 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.29 mW,159876.00 pJ,0.053 s,0.00 MiB,0.00 MiB,1,0,490000.00 samples/s,0.00 ms,2.29 mW,4619.67 pJ/it
|
||||
softmax/3d_last_axis,arch-a,PASS,PASS,0.056 s,0.00 MiB,0.00 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,0.056 s,0.00 MiB,0.00 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED
|
||||
softmax/basic,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.00 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,0.058 s,0.00 MiB,0.00 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED
|
||||
softmax/channel_axis,arch-a,PASS,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,0.067 s,0.00 MiB,0.00 MiB,3,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED
|
||||
softmax/large_dimension_1024,arch-a,PASS,PASS,0.059 s,0.01 MiB,0.01 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,0.054 s,0.01 MiB,0.01 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED
|
||||
softmax/negative_axis,arch-a,PASS,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,0.060 s,0.00 MiB,0.00 MiB,1,0,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED,UNSUPPORTED
|
||||
split/basic,arch-a,PASS,PASS,0.114 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,31554.00 pJ,0.067 s,0.00 MiB,0.00 MiB,1,0,2490000.00 samples/s,0.00 ms,2.30 mW,861.67 pJ/it
|
||||
split/equal_three_way,arch-a,PASS,PASS,0.060 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,44160.00 pJ,0.073 s,0.00 MiB,0.00 MiB,1,0,1780000.00 samples/s,0.00 ms,2.30 mW,1231.67 pJ/it
|
||||
split/negative_axis,arch-a,PASS,PASS,0.061 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.29 mW,84786.00 pJ,0.059 s,0.00 MiB,0.00 MiB,1,0,925000.00 samples/s,0.00 ms,2.29 mW,2413.67 pJ/it
|
||||
split/uneven_channel_axis_4d,arch-a,PASS,PASS,0.058 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.30 mW,18948.00 pJ,0.060 s,0.00 MiB,0.00 MiB,1,0,4170000.00 samples/s,0.00 ms,2.30 mW,491.67 pJ/it
|
||||
sub/after_gemm,arch-a,PASS,PASS,0.065 s,0.01 MiB,0.01 MiB,5,4,0.01 ms,104.70 mW,815012.96 pJ,0.065 s,0.01 MiB,0.01 MiB,6,4,145000.00 samples/s,0.01 ms,31.45 mW,216167.21 pJ/it
|
||||
sub/basic,arch-a,PASS,PASS,0.062 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.055 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
sub/broadcast_row,arch-a,PASS,PASS,0.059 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.077 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
sub/channel_broadcast_1024,arch-a,PASS,PASS,0.064 s,0.02 MiB,0.01 MiB,1,0,0.01 ms,78.12 mW,540030.00 pJ,0.058 s,0.02 MiB,0.01 MiB,1,0,145000.00 samples/s,0.01 ms,2.11 mW,13388.67 pJ/it
|
||||
sub/constant_lhs_broadcast,arch-a,PASS,PASS,0.057 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25188.00 pJ,0.061 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,656.67 pJ/it
|
||||
sub/leading_dimension_broadcast,arch-a,PASS,PASS,0.056 s,0.00 MiB,0.00 MiB,1,0,0.00 ms,78.22 mW,25266.00 pJ,0.057 s,0.00 MiB,0.00 MiB,1,0,3120000.00 samples/s,0.00 ms,2.23 mW,658.67 pJ/it
|
||||
|
||||
|
@@ -156,6 +156,42 @@ def gen_random_inputs(
|
||||
return arrays_in_order, arrays_by_name
|
||||
|
||||
|
||||
def generate_input_batch(onnx_inputs, first_inputs, batch_size, seed):
|
||||
if batch_size < 1:
|
||||
raise ValueError("batch size must be at least 1")
|
||||
if not onnx_inputs:
|
||||
return [first_inputs] * batch_size
|
||||
|
||||
batch = [first_inputs]
|
||||
for index in range(1, batch_size):
|
||||
sample, _ = gen_random_inputs(onnx_inputs, seed=seed + index)
|
||||
if all(np.array_equal(left, right) for left, right in zip(sample, batch[-1])):
|
||||
sample[0] = sample[0].copy()
|
||||
if sample[0].size == 0:
|
||||
raise ValueError("throughput validation cannot distinguish empty input tensors")
|
||||
if np.issubdtype(sample[0].dtype, np.bool_):
|
||||
sample[0].flat[0] = not sample[0].flat[0]
|
||||
elif np.issubdtype(sample[0].dtype, np.integer):
|
||||
info = np.iinfo(sample[0].dtype)
|
||||
value = sample[0].flat[0]
|
||||
sample[0].flat[0] = value + 1 if value < info.max else value - 1
|
||||
else:
|
||||
sample[0].flat[0] += 1
|
||||
batch.append(sample)
|
||||
return batch
|
||||
|
||||
|
||||
def write_input_batch_csv(path, input_batch):
|
||||
path = pathlib.Path(path)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", newline="", encoding="utf-8") as output:
|
||||
writer = csv.writer(output)
|
||||
for sample in input_batch:
|
||||
writer.writerow(
|
||||
np.concatenate([array.reshape(-1) for array in sample]) if sample else ()
|
||||
)
|
||||
|
||||
|
||||
def save_inputs_to_files(onnx_path, arrays_in_order, out_dir):
|
||||
"""
|
||||
Save arrays to CSV files. Returns (flags, files) where flags is a list
|
||||
@@ -201,3 +237,22 @@ def write_inputs_to_memory_bin(memory_bin_path, config_json_path, arrays_in_orde
|
||||
native = arr.astype(arr.dtype.newbyteorder("="), copy=False)
|
||||
f.seek(addr)
|
||||
f.write(native.tobytes(order="C"))
|
||||
|
||||
|
||||
def write_inputs_binary(path, arrays_in_order):
|
||||
"""Write one simulator input in graph-input order."""
|
||||
with open(path, "wb") as f:
|
||||
for arr in arrays_in_order:
|
||||
native = arr.astype(arr.dtype.newbyteorder("="), copy=False)
|
||||
f.write(native.tobytes(order="C"))
|
||||
|
||||
|
||||
def write_input_batch_binaries(input_batch, output_dir, transform=None):
|
||||
output_dir = pathlib.Path(output_dir)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
paths = []
|
||||
for index, sample in enumerate(input_batch):
|
||||
path = output_dir / f"input_{index}.bin"
|
||||
write_inputs_binary(path, [transform(sample[0])] if transform is not None else sample)
|
||||
paths.append(path)
|
||||
return paths
|
||||
|
||||
@@ -35,7 +35,7 @@ def parse_pimsim_nn_metrics(output):
|
||||
return metrics
|
||||
|
||||
|
||||
def export_raptor_latency_artifact(pim_dir, output_dir):
|
||||
def export_raptor_pimsim_artifact(pim_dir, output_dir):
|
||||
pim_dir = Path(pim_dir)
|
||||
output_dir = Path(output_dir)
|
||||
if output_dir.exists():
|
||||
|
||||
@@ -10,9 +10,18 @@ from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from colorama import Style, Fore
|
||||
from .gen_network_runner import gen_network_runner
|
||||
from .onnx_utils import gen_random_inputs, save_inputs_to_files, onnx_io, write_inputs_to_memory_bin, _ONNX_TO_NP
|
||||
from .onnx_utils import (
|
||||
_ONNX_TO_NP,
|
||||
gen_random_inputs,
|
||||
generate_input_batch,
|
||||
onnx_io,
|
||||
save_inputs_to_files,
|
||||
write_input_batch_binaries,
|
||||
write_input_batch_csv,
|
||||
write_inputs_to_memory_bin,
|
||||
)
|
||||
from .raptor import compile_with_raptor
|
||||
from .pimsim_nn import export_raptor_latency_artifact, parse_pimsim_nn_metrics, read_raptor_instruction_count
|
||||
from .pimsim_nn import export_raptor_pimsim_artifact, parse_pimsim_nn_metrics, read_raptor_instruction_count
|
||||
from .subprocess_utils import run_command_with_reporter
|
||||
|
||||
STAGE_TITLES = (
|
||||
@@ -36,7 +45,10 @@ STAGE_COLORS = {
|
||||
STAGE_TITLES[7]: Fore.BLUE,
|
||||
}
|
||||
STAGE_COUNT = len(STAGE_TITLES)
|
||||
GENERATED_DIR_NAMES = ("inputs", "outputs", "pimcomp", "raptor", "runner", "simulation")
|
||||
GENERATED_DIR_NAMES = (
|
||||
"inputs", "outputs", "pimcomp", "raptor", "runner", "simulation",
|
||||
"throughput_validation",
|
||||
)
|
||||
|
||||
MODE_FULL = "full"
|
||||
MODE_COMPILE_ONLY = "compile_only"
|
||||
@@ -77,11 +89,19 @@ def sanitize_output_name(name):
|
||||
@dataclass
|
||||
class ValidationResult:
|
||||
passed: bool
|
||||
latency_passed: bool | None = None
|
||||
throughput_passed: bool | None = None
|
||||
pim_pass_timings: dict[str, float] = field(default_factory=dict)
|
||||
pimsim_latency_ms: float | None = None
|
||||
pimsim_throughput_samples_s: float | None = None
|
||||
pimsim_power_mw: float | None = None
|
||||
pimsim_energy_pj: float | None = None
|
||||
pimsim_throughput_average_latency_ms: float | None = None
|
||||
pimsim_throughput_average_power_mw: float | None = None
|
||||
pimsim_throughput_average_energy_pj: float | None = None
|
||||
mode_metrics: dict[str, dict[str, float | int | None]] = field(default_factory=dict)
|
||||
pimsim_status: str = PIMSIM_SKIPPED
|
||||
throughput_pimsim_status: str = PIMSIM_SKIPPED
|
||||
compile_time_s: float | None = None
|
||||
host_memory_bytes: int | None = None
|
||||
cores_memory_bytes: int | None = None
|
||||
@@ -291,11 +311,16 @@ def pimcomp_compatibility_errors(config_path, *, core_count, crossbar_count, cro
|
||||
return errors
|
||||
|
||||
|
||||
def run_pimsim_nn(pimsim_nn_build_dir, pim_dir, config_path, reporter=None, timeout_sec=None):
|
||||
latency_artifact = export_raptor_latency_artifact(pim_dir, Path(pim_dir).parent / "pimsim_nn")
|
||||
def run_pimsim_nn(
|
||||
pimsim_nn_build_dir, pim_dir, config_path, execution_mode,
|
||||
reporter=None, timeout_sec=None, fast=True):
|
||||
pimsim_artifact = export_raptor_pimsim_artifact(pim_dir, Path(pim_dir).parent / "pimsim_nn")
|
||||
command = [pimsim_nn_build_dir / "ChipTest", pimsim_artifact, config_path, "--gui=false"]
|
||||
if fast:
|
||||
command.append("--fast")
|
||||
try:
|
||||
output = run_command(
|
||||
[pimsim_nn_build_dir / "ChipTest", latency_artifact, config_path, "--gui=false"],
|
||||
command,
|
||||
cwd=pimsim_nn_build_dir,
|
||||
reporter=reporter,
|
||||
timeout_sec=timeout_sec,
|
||||
@@ -307,10 +332,14 @@ def run_pimsim_nn(pimsim_nn_build_dir, pim_dir, config_path, reporter=None, time
|
||||
raise PimSimUnsupportedError(PIMSIM_UNSUPPORTED_VSOFTMAX) from exc
|
||||
raise
|
||||
metrics = parse_pimsim_nn_metrics(output)
|
||||
required = ("latency_ms", "average_power_mw", "average_energy_pj")
|
||||
required = (
|
||||
("latency_ms", "average_power_mw", "average_energy_pj")
|
||||
if execution_mode == "latency"
|
||||
else ("throughput", "average_latency_ms", "average_power_mw", "average_energy_pj")
|
||||
)
|
||||
if any(name not in metrics for name in required):
|
||||
raise RuntimeError("pimsim-nn output did not contain latency, average power, and average energy")
|
||||
return tuple(metrics[name] for name in required)
|
||||
raise RuntimeError(f"pimsim-nn output did not contain required {execution_mode} metrics")
|
||||
return metrics
|
||||
|
||||
|
||||
def clean_workspace_artifacts(workspace_dir, model_stem):
|
||||
@@ -327,6 +356,7 @@ def clean_workspace_artifacts(workspace_dir, model_stem):
|
||||
|
||||
for name in GENERATED_DIR_NAMES:
|
||||
remove_path(workspace_dir / name)
|
||||
remove_path(workspace_dir / "inputs.csv")
|
||||
|
||||
for suffix in (".onnx.mlir", ".so", ".tmp"):
|
||||
remove_path(workspace_dir / f"{model_stem}{suffix}")
|
||||
@@ -334,8 +364,10 @@ def clean_workspace_artifacts(workspace_dir, model_stem):
|
||||
return removed_paths
|
||||
|
||||
|
||||
def print_stage(reporter, model_index, model_total, model_name, title):
|
||||
color = STAGE_COLORS.get(title, Fore.WHITE)
|
||||
def print_stage(reporter, model_index, model_total, model_name, title, mode=None):
|
||||
if mode is not None:
|
||||
title = f"{title} ({mode.capitalize()})"
|
||||
color = STAGE_COLORS.get(title, STAGE_COLORS.get(title.split(" (", 1)[0], Fore.WHITE))
|
||||
reporter.log(Style.BRIGHT + color + f"[{title}]" + Style.RESET_ALL)
|
||||
reporter.set_stage(model_index, model_total, model_name, title)
|
||||
|
||||
@@ -374,11 +406,38 @@ def build_dump_ranges(config_path, outputs_descriptor):
|
||||
return ",".join(ranges)
|
||||
|
||||
|
||||
def run_pim_simulator(simulator_dir, pim_dir, output_bin_path, dump_ranges, reporter=None, timeout_sec=None):
|
||||
def build_pim_simulator_command(
|
||||
pim_dir, output_bin_path, dump_ranges, input_paths, mode="latency",
|
||||
batch_output_dir=None):
|
||||
if mode not in ("latency", "throughput"):
|
||||
raise ValueError(f"unknown simulator mode: {mode}")
|
||||
if not input_paths:
|
||||
raise ValueError("simulator requires at least one input")
|
||||
command = [
|
||||
"cargo", "run", "--no-default-features", "--release", "--package", "pim-simulator", "--bin", "pim-simulator",
|
||||
"--", "-f", str(pim_dir), "-o", str(output_bin_path), "-d", dump_ranges,
|
||||
"--mode", mode, "--batch-size", str(len(input_paths)),
|
||||
]
|
||||
if batch_output_dir is not None:
|
||||
command += ["--batch-output-dir", str(batch_output_dir)]
|
||||
for path in input_paths:
|
||||
command += ["--input", str(path)]
|
||||
return command
|
||||
|
||||
|
||||
def run_pim_simulator(
|
||||
simulator_dir, pim_dir, output_bin_path, dump_ranges, reporter=None,
|
||||
timeout_sec=None, input_paths=(), mode="latency", batch_output_dir=None):
|
||||
command = build_pim_simulator_command(
|
||||
pim_dir,
|
||||
output_bin_path,
|
||||
dump_ranges,
|
||||
input_paths,
|
||||
mode=mode,
|
||||
batch_output_dir=batch_output_dir,
|
||||
)
|
||||
run_command(
|
||||
["cargo", "run", "--no-default-features", "--release", "--package", "pim-simulator", "--bin", "pim-simulator",
|
||||
"--",
|
||||
"-f", str(pim_dir), "-o", str(output_bin_path), "-d", dump_ranges],
|
||||
command,
|
||||
cwd=simulator_dir,
|
||||
reporter=reporter,
|
||||
timeout_sec=timeout_sec,
|
||||
@@ -431,21 +490,121 @@ def validate_outputs(sim_arrays, runner_out_dir, outputs_descriptor, threshold,
|
||||
return all_passed
|
||||
|
||||
|
||||
def report_validation_failure(reporter, execution_name, stage, exc):
|
||||
reporter.suspend()
|
||||
print(
|
||||
Fore.RED + f"{execution_name.capitalize()} {stage} failed: "
|
||||
f"{type(exc).__name__}: {exc}" + Style.RESET_ALL,
|
||||
file=sys.stderr,
|
||||
flush=True,
|
||||
)
|
||||
reporter.resume()
|
||||
|
||||
|
||||
def validate_execution(
|
||||
execution, state, functional_data, workspace_dir, simulator_dir,
|
||||
pimsim_nn_build_dir, threshold, rtol, verbose, command_timeout_seconds,
|
||||
stage_context, pimsim_fast):
|
||||
reporter, model_index, model_total, model_name = stage_context
|
||||
name = execution["name"]
|
||||
pim_dir = execution["root"] / "pim"
|
||||
batch_size = execution["batch_size"]
|
||||
|
||||
if state["compiled"] and functional_data is not None:
|
||||
input_batch, input_paths, reference_dirs, outputs_descriptor = functional_data
|
||||
simulation_dir = workspace_dir / "simulation" / name
|
||||
try:
|
||||
print_stage(
|
||||
reporter, model_index, model_total, model_name,
|
||||
"Run Functional Simulation", name,
|
||||
)
|
||||
write_inputs_to_memory_bin(
|
||||
pim_dir / "memory.bin", pim_dir / "config.json", input_batch[0])
|
||||
simulation_dir.mkdir(parents=True, exist_ok=True)
|
||||
dump_ranges = build_dump_ranges(pim_dir / "config.json", outputs_descriptor)
|
||||
output_dir = simulation_dir / "outputs"
|
||||
run_pim_simulator(
|
||||
simulator_dir, pim_dir, simulation_dir / "out.bin", dump_ranges,
|
||||
reporter=reporter, timeout_sec=command_timeout_seconds,
|
||||
input_paths=input_paths[:batch_size], mode=name,
|
||||
batch_output_dir=output_dir)
|
||||
reporter.advance()
|
||||
|
||||
print_stage(
|
||||
reporter, model_index, model_total, model_name,
|
||||
"Compare Outputs", name,
|
||||
)
|
||||
reporter.suspend()
|
||||
try:
|
||||
iteration_results = [
|
||||
validate_outputs(
|
||||
parse_pim_simulator_outputs(
|
||||
output_dir / f"output_{index:06d}.bin", outputs_descriptor),
|
||||
reference_dirs[index], outputs_descriptor,
|
||||
threshold, rtol=rtol, verbose=verbose)
|
||||
for index in range(batch_size)
|
||||
]
|
||||
finally:
|
||||
reporter.resume()
|
||||
state["passed"] = all(iteration_results)
|
||||
reporter.advance()
|
||||
except Exception as exc:
|
||||
report_validation_failure(reporter, name, "functional validation", exc)
|
||||
|
||||
print_stage(
|
||||
reporter, model_index, model_total, model_name,
|
||||
"Run Non-functional Simulation", name,
|
||||
)
|
||||
config_path = execution["pimsim_config"]
|
||||
if state["compiled"] and pimsim_nn_build_dir is not None and config_path is not None:
|
||||
try:
|
||||
state["metrics"] = run_pimsim_nn(
|
||||
pimsim_nn_build_dir, pim_dir, config_path, name,
|
||||
reporter=reporter, timeout_sec=command_timeout_seconds,
|
||||
fast=pimsim_fast)
|
||||
state["pimsim_status"] = PIMSIM_DONE
|
||||
metric = (
|
||||
f"Latency: {state['metrics']['latency_ms']:.2f} ms"
|
||||
if name == "latency" else
|
||||
f"Throughput: {state['metrics']['throughput']:.2f} samples/s")
|
||||
energy_unit = "pJ" if name == "latency" else "pJ/it"
|
||||
print_info(
|
||||
reporter, f"{metric}, Power: {state['metrics']['average_power_mw']:.2f} mW, "
|
||||
f"Energy: {state['metrics']['average_energy_pj']:.2f} {energy_unit}")
|
||||
except PimSimUnsupportedError as exc:
|
||||
state["pimsim_status"] = PIMSIM_UNSUPPORTED
|
||||
print_info(reporter, str(exc))
|
||||
except Exception as exc:
|
||||
state["pimsim_status"] = PIMSIM_FAILED
|
||||
report_validation_failure(reporter, name, "non-functional validation", exc)
|
||||
elif not state["compiled"]:
|
||||
state["pimsim_status"] = PIMSIM_NOT_RUN
|
||||
else:
|
||||
print_info(reporter, "pimsim-nn non-functional simulation skipped")
|
||||
reporter.advance()
|
||||
|
||||
|
||||
def validate_network(network_onnx_path, raptor_path, onnx_include_dir,
|
||||
simulator_dir, crossbar_size, crossbar_count, core_count,
|
||||
raptor_extra_args,
|
||||
pimsim_nn_build_dir, pimsim_config_path,
|
||||
threshold, rtol,
|
||||
seed, reporter, model_index, model_total, verbose,
|
||||
command_timeout_seconds, mode):
|
||||
command_timeout_seconds, mode, throughput_pipeline=None,
|
||||
throughput_batch_size=4, throughput_pimsim_config_path=None,
|
||||
pimsim_fast=True):
|
||||
if throughput_pipeline is not None and throughput_batch_size < 2:
|
||||
raise ValueError("throughput validation requires batch size greater than 1")
|
||||
network_onnx_path = Path(network_onnx_path).resolve()
|
||||
raptor_path = Path(raptor_path).resolve()
|
||||
onnx_include_dir = Path(onnx_include_dir).resolve()
|
||||
simulator_dir = Path(simulator_dir).resolve()
|
||||
pimsim_enabled = pimsim_nn_build_dir is not None and pimsim_config_path is not None
|
||||
if pimsim_enabled:
|
||||
if pimsim_nn_build_dir is not None:
|
||||
pimsim_nn_build_dir = Path(pimsim_nn_build_dir).resolve()
|
||||
if pimsim_config_path is not None:
|
||||
pimsim_config_path = Path(pimsim_config_path).resolve()
|
||||
if throughput_pimsim_config_path is not None:
|
||||
throughput_pimsim_config_path = Path(throughput_pimsim_config_path).resolve()
|
||||
compile_extra_args = list(raptor_extra_args or [])
|
||||
owns_reporter = reporter is None
|
||||
reporter = reporter or ProgressReporter(model_total, stages_per_model=len(MODE_STAGE_TITLES[mode]), verbose=verbose)
|
||||
@@ -454,189 +613,209 @@ def validate_network(network_onnx_path, raptor_path, onnx_include_dir,
|
||||
raptor_dir = workspace_dir / "raptor"
|
||||
runner_dir = workspace_dir / "runner"
|
||||
runner_build_dir = runner_dir / "build"
|
||||
|
||||
if mode != MODE_RUN_ONLY:
|
||||
clean_workspace_artifacts(workspace_dir, network_onnx_path.stem)
|
||||
Path.mkdir(raptor_dir, exist_ok=True)
|
||||
Path.mkdir(raptor_dir, parents=True, exist_ok=True)
|
||||
Path.mkdir(runner_build_dir, parents=True, exist_ok=True)
|
||||
|
||||
reporter.log(Fore.CYAN + f"[{model_index}/{model_total}]" + Style.RESET_ALL +
|
||||
f" {Style.BRIGHT}Validating {network_onnx_path.name}{Style.RESET_ALL}")
|
||||
failed_with_exception = False
|
||||
stem = network_onnx_path.stem
|
||||
network_so_path = runner_dir / f"{stem}.so"
|
||||
network_mlir_path = raptor_dir / f"{stem}.onnx.mlir"
|
||||
runner_path = runner_build_dir / "runner"
|
||||
executions = [{
|
||||
"name": "latency",
|
||||
"root": raptor_dir,
|
||||
"batch_size": 1,
|
||||
"compile_args": compile_extra_args,
|
||||
"pimsim_config": pimsim_config_path,
|
||||
}]
|
||||
if throughput_pipeline is not None:
|
||||
throughput_args = [
|
||||
arg for arg in compile_extra_args if not str(arg).startswith("--pipeline=")
|
||||
] + [f"--pipeline={throughput_pipeline}"]
|
||||
executions.append({
|
||||
"name": "throughput",
|
||||
"root": raptor_dir / "throughput",
|
||||
"batch_size": throughput_batch_size,
|
||||
"compile_args": throughput_args,
|
||||
"pimsim_config": throughput_pimsim_config_path,
|
||||
})
|
||||
states = {
|
||||
execution["name"]: {
|
||||
"compiled": False,
|
||||
"passed": False,
|
||||
"metrics": {},
|
||||
"pimsim_status": PIMSIM_SKIPPED,
|
||||
"compile_time_s": 0.0,
|
||||
"resource_metrics": {},
|
||||
}
|
||||
for execution in executions
|
||||
}
|
||||
pim_pass_timings = {}
|
||||
compile_time_s = None
|
||||
compile_time_s = 0.0
|
||||
resource_metrics = {}
|
||||
|
||||
try:
|
||||
stem = network_onnx_path.stem
|
||||
network_so_path = runner_dir / f"{stem}.so"
|
||||
network_mlir_path = raptor_dir / f"{stem}.onnx.mlir"
|
||||
runner_path = runner_build_dir / "runner"
|
||||
pim_output_base = raptor_dir / stem
|
||||
|
||||
def compile_pim():
|
||||
nonlocal compile_time_s, resource_metrics
|
||||
started = time.perf_counter()
|
||||
timings = compile_with_raptor(
|
||||
network_onnx_path, raptor_path, pim_output_base, crossbar_size,
|
||||
crossbar_count, core_count=core_count,
|
||||
raptor_extra_args=compile_extra_args, cwd=raptor_dir,
|
||||
verbose=verbose, reporter=reporter,
|
||||
timeout_sec=command_timeout_seconds)
|
||||
compile_time_s = time.perf_counter() - started
|
||||
resource_metrics = collect_pim_resource_metrics(raptor_dir / "pim")
|
||||
return timings
|
||||
|
||||
reference_ready = False
|
||||
if mode != MODE_RUN_ONLY:
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Compile ONNX")
|
||||
network_so_path, network_mlir_path = compile_onnx_network(
|
||||
network_onnx_path, raptor_path, raptor_dir, runner_dir, reporter=reporter,
|
||||
timeout_sec=command_timeout_seconds)
|
||||
print_info(reporter, f"MLIR saved to {network_mlir_path}")
|
||||
print_info(reporter, f"Shared library saved to {network_so_path}")
|
||||
reporter.advance()
|
||||
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Build Runner")
|
||||
gen_network_runner(
|
||||
network_onnx_path,
|
||||
network_so_path,
|
||||
onnx_include_dir,
|
||||
entry="run_main_graph",
|
||||
out=runner_dir / "runner.c",
|
||||
verbose=False,
|
||||
)
|
||||
runner_path = build_onnx_runner(runner_dir, runner_build_dir, reporter=reporter,
|
||||
timeout_sec=command_timeout_seconds)
|
||||
print_info(reporter, f"Runner built at {runner_path}")
|
||||
reporter.advance()
|
||||
|
||||
if mode == MODE_COMPILE_ONLY:
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Compile PIM")
|
||||
pim_pass_timings = compile_pim()
|
||||
print_info(reporter, f"PIM artifacts saved to {raptor_dir / 'pim'}")
|
||||
reporter.advance()
|
||||
reporter.record_result(True)
|
||||
reporter.log(Style.BRIGHT + f"Result: {Fore.GREEN}PASS{Style.RESET_ALL}" + Style.RESET_ALL)
|
||||
return ValidationResult(
|
||||
passed=True, pim_pass_timings=pim_pass_timings,
|
||||
compile_time_s=compile_time_s, **resource_metrics)
|
||||
|
||||
if mode == MODE_RUN_ONLY:
|
||||
required_paths = [
|
||||
(network_so_path, "compiled reference shared library"),
|
||||
(network_mlir_path, "exported ONNX MLIR"),
|
||||
(runner_path, "built reference runner"),
|
||||
(raptor_dir / "pim" / "config.json", "compiled PIM artifacts"),
|
||||
]
|
||||
missing = [f"{description} at {path}" for path, description in required_paths if not path.exists()]
|
||||
if missing:
|
||||
raise FileNotFoundError("run-only mode requires existing artifacts:\n " + "\n ".join(missing))
|
||||
resource_metrics = collect_pim_resource_metrics(raptor_dir / "pim")
|
||||
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Generate Inputs")
|
||||
inputs_descriptor, outputs_descriptor = onnx_io(network_onnx_path)
|
||||
inputs_list, _inputs_dict = gen_random_inputs(inputs_descriptor, seed=seed)
|
||||
flags, _files = save_inputs_to_files(network_onnx_path, inputs_list, out_dir=workspace_dir / "inputs")
|
||||
print_info(reporter, f"Saved {len(inputs_list)} input file(s) to {workspace_dir / 'inputs'}")
|
||||
reporter.advance()
|
||||
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Run Reference")
|
||||
out_dir = workspace_dir / "outputs"
|
||||
Path.mkdir(out_dir, exist_ok=True)
|
||||
run_cmd = [runner_path, *flags]
|
||||
run_cmd += ["--save-csv-dir", f"{out_dir}"]
|
||||
run_command(run_cmd, cwd=runner_build_dir, reporter=reporter, timeout_sec=command_timeout_seconds)
|
||||
print_info(reporter, f"Reference outputs saved to {out_dir}")
|
||||
reporter.advance()
|
||||
|
||||
if mode != MODE_RUN_ONLY:
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Compile PIM")
|
||||
pim_pass_timings = compile_pim()
|
||||
print_info(reporter, f"PIM artifacts saved to {raptor_dir / 'pim'}")
|
||||
reporter.advance()
|
||||
|
||||
print_stage(
|
||||
reporter, model_index, model_total, network_onnx_path.name,
|
||||
"Run Functional Simulation")
|
||||
pim_dir = raptor_dir / "pim"
|
||||
write_inputs_to_memory_bin(pim_dir / "memory.bin", pim_dir / "config.json", inputs_list)
|
||||
simulation_dir = workspace_dir / "simulation"
|
||||
Path.mkdir(simulation_dir, exist_ok=True)
|
||||
dump_ranges = build_dump_ranges(pim_dir / "config.json", outputs_descriptor)
|
||||
output_bin_path = simulation_dir / "out.bin"
|
||||
run_pim_simulator(simulator_dir, pim_dir, output_bin_path, dump_ranges, reporter=reporter,
|
||||
timeout_sec=command_timeout_seconds)
|
||||
print_info(reporter, f"Functional simulation output saved to {output_bin_path}")
|
||||
reporter.advance()
|
||||
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Compare Outputs")
|
||||
sim_arrays = parse_pim_simulator_outputs(output_bin_path, outputs_descriptor)
|
||||
reporter.suspend()
|
||||
passed = validate_outputs(sim_arrays, out_dir, outputs_descriptor, threshold, rtol=rtol, verbose=verbose)
|
||||
reporter.resume()
|
||||
reporter.advance()
|
||||
|
||||
print_stage(
|
||||
reporter, model_index, model_total, network_onnx_path.name,
|
||||
"Run Non-functional Simulation")
|
||||
pimsim_latency_ms = None
|
||||
pimsim_power_mw = None
|
||||
pimsim_energy_pj = None
|
||||
pimsim_status = PIMSIM_SKIPPED
|
||||
if pimsim_enabled:
|
||||
try:
|
||||
pimsim_latency_ms, pimsim_power_mw, pimsim_energy_pj = run_pimsim_nn(
|
||||
pimsim_nn_build_dir,
|
||||
pim_dir,
|
||||
pimsim_config_path,
|
||||
reporter=reporter,
|
||||
timeout_sec=command_timeout_seconds,
|
||||
)
|
||||
pimsim_status = PIMSIM_DONE
|
||||
print_info(
|
||||
reporter,
|
||||
f"Latency: {pimsim_latency_ms:.6f} ms, "
|
||||
f"Power: {pimsim_power_mw:.6f} mW, "
|
||||
f"Energy: {pimsim_energy_pj:.6f} pJ")
|
||||
except PimSimUnsupportedError as exc:
|
||||
pimsim_status = PIMSIM_UNSUPPORTED
|
||||
print_info(reporter, str(exc))
|
||||
except Exception as exc:
|
||||
pimsim_status = PIMSIM_FAILED
|
||||
reporter.suspend()
|
||||
print(
|
||||
Fore.RED
|
||||
+ f"pimsim-nn non-functional simulation failed: {type(exc).__name__}: {exc}"
|
||||
+ Style.RESET_ALL,
|
||||
file=sys.stderr,
|
||||
flush=True,
|
||||
)
|
||||
reporter.resume()
|
||||
else:
|
||||
print_info(reporter, "pimsim-nn non-functional simulation skipped")
|
||||
reporter.advance()
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Compile ONNX")
|
||||
network_so_path, network_mlir_path = compile_onnx_network(
|
||||
network_onnx_path, raptor_path, raptor_dir, runner_dir,
|
||||
reporter=reporter, timeout_sec=command_timeout_seconds)
|
||||
print_info(reporter, f"MLIR saved to {network_mlir_path}")
|
||||
print_info(reporter, f"Shared library saved to {network_so_path}")
|
||||
reporter.advance()
|
||||
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Build Runner")
|
||||
gen_network_runner(
|
||||
network_onnx_path, network_so_path, onnx_include_dir,
|
||||
entry="run_main_graph", out=runner_dir / "runner.c", verbose=False)
|
||||
runner_path = build_onnx_runner(
|
||||
runner_dir, runner_build_dir, reporter=reporter,
|
||||
timeout_sec=command_timeout_seconds)
|
||||
print_info(reporter, f"Runner built at {runner_path}")
|
||||
reporter.advance()
|
||||
reference_ready = True
|
||||
except Exception as exc:
|
||||
report_validation_failure(reporter, "reference", "compilation", exc)
|
||||
else:
|
||||
required_paths = (network_so_path, network_mlir_path, runner_path)
|
||||
reference_ready = all(path.exists() for path in required_paths)
|
||||
if not reference_ready:
|
||||
report_validation_failure(reporter, "reference", "artifact lookup", FileNotFoundError(
|
||||
"run-only mode requires the compiled shared library, ONNX MLIR, and runner"))
|
||||
|
||||
for execution in executions:
|
||||
name = execution["name"]
|
||||
root = execution["root"]
|
||||
pim_dir = root / "pim"
|
||||
if mode == MODE_RUN_ONLY:
|
||||
states[name]["compiled"] = (pim_dir / "config.json").exists()
|
||||
if not states[name]["compiled"]:
|
||||
report_validation_failure(reporter, name, "artifact lookup", FileNotFoundError(
|
||||
f"run-only mode requires compiled PIM artifacts at {pim_dir}"))
|
||||
else:
|
||||
states[name]["resource_metrics"] = collect_pim_resource_metrics(pim_dir)
|
||||
if name == "latency":
|
||||
resource_metrics = states[name]["resource_metrics"]
|
||||
continue
|
||||
try:
|
||||
print_stage(
|
||||
reporter, model_index, model_total, network_onnx_path.name,
|
||||
"Compile PIM", name,
|
||||
)
|
||||
root.mkdir(parents=True, exist_ok=True)
|
||||
started = time.perf_counter()
|
||||
timings = compile_with_raptor(
|
||||
network_onnx_path, raptor_path, root / stem, crossbar_size,
|
||||
crossbar_count, core_count=core_count,
|
||||
raptor_extra_args=execution["compile_args"], cwd=root,
|
||||
verbose=verbose, reporter=reporter,
|
||||
timeout_sec=command_timeout_seconds)
|
||||
elapsed = time.perf_counter() - started
|
||||
compile_time_s += elapsed
|
||||
states[name]["compile_time_s"] = elapsed
|
||||
for label, duration in timings.items():
|
||||
pim_pass_timings[label] = pim_pass_timings.get(label, 0) + duration
|
||||
states[name]["compiled"] = True
|
||||
states[name]["resource_metrics"] = collect_pim_resource_metrics(pim_dir)
|
||||
if name == "latency":
|
||||
resource_metrics = states[name]["resource_metrics"]
|
||||
print_info(reporter, f"PIM artifacts saved to {pim_dir}")
|
||||
except Exception as exc:
|
||||
report_validation_failure(reporter, name, "compilation", exc)
|
||||
reporter.advance()
|
||||
|
||||
if mode == MODE_COMPILE_ONLY:
|
||||
for state in states.values():
|
||||
state["passed"] = reference_ready and state["compiled"]
|
||||
else:
|
||||
input_batch = input_paths = reference_dirs = outputs_descriptor = None
|
||||
try:
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Generate Inputs")
|
||||
inputs_descriptor, outputs_descriptor = onnx_io(network_onnx_path)
|
||||
first_inputs, _ = gen_random_inputs(inputs_descriptor, seed=seed)
|
||||
input_batch = generate_input_batch(
|
||||
inputs_descriptor, first_inputs,
|
||||
max(execution["batch_size"] for execution in executions), seed)
|
||||
write_input_batch_csv(workspace_dir / "inputs.csv", input_batch)
|
||||
input_paths = write_input_batch_binaries(input_batch, workspace_dir / "simulation" / "inputs")
|
||||
input_flags = [
|
||||
save_inputs_to_files(
|
||||
network_onnx_path, inputs,
|
||||
out_dir=workspace_dir / "inputs" / f"{index:06d}")[0]
|
||||
for index, inputs in enumerate(input_batch)
|
||||
]
|
||||
print_info(reporter, f"Saved {len(input_batch)} input sample(s) to {workspace_dir / 'inputs.csv'}")
|
||||
reporter.advance()
|
||||
|
||||
if not reference_ready:
|
||||
raise FileNotFoundError("reference runner is unavailable")
|
||||
print_stage(reporter, model_index, model_total, network_onnx_path.name, "Run Reference")
|
||||
reference_dirs = []
|
||||
for index, flags in enumerate(input_flags):
|
||||
reference_dir = workspace_dir / "outputs" / f"{index:06d}"
|
||||
reference_dir.mkdir(parents=True, exist_ok=True)
|
||||
run_command(
|
||||
[runner_path, *flags, "--save-csv-dir", str(reference_dir)],
|
||||
cwd=runner_build_dir, reporter=reporter,
|
||||
timeout_sec=command_timeout_seconds)
|
||||
reference_dirs.append(reference_dir)
|
||||
print_info(reporter, f"Reference outputs saved for {len(reference_dirs)} sample(s)")
|
||||
reporter.advance()
|
||||
except Exception as exc:
|
||||
report_validation_failure(reporter, "reference", "execution", exc)
|
||||
|
||||
functional_data = None
|
||||
if all(value is not None for value in (
|
||||
input_batch, input_paths, reference_dirs, outputs_descriptor)):
|
||||
functional_data = input_batch, input_paths, reference_dirs, outputs_descriptor
|
||||
stage_context = reporter, model_index, model_total, network_onnx_path.name
|
||||
for execution in executions:
|
||||
validate_execution(
|
||||
execution, states[execution["name"]], functional_data,
|
||||
workspace_dir, simulator_dir, pimsim_nn_build_dir,
|
||||
threshold, rtol, verbose, command_timeout_seconds,
|
||||
stage_context, pimsim_fast)
|
||||
|
||||
latency = states["latency"]
|
||||
throughput = states.get("throughput")
|
||||
passed = all(state["passed"] for state in states.values())
|
||||
latency_metrics = latency["metrics"]
|
||||
throughput_metrics = throughput["metrics"] if throughput else {}
|
||||
reporter.record_result(passed)
|
||||
status = Fore.GREEN + "PASS" + Style.RESET_ALL if passed else Fore.RED + "FAIL" + Style.RESET_ALL
|
||||
reporter.log(Style.BRIGHT + f"Result: {status}" + Style.RESET_ALL)
|
||||
mode_metrics = {
|
||||
name: {
|
||||
"compile_time_s": state["compile_time_s"] or None,
|
||||
**state["resource_metrics"],
|
||||
}
|
||||
for name, state in states.items()
|
||||
}
|
||||
return ValidationResult(
|
||||
passed=passed,
|
||||
latency_passed=latency["passed"],
|
||||
throughput_passed=throughput["passed"] if throughput else None,
|
||||
pim_pass_timings=pim_pass_timings,
|
||||
pimsim_latency_ms=pimsim_latency_ms,
|
||||
pimsim_power_mw=pimsim_power_mw,
|
||||
pimsim_energy_pj=pimsim_energy_pj,
|
||||
pimsim_status=pimsim_status,
|
||||
compile_time_s=compile_time_s,
|
||||
pimsim_latency_ms=latency_metrics.get("latency_ms"),
|
||||
pimsim_throughput_samples_s=throughput_metrics.get("throughput"),
|
||||
pimsim_power_mw=latency_metrics.get("average_power_mw"),
|
||||
pimsim_energy_pj=latency_metrics.get("average_energy_pj"),
|
||||
pimsim_throughput_average_latency_ms=throughput_metrics.get("average_latency_ms"),
|
||||
pimsim_throughput_average_power_mw=throughput_metrics.get("average_power_mw"),
|
||||
pimsim_throughput_average_energy_pj=throughput_metrics.get("average_energy_pj"),
|
||||
mode_metrics=mode_metrics,
|
||||
pimsim_status=latency["pimsim_status"],
|
||||
throughput_pimsim_status=(
|
||||
throughput["pimsim_status"] if throughput else PIMSIM_SKIPPED),
|
||||
compile_time_s=compile_time_s or None,
|
||||
**resource_metrics,
|
||||
)
|
||||
except Exception:
|
||||
failed_with_exception = True
|
||||
reporter.record_result(False)
|
||||
reporter.log(Style.BRIGHT + Fore.RED + "Result: FAIL" + Style.RESET_ALL)
|
||||
reporter.suspend()
|
||||
raise
|
||||
finally:
|
||||
if not failed_with_exception:
|
||||
reporter.log("=" * 72)
|
||||
reporter.log("=" * 72)
|
||||
if owns_reporter:
|
||||
reporter.finish()
|
||||
|
||||
@@ -15,7 +15,7 @@ REPO_ROOT = VALIDATION_DIR.parent
|
||||
if str(VALIDATION_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(VALIDATION_DIR))
|
||||
|
||||
from raptor_validation.onnx_utils import _ONNX_TO_NP, onnx_io, write_inputs_to_memory_bin
|
||||
from raptor_validation.onnx_utils import _ONNX_TO_NP, onnx_io, write_inputs_binary, write_inputs_to_memory_bin
|
||||
from raptor_validation.validate_one import (
|
||||
MODE_COMPILE_ONLY,
|
||||
build_dump_ranges,
|
||||
@@ -222,6 +222,8 @@ def run_reference_and_simulator(args, model_path: Path, tensor: np.ndarray):
|
||||
subprocess.run(runner_cmd, cwd=runner_build_dir, check=True)
|
||||
|
||||
write_inputs_to_memory_bin(pim_dir / "memory.bin", pim_dir / "config.json", [tensor])
|
||||
input_bin_path = simulation_dir / "input.bin"
|
||||
write_inputs_binary(input_bin_path, [tensor])
|
||||
dump_ranges = build_dump_ranges(pim_dir / "config.json", output_descriptors)
|
||||
output_bin_path = simulation_dir / "out.bin"
|
||||
run_pim_simulator(
|
||||
@@ -230,6 +232,7 @@ def run_reference_and_simulator(args, model_path: Path, tensor: np.ndarray):
|
||||
output_bin_path,
|
||||
dump_ranges,
|
||||
timeout_sec=args.command_timeout_seconds,
|
||||
input_paths=[input_bin_path],
|
||||
)
|
||||
|
||||
output_index, output_name, output_dtype_code, output_shape = output_descriptors[0]
|
||||
|
||||
@@ -33,18 +33,22 @@ sys.path.insert(0, str(VALIDATION_DIR))
|
||||
from raptor_validation.gen_network_runner import gen_network_runner # noqa: E402
|
||||
from raptor_validation.onnx_utils import ( # noqa: E402
|
||||
_ONNX_TO_NP,
|
||||
generate_input_batch,
|
||||
gen_random_inputs,
|
||||
onnx_io,
|
||||
save_inputs_to_files,
|
||||
write_input_batch_binaries,
|
||||
write_input_batch_csv,
|
||||
write_inputs_to_memory_bin,
|
||||
)
|
||||
from raptor_validation.raptor import compile_with_raptor # noqa: E402
|
||||
from raptor_validation.pimsim_nn import ( # noqa: E402
|
||||
export_raptor_latency_artifact,
|
||||
export_raptor_pimsim_artifact,
|
||||
parse_pimsim_nn_metrics,
|
||||
)
|
||||
from raptor_validation.validate_one import ( # noqa: E402
|
||||
STAGE_COLORS,
|
||||
build_pim_simulator_command,
|
||||
build_dump_ranges,
|
||||
parse_pim_simulator_outputs,
|
||||
)
|
||||
@@ -402,6 +406,29 @@ def generate_reference_outputs(
|
||||
return reference_dir
|
||||
|
||||
|
||||
def generate_reference_batch_outputs(
|
||||
runner_path: Path,
|
||||
runner_build_dir: Path,
|
||||
model_path: Path,
|
||||
input_batch: list[list[np.ndarray]],
|
||||
steps: list[StepRecord],
|
||||
args: argparse.Namespace,
|
||||
out_dir: Path,
|
||||
) -> list[Path]:
|
||||
return [
|
||||
generate_reference_outputs(
|
||||
runner_path,
|
||||
runner_build_dir,
|
||||
model_path,
|
||||
sample,
|
||||
steps,
|
||||
args,
|
||||
out_dir / f"batch_{index:06d}",
|
||||
)
|
||||
for index, sample in enumerate(input_batch)
|
||||
]
|
||||
|
||||
|
||||
def prepare_common_artifacts(
|
||||
args: argparse.Namespace,
|
||||
model_path: Path,
|
||||
@@ -500,31 +527,31 @@ def run_functional_validation(
|
||||
pim_dir: Path,
|
||||
config_path: Path,
|
||||
output_bin: Path,
|
||||
input_bins: list[Path],
|
||||
outputs_desc: list[tuple[int, str, int, list[int]]],
|
||||
reference_dir: Path,
|
||||
reference_dirs: list[Path],
|
||||
steps: list[StepRecord],
|
||||
args: argparse.Namespace,
|
||||
*,
|
||||
channel_last: bool = False,
|
||||
) -> CompareResult:
|
||||
dump_ranges = build_dump_ranges(config_path, outputs_desc)
|
||||
cmd = [
|
||||
"cargo",
|
||||
"run",
|
||||
"--no-default-features",
|
||||
"--release",
|
||||
"--package",
|
||||
"pim-simulator",
|
||||
"--bin",
|
||||
"pim-simulator",
|
||||
"--",
|
||||
"-f",
|
||||
str(pim_dir),
|
||||
"-o",
|
||||
str(output_bin),
|
||||
"-d",
|
||||
batch_size = len(input_bins)
|
||||
if batch_size == 0 or len(reference_dirs) != batch_size:
|
||||
raise ValueError(
|
||||
f"functional validation requires one input and reference per iteration, got "
|
||||
f"{batch_size} inputs and {len(reference_dirs)} references"
|
||||
)
|
||||
batch_output_dir = output_bin.parent / f"{output_bin.stem}_iterations"
|
||||
shutil.rmtree(batch_output_dir, ignore_errors=True)
|
||||
cmd = build_pim_simulator_command(
|
||||
pim_dir,
|
||||
output_bin,
|
||||
dump_ranges,
|
||||
]
|
||||
input_bins,
|
||||
args.pimsim_mode,
|
||||
batch_output_dir,
|
||||
)
|
||||
output_bin.parent.mkdir(parents=True, exist_ok=True)
|
||||
run_logged(
|
||||
label,
|
||||
@@ -534,13 +561,25 @@ def run_functional_validation(
|
||||
steps=steps,
|
||||
stage="Run Functional Simulation",
|
||||
)
|
||||
return compare_simulator_outputs(
|
||||
output_bin,
|
||||
outputs_desc,
|
||||
reference_dir,
|
||||
threshold=args.threshold,
|
||||
rtol=args.rtol,
|
||||
channel_last=channel_last,
|
||||
max_diffs: dict[str, float] = {}
|
||||
failed_iterations = []
|
||||
for index, reference_dir in enumerate(reference_dirs):
|
||||
result = compare_simulator_outputs(
|
||||
batch_output_dir / f"output_{index:06d}.bin",
|
||||
outputs_desc,
|
||||
reference_dir,
|
||||
threshold=args.threshold,
|
||||
rtol=args.rtol,
|
||||
channel_last=channel_last,
|
||||
)
|
||||
if not result.passed:
|
||||
failed_iterations.append(index)
|
||||
for name, diff in result.max_diffs.items():
|
||||
max_diffs[name] = max(max_diffs.get(name, 0.0), diff)
|
||||
return CompareResult(
|
||||
passed=not failed_iterations,
|
||||
max_diffs=max_diffs,
|
||||
error=(f"batch iterations failed: {failed_iterations}" if failed_iterations else None),
|
||||
)
|
||||
|
||||
|
||||
@@ -730,6 +769,7 @@ def export_pimcomp_for_rust(
|
||||
"adc_count": sim_info["config"]["adc_count"],
|
||||
"array_group_map": {},
|
||||
"inputs_addresses": [input_addr],
|
||||
"inputs_sizes": [input_tensor.nbytes],
|
||||
"outputs_addresses": [],
|
||||
}
|
||||
output_name_to_node = {node["name"]: node for node in node_list}
|
||||
@@ -887,6 +927,8 @@ def run_pimsim_nn(
|
||||
str(config_path),
|
||||
"--gui=false",
|
||||
]
|
||||
if not args.no_fast:
|
||||
cmd.append("--fast")
|
||||
output = run_logged(
|
||||
label,
|
||||
cmd,
|
||||
@@ -997,8 +1039,12 @@ def perf_status(perf: dict[str, Any]) -> str:
|
||||
return "DONE"
|
||||
|
||||
|
||||
def perf_value(perf: dict[str, Any], key: str) -> Any:
|
||||
return perf[key] if key in perf else "n/a"
|
||||
def perf_value(perf: dict[str, Any], key: str, unit: str = "") -> Any:
|
||||
value = perf.get(key)
|
||||
if value is None:
|
||||
return "n/a"
|
||||
formatted = f"{value:.2f}" if isinstance(value, float) else value
|
||||
return f"{formatted} {unit}" if unit else formatted
|
||||
|
||||
|
||||
def empty_instruction_summary(reason: str | None = None, error: str | None = None) -> dict[str, Any]:
|
||||
@@ -1170,24 +1216,24 @@ def write_report(
|
||||
if pimsim_mode == "throughput":
|
||||
lines.extend(
|
||||
[
|
||||
"| Compiler | Status | Throughput (samples/s) | Avg latency (ms) | Avg power (mW) | Avg energy (pJ/it) | Output count |",
|
||||
"| Compiler | Status | Avg latency | Throughput | Avg power | Avg energy | Output count |",
|
||||
"|---|---|---:|---:|---:|---:|---:|",
|
||||
f"| Raptor | {perf_status(raptor_perf)} | {perf_value(raptor_perf, 'throughput')} | {perf_value(raptor_perf, 'average_latency_ms')} | "
|
||||
f"{perf_value(raptor_perf, 'average_power_mw')} | {perf_value(raptor_perf, 'average_energy_pj')} | {perf_value(raptor_perf, 'output_count')} |",
|
||||
f"| PIMCOMP | {perf_status(pimcomp_perf)} | {perf_value(pimcomp_perf, 'throughput')} | {perf_value(pimcomp_perf, 'average_latency_ms')} | "
|
||||
f"{perf_value(pimcomp_perf, 'average_power_mw')} | {perf_value(pimcomp_perf, 'average_energy_pj')} | {perf_value(pimcomp_perf, 'output_count')} |",
|
||||
f"| Raptor | {perf_status(raptor_perf)} | {perf_value(raptor_perf, 'average_latency_ms', 'ms')} | {perf_value(raptor_perf, 'throughput', 'samples/s')} | "
|
||||
f"{perf_value(raptor_perf, 'average_power_mw', 'mW')} | {perf_value(raptor_perf, 'average_energy_pj', 'pJ/it')} | {perf_value(raptor_perf, 'output_count')} |",
|
||||
f"| PIMCOMP | {perf_status(pimcomp_perf)} | {perf_value(pimcomp_perf, 'average_latency_ms', 'ms')} | {perf_value(pimcomp_perf, 'throughput', 'samples/s')} | "
|
||||
f"{perf_value(pimcomp_perf, 'average_power_mw', 'mW')} | {perf_value(pimcomp_perf, 'average_energy_pj', 'pJ/it')} | {perf_value(pimcomp_perf, 'output_count')} |",
|
||||
"",
|
||||
]
|
||||
)
|
||||
else:
|
||||
lines.extend(
|
||||
[
|
||||
"| Compiler | Status | Latency (ms) | Avg power (mW) | Avg energy (pJ) |",
|
||||
"| Compiler | Status | Latency | Avg power | Avg energy |",
|
||||
"|---|---|---:|---:|---:|",
|
||||
f"| Raptor | {perf_status(raptor_perf)} | {perf_value(raptor_perf, 'latency_ms')} | "
|
||||
f"{perf_value(raptor_perf, 'average_power_mw')} | {perf_value(raptor_perf, 'average_energy_pj')} |",
|
||||
f"| PIMCOMP | {perf_status(pimcomp_perf)} | {perf_value(pimcomp_perf, 'latency_ms')} | "
|
||||
f"{perf_value(pimcomp_perf, 'average_power_mw')} | {perf_value(pimcomp_perf, 'average_energy_pj')} |",
|
||||
f"| Raptor | {perf_status(raptor_perf)} | {perf_value(raptor_perf, 'latency_ms', 'ms')} | "
|
||||
f"{perf_value(raptor_perf, 'average_power_mw', 'mW')} | {perf_value(raptor_perf, 'average_energy_pj', 'pJ')} |",
|
||||
f"| PIMCOMP | {perf_status(pimcomp_perf)} | {perf_value(pimcomp_perf, 'latency_ms', 'ms')} | "
|
||||
f"{perf_value(pimcomp_perf, 'average_power_mw', 'mW')} | {perf_value(pimcomp_perf, 'average_energy_pj', 'pJ')} |",
|
||||
"",
|
||||
]
|
||||
)
|
||||
@@ -1293,6 +1339,7 @@ def main():
|
||||
parser.add_argument("--mesh-cols", type=int)
|
||||
parser.add_argument("--pimsim-time-ms", type=int, default=1000)
|
||||
parser.add_argument("--pimsim-mode", choices=["latency", "throughput"], default="latency")
|
||||
parser.add_argument("--batch-size", type=int, default=128)
|
||||
parser.add_argument("--pimcomp-pipeline", choices=["element", "batch"])
|
||||
parser.add_argument("--pimcomp-model-name", help="Use a PIMCOMP built-in model name such as vgg16.")
|
||||
parser.add_argument(
|
||||
@@ -1316,6 +1363,11 @@ def main():
|
||||
help="Preserve the PIMCOMP side of an existing comparison report without rerunning it.",
|
||||
)
|
||||
parser.add_argument("--skip-pimsim-nn", action="store_true")
|
||||
parser.add_argument(
|
||||
"--no-fast",
|
||||
action="store_true",
|
||||
help="Disable fast pimsim-nn throughput convergence for authoritative experiments.",
|
||||
)
|
||||
parser.add_argument("--verbose-raptor-compile", action="store_true")
|
||||
parser.add_argument("--raptor-extra-arg", action="append", default=[])
|
||||
parser.add_argument(
|
||||
@@ -1328,6 +1380,10 @@ def main():
|
||||
parser.error("--reuse-pimcomp-dir and --reuse-pimcomp-report are mutually exclusive")
|
||||
if args.pimsim_time_ms <= 0:
|
||||
parser.error("--pimsim-time-ms must be positive")
|
||||
if args.batch_size <= 0:
|
||||
parser.error("--batch-size must be positive")
|
||||
if args.pimsim_mode == "throughput" and args.batch_size < 2:
|
||||
parser.error("throughput mode requires batch size greater than 1")
|
||||
if args.timeout_seconds < 0:
|
||||
parser.error("--timeout-seconds must be non-negative")
|
||||
if args.pimcomp_pipeline is None:
|
||||
@@ -1371,6 +1427,7 @@ def main():
|
||||
|
||||
runner_path: Path | None = None
|
||||
reference_dir: Path | None = None
|
||||
reference_dirs: list[Path] = []
|
||||
raptor_pim_dir: Path | None = None
|
||||
raptor_pimsim_dir: Path | None = None
|
||||
raptor_pass_timings: dict[str, float] = {}
|
||||
@@ -1506,6 +1563,33 @@ def main():
|
||||
"Reference outputs were skipped because the native runner or model inputs are not available.",
|
||||
)
|
||||
|
||||
input_batch = None
|
||||
raptor_input_bins: list[Path] = []
|
||||
pimcomp_input_bins: list[Path] = []
|
||||
if model_io is not None:
|
||||
batch_size = 1 if args.pimsim_mode == "latency" else args.batch_size
|
||||
input_batch = generate_input_batch(inputs_desc, runtime_inputs, batch_size, args.seed)
|
||||
write_input_batch_csv(out_dir / "inputs.csv", input_batch)
|
||||
raptor_input_bins = write_input_batch_binaries(input_batch, out_dir / "simulation/raptor_inputs")
|
||||
if args.pimsim_mode == "throughput":
|
||||
throughput_references = try_stage(
|
||||
failures,
|
||||
"Run throughput references",
|
||||
generate_reference_batch_outputs,
|
||||
runner_path,
|
||||
runner_path.parent,
|
||||
model_path,
|
||||
input_batch,
|
||||
steps,
|
||||
args,
|
||||
out_dir / "reference",
|
||||
) if runner_path is not None and runner_path.exists() else None
|
||||
if throughput_references is not None:
|
||||
reference_dirs = throughput_references
|
||||
reference_dir = out_dir / "reference"
|
||||
elif reference_dir is not None:
|
||||
reference_dirs = [reference_dir]
|
||||
|
||||
if not reuse_raptor and model_path.exists() and hardware["core_count"] > 0:
|
||||
compiled_raptor = try_stage(
|
||||
failures,
|
||||
@@ -1535,7 +1619,7 @@ def main():
|
||||
raptor_pim_dir / "config.json",
|
||||
runtime_inputs,
|
||||
)
|
||||
if wrote_inputs and reference_dir is not None and outputs_desc:
|
||||
if wrote_inputs and reference_dirs and outputs_desc:
|
||||
validation = try_stage(
|
||||
failures,
|
||||
"Functional Validation Raptor",
|
||||
@@ -1544,13 +1628,14 @@ def main():
|
||||
raptor_pim_dir,
|
||||
raptor_pim_dir / "config.json",
|
||||
out_dir / "simulation/out.bin",
|
||||
raptor_input_bins,
|
||||
outputs_desc,
|
||||
reference_dir,
|
||||
reference_dirs,
|
||||
steps,
|
||||
args,
|
||||
)
|
||||
raptor_validation = validation if validation is not None else failed_validation("Raptor validation failed")
|
||||
elif reference_dir is None:
|
||||
elif not reference_dirs:
|
||||
raptor_validation = skipped_validation("Reference outputs are not available")
|
||||
elif not outputs_desc:
|
||||
raptor_validation = skipped_validation("Output descriptors are not available")
|
||||
@@ -1616,7 +1701,14 @@ def main():
|
||||
"PIMCOMP functional export failed because model inputs are not available.",
|
||||
)
|
||||
|
||||
if not reuse_pimcomp and pimcomp_export_dir is not None and reference_dir is not None and outputs_desc:
|
||||
if input_batch is not None and pimcomp_export_dir is not None:
|
||||
pimcomp_input_bins = write_input_batch_binaries(
|
||||
input_batch,
|
||||
out_dir / "simulation/pimcomp_inputs",
|
||||
transform=flatten_pimcomp_input,
|
||||
)
|
||||
|
||||
if not reuse_pimcomp and pimcomp_export_dir is not None and reference_dirs and outputs_desc:
|
||||
validation = try_stage(
|
||||
failures,
|
||||
"Functional Validation PIMCOMP",
|
||||
@@ -1625,8 +1717,9 @@ def main():
|
||||
pimcomp_export_dir,
|
||||
pimcomp_export_dir / "config.json",
|
||||
out_dir / "simulation/pimcomp.out.bin",
|
||||
pimcomp_input_bins,
|
||||
outputs_desc,
|
||||
reference_dir,
|
||||
reference_dirs,
|
||||
steps,
|
||||
args,
|
||||
channel_last=True,
|
||||
@@ -1636,7 +1729,7 @@ def main():
|
||||
pass
|
||||
elif pimcomp_export_dir is None:
|
||||
pimcomp_validation = failed_validation("PIMCOMP functional export is not available")
|
||||
elif reference_dir is None:
|
||||
elif not reference_dirs:
|
||||
pimcomp_validation = failed_validation("Reference outputs are not available")
|
||||
else:
|
||||
pimcomp_validation = failed_validation("Output descriptors are not available")
|
||||
@@ -1673,7 +1766,7 @@ def main():
|
||||
raptor_pimsim_dir = try_stage(
|
||||
failures,
|
||||
"Export Raptor for pimsim-nn",
|
||||
export_raptor_latency_artifact,
|
||||
export_raptor_pimsim_artifact,
|
||||
raptor_pim_dir,
|
||||
out_dir / "raptor/pimsim_nn",
|
||||
)
|
||||
@@ -1777,6 +1870,8 @@ def main():
|
||||
"common_dir": str(common_dir),
|
||||
"reference_inputs": optional_path(common_dir / "inputs"),
|
||||
"reference_outputs": optional_path(reference_dir),
|
||||
"batch_inputs": optional_path(out_dir / "inputs.csv"),
|
||||
"batch_outputs": optional_path(out_dir / "simulation/out_iterations"),
|
||||
"reference_runner": optional_path(runner_path),
|
||||
"raptor_pim": optional_path(raptor_pim_dir),
|
||||
"raptor_pimsim_nn": optional_path(raptor_pimsim_dir),
|
||||
|
||||
@@ -108,10 +108,10 @@ def write_results_csv(
|
||||
"pimcomp_pipeline",
|
||||
"raptor_functional_validation",
|
||||
"pimcomp_functional_validation",
|
||||
"raptor_throughput_samples_s",
|
||||
"pimcomp_throughput_samples_s",
|
||||
"raptor_latency_ms",
|
||||
"pimcomp_latency_ms",
|
||||
"raptor_throughput_samples_s",
|
||||
"pimcomp_throughput_samples_s",
|
||||
"raptor_power_mw",
|
||||
"pimcomp_power_mw",
|
||||
"raptor_energy_pj",
|
||||
@@ -215,13 +215,13 @@ def write_results_csv(
|
||||
def performance_values(performance: dict) -> dict[str, float | None]:
|
||||
parsed = parse_pimsim_nn_metrics(performance.get("raw_output", ""))
|
||||
return {
|
||||
"throughput": performance.get("throughput") or parsed.get("throughput"),
|
||||
"latency": (
|
||||
performance.get("latency_ms")
|
||||
or performance.get("average_latency_ms")
|
||||
or parsed.get("latency_ms")
|
||||
or parsed.get("average_latency_ms")
|
||||
),
|
||||
"throughput": performance.get("throughput") or parsed.get("throughput"),
|
||||
"power": performance.get("average_power_mw") or parsed.get("average_power_mw"),
|
||||
"energy": performance.get("average_energy_pj") or parsed.get("average_energy_pj"),
|
||||
}
|
||||
@@ -261,7 +261,7 @@ def comparison_winner(mode: str, raptor: float, pimcomp: float) -> str:
|
||||
|
||||
|
||||
def format_value(value: float | None) -> str:
|
||||
return "NA" if value is None else f"{value:.6f}"
|
||||
return "NA" if value is None else f"{value:.2f}"
|
||||
|
||||
|
||||
def print_stage(title: str, color: str) -> None:
|
||||
@@ -323,7 +323,9 @@ def comparison_command(
|
||||
pipeline: int,
|
||||
pimcomp_pipeline: str,
|
||||
pimsim_time_ms: int,
|
||||
batch_size: int,
|
||||
timeout: float,
|
||||
fast: bool,
|
||||
reuse_raptor_report: Path | None = None,
|
||||
reuse_pimcomp_dir: Path | None = None,
|
||||
reuse_pimcomp_report: Path | None = None,
|
||||
@@ -351,6 +353,8 @@ def comparison_command(
|
||||
str(config),
|
||||
"--pimsim-mode",
|
||||
mode,
|
||||
"--batch-size",
|
||||
str(batch_size),
|
||||
*time_args,
|
||||
"--pimcomp-pipeline",
|
||||
pimcomp_pipeline,
|
||||
@@ -360,6 +364,7 @@ def comparison_command(
|
||||
"--timeout-seconds",
|
||||
str(timeout),
|
||||
"--fail-on-error",
|
||||
*([] if fast else ["--no-fast"]),
|
||||
*reuse_args,
|
||||
]
|
||||
|
||||
@@ -396,7 +401,9 @@ def comparison_command_for(
|
||||
spec.pipeline,
|
||||
spec.pimcomp_pipeline,
|
||||
args.pimsim_time_ms,
|
||||
args.batch_size,
|
||||
args.timeout_seconds,
|
||||
not args.no_fast,
|
||||
reuse_raptor_report=(
|
||||
report
|
||||
if args.only == "pimcomp"
|
||||
@@ -512,8 +519,14 @@ def main() -> int:
|
||||
parser.add_argument(
|
||||
"--pimsim-time-ms",
|
||||
type=int,
|
||||
default=100,
|
||||
help="throughput pimsim-nn horizon in ms (default: 100).",
|
||||
default=1000,
|
||||
help="throughput pimsim-nn convergence deadline in ms (default: 1000).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--batch-size",
|
||||
type=int,
|
||||
default=128,
|
||||
help="functional throughput batch size (default: 128).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--timeout-seconds",
|
||||
@@ -525,8 +538,8 @@ def main() -> int:
|
||||
"-j",
|
||||
"--jobs",
|
||||
type=int,
|
||||
default=os.cpu_count() or 1,
|
||||
help="Number of comparisons to run in parallel (default: all available CPUs).",
|
||||
default=4,
|
||||
help="Number of comparisons to run in parallel (default: 4).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--clean",
|
||||
@@ -534,6 +547,11 @@ def main() -> int:
|
||||
help="Remove generated comparison artifacts and result summaries, then exit.",
|
||||
)
|
||||
parser.add_argument("--dry-run", action="store_true", help="Print commands without modifying files.")
|
||||
parser.add_argument(
|
||||
"--no-fast",
|
||||
action="store_true",
|
||||
help="Disable fast pimsim-nn throughput convergence for authoritative experiments.",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
out_dir = args.out_dir.resolve() if args.out_dir is not None else None
|
||||
@@ -547,6 +565,8 @@ def main() -> int:
|
||||
parser.error("--jobs must be at least 1")
|
||||
if args.pimsim_time_ms <= 0:
|
||||
parser.error("--pimsim-time-ms must be positive")
|
||||
if args.batch_size <= 0:
|
||||
parser.error("--batch-size must be positive")
|
||||
if args.timeout_seconds < 0:
|
||||
parser.error("--timeout-seconds must be non-negative")
|
||||
comparisons_by_arch: dict[str, tuple[tuple[str, int, str], ...]] = {}
|
||||
|
||||
@@ -21,7 +21,7 @@ if sys.version_info < (3, 10):
|
||||
"Run it with a newer interpreter, for example your project venv Python."
|
||||
)
|
||||
|
||||
from raptor_validation.onnx_utils import _ONNX_TO_NP, onnx_io, write_inputs_to_memory_bin
|
||||
from raptor_validation.onnx_utils import _ONNX_TO_NP, onnx_io, write_inputs_binary, write_inputs_to_memory_bin
|
||||
from raptor_validation.validate_one import (
|
||||
MODE_COMPILE_ONLY,
|
||||
build_dump_ranges,
|
||||
@@ -137,6 +137,8 @@ def run_local_reference_and_simulator(args, network_dir: Path, network_onnx_path
|
||||
|
||||
tensor = np.loadtxt(paths["input_csv"], delimiter=",", dtype=np.float32).reshape(1, 3, 640, 640)
|
||||
write_inputs_to_memory_bin(paths["raptor_pim"] / "memory.bin", paths["raptor_pim"] / "config.json", [tensor])
|
||||
input_bin = paths["sim_dir"] / "input.bin"
|
||||
write_inputs_binary(input_bin, [tensor])
|
||||
|
||||
dump_ranges = build_dump_ranges(paths["raptor_pim"] / "config.json", output_descriptors)
|
||||
run_pim_simulator(
|
||||
@@ -145,6 +147,7 @@ def run_local_reference_and_simulator(args, network_dir: Path, network_onnx_path
|
||||
paths["sim_bin"],
|
||||
dump_ranges,
|
||||
timeout_sec=args.command_timeout_seconds,
|
||||
input_paths=[input_bin],
|
||||
)
|
||||
return paths, output_descriptors[0]
|
||||
|
||||
|
||||
@@ -272,6 +272,7 @@ def remote_case_paths(args, case_name: str):
|
||||
"input_csv": root / "real_image_validation" / "inputs" / f"{case_name}.csv",
|
||||
"ref_dir": root / "real_image_validation" / "reference" / case_name,
|
||||
"sim_dir": root / "real_image_validation" / "simulation" / case_name,
|
||||
"sim_input": root / "real_image_validation" / "simulation" / case_name / "input.bin",
|
||||
"sim_bin": root / "real_image_validation" / "simulation" / case_name / "out.bin",
|
||||
}
|
||||
|
||||
@@ -310,8 +311,10 @@ import numpy as np
|
||||
from pathlib import Path
|
||||
input_csv = Path({json.dumps(str(paths["input_csv"]))})
|
||||
pim_dir = Path({json.dumps(str(paths["raptor_pim"]))})
|
||||
input_bin = Path({json.dumps(str(paths["sim_input"]))})
|
||||
config = json.loads((pim_dir / "config.json").read_text())
|
||||
tensor = np.loadtxt(input_csv, delimiter=",", dtype=np.float32).reshape(1, 3, 640, 640)
|
||||
input_bin.write_bytes(tensor.tobytes(order="C"))
|
||||
with open(pim_dir / "memory.bin", "r+b") as f:
|
||||
f.seek(config["inputs_addresses"][0])
|
||||
f.write(tensor.tobytes(order="C"))
|
||||
@@ -327,7 +330,8 @@ PY
|
||||
f"export PATH=$HOME/.cargo/bin:$PATH && "
|
||||
f"cd {quoted_project}/backend-simulators/pim/pim-simulator && "
|
||||
f"cargo run --no-default-features --release --package pim-simulator --bin pim-simulator -- "
|
||||
f"-f {quoted_pim} -o {quoted_sim_bin} -d {dump_range}"
|
||||
f"-f {quoted_pim} -o {quoted_sim_bin} -d {dump_range} "
|
||||
f"--mode latency --input {shlex.quote(str(paths['sim_input']))}"
|
||||
)
|
||||
remote_bash(args.ssh_key, args.remote_host, sim_command)
|
||||
return paths
|
||||
|
||||
+133
-52
@@ -34,6 +34,8 @@ from raptor_validation.raptor import PIM_PASS_LABELS
|
||||
|
||||
DEFAULT_PIMCOMP_CONFIG = "arch-a"
|
||||
PIMCOMP_CONFIG_CHOICES = ("arch-a", "arch-b", "arch-c")
|
||||
THROUGHPUT_PIPELINE = 4
|
||||
THROUGHPUT_BATCH_SIZE = 4
|
||||
|
||||
|
||||
def discover_onnx_files(root):
|
||||
@@ -67,9 +69,13 @@ def run_validation_job(job):
|
||||
)
|
||||
except Exception as exc:
|
||||
print_validation_error(reporter, rel, exc)
|
||||
return ValidationResult(False, pimsim_status=PIMSIM_NOT_RUN)
|
||||
finally:
|
||||
reporter.finish()
|
||||
return ValidationResult(
|
||||
False,
|
||||
latency_passed=False,
|
||||
throughput_passed=False,
|
||||
pimsim_status=PIMSIM_NOT_RUN,
|
||||
throughput_pimsim_status=PIMSIM_NOT_RUN,
|
||||
)
|
||||
|
||||
sys.stdout.flush()
|
||||
sys.stderr.flush()
|
||||
@@ -89,6 +95,7 @@ def run_validation_job(job):
|
||||
os.dup2(saved_stderr, 2)
|
||||
os.close(saved_stdout)
|
||||
os.close(saved_stderr)
|
||||
reporter.finish()
|
||||
completed.append((str(rel), result, str(log_path) if log_path else None))
|
||||
return completed
|
||||
|
||||
@@ -135,10 +142,10 @@ def print_average_pim_pass_timings(pass_timing_sums, pass_timing_counts, total_t
|
||||
print(f" {'Total'.ljust(28)} {total_timing_sum / timed_benchmark_count:.4f}s")
|
||||
|
||||
|
||||
def format_pimsim_metric(result, value, unit):
|
||||
if result.pimsim_status == PIMSIM_DONE:
|
||||
return f"{value:.6f} {unit}"
|
||||
return result.pimsim_status
|
||||
def format_pimsim_metric(status, value, unit):
|
||||
if status == PIMSIM_DONE:
|
||||
return f"{value:.2f} {unit}"
|
||||
return status
|
||||
|
||||
|
||||
def format_memory(byte_count):
|
||||
@@ -147,6 +154,47 @@ def format_memory(byte_count):
|
||||
return f"{byte_count / (1 << 20):.2f} MiB"
|
||||
|
||||
|
||||
def print_results_table(title, headers, rows):
|
||||
widths = [max(len(header), *(len(row[index]) for row in rows))
|
||||
for index, header in enumerate(headers)]
|
||||
separator = "+-" + "-+-".join("-" * width for width in widths) + "-+"
|
||||
|
||||
def table_line(row):
|
||||
return "| " + " | ".join(
|
||||
value.ljust(widths[index]) if index < 3 else value.rjust(widths[index])
|
||||
for index, value in enumerate(row)) + " |"
|
||||
|
||||
print("\n" + Style.BRIGHT + Fore.CYAN + title + Style.RESET_ALL)
|
||||
print(separator)
|
||||
print(table_line(headers))
|
||||
print(separator)
|
||||
for row in rows:
|
||||
line = table_line(row)
|
||||
status = row[2].ljust(widths[2])
|
||||
color = Fore.GREEN if row[2] == "PASS" else Fore.RED
|
||||
print(line.replace(status, color + status + Style.RESET_ALL, 1))
|
||||
print(separator)
|
||||
|
||||
|
||||
def mode_common_metrics(result, mode):
|
||||
metrics = result.mode_metrics.get(mode, {})
|
||||
fallback = {
|
||||
"compile_time_s": result.compile_time_s if mode == "latency" else None,
|
||||
"host_memory_bytes": result.host_memory_bytes if mode == "latency" else None,
|
||||
"cores_memory_bytes": result.cores_memory_bytes if mode == "latency" else None,
|
||||
"used_core_count": result.used_core_count if mode == "latency" else None,
|
||||
"used_crossbar_count": result.used_crossbar_count if mode == "latency" else None,
|
||||
}
|
||||
values = {**fallback, **metrics}
|
||||
return (
|
||||
f"{values['compile_time_s']:.3f} s" if values["compile_time_s"] is not None else "-",
|
||||
format_memory(values["host_memory_bytes"]),
|
||||
format_memory(values["cores_memory_bytes"]),
|
||||
str(values["used_core_count"]) if values["used_core_count"] is not None else "-",
|
||||
str(values["used_crossbar_count"]) if values["used_crossbar_count"] is not None else "-",
|
||||
)
|
||||
|
||||
|
||||
def operation_label(relative_path):
|
||||
path = Path(relative_path)
|
||||
return str(path.parent) if path.parent != Path(".") else path.stem
|
||||
@@ -172,6 +220,8 @@ def main():
|
||||
"(default: arch-a).")
|
||||
ap.add_argument("--skip-non-functional-simulation", action="store_true",
|
||||
help="Skip non-functional simulation.")
|
||||
ap.add_argument("--no-fast", action="store_true",
|
||||
help="Disable fast pimsim-nn throughput convergence for authoritative experiments.")
|
||||
ap.add_argument("--threshold", type=float, default=1e-3,
|
||||
help="Absolute tolerance for per-element output comparison.")
|
||||
ap.add_argument("--relative-threshold", type=float, default=1e-5,
|
||||
@@ -187,8 +237,8 @@ def main():
|
||||
help="Additional argument to pass through to the Raptor compiler. Repeat as needed.")
|
||||
ap.add_argument("--command-timeout-seconds", type=float, default=1000000.0,
|
||||
help="Per-subprocess timeout in seconds for compiler, runner, and simulation commands.")
|
||||
ap.add_argument("-j", "--jobs", type=int, default=os.cpu_count() or 1,
|
||||
help="Number of model validations to run in parallel (default: all available CPUs).")
|
||||
ap.add_argument("-j", "--jobs", type=int, default=4,
|
||||
help="Number of model validations to run in parallel (default: 4).")
|
||||
ap.add_argument("--clean", action="store_true",
|
||||
help="Remove generated validation artifacts under each model workspace and exit.")
|
||||
mode_group = ap.add_mutually_exclusive_group()
|
||||
@@ -214,6 +264,9 @@ def main():
|
||||
script_dir / ".." / "backend-simulators" / "pim" / "pimsim-nn" / "build"
|
||||
)
|
||||
pimsim_config_path = pimcomp_configs_dir / a.pimcomp_config / "latency_config.json"
|
||||
throughput_pimsim_config_path = (
|
||||
pimcomp_configs_dir / a.pimcomp_config / "throughput_config_1000ms.json"
|
||||
)
|
||||
|
||||
if not operations_dir.is_dir():
|
||||
print(Fore.RED + f"Operations directory not found: {operations_dir}" + Style.RESET_ALL)
|
||||
@@ -252,6 +305,7 @@ def main():
|
||||
mode = MODE_RUN_ONLY
|
||||
|
||||
selected_pimsim_config = None
|
||||
selected_throughput_pimsim_config = None
|
||||
if not a.skip_non_functional_simulation:
|
||||
compatibility_errors = pimcomp_compatibility_errors(
|
||||
pimsim_config_path,
|
||||
@@ -269,6 +323,7 @@ def main():
|
||||
)
|
||||
else:
|
||||
selected_pimsim_config = pimsim_config_path
|
||||
selected_throughput_pimsim_config = throughput_pimsim_config_path
|
||||
|
||||
results = {} # relative_path -> ValidationResult
|
||||
pass_timing_sums = {label: 0.0 for _, label in PIM_PASS_LABELS}
|
||||
@@ -290,6 +345,10 @@ def main():
|
||||
"raptor_extra_args": raptor_extra_args,
|
||||
"pimsim_nn_build_dir": pimsim_nn_build_dir,
|
||||
"pimsim_config_path": selected_pimsim_config,
|
||||
"throughput_pipeline": THROUGHPUT_PIPELINE,
|
||||
"throughput_batch_size": THROUGHPUT_BATCH_SIZE,
|
||||
"throughput_pimsim_config_path": selected_throughput_pimsim_config,
|
||||
"pimsim_fast": not a.no_fast,
|
||||
"command_timeout_seconds": a.command_timeout_seconds,
|
||||
"threshold": a.threshold,
|
||||
"rtol": a.relative_threshold,
|
||||
@@ -356,44 +415,68 @@ def main():
|
||||
# Summary
|
||||
n_passed = sum(1 for result in results.values() if result.passed)
|
||||
n_total = len(results)
|
||||
headers = ("Operation", "Result", "Compile", "Host mem", "Cores mem",
|
||||
"Cores", "Xbars", "Latency", "Power", "Energy")
|
||||
rows = []
|
||||
latency_headers = ("Operation", "Arch", "Result", "Compile", "Host mem", "Cores mem",
|
||||
"Cores", "Xbars", "Latency", "Power", "Energy")
|
||||
throughput_headers = ("Operation", "Arch", "Result", "Compile", "Host mem", "Cores mem",
|
||||
"Cores", "Xbars", "Throughput", "Avg latency", "Avg power",
|
||||
"Avg energy")
|
||||
csv_headers = (
|
||||
"Operation", "Arch", "Result (l)", "Result (t)",
|
||||
"Compile (l)", "Host mem (l)", "Cores mem (l)", "Cores (l)", "Xbars (l)",
|
||||
"Latency (l)", "Power (l)", "Energy (l)",
|
||||
"Compile (t)", "Host mem (t)", "Cores mem (t)", "Cores (t)", "Xbars (t)",
|
||||
"Avg latency (t)", "Throughput (t)", "Avg power (t)", "Avg energy (t)",
|
||||
)
|
||||
latency_rows = []
|
||||
throughput_rows = []
|
||||
csv_rows = []
|
||||
for rel, result in results.items():
|
||||
rows.append((
|
||||
operation_label(rel), "PASS" if result.passed else "FAIL",
|
||||
f"{result.compile_time_s:.3f} s" if result.compile_time_s is not None else "-",
|
||||
format_memory(result.host_memory_bytes),
|
||||
format_memory(result.cores_memory_bytes),
|
||||
str(result.used_core_count) if result.used_core_count is not None else "-",
|
||||
str(result.used_crossbar_count) if result.used_crossbar_count is not None else "-",
|
||||
format_pimsim_metric(result, result.pimsim_latency_ms, "ms"),
|
||||
format_pimsim_metric(result, result.pimsim_power_mw, "mW"),
|
||||
format_pimsim_metric(result, result.pimsim_energy_pj, "pJ"),
|
||||
operation = operation_label(rel)
|
||||
latency_status = "PASS" if result.latency_passed else "FAIL"
|
||||
throughput_status = "PASS" if result.throughput_passed else "FAIL"
|
||||
latency_common = mode_common_metrics(result, "latency")
|
||||
throughput_common = mode_common_metrics(result, "throughput")
|
||||
latency_metrics = (
|
||||
format_pimsim_metric(result.pimsim_status, result.pimsim_latency_ms, "ms"),
|
||||
format_pimsim_metric(result.pimsim_status, result.pimsim_power_mw, "mW"),
|
||||
format_pimsim_metric(result.pimsim_status, result.pimsim_energy_pj, "pJ"),
|
||||
)
|
||||
throughput_metrics = (
|
||||
format_pimsim_metric(
|
||||
result.throughput_pimsim_status,
|
||||
result.pimsim_throughput_samples_s,
|
||||
"samples/s",
|
||||
),
|
||||
format_pimsim_metric(
|
||||
result.throughput_pimsim_status,
|
||||
result.pimsim_throughput_average_latency_ms,
|
||||
"ms",
|
||||
),
|
||||
format_pimsim_metric(
|
||||
result.throughput_pimsim_status,
|
||||
result.pimsim_throughput_average_power_mw,
|
||||
"mW",
|
||||
),
|
||||
format_pimsim_metric(
|
||||
result.throughput_pimsim_status,
|
||||
result.pimsim_throughput_average_energy_pj,
|
||||
"pJ/it",
|
||||
),
|
||||
)
|
||||
latency_rows.append((operation, a.pimcomp_config, latency_status, *latency_common, *latency_metrics))
|
||||
throughput_rows.append((operation, a.pimcomp_config, throughput_status, *throughput_common, *throughput_metrics))
|
||||
csv_rows.append((
|
||||
operation, a.pimcomp_config, latency_status, throughput_status,
|
||||
*latency_common, *latency_metrics,
|
||||
*throughput_common, *throughput_metrics,
|
||||
))
|
||||
widths = [max(len(header), *(len(row[index]) for row in rows))
|
||||
for index, header in enumerate(headers)]
|
||||
separator = "+-" + "-+-".join("-" * width for width in widths) + "-+"
|
||||
|
||||
def table_line(row):
|
||||
return "| " + " | ".join(
|
||||
value.ljust(widths[index]) if index < 2 else value.rjust(widths[index])
|
||||
for index, value in enumerate(row)) + " |"
|
||||
|
||||
print(separator)
|
||||
print(table_line(headers))
|
||||
print(separator)
|
||||
for row in rows:
|
||||
line = table_line(row)
|
||||
color = Fore.GREEN if row[1] == "PASS" else Fore.RED
|
||||
line = line.replace(row[1].ljust(widths[1]),
|
||||
color + row[1].ljust(widths[1]) + Style.RESET_ALL, 1)
|
||||
print(line)
|
||||
print(separator)
|
||||
print_results_table("Latency", latency_headers, latency_rows)
|
||||
print_results_table("Throughput", throughput_headers, throughput_rows)
|
||||
with (operations_dir / "validation_results.csv").open(
|
||||
"w", encoding="utf-8", newline=""
|
||||
) as results_file:
|
||||
csv.writer(results_file).writerows((headers, *rows))
|
||||
csv.writer(results_file, lineterminator="\n").writerows((csv_headers, *csv_rows))
|
||||
print("\n" + Style.BRIGHT + Fore.CYAN + "Summary" + Style.RESET_ALL)
|
||||
print(Style.BRIGHT + f"Passed: {n_passed}" + Style.RESET_ALL)
|
||||
print(Style.BRIGHT + f"Failed: {n_total - n_passed}" + Style.RESET_ALL)
|
||||
@@ -402,19 +485,17 @@ def main():
|
||||
for result in results.values()
|
||||
if result.pimsim_status == PIMSIM_DONE
|
||||
]
|
||||
pimsim_failed = sum(
|
||||
result.pimsim_status == PIMSIM_FAILED for result in results.values()
|
||||
)
|
||||
pimsim_skipped = sum(
|
||||
result.pimsim_status in (PIMSIM_SKIPPED, PIMSIM_NOT_RUN)
|
||||
pimsim_statuses = [
|
||||
status
|
||||
for result in results.values()
|
||||
)
|
||||
pimsim_unsupported = sum(
|
||||
result.pimsim_status == PIMSIM_UNSUPPORTED for result in results.values()
|
||||
)
|
||||
for status in (result.pimsim_status, result.throughput_pimsim_status)
|
||||
]
|
||||
pimsim_failed = pimsim_statuses.count(PIMSIM_FAILED)
|
||||
pimsim_skipped = sum(status in (PIMSIM_SKIPPED, PIMSIM_NOT_RUN) for status in pimsim_statuses)
|
||||
pimsim_unsupported = pimsim_statuses.count(PIMSIM_UNSUPPORTED)
|
||||
print(
|
||||
Style.BRIGHT
|
||||
+ f"pimsim-nn: {len(measured_latencies)} measured, "
|
||||
+ f"pimsim-nn: {pimsim_statuses.count(PIMSIM_DONE)} measured, "
|
||||
f"{pimsim_failed} failed, {pimsim_unsupported} unsupported, "
|
||||
f"{pimsim_skipped} skipped"
|
||||
+ Style.RESET_ALL
|
||||
@@ -422,7 +503,7 @@ def main():
|
||||
if measured_latencies:
|
||||
print(
|
||||
Style.BRIGHT
|
||||
+ f"Total latency: {sum(measured_latencies):.6f} ms"
|
||||
+ f"Total latency: {sum(measured_latencies):.2f} ms"
|
||||
+ Style.RESET_ALL
|
||||
)
|
||||
if a.verbose:
|
||||
|
||||
Reference in New Issue
Block a user