use std::env::{self, Args}; use std::mem; use std::num::NonZeroU64; use std::ops::RangeBounds; use std::time::Instant; use wgpu::{BufferUsages, SubmissionIndex, include_wgsl}; use wgpu::util::DeviceExt; use wgpu::wgc::command; const PALETTE: &[u8] = include_bytes!("../palette.txt"); const PALETTE_SIZE: usize = PALETTE.len(); const BATCH_SIZE : u64 = (1 << 16) - 64; const MAX_ITER : u64 = 1 << 32; // const BATCH_SIZE : u64 = 8; // const MAX_ITER : u64 = 2; const INPUT_SIZE:u64 = 512 / 8 * 8; const INSIZE_SIZE:u64= 32 / 8; const OUTPUT_SIZE:u64= 256 / 8; const INPUT_BUF_SIZE :u64= INPUT_SIZE * BATCH_SIZE; const SIZES_BUF_SIZE :u64= INSIZE_SIZE * BATCH_SIZE; const OUTPUT_BUF_SIZE:u64= OUTPUT_SIZE * BATCH_SIZE; fn get_arg(args: &mut Args, pname: &String , name: &'static str) -> String { args.next().unwrap_or_else(|| panic!("Argument missing: {}\nUsage: {} value prev bits", name, pname)).clone() } struct NonceIter { digits: Vec, } impl NonceIter { fn new_with_capacity(capacity: usize) -> Self { Self { digits: Vec::with_capacity(capacity) } } } impl Iterator for NonceIter { type Item = NonceElement; fn next(&mut self) -> Option { let mut carry = 1; let mut cursor = 0; while carry > 0 { if let Some(d) = self.digits.get_mut(cursor) { if *d + carry >= PALETTE_SIZE { *d = (*d + carry) % PALETTE_SIZE; } else { *d += carry; carry = 0; } } else { self.digits.push(carry); carry = 0; } cursor += 1; } Some(NonceElement::from_digits(&self.digits)) } } #[derive(Debug)] struct NonceElement { pub bytes: Vec } impl NonceElement { fn from_digits(digits: &Vec) -> Self { let mut bytes = Vec::with_capacity(digits.len()); for d in digits { bytes.push(PALETTE[*d]) } Self { bytes } } fn to_hash_input(&self, value: &String, prev: &String) -> String { let mut out = String::with_capacity(self.bytes.len() + value.len() + prev.len()); out += value; out += prev; out += str::from_utf8(self.bytes.as_slice()).expect("Invalid bytes in pallette"); out } } /// Swap chain. We Have Swap Chains At Home Edition. /// When given an index, returns the first tuple entry if the index is even, and the second if it's /// odd. fn swaptuple_get(tup :&(T,T), i: u64) -> &T { if i.is_multiple_of(2){ &tup.0 } else { &tup.1 } } /// Swap chain. We Have Swap Chains At Home Edition. Mutable Edition. /// When given an index, returns the first tuple entry if the index is even, and the second if it's /// odd, but now mutable. fn swaptuple_get_mut(tup :& mut (T,T), i: u64) -> & mut T { if i.is_multiple_of(2) { &mut tup.0 } else { &mut tup.1 } } fn main() { let mut args = env::args(); let pname: String = args.next().expect("Program name should always be included."); let value: String = get_arg(&mut args, &pname, "value"); let prev: String = get_arg(&mut args, &pname, "prev"); let bits: u16 = get_arg(&mut args, &pname, "bits").parse().unwrap(); // We first initialize an wgpu `Instance`, which contains any "global" state wgpu needs. // // This is what loads the vulkan/dx12/metal/opengl libraries. env_logger::init(); let instance = wgpu::Instance::new(&wgpu::InstanceDescriptor::from_env_or_default()); let areq = wgpu::RequestAdapterOptions {power_preference: wgpu::PowerPreference::HighPerformance, ..Default::default()}; let adapter = pollster::block_on(instance.request_adapter(&areq)) .expect("Failed to create adapter"); // Print out some basic information about the adapter. println!("Running on Adapter: {:#?}", adapter.get_info()); // Check to see if the adapter supports compute shaders. While WebGPU guarantees support for // compute shaders, wgpu supports a wider range of devices through the use of "downlevel" devices. let downlevel_capabilities = adapter.get_downlevel_capabilities(); if !downlevel_capabilities .flags .contains(wgpu::DownlevelFlags::COMPUTE_SHADERS) { panic!("Adapter does not support compute shaders"); } // We then create a `Device` and a `Queue` from the `Adapter`. // // The `Device` is used to create and manage GPU resources. // The `Queue` is a queue used to submit work for the GPU to process. let (device, queue) = pollster::block_on(adapter.request_device(&wgpu::DeviceDescriptor { label: None, required_features: wgpu::Features::empty(), required_limits: wgpu::Limits::downlevel_defaults(), experimental_features: wgpu::ExperimentalFeatures::disabled(), memory_hints: wgpu::MemoryHints::MemoryUsage, trace: wgpu::Trace::Off, })) .expect("Failed to create device"); // Create a shader module from our shader code. This will parse and validate the shader. // // `include_wgsl` is a macro provided by wgpu like `include_str` which constructs a ShaderModuleDescriptor. // If you want to load shaders differently, you can construct the ShaderModuleDescriptor manually. let module = device.create_shader_module(wgpu::include_wgsl!("shader_own.wgsl")); let mut iter = NonceIter::new_with_capacity((512 - prev.len() - value.len())); // Create a buffer with the data we want to process on the GPU. // // `create_buffer_init` is a utility provided by `wgpu::util::DeviceExt` which simplifies creating // a buffer with some initial data. // // We use the `bytemuck` crate to cast the slice of f32 to a &[u8] to be uploaded to the GPU. let input_data_buffer = device.create_buffer(&wgpu::BufferDescriptor { label: Some("Input"), size: INPUT_BUF_SIZE, usage: wgpu::BufferUsages::union(BufferUsages::STORAGE, BufferUsages::COPY_DST), mapped_at_creation: false, }); let data_size_buffer = device.create_buffer(&wgpu::BufferDescriptor { label: Some("Data Size"), size: SIZES_BUF_SIZE, usage: wgpu::BufferUsages::union(BufferUsages::STORAGE, BufferUsages::COPY_DST), mapped_at_creation: false, }); // Now we create a buffer to store the output data. let output_data_buffer = device.create_buffer(&wgpu::BufferDescriptor { label: Some("Output"), size: OUTPUT_BUF_SIZE, usage: wgpu::BufferUsages::STORAGE | wgpu::BufferUsages::COPY_SRC, mapped_at_creation: false, }); // Finally we create a buffer which can be read by the CPU. This buffer is how we will read // the data. We need to use a separate buffer because we need to have a usage of `MAP_READ`, // and that usage can only be used with `COPY_DST`. let download_buffer_0 = device.create_buffer(&wgpu::BufferDescriptor { label: Some("Download_0"), size: OUTPUT_BUF_SIZE, usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ, mapped_at_creation: false, }); // We use a second buffer to flip stuff let download_buffer_1 = device.create_buffer(&wgpu::BufferDescriptor { label: Some("Download_1"), size: OUTPUT_BUF_SIZE, usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ, mapped_at_creation: false, }); let dl_bufs = (download_buffer_0, download_buffer_1); // A bind group layout describes the types of resources that a bind group can contain. Think // of this like a C-style header declaration, ensuring both the pipeline and bind group agree // on the types of resources. let bind_group_layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { label: None, entries: &[ // Input buffer wgpu::BindGroupLayoutEntry { binding: 0, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Storage { read_only: true }, // This is the size of a single element in the buffer. min_binding_size: Some(NonZeroU64::new(512).unwrap()), has_dynamic_offset: false, }, count: None, }, wgpu::BindGroupLayoutEntry { binding: 1, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Storage { read_only: true }, // This is the size of a single element in the buffer. min_binding_size: Some(NonZeroU64::new(4).unwrap()), has_dynamic_offset: false, }, count: None, }, // Output buffer wgpu::BindGroupLayoutEntry { binding: 2, visibility: wgpu::ShaderStages::COMPUTE, ty: wgpu::BindingType::Buffer { ty: wgpu::BufferBindingType::Storage { read_only: false }, // This is the size of a single element in the buffer. min_binding_size: Some(NonZeroU64::new(32).unwrap()), has_dynamic_offset: false, }, count: None, }, ], }); // The bind group contains the actual resources to bind to the pipeline. // // Even when the buffers are individually dropped, wgpu will keep the bind group and buffers // alive until the bind group itself is dropped. let bind_group = device.create_bind_group(&wgpu::BindGroupDescriptor { label: None, layout: &bind_group_layout, entries: &[ wgpu::BindGroupEntry { binding: 0, resource: input_data_buffer.as_entire_binding(), }, wgpu::BindGroupEntry { binding: 1, resource: data_size_buffer.as_entire_binding(), }, wgpu::BindGroupEntry { binding: 2, resource: output_data_buffer.as_entire_binding(), }, ], }); // The pipeline layout describes the bind groups that a pipeline expects let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { label: None, bind_group_layouts: &[&bind_group_layout], immediate_size: 0, }); // The pipeline is the ready-to-go program state for the GPU. It contains the shader modules, // the interfaces (bind group layouts) and the shader entry point. let pipeline = device.create_compute_pipeline(&wgpu::ComputePipelineDescriptor { label: None, layout: Some(&pipeline_layout), module: &module, entry_point: Some("main"), compilation_options: wgpu::PipelineCompilationOptions::default(), cache: None, }); let mut nibbles: [u16; OUTPUT_SIZE as usize/ 8] = [0;4]; let mut b = bits as u16; for n in 0..4 { if b > 64 { nibbles[n] = 64; b -= 64; } else { nibbles[n] = b; break; } } // == Begin repeating part let mut found: Option<[u8;32]> = None; let mut found_input: Option = None; let mut nonces_1: Vec= Vec::with_capacity(BATCH_SIZE as usize); let mut nonces_2: Vec= Vec::with_capacity(BATCH_SIZE as usize); let mut nonce_bufs = (nonces_1, nonces_2); let mut prev_time: Instant = Instant::now(); let mut prev_submission_index = None; println!("starting!"); for x in 0 .. MAX_ITER { let data_upload_buffer = device.create_buffer(&wgpu::BufferDescriptor { label: Some("Upload Data"), size: INPUT_BUF_SIZE, usage: wgpu::BufferUsages::COPY_SRC | wgpu::BufferUsages::MAP_WRITE, mapped_at_creation: true, }); let size_upload_buffer = device.create_buffer(&wgpu::BufferDescriptor { label: Some("Upload Size"), size: SIZES_BUF_SIZE, usage: wgpu::BufferUsages::COPY_SRC | wgpu::BufferUsages::MAP_WRITE, mapped_at_creation: true, }); let mut input_slice = data_upload_buffer.get_mapped_range_mut(..); let mut len_slice = size_upload_buffer.get_mapped_range_mut(..); let nonces = swaptuple_get_mut(&mut nonce_bufs, x); for i in 0 .. BATCH_SIZE { if let Some(element) = iter.next() { let inputdata = element.to_hash_input(&value, &prev).into_bytes(); let inputlen = inputdata.len() as u32; let lenbytes = inputlen.to_le_bytes(); // println!("{}: {}, {}", i, inputlen*8, str::from_utf8(&inputdata).unwrap()); input_slice[(i * INPUT_SIZE) as usize .. (i*INPUT_SIZE + inputlen as u64) as usize].copy_from_slice(&inputdata); len_slice[(i * INSIZE_SIZE) as usize .. ((i+1) * INSIZE_SIZE) as usize].copy_from_slice(&lenbytes); // println!("In data {:?}", &input_slice[(i * INPUT_SIZE) as usize .. (i*INPUT_SIZE + inputlen as u64) as usize]); // println!("In length {:?}", &len_slice[(i * INSIZE_SIZE) as usize .. ((i+1) * INSIZE_SIZE) as usize]); nonces.push(element); } else { panic!("Infinite iter died") } } // println!("Hex slice:\n{}", hex::encode(&input_slice[..])); drop(input_slice); // The command encoder allows us to record commands that we will later submit to the GPU. drop(len_slice); data_upload_buffer.unmap(); size_upload_buffer.unmap(); let mut encoder = device.create_command_encoder(&wgpu::CommandEncoderDescriptor { label: None }); // We add a copy operation to the encoder. This will copy the data from the output buffer on the // GPU to the download buffer on the CPU. encoder.copy_buffer_to_buffer( &data_upload_buffer, 0, &input_data_buffer, 0, data_upload_buffer.size(), ); encoder.copy_buffer_to_buffer( &size_upload_buffer, 0, &data_size_buffer, 0, size_upload_buffer.size(), ); // A compute pass is a single series of compute operations. While we are recording a compute // pass, we cannot record to the encoder. let mut compute_pass = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor { label: None, timestamp_writes: None, }); // Set the pipeline that we want to use compute_pass.set_pipeline(&pipeline); // Set the bind group that we want to use compute_pass.set_bind_group(0, &bind_group, &[]); // Now we dispatch a series of workgroups. Each workgroup is a 3D grid of individual programs. // // We defined the workgroup size in the shader as 64x1x1. So in order to process all of our // inputs, we ceiling divide the number of inputs by 64. If the user passes 32 inputs, we will // dispatch 1 workgroups. If the user passes 65 inputs, we will dispatch 2 workgroups, etc. let rest = BATCH_SIZE % 64; let mut workgroup_count = BATCH_SIZE - rest / 64; if rest > 0 { workgroup_count += 1; } compute_pass.dispatch_workgroups(workgroup_count as u32, 1, 1); // Now we drop the compute pass, giving us access to the encoder again. drop(compute_pass); // We add a copy operation to the encoder. This will copy the data from the output buffer on the // GPU to the download buffer on the CPU. encoder.copy_buffer_to_buffer( &output_data_buffer, 0, swaptuple_get(&dl_bufs, x), 0, output_data_buffer.size(), ); // We finish the encoder, giving us a fully recorded command buffer. let command_buffer = encoder.finish(); let sub_index = queue.submit([command_buffer]); // println!("A pass has been submitted"); // Wait for the GPU to finish working on the submitted work. This doesn't work on WebGPU, so we would need // to rely on the callback to know when the buffer is mapped. if x > 0 { let nonces = swaptuple_get_mut(&mut nonce_bufs, x -1); let download_buffer = swaptuple_get(&dl_bufs, x-1); let buffer_slice = download_buffer.slice(..); buffer_slice.map_async(wgpu::MapMode::Read, |_| {}); device.poll(wgpu::PollType::Wait { submission_index: prev_submission_index, timeout: None }).unwrap(); // In this case we know exactly when the mapping will be finished, // so we don't need to do anything in the callback. let data = buffer_slice.get_mapped_range(); // println!("Out data {:?}", &data[..]); // println!("Out hash {:?}", hex::encode(&data[..])); let results: &[u8] = bytemuck::cast_slice(&data[..]); // println!("Full buffer: {}", hex::encode(results) ); for i in 0..(BATCH_SIZE as usize) { let result: &[u8;32] = &results[32*i..32*(i+1)].try_into().unwrap(); // println!("Hash of {}, {}", hex::encode(nonces.get(i).unwrap().bytes.clone()), hex::encode(bytemuck::cast_slice(result))); let mut correct = true; for n in 0..4 { if u64::from_be_bytes(result[n*8..(n+1)*8].try_into().unwrap()).leading_zeros() < nibbles[n] as u32{ correct = false; break; } } if correct { found = Some(*result); let nonce = nonces.get(i).expect("We pushed all those nonces, right?").bytes.clone(); found_input = Some(String::from_utf8(nonce).unwrap()); // break; } } if found.is_some() { break; } drop(data); download_buffer.unmap(); nonces.clear(); let now = Instant::now(); let delta = (now - prev_time).as_secs_f64(); let hashrate = BATCH_SIZE as f64 / delta; println!("\x1b[A\rFinished batch {}, {} hashes processed ({:.2} H/s) (delta: {}) ", x + 1, x * BATCH_SIZE, hashrate, delta); prev_time = Instant::now(); } prev_submission_index = Some(sub_index); } match found { Some(f) => { let hashbytes: &[u8] = bytemuck::cast_slice(&f); let nonce = found_input.unwrap(); println!("Found! {:?}", hex::encode(hashbytes)); println!("Full block: {}{}{}", value, prev , nonce); println!("Nonce: \"{}\"", nonce); } None => { println!("Could not find a valid hash") } } println!("Done for now") }