From a01eb2353803c1a14337ccf689f94b5b9c35f5a4 Mon Sep 17 00:00:00 2001 From: hachem Date: Wed, 10 Dec 2025 06:59:01 +0100 Subject: [add]: Kernel+Kernel Batching --- libpsi-core/src/core/kernel.rs | 222 ++++++++++++++++++++++++++++++++++++++++ libpsi-core/src/core/mod.rs | 2 + libpsi-core/src/core/runtime.rs | 74 +++++++++++++- libpsi-core/src/lib.rs | 1 + 4 files changed, 297 insertions(+), 2 deletions(-) create mode 100644 libpsi-core/src/core/kernel.rs (limited to 'libpsi-core') diff --git a/libpsi-core/src/core/kernel.rs b/libpsi-core/src/core/kernel.rs new file mode 100644 index 0000000..3d75425 --- /dev/null +++ b/libpsi-core/src/core/kernel.rs @@ -0,0 +1,222 @@ +use crate::{complex, Complex, Matrix}; +use rayon::prelude::*; + +#[derive(Clone)] +pub struct Kernel { + pub matrix: Matrix>, + pub targets: Vec, + pub name: String, +} + +impl Kernel { + pub fn new(name: &str, matrix: Matrix>, targets: Vec) -> Self { + Self { + matrix, + targets, + name: name.to_string(), + } + } + + pub fn num_qubits(&self) -> usize { + self.targets.len() + } + + pub fn can_fuse_with(&self, other: &Kernel) -> bool { + if self.targets.len() != 1 || other.targets.len() != 1 { + return false; + } + self.targets[0] == other.targets[0] + } + + pub fn fuse(&self, other: &Kernel) -> Option { + if !self.can_fuse_with(other) { + return None; + } + let fused_matrix = other.matrix.dot(&self.matrix)?; + Some(Kernel { + matrix: fused_matrix, + targets: self.targets.clone(), + name: format!("{}+{}", self.name, other.name), + }) + } +} + +pub struct KernelBatch { + kernels: Vec, + num_qubits: usize, +} + +impl KernelBatch { + pub fn new(num_qubits: usize) -> Self { + Self { + kernels: Vec::new(), + num_qubits, + } + } + + pub fn add(&mut self, kernel: Kernel) { + self.kernels.push(kernel); + } + + pub fn len(&self) -> usize { + self.kernels.len() + } + + pub fn is_empty(&self) -> bool { + self.kernels.is_empty() + } + + pub fn kernels(&self) -> &[Kernel] { + &self.kernels + } + + pub fn optimize(&mut self) { + if self.kernels.len() < 2 { + return; + } + + let mut optimized: Vec = Vec::with_capacity(self.kernels.len()); + let mut i = 0; + + while i < self.kernels.len() { + let current = &self.kernels[i]; + + if i + 1 < self.kernels.len() { + let next = &self.kernels[i + 1]; + if let Some(fused) = current.fuse(next) { + optimized.push(fused); + i += 2; + continue; + } + } + + optimized.push(current.clone()); + i += 1; + } + + self.kernels = optimized; + } + + pub fn execute(&self, state: &mut Vec>) { + for kernel in &self.kernels { + *state = apply_kernel(state, kernel, self.num_qubits); + } + } + + pub fn execute_parallel(&self, state: &mut Vec>) { + for kernel in &self.kernels { + *state = apply_kernel_parallel(state, kernel, self.num_qubits); + } + } +} + +fn apply_kernel(state: &[Complex], kernel: &Kernel, num_qubits: usize) -> Vec> { + let dim = 1 << num_qubits; + let g = kernel.targets.len(); + let gate_dim = 1 << g; + + let target_bits: Vec = kernel.targets.iter().map(|&t| num_qubits - 1 - t).collect(); + + let mut non_target_mask: usize = (1 << num_qubits) - 1; + for &pos in &target_bits { + non_target_mask &= !(1 << pos); + } + + let mut new_state = vec![complex!(0.0, 0.0); dim]; + + for i in 0..dim { + let mut target_idx = 0usize; + for (k, &pos) in target_bits.iter().enumerate() { + if (i >> pos) & 1 == 1 { + target_idx |= 1 << (g - 1 - k); + } + } + + let mut sum = complex!(0.0, 0.0); + + for j in 0..gate_dim { + let gate_elem = kernel.matrix.data[target_idx * gate_dim + j]; + + if gate_elem.real.abs() < 1e-15 && gate_elem.imaginary.abs() < 1e-15 { + continue; + } + + let mut source_idx = i & non_target_mask; + for (k, &pos) in target_bits.iter().enumerate() { + if (j >> (g - 1 - k)) & 1 == 1 { + source_idx |= 1 << pos; + } + } + + sum = sum + gate_elem * state[source_idx]; + } + + new_state[i] = sum; + } + + new_state +} + +fn apply_kernel_parallel( + state: &[Complex], + kernel: &Kernel, + num_qubits: usize, +) -> Vec> { + let dim = 1 << num_qubits; + let g = kernel.targets.len(); + let gate_dim = 1 << g; + + let target_bits: Vec = kernel.targets.iter().map(|&t| num_qubits - 1 - t).collect(); + + let mut non_target_mask: usize = (1 << num_qubits) - 1; + for &pos in &target_bits { + non_target_mask &= !(1 << pos); + } + + (0..dim) + .into_par_iter() + .map(|i| { + let mut target_idx = 0usize; + for (k, &pos) in target_bits.iter().enumerate() { + if (i >> pos) & 1 == 1 { + target_idx |= 1 << (g - 1 - k); + } + } + + let mut sum = complex!(0.0, 0.0); + + for j in 0..gate_dim { + let gate_elem = kernel.matrix.data[target_idx * gate_dim + j]; + + if gate_elem.real.abs() < 1e-15 && gate_elem.imaginary.abs() < 1e-15 { + continue; + } + + let mut source_idx = i & non_target_mask; + for (k, &pos) in target_bits.iter().enumerate() { + if (j >> (g - 1 - k)) & 1 == 1 { + source_idx |= 1 << pos; + } + } + + sum = sum + gate_elem * state[source_idx]; + } + + sum + }) + .collect() +} + +pub struct KernelBuilder { + num_qubits: usize, +} + +impl KernelBuilder { + pub fn new(num_qubits: usize) -> Self { + Self { num_qubits } + } + + pub fn num_qubits(&self) -> usize { + self.num_qubits + } +} diff --git a/libpsi-core/src/core/mod.rs b/libpsi-core/src/core/mod.rs index 697b0d7..343e319 100644 --- a/libpsi-core/src/core/mod.rs +++ b/libpsi-core/src/core/mod.rs @@ -2,6 +2,7 @@ pub mod circuit; pub mod classical_components; pub mod custom_gate; pub mod gates; +pub mod kernel; pub mod quantum_components; pub mod runtime; @@ -9,5 +10,6 @@ pub use circuit::*; pub use classical_components::*; pub use custom_gate::*; pub use gates::*; +pub use kernel::*; pub use quantum_components::*; pub use runtime::*; diff --git a/libpsi-core/src/core/runtime.rs b/libpsi-core/src/core/runtime.rs index 0a78f0f..69b9d01 100644 --- a/libpsi-core/src/core/runtime.rs +++ b/libpsi-core/src/core/runtime.rs @@ -1,4 +1,4 @@ -use super::{GateOp, QuantumGate, QuantumRegister, QuantumState}; +use super::{GateOp, Kernel, KernelBatch, QuantumGate, QuantumRegister, QuantumState}; use crate::gates::{ cp_matrix, crx_matrix, cry_matrix, crz_matrix, p_matrix, rx_matrix, ry_matrix, rz_matrix, u1_matrix, u2_matrix, u3_matrix, CNOT, CZ, FREDKIN, HADAMARD, @@ -9,7 +9,6 @@ use crate::maths::vector::Vector; use crate::{complex, Complex, Matrix}; use rayon::prelude::*; -/// Minimum number of qubits to enable parallelism (2^8 = 256 state vector elements) const PARALLEL_THRESHOLD: usize = 8; #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] @@ -17,6 +16,8 @@ pub enum Runtime { #[default] BasicRT, BasicRTMT, + BatchedRT, + BatchedRTMT, WFEvolution, WFEvolutionMT, GPUAccelerated, @@ -27,6 +28,8 @@ impl Runtime { match self { Runtime::BasicRT => Self::compute_basic(num_qubits, operations), Runtime::BasicRTMT => Self::compute_basic_mt(num_qubits, operations), + Runtime::BatchedRT => Self::compute_batched(num_qubits, operations, false), + Runtime::BatchedRTMT => Self::compute_batched(num_qubits, operations, true), Runtime::WFEvolution => { unimplemented!("WFEvolution (Schrödinger equation) runtime not yet implemented") } @@ -41,6 +44,73 @@ impl Runtime { } } + pub fn build_kernel_batch(num_qubits: usize, operations: &[GateOp]) -> KernelBatch { + let mut batch = KernelBatch::new(num_qubits); + + for op in operations { + if let Some(kernel) = Self::op_to_kernel(op) { + batch.add(kernel); + } + } + + batch + } + + fn op_to_kernel(op: &GateOp) -> Option { + let (matrix, targets, name): (Matrix>, Vec, &str) = match op { + GateOp::H(t) => (HADAMARD.matrix.clone(), vec![*t], "H"), + GateOp::X(t) => (PAULI_X.matrix.clone(), vec![*t], "X"), + GateOp::Y(t) => (PAULI_Y.matrix.clone(), vec![*t], "Y"), + GateOp::Z(t) => (PAULI_Z.matrix.clone(), vec![*t], "Z"), + GateOp::S(t) => (S_GATE.matrix.clone(), vec![*t], "S"), + GateOp::T(t) => (T_GATE.matrix.clone(), vec![*t], "T"), + GateOp::Sdg(t) => (SDG_GATE.matrix.clone(), vec![*t], "Sdg"), + GateOp::Tdg(t) => (TDG_GATE.matrix.clone(), vec![*t], "Tdg"), + GateOp::Sx(t) => (SX_GATE.matrix.clone(), vec![*t], "Sx"), + GateOp::Sxdg(t) => (SXDG_GATE.matrix.clone(), vec![*t], "Sxdg"), + GateOp::Rx(t, theta) => (rx_matrix(*theta), vec![*t], "Rx"), + GateOp::Ry(t, theta) => (ry_matrix(*theta), vec![*t], "Ry"), + GateOp::Rz(t, theta) => (rz_matrix(*theta), vec![*t], "Rz"), + GateOp::P(t, theta) => (p_matrix(*theta), vec![*t], "P"), + GateOp::U1(t, lambda) => (u1_matrix(*lambda), vec![*t], "U1"), + GateOp::U2(t, phi, lambda) => (u2_matrix(*phi, *lambda), vec![*t], "U2"), + GateOp::U3(t, theta, phi, lambda) => (u3_matrix(*theta, *phi, *lambda), vec![*t], "U3"), + GateOp::CNOT(c, t) => (CNOT.matrix.clone(), vec![*c, *t], "CNOT"), + GateOp::CZ(c, t) => (CZ.matrix.clone(), vec![*c, *t], "CZ"), + GateOp::SWAP(a, b) => (SWAP.matrix.clone(), vec![*a, *b], "SWAP"), + GateOp::CRx(c, t, theta) => (crx_matrix(*theta), vec![*c, *t], "CRx"), + GateOp::CRy(c, t, theta) => (cry_matrix(*theta), vec![*c, *t], "CRy"), + GateOp::CRz(c, t, theta) => (crz_matrix(*theta), vec![*c, *t], "CRz"), + GateOp::CP(c, t, theta) => (cp_matrix(*theta), vec![*c, *t], "CP"), + GateOp::CCNOT(c1, c2, t) => (TOFFOLI.matrix.clone(), vec![*c1, *c2, *t], "CCNOT"), + GateOp::CSWAP(c, t1, t2) => (FREDKIN.matrix.clone(), vec![*c, *t1, *t2], "CSWAP"), + GateOp::Measure(_, _) => return None, + GateOp::Custom(gate, tgts) => { + let qg = gate.to_quantum_gate(); + (qg.matrix, tgts.clone(), "Custom") + } + }; + + Some(Kernel::new(name, matrix, targets)) + } + + fn compute_batched(num_qubits: usize, operations: &[GateOp], parallel: bool) -> QuantumState { + let dim = 1 << num_qubits; + let mut state: Vec> = vec![complex!(0.0, 0.0); dim]; + state[0] = complex!(1.0, 0.0); + + let mut batch = Self::build_kernel_batch(num_qubits, operations); + batch.optimize(); + + if parallel && num_qubits >= PARALLEL_THRESHOLD { + batch.execute_parallel(&mut state); + } else { + batch.execute(&mut state); + } + + QuantumState::new(state) + } + fn compute_basic(num_qubits: usize, operations: &[GateOp]) -> QuantumState { let names: Vec = (0..num_qubits).map(|i| format!("q{}", i)).collect(); let leaked_names: &'static [String] = Box::leak(names.into_boxed_slice()); diff --git a/libpsi-core/src/lib.rs b/libpsi-core/src/lib.rs index 7b99170..ff3253c 100644 --- a/libpsi-core/src/lib.rs +++ b/libpsi-core/src/lib.rs @@ -11,5 +11,6 @@ pub use core::circuit::*; pub use core::classical_components::*; pub use core::custom_gate::*; pub use core::gates; +pub use core::kernel::*; pub use core::quantum_components::*; pub use core::runtime::*; -- cgit v1.3