Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

5 changes: 5 additions & 0 deletions crates/ppvm-tableau/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,10 @@ name = "ppvm-tableau"
version = "0.1.0"
edition = "2024"

[features]
default = []
rayon = ["dep:rayon"]

[dependencies]
bitvec = "1.0.1"
fxhash = "0.2.1"
Expand All @@ -11,6 +15,7 @@ itertools = "0.14.0"
num = "0.4.3"
ppvm-runtime = { version = "0.1.0", path = "../ppvm-runtime" }
rand = "0.10.0"
rayon = { version = "1.10", optional = true }

[dev-dependencies]
criterion = "0.7.0"
Expand Down
86 changes: 86 additions & 0 deletions crates/ppvm-tableau/benches/tableau.rs
Original file line number Diff line number Diff line change
Expand Up @@ -91,6 +91,92 @@ pub fn benchmark_suite_tableau(c: &mut Criterion, name: impl AsRef<str>) {
}

group.finish();

// Large-coefficient T-gate benchmarks: measures the cost of a single T gate
// applied to a pre-built state with a known number of coefficients.
// This isolates the coefficient branching hot path for rayon testing.
let mut large_group = c.benchmark_group("large-tgate");
large_group
.warm_up_time(Duration::from_secs(1))
.measurement_time(Duration::from_secs(5))
.sample_size(20);

// Use 128 qubits (u128 index) so we have enough room for many distinct T targets
type LargeTab = GeneralizedTableau<Byte8F64<2>, u128>;

for n_tgates in [12, 14, 16, 18] {
// Pre-build the state with (n_tgates - 1) T gates already applied,
// then benchmark applying the last one.
let mut setup: LargeTab = GeneralizedTableau::new(128, 1e-10);
for i in 0..n_tgates {
setup.h(i);
}
for i in 0..n_tgates - 1 {
setup.t(i);
}
let last_qubit = n_tgates - 1;
let n_coeffs = setup.coefficients.len();

large_group.bench_function(format!("single-t-on-{n_coeffs}-coeffs"), |b| {
b.iter_batched_ref(
|| setup.fork(Some(0)),
|tab| tab.t(last_qubit),
criterion::BatchSize::LargeInput,
);
});
}

large_group.finish();

// Combined benchmark: fused Clifford batches + variable T gates + measurement.
// Shows how fusion and rayon compose in a realistic circuit.
let mut combined_group = c.benchmark_group("fused-tgate-circuit");
combined_group
.warm_up_time(Duration::from_secs(1))
.measurement_time(Duration::from_secs(5))
.sample_size(20);

for n_tgates in [8, 12, 16] {
// Circuit: batched Cliffords → H (ensure branching) → T gates → batched Cliffords → measure.
// H is applied AFTER Cliffords to ensure T gates see non-stabilizer states and branch.
let n_qubits: usize = 85;

// Pre-build the Clifford portion so the setup cost is excluded
let mut setup = GeneralizedTableau::<Byte8F64<2>, u128>::new(n_qubits, 1e-10);
let block1: Vec<usize> = (0..17).collect();
let block2: Vec<usize> = (17..34).collect();
setup.sqrt_y_batch(&block1);
setup.sqrt_x_batch(&block2);
setup.cz_block_pairs(0, 17, 17);
// H on qubits that will get T gates — applied after Cliffords to guarantee branching
for i in 0..n_tgates {
setup.h(i);
}

combined_group.bench_function(format!("fused-{n_tgates}t-{n_qubits}q"), |b| {
b.iter_batched_ref(
|| setup.fork(Some(0)),
|tab| {
// T gates (coefficient branching — rayon target)
for i in 0..n_tgates {
tab.t(i);
}

// More batched Cliffords after T gates
tab.sqrt_x_adj_batch(&block1);
tab.sqrt_y_adj_batch(&block2);

// Measure all
for i in 0..n_qubits {
tab.measure(i);
}
},
criterion::BatchSize::LargeInput,
);
});
}

combined_group.finish();
}

pub fn tableau_scaling_benchmarks(c: &mut Criterion) {
Expand Down
129 changes: 129 additions & 0 deletions crates/ppvm-tableau/examples/profile_breakdown.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,129 @@
//! Profile the time breakdown inside branch_with_coefficients:
//! computation (phase + multiply) vs HashMap accumulation.

use std::time::Instant;

use fxhash::FxHashMap as HashMap;
use num::Zero;
use num::complex::{Complex, Complex64};

fn main() {
// Simulate the coefficient branching loop at various sizes
println!(
"{:>10} {:>10} {:>10} {:>10} {:>6}",
"N", "compute", "accum", "total", "%comp"
);

for n in [2048, 8192, 32768, 131072] {
// Create fake coefficient data
let items: Vec<(Complex<f64>, u128)> = (0..n)
.map(|i| (Complex::new(1.0 / n as f64, 0.0), i as u128))
.collect();

let stab_anticomm_bits: u128 = 0b1010_1010;
let destab_anticomm_bits: u128 = 0b0101_0101;
let odd_phase_mask: u128 = 0b1100_1100;
let phase_decomp: u8 = 1;
let coefficient_factor = Complex::new(0.85, 0.35);
let branch_factor = Complex::new(0.15, -0.35);

let phase_table: [Complex64; 4] = [
Complex64::new(1.0, 0.0),
Complex64::new(0.0, 1.0),
Complex64::new(-1.0, 0.0),
Complex64::new(0.0, -1.0),
];

let n_runs = 10;

// Phase 1: compute only (write to Vec, no HashMap)
let mut compute_ns = 0u128;
for _ in 0..n_runs {
let mut results = Vec::with_capacity(items.len());
let t0 = Instant::now();
for &(coeff, idx) in &items {
let branch_index = idx ^ stab_anticomm_bits;
let symplectic = (destab_anticomm_bits & idx).count_ones();
let mut phase = (2 * symplectic as u8) % 4;
let active = idx & stab_anticomm_bits;
let parity = (active & odd_phase_mask).count_ones() % 2;
phase = (phase + 2 * parity as u8) % 4;
let branch_phase = (phase + phase_decomp) % 4;
let phase_factor: Complex<f64> = phase_table[branch_phase as usize].into();
let branch_coeff = phase_factor * coeff * branch_factor;
let nonbranch_coeff = coeff * coefficient_factor;
results.push((branch_index, branch_coeff, idx, nonbranch_coeff));
}
compute_ns += t0.elapsed().as_nanos();
std::hint::black_box(&results);
}

// Phase 2: accumulate only (from pre-computed data)
let mut precomputed: Vec<(u128, Complex<f64>, u128, Complex<f64>)> =
Vec::with_capacity(items.len());
for &(coeff, idx) in &items {
let branch_index = idx ^ stab_anticomm_bits;
let symplectic = (destab_anticomm_bits & idx).count_ones();
let mut phase = (2 * symplectic as u8) % 4;
let active = idx & stab_anticomm_bits;
let parity = (active & odd_phase_mask).count_ones() % 2;
phase = (phase + 2 * parity as u8) % 4;
let branch_phase = (phase + phase_decomp) % 4;
let phase_factor: Complex<f64> = phase_table[branch_phase as usize].into();
precomputed.push((
branch_index,
phase_factor * coeff * branch_factor,
idx,
coeff * coefficient_factor,
));
}

let mut accum_ns = 0u128;
for _ in 0..n_runs {
let mut map: HashMap<u128, Complex<f64>> = HashMap::default();
map.reserve(2 * items.len());
let t0 = Instant::now();
for &(branch_idx, branch_coeff, idx, nonbranch_coeff) in &precomputed {
*map.entry(branch_idx).or_insert(Complex::zero()) += branch_coeff;
*map.entry(idx).or_insert(Complex::zero()) += nonbranch_coeff;
}
accum_ns += t0.elapsed().as_nanos();
std::hint::black_box(&map);
}

// Phase 3: combined (original pattern)
let mut total_ns = 0u128;
for _ in 0..n_runs {
let mut map: HashMap<u128, Complex<f64>> = HashMap::default();
let t0 = Instant::now();
for &(coeff, idx) in &items {
let branch_index = idx ^ stab_anticomm_bits;
let symplectic = (destab_anticomm_bits & idx).count_ones();
let mut phase = (2 * symplectic as u8) % 4;
let active = idx & stab_anticomm_bits;
let parity = (active & odd_phase_mask).count_ones() % 2;
phase = (phase + 2 * parity as u8) % 4;
let branch_phase = (phase + phase_decomp) % 4;
let phase_factor: Complex<f64> = phase_table[branch_phase as usize].into();
let branch_coeff = phase_factor * coeff * branch_factor;
let nonbranch_coeff = coeff * coefficient_factor;
*map.entry(branch_index).or_insert(Complex::zero()) += branch_coeff;
*map.entry(idx).or_insert(Complex::zero()) += nonbranch_coeff;
}
total_ns += t0.elapsed().as_nanos();
std::hint::black_box(&map);
}

let compute_us = compute_ns as f64 / n_runs as f64 / 1000.0;
let accum_us = accum_ns as f64 / n_runs as f64 / 1000.0;
let total_us = total_ns as f64 / n_runs as f64 / 1000.0;
println!(
"{:>10} {:>9.0}µ {:>9.0}µ {:>9.0}µ {:>5.1}%",
n,
compute_us,
accum_us,
total_us,
compute_us / total_us * 100.0,
);
}
}
78 changes: 78 additions & 0 deletions crates/ppvm-tableau/examples/profile_combined.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,78 @@
//! Profile the time breakdown of a fused circuit with variable T gates.

use std::time::Instant;

use ppvm_runtime::config::fx64hash::Byte8F64;
use ppvm_tableau::prelude::*;

type Tab = GeneralizedTableau<Byte8F64<2>, u128>;

fn main() {
let n_qubits = 85;

println!(
"{:>4} {:>10} {:>10} {:>10} {:>10} {:>6} {:>6} {:>6}",
"T#", "clifford", "t-gates", "measure", "total", "%clif", "%t", "%meas"
);

for n_tgates in [4, 8, 12, 14, 16] {
let block1: Vec<usize> = (0..17).collect();
let block2: Vec<usize> = (17..34).collect();

let n_runs = if n_tgates <= 12 { 5 } else { 2 };
let mut clif_ns = 0u128;
let mut tgate_ns = 0u128;
let mut meas_ns = 0u128;

for _ in 0..n_runs {
let mut tab: Tab = GeneralizedTableau::new(n_qubits, 1e-10);

// Clifford layer 1 (fused)
let t0 = Instant::now();
tab.sqrt_y_batch(&block1);
tab.sqrt_x_batch(&block2);
tab.cz_block_pairs(0, 17, 17);
for i in 0..n_tgates {
tab.h(i);
}
clif_ns += t0.elapsed().as_nanos();

// T gates
let t1 = Instant::now();
for i in 0..n_tgates {
tab.t(i);
}
tgate_ns += t1.elapsed().as_nanos();

// Clifford layer 2 (fused)
let t2 = Instant::now();
tab.sqrt_x_adj_batch(&block1);
tab.sqrt_y_adj_batch(&block2);
clif_ns += t2.elapsed().as_nanos();

// Measure
let t3 = Instant::now();
for i in 0..n_qubits {
tab.measure(i);
}
meas_ns += t3.elapsed().as_nanos();
}

let d = n_runs as f64;
let c = clif_ns as f64 / d / 1000.0;
let t = tgate_ns as f64 / d / 1000.0;
let m = meas_ns as f64 / d / 1000.0;
let total = c + t + m;
println!(
"{:>4} {:>9.0}µ {:>9.0}µ {:>9.0}µ {:>9.0}µ {:>5.1}% {:>5.1}% {:>5.1}%",
n_tgates,
c,
t,
m,
total,
c / total * 100.0,
t / total * 100.0,
m / total * 100.0,
);
}
}
Loading
Loading