Skew and insertion delay
You will learn
Why the clock does not reach every flop at once, and what the tree delays of a 7-series chip cost.
The clock does not reach every flop at once: it arrives through a tree, and the spread of arrival times is skew. cts.t27 models that tree with the teaching delays of an Artix-7 in its header: a BUFH at 0.05 ns, a BUFG at 0.1 ns, PLL jitter of 50 ps, and a maximum skew target of 100 ps. The recording runs the native t27c on it: 15 tests pass, 2 invariants are proved comptime. Skew is not noise to remove but a budget to spend: a well-balanced tree spends it evenly, which is what makes one shared edge possible at all.
Try it
In the recording, find the two buffer delays in the header; then in the spec frame find the maximum skew target and the test that compares buffer fanout.

t27c 0.4.0 on a laptop (macOS): 15 tests of the clock-tree spec pass natively, 2 invariants comptime; the header holds the Artix-7 teaching delays (BUFG 0.1 ns, BUFH 0.05 ns, PLL jitter 50 ps).
specs/fpga/cts.t27
// SPDX-License-Identifier: Apache-2.0
// t27/specs/fpga/cts.t27
// T27 Clock Tree Synthesis Specification
// PLL configuration, clock buffer trees, skew estimation
// Artix-7: BUFH=0.05ns, BUFG=0.1ns, PLL jitter=50ps, max skew=100ps
// Uses flat arrays + count fields (parser-compatible)
// phi^2 + 1/phi^2 = 3 | TRINITY
module CTS {
pub struct PllConfig {
name : &str,
input_mhz : u32,
output_mhz : u32,
multiply : u32,
divide : u32,
jitter_ps : u32,
}
fn pll_config(name: &str, input_mhz: u32, output_mhz: u32) -> PllConfig {
var m : u32 = 1;
var d : u32 = 1;
if input_mhz > 0 {
d = input_mhz;
m = output_mhz;
}
return PllConfig{
.name = name,
.input_mhz = input_mhz,
.output_mhz = output_mhz,
.multiply = m,
.divide = d,
.jitter_ps = 50,
};
}
fn pll_period_ps(pll: PllConfig) -> u32 {
if pll.output_mhz == 0 {
return 0;
}
return 1000000000 / pll.output_mhz;
}
pub struct ClockBuffer {
name : &str,
delay_ps : u32,
fanout : u32,
}
fn bufg(name: &str) -> ClockBuffer {
return ClockBuffer{ .name = name, .delay_ps = 100, .fanout = 32 };
}
fn bufh(name: &str) -> ClockBuffer {
return ClockBuffer{ .name = name, .delay_ps = 50, .fanout = 16 };
}
fn bufg_has_higher_fanout(b: ClockBuffer) -> bool {
return b.fanout >= 32;
}
pub struct ClockTree {
root : &str,
num_levels : u32,
total_buffers : u32,
max_skew_ps : u32,
}
fn clock_tree(root: &str, levels: u32, bufs: u32) -> ClockTree {
return ClockTree{
.root = root,
.num_levels = levels,
.total_buffers = bufs,
.max_skew_ps = 100,
};
}
fn tree_delay_ps(tree: ClockTree, buf_delay: u32) -> u32 {
return tree.num_levels * buf_delay;
}
fn skew_ok(tree: ClockTree, max_allowed_ps: u32) -> bool {
return tree.max_skew_ps <= max_allowed_ps;
}
pub struct CtsReport {
num_clocks : u32,
num_plls : u32,
total_buffers : u32,
worst_skew_ps : u32,
worst_latency_ps : u32,
has_violations : bool,
}
fn cts_ok(clocks: u32, plls: u32, bufs: u32, skew: u32, latency: u32) -> CtsReport {
return CtsReport{
.num_clocks = clocks,
.num_plls = plls,
.total_buffers = bufs,
.worst_skew_ps = skew,
.worst_latency_ps = latency,
.has_violations = false,
};
}
fn passed(r: CtsReport) -> bool {
return r.has_violations == false;
}
// === Auto tree estimation ===
fn est_buffers_needed(num_sinks: u32) -> u32 {
if num_sinks <= 16 {
return 1;
}
return num_sinks / 16 + 1;
}
fn est_tree_levels(num_sinks: u32) -> u32 {
if num_sinks <= 16 {
return 1;
}
if num_sinks <= 256 {
return 2;
}
return 3;
}
// === Validation ===
fn validate_pll(pll: PllConfig) -> u32 {
var errors : u32 = 0;
if pll.name == "" { errors = errors + 1; }
if pll.output_mhz == 0 { errors = errors + 1; }
return errors;
}
// === Tests ===
test pll_config_creation
given p = pll_config("sys_pll", 100, 200)
then p.input_mhz == 100
and p.output_mhz == 200
and pll_period_ps(p) == 5000000
test bufg_creation
given b = bufg("clk_buf")
then b.delay_ps == 100
and b.fanout == 32
and bufg_has_higher_fanout(b) == true
test bufh_creation
given b = bufh("clk_h")
then b.delay_ps == 50
and b.fanout == 16
and bufg_has_higher_fanout(b) == false
test clock_tree_creation
given t = clock_tree("clk", 2, 5)
then t.root == "clk"
and t.num_levels == 2
and t.total_buffers == 5
and t.max_skew_ps == 100
test tree_delay
given t = clock_tree("clk", 3, 8)
then tree_delay_ps(t, 100) == 300
test skew_ok_yes
given t = clock_tree("clk", 2, 5)
then skew_ok(t, 200) == true
test skew_ok_no
given t = clock_tree("clk", 2, 5)
then skew_ok(t, 50) == false
test cts_report_ok
given r = cts_ok(2, 1, 10, 80, 300)
then passed(r) == true
and r.has_violations == false
test est_buffers_one
then est_buffers_needed(10) == 1
test est_buffers_many
then est_buffers_needed(100) == 7
test est_tree_levels_one
then est_tree_levels(10) == 1
test est_tree_levels_two
then est_tree_levels(100) == 2
test est_tree_levels_three
then est_tree_levels(500) == 3
test validate_pll_ok
given p = pll_config("ok", 100, 200)
then validate_pll(p) == 0
test validate_pll_empty
given p = PllConfig{.name = "", .input_mhz = 100, .output_mhz = 0, .multiply = 1, .divide = 1, .jitter_ps = 50}
then validate_pll(p) > 0
// === Invariants ===
invariant bufg_delay_positive
given b = bufg("inv")
assert b.delay_ps > 0
invariant skew_non_negative
given t = clock_tree("inv", 2, 5)
assert t.max_skew_ps >= 0
bench buffer_estimation_latency
measure: nanoseconds to est_buffers_needed(50)
target: < 50ns
bench tree_level_estimation_latency
measure: nanoseconds to est_tree_levels(200)
target: < 50ns
}
// phi^2 + 1/phi^2 = 3 | TRINITY
All lessons
Module 1 · What a clock is
One edge, one world: what shares a clock edge shares a world, the period and the jitter of a real edge, and where the clock enters a board.
Module 2 · Clock trees
Skew and insertion delay, the global buffer network, and the trap of gating a clock with logic.
Module 3 · PLL and MMCM
Multiply and divide one clock into another, move its phase in steps of the VCO, and which clocks the analyzer treats as related.
Module 4 · Resets
Assert asynchronously, release synchronously: the three reset kinds, the release pipe, and the tree a reset grows.
Module 5 · Metastability
The setup-hold window, the mean time between failures in integer arithmetic, and the two flops that fix it.
Module 6 · Crossing many bits
Why a binary bus tears, why Gray code does not, and the handshake that moves a pulse between worlds.
Module 7 · The asynchronous FIFO
Pointers, flags and depth: the buffer that moves a stream between two clocks.
Module 8 · Constraints
The lines that tell the analyzer what a clock is, which paths not to check, and what the pins must meet.
Module 9 · On the board
A CDC report, one crossing captured at the flip-flops, and the bitstream diff that closes the course.