mirror of
https://github.com/logos-blockchain/research.git
synced 2026-08-07 19:53:10 +00:00
Acts on a correctness/completeness review of the countable uncle model and its report material. Correctness fixes in the report: - s3.4 quoted 0.998 for W_abs=10 at the 8s budget; the run says 0.9963. - s1 claimed both models >= 0.996 at U >= 1; countable U=2 delta=8 is 0.9955. Corrected to >= 0.995. - The s3.2 table presented two cells (U=1 at delta 16 and 32) as model differences. They are not resolvable: t = 0.46 and 0.47 over 5 replicates. The table now carries +-SEM and a t per cell. - s3.4 claimed the ~7-block-interval floor "carries over unchanged". Accuracy is still climbing past W=7 at every delay (8s: 0.989 -> 0.996), so the claim is dropped. The 32s curve is non-monotonic with replicate SD up to 0.22 and is now flagged as noise, not a trend. - 1-r was attributed to the first-fork restriction alone; it is the combined first-fork and capacity loss, which this measurement cannot separate. Hedged to match fig32's own axis label. Completeness: the U=0 negative control was swept but never reported. With no uncles the two models are identical by construction, yet they differ by -0.23 at delta_max=32 (t=2.1) because they draw independent RNG streams. That is the noise floor the rest of the grid must clear, and it is now in s3.2, s9, fig30 and the config header. New study (configs/fine-delay.yaml, scripts/plot_fine_delay.py, s3.2a, fig34/fig35): the design band delta_max 1-5 at 40 replicates, both models. Findings: every U >= 1 cell of both models lands in 0.998-1.001, flat in delay, while U=0 decays 0.810 -> 0.640. No individual cell resolves a model difference (widest 95% CI +-0.15pp; max t=2.59 vs Bonferroni 2.94 over 15 cells). Pooled across uncle caps the first-fork cost is monotone in delay and separates from zero only at delta_max=5 (-0.0014 +- 0.0007, t=3.7) -- below 0.15% everywhere in the band, against +-0.9% per-epoch sampling noise. Code: - deep_ref_share is identically 0 on every real countable run: for a chain block B the producer's chain below B is the counting chain below B, so the counting-side parent-on-chain re-check cannot reject what selection emitted. It is a drift alarm, not a rate. Documented as such in measure.py, the plot docstring and the config header, and pinned by a new end-to-end test. - Removed annotate_uncles: a second countable implementation that production never called, while carrying most of the selection test coverage. Tests now drive select_uncles_at_production through an annotate_via_production replay helper -- same assertions, live path. - Added tests for the two previously uncovered branches of the live selection: the pmin/below chain walk that resolves parent-on-chain for candidates whose parent sits below the window, and the occupied-slot exclusion built from the chain walk. - theory.q_effective and theory.window_miss_prob were unused and untested. Now used (the prediction figure reconstructs q_u through the identity the report quotes) and tested. The window_miss_prob test records that its "~ e^-W" docstring is the f->0 limit: the true decay is e^-1.017W at f=1/30, 16% off by W=10. - Shared sem()/recovery_rate() moved into figures_pernode.py; fig30 and fig33 regenerated with SEM error bars and the U=0 control curve. - Fixed the pre-existing E501 in bootstrap_dynamics.py; ruff clean. Report prose reworked to read standalone: the countable model is described as the rules under analysis and the former model as a labelled "unrestricted" comparison baseline, with no dated banners and no round-to-round narration. Tests: 209 passed (was 202). Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
84 lines
3.3 KiB
Python
84 lines
3.3 KiB
Python
import numpy as np
|
|
|
|
from tsi_sim import theory
|
|
|
|
F = 1 / 30
|
|
T = 10000
|
|
|
|
|
|
def test_expected_ratio_unbiased_at_q1():
|
|
assert abs(float(theory.expected_ratio(F, 1.0)) - 1.0) < 1e-12
|
|
|
|
|
|
def test_expected_ratio_monotone_in_q():
|
|
qs = np.linspace(0.5, 1.0, 20)
|
|
er = theory.expected_ratio(F, qs)
|
|
assert np.all(np.diff(er) > 0) # accuracy improves as q -> 1
|
|
assert np.all(er <= 1.0 + 1e-12) # always an underestimate
|
|
|
|
|
|
def test_variance_bound_matches_at_q1():
|
|
v = float(theory.variance_ratio(F, 1.0, T))
|
|
assert abs(v - theory.variance_bound(F, T)) < 1e-15
|
|
|
|
|
|
def test_optimal_beta_is_half_stability_bound():
|
|
for q in (0.7, 0.85, 0.95):
|
|
opt = float(theory.optimal_beta(F, q))
|
|
bound = float(theory.beta_stability_bound(F, q))
|
|
assert abs(opt - bound / 2) < 1e-12
|
|
|
|
|
|
def test_block_count_ceiling_above_one():
|
|
c = theory.block_count_ceiling(F)
|
|
assert 1.015 < c < 1.02 # -ln(1-1/30)/(1/30) ~ 1.01705
|
|
|
|
|
|
def test_fixed_point_bias_about_one_percent():
|
|
b = theory.fixed_point_bias(F)
|
|
assert abs(b - (F / (33 / 1000))) < 1e-12
|
|
assert 1.005 < b < 1.02
|
|
|
|
|
|
def test_q_effective_interpolates_between_q_and_one():
|
|
# r = 0 -> no recovery (chain-only q); r = 1 -> every wasted slot recovered (q_u = 1).
|
|
for q in (0.3, 0.65, 0.9):
|
|
assert abs(float(theory.q_effective(q, 0.0)) - q) < 1e-15
|
|
assert abs(float(theory.q_effective(q, 1.0)) - 1.0) < 1e-15
|
|
# monotone and strictly between for partial recovery
|
|
half = float(theory.q_effective(q, 0.5))
|
|
assert q < half < 1.0
|
|
|
|
|
|
def test_q_effective_round_trips_the_measured_recovery_rate():
|
|
# The identity the report quotes: given measured q and q_u, r = (q_u - q)/(1 - q)
|
|
# reconstructs q_u exactly. This is how plot_countable_vs_old.py derives its overlay.
|
|
for q, q_u in ((0.3129, 0.6147), (0.5380, 0.9874), (0.7485, 0.9992)):
|
|
r = (q_u - q) / (1.0 - q)
|
|
assert abs(float(theory.q_effective(q, r)) - q_u) < 1e-12
|
|
|
|
|
|
def test_q_effective_recovers_unbiased_equilibrium():
|
|
# Full recovery must lift the equilibrium to exactly 1 for any starting q.
|
|
for q in (0.3, 0.65, 0.9):
|
|
assert abs(float(theory.expected_ratio(F, theory.q_effective(q, 1.0))) - 1.0) < 1e-12
|
|
|
|
|
|
def test_window_miss_prob_decays_as_exp_minus_w():
|
|
# P(no canonical block in w_u = W/f slots) = (1-f)^(W/f) = exp(W * ln(1-f)/f).
|
|
# The docstring's "~ e^-W" is the f -> 0 limit: ln(1-f)/f = -(1 + f/2 + ...) = -1.0170
|
|
# at f = 1/30, so the true decay is slightly FASTER than e^-W, by a factor that grows
|
|
# with W (16% low by W = 10). Assert the exact form, and bracket the heuristic.
|
|
rate = np.log(1.0 - F) / F
|
|
assert -1.02 < rate < -1.0
|
|
for w_abs in (1.0, 3.0, 10.0):
|
|
p = float(theory.window_miss_prob(F, w_abs))
|
|
assert abs(p - (1.0 - F) ** (w_abs / F)) < 1e-15
|
|
assert abs(p - np.exp(rate * w_abs)) < 1e-15 # exact closed form
|
|
assert np.exp(-1.02 * w_abs) < p < np.exp(-w_abs) # brackets the e^-W heuristic
|
|
# the spec default W = 10 makes the window a negligible loss channel
|
|
assert float(theory.window_miss_prob(F, 10.0)) < 1e-4
|
|
# strictly decreasing in W
|
|
ps = [float(theory.window_miss_prob(F, w)) for w in (1, 2, 3, 5, 7, 10)]
|
|
assert all(a > b for a, b in zip(ps[:-1], ps[1:], strict=True))
|