From 189b150381c2069b5154e56ab9035e2120f9ddd1 Mon Sep 17 00:00:00 2001 From: Ruge Lin Date: Sat, 3 Oct 2026 19:42:55 +0800 Subject: [PATCH] Prove uniform precision-depth bounds with rectangular query allocation --- README.md | 27 +- REVIEW.md | 88 ++-- WORKSPACE.md | 144 +++--- docs/BLOCKED_BILINEAR_LOOKUP.md | 6 +- docs/OPEN_PROBLEM.md | 53 ++- docs/PARALLEL_DIRTY_LOOKUP.md | 6 +- docs/README.md | 1 + docs/RELATED_WORK.md | 42 +- docs/SOURCE_MAP.md | 14 + docs/UNARY_PHASE_GRADIENT.md | 5 +- docs/UNIFORM_PRECISION_DEPTH.md | 522 +++++++++++++++++++++ docs/VERIFICATION.md | 15 + tests/README.md | 1 + tests/test_rectangular_query_allocation.py | 145 ++++++ 14 files changed, 926 insertions(+), 143 deletions(-) create mode 100644 docs/UNIFORM_PRECISION_DEPTH.md create mode 100644 tests/test_rectangular_query_allocation.py diff --git a/README.md b/README.md index d161723..bc8fc6e 100644 --- a/README.md +++ b/README.md @@ -133,24 +133,23 @@ magnitude frames at $`b\ge L+n+8`$. With $`b\ge2(L+n+8)`$, it gives $`O(\sqrt{NL}+L\ell_*(n)+NL/b)`$ T gates, still $`G=O(NL)`$. Leaf-phase derivatives retain a separate QBP stream. -At fixed accuracy, [blocked bilinear lookup](docs/BLOCKED_BILINEAR_LOOKUP.md) -with charged unary phase-source groups gives one complete real-frame -circuit using two clean flags and $`b\ge17(L+n+7)`$, with +The [uniform-precision depth theorem](docs/UNIFORM_PRECISION_DEPTH.md) +gives one complete real-frame circuit using two clean flags and +$`b\ge17(L+n+7)`$, with absolute constants: ```math -T=O(\sqrt N+N/b),\qquad G=O(N),\qquad -D_T=O(N/b^2+n). +T=O(\sqrt{NL}+NL/b+nL),\qquad G=O(NL),\qquad +D_T=O(NL/b^2+nL). ``` -T-count is optimal in order. Count and depth match simultaneously through -$`b\le\sqrt{N/n}`$ when the interval is nonempty. Larger widths retain -the linear depth upper bound; depth optimality remains open there. -Source preparation, reuse, and return are charged. - -The [variable-precision hybrid](docs/PARALLEL_DIRTY_LOOKUP.md) retains its -existing bounds: at $`L=\Theta(n)`$, sufficient $`b=\Theta(n)`$ gives -optimal worst-case $`T=\Theta(N)`$ and $`D_T=\Theta(N/n)`$ in one -complete real-frame circuit. The high-precision endpoint remains open. +Count and depth match simultaneously through $`b\le\sqrt{N/n}`$ +when nonempty. For $`6\le L\le\log_2(n+2)/16`$, the same guarantee +improves to $`T=O(\sqrt{NL}+NL/b)`$ and $`D_T=O(NL/b^2+n)`$; +the matching interval extends through $`b\le\sqrt{NL/n}`$. +Source preparation, reuse, and return are charged. At $`L=\Theta(n)`$, +sufficient $`b=\Theta(n)`$ gives optimal worst-case $`T=\Theta(N)`$ +and $`D_T=\Theta(N/n)`$. Larger-width depth optimality and the +high-precision endpoint remain open; the general count bound retains nL. Beyond Hopf frames, **literal diagonals and general one-target U(2) multiplexors** attain $`\Theta(\sqrt{NL}+L+NL/b)`$ with one clean diff --git a/REVIEW.md b/REVIEW.md index b73c1b4..24de0a1 100644 --- a/REVIEW.md +++ b/REVIEW.md @@ -159,45 +159,72 @@ $`b\ge C(L+n+7+\sqrt{NL})`$, the [parallel dirty-lookup construction](docs/PARALLEL_DIRTY_LOOKUP.md) instead retains $`T=O(\sqrt{NL}+L\ell_*(n))`$ while attaining $`D_T=O(\min\{nL+n^2,L\ell_*(n)+n^3\})`$ in the same circuit. -At fixed accuracy, two clean and sufficiently large -$`\Theta(\sqrt N)`$ dirty workspace give count-optimal -$`O(\sqrt N)`$ T gates with -$`O(n\log\log(n+2))`$ T-depth, using the -[grouped-program schedule](docs/GROUPED_PROGRAM_PREFETCH.md#8-complete-frame-theorem-at-fixed-accuracy). -It combines conditional program reuse with chunked late-query indicators. -The earlier [dirty-counter hybrid](docs/DIRTY_SUM_COMPRESSION.md) -retains its variable-precision and variable-width scope below, where +The earlier +[grouped-program schedule](docs/GROUPED_PROGRAM_PREFETCH.md#8-complete-frame-theorem-at-fixed-accuracy) +combined conditional program reuse with chunked late-query indicators, +giving fixed-accuracy $`O(n\log\log(n+2))`$ T-depth and optimal-order +T-count at sufficient square-root dirty width. Its +[unary-source refinement](docs/UNARY_PHASE_GRADIENT.md) gives +$`O(n)`$ depth with charged source preparation and return. The +[blocked bilinear query](docs/BLOCKED_BILINEAR_LOOKUP.md) extends this +to $`O(N/b^2+n)`$ at every $`b\ge17(L+n+7)`$ for fixed L. +These are dependencies of the uniform-precision result below. The older +[dirty-counter hybrid](docs/DIRTY_SUM_COMPRESSION.md) uses ```math \chi(t)=\log_2(t+2). ``` -The T-depth need not be optimal, and Clifford depth remains charged -separately. -The [hybrid extension](docs/PARALLEL_DIRTY_LOOKUP.md#every-eligible-width-and-precision) covers every -$`L\ge6`$ and $`b\ge17(L+n+7)`$ with two clean qubits: +No matching large-width T-depth or elementary-depth conclusion follows. + +The [uniform-precision theorem](docs/UNIFORM_PRECISION_DEPTH.md) covers every +$`L\ge6`$ and $`b\ge17(L+n+7)`$ with two external clean flags: + +```math +T=O\!\left(\sqrt{NL}+\frac{NL}{b}+nL\right),\qquad G=O(NL). +``` ```math -T=O\!\left(\sqrt{NL}+\frac{NL}{b}+nL\right),\qquad G=O(NL), -\qquad D_T=O\!\left(\frac{NL}{b^2}+nL+n\chi(n)\right). +D_T=O\!\left(\frac{NL}{b^2}+nL\right). ``` -All three bounds hold for one complete real-frame circuit. When nonempty, -the interval $`17(L+n+7)\le b\le\sqrt{NL/(nL+n\chi(n))}`$ has matching -worst-case count $`T^\star=\Theta(NL/b)`$ and depth +All constants are absolute, and all three bounds hold for one complete +real-frame circuit. Its full initialized-isometry error includes every +returned work register and arbitrary dirty/reference inputs; no +intermediate work is reset. When nonempty, the interval +$`17(L+n+7)\le b\le\sqrt{N/n}`$ has matching worst-case count +$`T^\star=\Theta(NL/b)`$ and depth $`D_T^\star=\Theta(NL/b^2)`$. For inverse-polynomial error in N, $`L=\Theta(n)`$ and sufficient $`b=\Theta(n)`$ give $`T^\star=\Theta(N)`$, $`D_T^\star=\Theta(N/n)`$. -Outside the interval the extra $`nL`$ count term is retained; no -uniform count-optimality claim follows from this schedule. +At every eligible width, the sufficient condition $`L\le N/n^2`$ +absorbs $`nL`$ into $`\sqrt{NL}`$ and makes the count optimal in +order. The general theorem retains that extra count term; it does not +claim count optimality for all precisions. + +In the explicit low-precision range +$`6\le L\le\log_2(n+2)/16`$, the same chapter gives the stronger +absolute-constant bounds + +```math +T=O\!\left(\sqrt{NL}+\frac{NL}{b}\right),\qquad G=O(NL). +``` + +```math +D_T=O\!\left(\frac{NL}{b^2}+n\right). +``` -At fixed L, the hybrid retains $`T=O(\sqrt N+N/b)`$ at every eligible -width and gives $`D_T=O(N/b^2+n\chi(n))`$. Its matching interval -extends asymptotically to order $`\sqrt{N/(n\chi(n))}`$. -The [precision cap](docs/AMORTIZED_DIRTY_LOOKUP.md#capping-the-source-precision) -keeps total source depth at $`O(n\log(n+1))`$; the dirty-counter -queries account for the remaining term. Depth at square-root-scale -workspace is still not known to be optimal. +Both resources match their lower bounds throughout the larger interval +$`17(L+n+7)\le b\le\sqrt{NL/n}`$, when nonempty. The construction +uses rectangular blocked queries to retain the $`\sqrt{NL}`$ count +scale, and an explicit unary-source cutoff uniform in this precision +range. This includes the new regime of slowly growing +$`L=o(\log n)`$. For $`L=\Omega(\log n)`$, the older +[hybrid bound](docs/PARALLEL_DIRTY_LOOKUP.md#every-eligible-width-and-precision) +already absorbs its $`n\chi(n)`$ term into $`nL`$. Neither the +additive n nor nL is proved to be a general depth lower bound. +Unrestricted large-width depth and the high-precision constant-clean +endpoint remain open. The [error audit](docs/HOPF_ERROR_ACCUMULATION.md) distinguishes sharp square-sum stability of ideal angle perturbations from coherent linear @@ -211,16 +238,17 @@ reduces the full radial error quadratically on the same two flags. Charging its additional source calls and native phase words permits a smaller source-width cap, while preserving the displayed asymptotic bounds. -The [conditional geometric source](docs/CONDITIONAL_GEOMETRIC_SOURCE.md) -now reduces the precision component to logarithmic depth wherever the +The earlier [conditional geometric source](docs/CONDITIONAL_GEOMETRIC_SOURCE.md) +reduces the precision component to logarithmic depth wherever the active logical suffix supplies enough temporary clean work. Only the same two external clean flags are used. Its lookup and suffix-predicate costs remain charged in that source-only interface. The [grouped composition](docs/GROUPED_PROGRAM_PREFETCH.md) -now shares early prefetch and predicate costs, while the +shares early prefetch and predicate costs, while the [chunked indicator](docs/CHUNKED_DIRTY_INDICATOR.md) bounds the entire late-query depth by $`O(n)`$ at fixed accuracy. Together they yield -the improved complete-frame depth stated above. +the earlier $`O(n\log\log(n+2))`$ complete-frame schedule; the unary +source and rectangular blocked queries supply the subsequent refinements. Separately, the [two-layer obstruction](docs/SHALLOW_SOURCE_OBSTRUCTION.md) excludes arbitrarily accurate full-input replacement of the original source by two T layers, even with unrestricted Clifford interlayers and diff --git a/WORKSPACE.md b/WORKSPACE.md index 6708fbc..abf9e25 100644 --- a/WORKSPACE.md +++ b/WORKSPACE.md @@ -3,9 +3,9 @@ This is the entry point when a previous conversation or execution workspace is unavailable. Proofs and decisions live in the repository. -The 2026-10-03 width-sensitive bilinear pass starts from verified main -`b94c77d8d82881c2bb2efc155a1da1f99f27c92e`, after the charged -unary phase-source theorem. Check later commits before +The 2026-10-03 uniform-precision pass starts from verified main +`9ae9685154b498e3350f8a89262dfc1f02fb07d6`, after the width-sensitive +blocked bilinear theorem. Check later commits before continuing. The selected state-based Hopf QBP construction and its bounded-input audit are complete; the [consolidated theorem](docs/STATE_BASED_QBP_THEOREM.md) is their entry point. @@ -25,7 +25,8 @@ and finite checks, without large simulations, QRAM, resets inside a compiler execution, supplied catalysts, or hidden initialized work. 1. Begin active depth work with the - [width-sensitive bilinear theorem](docs/BLOCKED_BILINEAR_LOOKUP.md), + [uniform-precision theorem](docs/UNIFORM_PRECISION_DEPTH.md), + then the [width-sensitive bilinear theorem](docs/BLOCKED_BILINEAR_LOOKUP.md), then the [unary phase-source theorem](docs/UNARY_PHASE_GRADIENT.md#7-complete-frame-theorem-at-fixed-accuracy), then the [grouped complete-frame theorem](docs/GROUPED_PROGRAM_PREFETCH.md#8-complete-frame-theorem-at-fixed-accuracy), [chunked dirty indicator](docs/CHUNKED_DIRTY_INDICATOR.md), @@ -79,7 +80,8 @@ compiler execution, supplied catalysts, or hidden initialized work. | State-based real and complex QBP | Two compiler flags, fine state preparation, an exact-return coarse word, and corrected magnitude/phase decoders; [consolidated theorem](docs/STATE_BASED_QBP_THEOREM.md) | | Additional dirty banks | Improve the state preparation T bound while charging the exact coarse circuit; [banked proof](docs/COMPLEX_COARSE_COMPILER.md#8-additional-dirty-banks-improve-fine-state-preparation) | | State-based T-depth | Two complete schedules, one retaining the sharper count at a stronger dirty reservation; [depth proof](docs/STATE_QBP_DEPTH.md) and [fair comparison](docs/QBP_COST_COMPARISON.md#7-state-based-t-depth-comparison) | -| Complete-frame T-count and T-depth | Same-circuit bounds at every accuracy; matching in an explicit workspace range, including inverse-polynomial error; [amortized tradeoff](docs/AMORTIZED_DIRTY_LOOKUP.md) | +| Complete-frame T-count and T-depth | Uniform depth O(NL/b²+nL), retaining the same-circuit count bound at b at least 17(L+n+7); matching through sqrt(N/n); [uniform theorem](docs/UNIFORM_PRECISION_DEPTH.md) | +| Slowly growing precision | For 6 ≤ L ≤ log₂(n+2)/16, depth O(NL/b²+n) and count O(sqrt(NL)+NL/b), with absolute constants; matching through sqrt(NL/n); [uniform theorem](docs/UNIFORM_PRECISION_DEPTH.md) | | Fixed-accuracy width-dependent depth | Two external flags, T-count O(sqrt N+N/b), and depth O(N/b²+n) at b at least 17(L+n+7); both orders match through sqrt(N/n); [blocked bilinear theorem](docs/BLOCKED_BILINEAR_LOOKUP.md) | | Complete-frame error accumulation | Sharp ideal-angle stability and finite relative spectra; coherent linear leakage in actual shared-flag source layers; [scoped error audit](docs/HOPF_ERROR_ACCUMULATION.md) | | Filtered complete-frame source | Quadratic radial error on the same two flags, charged native selective phases, and a smaller source-precision cap; [filter proof](docs/HOPF_RADIAL_FILTER.md) | @@ -105,81 +107,60 @@ compiler improvement, not a general end-to-end gradient speedup. ## Current depth frontier -The selected next-step recommendations were accepted: retain optimal-order -T-count in the same circuit, and treat a logarithmic depth gap as a useful -bounded milestone. The subsequent revision selected the variable-accuracy -extension before another large-workspace routing attempt. That extension -is now complete. The [hybrid refinement](docs/PARALLEL_DIRTY_LOOKUP.md#every-eligible-width-and-precision) -gives, for every $`L\ge6`$, two clean flags, and -$`b\ge17(L+n+7)`$, one prescribed complete real-frame circuit with +The [uniform-precision theorem](docs/UNIFORM_PRECISION_DEPTH.md) gives, +for every $`L\ge6`$, two clean flags, and $`b\ge17(L+n+7)`$, +one prescribed complete real-frame circuit with absolute constants: ```math T=O\!\left(\sqrt{NL}+\frac{NL}{b}+nL\right),\qquad G=O(NL), -\qquad D_T=O\!\left(\frac{NL}{b^2}+nL+n\chi(n)\right). +\qquad D_T=O\!\left(\frac{NL}{b^2}+nL\right). ``` -Here - -```math -\chi(t)=\log_2(t+2). -``` - -When nonempty, the range -$`17(L+n+7)\le b\le\sqrt{NL/(nL+n\chi(n))}`$ has simultaneous -optimal-order $`T^\star=\Theta(NL/b)`$ and +When nonempty, $`17(L+n+7)\le b\le\sqrt{N/n}`$ is a simultaneous +matching interval, with $`T^\star=\Theta(NL/b)`$ and $`D_T^\star=\Theta(NL/b^2)`$. For $`L=\Theta(n)`$, or inverse-polynomial error in N, sufficiently large $`b=\Theta(n)`$ gives $`T^\star=\Theta(N)`$ and $`D_T^\star=\Theta(N/n)`$. -Outside the matching range, retain the $`nL`$ term and compare the -older grouped schedules before claiming count optimality. - -At fixed L, the [blocked bilinear theorem](docs/BLOCKED_BILINEAR_LOOKUP.md) -now gives $`T=O(\sqrt N+N/b)`$, $`G=O(N)`$, and -$`D_T=O(N/b^2+n)`$ at the same threshold. For sufficiently large -$`b=\Theta(n)`$ above the threshold, one circuit has optimal-order -$`T=\Theta(N/n)`$ and $`D_T=\Theta(N/n^2)`$. -The fixed-accuracy matching depth tradeoff -$`D_T^\star=\Theta(N/b^2)`$ now holds throughout -$`17(L+n+7)\le b\le\sqrt{N/n}`$ when this interval is nonempty. -The earlier loader places one low-address indicator echo around a multiplexed -family of linear shears. A two-pass dirty traversal and rank-reduced -controlled shears amortize both former per-batch logarithms. -Dirty selector stacks remain separate from live banks and indicators; -all lookup work returns exactly. Full-frame error and QBP substitution -are inherited unchanged. A [capped precision allocation](docs/AMORTIZED_DIRTY_LOOKUP.md#capping-the-source-precision) -now reduces accumulated source depth to $`O(nL+n\log(n+1))`$ without -increasing any layer's width or the weighted lookup costs. It retains -the full error guarantee using the sharper local isometry constant. -The total assigned precision is within $`5n`$ bits of the optimum under -that additive error certificate; this is not a frame-depth lower bound. -The original routed schedule retains an additive $`n^2`$ term. The -[dirty-counter hybrid](docs/PARALLEL_DIRTY_LOOKUP.md#6-a-polylogarithmic-depth-indicator-using-dirty-counters) -improves fixed-accuracy large-workspace depth to -$`O(n\chi(n))`$ while retaining optimal-order T-count. -The [unary phase-source theorem](docs/UNARY_PHASE_GRADIENT.md#7-complete-frame-theorem-at-fixed-accuracy) -first improved the fixed-accuracy, sufficient-square-root-width bound to +Outside the matching interval, retain the nL count term; the sufficient +condition $`L\le N/n^2`$ makes it absorb into $`\sqrt{NL}`$. + +There is a stronger uniform corollary for slowly growing precision: ```math -T=O_\eta(\sqrt N),\qquad G=O_\eta(N),\qquad -D_T=O_\eta(n), -\qquad b\ge C_\eta\sqrt N. +6\le L\le\frac{\log_2(n+2)}{16},\qquad +T=O\!\left(\sqrt{NL}+\frac{NL}{b}\right),\qquad +D_T=O\!\left(\frac{NL}{b^2}+n\right),\qquad G=O(NL). ``` -These are simultaneous bounds on one complete real-frame circuit using -two external clean flags. Early groups store one-hot programs and a charged -unary phase source in conditional logical zeros. Bilinear cyclic shifts -have constant T-depth; source preparation and its actual inverse occur -once per group. Chunked dirty indicators and capped source precision -control the entire remaining tail. The blocked bilinear refinement removes -the late word-bank route at smaller dirty widths and extends this result -to the displayed width-dependent curve. -The worst-case count is optimal in order. At square-root-scale dirty -width the depth lower bound remains only Omega(1). Fixed accuracy is -essential to this theorem; -the general-precision matching interval and endpoint remain unchanged. -The older general-precision schedules below remain useful where their -source or workspace costs are sharper. The separate state-based and -complex-frame schedules retain their existing contracts and bounds. +Its matching interval extends through $`b\le\sqrt{NL/n}`$. +The absolute constants and thresholds are independent of eta and L. +For fixed L this recovers the [blocked bilinear curve](docs/BLOCKED_BILINEAR_LOOKUP.md), +whose constants could depend on eta. At square-root-scale width the +fixed-accuracy depth upper bound is O(n), while the lower bound remains +Omega(1). No unrestricted depth optimality or endpoint improvement follows. + +The new proof uses the charged [unary source](docs/UNARY_PHASE_GRADIENT.md) +in early groups, with cutoff sublinear in n uniformly in the displayed +precision range. Rectangular indicator allocation keeps the tail's +weighted count at $`O(\sqrt{NL})`$ and its entire depth at +$`O(NL/b^2+n)`$. For larger L, the existing +[hybrid theorem](docs/PARALLEL_DIRTY_LOOKUP.md#every-eligible-width-and-precision) +already absorbs its $`n\log_2(n+2)`$ term into nL. No new native +query identity is needed. Source preparation and actual inverses are +charged; all dirty work returns, including arbitrary reference inputs. +The original sufficient threshold remains exactly $`17(L+n+7)`$. +The two-clean high-precision endpoint remains outside that threshold. + +Older schedules remain useful when their count or workspace requirements +are sharper. In those ledgers, $`\chi(t)=\log_2(t+2)`$. The [amortized compiler](docs/AMORTIZED_DIRTY_LOOKUP.md) +removed the per-batch logarithms but retained quadratic query depth. +Its capped precision allocation has total source depth +$`O(nL+n\log(n+1))`$ and is within $`5n`$ bits of its additive +error-certificate optimum. The dirty-counter hybrid, grouped program +reuse, unary source, and blocked bilinear query successively improve the +depth upper bound. Their scoped obstructions are not frame-depth lower +bounds. The separate state-based and complex-frame schedules retain +their existing contracts. The [state-based depth theorem](docs/STATE_QBP_DEPTH.md) closes the selected composition audit for both real and gauge-fixed complex Hopf QBP. Put @@ -275,7 +256,7 @@ software extensions below are not prerequisites for the stated theorem. | Variable-size native schedule | Optional software: general tables, predicates, reflections, and banked count/depth scheduling. The bounded component-to-gradient pass is complete | | General guarded decoder | Optional software: replace the general floating-point contractions with the proved certified arithmetic. Exact fixture decoders cover only their fixed target | | Constant-clean complete-frame endpoint | Still open independently of the state-based task. A new candidate must supply an explicit complete native identity and symbolic precision/workspace ledger before another fixture pass | -| Optimal T-depth | At fixed accuracy both match through $`b\le\sqrt{N/n}`$ above the literal threshold; the variable-L interval remains $`17(L+n+7)\le b\le\sqrt{NL/(nL+n\chi(n))}`$; larger-width optimality and further precision dependence remain unresolved | +| Optimal T-depth | Uniform matching through sqrt(N/n) above the literal threshold; for 6 ≤ L ≤ log₂(n+2)/16 this extends through sqrt(NL/n). Larger-width optimality remains open | Do not repeat the completed coefficient-to-row, two- and four-row lookup, one- and two-system-qubit amplification, or coherent residual-selection passes. @@ -506,15 +487,22 @@ indicator echo returns all masks. Chunk lengths and block sizes are chosen together, and every tail resource sum is charged. This extends the simultaneous fixed-accuracy matching interval to sqrt(N/n). -The next bounded question is the precision dependence of this route. -First expose the eta-dependent source width, conditional suffix cutoff, -and weighted query sums before asserting any uniform-L improvement. -No such extension is established here. A stronger unrestricted -large-width depth lower bound and the high-precision endpoint remain -separate. Do not repeat the completed selector, source-reuse, unary -source, or blocked-query proofs. Retain literal phases, actual inverses, -full work return, and charged preparation in any new schedule. -The completed modest-width +The [uniform precision pass](docs/UNIFORM_PRECISION_DEPTH.md) now exposes +the eta-dependent source width, conditional suffix cutoff, and weighted +query sums. Its rectangular allocation avoids the balanced query's +extra square root of precision in T-count. It proves the uniform bounds +and the stronger slowly growing precision corollary above. Bounded +integer checks cover allocation floors and workspace constraints; they +do not replace the asymptotic proof or implement a scalable compiler. + +A next depth pass should target the remaining large-width gap: identify +a concrete way to share work between unary groups or a full-frame +lower-bound invariant that permits arbitrary Clifford interlayers. +The linear upper bound is not itself a lower bound. The high-precision +endpoint remains a separate question. Do not repeat the completed +selector, source-reuse, unary source, blocked-query, or precision-splice +proofs. Retain literal phases, actual inverses, full work return, and +charged preparation in any new schedule. The completed modest-width matching theorem does not require solving the high-precision endpoint. For any new component, keep literal phases and actual inverses, declare all initialized inputs, include borrowed-work return in its isometry diff --git a/docs/BLOCKED_BILINEAR_LOOKUP.md b/docs/BLOCKED_BILINEAR_LOOKUP.md index 2ce8cea..c10424a 100644 --- a/docs/BLOCKED_BILINEAR_LOOKUP.md +++ b/docs/BLOCKED_BILINEAR_LOOKUP.md @@ -47,7 +47,11 @@ D_T^\star=\Theta_\eta(N/b^2). The additive n in the upper bound is not an unrestricted depth lower bound. Depth optimality at larger widths and the variable-accuracy complete-frame endpoint remain open. Constants may depend on the fixed -accuracy; this theorem does not assert a new all-precision bound. +accuracy in this chapter. The subsequent +[uniform-precision refinement](UNIFORM_PRECISION_DEPTH.md) exposes that +dependence and uses a rectangular allocation of the same query to prove +absolute-constant depth O(NL/b²+nL), with a stronger slowly growing +precision corollary. ## 1. A controlled bilinear output with one arbitrary dirty helper diff --git a/docs/OPEN_PROBLEM.md b/docs/OPEN_PROBLEM.md index 4082360..e47dade 100644 --- a/docs/OPEN_PROBLEM.md +++ b/docs/OPEN_PROBLEM.md @@ -40,7 +40,8 @@ hybrid depth bounds, write | T-depth with additional dirty banks | $`D_T=O(NL/b+\min\{nL+n^2,L\ell_*(n)+n^3\})`$ at $`a=2`$, $`b\ge2(L+n+7)`$, with $`T,G=O(NL)`$ | Choose between the layerwise and grouped [schedules](T_DEPTH_COMPILER.md); real frames; optimizing depth may increase T-count; no matching frontier established | | Simultaneous T-count and T-depth | $`T=O(\sqrt{NL}+L\ell_*(n))`$, $`D_T=O(\min\{nL+n^2,L\ell_*(n)+n^3\})`$, $`G=O(NL)`$, at $`a=2`$, $`b\ge C(L+n+7+\sqrt{NL})`$ | Same real-frame circuit, for sufficiently large fixed C; [parallel dirty lookup](PARALLEL_DIRTY_LOOKUP.md); T-depth optimality remains open | | Fixed-accuracy count and depth versus width | $`T=O(\sqrt N+N/b)`$, $`D_T=O(N/b^2+n)`$, $`G=O(N)`$, at $`a=2`$, fixed L, $`b\ge17(L+n+7)`$ | Same complete real-frame circuit; count is optimal in order, and depth is matching through $`b\le\sqrt{N/n}`$; [blocked bilinear lookup](BLOCKED_BILINEAR_LOOKUP.md) | -| Variable-accuracy count and depth | $`T=O(\sqrt{NL}+NL/b+nL)`$, $`D_T=O(NL/b^2+nL+n\chi(n))`$, $`G=O(NL)`$, at $`a=2`$, $`L\ge6`$, $`b\ge17(L+n+7)`$ | Same complete real-frame circuit; both are matching when $`b\le\sqrt{NL/(nL+n\chi(n))}`$; [hybrid composition](PARALLEL_DIRTY_LOOKUP.md#every-eligible-width-and-precision) | +| Uniform-precision count and depth | $`T=O(\sqrt{NL}+NL/b+nL)`$, $`D_T=O(NL/b^2+nL)`$, $`G=O(NL)`$, at $`a=2`$, $`L\ge6`$, $`b\ge17(L+n+7)`$ | Same complete real-frame circuit, with absolute constants; both are matching through $`b\le\sqrt{N/n}`$ when eligible; [uniform-precision theorem](UNIFORM_PRECISION_DEPTH.md) | +| Low-precision count and depth | $`T=O(\sqrt{NL}+NL/b)`$, $`D_T=O(NL/b^2+n)`$, $`G=O(NL)`$, at $`a=2`$, $`6\le L\le\log_2(n+2)/16`$, $`b\ge17(L+n+7)`$ | Absolute constants; count is optimal in order, and both resources match through $`b\le\sqrt{NL/n}`$ when the interval is nonempty; [low-precision theorem](UNIFORM_PRECISION_DEPTH.md) | | Fixed-accuracy large-width depth | $`T=O_\eta(\sqrt N)`$, $`G=O_\eta(N)`$, $`D_T=O_\eta(n)`$, at $`a=2`$, $`b\ge C_\eta\sqrt N`$ | Same complete real-frame circuit with charged unary source preparation/return; [unary theorem](UNARY_PHASE_GRADIENT.md#7-complete-frame-theorem-at-fixed-accuracy); T-count is optimal in order, depth lower bound remains $`\Omega(1)`$ | Take the best applicable construction. For fixed L, $`a=2`$ and @@ -49,6 +50,12 @@ $`T^\star=\Theta(N/n)`$, sharper than the grouped estimate. Extra clean qubits may be left unused; the depth schedules retain their separately proved two-clean allocation. +The uniform-precision count is optimal at every eligible width under the +sufficient condition $`L\le N/n^2`$, which absorbs its $`nL`$ term +into $`\sqrt{NL}`$. No count-optimality claim is made for all L. +The uniform and low-precision bounds include arbitrary dirty/reference +inputs and every returned work register on the same circuit. + All frame constructions preserve the prescribed completion and the [fixed-parameter QBP error contract](QBP_APPROXIMATION.md). They do not differentiate discrete synthesis. Complex leaf-phase derivatives remain a @@ -82,7 +89,8 @@ uses $`a=2`$ throughout and respects each sufficient allocation threshold. |---|---|---|---| | Fixed L, $`b=\Theta(n)`$ and $`b\ge17B_0`$ | $`\Omega(N/n^2)`$ | $`O(N/n^2)`$ with $`T=\Theta(N/n)`$ | Matching count and depth in one circuit; [amortized schedule](AMORTIZED_DIRTY_LOOKUP.md) | | Fixed L, $`17B_0\le b\le\sqrt{N/n}`$ | $`\Omega(N/b^2)`$ | $`O(N/b^2)`$ with optimal-order count | Matching throughout this interval when nonempty | -| Variable L, $`17B_0\le b\le\sqrt{NL/(nL+n\chi(n))}`$ | $`\Omega(NL/b^2)`$ | $`O(NL/b^2)`$ with $`T=\Theta(NL/b)`$ | Matching throughout this interval when nonempty | +| Variable L, $`17B_0\le b\le\sqrt{N/n}`$ | $`\Omega(NL/b^2)`$ | $`O(NL/b^2)`$ with $`T=\Theta(NL/b)`$ | Uniform-precision matching interval, when nonempty | +| $`6\le L\le\log_2(n+2)/16`$, $`17B_0\le b\le\sqrt{NL/n}`$ | $`\Omega(NL/b^2)`$ | $`O(NL/b^2)`$ with $`T=\Theta(NL/b)`$ | Larger low-precision matching interval, when nonempty | | $`L=\Theta(n)`$, sufficient $`b=\Theta(n)`$ | $`\Omega(N/n)`$ | $`O(N/n)`$ with $`T=\Theta(N)`$ | Matching at inverse-polynomial error in N for sufficiently large n | | Fixed L, $`2B_0\le b\lt17B_0`$ | $`\Omega(N/n^2)`$ | $`O(N/n)`$ | Earlier schedule remains the proved fallback at this literal reservation | | Fixed L, sufficiently large $`b=\Theta(\sqrt N)`$ | $`\Omega(1)`$ | $`O(n)`$ with $`T=O(\sqrt N)`$ | [Unary phase-source groups](UNARY_PHASE_GRADIENT.md#7-complete-frame-theorem-at-fixed-accuracy); depth lower bound remains unmatched | @@ -91,15 +99,15 @@ uses $`a=2`$ throughout and respects each sufficient allocation threshold. | Selected endpoint $`L=N,b=B_0`$ | $`\Omega(1)`$ | $`O(N\ell_*(n))`$ from $`D_T\le T`$ | The larger-bank depth theorem does not apply | The amortized schedule removes the earlier per-batch logarithm from -both indicator routing and chunk selection. The newer hybrid replaces +both indicator routing and chunk selection. The earlier hybrid replaced its additive $`n^2`$ term by $`n\chi(n)`$ at every eligible width -and precision, expanding the simultaneous matching interval above. +and precision. It remains a dependency of the uniform theorem. The [capped source precision](AMORTIZED_DIRTY_LOOKUP.md#capping-the-source-precision) reduces source depth to $`O(n\log(n+1))`$ at fixed L without changing the count or workspace orders. The quadratic contribution remaining in this schedule comes from query routing; it is not an unavoidable source cost. The exact dirty-counter indicator, [masked-sum refinement](DIRTY_SUM_COMPRESSION.md), -and bilinear-query hybrid now improve +and bilinear-query hybrid improved the square-root-width upper bound to $`O(n\chi(n))`$. The [grouped-program refinement](GROUPED_PROGRAM_PREFETCH.md#8-complete-frame-theorem-at-fixed-accuracy) gives $`O(n\log\log(n+2))`$ at fixed accuracy, with returned @@ -117,9 +125,18 @@ At square-root-scale dirty width, the depth lower bound is still constant, so depth optimality there remains open. The lower bound does not assume count optimality; the upper circuit also retains optimal-order T-count. No high-precision endpoint improvement follows. -For varying L outside the matching interval, the new count bound retains -its $`nL`$ term and need not match the lower bound. Earlier grouped and -parallel constructions remain eligible where they are sharper. +The [uniform-precision theorem](UNIFORM_PRECISION_DEPTH.md) now removes +the separate $`n\chi(n)`$ term: its depth is +$`O(NL/b^2+nL)`$ with absolute constants. For +$`L=\Omega(\log n)`$, the earlier hybrid already absorbed that term +into $`nL`$. The new regime includes slowly growing +$`L=o(\log n)`$: rectangular blocked queries preserve the +$`\sqrt{NL}`$ count scale, and a uniform unary cutoff gives the +stronger $`O(NL/b^2+n)`$ depth for +$`6\le L\le\log_2(n+2)/16`$. The general count bound retains its +$`nL`$ term; outside the sufficient condition $`L\le N/n^2`$ or a +matching interval, count optimality need not follow. Earlier grouped +and parallel constructions remain eligible where they are sharper. T-depth permits Clifford circuits of nonzero depth between its T layers. It is not total circuit depth or elapsed QBP execution time. Neither the @@ -155,16 +172,16 @@ square-sum-plus-quadratic composition bound. Source widths may then use $`m_d=L+4+\min\{n-d,\lceil\tfrac12\log_2(8n)\rceil\}`$. This halves the logarithmic coefficient in assigned source precision, but additional calls and phase words preserve the same asymptotic costs. -The [conditional geometric source](CONDITIONAL_GEOMETRIC_SOURCE.md) -now reduces the precision part to logarithmic depth when the active +The earlier [conditional geometric source](CONDITIONAL_GEOMETRIC_SOURCE.md) +reduces the precision part to logarithmic depth when the active suffix can reserve 7m temporary clean bits for source width m. It uses the same two external flags and an enlarged, charged reflection. With the unfiltered additive cap, source/reflection depth at fixed accuracy becomes $`O(n\log\log(n+2)+\log^2(n+2))`$ after the late -fallback. The [grouped-program theorem](GROUPED_PROGRAM_PREFETCH.md) -now supplies the missing query/predicate composition: exact cached-program +fallback. Its [grouped-program theorem](GROUPED_PROGRAM_PREFETCH.md) +supplies the query/predicate composition: exact cached-program erasure survives source leakage and changing internal addresses; chunked -dirty indicators avoid the late routing bottleneck. The full depth is +dirty indicators avoid the late routing bottleneck. This earlier depth is $`O(n\log\log(n+2))`$ with optimal-order count at sufficient square-root-scale dirty width and fixed accuracy. @@ -189,9 +206,13 @@ allocation proves the displayed $`O(n)`$ upper bound. The [blocked bilinear theorem](BLOCKED_BILINEAR_LOOKUP.md) now closes the width-dependent extension. It replaces the residual bank route by selected bilinear blocks with one returned dirty phase helper, then -charges the coupled block-size/chunk-length allocation. The next bounded -question is the precision dependence of this route; its unary source -width and suffix cutoff must be made explicit before any uniform-L claim. +charges the coupled block-size/chunk-length allocation. The +[uniform-precision chapter](UNIFORM_PRECISION_DEPTH.md) now proves the +precision extension: rectangular indicator dimensions retain the correct +precision-dependent count, and the explicit unary cutoff is sublinear +uniformly throughout its stated low-precision range. The remaining +questions concern unrestricted large-width depth and the high-precision +constant-clean endpoint; neither is resolved by these two-clean schedules. The [two-layer obstruction](SHALLOW_SOURCE_OBSTRUCTION.md) separately allows unrestricted Clifford interlayers: the original source at width diff --git a/docs/PARALLEL_DIRTY_LOOKUP.md b/docs/PARALLEL_DIRTY_LOOKUP.md index 4264baa..bb3fb51 100644 --- a/docs/PARALLEL_DIRTY_LOOKUP.md +++ b/docs/PARALLEL_DIRTY_LOOKUP.md @@ -67,8 +67,10 @@ At fixed accuracy, the [blocked bilinear refinement](BLOCKED_BILINEAR_LOOKUP.md) uses conditional unary groups and width-constrained bilinear blocks to give $`D_T=O(N/b^2+n)`$, retaining optimal-order T-count throughout $`b\ge17B_0`$. Count and depth match through $`b\le\sqrt{N/n}`$ -when nonempty. Its fixed-accuracy proof does not replace the uniform -precision statements above. +when nonempty. The subsequent [uniform-precision theorem](UNIFORM_PRECISION_DEPTH.md) +uses this hybrid at larger precision and a rectangular blocked allocation +at smaller precision, improving the additive depth to nL with absolute +constants throughout the same sufficient-width range. ## 1. An exact dirty indicator by conjugated routing diff --git a/docs/README.md b/docs/README.md index f63e0fd..198fc7e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -28,6 +28,7 @@ topic has one primary chapter below. | [Grouped program reuse](GROUPED_PROGRAM_PREFETCH.md) | Complete real-frame depth O(n log log n) at fixed accuracy and sufficient square-root dirty width; O(n) selector maintenance and scoped source-reuse audit | | [Unary phase-source groups](UNARY_PHASE_GRADIENT.md) | Complete real-frame T-depth O(n) at fixed accuracy and sufficient square-root dirty width, retaining optimal-order T-count; charged preparation and exact guarded cyclic shifts | | [Blocked bilinear lookup](BLOCKED_BILINEAR_LOOKUP.md) | Full-input selected bilinear blocks with returned dirty work; fixed-accuracy depth O(N/b²+n), optimal-order T-count, and matching range through sqrt(N/n) | +| [Uniform precision and depth](UNIFORM_PRECISION_DEPTH.md) | Absolute-constant depth O(NL/b²+nL); rectangular allocation gives O(NL/b²+n) for slowly growing precision with the same-circuit count guarantee | | [Chunked dirty indicator](CHUNKED_DIRTY_INDICATOR.md) | A tunable exact dirty-tree indicator and a summable late-query budget that removes the late routing bottleneck | | [Hopf error accumulation](HOPF_ERROR_ACCUMULATION.md) | Sharp ideal-angle stability, finite relative spectra, and coherent leakage in the actual shared-flag sources; scoped precision boundaries | | [Flag-echo audit](HOPF_FLAG_ECHO.md) | Exact errors of four diagonal Pauli echoes, their generic linear leakage, and an exact equal-mask exception | diff --git a/docs/RELATED_WORK.md b/docs/RELATED_WORK.md index 4a55ad1..d79ea5d 100644 --- a/docs/RELATED_WORK.md +++ b/docs/RELATED_WORK.md @@ -972,7 +972,47 @@ T^\star=\Theta_\eta(N/b),\quad D_T^\star=\Theta_\eta(N/b^2). This extends the fixed-accuracy matching window; it introduces no new external synthesis premise or depth lower-bound method. The preceding -variable-accuracy bounds remain separate and unchanged. The unary +variable-accuracy bounds remain valid; Section 21 gives a uniform +precision refinement. The unary square-root-width schedule remains a valid predecessor and corollary; unrestricted large-width depth optimality and the constant-clean high-precision endpoint remain open. No generic lookup priority is claimed. + +## 21. Uniform precision and rectangular query allocation (3 October 2026) + +The [uniform precision theorem](UNIFORM_PRECISION_DEPTH.md) uses the same +native selected-block query, dirty traversal and bilinear echo as Section +20. Rectangular block dimensions balance word precision against indicator +cost. The proof makes the unary-source cutoff uniform at low precision +and uses the existing hybrid when its source term absorbs the logarithmic +depth contribution. These are allocation and composition results; the +external ingredients in Sections 19–20 are unchanged, and no new native +query or synthesis premise is assumed. + +With $`L=\max\{6,\lceil\log_2(1/\eta)\rceil\}`$, two clean flags +and $`b\ge17(L+n+7)`$, one complete real-frame circuit has absolute, +precision-independent constants in + +```math +T=O\!\left(\sqrt{NL}+\frac{NL}{b}+nL\right),\qquad G=O(NL), +\qquad D_T=O\!\left(\frac{NL}{b^2}+nL\right). +``` + +For $`6\le L\le\log_2(n+2)/16`$, the sharper bounds are + +```math +T=O\!\left(\sqrt{NL}+\frac{NL}{b}\right),\qquad G=O(NL), +\qquad D_T=O\!\left(\frac{NL}{b^2}+n\right). +``` + +The existing count lower bounds divided by physical width give +simultaneous worst-case $`T^\star=\Theta(NL/b)`$ and +$`D_T^\star=\Theta(NL/b^2)`$ through $`b\le\sqrt{N/n}`$ +generally, and through $`b\le\sqrt{NL/n}`$ in the low-precision +regime, above the literal threshold and when the intervals are nonempty. +The low-precision count is optimal at every eligible width; the general +count retains nL and is asserted optimal outside its matching interval +only under a sufficient condition such as $`L\le N/n^2`$. +All preparation, queries, actual inverses and work return remain charged. +The selected high-precision endpoint and unrestricted large-width depth +optimality remain open. No priority claim follows from this composition. diff --git a/docs/SOURCE_MAP.md b/docs/SOURCE_MAP.md index 7093906..0637de3 100644 --- a/docs/SOURCE_MAP.md +++ b/docs/SOURCE_MAP.md @@ -203,6 +203,7 @@ is inherited. | R42 | state-based QBP T-depth composition | [depth proof](STATE_QBP_DEPTH.md) composes R21/R23's exact schedules, inherited from F2, with R36/R39's constant number of residual rotations and the actual exact-return coarse interpreter; with $`B_0=P+n+7`$, $`b\ge2B_0`$ gives $`D_T=O(NP/b+P+n^3)`$, $`T,G=O(NP)`$, while $`b\ge16(B_0+\sqrt{NP})`$ gives one circuit with $`T=O(\sqrt{NP}+P+n\sqrt N)`$, $`G=O(NP)`$, and $`D_T=O(P+n^3)`$; both real/complex task streams and oracle depth are charged; no new lookup primitive, depth optimality, total-runtime gain, or general emitter is claimed | | R43 | fixed-accuracy linear T-depth complete real frame | [unary phase-source proof](UNARY_PHASE_GRADIENT.md), 3 October 2026: coherent one-hot shifts, guarded bilinear work, unitary source preparation/return, and the Hopf angle-stability/group/query allocation give $`D_T=O_\eta(n)`$, $`T=O_\eta(\sqrt N)`$, $`G=O_\eta(N)`$ with two external clean flags and sufficient $`C_\eta\sqrt N`$ dirty work; F5/F8/F31/F36/F37 are attributed ingredients, F29/F38 are comparisons; no supplied catalyst, generic synthesis priority, matching depth lower bound, or high-precision endpoint follows | | R44 | fixed-accuracy dirty-width/depth tradeoff | [blocked bilinear lookup](BLOCKED_BILINEAR_LOOKUP.md), composed with R43's early groups: for fixed $`0\lt\eta\le1/64`$, $`L=\max\{6,\lceil\log_2(1/\eta)\rceil\}`$ and $`b\ge17(L+n+7)`$, one two-clean complete real-frame circuit has $`T=O_\eta(\sqrt N+N/b)`$, $`G=O_\eta(N)`$ and $`D_T=O_\eta(N/b^2+n)`$; simultaneous worst-case orders are $`\Theta_\eta(N/b)`$ and $`\Theta_\eta(N/b^2)`$ when $`b\le\sqrt{N/n}`$; the full-input leaf and allocation use existing F2/F8/F29/F30/F31 ingredients, and F3/F4 supply the lower-bound lineage; large-width depth optimality and the high-precision endpoint remain open | +| R45 | uniform precision and dirty-width depth tradeoff | [uniform composition](UNIFORM_PRECISION_DEPTH.md), 3 October 2026: at two clean flags, $`L\ge6`$ and $`b\ge17(L+n+7)`$, absolute constants give $`T=O(\sqrt{NL}+NL/b+nL)`$, $`G=O(NL)`$, $`D_T=O(NL/b^2+nL)`$; for $`L\le\log_2(n+2)/16`$, the sharper same-circuit bounds omit nL from T and replace it by n in depth. The matching width endpoints are respectively $`\sqrt{N/n}`$ and $`\sqrt{NL/n}`$ when eligible; this is rectangular allocation and uniform composition of R43/R44 with the existing hybrid, using the same external premises and F3/F4 lower bounds, with no new native query or high-precision endpoint claim | The [Hopf error audit](HOPF_ERROR_ACCUMULATION.md) derives a sharp ideal-angle stability recurrence and an exact finite relative-spectrum recursion from @@ -344,6 +345,19 @@ constants depend on fixed eta. Its enlarged matching window follows from the existing F3/F4 count lower bound divided by physical width, not a new depth lower-bound method. +The [uniform precision refinement](UNIFORM_PRECISION_DEPTH.md), added +**3 October 2026**, chooses rectangular blocks in R44's existing query +and makes the unary cutoff and the split between precision regimes +uniform. It reuses the same native circuits, source-return certificate, +and F2/F8/F29/F30/F31/F36/F37 ingredients; no new external synthesis +premise is introduced. R45's general count is optimal in order within +its matching interval, or at all eligible widths when $`L\le N/n^2`$; +its low-precision count is optimal throughout its stated regime. Both +matching intervals use the existing F3/F4 lower bounds. The earlier +theorems remain valid, and neither generic priority, unrestricted +large-width depth optimality, nor the selected high-precision endpoint +is claimed. + The [consolidated state-based QBP theorem](STATE_BASED_QBP_THEOREM.md) collects R36 and R38–R42 under one input, precision, workspace, and sampling contract. It introduces no additional compiler bound or diff --git a/docs/UNARY_PHASE_GRADIENT.md b/docs/UNARY_PHASE_GRADIENT.md index 8c0df41..1ba8b69 100644 --- a/docs/UNARY_PHASE_GRADIENT.md +++ b/docs/UNARY_PHASE_GRADIENT.md @@ -18,7 +18,10 @@ previous conjugated geometric reflection by a shallow implementation. The [blocked bilinear extension](BLOCKED_BILINEAR_LOOKUP.md) retains this source construction and proves fixed-accuracy depth $`O(N/b^2+n)`$ with optimal-order T-count throughout the original sufficient dirty-width -range. The square-root-width theorem below remains valid. +range. The [uniform-precision composition](UNIFORM_PRECISION_DEPTH.md) +subsequently bounds this source cutoff uniformly for slowly growing L +and combines it with a rectangular query allocation. The fixed-accuracy +square-root-width theorem below remains valid. ## 1. Local contract diff --git a/docs/UNIFORM_PRECISION_DEPTH.md b/docs/UNIFORM_PRECISION_DEPTH.md new file mode 100644 index 0000000..488b048 --- /dev/null +++ b/docs/UNIFORM_PRECISION_DEPTH.md @@ -0,0 +1,522 @@ +# A uniform precision-depth bound and a stronger low-precision range + +[Blocked bilinear query](BLOCKED_BILINEAR_LOOKUP.md) · [Unary group compiler](UNARY_PHASE_GRADIENT.md) · [Existing all-precision hybrid](PARALLEL_DIRTY_LOOKUP.md#every-eligible-width-and-precision) · [Capped source precision](AMORTIZED_DIRTY_LOOKUP.md#capping-the-source-precision) + +The blocked query can retain its depth bound while using unequal +indicator lengths to retain the frame's square-root precision +dependence in T-count. This changes the workspace allocation, not its native circuit +identities. The resulting tail schedule combines with the unary early +groups when precision is at most a small constant times log n. At larger +precision, the existing hybrid's logarithmic term is already absorbed +by its nL term. This gives one uniform theorem, with constants independent +of the requested accuracy. + +## 1. The uniform theorem and its stronger low-precision corollary + +Let $`n\ge1`$, $`N=2^n`$, and $`0\lt\eta\le1/64`$. Put + +```math +L=\max\{6,\lceil\log_2(1/\eta)\rceil\},\qquad B_0=L+n+7. +``` + +For every $`b\ge17B_0`$, every prescribed complete real Hopf frame W +has one coherent Clifford+T circuit V with two external clean flags and +at most b arbitrary dirty qubits such that + +```math +\|VJ_2-J_2(W\otimes I_b)\|\le\eta, +``` + +```math +T=O\!\left(\sqrt{NL}+\frac{NL}{b}+nL\right),\qquad +G=O(NL),\qquad +D_T=O\!\left(\frac{NL}{b^2}+nL\right). +``` + +All implied constants are absolute: none depends on n, L, eta, or b. +All bounds hold on the same circuit. The error is the full +initialized-isometry norm, including arbitrary dirty inputs, work return, +and reference correlations. Every quantum query and actual inverse is +charged. There are no measurements, resets, QRAM, supplied phase states, +or uncharged compiler primitives. T-depth permits arbitrary intervening +Clifford circuits; their elementary gate count remains included in G. + +The construction has a stronger bound throughout the explicit range + +```math +6\le L\le\frac1{16}\log_2(n+2): +``` + +```math +T=O\!\left(\sqrt{NL}+\frac{NL}{b}\right),\qquad +G=O(NL),\qquad +D_T=O\!\left(\frac{NL}{b^2}+n\right). +``` + +These constants are also absolute. The range is an asymptotic sufficient +condition, not a practical crossover estimate. In particular it contains +every fixed accuracy for sufficiently large n. Unlike a fixed-accuracy +statement with an unspecified eta-dependent constant, this corollary +controls the constants while L varies in the displayed range. + +The inherited lower bounds give simultaneous matching count and depth +on each nonempty interval + +```math +17B_0\le b\le\sqrt{N/n} +``` + +for the uniform theorem, and on the larger interval + +```math +17B_0\le b\le\sqrt{NL/n} +``` + +in the displayed low-precision range. On either relevant interval, + +```math +T^\star=\Theta(NL/b),\qquad D_T^\star=\Theta(NL/b^2). +``` + +These are worst-case bounds over prescribed complete real frames. No +unrestricted lower bound for the additive n or nL is asserted. The +large-width depth optimum and the high-precision complete-frame count +endpoint remain separate questions; the uniform theorem retains nL +in its general T-count. The sufficient condition $`L\le N/n^2`$ +absorbs that term into $`\sqrt{NL}`$, making the count optimal in +order at every eligible width in that precision range. + +## 2. An unequal indicator allocation preserves the block-depth bound + +The [blocked bilinear construction](BLOCKED_BILINEAR_LOOKUP.md#1-a-controlled-bilinear-output-with-one-arbitrary-dirty-helper) +implements a controlled rank-reduced bilinear output with one arbitrary +dirty helper. A two-pass dirty traversal selects its high-address block, +and the usual four-call, two-indicator echo extracts the selected table +entry. All those exact native words, phases, and arbitrary-input return +identities remain unchanged here. + +For a Q-row table, $`Q=2^r`$, with $`m\ge1`$ output bits, allocate +H and J indicator outputs and $`K=Q/(HJ)`$ high-address blocks. All +three sizes are powers of two. The generic proved ledger is + +```math +T=O(Km\min(H,J)+(H+J)P),\qquad +G=O(Qm+(H+J)P),\qquad +D_T=O(Km+D_H+D_J), +``` + +where $`P=(a+2)^3`$ for positive integer chunk cap a, and +$`D_H,D_J`$ are the two chunked indicator depths. Let the absolute +constant $`c_I\ge1`$ bound each indicator's T-count, Clifford count, +and dirty width, including its output word, by its output length times +$`c_IP`$. A sufficient simultaneous extra width is + +```math +c_I(H+J)P+p+1,\qquad p=\log_2K. +``` + +The last p bits are arbitrary dirty traversal selectors and the final +bit is the separate controlled-bilinear helper. Keep all these registers +disjoint from the address, output, and existing compiler reservation. +Assume the same sufficient extra-width condition + +```math +B\ge16c_I(P+r+1). +``` + +For $`Q\ge2`$, choose J as the largest power of two at most + +```math +\min\!\left\{Q,\sqrt{Qm},\frac{B}{8c_IP}\right\}, +``` + +and then set + +```math +H=\min\{J,Q/J\},\qquad K=Q/(HJ). +``` + +Both H and K are valid powers of two, $`H\le J`$, and $`HJ\le Q`$. +Thus these define disjoint low-address fields of lengths +$`\log_2H,\log_2J`$ and a remaining high field of length p. The +indicator pools use at most B divided by four; $`p+1\le r+1`$ +fits the remaining reservation. For $`Q=1`$, emit its sole row as a +Clifford X word with no work. + +Since $`H\le J`$, the selected bilinear contribution is +$`KmH=Qm/J`$. Power-of-two rounding yields + +```math +\frac{Qm}{J} +=O\!\left(m+\sqrt{Qm}+\frac{QmP}{B}\right). +``` + +The indicators have $`(H+J)P\le2JP\le2\sqrt{Qm}P`$ cost. +Consequently the same exact query has + +```math +T=O\!\left(m+\sqrt{Qm}P+\frac{QmP}{B}\right),\qquad +G=O(Qm+\sqrt{Qm}P). +``` + +The standalone +m count term is needed when $`m>Q`$: then the +cap $`J\le Q`$ prevents $`Qm/J`$ from falling below m. It will +be charged by the unweighted tail sum. + +The count cap preserves the claimed block-depth order. If +$`J\ge\sqrt Q`$, then $`H=Q/J`$ and $`K=1`$. Otherwise +$`H=J`$. If a count cap determines J, then +$`J\gt\sqrt Q/2`$, so $`K\lt4`$. If the width cap determines +J, then $`J\gt B/(16c_IP)`$ and +$`K\lt256c_I^2QP^2/B^2`$. This proves in all cases + +```math +K=O\!\left(1+\frac{QP^2}{B^2}\right),\qquad +D_T=O\!\left(m\left[1+\frac{QP^2}{B^2}\right]+D_H+D_J\right). +``` + +No initialized indicator, copied dirty-bit promise, or new native +controlled gate is introduced. Unequal indicator address lengths are +allowed by the generic echo and by the chunked-indicator theorem. The +matrix-coordinate changes are still charged separately for every +block and output bit, so G retains its $`Qm`$ term. + +## 3. Uniform tail bounds at the original capped precisions + +Write $`L'=L+2`$, the precision parameter for accuracy eta divided +by four. On a tail layer k steps from the leaves, use the original +full-input operator source at + +```math +m_k=L'+4+\min\{k,\lceil\log_2(8n)\rceil\},\qquad +Q_k=4N2^{-k},\qquad r_k=n-k+2. +``` + +Let the tail have at most M layers. Set + +```math +a_k=\min\{n+2,2^{\lfloor k/12\rfloor}\},\qquad +P_k=(a_k+2)^3. +``` + +Then + +```math +P_k\le(n+4)^3,\qquad P_k\le27\,2^{k/4},\qquad +m_k\le L'+4+k. +``` + +Use Section 2 for the entire m-bit output word. The two indicators are +shared by all its output-bit operations. Each indicator address length +is at most $`r_k\le n+1`$, even with unequal H and J. Its depth +therefore obeys the same bound as in the balanced case, + +```math +D_H+D_J=O\!\left(n(k+1)2^{-k/12}+\log(n+2)\right). +``` + +When the exponential chunk cap is smaller than the address length, +its inverse is within a fixed factor of $`2^{-k/12}`$ and its +logarithm is $`O(k+1)`$. Otherwise one chunk suffices and its depth +is logarithmic in the address length. A zero-bit indicator is Clifford. +Thus each completed tail query has + +```math +T_k=O\!\left(m_k+\sqrt{Q_km_k}P_k+ + \frac{Q_km_kP_k}{B}\right), +``` + +```math +G_k=O(Q_km_k+\sqrt{Q_km_k}P_k), +``` + +```math +D_{T,k}=O\!\left(m_k+\frac{Q_km_kP_k^2}{B^2} + +n(k+1)2^{-k/12}+\log(n+2)\right). +``` + +Every query fits B if +$`B\ge16c_I((n+4)^3+n+2)`$. This sufficient condition will be used +only in the large-width branch below. Actual inverses have the same +cost and return all their arbitrary dirty work. + +The weighted sums have universal constants for $`L'\ge6`$: + +```math +\sum_{k=1}^{M}\sqrt{Q_km_k}P_k=O(\sqrt{NL'}), +``` + +```math +\sum_{k=1}^{M}Q_km_kP_k=O(NL'),\qquad +\sum_{k=1}^{M}Q_km_kP_k^2=O(NL'),\qquad +\sum_{k=1}^{M}Q_km_k=O(NL'). +``` + +For the first sum, factor out $`\sqrt{NL'}`$ and sum +$`\sqrt{1+(k+4)/L'}\,2^{-k/4}`$. For the P-squared sum, it +suffices to sum $`(L'+4+k)2^{-k/2}`$; the other two decay faster. +All bounds hold uniformly for every $`M\le n`$. + +For the unweighted terms retain +$`m_k\le L'+4+\lceil\log_2(8n)\rceil`$. The finite sum +$`\sum_{k\ge1}(k+1)2^{-k/12}`$ gives the tail-query ledger + +```math +T=O\!\left(\sqrt{NL'}+\frac{NL'}B+ + M[L'+\log(n+2)]\right),\qquad G=O(NL'), +``` + +```math +D_T=O\!\left(\frac{NL'}{B^2}+n+ + M[L'+\log(n+2)]\right). +``` + +The constant number of queries in each amplified source layer changes +only constants. These replace the original table words exactly on the +full input space; the source's complete rejected action and error +certificate are unchanged. Serial source words and suffix predicates +add $`O(M[L'+\log(n+2)])`$ depth and polynomial local counts in +the regime where this lemma will be composed. + +## 4. Prove the stronger low-precision corollary + +Assume $`6\le L\le\log_2(n+2)/16`$. First consider + +```math +b\le\sqrt N/n. +``` + +The [original amortized theorem](AMORTIZED_DIRTY_LOOKUP.md#workspace-and-the-simultaneous-resource-ledger) +already supplies the required counts and depth +$`O(NL/b^2+nL+n^2)`$. Here $`NL/b^2\ge Ln^2`$ absorbs both +additive depth terms. Also $`nL=O(\sqrt{NL})`$ uniformly in the +present low-precision range: divide by $`\sqrt{NL}`$ and use +$`L\le\log_2(n+2)/16`$. The function +$`n\sqrt{\log_2(n+2)}/2^{n/2}`$ is uniformly bounded. Thus this +branch already obeys the stronger corollary. + +For the remaining case let $`b\gt\sqrt N/n`$. For sufficiently +large n, with an absolute threshold specified by the fixed circuit +constants, use the unary early groups followed by Section 3's tail. +The following estimates make the uniformity in eta explicit. + +### A cutoff uniformly smaller than n + +Choose q to be the least power of two satisfying + +```math +q\ge\frac{4\pi\sqrt{2n}}\eta, +\qquad \delta=\frac\eta{8n},\qquad \rho=\log_2 3. +``` + +Since $`1/\eta\le2^L`$, and because power-of-two rounding costs +at most a factor two, + +```math +q=O(2^L\sqrt n)=O((n+2)^{9/16}). +``` + +Use the unary local lemma's absolute reservation constant A, increased +to at least one, and its natural cutoff + +```math +M=\left\lceil16A(q^\rho+q\log_2q+\log_2q+1)\right\rceil, +\qquad C=16A. +``` + +This satisfies, uniformly over the entire low-precision range, + +```math +M=O\!\left((n+2)^{9\rho/16} + +(n+2)^{9/16}\log(n+2)\right), +\qquad \frac{9\rho}{16}\lt0.892\lt1, +``` + +```math +M[L+\log(n+2)]=o(n). +``` + +Thus $`M\lt n`$ for all sufficiently large n with one absolute +threshold, not an eta-dependent threshold. At remaining height +$`k>M`$, use + +```math +g=\left\lfloor\log_2\frac{k}{Cq}\right\rfloor, +\qquad w=q(2^g-1). +``` + +The existing unary reservation proof applies unchanged: its conditional +suffix work fits within k divided by eight, leaving the required outer +suffix. Its group height has + +```math +g\le\log_2q,\qquad +g\ge\lfloor(\rho-1)\log_2q\rfloor=\Omega(\log(n+2)). +``` + +All constants here are absolute. The upper bound uses $`q^2\ge n`$; +the lower bound uses the q-to-the-rho term in M. There are consequently +$`O(n/\log(n+2))`$ early groups. + +### Charge all early work uniformly + +The native phase-state preparation uses $`\ell=\log_2q`$ disjoint +one-qubit words with precision $`\delta/\ell`$. Its word-length +parameter is + +```math +1+\log_2((\ell+1)/\delta)=O(L+\log(n+2))=O(\log(n+2)). +``` + +This follows from $`1/\delta=8n/\eta\le8n2^L`$. Preparation, +actual unpreparation, unary conversion, internal translations, and +selectors therefore have $`O(\log(n+2))`$ T-depth per early group. +Prefetch, unload, and the outer-predicate pair have that same order. +Their total early depth is $`O(n)`$. + +The largest local source, convolution, selector, and preparation costs +are a fixed polynomial in n, uniformly over this precision range. +Summing them over at most n groups is still a fixed polynomial. Those +terms fit $`O(\sqrt{NL})`$ T gates and $`O(NL)`$ Clifford gates +with universal constants. This statement charges the actual precision +of every preparation word; it does not hide an eta-dependent native +synthesis cost. + +The early query program has $`w\le k/C`$. Its independently returned +parallel dirty pools have the same bound as in the unary proof, + +```math +O\!\left(\sqrt N(n+2)^3 + \sum_{k>M}k2^{-k/2}\right)=o(\sqrt N/n). +``` + +This estimate is uniform because $`q=\Omega(\sqrt n)`$ and hence +$`M=\Omega(n^{\rho/2})`$, with absolute constants. Their total +T-count is $`O(\sqrt N)`$ and their total Clifford count is +$`O(N)`$. They fit the large-width branch's available dirty pool and +remain disjoint from the logical suffix supplying conditional-zero work. + +### Allocate the tail and preserve the literal width threshold + +Let $`L'=L+2`$, $`B'_0=L'+n+7`$, and reserve two separate dirty +predicate helpers. The extra query pool has size + +```math +B=b-B'_0-2=b-B_0-4\ge b/2. +``` + +The last inequality follows from the original +$`b\ge17B_0`$. On this branch it gives +$`B\gt\sqrt N/(2n)`$. For sufficiently large n this exceeds +$`16c_I((n+4)^3+n+2)`$, with an absolute threshold. Thus every +unequal-indicator tail query satisfies Section 2's width condition. +The source base, predicate helpers, clean flags, and live query registers +are separately reserved; a returned query pool is reused only after +its completed call or actual inverse. + +After early grouping, at most M individual layers remain. Section 3 +therefore gives tail query depth + +```math +O\!\left(\frac{NL}{b^2}+n+M[L+\log(n+2)]\right) +=O\!\left(\frac{NL}{b^2}+n\right). +``` + +The sources and predicates add only the displayed sublinear term. +The query counts are +$`O(\sqrt{NL}+NL/b+M[L+\log(n+2)])`$ T gates and +$`O(NL)`$ Cliffords. The unweighted term, and the polynomial local +source/predicate counts, are absorbed into the same budgets. Combining +with the early groups proves all three low-precision bounds on one +circuit. + +### Full error and the universal finite-size fallback + +Early angle rounding costs at most eta divided by four by the +[complete-frame angular square-sum bound](HOPF_ERROR_ACCUMULATION.md#1-a-sharp-angle-error-bound-on-the-full-frame). +The native source preparation and actual inverse cost at most +$`2\delta`$ for an entire unary group, hence at most eta divided +by four over all early groups. This includes the source's full return, +not only its accepted action. The remaining capped source layers at +accuracy eta divided by four cost at most eta divided by four. +Their queries are exact replacements and add no error. Unitary +telescoping includes all leakage and reference correlations without +resetting any intermediate work. The final error is at most eta. + +Every sufficiently-large-n threshold above is absolute under the stated +low-precision restriction. For the finitely many smaller n, use the +existing all-precision hybrid at accuracy eta and width +$`b\ge17B_0`$. Its additive terms +$`nL+n\log(n+2)`$ are $`O(n)`$ on that finite low-precision set, +with one absolute constant. Its nL count term is uniformly absorbed by +$`\sqrt{NL}`$ as above. This supplies the stronger corollary for all +its stated n without an eta-dependent hidden constant or a larger +workspace threshold. + +## 5. Complete the uniform theorem and its matching statements + +When $`L\gt\log_2(n+2)/16`$, use the already proved +[all-precision hybrid](PARALLEL_DIRTY_LOOKUP.md#the-improved-matching-range). +Its same-circuit resources are + +```math +T=O\!\left(\sqrt{NL}+\frac{NL}{b}+nL\right),\qquad G=O(NL), +``` + +```math +D_T=O\!\left(\frac{NL}{b^2}+nL+n\log(n+2)\right). +``` + +Now $`n\log_2(n+2)\lt16nL`$, so its depth is +$`O(NL/b^2+nL)`$ with an absolute constant. On the complementary +range, Section 4's stronger result implies the uniform theorem since +$`L\ge6`$. Selecting these circuits by their declared precision +proves Section 1 for every eta and n. + +Under $`b\ge17B_0`$, physical width is $`\Theta(b)`$, and the +inherited complete-frame lower bounds give + +```math +T^\star=\Omega\!\left(\sqrt{NL}+L+\frac{NL}{b}\right),\qquad +D_T^\star=\Omega(NL/b^2). +``` + +For $`b\le\sqrt{N/n}`$, the general depth term nL is at most +$`NL/b^2`$. Also $`b\le\sqrt{NL}`$, so the square-root count +term is at most $`NL/b`$, and $`nL\le NL/b^2\le NL/b`$. +This proves the general matching interval. + +In the low-precision range, use its stronger depth and count bounds. +For $`b\le\sqrt{NL/n}`$, n is at most $`NL/b^2`$ and +$`\sqrt{NL}\le NL/b`$. This proves its larger simultaneous +matching interval. Both statements require their intervals to be +nonempty and concern worst-case complete frames. Neither asserts that +the additive depth term is necessary beyond that interval. + +## 6. Evidence and contribution boundary + +The only new local choice is the count-efficient cap on J and the +complementary definition of H. The controlled bilinear helper echo, +dirty block traversal, four-call indicator echo, and native phase +identities are exactly those of the blocked-query proof. No native +algorithm or correctness contract is altered. The tail source remains +the original full-input source at its proved capped precisions, and +the early source remains the proved unary group with its charged +preparation and actual inverse. + +The [bounded allocation checks](../tests/test_rectangular_query_allocation.py) +use integer arithmetic to check power-of-two rounding, asymmetric +address fields, sufficient dirty width, and the two branches of the +block-depth bound. They include odd address lengths, output words longer +than the table, a Clifford single-row query, and a balanced-allocation +family with an exact square-root-of-precision count penalty. The +[existing native query checks](../tests/test_blocked_bilinear_lookup.py) +cover literal phases and arbitrary helper return. Neither finite checks +nor integer resource ledgers replace the analytic uniform sums and +precision split proved above; they do not emit a scalable complete-frame +compiler or establish a new unrestricted depth lower bound. + +The [source map](SOURCE_MAP.md) records this allocation and composition +as R45; the [related-work scope](RELATED_WORK.md#21-uniform-precision-and-rectangular-query-allocation-3-october-2026) +identifies its inherited premises. No new external construction or +priority claim is needed. diff --git a/docs/VERIFICATION.md b/docs/VERIFICATION.md index 896094b..2f225f3 100644 --- a/docs/VERIFICATION.md +++ b/docs/VERIFICATION.md @@ -223,6 +223,21 @@ fixtures do not emit the scalable chunked indicators or establish the width-sensitive frame theorem by extrapolation; that resource and composition argument remains analytic. +The [rectangular allocation checks](../tests/test_rectangular_query_allocation.py) +audit the [uniform-precision allocation](UNIFORM_PRECISION_DEPTH.md) +with five exact arithmetic checks and the indicator constant normalized +to one. An independent search over feasible powers of two checks the +floor-based choice, address partition, simultaneous indicator pools, +and traversal reservation. Integer comparisons, including squared +positive remainders, check the block-count, selected-count, and indicator +bounds without approximating square roots. Boundary cases include odd +address length, output precision exceeding the row count, width-cap +transitions, and the helper-free single-row Clifford word. An exact +family shows the square allocation's square-root precision penalty in +its selected-middle cost. These finite certificates neither assign +unit constants to physical indicators nor establish asymptotic bounds +by fitted data; they test the allocation proof's arithmetic interfaces. + The [dirty-counter checks](../tests/test_counter_dirty_indicator.py) audit the two-adder signed increment, both modular-adder actions, cyclic routing, nested full-input echoes, actual inverses, and parallel native diff --git a/tests/README.md b/tests/README.md index f5f26b1..85e712c 100644 --- a/tests/README.md +++ b/tests/README.md @@ -58,6 +58,7 @@ theorem by numerical extrapolation. | [`test_parallel_dirty_lookup.py`](test_parallel_dirty_lookup.py) | Scratch-free routed indicators, actual-inverse orientation, literal phases, symbolic all-input return, complete native queries, and disjoint T layers | | [`test_bilinear_dirty_lookup.py`](test_bilinear_dirty_lookup.py) | Rectangular bilinear basis changes, shared-target native phases, the two-indicator query echo, actual inverses, and arbitrary dirty-input return; uses existing routed indicators | | [`test_blocked_bilinear_lookup.py`](test_blocked_bilinear_lookup.py) | Native dirty-controlled bilinear phases and selected blocks, symbolic blocked queries on every input variable, actual inverses, emitted T-layer ledgers, and missing-helper/selection counterchecks; no scalable chunked-query emitter | +| [`test_rectangular_query_allocation.py`](test_rectangular_query_allocation.py) | Exact power-of-two allocation, rectangular address partition, simultaneous pool reservation, integer resource inequalities, width/precision cap boundaries, and the square-allocation penalty; normalized analytic ledgers, no new native word | | [`test_dirty_indicator_depth.py`](test_dirty_indicator_depth.py) | Exact emitted parity phases, two-layer dirty-helper Toffoli, complete six-wire indicator, actual inverses, and no extra helper | | [`test_counter_dirty_indicator.py`](test_counter_dirty_indicator.py) | Two-adder signed increments, TTK and shortened RV modular-adder actions, cyclic echoes, arbitrary dirty offsets, full indicator return, and shared-address native scheduling; optimized RV depth is analytic | | [`test_readonly_dirty_increment.py`](test_readonly_dirty_increment.py) | Two-dirty-bit involution increment, both literal polarities, full native phases and actual inverses, shared-address scheduling, and wrong-order decrement witness; optimized increment depth is analytic | diff --git a/tests/test_rectangular_query_allocation.py b/tests/test_rectangular_query_allocation.py new file mode 100644 index 0000000..e862340 --- /dev/null +++ b/tests/test_rectangular_query_allocation.py @@ -0,0 +1,145 @@ +"""Exact finite certificates for the rectangular blocked-query allocation. + +These checks use the normalized abstract indicator ledger c_I=1. They +audit power-of-two rounding, reservations, and integer consequences of +the analytic inequalities; they do not assert that a literal indicator +has unit gate/width constants. The native query word is unchanged and +is checked in test_blocked_bilinear_lookup.py. No fitted asymptotic +exponent, floating square root, or large circuit simulation is used. +""" +from __future__ import annotations + +from math import isqrt +import unittest + +try: + from .test_operator_source_compiler import _basis_action +except ImportError: + from test_operator_source_compiler import _basis_action + + +def _allocation(r, m, a, budget): + """Test-only arithmetic transcription, checked against candidate search.""" + assert r >= 0 and m >= 1 and a >= 1 and budget >= 0 + q, overhead = 1 << r, (a + 2) ** 3 + if q == 1: + return dict(q=q, p=0, h=1, j=1, k=1, pool=0, helpers=0, direct=True) + assert budget >= 16 * (overhead + r + 1) + cap = min(q, isqrt(q * m), budget // (8 * overhead)) + j = 1 << (cap.bit_length() - 1) + h = min(j, q // j) + k = q // (h * j) + p = k.bit_length() - 1 + return dict(q=q, p=p, h=h, j=j, k=k, pool=overhead * (h + j), + helpers=p + 1, direct=False) + + +def _cases(): + """Finite edges around every width-cap transition through r=12.""" + for r in range(1, 13): + q = 1 << r + precisions = sorted({1, 2, 3, 4, 5, 7, 8, 9, 15, 16, 17, + q - 1, q, q + 1, 4 * q + 3}) + for a in (1, 2, 3, 7): + overhead = (a + 2) ** 3 + minimum = 16 * (overhead + r + 1) + budgets = {minimum, minimum + 1} + for exponent in range(r + 2): + transition = 8 * overhead * (1 << exponent) + budgets.update(b for b in (transition - 1, transition, transition + 1) + if b >= minimum) + for m in precisions: + for budget in sorted(budgets): + yield r, m, a, budget + + +class RectangularQueryAllocationTests(unittest.TestCase): + def test_exact_floor_choice_partition_and_simultaneous_reservation(self): + for r, m, a, budget in _cases(): + f = _allocation(r, m, a, budget) + q, overhead = 1 << r, (a + 2) ** 3 + # Independent finite feasibility search avoids relying on the + # floor/square-root implementation of the claimed allocation. + feasible = [1 << exponent for exponent in range(r + 1) + if (1 << (2 * exponent)) <= q * m + and 8 * overhead * (1 << exponent) <= budget] + self.assertEqual(f['j'], max(feasible)) + self.assertEqual(f['h'] * f['j'] * f['k'], q) + self.assertEqual(sum(x.bit_length() - 1 for x in (f['h'], f['j'], f['k'])), r) + self.assertTrue(all(x > 0 and x & (x - 1) == 0 + for x in (f['h'], f['j'], f['k']))) + self.assertLessEqual(f['h'], f['j']) + self.assertLessEqual(4 * f['pool'], budget) + self.assertLessEqual(16 * f['helpers'], budget) + self.assertLessEqual(16 * (f['pool'] + f['helpers']), 5 * budget) + + def test_integer_block_count_selected_count_and_indicator_bounds(self): + for r, m, a, budget in _cases(): + f = _allocation(r, m, a, budget) + q, overhead = 1 << r, (a + 2) ** 3 + self.assertLessEqual(f['k'] * budget ** 2, + max(4 * budget ** 2, 256 * q * overhead ** 2)) + selected = m * q // f['j'] + self.assertEqual(selected, f['k'] * m * f['h']) + # S <= 2m + 2 sqrt(Qm) + 16 QmP/B, with no irrational + # approximation: square only a strictly positive excess. + excess = selected * budget - 2 * m * budget - 16 * q * m * overhead + if excess > 0: + self.assertLessEqual(excess ** 2, 4 * budget ** 2 * q * m) + self.assertLessEqual((f['j'] * overhead) ** 2, q * m * overhead ** 2) + # The Clifford coordinate-change price is unchanged by shape. + self.assertEqual(f['k'] * m * f['h'] * f['j'], q * m) + + def test_odd_addresses_precision_cap_and_width_transition(self): + # An odd address does not force a high-block bit after rectangular + # allocation: the extra output precision permits H=4,J=8,K=1. + f = _allocation(r=5, m=2, a=1, budget=10_000) + self.assertEqual((f['h'], f['j'], f['k']), (4, 8, 1)) + # When Q> bit) & 1] + self.assertTrue(all(gate[1] < m for gate in word)) + for output in range(1 << m): + self.assertEqual(_basis_action(output, word), output ^ row) + self.assertEqual(_basis_action(output ^ row, list(reversed(word))), output) + + def test_balanced_square_has_exact_sqrt_precision_loss_in_selected_cost(self): + # This is a negative control for the selected-middle upper ledger, + # not a lower bound for arbitrary lookup circuits or total T-count. + for s in range(2, 7): + for t in range(1, s + 1): + q, m, overhead = 1 << (2 * s), 1 << (2 * t), 27 + budget = max(16 * (overhead + 2 * s + 1), + 8 * overhead * (1 << (s + t))) + f = _allocation(2 * s, m, 1, budget) + self.assertEqual((f['h'], f['j'], f['k']), + (1 << (s - t), 1 << (s + t), 1)) + rectangular_selected = m * q // f['j'] + balanced_selected = m * (1 << s) + self.assertEqual(rectangular_selected ** 2, q * m) + self.assertEqual(balanced_selected, rectangular_selected * (1 << t)) + self.assertEqual((balanced_selected // rectangular_selected) ** 2, m) + + +if __name__ == '__main__': + unittest.main()