@@ -443,34 +443,12 @@ impl Repr {
443443 self
444444 }
445445
446- /// Slow path for `Clone::clone`. Heap-resident values only
447- /// (capacity > 2 in absolute value).
446+ /// Slow path for `Clone::clone`: heap-resident values only (|capacity| > 2).
448447 ///
449- /// Allocates directly via `alloc::alloc::alloc` rather than going
450- /// through `Buffer::allocate` + `push_slice` + `mem::transmute`.
451- /// That bypass saves:
452- ///
453- /// * the redundant `0 < capacity <= MAX_CAPACITY` assertion inside
454- /// `Buffer::allocate_raw` (we just computed `new_cap` from a
455- /// bounded `len`);
456- /// * `push_slice`'s `len <= capacity - 0` check (we own the buffer
457- /// we just allocated, the only `len` ever written is the input
458- /// `len`);
459- /// * `Buffer::allocate_exact`'s second check on the same bound;
460- /// * the `mem::transmute(Buffer) -> Repr` shape and the conditional
461- /// sign-flip dance — the signed capacity is set directly.
462- ///
463- /// `#[inline(never)]` (without `#[cold]`) keeps the inline fast path
464- /// tiny at every `Clone::clone` instantiation and gives LLVM a stable
465- /// call boundary so the inline path's code-layout doesn't depend on
466- /// how aggressively the heap path gets considered for inlining.
467- ///
468- /// Removing the attribute lets LLVM occasionally inline this back —
469- /// which leaves the inline microbench unchanged but regresses
470- /// `ibig_clone/large` by ~5 pp (probably i-cache: the inlined
471- /// 30-instruction heap body lands inside the bench's hot loop).
472- /// `#[cold]` was also dropped (`shrinker_consider` gains ~3 pp without
473- /// it; see commit notes).
448+ /// Allocates directly instead of going through `Buffer::allocate` +
449+ /// `push_slice` + `mem::transmute`, skipping the redundant capacity/length
450+ /// bound checks and the sign-flip dance. Kept `#[inline(never)]` so the
451+ /// inline fast path in `clone` stays small.
474452 #[ inline( never) ]
475453 fn clone_heap ( & self ) -> Self {
476454 debug_assert ! ( self . capacity. get( ) . unsigned_abs( ) > 2 ) ;
@@ -535,15 +513,9 @@ impl Repr {
535513impl Clone for Repr {
536514 #[ inline]
537515 fn clone ( & self ) -> Self {
538- // Inline case (abs(capacity) <= 2): just copy the union and the
539- // signed capacity verbatim. Avoids the `sign_capacity` extract +
540- // `with_sign` re-apply round-trip the old implementation did.
541- //
542- // Profiled hegel-rust shrinker workloads spend ~13 % of total Ir
543- // in this function, almost entirely on the inline case (values
544- // are i64- or i128-sized). Inlining the trivial path here lets
545- // it disappear into the caller; the heavier heap path is split
546- // out into a `#[cold]` helper.
516+ // Inline case (|capacity| <= 2): copy the union and signed capacity
517+ // verbatim, avoiding the `sign_capacity` + `with_sign` round-trip the
518+ // old implementation did. The heap path is split into `clone_heap`.
547519 if self . capacity . get ( ) . unsigned_abs ( ) <= 2 {
548520 return Repr {
549521 data : ReprData {
@@ -564,11 +536,8 @@ impl Clone for Repr {
564536
565537 // SAFETY: see the comments inside the block
566538 unsafe {
567- // Fast path: src is inline. Same shape as `clone`'s fast path
568- // — copy union + signed capacity, with a rare dealloc when
569- // self was previously heap-resident. The shrinker's
570- // ChoiceNode = (IBig, IBig, IBig, IBig) clone pattern is
571- // dominated by inline-source clones.
539+ // Fast path: src is inline. Copy union + signed capacity, with a
540+ // rare dealloc when self was previously heap-resident.
572541 if src_cap_raw. unsigned_abs ( ) <= 2 {
573542 if cap > 2 {
574543 // release the old buffer if necessary
0 commit comments