rust-lang / rust-lang/rust

codegen: Unnecessary `memcpy` when returning a mutated `mut self`

Open
#157,685 2 comments 0 reactions 0 assignees View on GitHub

Nobody has claimed this yet.

A-codegen A-LLVM C-bug C-optimization I-slow
Dominant language
Rust
Stars
119k
Forks
16.1k
PR merge metrics
PR metrics pending

Description

When implementing an operator by delegating to its Assign version, two formulations that should be equivalent produce different assembly. Using mut self generates a redundant memcpy compared to using let mut ret = self.

Produces suboptimal assembly (Godbolt):

use std::ops::{BitAnd, BitAndAssign};

struct Foo([u128; 1]);

impl BitAndAssign<&Self> for Foo {
    #[unsafe(no_mangle)]
    fn bitand_assign(&mut self, other: &Self) {
        self.0[0] &= other.0[0];
    }
}

impl BitAnd<&Self> for Foo {
    type Output = Foo;
    #[unsafe(no_mangle)]
    fn bitand(mut self, other: &Self) -> Self::Output {
        self &= other;
        self
    }
}
Generated LLVM IR
define void @bitand(ptr dead_on_unwind noalias nofree noundef writable writeonly sret([16 x i8]) align 16 captures(none) dereferenceable(16) initializes((0, 16)) %_0, ptr dead_on_return noalias nofree noundef align 16 captures(none) dereferenceable(16) %self, ptr noalias nofree noundef readonly align 16 captures(none) dereferenceable(16) %other) unnamed_addr {
start:
  tail call void @llvm.experimental.noalias.scope.decl(metadata !29)
  tail call void @llvm.experimental.noalias.scope.decl(metadata !33)
  %_3.i = load i128, ptr %other, align 16
  %0 = load i128, ptr %self, align 16
  %1 = and i128 %0, %_3.i
  store i128 %1, ptr %self, align 16
  tail call void @llvm.memcpy.p0.p0.i64(ptr noundef nonnull align 16 dereferenceable(16) %_0, ptr noundef nonnull align 16 dereferenceable(16) %self, i64 16, i1 false)
  ret void
}

define void @bitand_assign(ptr noalias nofree noundef align 16 captures(none) dereferenceable(16) %self, ptr noalias nofree noundef readonly align 16 captures(none) dereferenceable(16) %other) unnamed_addr {
start:
  %_3 = load i128, ptr %other, align 16
  %0 = load i128, ptr %self, align 16
  %1 = and i128 %0, %_3
  store i128 %1, ptr %self, align 16
  ret void
}

declare void @llvm.memcpy.p0.p0.i64(ptr noalias writeonly captures(none), ptr noalias readonly captures(none), i64, i1 immarg) #2

declare void @llvm.experimental.noalias.scope.decl(metadata) #3
Generated amd64 assembly:
bitand:
        mov     rax, rdi
        movaps  xmm0, xmmword ptr [rsi]
        andps   xmm0, xmmword ptr [rdx]
        movaps  xmmword ptr [rsi], xmm0   ; store to self
        movaps  xmm0, xmmword ptr [rsi]   ; load from self
        movaps  xmmword ptr [rdi], xmm0   ; store to output (should have been direct)
        ret

bitand_assign:
        movaps  xmm0, xmmword ptr [rdi]
        andps   xmm0, xmmword ptr [rsi]
        movaps  xmmword ptr [rdi], xmm0
        ret

Produces optimal assembly (Godbolt):

use std::ops::{BitAnd, BitAndAssign};

struct Foo([u128; 1]);

impl BitAndAssign<&Self> for Foo {
    #[unsafe(no_mangle)]
    fn bitand_assign(&mut self, other: &Self) {
        self.0[0] &= other.0[0];
    }
}

impl BitAnd<&Self> for Foo {
    type Output = Foo;
    #[unsafe(no_mangle)]
    fn bitand(self, other: &Self) -> Self::Output {
        let mut ret = self;
        ret &= other;
        ret
    }
}
Generated LLVM IR
define void @bitand(ptr dead_on_unwind noalias nofree noundef writable writeonly sret([16 x i8]) align 16 captures(none) dereferenceable(16) initializes((0, 16)) %_0, ptr dead_on_return noalias nofree noundef readonly align 16 captures(none) dereferenceable(16) %self, ptr noalias nofree noundef readonly align 16 captures(none) dereferenceable(16) %other) unnamed_addr {
start:
  %ret.sroa.0.0.copyload = load i128, ptr %self, align 16
  %_3.i = load i128, ptr %other, align 16
  %0 = and i128 %_3.i, %ret.sroa.0.0.copyload
  store i128 %0, ptr %_0, align 16
  ret void
}

define void @bitand_assign(ptr noalias nofree noundef align 16 captures(none) dereferenceable(16) %self, ptr noalias nofree noundef readonly align 16 captures(none) dereferenceable(16) %other) unnamed_addr {
start:
  %_3 = load i128, ptr %other, align 16
  %0 = load i128, ptr %self, align 16
  %1 = and i128 %0, %_3
  store i128 %1, ptr %self, align 16
  ret void
}
Generated amd64 assembly:
bitand:
        mov     rax, rdi
        movaps  xmm0, xmmword ptr [rdx]
        andps   xmm0, xmmword ptr [rsi]
        movaps  xmmword ptr [rdi], xmm0
        ret

bitand_assign:
        movaps  xmm0, xmmword ptr [rdi]
        andps   xmm0, xmmword ptr [rsi]
        movaps  xmmword ptr [rdi], xmm0
        ret

Observed on rustc 1.98.0-nightly (cb46fbb8c 2026-06-08)

Contributor guide

Open the contributing guide

First steps

  1. Read the whole issue, then the project's contributing guide.
  2. Comment on the issue to say you are picking it up — it saves two people doing the same work.
  3. Fork the repository and make your change on a branch.
  4. Open a pull request that references the issue number.

Research direction

Start with the two BitAnd implementations and compile them on the reported rustc nightly with optimization enabled. Compare the generated LLVM IR and amd64 assembly, using the issue's Godbolt reproducer as the baseline. Done means the mut self formulation no longer emits the redundant memcpy while preserving the correct result.

Written by the indexing model from the issue text.

Assessment

Tech stack
rust
Domain
compilers, performance
Issue type
Bug
Difficulty
4/5
Estimated time
3-5 days
Activity status
Quiet
Clarity
Mostly clear
Newbie friendliness
45/100

Get new issues in your inbox

A short digest of beginner-friendly GitHub issues.