[CPU][Bugfix] Fix flaky ShortConv prefill test on ARM (uninitialized weights) (#47848)

Signed-off-by: Rahul Vishwakarma <Rahul.Vishwakarma2@ibm.com>
This commit is contained in:
Rahul Vishwakarma
2026-07-07 18:20:09 -07:00
committed by GitHub
parent e97c3cb303
commit f7efab58ec
@@ -60,6 +60,12 @@ def test_short_conv_forward_native_prefill(vllm_config):
layer = ShortConv(config=config, dim=dim, layer_idx=0, prefix=prefix)
layer.to("cpu")
# vLLM Linear layers allocate weights with torch.empty (uninitialized).
# On ARM these come back as zero-filled pages, so in_proj output is zero and
# the prefill state stays zero. Seed + init to make the test platform-safe.
torch.manual_seed(0)
for p in layer.parameters():
torch.nn.init.normal_(p)
dispatch_cpu_unquantized_gemm(layer.in_proj, remove_weight=False)
dispatch_cpu_unquantized_gemm(layer.out_proj, remove_weight=False)
@@ -117,6 +123,9 @@ def test_short_conv_forward_native_decode(vllm_config):
layer = ShortConv(config=config, dim=dim, layer_idx=0, prefix=prefix)
layer.to("cpu")
torch.manual_seed(0)
for p in layer.parameters():
torch.nn.init.normal_(p)
dispatch_cpu_unquantized_gemm(layer.in_proj, remove_weight=False)
dispatch_cpu_unquantized_gemm(layer.out_proj, remove_weight=False)