INNER CODE UNIT · Python
forward
Leeroo-AI/mergoo · mergoo/compose_layers.py:69
def forward(self, inputs: torch.Tensor):
gate_logits = self.gate(inputs)
weights, selected_experts = torch.topk(gate_logits, self.num_experts_per_tok)
weights = F.softmax(weights, dim=2, dtype=torch.float).to(inputs.dtype)
results = torch.zeros(
(inputs.shape[0], inputs.shape[1], self.out_features),
device=inputs.device,
dtype=inputs.dtype,
)
for ix, expert in enumerate(self.experts):
batch_idx, tok_idx, expert_idx = torch.where(selected_experts == ix)
results[batch_idx, tok_idx] += expert(inputs[batch_idx, tok_idx]) * weights[
batch_idx, tok_idx, expert_idx
].unsqueeze(-1)
return results
class LoRAMoeLayer(torch.nn.Module):