<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:content="http://purl.org/rss/1.0/modules/content/">
  <channel>
    <title>Mixture of Experts on Husky&#39;Log</title>
    <link>https://huskydoge.github.io/husky-blog/tags/mixture-of-experts/</link>
    <description>Recent content in Mixture of Experts on Husky&#39;Log</description>
    <image>
      <title>Husky&#39;Log</title>
      <url>https://huskydoge.github.io/avatar.png</url>
      <link>https://huskydoge.github.io/avatar.png</link>
    </image>
    <generator>Hugo -- 0.144.2</generator>
    <language>en</language>
    <lastBuildDate>Fri, 31 Jul 2026 00:00:00 -0700</lastBuildDate>
    <atom:link href="https://huskydoge.github.io/husky-blog/tags/mixture-of-experts/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Towards Looped Models Done Right</title>
      <link>https://huskydoge.github.io/husky-blog/posts/recursive_models/towards-looped-models-done-right/</link>
      <pubDate>Fri, 31 Jul 2026 00:00:00 -0700</pubDate>
      <guid>https://huskydoge.github.io/husky-blog/posts/recursive_models/towards-looped-models-done-right/</guid>
      <description>&lt;figure class=&#34;align-center &#34;&gt;&lt;img loading=&#34;lazy&#34; src=&#34;https://huskydoge.github.io/husky-blog/posts/recursive_models/towards-looped-models-done-right/cover.png#center&#34; class=&#34;zoomable-figure-image&#34; data-image-zoom-src=&#34;https://huskydoge.github.io/husky-blog/posts/recursive_models/towards-looped-models-done-right/cover.png&#34; tabindex=&#34;0&#34; role=&#34;button&#34;
             alt=&#34;Mohamed bin Zayed University of Artificial Intelligence Institute of Foundation Models banner&#34;/&gt;
&lt;/figure&gt;

&lt;p&gt;&lt;strong&gt;&lt;a href=&#34;https://huskydoge.github.io/&#34;&gt;Benhao Huang&lt;/a&gt;&lt;/strong&gt;‡†, Chufan Shi‡*, Junlin Chen‡, Shicheng Wen‡*,&lt;/p&gt;
&lt;p&gt;Zhengzhong Liu‡, Eric Xing‡, Xuezhe Ma‡*&lt;/p&gt;
&lt;p&gt;&lt;strong&gt;‡Institute of Foundation Models, *USC, †CMU&lt;/strong&gt;&lt;/p&gt;
&lt;hr&gt;
&lt;figure class=&#34;align-center &#34;&gt;&lt;img loading=&#34;lazy&#34; src=&#34;https://huskydoge.github.io/husky-blog/posts/recursive_models/towards-looped-models-done-right/fig-1-loop-moe-scaling.png#center&#34; class=&#34;zoomable-figure-image&#34; data-image-zoom-src=&#34;https://huskydoge.github.io/husky-blog/posts/recursive_models/towards-looped-models-done-right/fig-1-loop-moe-scaling.png&#34; tabindex=&#34;0&#34; role=&#34;button&#34;
             alt=&#34;Parameter scaling and benchmark performance of loop and feedforward MoE models&#34;/&gt;&lt;figcaption&gt;
            &lt;p&gt;&lt;strong&gt;Fig. 1 Parameter scaling and benchmark performance of loop and feedforward MoE models.&lt;/strong&gt; &lt;strong&gt;Left&lt;/strong&gt;: &lt;em&gt;Ouro&lt;/em&gt; MoE and &lt;em&gt;Huginn&lt;/em&gt; MoE share the same scale: 8.0B resident and 0.8B active parameters, corresponding to 32.0B resident-equivalent and 3.2B unrolled-active parameter applications. Unrolled counts measure parameter applications under weight reuse. &lt;strong&gt;Right&lt;/strong&gt;: &lt;em&gt;Huginn&lt;/em&gt; MoE outperforms &lt;em&gt;Ouro MoE&lt;/em&gt; overall, and approaches or surpasses the 112-layer feedforward MoE baseline on DROP, MATH500 and GSM8K, while remaining competitive on the other benchmarks. All models use the same data and matched training and inference FLOPs, while looped models require less memory.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
