<?xml version="1.0" ?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.4 20241031//EN"  "JATS-archivearticle1-4-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.4" xml:lang="en">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">elife</journal-id>
<journal-id journal-id-type="publisher-id">eLife</journal-id>
<journal-title-group>
<journal-title>eLife</journal-title>
</journal-title-group>
<issn publication-format="electronic" pub-type="epub">2050-084X</issn>
<publisher>
<publisher-name>eLife Sciences Publications, Ltd</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">101204</article-id>
<article-id pub-id-type="doi">10.7554/eLife.101204</article-id>
<article-id pub-id-type="doi" specific-use="version">10.7554/eLife.101204.2</article-id>
<article-version-alternatives>
<article-version article-version-type="publication-state">reviewed preprint</article-version>
<article-version article-version-type="preprint-version">1.5</article-version>
</article-version-alternatives>
<article-categories><subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
</subj-group>
</article-categories><title-group>
<article-title>Scale matters: Large language models with billions (rather than millions) of parameters better match neural representations of natural language</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Hong</surname>
<given-names>Zhuoqiao</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="author-notes" rid="n1">*</xref>
</contrib>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<contrib-id contrib-id-type="orcid" authenticated="true">https://orcid.org/0009-0000-8500-6054</contrib-id>
<name>
<surname>Wang</surname>
<given-names>Haocheng</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="author-notes" rid="n1">*</xref>
<email>kw1166@princeton.edu</email>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zada</surname>
<given-names>Zaid</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gazula</surname>
<given-names>Harshvardhan</given-names>
</name>
<xref ref-type="aff" rid="a2">2</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Turner</surname>
<given-names>David</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Aubrey</surname>
<given-names>Bobbi</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Niekerken</surname>
<given-names>Leonard</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Doyle</surname>
<given-names>Werner</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Devore</surname>
<given-names>Sasha</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dugan</surname>
<given-names>Patricia</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Friedman</surname>
<given-names>Daniel</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Devinsky</surname>
<given-names>Orrin</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid" authenticated="true">https://orcid.org/0000-0003-1247-1283</contrib-id>
<name>
<surname>Flinker</surname>
<given-names>Adeen</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Hasson</surname>
<given-names>Uri</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="author-notes" rid="n2">**</xref>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<contrib-id contrib-id-type="orcid" authenticated="true">https://orcid.org/0000-0001-7013-5275</contrib-id>
<name>
<surname>Nastase</surname>
<given-names>Samuel A</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="author-notes" rid="n2">**</xref>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Goldstein</surname>
<given-names>Ariel</given-names>
</name>
<xref ref-type="aff" rid="a4">4</xref>
<xref ref-type="author-notes" rid="n2">**</xref>
</contrib>
<aff id="a1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00hx57361</institution-id><institution>Department of Psychology and the Neuroscience Institute, Princeton University</institution></institution-wrap>, <city>Princeton</city>, <country country="US">United States</country></aff>
<aff id="a2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/042nb2s44</institution-id><institution>McGovern Institute for Brain Research, Massachusetts Institute of Technology</institution></institution-wrap>, <city>Cambridge</city>, <country country="US">United States</country></aff>
<aff id="a3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0190ak572</institution-id><institution>New York University Grossman School of Medicine</institution></institution-wrap>, <city>New York</city>, <country country="US">United States</country></aff>
<aff id="a4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03qxff017</institution-id><institution>Business School, Data Science Department and Cognitive Science Department, Hebrew University</institution></institution-wrap>, <city>Jerusalem</city>, <country country="IL">Israel</country></aff>
</contrib-group>
<contrib-group content-type="section">
<contrib contrib-type="editor">
<name>
<surname>Ding</surname>
<given-names>Nai</given-names>
</name>
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-3428-2723</contrib-id><role>Reviewing Editor</role>
<aff>
<institution-wrap>
<institution-id institution-id-type="ror">https://ror.org/00a2xv884</institution-id><institution>Zhejiang University</institution>
</institution-wrap>
<city>Hangzhou</city>
<country country="CN">China</country>
</aff>
</contrib>
<contrib contrib-type="senior_editor">
<name>
<surname>Bi</surname>
<given-names>Yanchao</given-names>
</name>
<contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0522-3372</contrib-id><role>Senior Editor</role>
<aff>
<institution-wrap>
<institution-id institution-id-type="ror">https://ror.org/02v51f717</institution-id><institution>Peking University</institution>
</institution-wrap>
<city>Beijing</city>
<country country="CN">China</country>
</aff>
</contrib>
</contrib-group>
<author-notes>
<fn id="n1" fn-type="equal"><label>*</label><p>Equal first author, alphabetical order</p></fn>
<fn id="n2" fn-type="equal"><label>**</label><p>Equal senior author</p></fn>
<fn fn-type="coi-statement"><p>Competing interests: No competing interests declared</p></fn>
</author-notes>
<pub-date date-type="original-publication" iso-8601-date="2024-10-22">
<day>22</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date date-type="update" iso-8601-date="2026-08-04">
<day>04</day>
<month>08</month>
<year>2026</year>
</pub-date>
<volume>13</volume>
<elocation-id>RP101204</elocation-id>
<pub-history>
<event>
<event-desc>Sent for review</event-desc>
<date date-type="sent-for-review" iso-8601-date="2024-07-18">
<day>18</day>
<month>07</month>
<year>2024</year>
</date>
</event>
<event>
<event-desc>Preprint posted</event-desc>
<date date-type="preprint" iso-8601-date="2024-10-16">
<day>16</day>
<month>10</month>
<year>2024</year>
</date>
<self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2024.06.12.598513"/>
</event>
<event>
<event-desc>Reviewed preprint v1</event-desc>
<date date-type="reviewed-preprint" iso-8601-date="2024-10-22">
<day>22</day>
<month>10</month>
<year>2024</year>
</date>
<self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.101204.1"/>
<self-uri content-type="editor-report" xlink:href="https://doi.org/10.7554/eLife.101204.1.sa4">eLife Assessment</self-uri>
<self-uri content-type="referee-report" xlink:href="https://doi.org/10.7554/eLife.101204.1.sa3">Reviewer #1 (Public review):</self-uri>
<self-uri content-type="referee-report" xlink:href="https://doi.org/10.7554/eLife.101204.1.sa2">Reviewer #2 (Public review):</self-uri>
<self-uri content-type="referee-report" xlink:href="https://doi.org/10.7554/eLife.101204.1.sa1">Reviewer #3 (Public review):</self-uri>
<self-uri content-type="author-comment" xlink:href="https://doi.org/10.7554/eLife.101204.1.sa0">Author response:</self-uri>
</event>
</pub-history>
<permissions>
<copyright-statement>© 2024, Hong et al</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Hong et al</copyright-holder>
<ali:free_to_read/>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<ali:license_ref>https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="elife-preprint-101204-v2.pdf"/>
<abstract><p>Recent research has used large language models (LLMs) to study the neural basis of naturalistic language processing in the human brain. LLMs have rapidly grown in complexity, leading to improved language processing capabilities. However, neuroscience researchers haven’t kept up with the quick progress in LLM development. Here, we utilized several families of transformer-based LLMs to investigate the relationship between model size and their ability to capture linguistic information in the human brain. Crucially, a subset of LLMs were trained on a fixed training set, enabling us to dissociate model size from architecture and training set size. We used electrocorticography (ECoG) to measure neural activity in epilepsy patients while they listened to a 30-minute naturalistic audio story. We fit electrode-wise encoding models using contextual embeddings extracted from each hidden layer of the LLMs to predict word-level neural signals. In line with prior work, we found that larger LLMs better capture the structure of natural language and better predict neural activity. We also found a logarithmic relationship where the encoding performance peaks in relatively earlier layers as model size increases. We also observed variations in the best-performing layer across different brain regions, corresponding to an organized language processing hierarchy.</p>
</abstract>
<funding-group>
<award-group id="par-1">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id>
<institution>HHS | National Institutes of Health (NIH)</institution>
</institution-wrap>
</funding-source>
<award-id>DP1HD091948</award-id>
<principal-award-recipient>
<name>
<surname>Hong</surname>
<given-names>Zhuoqiao</given-names>
</name><name>
<surname>Wang</surname>
<given-names>Haocheng</given-names>
</name><name>
<surname>Hasson</surname>
<given-names>Uri</given-names>
</name><name>
<surname>Goldstein</surname>
<given-names>Ariel Y</given-names>
</name>
</principal-award-recipient>
</award-group>



<award-group id="par-5">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id>
<institution>HHS | National Institutes of Health (NIH)</institution>
</institution-wrap>
</funding-source>
<award-id>R01DC022534</award-id>
<principal-award-recipient>
<name>
<surname>Nastase</surname>
<given-names>Samuel</given-names>
</name>
</principal-award-recipient>
</award-group>
<award-group id="par-6">
<funding-source>
<institution-wrap>
<institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id>
<institution>HHS | National Institutes of Health (NIH)</institution>
</institution-wrap>
</funding-source>
<award-id>R01NS109367</award-id>
<principal-award-recipient>
<name>
<surname>Flinker</surname>
<given-names>Adeen</given-names>
</name>
</principal-award-recipient>
</award-group>
</funding-group>
<custom-meta-group>
<custom-meta specific-use="meta-only">
<meta-name>publishing-route</meta-name>
<meta-value>prc</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
<notes>
<fn-group content-type="summary-of-updates">
<title>Summary of Updates:</title>
<fn fn-type="update"><p>Revised manuscript in response to reviewers. Added new analysis and updated figures.</p></fn>
</fn-group>
</notes>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>How has the functional architecture of the human brain come to support everyday language processing? Modeling the underlying neural basis that supports natural language processing has proven to be prohibitively challenging for many years. Deep learning has brought about a transformative shift in our ability to model natural language in recent years. Leveraging principles from statistical learning theory and using vast real-world datasets, deep learning algorithms can reproduce complex natural behaviors in visual perception, speech analyses, and even human-like conversations. With the recent emergence of large language models (LLMs), we are finally beginning to see explicit computational models that respect and reproduce the context-rich complexity of natural language and communication. LLMs rely on simple self-supervised objectives (e.g., next-word prediction) to learn to produce context-specific linguistic outputs from real-world corpora—and, in the process, implicitly encode the statistical structure of natural language into a multidimensional embedding space (<xref ref-type="bibr" rid="c29">Linzen &amp; Baroni, 2021</xref>; <xref ref-type="bibr" rid="c31">Manning et al., 2020</xref>; <xref ref-type="bibr" rid="c35">Pavlick, 2022</xref>).</p>
<p>Critically, there appears to be an alignment between the internal activity in LLMs for each word embedded in a natural text and the internal activity in the human brain while processing the same natural text. Indeed, recent studies have revealed that the internal, layer-by-layer representations learned by these models predict human brain activity during natural language processing better than any previous generations of models (<xref ref-type="bibr" rid="c9">Caucheteux &amp; King, 2022</xref>; <xref ref-type="bibr" rid="c18">Goldstein et al., 2022</xref>, <xref ref-type="bibr" rid="c16">2024</xref>; <xref ref-type="bibr" rid="c27">Kumar et al., 2024</xref>; <xref ref-type="bibr" rid="c39">Schrimpf et al., 2021</xref>).</p>
<p>LLMs, however, contain millions or billions of parameters, making them highly expressive learning algorithms. Combined with vast training text, these models can encode a rich array of linguistic structures—ranging from low-level morphological and syntactic operations to high-level contextual meaning—in a high-dimensional embedding space. Recent work has argued that the “size” of these models—the number of learnable parameters—is critical, as some linguistic competencies only emerge in larger models with more parameters (<xref ref-type="bibr" rid="c4">Bommasani et al., 2021</xref>; <xref ref-type="bibr" rid="c26">Kaplan et al., 2020</xref>; <xref ref-type="bibr" rid="c31">Manning et al., 2020</xref>; <xref ref-type="bibr" rid="c41">Sutton, 2019</xref>; C. <xref ref-type="bibr" rid="c50">Zhang et al., 2021</xref>). For instance, in-context learning (<xref ref-type="bibr" rid="c30">Liu et al., 2022</xref>; <xref ref-type="bibr" rid="c47">Xie et al., 2021</xref>) involves a model acquiring the ability to carry out a task for which it was not initially trained, based on a few-shot examples provided by the prompt. This capability is present in the bigger GPT-3 (<xref ref-type="bibr" rid="c6">Brown et al., 2020</xref>) but not in the smaller GPT-2, despite both models having similar architectures. This observation suggests that simply scaling up models produces more human-like language processing. Indeed, research in comparative neuroscience has suggested that uniquely human cognitive abilities emerged from scaling up the primate brain (<xref ref-type="bibr" rid="c22">Herculano-Houzel, 2012</xref>). While building and training LLMs with billions to trillions of parameters is an impressive engineering achievement, such artificial neural networks are tiny compared to cortical neural networks. In the human brain, each cubic millimeter of cortex contains a remarkable number of about 150 million synapses, and the language network can cover a few centimeters of the cortex (<xref ref-type="bibr" rid="c7">Cantlon &amp; Piantadosi, 2024</xref>).</p>
<p>Our study focuses on one crucial question: What is the relationship between the size of an LLM and how well it can predict linguistic information encoded in the brain? In this study, we define “model size” as the number of all trainable parameters in the model. In addition to size, we also consider the model’s expressivity: its capacity to predict the statistics of natural language. Perplexity measures expressivity by evaluating the average level of surprise or uncertainty the model attributes to a sequence of words. Larger models possess a greater capacity for expressing linguistic structure, which tends to yield lower (better) perplexity scores (<xref ref-type="bibr" rid="c37">Radford et al., 2019</xref>). In this paper, we hypothesized that larger models that capture linguistic structure more accurately (lower perplexity) would better capture neural activity.</p>
<p>To test this hypothesis, we used electrocorticography (ECoG) to measure neural activity in ten epilepsy patient participants while they listened to a 30-minute audio podcast. Invasive ECoG recordings more directly measure neural activity than non-invasive neuroimaging modalities like fMRI, with much higher temporal resolution. We extracted contextual embeddings at each hidden layer from multiple families of transformer-based LLMs, including GPT-2, GPT-Neo, OPT, and Llama 2 (<xref ref-type="bibr" rid="c3">Black et al., 2022</xref>; <xref ref-type="bibr" rid="c37">Radford et al., 2019</xref>; <xref ref-type="bibr" rid="c44">Touvron et al., 2023</xref>; S. <xref ref-type="bibr" rid="c51">Zhang et al., 2022</xref>), and fit electrode-wise encoding models to predict neural activity for each word in the podcast stimulus. We found that larger language models, with greater expressivity and lower perplexity, better predicted neural activity (<xref ref-type="bibr" rid="c2">Antonello et al., 2023</xref>). This result was consistent across all model families. Critically, we then focus on a particular family of models (GPT-Neo), which span a broad range of sizes and are trained on the same text corpora. This allowed us to assess the effect of scaling on the match between LLMs and the human brain while keeping the size of the training set constant.</p>
</sec>
<sec id="s2">
<title>Results</title>
<p>To investigate scaling effects between model size and the alignment of the model internal representations (embeddings) with brain activity, we utilized four families of transformer-based language models: GPT-2, GPT-Neo, OPT, and Llama 2 (<xref ref-type="bibr" rid="c3">Black et al., 2022</xref>; <xref ref-type="bibr" rid="c37">Radford et al., 2019</xref>; <xref ref-type="bibr" rid="c44">Touvron et al., 2023</xref>; S. <xref ref-type="bibr" rid="c51">Zhang et al., 2022</xref>). These models span 82 million to 70 billion parameters and 6 to 80 layers (<xref rid="tbl1" ref-type="table">Table 1</xref>). Different families of models vary in architectural details and are trained on different text corpora. To control for these confounding variables, we also focused on the GPT-Neo family (<xref ref-type="bibr" rid="c15">Gao et al., 2020</xref>) with a comprehensive range of models that vary only in size (but not training data), spanning 125 million to 20 billion parameters. For simplicity, we renamed the four models as “SMALL” (gpt-neo-125M), “MEDIUM” (gpt-neo-1.3B), “LARGE” (gpt-neo-2.7B), and “XL” (gpt-neox-20b).</p>
<table-wrap id="tbl1" orientation="portrait" position="float">
<label>Table 1.</label>
<caption><title>Summary of four families of open large language models: GPT-2, GPT-Neo, OPT, and Llama-2.</title>
<p>Context length is the maximum context length for the model, ranging from 1024 to 4096 tokens. The model name is the model’s name as it appears in the transformers package from Hugging Face (<xref ref-type="bibr" rid="c46">Wolf et al., 2019</xref>). Model size is the total number of parameters; M represents million, and B represents billion. The number of layers is the depth of the model, and the hidden embedding size is the internal width.</p></caption>
<graphic xlink:href="598513v5_tbl1.tif" mimetype="image/tiff"/>
</table-wrap>
<p>We collected ECoG data from ten epilepsy patients while they listened to a 30-minute audio podcast (<italic>So a Monkey and a Horse Walk into a Bar</italic>, 2017). We extracted high-frequency broadband power (70–200 Hz) in 200 ms bins at lags ranging from -2000 ms to +2000 ms relative to the onset of each word in the podcast stimulus. We ran a preliminary encoding analysis using non-contextual language embeddings (GloVe; <xref ref-type="bibr" rid="c36">Pennington et al., 2014</xref>) to select a subset of 160 language-sensitive electrodes across the cortical language network of eight patients; all subsequent analyses were performed within this set of electrodes (<xref ref-type="bibr" rid="c18">Goldstein et al., 2022</xref>). Using a podcast transcription, we next extracted contextual embeddings from each hidden layer across the four families of autoregressive large language models. We used the maximum context length of each word for each language model. We constructed linear, electrode-wise encoding models using contextual embeddings from every layer of each language model to predict neural activity for each word in the stimulus. We estimated and evaluated the encoding models using a 10-fold cross-validation procedure: ridge regression was used to estimate a weight matrix for predicting word-by-word neural signals in 9 out of 10 contiguous training segments of the podcast; for each electrode, we then calculated the Pearson correlation between predicted and actual word-by-word neural signals for the left-out test segment of the podcast. We repeated this analysis for 161 lags from -2,000 ms to 2,000 ms in 25 ms increments relative to word onset (<xref rid="fig1" ref-type="fig">Fig. 1</xref>).</p>
<fig id="fig1" position="float" orientation="portrait" fig-type="figure">
<label>Figure 1.</label>
<caption><title>Naturalistic language comprehension model comparison framework.</title>
<p><bold>A.</bold> Participants listened to a 30-minute story while undergoing ECoG recording. A word-level aligned transcript was obtained and served as input to four language models of varying size from the same GPT-Neo family. <bold>B.</bold> For every layer of each model, a separate linear regression encoding model was fitted on a training portion of the story to obtain regression weights that can predict each electrode separately. Then, the encoding models were tested on a held-out portion of the story and evaluated by measuring the Pearson correlation of their predicted signal with the actual signal. <bold>C.</bold> Encoding model performance (correlations) was measured as the average over electrodes and compared between the different language models.</p></caption>
<graphic xlink:href="598513v5_fig1.tif" mimetype="image/tiff"/>
</fig>
<p>Prior to encoding analysis, we measured the “expressiveness” of different language models—that is, their capacity to predict the structure of natural language. Perplexity quantifies expressivity as the average level of surprise or uncertainty the model assigns to a sequence of words. A lower perplexity value indicates a better alignment with linguistic statistics and a higher accuracy during next-word prediction. For each model, we computed perplexity values for the podcast transcript. Consistent with prior research (<xref ref-type="bibr" rid="c24">Hosseini et al., 2022</xref>; <xref ref-type="bibr" rid="c26">Kaplan et al., 2020</xref>), we found that perplexity decreases as model size increases (<xref rid="fig2" ref-type="fig">Fig. 2A</xref>). In simpler terms, we confirmed that larger models better predict the structure of natural language.</p>
<fig id="fig2" position="float" orientation="portrait" fig-type="figure">
<label>Figure 2.</label>
<caption><title>Model performance improves with increasing model size.</title><p><bold>A.</bold> The relationship between model size (measured as the number of parameters, shown on a log scale) and perplexity: as the model size increases, perplexity decreases. Each data point corresponds to a model. <bold>B.</bold> The relationship between model size (shown on a log scale) and brain encoding performance: correlations for each model are calculated by averaging the maximum correlations across all lags and layers across electrodes. As the model size increases, the encoding performance increases. Each data point corresponds to a model. The error bars represent standard error. <bold>C.</bold> For the GPT-Neo model family, the relationship between encoding performance and layer number. Encoding performance is best for intermediate layers. The shaded colors represent standard error. <bold>D</bold>. Same as C, but the layer number was transformed to a layer percentage for better model comparison.</p></caption>
<graphic xlink:href="598513v5_fig2.tif" mimetype="image/tiff"/>
</fig>
<sec id="s2a">
<title>Larger language models better predict brain activity</title>
<p>We compared encoding model performance across language models at different sizes. For each electrode, we obtained the maximum encoding performance correlation across all lags and layers, then averaged these correlations across electrodes to derive the overall maximum correlation for each model (<xref rid="fig2" ref-type="fig">Fig. 2B</xref>). Using ECoG neural signals with superior temporal resolution, we replicated the previous fMRI work reporting a logarithmic relationship between model size and encoding performance (<xref ref-type="bibr" rid="c2">Antonello et al., 2023</xref>), indicating that larger models better predict neural activity. We also observed a plateau in the maximal encoding performance, occurring around 7 billion parameters (<xref rid="fig2" ref-type="fig">Fig. 2B</xref>), with a decline in performance for the OPT-66B model (<xref ref-type="fig" rid="figs1">Fig. S1</xref>). The size of the contextual embedding varies across models depending on the model’s size and architecture. This can range from 762 in the smallest distill GPT2 model to 8192 in the largest LLAMA-2 70 billion parameter model. To control for the different embedding dimensionality across models, we standardized all embeddings to the same size using principal component analysis (PCA) and trained linear encoding models using ordinary least-squares (OLS) regression, replicating the logarithmic relationship but with significantly lower encoding performance overall (<xref ref-type="fig" rid="figs2">Fig. S2</xref>). The PC features are used by the OLS models only.</p>
<p>To dissociate model size and control for other confounding variables, we next focused on the GPT-Neo models and assessed layer-by-layer and lag-by-lag encoding performance. For each layer of each model, we identified the maximum encoding performance correlation across all lags and averaged this maximum correlation across electrodes (<xref rid="fig2" ref-type="fig">Fig. 2C</xref>). Additionally, we converted the absolute layer number into a percentage of the total number of layers to compare across models (<xref rid="fig2" ref-type="fig">Fig. 2D</xref>). We found that correlations for all four models typically peak at intermediate layers, forming an inverted U-shaped curve, corroborating with previous fMRI findings (<xref ref-type="bibr" rid="c8">Caucheteux et al., 2021</xref>; <xref ref-type="bibr" rid="c39">Schrimpf et al., 2021</xref>; <xref ref-type="bibr" rid="c43">Toneva &amp; Wehbe, 2019</xref>). Furthermore, we replicated the phenomenon observed by (<xref ref-type="bibr" rid="c2">Antonello et al., 2023</xref>), wherein smaller models (e.g. SMALL) achieve maximum encoding performance approximately three-quarters into the model, while larger models (e.g. XL) peak in relatively earlier layers before gradually declining. Leveraging the high temporal resolution of ECoG, we compared the encoding performance of models across various lags relative to word onset. We identified the optimal layer for each electrode and model and then averaged the encoding performance across electrodes. We found that XL significantly outperformed SMALL in encoding models for most lags from 2000 ms before word onset to 575 ms after word onset (<xref ref-type="fig" rid="figs3">Fig. S3</xref>). To establish a general baseline for encoding performance, we built encoding models using embeddings from the SMALL model with randomly initialized weights. The trained SMALL model exhibits significantly higher encoding performance across all layers compared to the untrained SMALL model (<xref ref-type="fig" rid="figs4">Fig. S4</xref>). We also assessed the encoding performance of contextual embeddings from LLMs against classic speech features and static GloVe embeddings (<xref ref-type="table" rid="tbls1">Table S1</xref>). The SMALL and XL embeddings achieved markedly higher encoding correlations than the speech features and GloVe embeddings (<xref ref-type="fig" rid="figs5">Fig. S5</xref>). We also built encoding models using subsets of the data and found that encoding performance increases as the volume of training data increases (<xref ref-type="fig" rid="figs6">Fig. S6</xref>).</p>
</sec>
<sec id="s2b">
<title>Encoding model performance across electrodes and brain regions</title>
<p>Next, we examined the differences in the encoding model across electrodes and brain regions. For each of the 160 electrodes, we identified the maximum encoding performance correlation across all lags and layers (<xref rid="fig3" ref-type="fig">Fig. 3A</xref>). Consistent with prior studies (<xref ref-type="bibr" rid="c18">Goldstein et al., 2022</xref>, <xref ref-type="bibr" rid="c17">2025</xref>), our encoding model for SMALL achieved the highest correlations in superior temporal gyrus (STG) and inferior frontal gyrus (IFG). We then compared the encoding performances between SMALL and the other three models, plotting the percent change in encoding performance relative to SMALL for each electrode (<xref rid="fig3" ref-type="fig">Fig. 3B</xref>). Across all three comparisons, we observed significantly higher encoding performance for the larger models in approximately one-third of the 160 electrodes (two-sided pairwise t-test across cross-validation folds for each electrode, <italic>p</italic> &lt; 0.05, Bonferroni corrected).</p>
<fig id="fig3" position="float" orientation="portrait" fig-type="figure">
<label>Figure 3.</label>
<caption><title>Model performance improves with increasing model size across electrodes and ROIs.</title>
<p><bold>A.</bold> Maximum correlation per electrode for SMALL. The encoding model achieves the highest correlations in STG and IFG. <bold>B.</bold> For MEDIUM, LARGE, and XL, the percentage difference in correlation relative to SMALL for all electrodes with significant encoding differences. The encoding performance is significantly higher for the bigger models for almost all electrodes across the brain (pairwise t-test across cross-validation folds). <bold>C.</bold> Maximum encoding correlations for SMALL and XL for each ROI (mSTG, aSTG, BA44, BA45, and TP area). The encoding performance is significantly higher for XL for all ROIs except TP. Each data point corresponds to an electrode in the corresponding ROI. <bold>D.</bold> Percent difference in correlation relative to SMALL for all ROIs. As model size increases, the percent change in encoding performance also increases for mSTG, aSTG, and BA44. After the medium model, the percent change in encoding performance plateaus for BA45 and TP. The shaded colors represent standard error.</p></caption>
<graphic xlink:href="598513v5_fig3.tif" mimetype="image/tiff"/>
</fig>
<p>We then compared the maximum correlations between SMALL and XL models across five regions of interest (ROIs) across the cortical language network (<xref ref-type="fig" rid="figs7">Fig. S7</xref>): middle superior temporal gyrus (mSTG, n = 28 electrodes), anterior superior temporal gyrus (aSTG, n = 13 electrodes), Brodmann area 44 (BA44, n = 19 electrodes), Brodmann area 45 (BA45, n = 26 electrodes), and temporal pole (TP, n = 6 electrodes). Encoding performance for the XL model significantly surpassed that of the SMALL model in mSTG, aSTG, BA44, and BA45 (<xref rid="fig3" ref-type="fig">Fig. 3C</xref>, <xref ref-type="table" rid="tbls2">Table S2</xref>). Additionally, we calculated the percent change in encoding performance relative to SMALL for each brain region by averaging across electrodes and plotting against model size (<xref rid="fig3" ref-type="fig">Fig. 3D</xref>). As model size increases, the fit to the brain nominally increases across all observed regions. However, the increase plateaued after the Medium model for regions BA45 and TP.</p>
</sec>
<sec id="s2c">
<title>The best layer for encoding performance varies with model size</title>
<p>In the previous analyses, we observed that encoding performance peaks at intermediate to later layers for some models and relatively earlier layers for others (<xref rid="fig1" ref-type="fig">Fig. 1C</xref>, 1D). To examine this phenomenon more closely, we selected the best layer for each electrode based on its maximum encoding performance across lags. To account for the variation in depth across models, we computed the best layer as the percentage of each model’s overall depth. We found that as models increase in size, peak encoding performance tends to occur in relatively earlier layers, being closer to the input in larger models (<xref rid="fig4" ref-type="fig">Fig. 4A</xref>). This was consistent across multiple model families, where we found a logarithmic relationship between model size and best encoding layers (<xref rid="fig4" ref-type="fig">Fig. 4B</xref>).</p>
<fig id="fig4" position="float" orientation="portrait" fig-type="figure">
<label>Figure 4.</label>
<caption><title>Relative layer preference varies with model size.</title><p><bold>A.</bold> Relative layer (in percentage of total number of layers) with peak encoding performance for all four GPT-Neo models: the larger the model size, the earlier relative layer where the encoding performance peaks. <bold>B.</bold> The relationship between model size (shown on a log scale) and best encoding layer (in percentage) for all four model families: as the model size increases, the best encoding layer (in percentage) decreases, although the rate of decrease is different between model families. We estimate a linear regression model per model family of the form: best percent layer ∼ log(model size). The slopes (<italic>β</italic>) indicate the decrease in the relative best-performing layer at increasing log model size; p-values are obtained from a Wald test against the null hypothesis that the slope is 0. Each data point corresponds to a model. <bold>C.</bold> Best relative encoding layer (in percentage) for all four GPT-Neo models. <bold>D</bold>. Best encoding layer for XL with electrodes that peak in the first half of the model (Layer 0 to 22). <bold>E.</bold> Best encoding layer (in percentage) for SMALL and XL for each ROI (mSTG, aSTG, BA44, BA45, and TP). Each data point corresponds to an electrode in the corresponding ROI.</p></caption>
<graphic xlink:href="598513v5_fig4.tif" mimetype="image/tiff"/>
</fig>
<p>We further observed variations of the best encoding layers across the brain within the same model. We found that the language processing hierarchy was better reflected in the best encoding layer preference for smaller than for larger models (<xref rid="fig4" ref-type="fig">Fig. 4C</xref>). Specifically, in the SMALL model, peak encoding was observed in earlier layers for STG electrodes and in later layers for IFG electrodes (<xref rid="fig4" ref-type="fig">Fig.4C</xref>, <xref ref-type="table" rid="tbls3">Table S3</xref>). A similar trend is evident in MEDIUM and partially in LARGE models, but not in the XL model, where the majority of electrodes exhibited peak encoding in the first 25% of all layers (<xref rid="fig4" ref-type="fig">Fig. 4C</xref>). However, despite the XL model showing less variance in the best layer distributions across cortex, we found the same hierarchy present for the first half of the model (layers 0–22, <xref rid="fig4" ref-type="fig">Fig. 4D</xref>). In this analysis, we observed that the best relative layer nominally increases from mSTG electrodes (M = 21.916, SD = 10.556) to aSTG electrodes (M = 29.720, SD = 17.979) to BA45 (M = 30.157, SD = 16.039) and TP electrodes (M = 31.061, SD = 16.305), and finally to BA44 electrodes (M = 36.962, SD = 13.140, <xref rid="fig4" ref-type="fig">Fig. 4E</xref>).</p>
</sec>
<sec id="s2d">
<title>The best lag for encoding performance does not vary with model size</title>
<p>Leveraging the high temporal resolution of ECoG, we investigated whether peak lag for encoding performance relative to word onset is affected by model size. For each ROI, we identified the optimal layer for each electrode in the ROI and then averaged the encoding performance (<xref rid="fig5" ref-type="fig">Fig. 5A</xref>). Within the SMALL model, we observed a trend where putatively lower-level regions of the language processing hierarchy peak earlier relative to word onset: mSTG encoding performance peaks around 25 ms before word onset, followed by aSTG encoding peak 225 ms after onset, and subsequently TP, BA44, and BA45 peak at approximately 350 ms. Within the XL model, we observed a similar trend, with mSTG encoding peaking first, followed by aSTG encoding peak, and finally, TP, BA44, and BA45 encodings. We also identified the lags when the encoding performance peaks for each electrode and visualized them on the brain map (<xref rid="fig5" ref-type="fig">Fig. 5B</xref>). Notably, the optimal lags for each electrode do not exhibit significant variation when transitioning from SMALL to XL.</p>
<fig id="fig5" position="float" orientation="portrait" fig-type="figure">
<label>Figure 5.</label>
<caption><title>Encoding performance across lags does not vary with model size.</title><p><bold>A.</bold> Average ROI encoding performance for SMALL and XL models. mSTG encoding peaks first before word onset, then aSTG peaks after word onset, followed by BA44, BA45, and TP encoding peaks at around 400 ms after onset. The dots represent the peak lag for each ROI. <bold>B.</bold> Lag with best encoding performance correlation for each electrode, using SMALL and XL model embeddings. Only electrodes with the best lags that fall within 600 ms before and after word onset are plotted.</p></caption>
<graphic xlink:href="598513v5_fig5.tif" mimetype="image/tiff"/>
</fig>
</sec>
</sec>
<sec id="s3">
<title>Discussion</title>
<p>In this study, we investigated how the quality of model-based predictions of neural activity scales with LLM model size (i.e., the number of parameters across layers). Prior studies have shown that encoding models constructed from the internal embeddings of LLMs provide remarkably good predictions of neural activity during natural language comprehension (<xref ref-type="bibr" rid="c9">Caucheteux &amp; King, 2022</xref>; <xref ref-type="bibr" rid="c18">Goldstein et al., 2022</xref>; <xref ref-type="bibr" rid="c39">Schrimpf et al., 2021</xref>). Corroborating prior work using fMRI (<xref ref-type="bibr" rid="c2">Antonello et al., 2023</xref>), across a range of models with 82 million to 70 billion parameters, we found that larger models are better aligned with neural activity. This result was consistent across several autoregressive LLM families varying in architectural details and training corpora and within a single model family trained on the same corpora and varying only in size. We suspect that the improved alignment with brain activity in larger models is driven by their increased expressivity and sensitivity to nuanced linguistic structure present in large-scale naturalistic datasets (<xref ref-type="bibr" rid="c2">Antonello et al., 2023</xref>). Our findings indicate that this trend does not trivially result from arbitrarily increasing model complexity: (a) models of varying size were estimated using regularized regression and evaluated using out-of-sample prediction to minimize the risks of overfitting, and (b) we obtained qualitatively similar results with explicitly matched dimensionality using PCA. Combined with the observation that larger models yield lower perplexity, our findings suggest that larger models’ capacity for learning the structure of natural language also yields better predictions of brain activity. By leveraging their increased size and representational power, these models have the potential to provide valuable insights into the mechanisms underlying language comprehension.</p>
<p>We focused on a particular family of models (GPT-Neo) trained on the same corpora and varying only in size to investigate how model size impacts layerwise encoding performance across lags and ROIs. We found that model-brain alignment improves consistently with increasing model size across the cortical language network. However, the increase plateaued after the MEDIUM model for regions BA45 and TP, possibly due to already high encoding correlations for the SMALL model and a small number of electrodes in the area, respectively.</p>
<p>A more detailed investigation of layerwise encoding performance revealed a logarithmic relationship where peak encoding performance tends to occur in relatively earlier layers as both model size and expressivity increase (<xref ref-type="bibr" rid="c33">Mischler et al., 2024</xref>). This is an unexpected extension of prior work on both language (<xref ref-type="bibr" rid="c9">Caucheteux &amp; King, 2022</xref>; <xref ref-type="bibr" rid="c27">Kumar et al., 2024</xref>; <xref ref-type="bibr" rid="c39">Schrimpf et al., 2021</xref>; <xref ref-type="bibr" rid="c43">Toneva &amp; Wehbe, 2019</xref>) and vision (<xref ref-type="bibr" rid="c25">Jiahui et al., 2023</xref>), where peak encoding performance was found at late-intermediate layers. Moreover, we observed variations in best relative layers across different brain regions, corresponding to a language processing hierarchy. This is particularly evident in smaller models and early layers of larger models. The inverted U-shaped trend of encoding performance commonly found in previous research is likely due to a “two-phase abstraction process” within LLMs (<xref ref-type="bibr" rid="c10">Cheng &amp; Antonello, 2024</xref>; <xref ref-type="bibr" rid="c11">Csordás et al., 2025</xref>). In the early and intermediate layers of the model, a composition phase occurs, where low-level input features become increasingly abstract and contextualized. The intermediate layers of the model show the highest correlation with brain activity, presumably because they capture complex semantic and contextual information in a way that generalizes well across a variety of tasks (including prediction of human neural activity) (<xref ref-type="bibr" rid="c1">Antonello &amp; Huth, 2024</xref>). Subsequently, a prediction phase happens in the later layers of the model, where the representations become more specialized for the LLM’s specific training objective (e.g., next-word prediction). This specialization can effectively constrict the more generalized feature representations, making these layers less optimal for predicting brain activity. Our results indicate that the initial composition phase does not take up more layers as models scale up in size. Larger models develop the necessary rich, abstract representations in the same number of layers as smaller models. Thus, as LLMs increase in size, the later layers of the model may contain representations that are increasingly divergent from the more general linguistic processing captured in brain activity. It is also possible that the later layers of larger models are overall underutilized and may not significantly contribute to benchmark performances during inference (<xref ref-type="bibr" rid="c11">Csordás et al., 2025</xref>; <xref ref-type="bibr" rid="c13">Fan et al., 2024</xref>; <xref ref-type="bibr" rid="c19">Gromov et al., 2024</xref>). Leveraging the high temporal resolution of ECoG, we found that putatively lower-level regions of the language processing hierarchy peak earlier than higher-level regions. However, we did not observe variations in the optimal lags for encoding performance across different model sizes. Since we exclusively employ textual LLMs, which lack inherent temporal information due to their discrete token-based nature, future studies utilizing multimodal LLMs integrating continuous audio or video streams, like Whisper or WavLM, may better unravel the relationship between model size and temporal dynamic representations in LLMs (<xref ref-type="bibr" rid="c17">Goldstein et al., 2025</xref>; <xref ref-type="bibr" rid="c32">Millet et al., 2023</xref>; <xref ref-type="bibr" rid="c45">Vaidya et al., 2022</xref>).</p>
<p>Our podcast stimulus comprised ∼5,000 words over a roughly 30-minute episode. Although this is a rich language stimulus, naturalistic stimuli of this kind have relatively low power for modeling infrequent linguistic structures (<xref ref-type="bibr" rid="c20">Hamilton &amp; Huth, 2020</xref>). While perplexity for the podcast stimulus continued to decrease for larger models, we observed a plateau in predicting brain activity for the largest LLMs. The largest models learn to capture relatively nuanced or rare linguistic structures, but these may occur too infrequently in our stimulus to capture much variance in brain activity. Using a subsampling approach, we found that encoding performance also scales with the volume (and diversity) of language stimuli. Encoding performance may continue to increase for the largest models with more extensive stimuli (<xref ref-type="bibr" rid="c2">Antonello et al., 2023</xref>), motivating future work to pursue dense sampling with numerous, diverse naturalistic stimuli (<xref ref-type="bibr" rid="c17">Goldstein et al., 2025</xref>; <xref ref-type="bibr" rid="c28">LeBel et al., 2023</xref>).</p>
<p>The advent of deep learning has marked a tectonic shift in how we model brain activity in more naturalistic contexts, such as real-world language comprehension (<xref ref-type="bibr" rid="c21">Hasson et al., 2020</xref>; <xref ref-type="bibr" rid="c38">Richards et al., 2019</xref>). Traditionally, neuroscience has sought to extract a limited set of interpretable rules to explain brain function. However, deep learning introduces a new class of highly parameterized models that can challenge and enhance our understanding. The vast number of parameters in these models allows them to achieve human-like performance on complex tasks like language comprehension and production. It is important to note that LLMs have fewer parameters than the number of synapses in any human cortical functional network. Furthermore, the complexity of what these models learn enables them to process natural language in real-life contexts as effectively as the human brain does. Thus, the explanatory power of these models is in achieving such expressivity based on relatively simple computations in pursuit of a relatively simple objective function (e.g., next-word prediction). As in the human brain, while scaling alone may yield emergent cognitive abilities (<xref ref-type="bibr" rid="c7">Cantlon &amp; Piantadosi, 2024</xref>; <xref ref-type="bibr" rid="c22">Herculano-Houzel, 2012</xref>), specialized architectural features likely also play a critical role (<xref ref-type="bibr" rid="c14">Friederici &amp; Becker, 2025</xref>). As we continue to develop larger, more sophisticated models, the scientific community is tasked with advancing a framework for understanding these models to better understand the intricacies of the neural code that supports natural language processing in the human brain.</p>
</sec>
<sec id="s4">
<title>Materials and Methods</title>
<sec id="s4a">
<title>Participants</title>
<p>Ten patients (6 female, 20-48 years old) with treatment-resistant epilepsy undergoing intracranial monitoring with subdural grid and strip electrodes for clinical purposes participated in the study. Two patients consented to have an FDA-approved hybrid clinical research grid implanted, which includes standard clinical electrodes and additional electrodes between clinical contacts. The hybrid grid provides a broader spatial coverage while maintaining the same clinical acquisition or grid placement. All participants provided informed consent following the protocols approved by the Institutional Review Board of the New York University Grossman School of Medicine. The patients were explicitly informed that their participation in the study was unrelated to their clinical care and that they had the right to withdraw from the study at any time without affecting their medical treatment. One patient was removed from further analyses due to excessive epileptic activity and low SNR across all experimental data collected during the day.</p>
</sec>
<sec id="s4b">
<title>Stimuli</title>
<p>Participants listened to a 30-minute auditory story stimulus, “So a Monkey and a Horse Walk Into a Bar: Act One, Monkey in the Middle,” (<italic>So a Monkey and a Horse Walk into a Bar</italic>, 2017) from the This American Life Podcast (<xref ref-type="bibr" rid="c40">Chivvis, 2017</xref>). The audio narrative is 30 minutes long and consists of approximately 5000 words. Participants were not explicitly aware that we would examine word prediction in our subsequent analyses. The onset of each word was marked using the Penn Phonetics Lab Forced Aligner (<xref ref-type="bibr" rid="c48">Yuan &amp; Liberman, 2008</xref>) and manually validated and adjusted as needed. The stimulus and alignment processes are described in prior work (<xref ref-type="bibr" rid="c18">Goldstein et al., 2022</xref>). In this study, we use the term “structure” to refer to a variety of linguistic patterns (e.g., morphology, syntax, semantics, context) that LLMs encode.</p>
</sec>
<sec id="s4c">
<title>Data acquisition and preprocessing</title>
<p>Across all patients, 1106 electrodes were placed on the left and 233 on the right hemispheres (signal sampled at or downsampled to 512 Hz). Brain activity was recorded from a total of 1339 intracranially implanted subdural platinum-iridium electrodes embedded in silastic sheets (2.3mm diameter contacts, Ad-Tech Medical Instrument; for the hybrid grids, 64 standard contacts had a diameter of 2 mm and an additional 64 contacts were 1mm diameter, PMT corporation, Chananssen, MN). We also preprocessed the neural data to get the power in the high-gamma-band activity (70-200 HZ). The full description of ECoG recording procedure is provided in prior work (<xref ref-type="bibr" rid="c18">Goldstein et al., 2022</xref>).</p>
<p>Electrode-wise preprocessing consisted of four main stages: First, large spikes exceeding four quartiles above and below the median were removed, and replacement samples were imputed using cubic interpolation. Second, the data were re-referenced using common average referencing. Third, 6-cycle wavelet decomposition was used to compute the high-frequency broadband (HFBB) power in the 70–200 Hz band, excluding 60, 120, and 180 Hz line noise. In addition, the HFBB time series of each electrode was log-transformed and z-scored. Fourth, the signal was smoothed using a Hamming window with a kernel size of 50 ms. The filter was applied in both the forward and reverse directions to maintain the temporal structure. Additional preprocessing details can be found in prior work (<xref ref-type="bibr" rid="c18">Goldstein et al., 2022</xref>).</p>
</sec>
<sec id="s4d">
<title>Electrode selection</title>
<p>We used a nonparametric statistical procedure with correction for multiple comparisons(<xref ref-type="bibr" rid="c34">Nichols &amp; Holmes, 2002</xref>) to identify significant electrodes. We randomized each electrode’s signal phase at each iteration by sampling from a uniform distribution. This disconnected the relationship between the words and the brain signal while preserving the autocorrelation in the signal. We then performed the encoding procedure for each electrode (for all lags). We used GloVe embeddings for electrode selection to avoid biasing our main results toward a particular LLM. We repeated the encoding process 5000 times. After each iteration, the encoding model’s maximal value across all lags was retained for each electrode. We then took the maximum value for each permutation across electrodes. This resulted in a distribution of 5000 values, which was used to determine the significance for all electrodes. For each electrode, a <italic>p</italic>-value was computed as the percentile of the non-permuted encoding model’s maximum value across all lags from the null distribution of 5000 maximum values. Performing a significance test using this randomization procedure evaluates the null hypothesis that there is no systematic relationship between the brain signal and the corresponding word embedding. This procedure yielded a <italic>p</italic>-value per electrode, corrected for the number of models tested across all lags within an electrode. To further correct for multiple comparisons across all electrodes, we used a false-discovery rate (FDR). Electrodes with <italic>q</italic>-values less than .01 are considered significant. This procedure identified 160 electrodes from eight patients in the left hemisphere’s early auditory, motor cortex, and language areas.</p>
</sec>
<sec id="s4e">
<title>Perplexity</title>
<p>We computed the perplexity values for each LLM using our story stimulus, employing a stride length half the maximum token length of each model (stride 512 for GPT-2 models, stride 1024 for GPT-Neo models, stride 1024 for OPT models, and stride 2048 for Llama-2 models). These stride values yield the lowest perplexity value for each model. We also replicated our results on fixed stride length across model families (stride 512, 1024, 2048, 4096).</p>
</sec>
<sec id="s4f">
<title>Contextual embeddings</title>
<p>We extracted contextual embeddings from all layers of four families of autoregressive large language models. The GPT-2 family, particularly <italic>gpt2-xl</italic>, has been extensively used in previous encoding studies (<xref ref-type="bibr" rid="c18">Goldstein et al., 2022</xref>; <xref ref-type="bibr" rid="c39">Schrimpf et al., 2021</xref>). Here we include distilGPT-2 as part of the GPT-2 family. The GPT-Neo family, released by EleutherAI (<xref ref-type="bibr" rid="c3">Black et al., 2022</xref>), features three models plus GPT-Neox-20b, all trained on the Pile dataset (<xref ref-type="bibr" rid="c15">Gao et al., 2020</xref>). The OPT and Llama-2 families are released by MetaAI (<xref ref-type="bibr" rid="c44">Touvron et al., 2023</xref>; S. <xref ref-type="bibr" rid="c51">Zhang et al., 2022</xref>). For Llama-2, we use the pre-trained versions before any reinforcement learning from human feedback. All models we used are implemented in the HuggingFace environment (<xref ref-type="bibr" rid="c46">Wolf et al., 2019</xref>). We define “model size” as the combined width of a model’s hidden layers and its number of layers, determining the total parameters. We first converted the words from the raw transcript (including punctuation and capitalization) to tokens comprising whole words or sub-words (e.g., (1) there’s → (1) there (2) ’s). All models within the same model family adhere to the same tokenizer convention, except for GPT-Neox-20B, which utilizes a different tokenizer (<xref ref-type="bibr" rid="c3">Black et al., 2022</xref>). For each token, we utilized a context window with the maximum context length of each language model containing prior tokens from the podcast (i.e., the token and its history) and extracted the embedding for the final token in the sequence (i.e., the token itself). To facilitate a fair comparison of the encoding effect across different models, we aligned all tokens in the story across all models. We averaged the token embeddings if a word is split into several tokens, resulting in one embedding per word for each model.</p>
<p>Transformer-based language model consists of blocks containing a self-attention sub-block and a subsequent feedforward sub-block. The output of a block is obtained through a residual connection applied to the sum of the block’s input and the output of the feedforward sub-block. The self-attention output is added to this sum later in the layer normalization step. This output is commonly referred to as the “hidden state” of language models. This hidden state is considered the contextual embedding for the preceding block. For convenience, we refer to the blocks as “layers”; that is, the hidden state at the output of block 3 is referred to as the contextual embedding for layer 3. To generate the contextual embeddings for each layer, we store each layer’s hidden state for each word in the input text. Fortunately, the HuggingFace implementation of those language models automatically stores these hidden states when a forward pass of the model is conducted. Different models have different numbers of layers and embeddings of different dimensionality. For instance, gpt-neo-125M (SMALL) has 12 layers, and the embeddings at each layer are 768-dimensional vectors. Since we generate an embedding for each word at every layer, this results in 12 768-dimensional embeddings per word.</p>
</sec>
<sec id="s4g">
<title>Static embeddings</title>
<p>We constructed static embeddings with classic speech features and GloVe to compare with the encoding performance of LLMs. First, we extracted features capturing lower-level acoustic qualities of speech. Using the stimulus transcript as input, we created one-hot vectors for phonetic and articulatory features. Phoneme classes (39 total classes) were obtained from the Carnegie Mellon Pronouncing Dictionary (<italic>The CMU Pronouncing Dictionary</italic>, n.d.). We further classified the phonemes based on their place of articulation (9 classes), manner of articulation (9 classes), and voiced or voiceless status (3 classes), according to the general American English consonants of the International Phonetic Alphabet. Given that each word consists of multiple phonemes, we averaged the one-hot vectors for all phonetic and articulatory features for each word. Second, we extracted linguistic features using spaCy (<xref ref-type="bibr" rid="c23">Honnibal et al., 2020</xref>), including part of speech (17 classes), tag (50 classes), function or content word (3 classes), dependency (45 classes), whether the word is an alpha character (binary), and whether the word is a stop word (binary). We also extracted prefix (30 classes) and suffix (44 classes) information using the Cambridge Dictionary. We constructed one-hot vectors for each multi-class feature and one-dimensional vectors for each binary feature. Third, for each word, we obtained word frequency from the Google Web Trillion Word Corpus (<xref ref-type="bibr" rid="c5">Brants &amp; Franz, 2006</xref>) and from our own dataset. Fourth, we generated static word embeddings of dimension 50 using GloVe (<xref ref-type="bibr" rid="c36">Pennington et al., 2014</xref>).</p>
</sec>
<sec id="s4h">
<title>Encoding models</title>
<p>Linear encoding models were estimated at each lag (-2000 ms to 2000 ms in 25-ms increments) relative to word onset (0 ms) to predict the brain activity for each word from the corresponding contextual embedding. Before fitting the encoding model, we smoothed the signal using a rolling 200-ms window (i.e., for each lag, the model learns to predict the average single +-100 ms around the lag). We estimated and evaluated the encoding models using a 10-fold cross-validation procedure: ridge regression was used to estimate a weight matrix for predicting word-by-word neural signals in 9 out of 10 contiguous training segments of the podcast; for each electrode, we then calculated the Pearson correlation between predicted and actual neural signals for the left-out test segment of the podcast. For each ridge regression model (for each fold, lag, and electrode), the alpha parameter is determined by cross-validation using the “RidgeCV” method from the “himalaya” package (<xref ref-type="bibr" rid="c12">Dupré la Tour et al., 2022</xref>). This procedure was performed for all layers of contextual embeddings from each LLM. To control for the different embedding dimensionality across models, we standardized all embeddings to the same size using principal component analysis (PCA) and trained linear encoding models using ordinary least-squares (OLS) regression. The PC features are used by the OLS models only.</p>
</sec>
<sec id="s4i">
<title>Dimensionality reduction</title>
<p>To control for the different hidden embedding sizes across models, we standardized all embeddings to the same size using principal component analysis (PCA) and trained linear regression encoding models using ordinary least-squares regression, replicating all results (<xref ref-type="fig" rid="figs2">Fig. S2</xref>). This procedure effectively focuses our subsequent analysis on the 50 orthogonal dimensions in the embedding space that account for the most variance in the stimulus. We compute PCA separately on the training and testing set to avoid data leakage.</p>
</sec>
</sec>
</body>
<back>
<sec id="das" sec-type="data-availability">
<title>Data availability</title>
<p>We have recently made the data publicly available (<xref ref-type="bibr" rid="c49">Zada et al., 2025</xref>). We have also provided tutorials for preprocessing the data and training encoding models: <ext-link ext-link-type="uri" xlink:href="https://hassonlab.github.io/podcast-ecog-tutorials">https://hassonlab.github.io/podcast-ecog-tutorials</ext-link>. For this specific project, the analysis code is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/hassonlab/247-pickling/tree/scaling-paper-0">https://github.com/hassonlab/247-pickling/tree/scaling-paper-0</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://github.com/hassonlab/247-encoding/tree/scaling-paper-1">https://github.com/hassonlab/247-encoding/tree/scaling-paper-1</ext-link>.</p>
</sec>
<sec sec-type="supplementary" id="supplementary20">
<title>Supplementary figures and tables</title>
<fig id="figs1" position="float" orientation="portrait" fig-type="figure">
<label>Supplementary Figure 1.</label>
<caption><title>Heatmap comparing encoding performance between models.</title>
<p>Encoding performance for a model is represented by the maximum correlation across lags and layers per electrode. Paired t-tests are performed across electrodes (df = 159). The result is blue if t &lt; 0, meaning the larger model (the model with a higher number of parameters, represented on the x-axis) outperforms the smaller model (the model with a smaller number of parameters, represented on the y-axis). The result is red if t &gt; 0, meaning the smaller model outperforms the larger model. The shades of the colors represent significance (p &lt; 0.001 or p &lt; 0.01 or not significant, two-sided, FDR corrected). There is a positive relationship between model size and encoding performance when models are smaller than 3 billion parameters. When models are larger than 7 billion parameters, we observed a plateau in the maximal encoding performance.</p></caption>
<graphic xlink:href="598513v5_figs1.tif" mimetype="image/tiff"/>
</fig>
<fig id="figs2" position="float" orientation="portrait" fig-type="figure">
<label>Supplementary Figure 2.</label>
<caption><title>Model performance improves with increasing model size.</title><p>To control for the different embedding dimensionality across models, we standardized all embeddings to the same size using principal component analysis (PCA) and trained linear encoding models using ordinary least-squares (OLS) regression (cf. <xref rid="fig2" ref-type="fig">Fig. 2</xref>). <bold>A.</bold> Replication of <xref rid="fig2" ref-type="fig">Fig. 2B</xref>, the relationship between model size (shown on a log scale) and brain encoding performance: encoding performance increases as model size increases. Each data point corresponds to a model. <bold>B.</bold> Ridge regression encoding outperforms PCA + OLS regression encoding for all 20 transformer-based language models (paired two-sided <italic>t</italic>-test across electrodes, df = 159, <italic>p</italic> &lt; 0.001, Bonferroni corrected). Each data point corresponds to a model.</p></caption>
<graphic xlink:href="598513v5_figs2.tif" mimetype="image/tiff"/>
</fig>
<fig id="figs3" position="float" orientation="portrait" fig-type="figure">
<label>Supplementary Figure 3.</label>
<caption><title>Lag-wise encoding for the GPT-Neo Family.</title><p><bold>Top.</bold> Lag-wise encoding for all four models of the GPT-Neo family, averaged across electrodes. The dots represent lags where XL significantly outperformed Small (paired two-sided <italic>t</italic>-test across electrodes, df = 159, <italic>p</italic> &lt; 0.001, Bonferroni corrected). XL significantly outperformed Small in encoding models for most lags from 2000 ms before word onset to 575 ms after word onset. <bold>Bottom.</bold> Lag-wise encoding difference for the three bigger models compared to SMALL, averaged across electrodes.</p></caption>
<graphic xlink:href="598513v5_figs3.tif" mimetype="image/tiff"/>
</fig>
<fig id="figs4" position="float" orientation="portrait" fig-type="figure">
<label>Supplementary Figure 4.</label>
<caption><title>The relationship between encoding performance and layer number for pretrained and untrained SMALL.</title>
<p>For pretrained SMALL, encoding performance is best for intermediate layers. For untrained SMALL with randomly initialized weights, encoding performance is best for the 0th layer. Encoding performance is significantly higher for pretrained SMALL than for untrained SMALL for every layer (paired two-sided t-test across electrodes, df = 159, p &lt; 0.001, Bonferroni corrected). The shaded colors represent standard error.</p></caption>
<graphic xlink:href="598513v5_figs4.tif" mimetype="image/tiff"/>
</fig>
<fig id="figs5" position="float" orientation="portrait" fig-type="figure">
<label>Supplementary Figure 5.</label>
<caption><title>Contextual embeddings from LLMs outperform classical speech features and GloVe embeddings.</title>
<p><bold>A</bold>. Maximum encoding performance (across lags) for acoustic features (60 dimensions), linguistic features (191 dimensions), word frequency (2 dimensions), acoustic features + linguistic features + word frequency (253 dimensions), GloVe (50 dimensions), SMALL (768 dimensions), and XL (6144 dimensions) using ridge regression. SMALL and XL showed significantly better performance than other embeddings (paired two-sided t-test across electrodes, df = 159, p &lt; 0.001). <bold>B</bold>. Replication of A using PCA and OLS regression. To control for the different dimensions of the different feature spaces, we standardized all feature sets to 50 dimensions (2 dimensions for word frequency) using PCA and trained OLS regression encoding models. SMALL and XL showed significantly better performance than other embeddings (paired two-sided t-test across electrodes, df = 159, p &lt; 0.001). <bold>C</bold>. Encoding performance across lags for all features. Shaded colors represent standard error across electrodes. <bold>D.</bold> Replication of C using PCA and OLS regression.</p></caption>
<graphic xlink:href="598513v5_figs5.tif" mimetype="image/tiff"/>
</fig>
<fig id="figs6" position="float" orientation="portrait" fig-type="figure">
<label>Supplementary Figure 6.</label>
<caption><title>Model performance improves with increasing dataset size.</title>
<p><bold>A.</bold> For SMALL, the relationship between the percentage of training data and brain encoding performance across layers. Encoding performance increases as the training dataset size increases. The shaded colors represent standard error across electrodes. B. Same as A, but for model XL.</p></caption>
<graphic xlink:href="598513v5_figs6.tif" mimetype="image/tiff"/>
</fig>
<fig id="figs7" position="float" orientation="portrait" fig-type="figure">
<label>Supplementary Figure 7.</label>
<caption><title>Brain map of electrodes in five regions of interest (ROIs) across the cortical language network:</title><p>middle superior temporal gyrus (mSTG, n = 28 electrodes), anterior superior temporal gyrus (aSTG, n = 13 electrodes), Brodmann area 44 (BA44, n = 19 electrodes), Brodmann area 45 (BA45, n = 26 electrodes), and temporal pole (TP, n = 6 electrodes).</p></caption>
<graphic xlink:href="598513v5_figs7.tif" mimetype="image/tiff"/>
</fig>
<fig id="figs8" position="float" orientation="portrait" fig-type="figure">
<label>Supplementary Figure 8.</label>
<caption><title>The optimal lags for each electrode do not exhibit significant variation when transitioning between SMALL and XL models.</title>
<p><bold>A.</bold> Scatter plot of best-performing lag for SMALL and XL models, colored by max correlation. Each data point corresponds to an electrode. <bold>B.</bold> Scatter plot of best-performing lag for SMALL and XL models, colored by ROIs. Each data point corresponds to an electrode. Only the electrodes in <xref ref-type="fig" rid="figs7">Fig. S7</xref> are included.</p></caption>
<graphic xlink:href="598513v5_figs8.tif" mimetype="image/tiff"/>
</fig>
<table-wrap id="tbls1" orientation="portrait" position="float">
<label>Supplementary Table 1.</label>
<caption><title>Classic speech features.</title><p>Lower-level acoustic features include phoneme (39 classes), place of articulation (9 classes), manner of articulation (9 classes), and voiced or voiceless (3 classes). Linguistic features include part of speech (17 classes), tag (50 classes), function or content (3 classes), prefix (30 classes), suffix (44 classes), dependency (45 classes), whether the word is an alpha character (binary), and whether the word is a stop word (binary). Word frequency includes the frequency of the word in the Google Web Trillion Word Corpus and in our dataset.</p></caption>
<graphic xlink:href="598513v5_tbls1.tif" mimetype="image/tiff"/>
</table-wrap>
<table-wrap id="tbls2" orientation="portrait" position="float">
<label>Supplementary Table 2.</label>
<caption><title>Summary statistics and paired <italic>t</italic>-test results for maximum correlations between SMALL and XL models across five regions of interest.</title><p>Encoding performance for the XL model significantly surpassed that of the SMALL model in whole brain, mSTG, aSTG, BA44, and BA45.</p></caption>
<graphic xlink:href="598513v5_tbls2.tif" mimetype="image/tiff"/>
</table-wrap>
<table-wrap id="tbls3" orientation="portrait" position="float">
<label>Supplementary Table 3.</label>
<caption><title>Summary statistics and paired <italic>t</italic>-test results for best-performing layers (in percentage) for the SMALL model across five regions of interest.</title><p>The best-performing layer (in percentage) occurred earlier for electrodes in mSTG and aSTG and later for electrodes in BA44, BA45, and TP.</p></caption>
<graphic xlink:href="598513v5_tbls3.tif" mimetype="image/tiff"/>
</table-wrap>
</sec>
<ack>
<title>Acknowledgements</title>
<p>This work was supported by the National Institutes of Health under award numbers DP1HD091948 (to A.G., Z.H., H.W., Z.Z., B.A., L.N., A.F., and U.H.), R01NS109367 (to A.F.), and R01DC022534 (to S.A.N.), Finding a Cure for Epilepsy and Seizures (FACES), and Schmidt Futures Foundation DataX Fund. Z.H. devised the project, performed experimental design and data analysis, and wrote the article; H.W. devised the project, performed experimental design and data analysis, and wrote the article; Z.Z. devised the project, performed experimental design and data analysis, and critically revised the article; H.G. performed data analysis; B.A. performed data analysis; L.N. performed data analysis; W.D. devised the project; S.D. devised the project; P.D. devised the project; D.F. devised the project; O.D. devised the project; A.F. devised the project; U.H. devised the project, performed experimental design, and critically revised the article; S.A.N. devised the project, performed experimental design, wrote and critically revised the article; A.G. devised the project, performed experimental design, and critically revised the article.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="c1"><label>1.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Antonello</surname>, <given-names>R</given-names></string-name>, &amp; <string-name><surname>Huth</surname>, <given-names>A</given-names></string-name></person-group>. (<year>2024</year>). <article-title>Predictive coding or just feature discovery? An alternative account of why language models fit brain data</article-title>. <source>Neurobiology of Language (Cambridge, Mass.)</source>, <volume>5</volume>(<issue>1</issue>), <fpage>64</fpage>–<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1162/nol_a_00087</pub-id> <pub-id pub-id-type="pmid">38645616</pub-id></mixed-citation></ref>
<ref id="c2"><label>2.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Antonello</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Huth</surname>, <given-names>A</given-names></string-name>, &amp; <string-name><surname>Vaidya</surname>, <given-names>A</given-names></string-name></person-group>. (<year>2023</year>). <article-title>Scaling laws for language encoding models in fMRI</article-title>. <source>Advances in Neural Information Processing Systems</source> <volume>36</volume>, <fpage>21895</fpage>–<lpage>21907</lpage>. <pub-id pub-id-type="doi">10.48550/arxiv.2305.11863</pub-id> <pub-id pub-id-type="pmid">39035676</pub-id></mixed-citation></ref>
<ref id="c3"><label>3.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Black</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Biderman</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Hallahan</surname>, <given-names>E</given-names></string-name>, <string-name><surname>Anthony</surname>, <given-names>Q</given-names></string-name>, <string-name><surname>Gao</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Golding</surname>, <given-names>L</given-names></string-name>, <string-name><surname>He</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Leahy</surname>, <given-names>C</given-names></string-name>, <string-name><surname>McDonell</surname>, <given-names>K</given-names></string-name>, <string-name><surname>Phang</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Pieler</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Prashanth</surname>, <given-names>US</given-names></string-name>, <string-name><surname>Purohit</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Reynolds</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Tow</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>B</given-names></string-name>, &amp; <string-name><surname>Weinbach</surname>, <given-names>S</given-names></string-name></person-group>. (<year>2022</year>). <article-title>GPT-NeoX-20B: An Open-Source Autoregressive Language Model</article-title>. <source>Proceedings of BigScience Episode #5 -- Workshop on Challenges &amp;amp; Perspectives in Creating Large Language Models</source> <conf-name>Proceedings of BigScience Episode #5 -- Workshop on Challenges &amp; Perspectives in Creating Large Language Models</conf-name>. <pub-id pub-id-type="doi">10.18653/v1/2022.bigscience-1.9</pub-id></mixed-citation></ref>
<ref id="c4"><label>4.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Bommasani</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Hudson</surname>, <given-names>DA</given-names></string-name>, <string-name><surname>Adeli</surname>, <given-names>E</given-names></string-name>, <string-name><surname>Altman</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Arora</surname>, <given-names>S</given-names></string-name>, <string-name><surname>von Arx</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Bernstein</surname>, <given-names>MS</given-names></string-name>, <string-name><surname>Bohg</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Bosselut</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Brunskill</surname>, <given-names>E</given-names></string-name>, <string-name><surname>Brynjolfsson</surname>, <given-names>E</given-names></string-name>, <string-name><surname>Buch</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Card</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Castellon</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Chatterji</surname>, <given-names>N</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Creel</surname>, <given-names>K</given-names></string-name>, <string-name><surname>Davis</surname>, <given-names>JQ</given-names></string-name>, <string-name><surname>Demszky</surname>, <given-names>D</given-names></string-name>, <etal>…</etal> <string-name><surname>Liang</surname>, <given-names>P</given-names></string-name></person-group> (<year>2021</year>). <article-title>On the Opportunities and Risks of Foundation Models</article-title>. <source>arXiv</source> <pub-id pub-id-type="doi">10.48550/arxiv.2108.07258</pub-id></mixed-citation></ref>
<ref id="c5"><label>5.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Brants</surname>, <given-names>T</given-names></string-name>, &amp; <string-name><surname>Franz</surname>, <given-names>A</given-names></string-name></person-group>. (<year>2006</year>). <article-title><italic>Web 1T 5-gram Version 1</italic> [Dataset]</article-title>. <source>Linguistic Data Consortium</source>. <pub-id pub-id-type="doi">10.35111/CQPA-A498</pub-id></mixed-citation></ref>
<ref id="c6"><label>6.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Brown</surname>, <given-names>TB</given-names></string-name>, <string-name><surname>Mann</surname>, <given-names>B</given-names></string-name>, <string-name><surname>Ryder</surname>, <given-names>N</given-names></string-name>, <string-name><surname>Subbiah</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Kaplan</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Dhariwal</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Neelakantan</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Shyam</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Sastry</surname>, <given-names>G</given-names></string-name>, <string-name><surname>Askell</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Agarwal</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Herbert-Voss</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Krueger</surname>, <given-names>G</given-names></string-name>, <string-name><surname>Henighan</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Child</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Ramesh</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Ziegler</surname>, <given-names>DM</given-names></string-name>, <string-name><surname>Wu</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Winter</surname>, <given-names>C</given-names></string-name>, <etal>…</etal> <string-name><surname>Amodei</surname>, <given-names>D</given-names></string-name></person-group>. (<year>2020</year>). <article-title>Language Models are Few-Shot Learners</article-title>. <source>arXiv</source> <pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id></mixed-citation></ref>
<ref id="c7"><label>7.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cantlon</surname>, <given-names>JF</given-names></string-name>, &amp; <string-name><surname>Piantadosi</surname>, <given-names>ST</given-names></string-name></person-group>. (<year>2024</year>). <article-title>Uniquely human intelligence arose from expanded information capacity</article-title>. <source>Nature Reviews Psychology</source>, <fpage>1</fpage>–<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1038/s44159-024-00283-3</pub-id></mixed-citation></ref>
<ref id="c8"><label>8.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Caucheteux</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Gramfort</surname>, <given-names>A</given-names></string-name>, &amp; <string-name><surname>King</surname>, <given-names>J.-R</given-names></string-name></person-group>. (<year>2021</year>). <article-title>GPT-2’s activations predict the degree of semantic comprehension in the human brain</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2021.04.20.440622</pub-id></mixed-citation></ref>
<ref id="c9"><label>9.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Caucheteux</surname>, <given-names>C</given-names></string-name>, &amp; <string-name><surname>King</surname>, <given-names>J.-R</given-names></string-name></person-group>. (<year>2022</year>). <article-title>Brains and algorithms partially converge in natural language processing</article-title>. <source>Communications Biology</source>, <volume>5</volume>(<issue>1</issue>), <fpage>134</fpage>. <pub-id pub-id-type="doi">10.1038/s42003-022-03036-1</pub-id> <pub-id pub-id-type="pmid">35173264</pub-id></mixed-citation></ref>
<ref id="c10"><label>10.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Cheng</surname>, <given-names>E</given-names></string-name>, &amp; <string-name><surname>Antonello</surname>, <given-names>RJ</given-names></string-name></person-group>. (<year>2024</year>). <article-title>Evidence from fMRI supports a two-phase abstraction process in language models</article-title>. <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/ARXIV.2409.05771</pub-id></mixed-citation></ref>
<ref id="c11"><label>11.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Csordás</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Manning</surname>, <given-names>CD</given-names></string-name>, &amp; <string-name><surname>Potts</surname>, <given-names>C</given-names></string-name></person-group>. (<year>2025</year>). <article-title>Do language models use their depth efficiently?</article-title> <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/ARXIV.2505.13898</pub-id></mixed-citation></ref>
<ref id="c12"><label>12.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Dupré la Tour</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Eickenberg</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Nunez-Elizalde</surname>, <given-names>AO</given-names></string-name>, &amp; <string-name><surname>Gallant</surname>, <given-names>JL</given-names></string-name></person-group> (<year>2022</year>). <article-title>Feature-space selection with banded ridge regression</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2022.05.05.490831</pub-id></mixed-citation></ref>
<ref id="c13"><label>13.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Fan</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Jiang</surname>, <given-names>X</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>X</given-names></string-name>, <string-name><surname>Meng</surname>, <given-names>X</given-names></string-name>, <string-name><surname>Han</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Shang</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Sun</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>Y</given-names></string-name>, &amp; <string-name><surname>Wang</surname>, <given-names>Z</given-names></string-name></person-group> (<year>2024</year>). <article-title>Not all Layers of LLMs are Necessary during Inference</article-title>. In <conf-name>Proceedings of the Thirty-ThirdInternational Joint Conference on Artificial Intelligence</conf-name> <pub-id pub-id-type="doi">10.24963/ijcai.2024/566</pub-id></mixed-citation></ref>
<ref id="c14"><label>14.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Friederici</surname>, <given-names>AD</given-names></string-name>, &amp; <string-name><surname>Becker</surname>, <given-names>Y</given-names></string-name></person-group>. (<year>2025</year>). <article-title>The core language network separated from other networks during primate evolution</article-title>. <source>Nature Reviews. Neuroscience</source>, <volume>26</volume>(<issue>2</issue>), <fpage>131</fpage>–<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1038/s41583-024-00897-9</pub-id> <pub-id pub-id-type="pmid">39702531</pub-id></mixed-citation></ref>
<ref id="c15"><label>15.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Gao</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Biderman</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Black</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Golding</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Hoppe</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Foster</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Phang</surname>, <given-names>J</given-names></string-name>, <string-name><surname>He</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Thite</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Nabeshima</surname>, <given-names>N</given-names></string-name>, <string-name><surname>Presser</surname>, <given-names>S</given-names></string-name>, &amp; <string-name><surname>Leahy</surname>, <given-names>C</given-names></string-name></person-group> (<year>2020</year>). <article-title>The Pile: An 800GB Dataset of Diverse Text for Language Modeling</article-title>. <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2101.00027</pub-id></mixed-citation></ref>
<ref id="c16"><label>16.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Goldstein</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Grinstein-Dabush</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Schain</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Hong</surname>, <given-names>Z</given-names></string-name>, <string-name><surname>Aubrey</surname>, <given-names>B</given-names></string-name>, <string-name><surname>Schain</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Nastase</surname>, <given-names>SA</given-names></string-name>, <string-name><surname>Zada</surname>, <given-names>Z</given-names></string-name>, <string-name><surname>Ham</surname>, <given-names>E</given-names></string-name>, <string-name><surname>Feder</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Gazula</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Buchnik</surname>, <given-names>E</given-names></string-name>, <string-name><surname>Doyle</surname>, <given-names>W</given-names></string-name>, <string-name><surname>Devore</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Dugan</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Reichart</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Friedman</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Brenner</surname>, <given-names>M</given-names></string-name>, <etal>…</etal> <string-name><surname>Hasson</surname>, <given-names>U</given-names></string-name></person-group>. (<year>2024</year>). <article-title>Alignment of brain embeddings and artificial contextual embeddings in natural language points to common geometric patterns</article-title>. <source>Nature Communications</source>, <volume>15</volume>(<issue>1</issue>), <fpage>2768</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-024-46631-y</pub-id> <pub-id pub-id-type="pmid">38553456</pub-id></mixed-citation></ref>
<ref id="c17"><label>17.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Goldstein</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Niekerken</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Schain</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Zada</surname>, <given-names>Z</given-names></string-name>, <string-name><surname>Aubrey</surname>, <given-names>B</given-names></string-name>, <string-name><surname>Sheffer</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Nastase</surname>, <given-names>SA</given-names></string-name>, <string-name><surname>Gazula</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Singh</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Rao</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Choe</surname>, <given-names>G</given-names></string-name>, <string-name><surname>Kim</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Doyle</surname>, <given-names>W</given-names></string-name>, <string-name><surname>Friedman</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Devore</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Dugan</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Hassidim</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Brenner</surname>, <given-names>M</given-names></string-name>, <etal>…</etal> <string-name><surname>Hasson</surname>, <given-names>U</given-names></string-name></person-group>. (<year>2025</year>). <article-title>A unified acoustic-to-speech-to-language embedding space captures the neural basis of natural language processing in everyday conversations</article-title>. <source>Nature Human Behaviour</source>. <pub-id pub-id-type="doi">10.1038/s41562-025-02105-9</pub-id> <pub-id pub-id-type="pmid">40055549</pub-id></mixed-citation></ref>
<ref id="c18"><label>18.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Goldstein</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Zada</surname>, <given-names>Z</given-names></string-name>, <string-name><surname>Buchnik</surname>, <given-names>E</given-names></string-name>, <string-name><surname>Schain</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Price</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Aubrey</surname>, <given-names>B</given-names></string-name>, <string-name><surname>Nastase</surname>, <given-names>SA</given-names></string-name>, <string-name><surname>Feder</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Emanuel</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Cohen</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Jansen</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Gazula</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Choe</surname>, <given-names>G</given-names></string-name>, <string-name><surname>Rao</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Kim</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Casto</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Fanda</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Doyle</surname>, <given-names>W</given-names></string-name>, <string-name><surname>Friedman</surname>, <given-names>D</given-names></string-name>, <etal>…</etal> <string-name><surname>Hasson</surname>, <given-names>U</given-names></string-name></person-group>. (<year>2022</year>). <article-title>Shared computational principles for language processing in humans and deep language models</article-title>. <source>Nature Neuroscience</source>, <volume>25</volume>(<issue>3</issue>), <fpage>369</fpage>–<lpage>380</lpage>. <pub-id pub-id-type="doi">10.1038/s41593-022-01026-4</pub-id> <pub-id pub-id-type="pmid">35260860</pub-id></mixed-citation></ref>
<ref id="c19"><label>19.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Gromov</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Tirumala</surname>, <given-names>K</given-names></string-name>, <string-name><surname>Shapourian</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Glorioso</surname>, <given-names>P</given-names></string-name>, &amp; <string-name><surname>Roberts</surname>, <given-names>DA</given-names></string-name></person-group> (<year>2024</year>). <article-title>The Unreasonable Ineffectiveness of the Deeper Layers</article-title>. <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/arxiv.2403.17887</pub-id></mixed-citation></ref>
<ref id="c20"><label>20.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hamilton</surname>, <given-names>LS</given-names></string-name>, &amp; <string-name><surname>Huth</surname>, <given-names>AG</given-names></string-name></person-group>. (<year>2020</year>). <article-title>The revolution will not be controlled: natural stimuli in speech neuroscience</article-title>. <source>Language, Cognition and Neuroscience</source>, <volume>35</volume>(<issue>5</issue>), <fpage>573</fpage>–<lpage>582</lpage>. <pub-id pub-id-type="doi">10.1080/23273798.2018.1499946</pub-id> <pub-id pub-id-type="pmid">32656294</pub-id></mixed-citation></ref>
<ref id="c21"><label>21.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hasson</surname>, <given-names>U</given-names></string-name>, <string-name><surname>Nastase</surname>, <given-names>SA</given-names></string-name>, &amp; <string-name><surname>Goldstein</surname>, <given-names>A</given-names></string-name></person-group>. (<year>2020</year>). <article-title>Direct Fit to Nature: An Evolutionary Perspective on Biological and Artificial Neural Networks</article-title>. <source>Neuron</source>, <volume>105</volume>(<issue>3</issue>), <fpage>416</fpage>–<lpage>434</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2019.12.002</pub-id> <pub-id pub-id-type="pmid">32027833</pub-id></mixed-citation></ref>
<ref id="c22"><label>22.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Herculano-Houzel</surname>, <given-names>S</given-names></string-name></person-group>. (<year>2012</year>). <article-title>The remarkable, yet not extraordinary, human brain as a scaled-up primate brain and its associated cost</article-title>. <source>Proceedings of the National Academy of Sciences of the United States of America</source>, <volume>109</volume> (<issue>Suppl 1</issue>), <fpage>10661</fpage>–<lpage>10668</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1201895109</pub-id> <pub-id pub-id-type="pmid">22723358</pub-id></mixed-citation></ref>
<ref id="c23"><label>23.</label><mixed-citation publication-type="software"><person-group person-group-type="author"><string-name><surname>Honnibal</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Montani</surname>, <given-names>I</given-names></string-name>, <string-name><surname>Van Landeghem</surname>, <given-names>S</given-names></string-name>, &amp; <string-name><surname>Boyd</surname>, <given-names>A</given-names></string-name></person-group> (<year>2020</year>). <source>spaCy: Industrial-strength Natural Language Processing in Python</source>. <publisher-name>Explosion</publisher-name></mixed-citation></ref>
<ref id="c24"><label>24.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Hosseini</surname>, <given-names>EA</given-names></string-name>, <string-name><surname>Schrimpf</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>Y</given-names></string-name>, <string-name><surname>Bowman</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Zaslavsky</surname>, <given-names>N</given-names></string-name>, &amp; <string-name><surname>Fedorenko</surname>, <given-names>E</given-names></string-name></person-group>. (<year>2022</year>). <article-title>Artificial neural network language models align neurally and behaviorally with humans even after a developmentally realistic amount of training</article-title>. <source>bioRxiv</source> (p. <elocation-id>2022.10.04.510681</elocation-id>). <pub-id pub-id-type="doi">10.1101/2022.10.04.510681</pub-id></mixed-citation></ref>
<ref id="c25"><label>25.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jiahui</surname>, <given-names>G</given-names></string-name>, <string-name><surname>Feilong</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Visconti di Oleggio Castello</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Nastase</surname>, <given-names>SA</given-names></string-name>, <string-name><surname>Haxby</surname>, <given-names>JV</given-names></string-name>, &amp; <string-name><surname>Gobbini</surname>, <given-names>MI</given-names></string-name></person-group> (<year>2023</year>). <article-title>Modeling naturalistic face processing in humans with deep convolutional neural networks</article-title>. <source>Proceedings of the National Academy of Sciences of the United States of America</source>, <volume>120</volume>(<issue>43</issue>), <fpage>e2304085120</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.2304085120</pub-id> <pub-id pub-id-type="pmid">37847731</pub-id></mixed-citation></ref>
<ref id="c26"><label>26.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Kaplan</surname>, <given-names>J</given-names></string-name>, <string-name><surname>McCandlish</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Henighan</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Brown</surname>, <given-names>TB</given-names></string-name>, <string-name><surname>Chess</surname>, <given-names>B</given-names></string-name>, <string-name><surname>Child</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Gray</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Radford</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Wu</surname>, <given-names>J</given-names></string-name>, &amp; <string-name><surname>Amodei</surname>, <given-names>D</given-names></string-name></person-group> (<year>2020</year>). <article-title>Scaling Laws for Neural Language Models</article-title>. <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/arxiv.2001.08361</pub-id></mixed-citation></ref>
<ref id="c27"><label>27.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kumar</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Sumers</surname>, <given-names>TR</given-names></string-name>, <string-name><surname>Yamakoshi</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Goldstein</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Hasson</surname>, <given-names>U</given-names></string-name>, <string-name><surname>Norman</surname>, <given-names>KA</given-names></string-name>, <string-name><surname>Griffiths</surname>, <given-names>TL</given-names></string-name>, <string-name><surname>Hawkins</surname>, <given-names>RD</given-names></string-name>, &amp; <string-name><surname>Nastase</surname>, <given-names>SA</given-names></string-name></person-group>. (<year>2024</year>). <article-title>Shared functional specialization in transformer-based language models and the human brain</article-title>. <source>Nature Communications</source>, <volume>15</volume>(<issue>1</issue>), <fpage>5523</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-024-49173-5</pub-id> <pub-id pub-id-type="pmid">38951520</pub-id></mixed-citation></ref>
<ref id="c28"><label>28.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>LeBel</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Wagner</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Jain</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Adhikari-Desai</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Gupta</surname>, <given-names>B</given-names></string-name>, <string-name><surname>Morgenthal</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Tang</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Xu</surname>, <given-names>L</given-names></string-name>, &amp; <string-name><surname>Huth</surname>, <given-names>AG</given-names></string-name></person-group>. (<year>2023</year>). <article-title>A natural language fMRI dataset for voxelwise encoding models</article-title>. <source>Scientific Data</source>, <volume>10</volume>(<issue>1</issue>), <fpage>555</fpage>. <pub-id pub-id-type="doi">10.1038/s41597-023-02437-z</pub-id> <pub-id pub-id-type="pmid">37612332</pub-id></mixed-citation></ref>
<ref id="c29"><label>29.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Linzen</surname>, <given-names>T</given-names></string-name>, &amp; <string-name><surname>Baroni</surname>, <given-names>M</given-names></string-name></person-group>. (<year>2021</year>). <article-title>Syntactic Structure from Deep Learning</article-title>. <source>Annual Review of Linguistics</source>, <volume>7</volume>(<issue>1</issue>), <fpage>195</fpage>–<lpage>212</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-linguistics-032020-051035</pub-id></mixed-citation></ref>
<ref id="c30"><label>30.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Liu</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Shen</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>Y</given-names></string-name>, <string-name><surname>Dolan</surname>, <given-names>B</given-names></string-name>, <string-name><surname>Carin</surname>, <given-names>L</given-names></string-name>, &amp; <string-name><surname>Chen</surname>, <given-names>W</given-names></string-name></person-group>. (<year>2022</year>). <article-title>What makes good in-context examples for GPT-3?</article-title> <conf-name>Proceedings of Deep Learning Inside Out (DeeLIO 2022): The 3rd Workshop on Knowledge Extraction and Integration for Deep Learning Architectures.</conf-name>. <pub-id pub-id-type="doi">10.18653/v1/2022.deelio-1.10</pub-id></mixed-citation></ref>
<ref id="c31"><label>31.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Manning</surname>, <given-names>CD</given-names></string-name>, <string-name><surname>Clark</surname>, <given-names>K</given-names></string-name>, <string-name><surname>Hewitt</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Khandelwal</surname>, <given-names>U</given-names></string-name>, &amp; <string-name><surname>Levy</surname>, <given-names>O</given-names></string-name></person-group>. (<year>2020</year>). <article-title>Emergent linguistic structure in artificial neural networks trained by self-supervision</article-title>. <source>Proceedings of the National Academy of Sciences of the United States of America</source>, <volume>117</volume>(<issue>48</issue>), <fpage>30046</fpage>–<lpage>30054</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1907367117</pub-id> <pub-id pub-id-type="pmid">32493748</pub-id></mixed-citation></ref>
<ref id="c32"><label>32.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Millet</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Caucheteux</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Orhan</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Boubenec</surname>, <given-names>Y</given-names></string-name>, <string-name><surname>Gramfort</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Dunbar</surname>, <given-names>E</given-names></string-name>, <string-name><surname>Pallier</surname>, <given-names>C</given-names></string-name>, &amp; <string-name><surname>King</surname>, <given-names>J.-R</given-names></string-name></person-group>. (<year>2023</year>). <article-title>Toward a realistic model of speech processing in the brain with self-supervised learning</article-title>. <conf-name>NeurIPS</conf-name>. <ext-link ext-link-type="uri" xlink:href="https://neurips.cc/virtual/2022/poster/54632">https://neurips.cc/virtual/2022/poster/54632</ext-link></mixed-citation></ref>
<ref id="c33"><label>33.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mischler</surname>, <given-names>G</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>YA</given-names></string-name>, <string-name><surname>Bickel</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Mehta</surname>, <given-names>AD</given-names></string-name>, &amp; <string-name><surname>Mesgarani</surname>, <given-names>N</given-names></string-name></person-group>. (<year>2024</year>). <article-title>Contextual feature extraction hierarchies converge in large language models and the brain</article-title>. <source>Nature Machine Intelligence</source>, <volume>6</volume>(<issue>12</issue>), <fpage>1467</fpage>–<lpage>1477</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-024-00925-4</pub-id></mixed-citation></ref>
<ref id="c34"><label>34.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Nichols</surname>, <given-names>TE</given-names></string-name>, &amp; <string-name><surname>Holmes</surname>, <given-names>AP</given-names></string-name></person-group>. (<year>2002</year>). <article-title>Nonparametric permutation tests for functional neuroimaging: a primer with examples</article-title>. <source>Human Brain Mapping</source>, <volume>15</volume>(<issue>1</issue>), <fpage>1</fpage>–<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1002/hbm.1058</pub-id> <pub-id pub-id-type="pmid">11747097</pub-id></mixed-citation></ref>
<ref id="c35"><label>35.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pavlick</surname>, <given-names>E</given-names></string-name></person-group>. (<year>2022</year>). <article-title>Semantic structure in deep learning</article-title>. <source>Annual Review of Linguistics</source>, <volume>8</volume>(<issue>1</issue>), <fpage>447</fpage>–<lpage>471</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-linguistics-031120-122924</pub-id></mixed-citation></ref>
<ref id="c36"><label>36.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Pennington</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Socher</surname>, <given-names>R</given-names></string-name>, &amp; <string-name><surname>Manning</surname>, <given-names>C</given-names></string-name></person-group>. (<year>2014</year>). <article-title>Glove: Global vectors for word representation</article-title>. <conf-name>Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)</conf-name>. <pub-id pub-id-type="doi">10.3115/v1/d14-1162</pub-id></mixed-citation></ref>
<ref id="c37"><label>37.</label><mixed-citation publication-type="report"><person-group person-group-type="author"><string-name><surname>Radford</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Wu</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Child</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Luan</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Amodei</surname>, <given-names>D</given-names></string-name>, &amp; <string-name><surname>Sutskever</surname>, <given-names>I</given-names></string-name></person-group> (<year>2019</year>). <source>Language Models are Unsupervised Multitask Learners</source>. <publisher-name>OpenAI</publisher-name> <ext-link ext-link-type="uri" xlink:href="https://cdn.openai.com/better-language-models/language_models_are_unsupervised_multitask_learners.pdf">https://cdn.openai.com/better-language-models/language_models_are_unsupervised_multitask_learners.pdf</ext-link></mixed-citation></ref>
<ref id="c38"><label>38.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Richards</surname>, <given-names>BA</given-names></string-name>, <string-name><surname>Lillicrap</surname>, <given-names>TP</given-names></string-name>, <string-name><surname>Beaudoin</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Bengio</surname>, <given-names>Y</given-names></string-name>, <string-name><surname>Bogacz</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Christensen</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Clopath</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Costa</surname>, <given-names>RP</given-names></string-name>, <string-name><surname>de Berker</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Ganguli</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Gillon</surname>, <given-names>CJ</given-names></string-name>, <string-name><surname>Hafner</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Kepecs</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Kriegeskorte</surname>, <given-names>N</given-names></string-name>, <string-name><surname>Latham</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Lindsay</surname>, <given-names>GW</given-names></string-name>, <string-name><surname>Miller</surname>, <given-names>KD</given-names></string-name>, <string-name><surname>Naud</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Pack</surname>, <given-names>CC</given-names></string-name>, <etal>…</etal> <string-name><surname>Kording</surname>, <given-names>KP</given-names></string-name></person-group> (<year>2019</year>). <article-title>A deep learning framework for neuroscience</article-title>. <source>Nature Neuroscience</source>, <volume>22</volume>(<issue>11</issue>), <fpage>1761</fpage>–<lpage>1770</lpage>. <pub-id pub-id-type="doi">10.1038/s41593-019-0520-2</pub-id> <pub-id pub-id-type="pmid">31659335</pub-id></mixed-citation></ref>
<ref id="c39"><label>39.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Schrimpf</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Blank</surname>, <given-names>IA</given-names></string-name>, <string-name><surname>Tuckute</surname>, <given-names>G</given-names></string-name>, <string-name><surname>Kauf</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Hosseini</surname>, <given-names>EA</given-names></string-name>, <string-name><surname>Kanwisher</surname>, <given-names>N</given-names></string-name>, <string-name><surname>Tenenbaum</surname>, <given-names>JB</given-names></string-name>, &amp; <string-name><surname>Fedorenko</surname>, <given-names>E</given-names></string-name></person-group>. (<year>2021</year>). <article-title>The neural architecture of language: Integrative modeling converges on predictive processing</article-title>. <source>Proceedings of the National Academy of Sciences of the United States of America</source>, <volume>118</volume>(<issue>45</issue>). <pub-id pub-id-type="doi">10.1073/pnas.2105646118</pub-id> <pub-id pub-id-type="pmid">34737231</pub-id></mixed-citation></ref>
<ref id="c40"><label>40.</label><mixed-citation publication-type="web"><person-group person-group-type="author"><string-name><surname>Chivvis</surname> <given-names>D</given-names></string-name></person-group>. <article-title>So a monkey and a horse walk into a bar</article-title> (<date-in-citation><year>2017</year>, <month>November</month>, 10<day>10</day></date-in-citation>). <source>This American Life</source>. <ext-link ext-link-type="uri" xlink:href="https://www.thisamericanlife.org/631/so-a-monkey-and-a-horse-walk-into-a-bar">https://www.thisamericanlife.org/631/so-a-monkey-and-a-horse-walk-into-a-bar</ext-link></mixed-citation></ref>
<ref id="c41"><label>41.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Sutton</surname>, <given-names>R</given-names></string-name></person-group>. (<year>2019</year>). <article-title>The bitter lesson</article-title>. <source>Incomplete Ideas</source>.</mixed-citation></ref>
<ref id="c42"><label>42.</label><mixed-citation publication-type="web"><person-group person-group-type="author"><collab-name>Carnegie Mellon University</collab-name></person-group>. (<year>n.d.</year>). <source>The CMU Pronouncing Dictionary</source> Retrieved May 27, 2025, from <ext-link ext-link-type="uri" xlink:href="http://www.speech.cs.cmu.edu/cgi-bin/cmudict">http://www.speech.cs.cmu.edu/cgi-bin/cmudict</ext-link></mixed-citation></ref>
<ref id="c43"><label>43.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Toneva</surname>, <given-names>M</given-names></string-name>, &amp; <string-name><surname>Wehbe</surname>, <given-names>L</given-names></string-name></person-group>. (<year>2019</year>). <article-title>Interpreting and improving natural-language processing (in machines) with natural language-processing (in the brain)</article-title>. <conf-name>Advances in Neural Information Processing Systems</conf-name>, <volume>32</volume>. <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2019/hash/749a8e6c231831ef7756db230b4359c8-Abstract.html">https://proceedings.neurips.cc/paper/2019/hash/749a8e6c231831ef7756db230b4359c8-Abstract.html</ext-link></mixed-citation></ref>
<ref id="c44"><label>44.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Touvron</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Martin</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Stone</surname>, <given-names>K</given-names></string-name>, <string-name><surname>Albert</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Almahairi</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Babaei</surname>, <given-names>Y</given-names></string-name>, <string-name><surname>Bashlykov</surname>, <given-names>N</given-names></string-name>, <string-name><surname>Batra</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Bhargava</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Bhosale</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Bikel</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Blecher</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Ferrer</surname>, <given-names>CC</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Cucurull</surname>, <given-names>G</given-names></string-name>, <string-name><surname>Esiobu</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Fernandes</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Fu</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Fu</surname>, <given-names>W</given-names></string-name>, <etal>…</etal> <string-name><surname>Scialom</surname>, <given-names>T</given-names></string-name></person-group> (<year>2023</year>). <article-title>Llama 2: Open Foundation and Fine-Tuned Chat Models</article-title>. <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/arxiv.2307.09288</pub-id></mixed-citation></ref>
<ref id="c45"><label>45.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Vaidya</surname>, <given-names>AR</given-names></string-name>, <string-name><surname>Jain</surname>, <given-names>S</given-names></string-name>, &amp; <string-name><surname>Huth</surname>, <given-names>AG</given-names></string-name></person-group>. (<year>2022</year>). <article-title>Self-supervised models of audio effectively explain human cortical responses to speech</article-title>. <conf-name>Icml</conf-name> <volume>2022</volume>. <pub-id pub-id-type="doi">10.48550/ARXIV.2205.14252</pub-id></mixed-citation></ref>
<ref id="c46"><label>46.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Wolf</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Debut</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Sanh</surname>, <given-names>V</given-names></string-name>, <string-name><surname>Chaumond</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Delangue</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Moi</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Cistac</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Rault</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Louf</surname>, <given-names>R</given-names></string-name>, <string-name><surname>Funtowicz</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Davison</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Shleifer</surname>, <given-names>S</given-names></string-name>, <string-name><surname>von Platen</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Ma</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Jernite</surname>, <given-names>Y</given-names></string-name>, <string-name><surname>Plu</surname>, <given-names>J</given-names></string-name>, <string-name><surname>Xu</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Scao</surname>, <given-names>TL</given-names></string-name>, <string-name><surname>Gugger</surname>, <given-names>S</given-names></string-name>, <etal>…</etal> <string-name><surname>Rush</surname>, <given-names>AM</given-names></string-name></person-group> (<year>2019</year>). <article-title>HuggingFace’s transformers: State-of-the-art natural language processing</article-title>. <source>arXiv</source> <pub-id pub-id-type="doi">10.48550/ARXIV.1910.03771</pub-id></mixed-citation></ref>
<ref id="c47"><label>47.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Xie</surname>, <given-names>SM</given-names></string-name>, <string-name><surname>Raghunathan</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Liang</surname>, <given-names>P</given-names></string-name>, &amp; <string-name><surname>Ma</surname>, <given-names>T</given-names></string-name></person-group>. (<year>2021</year>). <article-title>An explanation of in-context learning as implicit Bayesian inference</article-title>. <source>arXiv</source> <pub-id pub-id-type="doi">10.48550/arXiv.2111.02080</pub-id></mixed-citation></ref>
<ref id="c48"><label>48.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yuan</surname>, <given-names>J</given-names></string-name>, &amp; <string-name><surname>Liberman</surname>, <given-names>M</given-names></string-name></person-group>. (<year>2008</year>). <article-title>Speaker identification on the SCOTUS corpus</article-title>. <source>The Journal of the Acoustical Society of America</source>, <volume>123</volume>(<issue>5</issue>), <fpage>3878</fpage>–<lpage>3878</lpage>. <pub-id pub-id-type="doi">10.1121/1.2935783</pub-id></mixed-citation></ref>
<ref id="c49"><label>49.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zada</surname>, <given-names>Z</given-names></string-name>, <string-name><surname>Nastase</surname>, <given-names>SA</given-names></string-name>, <string-name><surname>Aubrey</surname>, <given-names>B</given-names></string-name>, <string-name><surname>Jalon</surname>, <given-names>I</given-names></string-name>, <string-name><surname>Michelmann</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>H</given-names></string-name>, <string-name><surname>Hasenfratz</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Doyle</surname>, <given-names>W</given-names></string-name>, <string-name><surname>Friedman</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Dugan</surname>, <given-names>P</given-names></string-name>, <string-name><surname>Melloni</surname>, <given-names>L</given-names></string-name>, <string-name><surname>Devore</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Flinker</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Devinsky</surname>, <given-names>O</given-names></string-name>, <string-name><surname>Goldstein</surname>, <given-names>A</given-names></string-name>, &amp; <string-name><surname>Hasson</surname>, <given-names>U</given-names></string-name></person-group>. (<year>2025</year>). <article-title>The “Podcast” ECoG dataset for modeling neural activity during natural language comprehension</article-title>. <source>Scientific Data</source>, <volume>12</volume>(<issue>1</issue>), <fpage>1135</fpage>. <pub-id pub-id-type="doi">10.1038/s41597-025-05462-2</pub-id> <pub-id pub-id-type="pmid">40610484</pub-id></mixed-citation></ref>
<ref id="c50"><label>50.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Bengio</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Hardt</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Recht</surname>, <given-names>B</given-names></string-name>, &amp; <string-name><surname>Vinyals</surname>, <given-names>O</given-names></string-name></person-group>. (<year>2021</year>). <article-title>Understanding deep learning (still) requires rethinking generalization</article-title>. <source>Communications of the ACM</source>, <volume>64</volume>(<issue>3</issue>), <fpage>107</fpage>–<lpage>115</lpage>. <pub-id pub-id-type="doi">10.1145/3446776</pub-id></mixed-citation></ref>
<ref id="c51"><label>51.</label><mixed-citation publication-type="preprint"><person-group person-group-type="author"><string-name><surname>Zhang</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Roller</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Goyal</surname>, <given-names>N</given-names></string-name>, <string-name><surname>Artetxe</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Dewan</surname>, <given-names>C</given-names></string-name>, <string-name><surname>Diab</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>X</given-names></string-name>, <string-name><surname>Lin</surname>, <given-names>XV</given-names></string-name>, <string-name><surname>Mihaylov</surname>, <given-names>T</given-names></string-name>, <string-name><surname>Ott</surname>, <given-names>M</given-names></string-name>, <string-name><surname>Shleifer</surname>, <given-names>S</given-names></string-name>, <string-name><surname>Shuster</surname>, <given-names>K</given-names></string-name>, <string-name><surname>Simig</surname>, <given-names>D</given-names></string-name>, <string-name><surname>Koura</surname>, <given-names>PS</given-names></string-name>, <string-name><surname>Sridhar</surname>, <given-names>A</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>T</given-names></string-name>, &amp; <string-name><surname>Zettlemoyer</surname>, <given-names>L</given-names></string-name></person-group> (<year>2022</year>). <article-title>OPT: Open Pre-trained Transformer Language Models</article-title>. <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/arxiv.2205.01068</pub-id></mixed-citation></ref>
</ref-list>
</back>
<sub-article id="sa0" article-type="editor-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.101204.2.sa4</article-id>
<title-group>
<article-title>eLife Assessment</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ding</surname>
<given-names>Nai</given-names>
</name>
<role specific-use="editor">Reviewing Editor</role>
<contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0003-3428-2723</contrib-id>
<aff>
<institution-wrap>
<institution-id institution-id-type="ror">https://ror.org/00a2xv884</institution-id><institution>Zhejiang University</institution>
</institution-wrap>
<city>Hangzhou</city>
<country>China</country>
</aff>
</contrib>
</contrib-group>
<kwd-group kwd-group-type="claim-importance">
<kwd>Important</kwd>
</kwd-group>
<kwd-group kwd-group-type="evidence-strength">
<kwd>Solid</kwd>
</kwd-group>
</front-stub>
<body>
<p>This <bold>important</bold> study investigates how the size of an LLM may influence its ability to model the human neural response to language recorded by ECoG. Overall, <bold>solid</bold> evidence is provided that larger language models can better predict the human ECoG response. This study will be of interest to both neuroscientists and psychologists who work on language comprehension and computer scientists working on LLMs.</p>
</body>
</sub-article>
<sub-article id="sa1" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.101204.2.sa3</article-id>
<title-group>
<article-title>Reviewer #1 (Public review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>Summary:</p>
<p>The authors perform an analysis of the relationship between the size of an LMM and the predictive performance of an ECoG encoding model made using the representations from that LMM. They find a logarithmic relationship between model size and prediction performance, consistent with previous findings in fMRI. They additionally observe that as the model size increases, the location of the &quot;peak&quot; encoding performance typically moves further back into the model in terms of percent layer depth, an interesting result worthy of further analysis into these representations.</p>
<p>Strengths:</p>
<p>The evidence is quite convincing, consistent across model families and complementary to other work in this field. This sort of analysis for ECoG is needed and supports the decade-long enduring trend of the &quot;virtuous cycle&quot; between neuroscience and AI research, where more powerful AI models have consistently yielded more effective predictions of responses in the brain. The lag analysis showing that optimal lags do not change with model size is a nice result using the higher temporal resolution of ECoG compared to other methods like fMRI.</p>
<p>Comments on revised version.</p>
<p>After the latest revision, I am pleased to remove my previous remarks about weaknesses of the paper, as I believe the additional data scaling analysis, discussion of layerwise trends, and other additional commentary makes the paper a compelling addition to the literature.</p>
</body>
</sub-article>
<sub-article id="sa2" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.101204.2.sa2</article-id>
<title-group>
<article-title>Reviewer #2 (Public review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>Summary:</p>
<p>This paper investigates whether large language models (LLMs) of increasing size more accurately align with brain activity during naturalistic language comprehension. The authors extracted word embeddings from LLMs for each word in a 30-minute story and regressed them against electrocorticography (ECoG) activity time-locked to each word as participants listened to the story. The findings reveal that larger LLMs more effectively predict ECoG activity, reflecting the scaling laws observed in other natural language processing tasks.</p>
<p>Strengths:</p>
<p>(1) The study compared model activity with ECoG recordings, which offer much better temporal resolution than other neuroimaging methods, allowing for the examination of model encoding performance across various lags relative to word onset.</p>
<p>(2) The range of LLMs tested is comprehensive, spanning from 82 million to 70 billion parameters. This serves as a valuable reference for researchers selecting LLMs for brain encoding and decoding studies.</p>
<p>(3) The regression methods used are well-established in prior research, and the results demonstrate a convincing scaling law for the brain encoding ability of LLMs. The consistency of these results after PCA dimensionality reduction further supports the claim.</p>
<p>Comments on revised version.</p>
<p>I thank the authors very much for their efforts in addressing my comments. One remaining concern is the extent of the paper's conceptual advance. Several recent studies have made broadly similar claims regarding the increasing alignment between large language models and human language processing, although using fMRI data (Antonello et al., 2023; Gao et al., 2025). I would therefore encourage the authors to more clearly articulate what additional insights are gained from using ECoG. Clarifying this point would help better establish the novelty and contribution of the present study.</p>
<p>Antonello, R. J., Vaidya, A. R., &amp; Huth, A. G. (2023). Scaling laws for language encoding models in fMRI. Advances in Neural Information Processing Systems, 36, 21895-21907.</p>
<p>Gao, C., Ma, Z., Chen, J., Li, P., Huang, S., &amp; Li, J. (2025). Increasing alignment of large language models with language processing in the human brain. Nature Computational Science, 5(11), 1080-1090.</p>
</body>
</sub-article>
<sub-article id="sa3" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.101204.2.sa1</article-id>
<title-group>
<article-title>Reviewer #3 (Public review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>This manuscript studies the connection between neural activity collected through electrocorticography and hidden vector representations from autoregressive language models, with the specific aim of studying the influence of language model size on this connection. Neural activity was measured from subjects that listened to a segment from a podcast, and the representations from language models were calculated using the written transcription as the input text. The ability of vector representations to predict neural activity was evaluated using 10-fold cross-validation with ridge regression models.</p>
<p>The main results are that (as well summarized in section headings):</p>
<p>
(1) Larger models predict neural activity better.</p>
<p>(2) The ability of language model representations to predict neural activity differs across electrodes and brain regions.</p>
<p>(3) The layer that best predicts neural activity differs according to model size, with the &quot;SMALL&quot; model showing a correspondence between layer number and the language processing hierarchy.</p>
<p>(4) There seems to be a similar relationship between the time lag and the ability of language model representations to predict neural activity across models.</p>
<p>Strengths:</p>
<p>(1) The experimental and modeling protocols generally seem solid, which yielded results that answer the authors' primary research question.</p>
<p>(2) Electrocorticography data is especially hard to collect, so these results make a nice addition to recent functional magnetic resonance imaging studies.</p>
<p>Weaknesses:</p>
<p>(1) The interpretation of some results seems unjustified, although this may just be a presentational issue.</p>
<p>a) Figure 2B: The authors interpret the results as &quot;a plateau in the maximal encoding performance,&quot; when some readers might interpret this rather as a decline after 13 billion parameters. Can this be further supported by a significance test like that shown in Figure 4B?</p>
<p>b) Figure S1A: It looks like the drop in PCA max correlation is larger for larger models, which may suggest to some readers that the same trend observed for ridge max correlation may not hold, contra the authors' claim that all results replicate. Why not include a similar figure as Figure 2B as part of Figure S1?</p>
<p>(2) Discussion of what might be driving the main result about the influence of model size appears to be missing (cf. the authors aim to provide an explanation of what seems to drive the influence of the layer location in Paragraph 3 of the Discussion section). What explanations have been proposed in the previous functional magnetic resonance imaging studies? Do those explanations also hold in the context of this study?</p>
<p>(3) The GloVe-based selection of language-sensitive electrodes (at least to me) isn't explained/motivated clearly enough (I think a more detailed explanation should be included in the Materials and Methods section). If the electrodes are selected based on GloVe embeddings, then isn't the main experiment just showing that representations from larger language models track more closely with GloVe embeddings? What justifies this methodology?</p>
<p>(4) (Minor weakness) The main experiments are largely replications of previous functional magnetic resonance imaging studies, with the exception of the one lag-based analysis. Is there anything else that the electrocorticography data can reveal that functional magnetic resonance imaging data can't?</p>
<p>Comments on revised version.</p>
<p>I reread the manuscript, my previous review, and the authors' response to it. I thank the authors for clarifying any misunderstanding from my end (e.g. the different LLM tokenizers) and feel that the authors addressed my concerns very carefully.</p>
</body>
</sub-article>
<sub-article id="sa4" article-type="author-comment">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.101204.2.sa0</article-id>
<title-group>
<article-title>Author response:</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Hong</surname>
<given-names>Zhuoqiao</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Haocheng</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zada</surname>
<given-names>Zaid</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gazula</surname>
<given-names>Harshvardhan</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Turner</surname>
<given-names>David</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Aubrey</surname>
<given-names>Bobbi</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Niekerken</surname>
<given-names>Leonard</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Doyle</surname>
<given-names>Werner</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Devore</surname>
<given-names>Sasha</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dugan</surname>
<given-names>Patricia</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Friedman</surname>
<given-names>Daniel</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Devinsky</surname>
<given-names>Orrin</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Flinker</surname>
<given-names>Adeen</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hasson</surname>
<given-names>Uri</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nastase</surname>
<given-names>Samuel A</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Goldstein</surname>
<given-names>Ariel</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>The following is the authors’ response to the original reviews</p>
<disp-quote content-type="editor-comment">
<p><bold>Public Reviews:</bold></p>
<p><bold>Reviewer #1 (Public review):</bold></p>
<p>Summary:</p>
<p>The authors perform an analysis of the relationship between the size of an LMM and the predictive performance of an ECoG encoding model made using the representations from that LMM. They find a logarithmic relationship between model size and prediction performance, consistent with previous findings in fMRI. They additionally observe that as the model size increases, the location of the &quot;peak&quot; encoding performance typically moves further back into the model in terms of percent layer depth, an interesting result worthy of further analysis into these representations.</p>
<p>Strengths:</p>
<p>The evidence is quite convincing, consistent across model families, and complementary to other work in this field. This sort of analysis for ECoG is needed and supports the decade-long enduring trend of the &quot;virtuous cycle&quot; between neuroscience and AI research, where more powerful AI models have consistently yielded more effective predictions of responses in the brain. The lag analysis showing that optimal lags do not change with model size is a nice result using the higher temporal resolution of ECoG compared to other methods like fMRI.</p>
</disp-quote>
<p>We thank the reviewer for their thoughtful assessment! We agree that the “virtuous cycle” between neuroscience and AI research has been, and will continue to be, a driving force in advancing our understanding of brain function through more powerful predictive models. We are especially pleased that the reviewer appreciated the lag analysis, as we view this as a valuable complement to the existing fMRI work.</p>
<disp-quote content-type="editor-comment">
<p>Weaknesses:</p>
<p>I would have liked to have seen the data scaling trends explored a bit too, as this is somewhat analogous to the main scaling results. While better performance with more data might be unsurprising, showing good data scaling would be a strong and useful justification for additional data collection in the field, especially given the extremely limited amount of existing language ECoG data. I realize that the data here is somewhat limited (only 30 minutes per subject), but authors could still in principle train models on subsets of this data.</p>
</disp-quote>
<p>We thank the reviewer for their valuable suggestion. For the revised manuscript, we performed a new analysis where we trained encoding models using subsets of the data (randomly sampling contiguous chunks of 50%, 25%, and 10% of all words in each of the training folds) and tested these models on all words in the test fold. As expected, we found that encoding performance increases as the training dataset size increases, suggesting that model performance scales with data quantity even within the constraints of our relatively small dataset. This result reinforces the importance of collecting dense ECoG data. We have added the following text to our Results section: “We also built encoding models using subsets of the data and found that encoding performance increases as the volume of training data increases (Fig. S6)” and included the results as a supplementary figure 6 in the revised manuscript.</p>
<disp-quote content-type="editor-comment">
<p>Separately, it would be nice to have better justification of some of these trends, in particular the peak layerwise encoding performance trend and the overall upside-down U-trend of encoding performance across layers more generally. There is clearly something very fundamental going on here, about the nature of abstraction patterns in LLMs and in the brain, and this result points to that. I don't see the lack of justification here as a critical issue, but the paper would certainly be better with some theoretical explanation for why this might be the case.</p>
</disp-quote>
<p>We thank the reviewer for this insightful comment. The general inverted U-shaped trend of encoding performance across layers has been a frequently observed phenomenon in studies comparing LLM representations to brain activity (Goldstein, Ham, et al., 2025; Schrimpf et al., 2021). A potential explanation is the existence of a “two-phase abstraction process” within LLMs (Cheng &amp; Antonello, 2024; Csordás et al., 2025). In the initial layers, models begin by processing relatively low-level input features. As layers get deeper, representations become increasingly abstract and richly contextualized in semantic features relevant for understanding language. These intermediate layers often show the highest correlation with brain activity in language areas, presumably because they capture complex semantic and contextual information in a way that generalizes well across a variety of tasks (including prediction of human neural activity) (Antonello &amp; Huth, 2024). Subsequently, a prediction phase happens in the later layers, where the representations become more specialized for the LLM's specific training objective (e.g., next-word prediction). This specialization can effectively constrict the more generalized feature representations, making these layers less optimal for predicting brain activity. These observations suggest that it is primarily the abstractive, contextual features developed in the intermediate layers of LLMs that drive their alignment with brain activity. As models become more potent at prediction, their most predictive layers (for the LLM’s natural language task) and their most generalizable layers (for brain activity) can diverge.</p>
<p>A key finding in our study is that the initial processing phase does not scale and take up more layers as models scale up in size and layers. Larger models develop the necessary rich, abstract representations in the same number of layers as smaller models. Consequently, the prediction phase may begin relatively earlier in these larger models, and the later layers could develop highly specialized representations that are increasingly divergent from the more general linguistic processing captured in brain activity. For example, these layers may specialize in capturing very specific patterns of language (thus lowering their perplexity) that do not actually occur often or at all in our naturalistic dataset. It is also possible that the later layers of larger models are overall underutilized and do not contribute as much to linguistic processing and next-word prediction (Csordás et al., 2025).</p>
<p>We have added the following text to our Discussion section:</p>
<p>“The inverted U-shaped trend of encoding performance commonly found in previous research is likely due to a &quot;two-phase abstraction process&quot; within LLMs (Cheng &amp; Antonello, 2024; Csordás et al., 2025). In the early and intermediate layers of the model, a composition phase occurs, where low-level input features become increasingly abstract and contextualized. The intermediate layers of the model show the highest correlation with brain activity, presumably because they capture complex semantic and contextual information in a way that generalizes well across a variety of tasks (including prediction of human neural activity) (Antonello &amp; Huth, 2024). Subsequently, a prediction phase happens in the later layers of the model, where the representations become more specialized for the LLM's specific training objective (e.g., next-word prediction). This specialization can effectively constrict the more generalized feature representations, making these layers less optimal for predicting brain activity. Our results indicate that the initial composition phase does not take up more layers as models scale up in size. Larger models develop the necessary rich, abstract representations in the same number of layers as smaller models. Thus, as LLMs increase in size, the later layers of the model may contain representations that are increasingly divergent from the more general linguistic processing captured in brain activity. It is also possible that the later layers of larger models are overall underutilized and may not significantly contribute to benchmark performances during inference (Csordás et al., 2025; Fan et al., 2024; Gromov et al., 2024).”</p>
<disp-quote content-type="editor-comment">
<p>Lastly, I would have wanted to see a similar analysis here done for audio encoding models using Whisper or WavLM as this is the modality where you might see real differences between ECoG and other slower scanning approaches. Again, I do not see this omission as a fundamental issue, but it does seem like the sort of analysis for which the higher temporal resolution of ECoG might grant some deeper insight.</p>
</disp-quote>
<p>We appreciate this suggestion. In a separate project, we focused on multimodal audio-to-speech-to-language large language models (LLMs), building encoding models using Whisper embeddings (from both the encoder and decoder stacks) to predict electrocorticographic (ECoG) signals during naturalistic conversations (Goldstein, Wang, et al., 2025). The higher temporal resolution of ECoG enables us to trace the temporal flow of information from the superior temporal gyrus (STG) and somatomotor areas (SM) to the inferior frontal gyrus (IFG) during speech comprehension. Conversely, during speech production, encoding in IFG peaked significantly earlier than in the STG and SM. We agree that scaling encoding models using multimodal approaches and our ECoG conversation datasets could yield valuable insights, and we look forward to exploring this in future work. However, we feel that the added complexity of multimodal encoding models falls beyond the scope of this paper.</p>
<p>We have modified the following text to our Discussion section:</p>
<p>“Since we exclusively employ textual LLMs, which lack inherent temporal information due to their discrete token-based nature, future studies utilizing multimodal LLMs integrating continuous audio or video streams, like Whisper or WavLM may better unravel the relationship between model size and temporal dynamic representations in LLMs (Goldstein, Wang, et al., 2025; Millet et al., 2023; Vaidya et al., 2022).”</p>
<disp-quote content-type="editor-comment">
<p><bold>Reviewer #2 (Public review):</bold></p>
<p>Summary:</p>
<p>This paper investigates whether large language models (LLMs) of increasing size more accurately align with brain activity during naturalistic language comprehension. The authors extracted word embeddings from LLMs for each word in a 30-minute story and regressed them against electrocorticography (ECoG) activity time-locked to each word as participants listened to the story. The findings reveal that larger LLMs more effectively predict ECoG activity, reflecting the scaling laws observed in other natural language processing tasks.</p>
<p>Strengths:</p>
<p>(1) The study compared model activity with ECoG recordings, which offer much better temporal resolution than other neuroimaging methods, allowing for the examination of model encoding performance across various lags relative to word onset.</p>
<p>(2) The range of LLMs tested is comprehensive, spanning from 82 million to 70 billion parameters. This serves as a valuable reference for researchers selecting LLMs for brain encoding and decoding studies.</p>
<p>(3) The regression methods used are well-established in prior research, and the results demonstrate a convincing scaling law for the brain encoding ability of LLMs. The consistency of these results after PCA dimensionality reduction further supports the claim.</p>
</disp-quote>
<p>We thank the reviewer for their thoughtful and positive feedback.</p>
<disp-quote content-type="editor-comment">
<p>Weaknesses:</p>
<p>(1) Some claims of the paper are less convincing. The authors suggested that &quot;scaling could be a property that the human brain, similar to LLMs, can utilize to enhance performance&quot;, however, many other animals have brains with more neurons than the human brain, making it unlikely that simple scaling alone leads to better language performance.</p>
</disp-quote>
<p>We thank the reviewer for this insightful comment. We agree that simply having more neurons does not automatically confer more complex or human-like cognitive or linguistic capabilities. This suggestion deserves a more nuanced treatment than we had included in the original manuscript.</p>
<p>Research in comparative neuroscience has argued that human cognitive abilities emerge from scaling up the primate brain (Herculano-Houzel, 2012). However, the critical aspect is not merely the number of neurons, but how these neurons contribute to computational power within a specific evolutionary and cultural context. The uniqueness of human cognition has been argued to result from a global adaptation for increased information processing capacity (Cantlon &amp; Piantadosi, 2024). Moreover, the language network in humans is likely grounded in the evolution of particular structural networks in the primate brain (Friederici &amp; Becker, 2025). This suggests that the way brain regions are connected and the expansion of certain pathways are critical, not just the overall scale. Furthermore, the specialized structure must be tuned by its learning environment and training data. For example, both humans and LLMs learn from language data generated by other humans, which reflects world knowledge that has accumulated over many generations.</p>
<p>We have modified the following text in the Introduction:</p>
<p>“Research in comparative neuroscience has suggested that uniquely human cognitive abilities emerge from scaling up the primate brain (Herculano-Houzel, 2012).”</p>
<p>We also added a caveat to the Discussion on this point:</p>
<p>“As in the human brain, while scaling alone may yield emergent cognitive abilities (Cantlon &amp; Piantadosi, 2024; Herculano-Houzel, 2012), specialized architectural features likely also play a critical role (Friederici &amp; Becker, 2025).”</p>
<disp-quote content-type="editor-comment">
<p>Additionally, the authors claim that their results show 'larger models better predict the structure of natural language.' However, it remains unclear to what extent the embeddings of LLMs capture the &quot;structure&quot; of language better than the lexical semantics of language.</p>
</disp-quote>
<p>We appreciate the reviewer's point about how well LLM embeddings capture the &quot;structure&quot; of language versus just lexical semantics. It's true that distinguishing these aspects is complex. From our perspective, a model's ability to predict/produce natural language entails that the model captures various levels of linguistic structure, including morphology, syntax, semantics, and contextual dependencies. We use &quot;structure&quot; inclusively in this sense. A model cannot achieve high predictive accuracy without representing, to some extent, all of these structural elements (Linzen &amp; Baroni, 2021; Manning et al., 2020; Pavlick, 2022). There is a very active field of research into understanding exactly <italic>how</italic> these models represent these different structures of language (Ameisen et al., 2025; Chemla et al., 2024; Elhage et al., 2021, 2022; Hewitt &amp; Manning, 2019). Our results confirm the core trend that larger models tend to better reproduce the various structures of language (i.e., yield lower perplexity; Fig. 2A).</p>
<p>In previous work, we have shown that LLM embeddings better predict neural activity during natural language processing than lexical embeddings (e.g., GloVe) that do not contain other elements of linguistic structure (Goldstein et al., 2022; Kumar et al., 2024; Zada et al., 2024). In response to the following comment, we also compare LLMs to simpler models capturing specific speech and language features (see next comment). To clarify our intended use of the word “structure”, we’ve added a brief explanation in the Methods section:</p>
<p>“In this study, we use the term “structure” to refer to a variety of linguistic patterns (e.g., morphology, syntax, semantics, context) that LLMs encode in order to better predict natural language.”</p>
<disp-quote content-type="editor-comment">
<p>(2) The study lacks control LLMs with randomly initialized weights and control regressors, such as word frequency and phonetic features of speech, making it unclear what the baseline is for the model-brain correlation.</p>
</disp-quote>
<p>We’ve added several supplementary analyses to the revised manuscript to address these concerns. To establish a baseline, we extracted embeddings from each layer of the SMALL model with randomly initialized weights and constructed encoding models. The encoding performance is significantly higher for pretrained SMALL than for untrained SMALL for every layer (Fig. S4). For the untrained model, the performance is the highest for the 0th layer and decreases in subsequent layers. This is because at the 0th layer, every instance of the same word receives an identical, albeit random, embedding (See Supplementary Figure 4).</p>
<p>We also compared the encoding performance of LLMs with more classical speech/language features and static GloVe embeddings (Goldstein, Wang, et al., 2025; Kumar et al., 2024). First, we extracted features capturing lower-level speech features. Using the stimulus transcript as input, we created one-hot vectors for phonetic and articulatory features. Phoneme classes (39 total classes) were obtained from the Carnegie Mellon Pronouncing Dictionary (The CMU Pronouncing Dictionary, n.d.). We further classified the phonemes based on their place of articulation (9 classes), manner of articulation (9 classes), and voiced or voiceless status (3 classes), according to the general American English consonants of the International Phonetic Alphabet. Given that each word consists of multiple phonemes, we averaged the one-hot vectors for all phonetic and articulatory features for each word.</p>
<p>Second, we extracted linguistic features using spaCy (Honnibal et al., 2020), including part of speech (17 classes), tag (50 classes), function or content word (3 classes), dependency (45 classes), whether the word is an alpha character (binary), and whether the word is a stop word (binary). We also extracted prefix (30 classes) and suffix (44 classes) information using the Cambridge Dictionary. We constructed one-hot vectors for each multi-class feature and one-dimensional vectors for each binary feature.</p>
<p>Third, for each word, we obtained word frequency from the Google Web Trillion Word Corpus (Brants &amp; Franz, 2006) and from our own dataset.</p>
<p>Fourth, we generated static word embeddings of dimension 50 using GloVe (Pennington et al., 2014).</p>
<p>We then built encoding models in the same way as the contextual embeddings for each of the three categories of speech features, all speech features concatenated, and the GloVe embeddings. To control for the different dimensions of the embeddings, we also standardized all embeddings to the same size (50 dimensions) using principal component analysis (PCA) and trained linear encoding models using ordinary least-squares (OLS) regression. For both ridge and OLS encoding, our contextual embeddings from LLMs showed significantly better performance than the classic speech features and GloVe embeddings.</p>
<p>We have added the following text to our manuscript and updated our Figures S4, S5, Table S1, and the methods section:</p>
<p>“To establish a general baseline for encoding performance, we built encoding models using embeddings from the SMALL model with randomly initialized weights. The trained SMALL model exhibits significantly higher encoding performance across all layers compared to the untrained SMALL model (Fig. S4). We also assessed the encoding performance of contextual embeddings from LLMs against classic speech features and static GloVe embeddings (Table S1). The SMALL and XL embeddings achieved markedly higher encoding correlations than the speech features and GloVe embeddings (Fig. S5).”</p>
<disp-quote content-type="editor-comment">
<p>(3) The finding that peak encoding performance tends to occur in relatively earlier layers in larger models is somewhat surprising and requires further explanation. Since more layers mean more parameters, if the later layers diverge from language processing in the brain, it raises the question of what aspects of the larger models make them more brain-like.</p>
</disp-quote>
<p>We thank the reviewer for this insightful comment; this point was also highlighted by Reviewer 1. We agree that this result is somewhat surprising, and we aim to provide a more detailed explanation in the revised manuscript. The general inverted U-shaped trend of encoding performance across layers has been a frequently observed phenomenon in studies comparing LLM representations to brain activity (Goldstein, Ham, et al., 2025; Schrimpf et al., 2021). A potential explanation is the existence of a “two-phase abstraction process” within LLMs (Cheng &amp; Antonello, 2024; Csordás et al., 2025). In the initial layers, models begin by processing relatively low-level input features. As layers get deeper, representations become increasingly abstract and richly contextualized in semantic features relevant for understanding language. These intermediate layers often show the highest correlation with brain activity in language areas, presumably because they capture complex semantic and contextual information in a way that generalizes well across a variety of tasks (including prediction of human neural activity) (Antonello &amp; Huth, 2024). Subsequently, a prediction phase happens in the later layers, where the representations become more specialized for the LLM's specific training objective (e.g., next-word prediction). This specialization can effectively constrict the more generalized feature representations, making these layers less optimal for predicting brain activity. These observations suggest that it is primarily the abstractive, contextual features developed in the intermediate layers of LLMs that drive their alignment with brain activity. As models become more potent at prediction, their most predictive layers (for the LLM’s natural language task) and their most generalizable layers (for brain activity) can diverge.</p>
<p>A key finding in our study is that the initial processing phase does not scale and take up more layers as models scale up in size and layers. Larger models develop the necessary rich, abstract representations in the same number of layers as smaller models.</p>
<p>Consequently, the prediction phase may begin relatively earlier in these larger models, and the later layers could develop highly specialized representations that are increasingly divergent from the more general linguistic processing captured in brain activity. For example, these layers may specialize in capturing very specific patterns of language (thus lowering their perplexity) that do not actually occur often or at all in our naturalistic dataset. It is also possible that the later layers of larger models are overall underutilized and do not contribute as much to linguistic processing and next-word prediction (Csordás et al., 2025).</p>
<p>We have added the following text to our Discussion section:</p>
<p>“The inverted U-shaped trend of encoding performance commonly found in previous research is likely due to a &quot;two-phase abstraction process&quot; within LLMs (Cheng &amp; Antonello, 2024; Csordás et al., 2025). In the early and intermediate layers of the model, a composition phase occurs, where low-level input features become increasingly abstract and contextualized. The intermediate layers of the model show the highest correlation with brain activity, presumably because they capture complex semantic and contextual information in a way that generalizes well across a variety of tasks (including prediction of human neural activity) (Antonello &amp; Huth, 2024). Subsequently, a prediction phase happens in the later layers of the model, where the representations become more specialized for the LLM's specific training objective (e.g., next-word prediction). This specialization can effectively constrict the more generalized feature representations, making these layers less optimal for predicting brain activity. Our results indicate that the initial composition phase does not take up more layers as models scale up in size. Larger models develop the necessary rich, abstract representations in the same number of layers as smaller models. Thus, as LLMs increase in size, the later layers of the model may contain representations that are increasingly divergent from the more general linguistic processing captured in brain activity. It is also possible that the later layers of larger models are overall underutilized and may not significantly contribute to benchmark performances during inference (Csordás et al., 2025; Fan et al., 2024; Gromov et al., 2024).”</p>
<disp-quote content-type="editor-comment">
<p><bold>Reviewer #3 (Public review):</bold></p>
<p>This manuscript studies the connection between neural activity collected through electrocorticography and hidden vector representations from autoregressive language models, with the specific aim of studying the influence of language model size on this connection. Neural activity was measured from subjects who listened to a segment from a podcast, and the representations from language models were calculated using the written transcription as the input text. The ability of vector representations to predict neural activity was evaluated using 10-fold cross-validation with ridge regression models.</p>
<p>The main results are that (as well summarized in section headings):</p>
<p>(1) Larger models predict neural activity better.</p>
<p>(2) The ability of language model representations to predict neural activity differs across electrodes and brain regions.</p>
<p>(3) The layer that best predicts neural activity differs according to model size, with the &quot;SMALL&quot; model showing a correspondence between layer number and the language processing hierarchy.</p>
<p>(4) There seems to be a similar relationship between the time lag and the ability of language model representations to predict neural activity across models.</p>
<p>Strengths:</p>
<p>(1) The experimental and modeling protocols generally seem solid, which yielded results that answer the authors' primary research question.</p>
<p>(2) Electrocorticography data is especially hard to collect, so these results make a nice addition to recent functional magnetic resonance imaging studies.</p>
</disp-quote>
<p>We thank the reviewer for their thoughtful and positive feedback.</p>
<disp-quote content-type="editor-comment">
<p>Weaknesses:</p>
<p>(1) The interpretation of some results seems unjustified, although this may just be a presentational issue.</p>
<p>(a) Figure 2B: The authors interpret the results as &quot;a plateau in the maximal encoding performance,&quot; when some readers might interpret this rather as a decline after 13 billion parameters. Can this be further supported by a significance test like that shown in Figure 4B?</p>
</disp-quote>
<p>We agree that this could be a subjective interpretation, so we conducted an additional analysis. We performed paired two-sided <italic><italic>t</italic></italic>-tests between best layer encoding performances averaged across electrodes (df = 159 electrodes), comparing all models with larger models. We found that after 13 billion parameters, only the encoding performance for OPT-66B, the largest model in the OPT family, is significantly worse than the encoding performance of some other smaller models, supporting the claim that the maximal encoding performance declines after 13 billion parameters. However, we did not find conclusive statistical evidence of a decline in encoding performance for other model families.</p>
<p>We have added the statistical results as Supplementary Figure 1.</p>
<p>We have also modified the following text in the manuscript:</p>
<p>“We also observed a plateau in the maximal encoding performance, occurring around 7 billion parameters (Fig. 2B), with a decline in performance for the OPT-66B model (Fig. S1).”</p>
<disp-quote content-type="editor-comment">
<p>(b) Figure S1A: It looks like the drop in PCA max correlation is larger for larger models, which may suggest to some readers that the same trend observed for ridge max correlation may not hold, contra the authors' claim that all results replicate. Why not include a similar figure as Figure 2B as part of Figure S1?</p>
</disp-quote>
<p>PCA is an unsupervised dimensionality reduction technique and may discard model features with small eigenvalues that nonetheless contribute to encoding performance. Ridge regression, a supervised method, can capitalize on these features. We suspect that this is why there appears to be a larger drop in model performance for larger models with PCA than with ridge regression. We replicated the logarithmic relationship between model size and encoding performance using PCA and ordinary least-squares (OLS) regression encoding models. We have updated Supplementary Figure 2.</p>
<disp-quote content-type="editor-comment">
<p>(2) Discussion of what might be driving the main result about the influence of model size appears to be missing (cf. the authors aim to provide an explanation of what seems to drive the influence of the layer location in Paragraph 3 of the Discussion section). What explanations have been proposed in the previous functional magnetic resonance imaging studies? Do those explanations also hold in the context of this study?</p>
</disp-quote>
<p>We suspect that the increased expressivity of larger models - that is, their improved sensitivity to nuanced structure in natural language - yields improved alignment to brain activity (given large enough samples of brain activity) (Antonello et al., 2023). This effect persists even when dimensionality is tightly controlled in our PCA-based analysis, indicating that the improved alignment with the brain is not a modeling artifact of dimensionality alone, but results from the structural representations learned by these larger models.</p>
<p>We have added the following text to our Discussion section:</p>
<p>“We suspect that the improved alignment with brain activity in larger models is driven by their increased expressivity and sensitivity to nuanced linguistic structure present in large-scale naturalistic datasets (Antonello et al., 2023).”</p>
<disp-quote content-type="editor-comment">
<p>(3) The GloVe-based selection of language-sensitive electrodes (at least to me) isn't explained/motivated clearly enough (I think a more detailed explanation should be included in the Materials and Methods section). If the electrodes are selected based on GloVe embeddings, then isn't the main experiment just showing that representations from larger language models track more closely with GloVe embeddings? What justifies this methodology?</p>
</disp-quote>
<p>We selected electrodes based on previously established methods (Goldstein et al., 2022). Our use of GloVe embeddings for electrode selection does not imply that larger language model representations are simply more closely aligned with GloVe embeddings. On the contrary, contextual embeddings from LLMs, which incorporate the word’s previous context, consistently outperform static embeddings like GloVe or word2vec (Fig. S3). Selecting electrodes using LLM embeddings would likely result in a slightly different, potentially larger set of electrodes (Goldstein et al., 2022), but would be more circular (Kriegeskorte et al., 2009). The GloVe-based electrode selection represents a more conservative approach by identifying words encoding linguistic content without biasing the selection directly toward any LLMs.</p>
<p>We have added the following text to our Method section:</p>
<p>“We used GloVe embeddings for electrode selection to avoid biasing our main results toward a particular LLM.”</p>
<disp-quote content-type="editor-comment">
<p>(4) (Minor weakness) The main experiments are largely replications of previous functional magnetic resonance imaging studies, with the exception of the one lag-based analysis. Is there anything else that the electrocorticography data can reveal that functional magnetic resonance imaging data can't?</p>
</disp-quote>
<p>We thank the reviewer for this thoughtful question. While we agree that a key contribution of our work corroborates previous fMRI findings, we would argue that using ECoG is not merely a replication but a crucial validation and extension of that work. It is important to validate these effects across distinct measurement modalities. In our work, we further observed a novel trend where the peak encoding performance tends to occur in relatively earlier layers for larger models. This is supported by recent studies suggesting that later layers of large LLMs may not significantly contribute to benchmark performance (Csordás et al., 2025). While scaling has been an effective method to improve LLM performance, including in encoding models, future research should explore the potential underutilization of the later layers as models scale.</p>
<p>Furthermore, ECoG data offers temporal resolution on the order of milliseconds, far superior to fMRI’s. Although we did not observe a relationship between model size and temporal lags in this study, future work should investigate the temporal dynamics of encoding that are accessible with ECoG (Goldstein, Ham, et al., 2025; Goldstein, Wang, et al., 2025).</p>
<disp-quote content-type="editor-comment">
<p><bold>Recommendations for the authors:</bold></p>
<p><bold>Reviewer #1 (Recommendations for the authors):</bold></p>
<p>Thank you to the authors for the fun and personally useful read.</p>
<p>I see in Supplementary Figure 1 the authors show a comparison of the performance between OLS vs. Ridge regression. Is the OLS model the only one that is working over PC features, or are both models using PC features? The current text is a bit unclear. My current understanding is that the comparison is between (OLS + PCA) and (Ridge with no PCA), but I am not sure.</p>
</disp-quote>
<p>The OLS model is the only one that works over PC features, following previous methods (Goldstein et al., 2022).</p>
<p>We have added the following text to our Results and Methods section for clarity:</p>
<p>“To control for the different embedding dimensionality across models, we standardized all embeddings to the same size using principal component analysis (PCA) and trained linear encoding models using ordinary least-squares (OLS) regression, replicating the logarithmic relationship but with significantly lower encoding performance overall (Fig. S2). The PC features are used by the OLS models only.”</p>
<disp-quote content-type="editor-comment">
<p>Clarification in the text would be appropriate. If this is the correct understanding, the authors should note in the main text that the ridge approach is more effective than the PCA approach, which is still the dominant approach to building linear encoding models in the field for some unjustifiable reason.</p>
</disp-quote>
<p>We thank the reviewer for pointing out the confusion. We have updated Supplementary Figure 2.</p>
<disp-quote content-type="editor-comment">
<p>How were the alpha values for ridge regression determined? Do you use the same ridge parameter for all electrodes or fit a different parameter for each electrode? This is not mentioned anywhere.</p>
</disp-quote>
<p>The alpha values are determined by cross-validation using the “RidgeCV” method from the “himalaya” package (Dupré la Tour et al., 2022). Specifically, we perform a grid search over cross-validation folds in the training data to find the best-performing alpha. The alpha parameter is specific to each ridge regression model, meaning each fold, lag, and electrode has a different alpha parameter.</p>
<p>We have added the following text to our manuscript:</p>
<p>“For each ridge regression model (for each fold, lag, and electrode), the alpha parameter is determined by cross-validation using the “RidgeCV” method from the “himalaya” package (Dupré la Tour et al., 2022).”</p>
<disp-quote content-type="editor-comment">
<p>It's not entirely clear to me how the authors handle tokens that do not terminate in words (such as the &quot;there&quot; + &quot;'s&quot; example in the text). My current reading of the text is that authors essentially ignore these half-word embeddings, doing one forward pass per word, rather than per token, but the current text is somewhat ambiguous.</p>
</disp-quote>
<p>If a word is tokenized into several tokens, like “there” and “‘s”, we average the token embeddings to get a word embedding.</p>
<p>We have added the following text to our Method section:</p>
<p>“To facilitate a fair comparison of the encoding effect across different models, we aligned all tokens in the story across all models. We averaged the token embeddings if a word is split into multiple tokens, resulting in one embedding per word for each model.”</p>
<disp-quote content-type="editor-comment">
<p>The authors describe the scaling relationship they find as a &quot;log-linear&quot; relationship. I believe this is a misnomer derived from the original paper describing this relationship in fMRI as log-linear (Antonello et al.) The correct term is simply &quot;logarithmic&quot;, and for what it's worth, the authors of the original fMRI work have made this correction as well.</p>
</disp-quote>
<p>Thank you! We have made this correction.</p>
<disp-quote content-type="editor-comment">
<p>Is the data publicly available? If not, there should be some basic justification as to why (consent reasons, etc.).</p>
</disp-quote>
<p>We have recently made the data publicly available (Zada et al., 2025). We have also provided tutorials for preprocessing the data and training encoding models: <ext-link ext-link-type="uri" xlink:href="https://hassonlab.github.io/podcast-ecog-tutorials">https://hassonlab.github.io/podcast-ecog-tutorials</ext-link>. For this specific project, the analysis code is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/hassonlab/247-pickling/tree/scaling-paper-0">https://github.com/hassonlab/247-pickling/tree/scaling-paper-0</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://github.com/hassonlab/247-encoding/tree/scaling-paper-1">https://github.com/hassonlab/247-encoding/tree/scaling-paper-1</ext-link>.</p>
<disp-quote content-type="editor-comment">
<p>The authors assert that ECoG has &quot;superior spatiotemporal resolution&quot;. While this is unquestionably true for temporal resolution, the story is a bit more complicated for spatial resolution, where ECoG has far less cortical coverage than fMRI. Perhaps this sentence should be revised.</p>
</disp-quote>
<p>Thank you for pointing out the typo! We have changed it to “superior temporal resolution”.</p>
<disp-quote content-type="editor-comment">
<p>Minor Points:</p>
<p>The bolded title of Figure 3 probably shouldn't be bolded, as this is just actually the title of Figure 3A.</p>
</disp-quote>
<p>Fixed.</p>
<disp-quote content-type="editor-comment">
<p>Figure 4d is has a typo: &quot;Best Encoidng Layer&quot;.</p>
</disp-quote>
<p>Fixed.</p>
<disp-quote content-type="editor-comment">
<p><bold>Reviewer #2 (Recommendations for the authors):</bold></p>
<p>The authors could consider adding control regressors such as word rate, word frequency, phonetic features, and syntactic features like node counts, as well as control LLMs of comparable size to serve as baselines. The authors could also include correlation analyses of the embeddings from different layers of the same LLM to further illustrate how distinct the layers are within the models.</p>
</disp-quote>
<p>We have added untrained LLM embeddings as a baseline and included a comparison of encoding models between LLM contextual embeddings and classical speech features. We have also performed some preliminary correlation analyses of embeddings. In some models, we found evidence of the “two-phase abstraction process” (Cheng &amp; Antonello, 2024). However, the result is inconclusive across different LLM families. Since each LLM layer accesses and modifies the residual stream (Elhage et al., 2021), the embeddings across layers are inherently correlated. Future work could instead explore the isolated transformations within each layer to illustrate the distinct information across layers (Kumar et al., 2024).</p>
<disp-quote content-type="editor-comment">
<p>The analysis codes and data should be made available.</p>
</disp-quote>
<p>We have recently made the data publicly available (Zada et al., 2025). We have also provided tutorials for preprocessing the data and training encoding models: <ext-link ext-link-type="uri" xlink:href="https://hassonlab.github.io/podcast-ecog-tutorials">https://hassonlab.github.io/podcast-ecog-tutorials</ext-link>. For this specific project, the analysis code is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/hassonlab/247-pickling/tree/scaling-paper-0">https://github.com/hassonlab/247-pickling/tree/scaling-paper-0</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://github.com/hassonlab/247-encoding/tree/scaling-paper-1">https://github.com/hassonlab/247-encoding/tree/scaling-paper-1</ext-link>.</p>
<disp-quote content-type="editor-comment">
<p><bold>Reviewer #3 (Recommendations for the authors):</bold></p>
<p>Most of my concrete recommendations are in the public review. Below are some additional minor ones:</p>
<p>(1) Introduction: &quot;Remarkably, these models learn from much the same shared space as humans: from real-world language generated by humans.&quot;</p>
<p>I think this is an extremely strong claim due to e.g. the different nature of child-directed speech vs. written text corpora, the lack of multimodality and grounding in language models, etc. I might suggest re-wording this sentence or removing it entirely.</p>
</disp-quote>
<p>We thank the reviewer for their suggestion! We have removed the sentence from the manuscript.</p>
<disp-quote content-type="editor-comment">
<p>(2) Introduction: &quot;EleutherAI, n.d.&quot; reference for GPT-Neo</p>
<p>GPT-NeoX-20B has an associated paper, which the authors might cite instead: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2022.bigscience-1.9">https://aclanthology.org/2022.bigscience-1.9</ext-link></p>
</disp-quote>
<p>Thank you! We have added the reference for GPT-NeoX-20B (Black et al., 2022).</p>
<disp-quote content-type="editor-comment">
<p>(3) Figure 4D: Encoidng -&gt; Encoding</p>
</disp-quote>
<p>Fixed.</p>
<disp-quote content-type="editor-comment">
<p>(4) Materials and Methods, Contextual embeddings: &quot;except for GPT-Neox-20b, which assigns additional tokens to whitespace characters.&quot;</p>
<p>What do the authors mean by &quot;additional tokens to whitespace characters?&quot; The tokenizer for GPT-NeoX-20B works in much the same way as that of GPT-Neo, just with a different vocabulary set.</p>
<p>&gt;&gt;&gt; t1 = AutoTokenizer.from_pretrained(&quot;EleutherAI/gpt-neo-125M&quot;)</p>
<p>&gt;&gt;&gt; t2 = AutoTokenizer.from_pretrained(&quot;EleutherAI/gpt-neox-20b&quot;)</p>
<p>&gt;&gt;&gt; t1.convert_ids_to_tokens(t1(&quot;The quick brown fox jumps over the lazy dog.&quot;).input_ids) ['The', 'Ġquick', 'Ġbrown', 'Ġfox', 'Ġjumps', 'Ġover', 'Ġthe', 'Ġlazy', 'Ġdog', '.']</p>
<p>&gt;&gt;&gt; t2.convert_ids_to_tokens(t2(&quot;The quick brown fox jumps over the lazy dog.&quot;).input_ids)</p>
<p>['The', 'Ġquick', 'Ġbrown', 'Ġfox', 'Ġjumps', 'Ġover', 'Ġthe', 'Ġlazy', 'Ġdog', '.']</p>
<p>If the authors are referring to Ġ as the &quot;additional token to whitespace characters,&quot; then these are in all other tokenizers as well (not only that for GPT-Neo, but also those for GPT-2 and OPT).</p>
</disp-quote>
<p>We agree that “additional tokens to whitespace characters” is an oversimplification. The GPT-Neo model family, which includes the 125M, 1.3B, and 2.7B models, utilizes the same Byte Pair Encoding (BPE) tokenizer as GPT-2. This common tokenizer has a vocabulary size of 50,257 tokens, providing compatibility and seamless integration across the models.</p>
<p>The GPT-NeoX-20B model introduces a modified tokenizer to address limitations observed in the GPT-2 tokenizer (Black et al., 2022). As detailed in Section 3.2, this new tokenizer incorporates a few key improvements:</p>
<p>(1) New BPE tokenizer: A more general-purpose BPE tokenizer was trained using the Pile dataset.</p>
<p>(2) Space Delimitation: Unlike the GPT-2 tokenizer, which treats tokenization at the start of a string as a non-space-delimited token, the GPT-NeoX-20B tokenizer applies consistent space delimitation regardless. This change resolves inconsistencies related to the presence of prefix spaces in the tokenization input.</p>
<p>(3) Whitespace Handling: The tokenizer includes tokens for repeated space characters (up to 24 consecutive spaces), enhancing efficiency in tokenizing text with substantial whitespace, such as program source code or LaTeX documents.</p>
<p>These modifications result in the GPT-NeoX-20B tokenizer representing the Pile validation set with approximately 10% fewer tokens than the GPT-2 tokenizer. This efficiency gain is particularly beneficial for processing texts with extensive whitespace.</p>
<p>In our analysis, we extracted embeddings by setting `add_prefix_space = True` to all tokenizers, so space delimitation does not result in tokenizer differences. We highlight here examples of the other two tokenizer differences using the Huggingface `AutoTokenizer`:</p>
<p>&gt;&gt;&gt; t1 = AutoTokenizer.from_pretrained(&quot;EleutherAI/gpt-neo-125M&quot;)</p>
<p>&gt;&gt;&gt; t2 = AutoTokenizer.from_pretrained(&quot;EleutherAI/gpt-neox-20b&quot;)</p>
<p>&gt;&gt;&gt; t1.convert_ids_to_tokens(t1(&quot;The Downing Street.&quot;).input_ids) ['The', 'ĠDowning', 'ĠStreet']</p>
<p>&gt;&gt;&gt; t2.convert_ids_to_tokens(t2(&quot;The Downing Street.&quot;).input_ids)</p>
<p>['The', 'ĠDown', 'ing', 'ĠStreet']</p>
<p>&gt;&gt;&gt; t1.convert_ids_to_tokens(t1(&quot;Hello !&quot;).input_ids)</p>
<p>['Hello', 'Ġ', 'Ġ', 'Ġ', 'Ġ', 'Ġ', 'Ġ', 'Ġ!']</p>
<p>&gt;&gt;&gt; t2.convert_ids_to_tokens(t2(&quot;Hello !&quot;).input_ids)</p>
<p>['Hello', ' ', '!']</p>
<p>More examples showing the differences between the GPT-2 tokenizer and the GPT-NeoX-20B tokenizer can be found in Appendix F: Tokenizer Analysis (Black et al., 2022).</p>
<p>We have added the following text to our manuscript for simplicity:</p>
<p>“All models within the same model family adhere to the same tokenizer convention, except for GPT-Neox-20B, which utilizes a different tokenizer (Black et al., 2022).”</p>
<p>References</p>
<p>Ameisen, E., Lindsey, J., Pearce, A., Gurnee, W., Turner, N. L., Chen, B., Citro, C., Abrahams, D.,  Carter, S., Hosmer, B., Marcus, J., Sklar, M., Templeton, A., Bricken, T., McDougall, C.,  Cunningham, H., Henighan, T., Jermyn, A., Jones, A., … Batson, J. (2025). Circuit Tracing:  Revealing Computational Graphs in Language Models. Transformer Circuits Thread. https://transformer-circuits.pub/2025/attribution-graphs/methods.html</p>
<p>Antonello, R., &amp; Huth, A. (2024). Predictive coding or just feature discovery? An alternative account of why language models fit brain data. Neurobiology of Language (Cambridge, Mass.), 5(1), 64–79.</p>
<p>Antonello, R., Vaidya, A., &amp; Huth, A. G. (2023). Scaling laws for language encoding models in fMRI. NeurIPS 2023. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.48550/ARXIV.2305.11863">https://doi.org/10.48550/ARXIV.2305.11863</ext-link></p>
<p>Black, S., Biderman, S., Hallahan, E., Anthony, Q., Gao, L., Golding, L., He, H., Leahy, C., McDonell,  K., Phang, J., Pieler, M., Prashanth, U. S., Purohit, S., Reynolds, L., Tow, J., Wang, B., &amp; Weinbach, S. (2022). GPT-NeoX-20B: An Open-Source Autoregressive Language Model.  Proceedings of BigScience Episode #5 -- Workshop on Challenges &amp; Perspectives in Creating Large Language Models. Proceedings of BigScience Episode #5 -- Workshop on Challenges &amp; Perspectives in Creating Large Language Models, virtual+Dublin. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.18653/v1/2022.bigscience-1.9">https://doi.org/10.18653/v1/2022.bigscience-1.9</ext-link></p>
<p>Brants, T., &amp; Franz, A. (2006). Web 1T 5-gram Version 1 [Dataset]. Linguistic Data Consortium. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.35111/CQPA-A498">https://doi.org/10.35111/CQPA-A498</ext-link></p>
<p>Cantlon, J. F., &amp; Piantadosi, S. T. (2024). Uniquely human intelligence arose from expanded information capacity. Nature Reviews Psychology, 3(4), 275–293.</p>
<p>Chemla, E., D’Ascoli, S., Diego-Simón, P., King, J.-R., &amp; Lakretz, Y. (2024). A Polar coordinate system represents syntax in large language models. Advances in Neural Information Processing Systems 37, 105375–105396.</p>
<p>Cheng, E., &amp; Antonello, R. J. (2024). Evidence from fMRI supports a two-phase abstraction process in language models. In arXiv [cs.CL]. arXiv. <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2409.05771">http://arxiv.org/abs/2409.05771</ext-link></p>
<p>Csordás, R., Manning, C. D., &amp; Potts, C. (2025). Do language models use their depth efficiently?  In arXiv [cs.LG]. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.48550/ARXIV.2505.13898">https://doi.org/10.48550/ARXIV.2505.13898</ext-link></p>
<p>Dupré la Tour, T., Eickenberg, M., Nunez-Elizalde, A. O., &amp; Gallant, J. L. (2022). Feature-space selection with banded ridge regression. In bioRxiv. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1101/2022.05.05.490831">https://doi.org/10.1101/2022.05.05.490831</ext-link></p>
<p>Elhage, N., Hume, T., Olsson, C., Schiefer, N., Henighan, T., Kravec, S., Hatfield-Dodds, Z., Lasenby, R., Drain, D., Chen, C., Grosse, R., McCandlish, S., Kaplan, J., Amodei, D., Wattenberg, M., &amp; Olah, C. (2022). Toy Models of Superposition. Transformer Circuits Thread.</p>
<p>Elhage, N., Nanda, N., Olsson, C., Henighan, T., Joseph, N., Mann, B., Askell, A., Bai, Y., Chen, A.,  Conerly, T., DasSarma, N., Drain, D., Ganguli, D., Hatfield-Dodds, Z., Hernandez, D., Jones, A.,  Kernion, J., Lovitt, L., Ndousse, K., … Olah, C. (2021). A Mathematical Framework for Transformer Circuits. Transformer Circuits Thread.</p>
<p>Fan, S., Jiang, X., Li, X., Meng, X., Han, P., Shang, S., Sun, A., Wang, Y., &amp; Wang, Z. (2024). Not all Layers of LLMs are Necessary during Inference. In arXiv [cs.CL]. arXiv. <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2403.02181">http://arxiv.org/abs/2403.02181</ext-link></p>
<p>Friederici, A. D., &amp; Becker, Y. (2025). The core language network separated from other networks during primate evolution. Nature Reviews. Neuroscience, 26(2), 131–132.</p>
<p>Goldstein, A., Ham, E., Schain, M., Nastase, S. A., Aubrey, B., Zada, Z., Grinstein-Dabush, A.,  Gazula, H., Feder, A., Doyle, W., Devore, S., Dugan, P., Friedman, D., Brenner, M., Hassidim, A., Matias, Y., Devinsky, O., Siegelman, N., Flinker, A., … Hasson, U. (2025). Temporal structure of natural language processing in the human brain corresponds to layered hierarchy of large language models. Nature Communications, 16(1), 10529.</p>
<p>Goldstein, A., Wang, H., Niekerken, L., Schain, M., Zada, Z., Aubrey, B., Sheffer, T., Nastase, S. A., Gazula, H., Singh, A., Rao, A., Choe, G., Kim, C., Doyle, W., Friedman, D., Devore, S., Dugan, P., Hassidim, A., Brenner, M., … Hasson, U. (2025). A unified acoustic-to-speech-to-language embedding space captures the neural basis of natural language processing in everyday conversations. Nature Human Behaviour. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41562-025-02105-9">https://doi.org/10.1038/s41562-025-02105-9</ext-link></p>
<p>Goldstein, A., Zada, Z., Buchnik, E., Schain, M., Price, A., Aubrey, B., Nastase, S. A., Feder, A.,  Emanuel, D., Cohen, A., Jansen, A., Gazula, H., Choe, G., Rao, A., Kim, C., Casto, C., Fanda, L., Doyle, W., Friedman, D., … Hasson, U. (2022). Shared computational principles for language processing in humans and deep language models. Nature Neuroscience, 25(3), 369–380.</p>
<p>Gromov, A., Tirumala, K., Shapourian, H., Glorioso, P., &amp; Roberts, D. A. (2024). The Unreasonable Ineffectiveness of the Deeper Layers. In arXiv [cs.CL]. arXiv. <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2403.17887">http://arxiv.org/abs/2403.17887</ext-link></p>
<p>Herculano-Houzel, S. (2012). The remarkable, yet not extraordinary, human brain as a scaled-up primate brain and its associated cost. Proceedings of the National Academy of Sciences of the United States of America, 109 Suppl 1(supplement_1), 10661–10668.</p>
<p>Hewitt, J., &amp; Manning, C. D. (2019). A Structural Probe for Finding Syntax in Word Representations. In J. Burstein, C. Doran, &amp; T. Solorio (Eds.), Proceedings of the 2019 Conference of the North (pp. 4129–4138). Association for Computational Linguistics.</p>
<p>Honnibal, M., Montani, I., Van Landeghem, S., &amp; Boyd, A. (2020). spaCy: Industrial-strength Natural Language Processing in Python.</p>
<p>Kriegeskorte, N., Simmons, W. K., Bellgowan, P. S. F., &amp; Baker, C. I. (2009). Circular analysis in systems neuroscience: the dangers of double dipping. Nature Neuroscience, 12(5),  535–540.</p>
<p>Kumar, S., Sumers, T. R., Yamakoshi, T., Goldstein, A., Hasson, U., Norman, K. A., Griffiths, T. L., Hawkins, R. D., &amp; Nastase, S. A. (2024). Shared functional specialization in transformer-based language models and the human brain. Nature Communications, 15(1), 5523.</p>
<p>Linzen, T., &amp; Baroni, M. (2021). Syntactic Structure from Deep Learning. Annual Review of Linguistics, 7(1), 195–212.</p>
<p>Manning, C. D., Clark, K., Hewitt, J., Khandelwal, U., &amp; Levy, O. (2020). Emergent linguistic structure in artificial neural networks trained by self-supervision. Proceedings of the National Academy of Sciences of the United States of America, 117(48), 30046–30054.</p>
<p>Millet, J., Caucheteux, C., Orhan, P., Boubenec, Y., Gramfort, A., Dunbar, E., Pallier, C., &amp; King, J.-R. (2023). Toward a realistic model of speech processing in the brain with self-supervised learning. NeurIPS 2022. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.48550/ARXIV.2206.01685">https://doi.org/10.48550/ARXIV.2206.01685</ext-link></p>
<p>Pavlick, E. (2022). Semantic structure in deep learning. Annual Review of Linguistics, 8(1),  447–471.</p>
<p>Pennington, J., Socher, R., &amp; Manning, C. (2014). Glove: Global vectors for word representation.  Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP). Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), Doha, Qatar. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3115/v1/d14-1162">https://doi.org/10.3115/v1/d14-1162</ext-link></p>
<p>Schrimpf, M., Blank, I. A., Tuckute, G., Kauf, C., Hosseini, E. A., Kanwisher, N., Tenenbaum, J. B., &amp; Fedorenko, E. (2021). The neural architecture of language: Integrative modeling converges on predictive processing. Proceedings of the National Academy of Sciences of the United States of America, 118(45), e2105646118.</p>
<p>The CMU Pronouncing Dictionary. (n.d.). Retrieved May 27, 2025, from <ext-link ext-link-type="uri" xlink:href="http://www.speech.cs.cmu.edu/cgi-bin/cmudict">http://www.speech.cs.cmu.edu/cgi-bin/cmudict</ext-link></p>
<p>Vaidya, A. R., Jain, S., &amp; Huth, A. G. (2022). Self-supervised models of audio effectively explain human cortical responses to speech. ICML 2022. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.48550/ARXIV.2205.14252">https://doi.org/10.48550/ARXIV.2205.14252</ext-link></p>
<p>Zada, Z., Goldstein, A., Michelmann, S., Simony, E., Price, A., Hasenfratz, L., Barham, E., Zadbood,  A., Doyle, W., Friedman, D., Dugan, P., Melloni, L., Devore, S., Flinker, A., Devinsky, O., Nastase, S. A., &amp; Hasson, U. (2024). A shared model-based linguistic space for transmitting our thoughts from brain to brain in natural conversations. Neuron, S0896627324004604. Zada, Z., Nastase, S. A., Aubrey, B., Jalon, I., Michelmann, S., Wang, H., Hasenfratz, L., Doyle, W.,  Friedman, D., Dugan, P., Melloni, L., Devore, S., Flinker, A., Devinsky, O., Goldstein, A., &amp; Hasson, U. (2025). The “Podcast” ECoG dataset for modeling neural activity during natural language comprehension. Scientific Data, 12(1), 1135.</p>
</body>
</sub-article>
</article>