<?xml version='1.0' encoding='utf-8'?>
<?xml-stylesheet type="text/xsl" href="/v2/static/oai2.xsl"?>
<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd">
  <responseDate>2026-10-07T10:23:06Z</responseDate>
  <request identifier="oai:figshare.com:article/33963382" metadataPrefix="oai_dc" verb="GetRecord">https://api.figshare.com/v2/oai</request>
  <GetRecord>
    <record>
      <header>
        <identifier>oai:figshare.com:article/33963382</identifier>
        <datestamp>2026-10-01T14:54:22Z</datestamp>
        <setSpec>category_28864</setSpec>
        <setSpec>item_type_12</setSpec>
        <setSpec>month_year_10_2026</setSpec>
      </header>
      <metadata>
        <oai_dc:dc xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"  xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
          <dc:title>Behavioral Success Without Internalization</dc:title>
          <dc:creator>Yibo Chen (24724807)</dc:creator>
          <dc:subject>Artificial intelligence not elsewhere classified</dc:subject>
          <dc:subject>Large Language Model</dc:subject>
          <dc:subject>Cognitive Intelligence</dc:subject>
          <dc:description>&lt;p dir="ltr"&gt;This preprint studies when a language model’s own scientific judgement becomes behaviorally binding on its later decisions. Using controlled scientific-result triage tasks, we separate whether models notice expectation-violating findings, judge their scientific consequence, assess verification needs, and choose subsequent research actions. We find that additional reasoning can substantially improve consequence judgements without comparably changing action, and that internally represented consequence states can causally control the model’s expressed judgement while having little effect on downstream policy. Training further reveals a distinction between behavioral improvement and internalization: ordinary supervised fine-tuning can substantially improve both judgement and action accuracy without creating a strong causal judgement-to-policy dependency, while counterfactual training can teach models to condition on explicit judgement representations without making their own endogenous judgement behaviorally binding. The work motivates a stronger operational notion of internalization based on causal influence over downstream policy.&lt;/p&gt;</dc:description>
          <dc:date>2026-10-01T14:54:22Z</dc:date>
          <dc:type>Text</dc:type>
          <dc:type>Preprint</dc:type>
          <dc:identifier>10.6084/m9.figshare.33963382.v3</dc:identifier>
          <dc:relation>https://figshare.com/articles/preprint/Behavioral_Success_Without_Internalization/33963382</dc:relation>
          <dc:rights>CC BY 4.0</dc:rights>
        </oai_dc:dc>
      </metadata>
    </record>
  </GetRecord>
</OAI-PMH>
