<?xml version='1.0' encoding='utf-8'?>
<?xml-stylesheet type="text/xsl" href="/v2/static/oai2.xsl"?>
<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd">
  <responseDate>2026-09-18T17:51:32Z</responseDate>
  <request identifier="oai:figshare.com:article/31451836" metadataPrefix="oai_dc" verb="GetRecord">https://api.figshare.com/v2/oai</request>
  <GetRecord>
    <record>
      <header>
        <identifier>oai:figshare.com:article/31451836</identifier>
        <datestamp>2025-12-01T00:00:00Z</datestamp>
        <setSpec>portal_693</setSpec>
        <setSpec>item_type_8</setSpec>
        <setSpec>month_year_12_2025</setSpec>
      </header>
      <metadata>
        <oai_dc:dc xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"  xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
          <dc:title>Imitation Learning with Superhuman Policy Gradient Optimization for Sequential Cancer Treatment Decisions</dc:title>
          <dc:creator>Filippo Corna (23292115)</dc:creator>
          <dc:subject>Artificial Intelligence</dc:subject>
          <dc:description>We propose a simulator-driven imitation learning framework for sequential deci- sion making in head and neck cancer (HNC) treatment. Our method, Superhu- man Policy Gradient Optimization (SPGO), integrates inverse reinforcement learning principles with policy gradient updates to derive three-stage treat- ment policies directly from recorded physician decisions. A pre-trained clinical simulator—combining a variational autoencoder (VAE) and gradient boosting (XGBoost) models—generates complete, temporally consistent patient trajec- tories, enabling safe and reproducible training.
Unlike conventional behavior cloning, SPGO optimizes a subdominance loss that explicitly rewards surpassing the expert across multiple clinical outcomes, including relapse at year three and patient-reported toxicities at multiple follow- up times. We systematically compare six subdominance configurations (absolute vs. relative, sum vs. max aggregation, per-feature vs. max-only α updates) to assess how loss design affects convergence and treatment quality.
Our best configuration—relative differences with sum aggregation and per- feature α updates—achieves over 70% superhuman dominance across clinically relevant features on held-out patients. The learned policies reproduce expert decisions on acute measures while significantly reducing predicted late toxicities and relapse risk, demonstrating generalization beyond the training distribution.</dc:description>
          <dc:date>2025-12-01T00:00:00Z</dc:date>
          <dc:type>Text</dc:type>
          <dc:type>Thesis</dc:type>
          <dc:identifier>10.25417/uic.31451836.v1</dc:identifier>
          <dc:relation>https://figshare.com/articles/thesis/Imitation_Learning_with_Superhuman_Policy_Gradient_Optimization_for_Sequential_Cancer_Treatment_Decisions/31451836</dc:relation>
          <dc:rights>In Copyright</dc:rights>
          <dc:rights>Open Access after 2028-01-01</dc:rights>
        </oai_dc:dc>
      </metadata>
    </record>
  </GetRecord>
</OAI-PMH>
