<?xml version="1.0" encoding="UTF-8"?><?xml-stylesheet type="text/xsl" href="static/style.xsl"?><OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd"><responseDate>2026-09-19T00:26:52Z</responseDate><request verb="GetRecord" identifier="oai:dspace.mit.edu:1721.1/140090" metadataPrefix="dim">https://dspace.mit.edu/server/oai/request</request><GetRecord><record><header><identifier>oai:dspace.mit.edu:1721.1/140090</identifier><datestamp>2022-02-08T03:35:38Z</datestamp><setSpec>com_1721.1_7582</setSpec><setSpec>com_1721.1_7581</setSpec><setSpec>col_1721.1_131023</setSpec></header><metadata><dim:dim xmlns:dim="http://www.dspace.org/xmlns/dspace/dim" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:doc="http://www.lyncode.com/xoai" xsi:schemaLocation="http://www.dspace.org/xmlns/dspace/dim http://www.dspace.org/schema/dim.xsd">
   <dim:field mdschema="dc" element="contributor" qualifier="advisor">How, Jonathan P.</dim:field>
   <dim:field mdschema="dc" element="contributor" qualifier="author">Abdulhai, Marwa</dim:field>
   <dim:field mdschema="dc" element="contributor" qualifier="department">Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science</dim:field>
   <dim:field mdschema="dc" element="date" qualifier="accessioned">2022-02-07T15:23:31Z</dim:field>
   <dim:field mdschema="dc" element="date" qualifier="available">2022-02-07T15:23:31Z</dim:field>
   <dim:field mdschema="dc" element="date" qualifier="issued">2021-09</dim:field>
   <dim:field mdschema="dc" element="date" qualifier="submitted">2021-11-03T19:25:27.883Z</dim:field>
   <dim:field mdschema="dc" element="identifier" qualifier="uri">https://hdl.handle.net/1721.1/140090</dim:field>
   <dim:field mdschema="dc" element="description" qualifier="abstract">Hierarchical reinforcement learning has focused on discovering temporally extended actions (options) to provide efficient solutions for long-horizon decision-making problems with sparse rewards. One promising approach that learns these options end-toend in this setting is the option-critic (OC) framework. However, there are several practical limitations of this method, including the lack of diversity between the learned sub-policies and sample inefficiency. This thesis shows that the OC framework does not decompose problems into smaller and largely independent components, but instead increases the problem complexity with each option by considering the entire state space during learning. To address this issue, we introduce state abstracted option-critic (SOC), a new framework that considers both temporal and state abstraction to effectively reduce the problem complexity in sparse reward settings. Our contribution includes learning a factored state space to enable each option to map to a sub-section of the state space. We test our method against hierarchical, nonhierarchical, and state abstraction baselines to demonstrate better sample efficiency and higher overall performance in both image and large vector-state representations under sparse reward settings.</dim:field>
   <dim:field mdschema="dc" element="description" qualifier="degree">M.Eng.</dim:field>
   <dim:field mdschema="dc" element="publisher">Massachusetts Institute of Technology</dim:field>
   <dim:field mdschema="dc" element="rights">In Copyright - Educational Use Permitted</dim:field>
   <dim:field mdschema="dc" element="rights">Copyright MIT</dim:field>
   <dim:field mdschema="dc" element="rights" qualifier="uri">http://rightsstatements.org/page/InC-EDU/1.0/</dim:field>
   <dim:field mdschema="dc" element="title">Factored State Abstraction for Option Learning</dim:field>
   <dim:field mdschema="dc" element="type">Thesis</dim:field>
   <dim:field mdschema="dc" element="format" qualifier="mimetype">application/pdf</dim:field>
   <dim:field mdschema="mit" element="thesis" qualifier="degree">Master</dim:field>
   <dim:field mdschema="thesis" element="degree" qualifier="name">Master of Engineering in Electrical Engineering and Computer Science</dim:field>
   <dim:field mdschema="dspace" element="entity" qualifier="type">Publication</dim:field>
   <dim:field mdschema="others" element="access-status">unknown</dim:field>
   <dim:field mdschema="others" element="access-status">unknown</dim:field>
   <dim:field mdschema="cerif" element="openaire" authority="" confidence="-1">&lt;Publication xmlns="https://www.openaire.eu/cerif-profile/1.1/" id="80c38890-d5a9-45a8-ad1f-811434452048">
	&lt;Type xmlns="https://www.openaire.eu/cerif-profile/vocab/COAR_Publication_Types">http://purl.org/coar/resource_type/c_1843&lt;/Type>
   	&lt;Title>Factored State Abstraction for Option Learning&lt;/Title>
   	&lt;PublishedIn>
    	&lt;Publication>
      	&lt;/Publication>
   	&lt;/PublishedIn>
   	&lt;PublicationDate>2021-09&lt;/PublicationDate>
   	&lt;Authors>
      	&lt;Author>
        	&lt;DisplayName>Abdulhai, Marwa&lt;/DisplayName>
         	&lt;Affiliation>
         		&lt;OrgUnit>
         		&lt;/OrgUnit>
         	&lt;/Affiliation>
      	&lt;/Author>
	&lt;/Authors>
   	&lt;Editors>
	&lt;/Editors>
    &lt;Publishers>
        &lt;Publisher>
            &lt;DisplayName>Massachusetts Institute of Technology&lt;/DisplayName>
            &lt;OrgUnit />
        &lt;/Publisher>
    &lt;/Publishers>
    &lt;License>http://rightsstatements.org/page/InC-EDU/1.0/&lt;/License>
   	&lt;Abstract>Hierarchical reinforcement learning has focused on discovering temporally extended actions (options) to provide efficient solutions for long-horizon decision-making problems with sparse rewards. One promising approach that learns these options end-toend in this setting is the option-critic (OC) framework. However, there are several practical limitations of this method, including the lack of diversity between the learned sub-policies and sample inefficiency. This thesis shows that the OC framework does not decompose problems into smaller and largely independent components, but instead increases the problem complexity with each option by considering the entire state space during learning. To address this issue, we introduce state abstracted option-critic (SOC), a new framework that considers both temporal and state abstraction to effectively reduce the problem complexity in sparse reward settings. Our contribution includes learning a factored state space to enable each option to map to a sub-section of the state space. We test our method against hierarchical, nonhierarchical, and state abstraction baselines to demonstrate better sample efficiency and higher overall performance in both image and large vector-state representations under sparse reward settings.&lt;/Abstract>
	&lt;Access xmlns="http://purl.org/coar/access_right" 
    >
    &lt;/Access>
&lt;/Publication>
</dim:field>
</dim:dim>
</metadata></record></GetRecord></OAI-PMH>