<?xml version="1.0" encoding="UTF-8"?>
<?xml-stylesheet type="text/xsl" href="/oai-pmh.xsl"?>
<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd">
  <responseDate>2026-09-20T01:50:37Z</responseDate>
  <request identifier="oai:www.ideals.illinois.edu:2142/124357" metadataPrefix="etdms" verb="GetRecord">https://www.ideals.illinois.edu/oai-pmh</request>
  <GetRecord>
    <record>
      <header>
        <identifier>oai:www.ideals.illinois.edu:2142/124357</identifier>
        <datestamp>2024-09-16</datestamp>
        <setSpec>col_2142_5131</setSpec>
        <setSpec>col_2142_10761</setSpec>
        <setSpec>com_2142_5130</setSpec>
        <setSpec>com_2142_10755</setSpec>
        <setSpec>com_2142_234</setSpec>
      </header>
      <metadata>
        <thesis xmlns="http://www.ndltd.org/standards/metadata/etdms/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:dc="http://purl.org/dc/elements/1.1/" xsi:schemaLocation="http://www.ndltd.org/standards/metadata/etdms/1.1/ http://www.ndltd.org/standards/metadata/etdms/1.1/etdms11.xsd http://purl.org/dc/elements/1.1/ http://www.ndltd.org/standards/metadata/etdms/1.1/etdmsdc.xsd">
          <dc:contributor>Hoiem, Derek</dc:contributor>
          <dc:date>2024-05</dc:date>
          <dc:format>application/pdf</dc:format>
          <dc:language>en</dc:language>
          <dc:type>text</dc:type>
          <dc:description>Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2024-09-16 without embargo terms</dc:description>
          <dc:description>The student, Joshua Levine, accepted the attached license on 2024-04-22 at 09:10.</dc:description>
          <dc:description>The student, Joshua Levine, submitted this Thesis for approval on 2024-04-22 at 09:26.</dc:description>
          <dc:description>This Thesis was approved for publication on 2024-04-29 at 09:06.</dc:description>
          <dc:description>DSpace SAF Submission Ingestion Package generated from Vireo submission #20529 on 2024-09-16 at 00:35:42</dc:description>
          <dc:title>Methods for generating visual programs with optimizable vision models</dc:title>
          <dc:creator>Levine, Joshua</dc:creator>
          <dc:date>2024-04-29</dc:date>
          <dc:subject>Visual Programming</dc:subject>
          <dc:subject>Visual Question Answering</dc:subject>
          <dc:subject>Large Language Models</dc:subject>
          <dc:subject>Computer Vision</dc:subject>
          <dc:subject>Program Generation</dc:subject>
          <dc:description>End-to-end vision-language models often fail to handle compositional tasks, necessitating alternative approaches for more complex problem-solving. Leveraging the visual programming paradigm, we propose a novel method for composing foundational vision models through program generation to tackle compositional tasks effectively. We investigate prompting and execution strategies that enable the synthesis of fine-tunable code by trainable large language models aimed at improving the effectiveness of the programs in solving vision-language tasks. Capitalizing on the robust compositional reasoning capabilities of large language models (LLMs), we employ pre-trained LLMs to architect programs constructed using a catalog of pre-defined atomic functions. These atomic functions, implemented with pre-trained vision models, serve as the building blocks for the visual programs generated by our system. Our methodology supports programs in various formats, always offering the flexibility to fine-tune the constituent vision models and the LLM code generator. This study concentrates on image-based question-answering. This focus underscores the critical need for advanced compositional reasoning in interpreting and responding to complex visual queries. Our evaluation encompasses the executability and correctness of the produced programs, providing a comprehensive assessment of our approach's effectiveness. This paper lays the groundwork for a subsequent investigation into the joint training of the LLMs and atomic functions, setting the stage for significant advancements in program generation and compositional reasoning in computer vision.</dc:description>
          <dc:type>Text</dc:type>
          <dc:language>eng</dc:language>
          <dc:identifier>https://hdl.handle.net/2142/124357</dc:identifier>
          <dc:rights>Copyright 2024 Joshua Levine</dc:rights>
          <degree>
            <name>M.S.</name>
            <level>Thesis</level>
            <discipline>Computer Science</discipline>
            <grantor>University of Illinois at Urbana-Champaign</grantor>
            <department>Computer Science</department>
          </degree>
        </thesis>
      </metadata>
    </record>
  </GetRecord>
</OAI-PMH>
