<?xml version="1.0" encoding="UTF-8"?>
<?xml-stylesheet type="text/xsl" href="/oai-pmh.xsl"?>
<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd">
  <responseDate>2026-09-19T21:42:18Z</responseDate>
  <request identifier="oai:www.ideals.illinois.edu:2142/98359" metadataPrefix="etdms" verb="GetRecord">https://www.ideals.illinois.edu/oai-pmh</request>
  <GetRecord>
    <record>
      <header>
        <identifier>oai:www.ideals.illinois.edu:2142/98359</identifier>
        <datestamp>2023-07-11</datestamp>
        <setSpec>col_2142_10761</setSpec>
        <setSpec>col_2142_5131</setSpec>
        <setSpec>com_2142_10755</setSpec>
        <setSpec>com_2142_234</setSpec>
        <setSpec>com_2142_5130</setSpec>
      </header>
      <metadata>
        <thesis xmlns="http://www.ndltd.org/standards/metadata/etdms/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:dc="http://purl.org/dc/elements/1.1/" xsi:schemaLocation="http://www.ndltd.org/standards/metadata/etdms/1.1/ http://www.ndltd.org/standards/metadata/etdms/1.1/etdms11.xsd http://purl.org/dc/elements/1.1/ http://www.ndltd.org/standards/metadata/etdms/1.1/etdmsdc.xsd">
          <dc:contributor>Hoiem, Derek</dc:contributor>
          <dc:contributor>Hoiem, Derek</dc:contributor>
          <dc:contributor>Lazebnik, Svetlana</dc:contributor>
          <dc:contributor>Forsyth, David</dc:contributor>
          <dc:contributor>Parikh, Devi</dc:contributor>
          <dc:creator>Shih, Kevin Jonathan</dc:creator>
          <dc:date>2017-09-29T17:56:40Z</dc:date>
          <dc:date>2017-09-29T17:56:40Z</dc:date>
          <dc:date>2017-07-11</dc:date>
          <dc:date>2017-08</dc:date>
          <dc:description>Knowing where to look in an image can significantly improve performance in computer vision tasks by eliminating irrelevant information from the rest of the input image, and by breaking down complex scenes into simpler and more familiar sub-components. We show that a framework for identifying multiple task-relevant regions can be learned in current state-of-the-art deep network architectures, resulting in significant gains in several visual prediction tasks. We will demonstrate both directly and indirectly supervised models for selecting image regions and show how they can improve performance over baselines by means of focusing on the right areas.</dc:description>
          <dc:description>Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2017-09-29 without embargo terms</dc:description>
          <dc:description>The student, Kevin Shih, accepted the attached license on 2017-07-10 at 12:45.</dc:description>
          <dc:description>The student, Kevin Shih, submitted this Dissertation for approval on 2017-07-10 at 13:18.</dc:description>
          <dc:description>This Dissertation was approved for publication on 2017-07-11 at 15:03.</dc:description>
          <dc:description>DSpace SAF Submission Ingestion Package generated from Vireo submission #11368 on 2017-09-29 at 11:29:29</dc:description>
          <dc:description>Made available in DSpace on 2017-09-29T17:56:40Z (GMT). No. of bitstreams: 2
SHIH-DISSERTATION-2017.pdf: 35992565 bytes, checksum: 0236e3afe4b94ec89729250662a7eb76 (MD5)
LICENSE.txt: 4207 bytes, checksum: 850b64b383db31c4fc6d39801a3eab05 (MD5)
  Previous issue date: 2017-07-11</dc:description>
          <dc:format>application/pdf</dc:format>
          <dc:identifier>http://hdl.handle.net/2142/98359</dc:identifier>
          <dc:language>en</dc:language>
          <dc:rights>Copyright 2017 Kevin Jonathan Shih</dc:rights>
          <dc:subject>Computer vision</dc:subject>
          <dc:subject>Visual attention</dc:subject>
          <dc:subject>Visual question answering (VQA)</dc:subject>
          <dc:subject>Keypoint localization</dc:subject>
          <dc:subject>Part localization</dc:subject>
          <dc:subject>Image recognition</dc:subject>
          <dc:subject>Fine-grained image recognition</dc:subject>
          <dc:subject>Deep learning</dc:subject>
          <dc:subject>Multi-task learning</dc:subject>
          <dc:subject>Machine learning</dc:subject>
          <dc:title>Learning visual tasks with selective attention</dc:title>
          <dc:type>text</dc:type>
          <dc:type>text</dc:type>
          <degree>
            <department>Computer Science</department>
            <discipline>Computer Science</discipline>
            <grantor>University of Illinois at Urbana-Champaign</grantor>
            <level>Dissertation</level>
            <name>Ph.D.</name>
          </degree>
        </thesis>
      </metadata>
    </record>
  </GetRecord>
</OAI-PMH>
