<?xml version="1.0" encoding="utf-8" standalone="no"?>
<?xml-stylesheet type='text/xsl' href='/oai-pmh/oai2.xsl'?>
<OAI-PMH xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns="http://www.openarchives.org/OAI/2.0/">
  <responseDate>2026-10-10</responseDate>
  <request verb="GetRecord" identifier="oai:researchdata.se:doi-10-17044-scilifelab-19375172/0" metadataPrefix="oai_dc">https://api.researchdata.se/oai-pmh</request>
  <GetRecord>
    <record>
      <header>
        <identifier>oai:researchdata.se:doi-10-17044-scilifelab-19375172/0</identifier>
        <datestamp>2023-05-31</datestamp>
        <setSpec>subject:ssif:3</setSpec>
        <setSpec>subject:ssif:10203</setSpec>
        <setSpec>subject:ssif:102</setSpec>
        <setSpec>subject:ssif:1</setSpec>
        <setSpec>principal:slug:stockholm-university</setSpec>
      </header>
      <metadata>
        <oai_dc:dc xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
          <dc:type>info:eu-repo/semantics/other</dc:type>
          <dc:type>http://purl.org/dc/dcmitype/Dataset</dc:type>
          <dc:identifier>https://doi.org/10.17044/SCILIFELAB.19375172</dc:identifier>
          <dc:title xml:lang="en">Modelling of Large Protein Complexes</dc:title>
          <dc:creator>https://orcid.org/0000-0003-3439-1866</dc:creator>
          <dc:creator>https://orcid.org/0000-0002-7115-9751</dc:creator>
          <dc:subject xml:lang="en">Medical and Health Sciences</dc:subject>
          <dc:subject xml:lang="sv">Medicin och hälsovetenskap</dc:subject>
          <dc:subject xml:lang="en">Bioinformatics (Computational Biology)</dc:subject>
          <dc:subject xml:lang="sv">Bioinformatik (beräkningsbiologi)</dc:subject>
          <dc:description xml:lang="en">AlphaFold and AlphaFold-multimer can predict the structure of single- and multiple chain proteins with very high accuracy. However, predicting protein complexes with more than a handful of chains is still unfeasible, as the accuracy rapidly decreases with the number of chains and the protein size is limited by the memory on a GPU. Nevertheless, it might be possible to predict the structure of large complexes starting from predictions of subcomponents. Here, we take a graph traversal approach to assemble 175 protein complexes with 10-30 chains using predictions of subcomponents. We compute paths through a complex graph constructed of subcomponents using Monte Carlo Tree Search and assemble these in a stepwise fashion. Using subcomponents predicted from all possible trimeric interactions, 91 complexes (52%) are assembled to completion. We create a scoring function, mpDockQ, that can distinguish if assemblies are complete and predict their accuracy. Selecting complete complexes with TM-score ≥0.9 at FPR 10% using mpDockQ results in 20 complete complexes with a median TM-score of 0.92. The complete assembly protocol, starting from the sequences, is freely available at: https://gitlab.com/patrickbryant1/molpc  

The repository here contains MSAs and predicted subcomponents to reproduce the assembly for the "all-trimer" approach.</dc:description>
          <dc:rights>https://creativecommons.org/licenses/by/4.0/</dc:rights>
          <dc:publisher xml:lang="en">Stockholm University</dc:publisher>
          <dc:publisher xml:lang="sv">Stockholms universitet</dc:publisher>
          <dc:date>2022-03-22T00:00:00Z</dc:date>
        </oai_dc:dc>
      </metadata>
    </record>
  </GetRecord>
</OAI-PMH>