-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathfull_paper_test_2026-05-05_22-02-51.jsonl
More file actions
6 lines (6 loc) · 196 KB
/
Copy pathfull_paper_test_2026-05-05_22-02-51.jsonl
File metadata and controls
6 lines (6 loc) · 196 KB
1
2
3
4
5
6
{"pmid": "36851914", "meta": {"content": {"abstract": "One of the key features of intrinsically disordered regions (IDRs) is their ability to interact with a broad range of partner molecules. Multiple types of interacting IDRs were identified including molecular recognition fragments (MoRFs), short linear sequence motifs (SLiMs), and protein-, nucleic acids- and lipid-binding regions. Prediction of binding IDRs in protein sequences is gaining momentum in recent years. We survey 38 predictors of binding IDRs that target interactions with a diverse set of partners, such as peptides, proteins, RNA, DNA and lipids. We offer a historical perspective and highlight key events that fueled efforts to develop these methods. These tools rely on a diverse range of predictive architectures that include scoring functions, regular expressions, traditional and deep machine learning and meta-models. Recent efforts focus on the development of deep neural network-based architectures and extending coverage to RNA, DNA and lipid-binding IDRs. We analyze availability of these methods and show that providing implementations and webservers results in much higher rates of citations/use. We also make several recommendations to take advantage of modern deep network architectures, develop tools that bundle predictions of multiple and different types of binding IDRs, and work on algorithms that model structures of the resulting complexes.", "keywords": ["CAID, Critical Assessment of Intrinsic Disorder", "CASP, Critical Assessment of techniques for protein Structure Prediction", "DL, deep learning", "Disordered binding regions", "IDP, intrinsically disordered protein", "IDR, intrinsically disordered region", "Intrinsic disorder", "ML, machine learning", "MoRF, molecular recognition fragment", "Molecular recognition features", "NN, neural network", "Protein-lipid interactions", "Protein-nucleic acids interactions", "Protein-protein interactions", "SLiM, short linear sequence motif", "Short linear motifs"], "mesh_terms": [], "pub_types": ["Journal Article", "Review"]}, "contributors": {"medline": {"affiliations": ["Department of Computer Science, Virginia Commonwealth University, USA.", "Department of Biological Sciences, Purdue University, West Lafayette, IN 47907, USA.", "Department of Computer Science, Purdue University, West Lafayette, IN 47907, USA.", "Department of Computer Science, Virginia Commonwealth University, USA."], "auids": [], "full_names": ["Basu, Sushmita", "Kihara, Daisuke", "Kurgan, Lukasz"], "short_names": ["Basu S", "Kihara D", "Kurgan L"]}, "xml": [{"affiliations": ["Department of Computer Science, Virginia Commonwealth University, USA."], "full_name": "Basu, Sushmita", "identifiers": [], "short_name": "Basu S"}, {"affiliations": ["Department of Biological Sciences, Purdue University, West Lafayette, IN 47907, USA.", "Department of Computer Science, Purdue University, West Lafayette, IN 47907, USA."], "full_name": "Kihara, Daisuke", "identifiers": [], "short_name": "Kihara D"}, {"affiliations": ["Department of Computer Science, Virginia Commonwealth University, USA."], "full_name": "Kurgan, Lukasz", "identifiers": [], "short_name": "Kurgan L"}]}, "identity": {"doi": "10.1016/j.csbj.2023.02.018", "pmid": "36851914", "title": "Computational prediction of disordered binding regions."}, "links": {"cites": ["41689628", "41023685", "40756902", "40728621", "40728620", "40728619", "40427554", "40325419", "39833102", "39819844", "39811792", "39576584", "39576583", "39576581", "39243101", "39202449", "38213902", "38146538", "38067593", "37933852", "37892124", "37740110", "37140058", "36979465"], "entrez": {}, "external": [{"attribute": "free resource", "category": "Full Text Sources", "linkname": "", "provider": "Elsevier Science", "url": "https://linkinghub.elsevier.com/retrieve/pii/S2001-0370(23)00064-8"}, {"attribute": "free resource", "category": "Full Text Sources", "linkname": "", "provider": "PubMed Central", "url": "https://pmc.ncbi.nlm.nih.gov/articles/pmid/36851914/"}, {"attribute": "free resource", "category": "Full Text Sources", "linkname": "", "provider": "Europe PMC", "url": "https://europepmc.org/abstract/MED/36851914"}, {"attribute": "free resource", "category": "Research Materials", "linkname": "", "provider": "NCI CPTC Antibody Characterization Program", "url": "https://antibodies.cancer.gov/detail/CPTC-FHL1-1"}], "pmc": ["9957716"], "refs": ["36212542", "35883444", "35832624", "35609776", "35489069", "35367597", "35356546", "34905768", "34894985", "34850135", "34830151", "34513334", "34487791", "34487138", "34453465", "34391457", "34290238", "34265844", "34029314", "33875888", "33875885", "33537726", "33267349", "33237329", "33211854", "33119734", "32961749", "32920048", "32696352", "32621228", "32517331", "32173600", "32089835", "32019410", "31889261", "31870849", "31713636", "31680160", "31660849", "31616935", "31521235", "31504193", "31486849", "31415569", "31329574", "31207240", "31007871", "30979084", "30941889", "30866736", "30865258", "30657889", "30550782", "30324701", "30298407", "30099775", "29860432", "29806170", "29626537", "29360926", "29257115", "29136219", "29042212", "29028931", "28818512", "28776938", "28701416", "28601983", "28589442", "28516009", "28516007", "28453683", "28430951", "28394890", "28387819", "28369666", "28367366", "28334258", "28232901", "28155710", "27911701", "27787828", "27787820", "27587688", "27473064", "27174932", "27037624", "26651072", "26517836", "26426014", "26297830", "26287166", "26109352", "26073260", "25792551", "25637562", "25531225", "25391399", "25361972", "25246432", "25038412", "24939692", "24926813", "24727949", "24532081", "24093637", "24019881", "23946100", "23942625", "23534882", "23203878", "23170892", "22977176", "22702725", "22689782", "22661649", "22624656", "22543956", "22190692", "22079048", "22067451", "21928402", "21909575", "21874205", "21874190", "21646342", "21622654", "21152297", "20823312", "20509167", "20007729", "19774619", "19717576", "19597536", "19594871", "19412530", "19260013", "18991772", "18573080", "18426805", "17973494", "17912346", "17488107", "17145717", "16935303", "16889422", "16187360", "16156658", "16094605", "15955783", "15955779", "15943979", "15769473", "15310560", "15044227", "15019783", "14579348", "14579346", "12824398", "11381529", "11093259", "10869041", "10592235", "10550212", "10493868", "9254694", "7952898"], "review": ["36851914", "31521235", "31007871", "39789785", "35356546", "35098694", "31441158", "39811792", "39093570", "36959451", "35139468", "35350759", "34390736", "36833360", "28589442", "34418423", "33248138", "22280012", "26387106", "33243935", "31709918"], "similar": ["36851914", "31521235", "36458437", "31007871", "34905768", "40944448", "27881680", "39763873", "41534519", "39789785", "38166858", "36908253", "35356546", "35883444", "40031750", "35098694", "38701796", "37709009", "32696395", "39576583", "31441158", "36210722", "39811792", "31504193", "39032884", "39614773", "39576585", "39093570", "36959451", "35139468", "30324701", "32702119", "35350759", "25424537", "39720898", "38895487", "30866736", "39446390", "39253485", "31660849", "22689782", "40631955", "34894985", "32824743", "34390736", "40403066", "36272675", "36833360", "31207240", "28589442", "35767567", "34179096", "34418423", "30878482", "24727949", "31889261", "33248138", "26149687", "40441416", "24115198", "37883249", "22280012", "24678734", "31797595", "21152297", "26387106", "18831774", "36929465", "16935303", "25391399", "40254833", "32696352", "41378882", "40244295", "33866372", "41667449", "36291695", "29042212", "33087759", "32743159", "33243935", "31908009", "37674132", "35404090", "31709918", "38987470", "40756902", "20522251", "40867524", "27787828", "36008992", "41276815", "40054813", "21647374", "39571839", "23233352", "40650026", "39933697", "35310482", "32112084"], "text_mined": [{"category": "General", "source": "full_text", "url": "http://biomine.cs.vcu.edu/servers/MoRFpred/316"}, {"category": "General", "source": "full_text", "url": "https://gsponerlab.msl.ubc.ca/software/morf_chibi/75"}, {"category": "General", "source": "full_text", "url": "http://bioinf.cs.ucl.ac.uk/disopred638"}, {"category": "General", "source": "full_text", "url": "https://gsponerlab.msl.ubc.ca/software/morf_chibi/35"}, {"category": "General", "source": "full_text", "url": "http://biomine.cs.vcu.edu/servers/fMoRFpred/112"}, {"category": "General", "source": "full_text", "url": "https://gsponerlab.msl.ubc.ca/software/morf_chibi/93"}, {"category": "GitHub", "source": "full_text", "url": "https://github.com/roneshsharma/Predict-MoRFs23"}, {"category": "GitHub", "source": "full_text", "url": "https://github.com/roneshsharma/MoRFpred-plus/wiki/MoRFpred-plus41"}, {"category": "General", "source": "full_text", "url": "http://www.alok-ai-lab.com/tools/opal/50"}, {"category": "General", "source": "full_text", "url": "http://www.alok-ai-lab.com/tools/opal_plus/27"}, {"category": "GitHub", "source": "full_text", "url": "https://github.com/HHJHgithub/MoRFs_MPM6"}, {"category": "General", "source": "full_text", "url": "http://sparks-lab.org/jack/server/SPOT-MoRF/index.php24"}, {"category": "General", "source": "full_text", "url": "http://anchor.enzim.hu566"}, {"category": "General", "source": "full_text", "url": "http://iupred2a.elte.hu754"}, {"category": "General", "source": "full_text", "url": "https://idrbind.msl.ubc.ca/1"}, {"category": "General", "source": "full_text", "url": "http://bioware.ucd.ie/∼compass/biowareweb/Server_pages/api.php173"}, {"category": "General", "source": "full_text", "url": "http://www.slimsuite.unsw.edu.au/servers/slimprob.php10"}, {"category": "General", "source": "full_text", "url": "http://bioware.ucd.ie/∼compass/biowareweb/77"}, {"category": "General", "source": "full_text", "url": "http://bioware.ucd.ie/∼compass/biowareweb/74"}, {"category": "General", "source": "full_text", "url": "http://bioware.ucd.ie/∼compass/biowareweb/84"}, {"category": "General", "source": "full_text", "url": "http://rest.slimsuite.unsw.edu.au/qslimfinder17"}, {"category": "General", "source": "full_text", "url": "http://slim.icr.ac.uk/slimsearch/68"}, {"category": "General", "source": "full_text", "url": "http://biomine.cs.vcu.edu/servers/DisoRDPbind/118"}, {"category": "General", "source": "full_text", "url": "http://biomine.cs.vcu.edu/servers/flDPnn/37"}, {"category": "General", "source": "full_text", "url": "https://www.csuligroup.com/DeepDISOBind/6"}, {"category": "General", "source": "full_text", "url": "http://biomine.cs.vcu.edu/servers/DisoLipPred/6"}, {"category": "General", "source": "full_text", "url": "http://memdis.ttk.hu/"}, {"category": "GitHub", "source": "full_text", "url": "https://github.com/brgenzim/MemDis1"}, {"category": "General", "source": "full_text", "url": "https://idpcentral.org/caid"}]}, "metadata": {"entrez_date": "2023/03/01 06:00", "fetched_at": "2026-05-02 18:19:10"}, "source": {"journal_abbrev": ["Comput Struct Biotechnol J"], "journal_title": ["Computational and structural biotechnology journal"], "pub_date": "2023", "pub_types": ["Journal Article", "Review"], "pub_year": "2023"}}, "content": {"title": "Computational prediction of disordered binding regions", "body": [{"title": "Abstract", "content": ["One of the key features of intrinsically disordered regions (IDRs) is their ability to interact with a broad range of partner molecules. Multiple types of interacting IDRs were identified including molecular recognition fragments (MoRFs), short linear sequence motifs (SLiMs), and protein-, nucleic acids- and lipid-binding regions. Prediction of binding IDRs in protein sequences is gaining momentum in recent years. We survey 38 predictors of binding IDRs that target interactions with a diverse set of partners, such as peptides, proteins, RNA, DNA and lipids. We offer a historical perspective and highlight key events that fueled efforts to develop these methods. These tools rely on a diverse range of predictive architectures that include scoring functions, regular expressions, traditional and deep machine learning and meta-models. Recent efforts focus on the development of deep neural network-based architectures and extending coverage to RNA, DNA and lipid-binding IDRs. We analyze availability of these methods and show that providing implementations and webservers results in much higher rates of citations/use. We also make several recommendations to take advantage of modern deep network architectures, develop tools that bundle predictions of multiple and different types of binding IDRs, and work on algorithms that model structures of the resulting complexes."], "subsections": []}, {"title": "Introduction", "content": ["Intrinsically disordered regions (IDRs) are segments in a protein sequence that lack stable structure under physiological conditions 1, 2, 3, 4. Intrinsically disordered proteins (IDPs) include one or more IDRs, and they could be fully disordered when an IDR covers the entire chain. IDPs are found across all domains of life, with a larger abundance in eukaryotic proteomes 5, 6, 7, 8. They play important roles in a plethora of cellular activities, complementing functions of the structured proteins and domains 9, 10, 11. Examples include cellular signaling and its regulation, translation, transcription, and phase separation 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22. Being involved in key regulatory pathways, mis-regulation of IDPs and IDRs was shown to be associated with several human diseases 23, 24, 25, 26. Many of functions of IDPs involve interactions with a broad spectrum of partner molecules, including proteins, nucleic acids, lipids, metals, ions, carbohydrates and small-molecules 19, 21, 27, 28, 29, 30, 31. In that context, conformational plasticity of IDRs provides them with certain advantages compared to structured regions, such as ability of a single IDR to interact with multiple different partners, leading to an enrichment of IDPs among the hub proteins in the protein interaction networks 32, 33, 34, 35. Multiple types of interacting IDRs were categorized and characterized in the literature. Two of these types concern relative short sequences regions, molecular recognition fragments (MoRFs) and short linear sequence motifs (SLiMs). MoRFs are short IDRs that undergo disorder-to-order transition when interacting with proteins and peptides, i.e., they “morph” from disorder to order upon binding 36, 37, 38. Their length range varies across studies, with some works limiting their length to between 10 and 70 residues 37, 38, and other studies considering much shorter, 5–25 residues long, regions 36, 39. MoRFs are subdivided into multiple classes including α-MoRFs, β-MoRFs, γ-MoRFs and complex-MoRFs, based on the type of the secondary structure that they fold into upon binding, i.e., α -helix, β-sheet, irregular structures, and mixed secondary structures, respectively. SLiMs are relatively short sequence motifs represented by regular expressions that are found across multiple proteins 40, 41, 42. Majority of SLiMs are between 3 and 15 residues in length and many of them are disordered. They are associated with a variety of molecular interactions, primarily being involved in interactions with proteins and nucleic-acids [43]. Recent update of the ELM resource, a repository of eukaryotic linear motifs, reports over 3500 SLiMs that were curated from literature [40]. Moreover, human proteome was predicted to contain over 1 million binding motifs [44]. Another type of binding IDR called protean segments is defined by the IDEAL database [45]. These are short segments that are disordered in an unbound form and undergo folding upon binding with a partner molecule. The protean segments overlap with MoRFs and SLiMs but they are not limited in length like MoRFs, and do not have to be defined by regular expressions like SLiMs. The above three classes of binding IDRs are defined by their sequence features (length and motifs), modes of interactions with the partner molecule (coupled binding and folding), and binding to specific types of partners (proteins, peptides and nucleic acids). However, some interacting IDRs can be long, may not involve motifs, and may bind a variety of other molecules 30, 46. For instance, IDRs longer than 30 residues that bind proteins and peptides were classified as protein-binding IDRs [47].", "While a huge number of binding IDRs occur in nature, only a relative handful of them has been annotated by biochemical experiments. More specifically, a few hundred IDRs with binding information are available in the DisProt database, the largest repository of functionally annotated IDRs [30]. Computational methods can help with closing this annotation gap. The limited collection of annotated binding IDRs can be used to develop and evaluate computational predictors, which then can be utilized to predict these regions for the millions of protein sequences that remain unannotated. This approach relies on the fact that the disordered nature of IDRs is intrinsic (i.e., encoded) in their underlying sequences 4, 48, 49, 50, making them predictable from the sequence. This has motivated development of numerous methods that accurately predict IDRs from the protein sequence 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, with over 100 methods that were developed to date [62]. Recent research has shifted from building disorder predictors to developing methods that predict binding IDRs. Similar to IDRs, recent study shows that binding IDRs also have compositional bias in their sequences [48], suggesting that they can be predicted directly from the sequence. Significance of these predictors is reflected by the inclusion of the assessment of the binding IDRs predictions in the recently completed community-organized Critical Assessment of Intrinsic disorder (CAID) experiment [63]. The CAID experiment evaluated 11 predictors of disordered binding regions; we discuss further details later.", "Predictors of intrinsic disorder have been comprehensively surveyed and analyzed in a large number of studies 4, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77. They were evaluated in a several comparative assessments, most notably as part of the community-driven efforts including the Critical Assessment of techniques for protein Structure Prediction (CASP) experiments 78, 79, 80, 81 and more recently the CAID experiment [63]. Disorder prediction was part of CASP between CASP5 in 2002 that evaluated six methods [81] and CASP10 in 2012 which assessed 28 predictors [78], compared to CAID that was performed in 2018 and compared 32 disorder predictors. In contrast, only a few reviews focus on prediction of binding IDRs while over three dozen of these methods were developed. A survey that covered 12 predictors of binding IDRs that were discussed together with over 30 disorder predictors was published in 2017 [82]. Two articles were published in 2019 83, 84. The first overviews 20 predictors of binding IDRs that target MoRFs, SLiMs, and other protein-binding IDRs, while omitting methods that target other types of interactions [83]. The second is a book chapter that describes 22 predictors of MoRFs, SLiMs, protein and nucleic acid binding regions, largely overlapping in scope with the other study [84]. We note that prediction of disordered binding region is gaining momentum in recent years, with 13 methods published since 2019. These factors motivate this systematic survey of predictors of binding IDRs. We provide a historical perspective, comprehensively enumerate current tools, categorize them based on architectures and their predictive targets, details predictive architectures for several tools that secured best results in the CAID experiment, highlight a few interesting observations concerning availability and impact of these tools, and offer several recommendations. Moreover, we fill the gap created after the surveys from 2019 and cover 38 methods that target a diverse set of ligands including peptides, proteins, RNA, DNA and lipids."], "subsections": []}, {"title": "Historical overview", "content": ["We perform an exhaustive literature search to identify a comprehensive collection of predictors of binding IDRs. We consider three main sources: i) extraction of methods that are covered in articles that focus on the prediction of binding IDRs and disorder functions 82, 83, 84, 124; ii) manual search of citations to these methods; and iii) manual search of the results produced by a relevant and broad PubMed search: (((Intrinsically disordered proteins) AND ((binding region) OR (binding residue)) AND (identification)), (((Intrinsic disorder) AND (binding region) AND (predictor)) OR ((MoRF) AND (prediction)), (((Intrinsic disorder) AND (short linear motif) AND (prediction)), ((Intrinsically disordered proteins) AND ((RNA binding) OR (DNA binding) OR (nucleotide binding))) AND ((binding region) OR (binding residue)) AND ((prediction) OR (identification)), ((Intrinsically disordered proteins) AND (lipid binding)) AND ((binding region) OR (binding residue)) AND ((prediction) OR (identification)). We combine results from these sources and remove duplicates, which results in a list of 38 methods that were published between 2005 to June 2022 (Table 1) 36, 39, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116, 117, 118, 119, 120, 121, 122, 123. We first provide a historical overview of this area of research, which we follow by a discussion of several key aspects of these computational tools including their predictive models, popularity and availability.Table 1Predictors of binding IDRs grouped by types: MoRFs, protein-binding IDRs, SLiMs, protein/DNA/RNA-binding IDRs and lipid-binding IDRs. Methods are sorted chronologically in each type. 'Predictive architecture' column covers scoring functions (SF), regular expressions (Regex), hidden markov model (HMM)shallow machine learning (ML) and deep machine learning (DL). Specific ML and DL algorithms include neural network (NN), support vector machine (SVM), XGBoost, naive Bayes (NB), linear regression (LR), deep NN (dNN), long-short term memory network (LSTM) and multi-layer perceptron NN (mlpNN). 'Availability' covers webserver (WS), source code (SC), both (WS+SC) and never implemented (NA). 'URL' gives pages where a given method was available as of July 2022. 'Citations' includes total citations with annual citations inside brackets; these data were collected from Google Scholar in July 2022. For methods published in multiple articles, we use the reference with the highest citation count to avoid duplicate counting.Table 1Target of predictionMethod name (year published)ReferencePredictive architectureAvailabilityURLCitationstotal (per year)MoRFsα-MoRFpred (2005)[39]ML (NN)NANA647 (39)α-MoRFpred II (2007)[85]ML(NN)NANA317 (22)retro-MoRFs (2010)[86]SF (alignment)NANA43 (4)MoRFpred (2012)87, 88ML (SVM)WShttp://biomine.cs.vcu.edu/servers/MoRFpred/316 (32)MFSPSSMpred (2013)[89]ML (SVM)WS+SCThe website does not work as of July 202255 (7)MoRFCHiBi (2015)[90]ML (SVM)WS+SChttps://gsponerlab.msl.ubc.ca/software/morf_chibi/75 (11)DISOPRED3 (2015)[91]ML (SVM)WS+SChttp://bioinf.cs.ucl.ac.uk/disopred638 (92)MoRFCHiBi_Web (2015)[92]ML (NB)WS+SChttps://gsponerlab.msl.ubc.ca/software/morf_chibi/35 (5)fMoRFpred (2016)[36]ML (SVM)WShttp://biomine.cs.vcu.edu/servers/fMoRFpred/112 (19)MoRFCHiBi SYSTEM (2016)[93]ML (NB)WS+SChttps://gsponerlab.msl.ubc.ca/software/morf_chibi/93 (16)Predict-MoRFs (2016)[94]ML (SVM)SChttps://github.com/roneshsharma/Predict-MoRFs23 (4)Fang et al. (2018)[95]ML (SVM)NANA6 (2)MoRFPred-plus (2018)[96]ML (SVM)SChttps://github.com/roneshsharma/MoRFpred-plus/wiki/MoRFpred-plus41 (11)OPAL (2018)[97]ML (SVM)WS+SChttp://www.alok-ai-lab.com/tools/opal/50 (13)OPAL+ (2019)[98]ML (SVM)WS+SChttp://www.alok-ai-lab.com/tools/opal_plus/27 (9)en_DCNNMoRF (2019)[99]DL (dNN)WSThe website does not work as of July 20229 (3)MoRFMPM (2019)[100]ML (MPM)SChttps://github.com/HHJHgithub/MoRFs_MPM6 (2)MoRFPred_en (2019)[101]DL (dNN +dNN+SVM)WSThe website does not work as of July 20225 (2)MoRFMLP (2019)[102]DL (mlpNN + NB)NANA6 (2)SPOT-MoRF (2020)[103]DL (dNN)WS+SChttp://sparks-lab.org/jack/server/SPOT-MoRF/index.php24 (12)MoRFCNN (2021)[104]DL (dNN)NANA0 (0)Protein-bindingANCHOR (2009)105, 106SFWS+SChttp://anchor.enzim.hu566 (44)ANCHOR2 (2018)[107]SFWS+SChttp://iupred2a.elte.hu754 (189)IDRBind (2019)[108]ML (XGBoost)WShttps://idrbind.msl.ubc.ca/1 (1)SLiMsSLiMFinder (2007)[109]RegexWShttp://bioware.ucd.ie/∼compass/biowareweb/Server_pages/api.php173 (12)SLiMProb (SLiMSearch 1.0) (2010)[110]RegexWShttp://www.slimsuite.unsw.edu.au/servers/slimprob.php10 (1)SLiMSearch 2.0 (2011)[111]RegexWShttp://bioware.ucd.ie/∼compass/biowareweb/77 (7)SLiMPred (2012)[112]ML (NN)WShttp://bioware.ucd.ie/∼compass/biowareweb/74 (8)SLiMPrints (2012)[113]SF (alignment)WShttp://bioware.ucd.ie/∼compass/biowareweb/84 (9)PepBindPred (2013)[114]ML (NN)WSThe website does not work as of July 202231 (4)QSLiMFinder (2015)[115]RegexWShttp://rest.slimsuite.unsw.edu.au/qslimfinder17 (3)Song et al. (2015)[116]discriminative HMMNANA2 (1)SLiMSearch4.0 (2017)[117]RegexWShttp://slim.icr.ac.uk/slimsearch/68 (14)Protein/DNA/RNA-bindingDisoRDPbind (2015)118, 119ML (LR)WShttp://biomine.cs.vcu.edu/servers/DisoRDPbind/118 (17)flDPnn (2021)[120]DL (dNN)WS+SChttp://biomine.cs.vcu.edu/servers/flDPnn/37 (37)DeepDISObind (2022)[121]DL (dNN)WShttps://www.csuligroup.com/DeepDISOBind/6 (6)Lipid-bindingDisoLipPred (2021)[122]DL (dNN)WShttp://biomine.cs.vcu.edu/servers/DisoLipPred/6 (6)MemDis (2021)[123]DL (dNN)WS+SChttp://memdis.ttk.hu/ https://github.com/brgenzim/MemDis1 (1)"], "subsections": [{"title": "Historical progress in coverage of different types of interacting IDRs", "content": ["We summarize historical overview in Fig. 1. The initial focus was primarily confined to the prediction of MoRFs and SLiMs, with 10 out of the 11 methods that were published before 2015 targeting these two types of binding IDRs. The very first method is α-MoRFpred that was developed by Keith Dunker’s group in 2005 [39]. It predicts α-helix-forming MoRFs by relying on the PONDR VL-XT-generated disorder predictions [58]. The main challenge at this point was lack of annotated MoRF regions, which had to be manually compiled from the data available in Protein Data Bank (PDB) 125, 126. The α-MoRFpred was developed using a small dataset of 14 MoRFs from 12 proteins, which were unlikely to represent a broader population of MoRF regions. An improved version of this algorithm, α-MoRFpred-II, was published two years later [85]. This predictor utilized a larger training dataset (102 MoRF regions from 99 proteins) and a machine learning algorithm, a shallow feed-forward neural network. However, implementation of the resulting predictor was not released, limiting its potential applications. The year 2012 marks the release of MoRFpred 87, 88, the first predictor that tackles prediction of generic MoRFs, irrespective of their type (as compared to α-MoRFs). This method has a more advanced design compared to the earlier tools. It uses a comprehensive sequence-derived input, which includes evolutionary profile and putative disorder, solvent accessibility and B-factors, that is processed by a support vector machine model. The model was trained on a large dataset of over 400 proteins with MoRF regions and the resulting predictor was released as a publicly accessible webserver, which is operational to this date.Fig. 1Timeline of the development of predictors of binding IDRs. Color-coded bars denote different prediction targets including MoRFs (blue), protein-binding regions (red), SLiMs (yellow), protein/DNA/RNA-binding regions (grey) and lipid-binding (orange) regions. Dark green callouts show major events that drive the development of these predictors. Light green callouts identify the first predictor for each ligand type.Fig. 1", "Tools that extract/predict SLiMs were being developed in parallel to the efforts that target prediction of MoRFs. SLiMFinder, the first predictor of SLiMs, was published by Denis Shields’s lab in 2007 [109]. This method utilizes the SLiMBuild algorithm that constructs motifs, ranks them by their probability, and estimates their statistical significance. SLiMFinder offers options to restricts motif finding to specific regions of the protein sequence, such as IDRs that it predicts with the IUPred method [59], and is available in the form of a convenient webserver. Several other methods that produce SLiMs were developed subsequently, with majority of them including SLiMSearch 1.0 [110], SLiMSearch 2.0 [111], SLiMPred [112], SLiMPrints [113], PepBindPred [114] and SLiMSearch 4.0 [117] developed by the labs of Denis Shields and Norman Davey.", "With growing interest in prediction of binding IDRs, the focus has gradually shifted towards prediction of IDRs that interact with specific ligands, such as proteins, RNA, DNA, and lipids. ANCHOR, which was published by Zsuzsanna Dosztanyi’s lab in 2009, is the first method that predicts protein-binding IDRs 105, 106. ANCHOR is based on a scoring function that was derived by comparing disordered binding residues between their bound and unbound states. The prediction process is very fast and this method is available as a source code and a webserver. These factors undoubtedly contribute to high levels of popularity of this tool. DisoRDPbind, which was released by Lukasz Kurgan’s lab in 2015, is the first method that predicts nucleic acid binding IDRs 118, 119, 127. This tool relies on three relatively simple logistic regressions that are used to predict protein-binding, RNA-binding, and DNA-binding IDRs. The only other tools that target prediction of the nucleic acid-binding IDRs are flDPnn and DeepDISOBind that were released very recently 120, 121. They improve over the DisoRDPbind’s model by utilizing more sophisticated deep neural networks. The newest addition to the toolbox of predictors of binding IDRs are the two tools that predict lipid-binding IDRs, DisoLipPred [122] and MemDis [123], which were released in 2021. Interestingly, they complement each other since MemDis focuses on IDRs in trans-membrane proteins while DisoLipPred predicts lipid-binding IDRs that specifically exclude trans-membrane regions. Lastly, we note that there are no predictors for the protean regions."], "subsections": []}, {"title": "Major events", "content": ["The timeline in Fig. 1 can be divided into two distinct periods, a first-generation period before 2015 and a second-generation period that started in 2015. The first-generation period is characterized by a relatively slower pace of the development efforts, with on average 1.1 new methods published per year, and focus on a small subset of the binding IDR types, such as MoRFs and SLiMs. The efforts intensified in the second-generation period, with on average 3.4 methods published per year and a broader coverage of binding IDR types, which include MoRFs, SLiMs, protein-binding, nucleic acid-binding, and lipid-binding IDRs. This increase results from an improved availability of ground-truth annotations of binding IDRs. The early methods, such as α-MoRFpred, α-MoRFpred-II, MoRFpred, and MoRFCHiBi, primarily relied on parsing data from PDB, which is rather difficult since it requires processing atomic-level data, aggregation at residue level and comparing across multiple structures given that PDB files are redundant and often cover fragments of protein sequences. Moreover, these data are also limited since PDB centers on providing access to structured proteins and regions. The first database of disordered proteins, DisProt, was established in 2005 128, 129. It started with a few hundred IDPs that were annotated based on published experimental data. It took several years before the annotations of binding were added and a sufficiently large number of these annotations was collected. By early 2010 s the amount of the accumulated binding IDRs was sufficient to develop and test predictive tools, and the second-generation tools, such as DisoRDPbind, flDPnn, DeepDISObind, DisoLipPred, and MemDis, rely on DisProt to source training and test datasets. These annotations are easier to collect compared to PDB data since they are reported at the residue level and mapped into full protein sequences. Moreover, they are more diverse, allowing to collect data to develop methods for more types of binding IDRs.", "Besides the development and growth of DisProt, the other significant event that stimulates efforts to develop predictors of binding IDRs is the CAID experiment, which was held in 2018 and included evaluation of the these predictors [63]. CAID is the first community-driven evaluation of accuracy of predictions of binding IDRs, which suggests growing interest in this area. Several best-performing methods secured area under the ROC curve (AUC) values> 0.7, including ANCHOR2 [107] with AUC = 0.742, DisoRDPbind’s model for the protein-binding IDRs [118] with AUC = 0.729, MoRFCHiBi_Light\n[93] with AUC = 0.720, and MoRFCHiBi_Web\n[92] with AUC = 0.702. Overall, among the 11 methods which participated in the CAID’s binding IDR prediction assessment, five perform above a baseline level: ANCHOR2, DisoRDPbind, the two versions of MoRFCHiBi, and OPAL [97]. We refrain from reporting predictive performance of individual methods based on their respective publications since these results should not be directly compared due to differences in the datasets, metrics and test procedures used. We also note several drawbacks of CAID. It performs evaluation of binding predictions in a ligand agnostic way, i.e., different types of binding IDRs were clumped together. We note that the five above-baseline methods target prediction of protein-binding IDRs, benefitting from the fact that 72% of the binding annotations in the CAID dataset are protein-binding. Overall, this challenge shows substantial potential for future improvements. Interestingly, some of these limitations are being addressed in the currently pending CAID2 experiment (https://idpcentral.org/caid). CAID2 expands the assessment of predictions of binding IDRs by introducing assessment of ligand-specific prediction that cover protein-binding and nucleic-acid binding. This will likely result in a further growth in the efforts to generate more diverse and more accurate methods."], "subsections": []}]}, {"title": "Predictors of disordered binding regions", "content": ["Table 1 covers several important aspects of the 38 predictors of binding IDRs, such as their predictive architectures, modes of availability, and popularity quantified with citations. We categorize these methods into five groups based on the target of their predictions: MoRFs, SLiMs, protein-binding regions, lipid binding regions, and protein/DNA/RNA-binding regions. The methods in the latter category identify three types of binding IDRs, those that interact with proteins, with DNA, and with RNA."], "subsections": [{"title": "Predictors of MoRFs", "content": ["The largest group of predictors of binding IDRs focuses on the MoRF regions, with 21 out of the 38 methods (55%) in this category (Table 1). The defining feature of MoRFs is their ability to transition to structured conformation upon binding to proteins and peptides, which implies that the underlying interaction-dependent structure differentiates them from other binding IDRs. While the first MoRF predictor targeted α-MoRFs, majority of the subsequent tools were designed to target all types of MoRFs, irrespective of how they fold upon binding.", "The most popular (i.e., based on annual number of citations listed in Table 1) and available to the end users MoRF predictors include MoRFpred [87], MoRFCHiBi\n[90], DISOPRED3 [91], fMoRFpred [36] and OPAL [97]. We briefly summarize MoRFpred in Section 2.1. MoRFCHiBi was first published in 2015 and has been successively improved by the same authors 90, 92, 93, ultimately resulting in the MoRFCHiBi SYSTEM that is composed of three predictors: MoRFCHiBi, MoRFCHiBi_Light and MoRFCHiBi_Web\n[93]. MoRFCHiBi_Light and MoRFCHiBi_Web rely on predictions from MoRFCHiBi, but MoRFCHiBi_Light does not utilize computationally expensive PSSM profiles, which makes it much faster than MoRFCHiBi_Web. Thus, users of the MoRFCHiBi SYSTEM have an option to apply a fast MoRFCHiBi_Light version or slower and more accurate MoRFCHiBi_Web version.", "DISOPRED3 is a popular predictor of disorder that includes an option to predict MoRF regions [91]. The disorder predictor uses a small neural network to combine SVM-based DISOPRED2 model [130], neural network specialized to predict long IDRs, and a nearest neighbor-based classifier that takes advantage of similarity to annotations in a training dataset. DISOPRED3 applies a separate SVM-based model that uses information extracted from the input sequence and its PSSM profile to predict MoRFs.", "Another popular MoRFs predictor is OPAL [97]. This is a meta-predictor that averages results produced by two MoRF predictors: MoRFCHiBi and a relatively slow PROMIS [97]. The fMoRFpred tool represent an opposite approach, with a simpler architecture and fast runtime [36]. This method utilizes a basic SVM-based model that relies on fast-to-compute putative disorder predicted with IUPred [131] and putative secondary structure generated with the fast single-sequence version of PSIPRED [132]."], "subsections": []}, {"title": "Predictors of SLiMs", "content": ["Majority of predictors that target SLiMs rely on regular expressions to identify these motifs in protein sequences. This is the second most populous category of predictors, with 9 methods published to date (Table 1). The most popular and available to the end users SLiMs predictors include SLiMFinder [109], which we described in Section 2.1, and SLiMSearch 4.0 [117]. The latter tool is a successor of the SLiMSearch 1.0 [110] and SLiMSearch 2.0 [111] methods. SLiMSearch 4.0 is an advanced framework that identifies SLiMs using likelihood-based scoring of motifs, sequence conservation, functional enrichment analysis using Gene Ontology (GO) terms, and filters that consider putative disorder generated with IUPred, surface accessibility when structure is available, and overlap with Pfam domains [117]. Moreover, SLiMSearch 4.0 identifies SLiMs in a taxonomy-aware manner, focusing on around 70 model species that include human, yeast, mouse, fruit fly, C. elegans, and A. thaliana. We also note a recently released SLiMSuite package [133], which provides convenient access to multiple tools for discovery and characterization of SLiMs: SLiMProb [110](also known as SLiMSearch 1.0), SLiMFinder [109] and QSLiMFinder [115]. Besides these regular expression-based tools, there are two methods that utilize machine-learning models to predict SLiMs: SLiMPred [112] and PepBindPred [114]. Both methods apply bidirectional recurrent neural network models and rely on information extracted from sequence-derived predictions of secondary structure, intrinsic disorder and solvent accessibility. PepBindPred additionally performs docking between the interacting molecules."], "subsections": []}, {"title": "Predictors of protein, RNA, DNA and lipid-binding regions", "content": ["There are three predictors which target protein-binding IDRs: ANCHOR 105, 106, which we discussed in Section 2.1, ANCHOR2 [107] and IDRBind [108]. The two ANCHOR methods are arguably the most popular predictors of binding IDRs. ANCHOR2 improves over ANCHOR by extending its scoring function with additional terms, which results in a more accurate model.", "Recent years observed the push to develop methods that predict IDRs that interact with nucleic acids and lipids. There are three tools that predict DNA/RNA/protein-binding IDRs: DisoRDPbind 118, 119, flDPnn [120], and DeepDisoBind [121], and two tools that predict lipid-binding IDRs: DisoLipPred [122] and MemDis [123]. These methods, with the exception of DisoRDPbind, apply state-of-the-art deep learning models that we explore in Section 3.4."], "subsections": []}, {"title": "Predictive architectures", "content": ["We identify five categories of predictive architectures that are used to implement predictors of binding IDRs: scoring functions (SF), regular expressions (regex), shallow machine learning (ML) algorithms, deep-learning (DL) algorithms and meta-predictors; see “predictive architecture” column in Table 1. These categories are in line with similar analyses for the disorder predictors 67, 71, 72, 74.", "The SF-based models use pre-defined functions to combine evolutionary and biochemical features that are estimated from protein sequences. Key characteristics of these functions are that they utilize relatively few parameters and rely on explicit formulas that are typically derived from biophysical principles underlying interactions. Examples include retro-MoRF [86] and SLiMPrints [113] that utilize scoring functions based on the conservation extracted from multiple sequence alignments, and ANCHOR and ANCHOR2 that use interaction energy-based features 105, 107.", "Regex-based models are exclusively used for the prediction of SLiMs 109, 110, 111, 115, 117. Regex is a sequential combination of symbols and characters that represents a pattern for a short string that can be efficiently searched in a longer string (i.e., amino acid sequence). Using regex, prediction of SLiMs boils down to search for short motifs in a given protein sequence, followed by ranking to find statistically significant hits, and filtering to identify motifs in a specific part of the sequence, e.g., disordered region. Prominent examples of the regex-based predictors include SLiMFinder [109] and SLiMSearch 4.0 [117].", "ML and DL, the two most numerous categories, utilize machine learning algorithms to generate predictive models from training datasets. There are 28 of them in total including 9 DL models and 19 ML models. These algorithms depend on the quality and size of the training datasets since they utilize the ground truth from these datasets to optimize predictive models, such that they minimize differences between predictions and the corresponding grounds truth. Shallow ML algorithm are the traditional classifiers that in general produce smaller models and require less training data than the deep learning algorithms. Over a half of the shallow ML methods (i.e., 10 out of 19) utilize models produced with the support vector machine (SVM) algorithm 36, 87, 89, 90, 91, 94, 95, 96, 97, 98. Other algorithms include linear regression 118, 119, naïve Bayes 92, 93, XGBoost [108], minimax probability machine [100], and shallow neural networks 39, 85, 112, 114. The DL algorithms are neural networks with topologies that include multiple/many hidden layers and which also typically use more sophisticated types of neurons and utilize modern types of architectures, such as convolutional and recurrent networks. These models usually involve a large number of parameters (i.e., weights associated with the connections between neurons in the network) and thus they need large datasets to properly train these parameters. The DL-based predictors of binding IDRs apply a variety of architectures including convolutional 99, 101, 104, 121, bidirectional recurrent [122], recurrent Long Short-Term Memory (LSTM) [103], hybrid of convolutional and recurrent LSTM [123], as well as deep fully-connected perceptron network 102, 120. We note that methods developed since 2020 exclusively utilize the DL models. Part of the reason why these models could be developed is that a sufficient amount of training data has become available in recent years, driven mostly by the substantial growth of the DisProt database. When a sufficient amount of training data became available and given the breakthroughs in the designs of deep network architectures in the past decade and the resulting high-levels of their predictive performance, unsurprisingly, researchers in this field have shifted to adopt DL algorithms instead of traditional ML. This is likely also motivated by the recent influx and success of DL-based predictors of intrinsic disorder. Notably, the top-performing disorder predictors in CAID [134] include flDPnn [51], SPOT-Disorder2 [52], rawMSA [53] and AUCpred [54], all of which rely on the DL models. Furthermore, recent empirical study finds that the DL models in general produce more accurate disorder predictions when compared to the shallow ML models [64], which provides a strong justification to develop these models for prediction of binding IDRs.", "Finally, there are several meta-predictors which are defined as methods that combine predictions of binding IDRs produced by multiple predictors. The underlying objective is to provide more accurate results when compared to the results produced by the input predictors. This approach was used to develop several popular and accurate disorder predictors 135, 136, 137, 138, 139, 140, 141. We identify four meta-predictors, all of which predict MoRFs, including MoRFCHiBi_Web, MoRFCHiBi SYSTEM, OPAL and OPAL+ 90, 93, 97, 98. The focus on MoRFs can be explained by the fact that the most and large number of predictors target this category of binding IDRs, providing a deep pool of input predictions for the meta-method.", "Lastly, we detail predictive models of the five methods that performed well in the CAID experiment [63]: ANCHOR2, DisoRDPbind, two versions of MoRFCHiBi method, and OPAL. ANCHOR2 [107] is the SF-based model that improves over its predecessor, ANCHOR 105, 106. ANCHOR implements SF that quantifies differences in basic biophysical properties of disorder binding residues between their bound and unbound state. It combines the putative disorder information generated by IUPred with estimates of pairwise interaction energy of disordered residues with globular proteins and local disordered sequence segments. ANCHOR2 uses a computationally efficient linear function to combine the interaction energy estimation from ANCHOR with two new terms that estimate energy for interaction with binding surface of globular proteins and presence of a disordered sequence. This results in a more accurate model that still retains the small computational footprint of ANCHOR.", "DisoRDPbind [118] is a shallow ML method that utilizes three logistic regression models to predict RNA-binding, DNA-binding and protein-binding propensities, one regression for each ligand type. These regressions use a common input profile generated from the sequence that includes information about hydrophobicity and net charge, putative disorder produced with IUPred [59], putative secondary structure generated by a single-sequence version of PSIPRED [142], and sequence complexity computed by the SEG algorithm [143]. This profile is processed to generate inputs for the regressions using sliding-windows with sizes that are optimized for specific ligand types.", "MoRFCHiBi SYSTEM [93] is also a shallow ML predictor but it features a multi-layer architecture. The bottom layer implements the base MoRFCHiBi model that uses a Bayes rule to combine MoRF predictions from two SVM models, one that is trained directly on sequences and the other that relies on similarities between sequences. The second layer implements the MoRFCHiBi_Light prediction [93] by using a Bayesian model to fuse predictions from the base MoRFCHiBi with the predictions of disorder from ESpritz-DisProt [55]. The third layer implements the MoRFCHiBi_Web prediction [93] that again uses a Bayesian model to combine the base MoRFCHiBi, the ESpritz-DisProt predictions and the conservation derived from the sequence using PSI-BLAST [144]. Benchmarking done by the authors suggests that MoRFCHiBi_Light produces more accurate predictions than the base MoRFCHiBi, while MoRFCHiBi_Web further increases accuracy but at substantially higher computational cost due to the calculation of the conservation [93].", "OPAL [97] is a meta-predictor that averages results produced by the base MoRFCHiBi model and PROMIS, a relatively slow MoRF predictor developed by the authors of OPAL. PROMIS predicts MoRFs using an SVM model based on putative solvent accessibility, secondary structure and torsional angles predicted from the input sequence with SPIDER2 [145] and a PSSM profile generated from the sequence with PSI-BLAST. The need to compute the PSSM profiles results in a relatively long runtime.", "We highlight the fact that these models are rather diverse. They utilize a variety of predictive architectures and different inputs that are derived from the sequence. They also vary in terms of their runtime. The CAID experiment reports that ANCHOR2 and DisoRDPbind take around 1 second to predict one protein, MoRFCHiBi_Light takes a few seconds, and the other two methods require two order of magnitude more runtime due to the use of PSI-BLAST, i.e., about 100 seconds for MoRFCHiBi_Web and over 500 seconds for OPAL [63]."], "subsections": []}, {"title": "Availability and impact", "content": ["Availability of these predictors to a broad scientific user-group is an important factor to facilitate research on binding IDRs. Table 1 provides details on implementations and whether they are currently available, i.e., as of July 2022 when we collected these data. There are two types of implementations: webserver (WS) and source code (SC). WS is available online via a web browser or programmable interface, typically does not require installation of any software, and performs all computations on the server side. While webservers are usually accessed via webpages, in a few cases (e.g., SLiMPred, SLiMPrints and QSLiMFinder) the access is based on the representational state transfer (REST) interface. SC have to be downloaded, installed/compiled and run on user’s hardware. While WSs are easier to use, they are typically limited to prediction of a single or a few proteins at the time and could be difficult to embed into other bioinformatics platforms if they lack programmable interface. On the other hand, SC usually can be setup to perform predictions on a larger scale and is easier to incorporate into other bioinformatics software, but it can be challenging to install and requires hardware to run. We collect the location of these WS and SC resources as per information given in the respective publications and check their availability. We find 27 methods that have working WS and/or SC implementations. Among them 13 methods are available solely as WS, 3 as SC and 11 as both WS and SC. There are 4 methods which were once functional but as of July 2022 did not work, and 7 methods that were never implemented for public use. The corresponding 71% rate of availability (27 out of 38 methods) is relatively high, higher than the 65% availability rate for disorder predictors [62], and much higher than the approximately 40% rate for other related predictors of protein-binding and nucleic acid binding residues 146, 147.", "We analyze impact/use of these methods, which we quantify in using citations collected from Google Scholar as of July 2022 (Table 1). We provide total number of citations as well as an annual count, where the latter is a better metric to compare impact/use of different methods. The predictors published from 2020 onwards are too new to reliably measure their citation data, hence, we exclude them from the below analysis. We find that methods which offer WS and/or SC implementation are cited much more often (median annual citations = 12), compared to methods which were never made available (median annual citations = 3). Moreover, among the methods which are currently functional, the tools that provide both WS and SC are cited more (median annual citations = 16) compared to the methods that provide only WS (median annual citations = 12) and only SC (median annual citations = 4). The higher popularity of predictors implemented as WSs is because they are arguably more convenient for majority of users who have limited computational resources and are less computer savvy to be able to install and run software locally. Methods with no implementations suffer low citations, revealing that availability directly influences the level of use and impact. These observations suggest that future methods should be made available as both WS and SC to maximize impact. Moreover, we find that among the methods which have/had WS and/or SC implementations, the ones which are currently non-functional receive median annual citations of 4, which is 4 times lower than the functional methods. This means that it is vitally important to maintain availability after methods are released.", "We briefly discuss impact/use of individual tools. Predictor of protein-binding IDRs, ANCHOR2, is the most highly cited method, both in terms of annual citations (189) and total citations (754). We note that ANCHOR2′s publication also introduces a popular disorder predictor, IUPred2, which likely inflates the above number; this is also why we use median annual citations to compare groups of tools. There are 9 predictors that were cited over 100 times and 4 of them were cited over 500 times. These observations should be considered with a pinch of salt, since these tools were published in 2016 or earlier and had more time to accumulate citations when compared to newer methods. However, this reveals a significant amount of interest in using these methods."], "subsections": []}]}, {"title": "Summary and outlook", "content": ["IDRs interact with many different molecular partners including proteins, DNA, RNA, lipids, small molecules, carbohydrates, and metals. The knowledge of these interactions is rather limited, which motivates development of computations tools that predicts them from the readily available protein sequences. This comprehensive survey of sequence-based predictors of binding IDRs covers a wide range of interacting partners. We identify and summarize a large collection of 38 predictors that consider 5 different types of interacting IDRs. The MoRF predictors are the largest category with 21 methods, followed by 9 SLiM predictors, 3 predictors of protein-binding IDRs, 3 methods that predict protein/DNA/RNA binding IDRs and 2 predictors of lipid-binding IDRs. We find that these methods rely on a diverse range of predictive architectures that include scoring functions, regular expressions, machine learning models and meta-predictors, where about three-quarters of them utilize machine learning algorithms. We observe a couple of recent trends to develop deep network-based models and to extend coverage to new types of interacting IDRs, such as RNA, DNA and lipid binding regions. We also note a high rate of availability of these methods, with over 70% that are provided to the end users as either webservers and/or standalone code. Furthermore, we analyze relation between availability and impact/use of these methods. We find that methods which are more broadly available, as both webserver and source code, are substantially more cited/used when compared to those that are available in either format, while methods that do not offer a publicly available implementation suffer low use/citations. Moreover, we also find that the availability should be maintained since tools that were originally made available and are currently not functional observe a large drop in the use/citations. The latter observations strongly suggest that future predictors should be made available in both formats upon publication and should be maintained after publication.", "While IDPs interact with a broad range of molecular partners, we show that the current predictors are largely focused on two types of binding IDRs, MoRFs and SLiMs. A particularly acute situation concerns prediction of nucleic acid and lipid-binding IDRs, where only a handful of methods are available. The prediction of small molecules-, carbohydrates-, and metal-binding IDRs is not feasible at the moment, given a very small amount of ground truth data. The need to develop new predictors of DNA and RNA binding regions is further motivated by the inclusion of this prediction category in the pending CAID2 experiment. Consequently, one of the key future directions would be to diversify the development efforts to more uniformly cover different types of binding IDRs.", "Results of the recently completed CAID assessment show that predictors of binding IDRs offer modest levels of predictive performance [63], suggesting that there is a large room for improvement. We observe that none of the methods that participated in this evaluation use deep learning models. The recent influx of the deep learning-based predictors of binding IDRs will likely result in improved predictive quality. This claim stems from a recent study that empirically demonstrates that deep learning-based predictors of intrinsic disorder significantly outperform other types of models [64]. The drive to use deep learning models is also motivated by the growing and successful use of these models in related areas of bioinformatics [148], such as prediction of protein-protein interactions 149, 150, 151 and protein function 152, 153, 154. We envision that majority of future predictors of binding IDRs will likely rely on deep neural networks. We encourage the developers to consider modern network topologies, such as the recently developed transformers [155], that were used to very accurately predict protein structures [156].", "Some IDPs include IDRs that interact with different types of ligands and yet most of the current methods cover a single ligand type. Consequently, users are forced to use multiple methods and convert between different output formats to obtain a complete prediction. These difficulties could be alleviated with solutions that bundle multiple predictors, however, the only such solution to date is the DEPICTER webserver [157]. Moreover, there are only a handful of methods that predict IDRs that bind to multiple ligand types, such as DisoRDPbind, flDPnn and DeepDISObind, that target protein, RNA and DNA-binding IDRs. Consequently, we advocate for the development of new tools that address predictions of multiple and many different types of binding IDRs. Furthermore, some IDRs can bind multiple partner types, which corresponds to multi-label (multi-output) learning. Prediction of such multifunctional IDRs is possible with the DMRpred method, although this tool does not provide types of binding partners [158]. Thus, new tools that would cast this prediction as multi-labels problem should be developed. We note that multi-labels predictors are widely used in related areas, such as prediction of subcellular localization 159, 160, 161, 162, nucleic acid binding proteins [163], enzymatic functions [164], and ion channel types [165].", "Prediction of the binding IDRs in protein sequences should be followed by modelling structures of the corresponding complexes (i.e., IDRs fold upon binding). While computational protein docking has been extensively pursued over the past several years [166], studies that investigate docking with IDPs are lagging behind since IDPs are difficult to model. Daisuke Kihara’s lab developed a pioneering approach for IDP-protein docking, IDP-LZerD 167, 168. This method produces a docking model from the 3D structure of the receptor and the sequence of interacting IDP. Docking an IDP is conceptually similar to protein-small peptide docking, but technically more challenging because conformation of the IDP on the receptor’s surface has to be predicted. In IDP-LZerD, this is done by docking and stitching short protein fragments taken from the binding IDR. Moreover, a recent benchmark study that evaluates three methods capable of docking with IDPs, IDP-LZerD 167, 168, CABS-Dock [169] and AlphaFold-Multimer [170], shows that they accurately identify location of the binding site but struggle with atomic-levels details of the structure [171], suggesting that further research is needed.", "Lastly, databases like D2P2\n[172], MobiDB 173, 174, 175, 176 and DescribePROT [177] provide convenient access to pre-computed predictions of disorder for millions of proteins. However, they typically contain a limited number of binding IDR predictions, with DescribePROT covering the most diverse range that includes putative protein, RNA and DNA-binding IDRs. This coverage should be extended in the future as more methods that cover a broader range of binding IDRs will be developed. In turn, this effort motivates the development of runtime-efficient predictors that can be used to perform predictions on such large scale. Examples of current fast tools include ANCHOR2, DisoRDPbind and fMoRFpred, that were shown to produce predictions in about 1 second per protein in the CAID experiment [63]."], "subsections": []}, {"title": "Funding", "content": ["LK was funded in part by the 10.13039/100000001National Science Foundation (DBI2146027 and IIS2125218) and the Robert J. Mattauch Endowment funds. DK acknowledges supports from the 10.13039/100000002National Institutes of Health (R01GM133840 and 3R01GM133840–02S1) and the 10.13039/100000001National Science Foundation (CMMI1825941, MCB1925643, DBI2146026, IIS2211598, DMS2151678, and DBI2003635)."], "subsections": []}, {"title": "CRediT authorship contribution statement", "content": ["Sushmita Basu: Data curation; Formal analysis; Investigation; Methodology; Validation; Writing – original draft. Lukasz Kurgan: Conceptualization; Data curation; Formal analysis; Funding acquisition; Investigation; Project administration; Supervision; Validation; Writing – original draft; Writing – review & editing. Daisuke Kihara: Conceptualization; Funding acquisition; Investigation; Supervision; Writing – original draft; Writing – review & editing."], "subsections": []}, {"title": "Conflicts of interest", "content": ["The authors declare no conflicts of interest."], "subsections": []}]}}
{"pmid": "38701796", "meta": {"content": {"abstract": "Despite their lack of a rigid structure, intrinsically disordered regions (IDRs) in proteins play important roles in cellular functions, including mediating protein-protein interactions. Therefore, it is important to computationally annotate IDRs with high accuracy. In this study, we present Disordered Region prediction using Bidirectional Encoder Representations from Transformers (DR-BERT), a compact protein language model. Unlike most popular tools, DR-BERT is pretrained on unannotated proteins and trained to predict IDRs without relying on explicit evolutionary or biophysical data. Despite this, DR-BERT demonstrates significant improvement over existing methods on the Critical Assessment of protein Intrinsic Disorder (CAID) evaluation dataset and outperforms competitors on two out of four test cases in the CAID 2 dataset, while maintaining competitiveness in the others. This performance is due to the information learned during pretraining and DR-BERT's ability to use contextual information.", "keywords": ["IDP", "IDR", "deep learning", "disorder", "machine learning", "protein language model", "protein structure prediction"], "mesh_terms": ["*Intrinsically Disordered Proteins/chemistry/metabolism", "Databases, Protein", "Models, Molecular", "Computational Biology/methods", "Protein Conformation", "Molecular Sequence Annotation", "Algorithms"], "pub_types": ["Journal Article", "Research Support, Non-U.S. Gov't", "Research Support, U.S. Gov't, Non-P.H.S."]}, "contributors": {"medline": {"affiliations": ["Department of Bioengineering, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA; Carl R. Woese Institute for Genomic Biology, Urbana, IL 61801, USA. Electronic address: nambiar4@illinois.edu.", "Carl R. Woese Institute for Genomic Biology, Urbana, IL 61801, USA; Department of Computer Science, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA.", "Carl R. Woese Institute for Genomic Biology, Urbana, IL 61801, USA; Department of Computer Science, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA.", "Department of Bioengineering, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA; Carl R. Woese Institute for Genomic Biology, Urbana, IL 61801, USA; Department of Physics, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA; Computing, Environment and Life Sciences, Argonne National Laboratory, Lemont, IL 60439, USA. Electronic address: maslov@illinois.edu."], "auids": [], "full_names": ["Nambiar, Ananthan", "Forsyth, John Malcolm", "Liu, Simon", "Maslov, Sergei"], "short_names": ["Nambiar A", "Forsyth JM", "Liu S", "Maslov S"]}, "xml": [{"affiliations": ["Department of Bioengineering, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA; Carl R. Woese Institute for Genomic Biology, Urbana, IL 61801, USA. Electronic address: nambiar4@illinois.edu."], "full_name": "Nambiar, Ananthan", "identifiers": [], "short_name": "Nambiar A"}, {"affiliations": ["Carl R. Woese Institute for Genomic Biology, Urbana, IL 61801, USA; Department of Computer Science, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA."], "full_name": "Forsyth, John Malcolm", "identifiers": [], "short_name": "Forsyth JM"}, {"affiliations": ["Carl R. Woese Institute for Genomic Biology, Urbana, IL 61801, USA; Department of Computer Science, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA."], "full_name": "Liu, Simon", "identifiers": [], "short_name": "Liu S"}, {"affiliations": ["Department of Bioengineering, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA; Carl R. Woese Institute for Genomic Biology, Urbana, IL 61801, USA; Department of Physics, University of Illinois Urbana-Champaign, Urbana, IL 61801, USA; Computing, Environment and Life Sciences, Argonne National Laboratory, Lemont, IL 60439, USA. Electronic address: maslov@illinois.edu."], "full_name": "Maslov, Sergei", "identifiers": [], "short_name": "Maslov S"}]}, "identity": {"doi": "10.1016/j.str.2024.04.010", "pmid": "38701796", "title": "DR-BERT: A protein language model to annotate disordered regions."}, "links": {"cites": ["41689628", "41523656", "41452662", "41365991", "40794952", "40601248", "40577786", "40301326", "39720898", "38746434", "38040454"], "entrez": {}, "external": [{"attribute": "free resource", "category": "Full Text Sources", "linkname": "", "provider": "Elsevier Science", "url": "https://linkinghub.elsevier.com/retrieve/pii/S0969-2126(24)00136-9"}], "pmc": [], "refs": [], "review": ["38701796", "39093570", "31521235", "31441158", "36851914", "30657889", "34418423", "35882686", "23466039", "33248138", "36959451", "38296708", "40081192", "34390736", "28589442", "32034199", "35139468", "37657994"], "similar": ["38701796", "39576583", "37992088", "40465350", "39576585", "40254833", "34830151", "38166858", "38079339", "37883249", "39093570", "40859602", "31521235", "36210722", "40944448", "38812466", "33875885", "30135358", "36272675", "39571839", "40133781", "27115638", "40854226", "40054813", "40662788", "32112084", "31441158", "33866372", "32535960", "36908253", "36851914", "41378882", "34290238", "39246251", "31207240", "32702119", "36315580", "25391399", "28394890", "32696355", "38728811", "28934129", "30657889", "40828855", "34418423", "31797595", "34626492", "38987470", "39586094", "28369666", "35882686", "23466039", "33248138", "40577786", "38297118", "40133791", "38747347", "36008992", "40493668", "30878482", "23946100", "40441416", "36959451", "27787828", "32723719", "32696395", "40221640", "39446390", "39688476", "40440586", "39032884", "40244295", "38296708", "29791766", "40867524", "39910649", "38781245", "34048569", "32599863", "40081192", "34905768", "37498545", "33936509", "38435559", "31724233", "40138319", "39708500", "33719450", "34383671", "34390736", "28589442", "35094056", "39761555", "32173600", "32034199", "40133775", "39718444", "35139468", "37657994", "35102880"], "text_mined": []}, "metadata": {"entrez_date": "2024/05/04 11:50", "fetched_at": "2026-05-02 18:19:10"}, "source": {"journal_abbrev": ["Structure"], "journal_title": ["Structure (London, England : 1993)"], "pub_date": "2024 Aug 8", "pub_types": ["Journal Article", "Research Support, Non-U.S. Gov't", "Research Support, U.S. Gov't, Non-P.H.S."], "pub_year": "2024"}}, "content": {}}
{"pmid": "39763873", "meta": {"content": {"abstract": "Intrinsically disordered proteins or regions (IDPs/IDRs) adopt diverse binding modes with different partners, from coupled-folding-and-binding, to fuzzy binding, to fully-disordered binding. Characterizing IDR interfaces is challenging experimentally and computationally. The state-of-the-art AlphaFold-multimer and AlphaFold3 can be used to predict IDR binding sites, although they are less accurate at their benchmarked confidence cutoffs. Here, we developed Disobind, a deep-learning method that predicts inter-protein contact maps and interface residues for an IDR and its partner, given their sequences. It uses sequence embeddings from the ProtT5 protein language model. Disobind outperforms state-of-the-art interface predictors for IDRs. It also outperforms AlphaFold-multimer and AlphaFold3 at multiple confidence cutoffs. Combining Disobind and AlphaFold-multimer predictions further improves the performance. In contrast to current methods, Disobind considers the context of the binding partner and does not depend on structures and multiple sequence alignments. Its predictions can be used to localize IDRs in large assemblies and characterize IDR-mediated interactions.", "keywords": ["Intrinsically disordered proteins (IDP)", "deep learning (DL)", "intrinsically disordered regions (IDR)", "protein language model (pLMs)", "protein structure"], "mesh_terms": [], "pub_types": ["Journal Article", "Preprint"]}, "contributors": {"medline": {"affiliations": ["National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore, India 560065.", "National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore, India 560065.", "National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore, India 560065."], "auids": ["ORCID: 0009-0006-3190-7977", "ORCID: 0000-0002-1238-7041", "ORCID: 0000-0002-9061-8407"], "full_names": ["Majila, Kartik", "Ullanat, Varun", "Viswanath, Shruthi"], "short_names": ["Majila K", "Ullanat V", "Viswanath S"]}, "xml": [{"affiliations": ["National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore, India 560065."], "full_name": "Majila, Kartik", "identifiers": ["0009-0006-3190-7977"], "short_name": "Majila K"}, {"affiliations": ["National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore, India 560065."], "full_name": "Ullanat, Varun", "identifiers": ["0000-0002-1238-7041"], "short_name": "Ullanat V"}, {"affiliations": ["National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore, India 560065."], "full_name": "Viswanath, Shruthi", "identifiers": ["0000-0002-9061-8407"], "short_name": "Viswanath S"}]}, "identity": {"doi": "10.1101/2024.12.19.629373", "pmid": "39763873", "title": "A deep learning method for predicting interactions for intrinsically disordered regions of proteins."}, "links": {"cites": [], "entrez": {}, "external": [{"attribute": "free resource", "category": "Full Text Sources", "linkname": "", "provider": "PubMed Central", "url": "https://pmc.ncbi.nlm.nih.gov/articles/pmid/39763873/"}, {"attribute": "free resource", "category": "Full Text Sources", "linkname": "", "provider": "Cold Spring Harbor Laboratory", "url": "https://doi.org/10.1101/2024.12.19.629373"}], "pmc": ["11702703"], "refs": ["40403066", "40133781", "40080667", "39756261", "39740355", "39665266", "39633723", "39548826", "39510344", "39470701", "39446390", "39115813", "38866950", "38781245", "38718835", "38649368", "38547871", "38391029", "38330793", "38238291", "38225382", "38102755", "37957331", "37956700", "37904608", "37904585", "37878721", "36815589", "36478084", "36379943", "36040254", "35900023", "35679397", "35609776", "35238773", "35104320", "34982960", "34905768", "34845384", "34448830", "34433091", "34418423", "34390736", "34265844", "34232869", "34139171", "33876751", "33693581", "33344917", "33186584", "32661420", "32621228", "32321157", "32112804", "31906602", "31831797", "31819266", "30355150", "30066087", "29385418", "29069470", "29036655", "29035372", "28597296", "28516010", "27794553", "25038412", "24606139", "24178034", "24028092", "23988124", "23665034", "23203869", "22466611", "22272186", "21474068", "19412530", "16935303", "16406523", "12580598", "10837058", "10550212"], "review": ["39763873", "31441158", "36851914", "35139468", "34418423", "39093570", "36833360", "33194509", "38296708", "33248138", "34358545", "33243935", "39740355", "36959451", "34390736", "33567297", "35882686", "31709918", "37657994", "40650026", "39205298"], "similar": ["39763873", "41534519", "39446390", "31441158", "36304335", "36851914", "40944448", "30058229", "37883249", "40034137", "37878721", "35139468", "30878482", "28780862", "36272675", "39253485", "34418423", "33087759", "31207240", "38866950", "31307006", "32696395", "41454828", "23142703", "40756902", "39093570", "36908253", "31724233", "36833360", "33194509", "33936509", "38296708", "29763584", "35609776", "40254833", "32112084", "38701796", "36153968", "34461937", "40403066", "41174305", "32824743", "40221640", "30841624", "33248138", "34358545", "39933697", "26149687", "32743159", "32702119", "33243935", "34905768", "34825861", "40244295", "38166858", "40133775", "39740355", "37498545", "29066345", "38987470", "39433443", "36959451", "34390736", "38895487", "33567297", "41378882", "33036302", "34179096", "32722039", "36210722", "36005688", "28381244", "40574713", "35882686", "40441416", "39586094", "38079339", "37992088", "24115198", "34710113", "41053908", "33267376", "31709918", "36315580", "35310482", "32553192", "21647374", "32535960", "33719450", "31797594", "37657994", "29716994", "40650026", "39205298", "30366362", "40997106", "38942776", "31003202", "31756456", "40317235"], "text_mined": [{"category": "General", "source": "full_text", "url": "https://doi.org/10.5281/zenodo.14504762"}, {"category": "GitHub", "source": "full_text", "url": "https://github.com/isblab/disobind"}]}, "metadata": {"entrez_date": "2025/01/07 06:22", "fetched_at": "2026-05-02 18:19:10"}, "source": {"journal_abbrev": ["bioRxiv"], "journal_title": ["bioRxiv : the preprint server for biology"], "pub_date": "2025 Aug 15", "pub_types": ["Journal Article", "Preprint"], "pub_year": "2025"}}, "content": {"title": "A deep learning method for predicting interactions for intrinsically disordered regions of proteins", "body": [{"title": "Summary", "content": ["Intrinsically disordered proteins or regions (IDPs/IDRs) adopt diverse binding modes with different partners, from coupled-folding-and-binding, to fuzzy binding, to fully-disordered binding. Characterizing IDR interfaces is challenging experimentally and computationally. The state-of-the-art AlphaFold-multimer and AlphaFold3 can be used to predict IDR binding sites, although they are less accurate at their benchmarked confidence cutoffs. Here, we developed Disobind, a deep-learning method that predicts inter-protein contact maps and interface residues for an IDR and its partner, given their sequences. It uses sequence embeddings from the ProtT5 protein language model. Disobind outperforms state-of-the-art interface predictors for IDRs. It also outperforms AlphaFold-multimer and AlphaFold3 at multiple confidence cutoffs. Combining Disobind and AlphaFold-multimer predictions further improves the performance. In contrast to current methods, Disobind considers the context of the binding partner and does not depend on structures and multiple sequence alignments. Its predictions can be used to localize IDRs in large assemblies and characterize IDR-mediated interactions."], "subsections": []}, {"title": "Introduction", "content": ["Intrinsically disordered proteins or regions (IDPs or IDRs) lack a well-defined three-dimensional structure in their monomeric state (Oldfield & Dunker, 2014; Wright & Dyson, 1999). They exist as an ensemble of interconverting conformers in equilibrium and hence are structurally heterogeneous (Lindorff-Larsen & Kragelund, 2021; Oldfield & Dunker, 2014; Uversky, 2013b). Here, we use the term IDR for both IDPs and IDRs. The heterogeneity of IDRs provides several functional advantages including the ability to overcome steric restrictions, have a large capture radius, undergo functional misfolding, and interact with multiple partners (Fonin et al., 2018). They play critical roles in cellular processes including signaling, intracellular transport, protein folding, and condensate formation (Fonin et al., 2018; Uversky, 2013b).", "IDRs are known to mediate a large number of protein-protein interactions (Tompa et al., 2014). They exhibit considerable diversity in their binding modes within complexes (Holehouse & Kragelund, 2024; Olsen et al., 2017). Upon binding to a partner protein, they may adopt an ordered state (disorder-to-order or DOR) or may remain disordered (disorder-to-disorder or DDR) (Holehouse & Kragelund, 2024; Miskei et al., 2017, 2020; Orand & Jensen, 2025). Moreover, depending on the binding partner, the same IDR may undergo DOR and DDR transitions (Miskei et al., 2020; Orand & Jensen, 2025). For example, IDRs with short linear motifs (SLiMs) may form multivalent interactions with their partner, exchanging between conformations where one or more motifs bind at a time (Orand & Jensen, 2025). This heterogeneity in binding makes their structural characterization in complexes challenging.", "Here, we develop a deep learning method to determine the binding sites of an IDR and an interacting partner protein given their sequences. Several methods have been developed for predicting interface residues, inter-protein contact maps, and structures of complexes formed by two proteins. Some of them are limited by the requirement of a structure of the input proteins (Christoffer & Kihara, 2020; Dai & Bailey-Kellogg, 2021; Gainza et al., 2020; Guo et al., 2022; Lin et al., 2023; Pittala & Bailey-Kellogg, 2020). Sequence-based methods that require a multiple sequence alignment (MSA) are often limited by the low sequence conservation of IDRs (Abramson et al., 2024; Bret et al., 2024; Jumper et al., 2021; B.-G. Lee et al., 2020). Other sequence-based methods for IDRs use physicochemical features or embeddings from protein language models (pLMs) but do not account for the partner protein (Jahn et al., 2024; Mészáros et al., 2009; Singh et al., 2022; F. Zhang et al., 2022). However, the knowledge of the partner protein is crucial in understanding IDR interactions as these are context-dependent. To this end, FINCHES predicts intermolecular interactions for two IDRs based on their chemical specificity, given their sequences. However, it is currently limited to IDRs that remain disordered upon binding (Ginell et al., 2025).", "State-of-the-art methods like AlphaFold2 (AF2) and AlphaFold3 (AF3) tend to provide low-confidence predictions for IDRs (Abramson et al., 2024; Evans et al., 2022; Jumper et al., 2021). Even though the low-confidence regions predicted by AF2 often correspond to the presence of IDRs, the predicted structure cannot be considered a representative IDR structure (Alderson et al., 2022; Ruff & Pappu, 2021). Recent studies showed that AlphaFold-multimer (referred to as AF2 hereon) could predict the structures of complexes where the IDR undergoes a DOR transition, including protein-peptide complexes and domain-motif complexes (Bret et al., 2024; B.-G. Lee et al., 2020; Omidi et al., 2024). However, the results may be sensitive to the inputs, such as the sequence fragment size, the fragment delimitation, and the alignment mode for the MSA (Bret et al., 2024; B.-G. Lee et al., 2020). In general, predicting the binding interfaces for IDRs at high resolution remains a challenge for current methods (Verburgt et al., 2022).", "Our method, Disobind, predicts the inter-protein contact maps and interface residues for an IDR and its partner from their sequences. Predicting inter-protein contact maps for IDRs presents several challenges. The structural data for IDRs in complexes is limited, the IDRs may retain their heterogeneity in complexes, and the inter-protein contact maps can be sparse. Here, we leverage the pLM ProtT5 to train our model with limited data, without the need for large, resource-intensive architectures. Predicting coarse-grained inter-protein contact maps and interface residues reduces problems associated with sparsity and heterogeneity. Disobind performs better than state-of-the-art interface predictors for IDRs and better than AF2 and AF3 across multiple confidence cutoffs used in this study. Combining the Disobind and AF2 predictions further improves the performance. We demonstrate the performance of Disobind+AF2 on several IDR complexes, on multiple examples, including several DOR and DDR complexes. Predictions from the method can be used to localize IDRs in integrative structures of large assemblies. They can be used to characterize protein-protein interactions involving IDRs, identify novel motifs, and modulate IDR-mediated interactions."], "subsections": []}, {"title": "Results", "content": [], "subsections": [{"title": "Disobind dataset creation", "content": ["Given a pair of protein sequences, at least one of which is an IDR, Disobind predicts binary inter-protein contact maps and interface residues for the pair (Fig 1). We compile our dataset by gathering structures of IDR-containing complexes from an array of existing IDR databases and datasets, including DIBS, MFIB, FuzDB, PDBtot, PDBcdr, DisProt, IDEAL, and MobiDB, covering a range of IDR binding modes (Fig 1) (See Methods). Missing residues in the structures are excluded from the input sequences, resulting in sequence fragments for each PDB chain. Further, we restrict the maximum length of a sequence fragment to 100 residues for computational reasons.", "We classify a sequence fragment as an IDR based on annotations supported by experimental evidence and sequence-based homology obtained from DisProt, IDEAL, and MobiDB; an IDR fragment has at least 20% residues annotated as disordered (See Methods). Next, we obtain inter-chain binary complexes from each PDB structure, in which at least one of the two fragments is an IDR (Fig 1). We obtain contact maps for all binary complexes. The contact maps are grouped based on the UniProt accessions of the corresponding sequence fragment pair. Next, we merge all contact maps with overlapping sequences in a group using a logical OR operation to obtain a merged contact map, ensuring that the length of each sequence fragment is up to 100 residues. Together, the sequence fragments for the pair and the corresponding merged contact map form a merged binary complex. Disobind is trained on a dataset of merged binary complexes with the sequence fragments as input and the merged contact map as output. The merged contact map encompasses contacts formed across all available structures of the complex, accounting for the multiple conformations of an IDR in a complex. This simplified output representation avoids making any assumptions about binding mode, allowing it to be applicable to DOR and DDR cases. Further, this provides for a less sparse output representation compared to an ensemble of contact maps, while also mitigating the bias caused by having an unequal number of structures for different sequence fragment pairs.", "We then create an out-of-distribution (OOD) test set comprising merged binary complexes where the input sequence fragments share less than 20% identity with the Disobind dataset and with the AlphaFold2 PDB70 dataset. The remaining complexes are further split into a train, validation, and an in-distribution (ID) test set. The ID and OOD test sets comprise 297 and 52 merged binary complexes respectively. Adding more complexes to the OOD test set may be achieved by increasing the sequence identity cutoff but would also increase similarity to the AF2 training dataset, which we attempted to avoid."], "subsections": []}, {"title": "Model architecture", "content": ["Disobind uses a shallow feedforward neural network architecture with the sequence embeddings obtained from a protein language model as input (Fig 2). The input embeddings are projected to a lower dimension using a projection block. Next, in the interaction block, the projected embeddings are used to compute the outer product and outer difference, which are subsequently concatenated along the feature dimension. This allows the model to capture complementarity between the input sequences and yields an interaction tensor. This interaction tensor is processed by a multi-layer perceptron (MLP) for contact map prediction. For interface residue prediction, the interaction tensor is first reduced by computing row-wise and column-wise averages and further processed by an MLP. Finally, an output block provides element-wise sigmoid scores, with a score greater than 0.5 representing a contact or interface residue. The outputs are binary contact maps or binary interface residue predictions on the fragments. The model is trained with the singularity-enhanced loss (SE loss) (Si & Yan, 2021) using the AdamW optimizer with weight decay (Fig S1). This loss is a modified form of the binary cross-entropy (BCE) loss used for class-imbalanced training. We varied the hyperparameters of the network including the projection dimension, the number of layers in the MLP, and the SE loss parameters (Table S1–S4, Fig S1). We use recall, precision, and F1-score to evaluate model performance."], "subsections": []}, {"title": "Evaluating Disobind", "content": [], "subsections": [{"title": "Inter-protein contact map prediction", "content": ["First, we evaluate Disobind on predicting residue-wise inter-protein contact maps on the ID and OOD test sets. Disobind achieves an F1-score of 0.57 and 0.33 on the ID and OOD test sets respectively (Table 1 and Table S5, contact map prediction at coarse-grained (CG) resolution 1). It performs better than a random baseline (see Methods) on the OOD test set (Table 1). However, predicting residue-wise inter-protein contact maps is challenging due to their sparsity (Table S6). This leads to a class imbalance, biasing the model to predict just the majority class, i.e., 0’s or non-contacts. The heterogeneity of IDRs in complexes poses another challenge. For example, the multivalency of IDRs may result in different sets of contacts in different conformations with the same partner (J. Zhang et al., 2024)."], "subsections": []}, {"title": "Interface residue prediction", "content": ["As an alternative to predicting inter-protein residue-wise contact maps, next, we predict interface residues for the IDR and its partner (Fig 3a). For two input sequence fragments of length L1 and L2, this reduces the number of predicted elements from L1 × L2 in the contact map prediction to L1 + L2 in the interface residue prediction (Fig 3b). This helps mitigate the class imbalance as the interface residue predictions are less sparse (Fig 3a, Fig 3b\nTable S6, interface residue prediction at coarse-grained (CG) resolution 1). Interface residue prediction is also less affected by the heterogeneity of IDRs in complexes. The prediction for each residue is simplified to determining whether it binds to the partner, rather than identifying the specific residues within the partner it interacts with.", "As expected, Disobind performs better in interface residue prediction than in contact map prediction. It achieves an F1 score of 0.70 and 0.48 on the ID and OOD test sets respectively (Fig S2, Table 1 and Table S5, interface residue prediction at coarse-grained (CG) resolution 1). It performs better than a random baseline evaluated on the OOD test set (Table 1)."], "subsections": []}, {"title": "Coarse-graining improves the performance", "content": ["To further improve the performance of our model, we predict coarse-grained contact maps and interface residues (Fig 3a, Fig 3b). Coarse-graining has been widely used for modeling biological systems (Arvindekar et al., 2024; Noid, 2013). Coarse-graining over a set of residue pairs in a contact map or a set of residues in the interfaces further reduces the sparsity and class imbalance, and mitigates the problems posed by the multivalency of IDR interactions (Fig 3a, Fig 3b, Table S6). Notably, we coarse-grain over embeddings from pLMs, which provide context-aware, abstract representations of the sequence. Further, coarse-graining along the sequence is well-suited for the modular and patch-driven nature of IDR interactions, that are mediated by charge patterning, attractive sticker regions, conserved short linear motifs (SLiMs) and molecular recognition features (MoRFs), and/or less conserved low-complexity regions (LCRs) (Christensen et al., 2019; Cumberworth et al., 2013; Holehouse & Kragelund, 2024; Mohan et al., 2006).", "We train Disobind to predict coarse-grained contact maps and coarse-grained interface residues at resolutions of 5 and 10 contiguous residues along the backbone. The coarse-grained embeddings are obtained by averaging the input embeddings (Elnaggar et al., 2022). As expected, coarse-graining further improves the performance of Disobind on the ID and OOD test sets for both the contact map and interface residue prediction (Fig S2, Table 1, Table S5). For all cases, Disobind performs better than a random baseline on the OOD test set."], "subsections": []}]}, {"title": "Comparison to AlphaFold2 and AlphaFold3", "content": ["We further compare the performance of Disobind with the state-of-the-art methods AF2 and AF3 on the OOD test set. For evaluating AF2 and AF3 outputs, we mask out interactions predicted at low confidence based on the interface-predicted template matching (ipTM) score, predicted local distance difference test (pLDDT), and predicted aligned error (PAE) metrics (see Methods)."], "subsections": [{"title": "Using different ipTM cutoffs for AF2 and AF3", "content": ["The ipTM score is a metric provided by AF2 and AF3 to assess the confidence in the predicted interfaces in a complex (Abramson et al., 2024; Evans et al., 2022). A prediction with an ipTM score greater than 0.75 is considered highly confident, while an ipTM lower than 0.6 indicates a likely failed prediction (O’Reilly et al., 2023; Yin et al., 2022). We first evaluate the performance of AF2 and AF3 with an ipTM cutoff of 0.75. Disobind performs better than AF2 and AF3 for all tasks, i.e., contact map and interface residue predictions across different coarse-grained resolutions (Table 1, ipTM = 0.75). Disordered residues, however, are known to impact the pTM and ipTM metrics negatively (Abramson et al., 2024; Magana & Kovalevskiy, 2024). Considering this, we relaxed the ipTM cutoff to 0.4 to assess the performance of AF2 and AF3, keeping the per-residue metric, i.e. pLDDT and PAE cutoffs as is, guided by recent benchmarks and the AF3 guide (Abramson et al., 2024; Omidi et al., 2024). Here also, Disobind performs better than both AF2 and AF3 for both contact map and interface residue prediction (Table 1, ipTM = 0.4). AF3 performs worse than both Disobind and AF2. Furthermore, this trend remains more or less the same even when we do not apply an ipTM cutoff (Table 1, ipTM = 0.0). Additionally, we show that predictions derived from the structure and the contact probabilities are largely similar, indicating a good agreement between the predicted structure and the contact probabilities (Table S7, see supplementary text)."], "subsections": []}, {"title": "AF2 performs better than AF3", "content": ["Similar to Disobind, coarse-graining improves the performance of both AF2 and AF3 for both contact map and interface predictions (Table 1). Across all tasks, AF2 performs better than AF3. Moreover, AF3 predictions have lower confidence as measured by their ipTM scores (Fig S3). For 35 of 52 OOD test set entries, the ipTM score of the best AF2 model was greater than that of the best AF3 model. Overall, only 8 of 52 predictions from AF2 and 2 of 52 predictions from AF3 had a high confidence prediction, i.e., ipTM score higher than 0.75."], "subsections": []}, {"title": "Combining Disobind and AlphaFold2 predictions", "content": ["Next, we sought to combine the predictions from Disobind and AF2, to see if we could further improve over either method. We combine the predictions from Disobind and AF2 using a logical OR operation and evaluate the performance on the OOD test set. The combined model, “Disobind+AF2”, performs better than either of the methods (Table 2)."], "subsections": []}, {"title": "Performance by residue types", "content": ["Next, we assess the performance of Disobind, AF2, and Disobind+AF2 by residue type to examine which regions each method tends to perform better on (See Methods). Disobind outperforms AF2 on both the contact map and interface residue prediction in disordered regions, whereas AF2 performs slightly better than Disobind for ordered regions (Table S8). Importantly, the combined model Disobind+AF2 performs better for both disordered and ordered regions, likely because the two models learn complementary information by focusing on different regions.", "Further, Disobind predictions are more accurate than AF2 for disorder-promoting and hydrophobic, whereas AF2 and Disobind perform similarly for aromatic and polar residues (Table S9). Disobind is more accurate than AF2 in predicting contacts and interface residues involving LIPs (Table S9). The combined model is better across all residue types (Table S9)."], "subsections": []}]}, {"title": "Comparison with interface predictors for IDRs", "content": ["We compared Disobind with partner-independent interface predictors for IDRs, including AIUPred, MORFchibi, and DeepDISOBind, which are comparable to the state-of-the-art in the Critical Assessments of protein Intrinsic Disorder prediction (CAID2) (Conte et al., 2023; Erdős et al., 2025; Malhis & Gsponer, 2025; F. Zhang et al., 2022). For a fair comparison, we evaluated interface residue predictions for only the IDR protein in the OOD set (See Methods). Disobind outperforms AIUPred, MORFchibi, and DeepDISOBind at predicting interface residues for the IDR (Table 3)."], "subsections": [{"title": "Input Embeddings", "content": ["Next, we investigate the effect of using embeddings from various pLMs and the different embedding types on the performance of Disobind."], "subsections": []}, {"title": "Protein language models allow for a shallow architecture", "content": ["Protein language models (pLMs) provide context-aware representations of protein sequences that facilitate training models for downstream tasks like contact map prediction via transfer learning. They allow using a shallow downstream architecture when training with limited data (Bepler & Berger, 2021; Jahn et al., 2024). Several pLMs have been developed including the ESM series of models (Rives et al., 2021), ProstT5 (Heinzinger et al., 2024), ProtT5 (Elnaggar et al., 2022), ProtBERT (Elnaggar et al., 2022), and ProSE (Bepler & Berger, 2019).", "The pLMs differ in the model architecture, the training datasets, and the training regime. These can affect their ability to learn contextual representations for protein sequences. ProSE uses a bidirectional LSTM trained on UniRef protein sequences and structures with a multitask loss combining similarity and contact prediction objectives (Bepler & Berger, 2019). The other pLMs use variants of the transformer architecture. ProtBERT uses the Bidirectional Encoder Representations from Transformers (BERT) architecture, an encoder-only model, trained on the UniRef100 and BFD100 datasets. It is trained to reconstruct corrupted tokens given the context of the neighbouring tokens (Elnaggar et al., 2022). ProtT5 uses the Text-to-Text Transfer Transformer (T5) architecture, an encoder-decoder model, trained on the UniRef50 and BFD100 datasets. In contrast to the BERT-style training objective, ProtT5 is trained to reconstruct spans of corrupted tokens (Elnaggar et al., 2022; Raffel et al., 2023). ProstT5 is a newer version of ProtT5, fine-tuned with AlphaFold2 predicted protein structures (Heinzinger et al., 2023).", "Disobind uses embeddings from ProtT5, one of the best performing pLMs in our comparison (Table S10). The T5-based models, ProtT5 and ProstT5, perform similarly, but outperform the BERT-styled ProtBERT and LSTM-based ProSE models. Due to limitations in memory, we could not generate ESM2-650M-global embeddings for the entire training set. Hence, we do not include ESM in our comparison."], "subsections": []}, {"title": "Global embeddings perform better than local embeddings", "content": ["We then evaluate whether models trained with global sequence embeddings perform better than those with local embeddings. Local embeddings are obtained by providing the sequence of the fragment as input to the pLM. Global embeddings, in contrast, are obtained by providing the entire protein sequence corresponding to the fragment as input to the pLM, followed by extracting the embeddings corresponding to the fragment (Fig S4). Models with global embeddings perform better than those with local embeddings across all the pLMs we tested (Table S10). Using the global embeddings provides the context of the complete protein sequence corresponding to the fragment, plausibly resulting in better performance. This context may contain information relevant to binding: for example, the binding of IDRs is known to be affected by the flanking regions (Olsen et al., 2017)."], "subsections": []}]}, {"title": "Case studies", "content": ["IDRs enable dynamic interactions, molecular recognition, and functional adaptability, making them a crucial component of various signaling and regulatory pathways. We used Disobind+AF2 for predicting interface residues for several IDR complexes (Fig 4, Fig S5). Here, we discuss in detail two cases in which the IDR becomes ordered upon binding (DOR) and two cases where the IDR remains disordered (DDR). A larger set of NMR examples of mostly DDR cases from MobiDB, PEDS, BMRB, and LLPSdb are also shown separately (Fig S5) (Ghafouri et al., 2024; Hoch et al., 2023; Li et al., 2020; Piovesan et al., 2024).", "The cellular prion protein (PrPC) plays an important role in several cellular processes, including synapse formation, regulating circadian rhythm, and maintaining ion homeostasis (Kraus et al., 2021; Sigurdson et al., 2019). It is known to convert to an infectious PrPSc form, which forms amyloid fibrils associated with several diseases, including Creutzfeldt-Jakob disease and bovine spongiform encephalopathy (Kraus et al., 2021; Sigurdson et al., 2019). PrPC has an N-terminal disordered region and a primarily α-helical C-terminal region, whereas the conversion to PrPSc involves significant structural changes, resulting in a β-sheet-rich structure. Disobind+AF2 could accurately predict the interface residues in the β-sheet-rich PrPSc dimer, achieving an F1-score of 0.71 (Fig 4a).", "Chemokines are small secretary molecules that play several important roles, including immune cell trafficking, wound healing, and lymphoid cell development (Rossi & Zlotnik, 2000; Sepuru et al., 2020). Dysregulation of chemokine release is associated with several diseases, including chronic pancreatitis, inflammatory bowel disease, and psoriasis (Rossi & Zlotnik, 2000; Sepuru et al., 2020). Chemokines bind to chemokine receptors that belong to the G-protein coupled receptor (GPCR) family. Tha latter contains an N-terminal disordered region that becomes structured and adopts an extended conformation upon binding. Disobind + AF2 accurately predicts the interface residues for the chemokine Interleukin-8 binding to the N-terminus of its cognate receptor, CXCR1 with an F1 score of 0.73 (Fig 4b).", "The death domain-associated protein 6 (Daxx) is a transcriptional coregulator that plays a key role in apoptosis (Chang et al., 2011; Salomoni & Khelifi, 2006). It contains a C-terminal disordered region that attains an extended conformation upon binding to its partner SUMO-1 (Chang et al., 2011). This interaction is crucial for its localization to nuclear PML bodies, the absence of which leads to disruption of PML nuclear bodies as seen in acute myelocytic leukaemia (Salomoni & Khelifi, 2006). Disobind+AF2 could accurately capture interface residues in this region, achieving an F1-score of 0.81 (Fig 4c).", "UV-stimulated scaffold protein A (UVSSA) is a crucial component for transcription-coupled nucleotide excision repair (TC-NER) pathway (Okuda et al., 2017; Schwertman et al., 2012). Disruption of the TC-NER pathway is associated with various diseases, including xeroderma pigmentosa and UV-sensitive syndrome (Okuda et al., 2017). In complex with USP7, UVSSA is involved in processing the stalled RNA pol II and recruiting TFIIH. A short acidic region in UVSSA is an IDR and interacts with the N-terminal PH domain in the p62 subunit of the TFIIH complex, which is crucial for the recruitment of TFIIH to the TCR site (Okuda et al., 2017). Disobind+AF2 was able to capture the interface residues in this case with an F1-score of 0.71 (Fig 4d). In summary, across all the cases, our combined model achieves high accuracy in predicting interface residues for IDR complexes."], "subsections": []}]}, {"title": "Discussion", "content": ["Here, we discuss about the uses and limitations of Disobind, challenges involved, the performance of AlphaFold, and future directions."], "subsections": [{"title": "Uses and limitations", "content": ["Predictions from Disobind+AF2 may be used to improve the localization of IDRs in integrative models of large assemblies, our primary motivation for developing the method. Macromolecular assemblies contain significant portions of IDRs, for example, the Fg Nups in the nuclear pore complex, the MBD3-IDR in the nucleosome remodeling and deacetylase (NuRD) complex, and the N-terminus of Plakophilin1 in the desmosome (Akey et al., 2022; Arvindekar et al., 2022; Pasani et al., 2024). These regions typically lack data for structural modeling, resulting in integrative models of poor precision (Arvindekar et al., 2022; Pasani et al., 2024). The predictions from Disobind+AF2 can be used in integrative modeling methods such as IMP, HADDOCK, and Assembline as inter-protein distance restraints (Dominguez et al., 2003; Rantos et al., 2022; Russel et al., 2012). Even the coarse-grained contact map and interface residue predictions would be useful in such cases to improve the precision of these regions in the integrative model. Further, the predicted contacts can be combined with molecular dynamics (MD) simulations to generate ensembles of IDRs in complexes, providing mechanistic insights into their dynamic behaviour. Additionally, our method can be used to characterize interactions involving IDRs across proteomes. This may aid in identifying new binding motifs for IDRs, potentially linked to their sub-cellular localization or function. Finally, predictions from our method may aid in modulating interactions involving IDRs, for example, by suggesting plausible mutations.", "However, our method also has several limitations. First, it is limited to binary IDR-partner complexes. For complexes formed by three or more subunits, binary predictions from Disobind would need to be combined. Second, it assumes that the IDR and its partner are known to bind. Disobind cannot reliably distinguish binders from non-binders (see Supplementary text). Third, the accuracy of the predictions depends on the ability of the pLM to provide accurate representations of the IDR and its partner. Finally, our method cannot be used to assess the effects of post-translational modifications as the pLM used does not distinguish post-translationally modified amino acids."], "subsections": []}, {"title": "Challenges", "content": ["Predicting contact maps and interface residues for IDRs in a complex with a partner is challenging. First, a limited number of experimental structures are available for IDRs in complexes (Jahn et al., 2024; J. Zhang et al., 2024). Second, IDRs adopt an ensemble of conformations, and the available structures may only partially capture this conformational diversity. Third, inter-protein contact maps are typically sparse, with only a few residues forming contacts. Although not sufficient, we gather all available structures for IDRs in complexes. With these, we create a dataset of merged binary complexes for training Disobind. Using merged binary complexes helps overcome the issue of sparsity associated with training on an ensemble of contact maps. Predicting interface residues instead of contacts and using coarse-graining helps overcome the challenges associated with the multivalency of IDRs and the sparsity of inter-protein contact maps (Fig 3a)."], "subsections": []}, {"title": "Performance of AF2 and AF3", "content": ["Several studies indicate that the low-confidence regions in AF2 predictions overlap with the presence of disordered regions, although these low-confidence regions cannot be considered as a representative conformation for IDRs (Alderson et al., 2022; Escobedo et al., 2023; Ruff & Pappu, 2021). Despite the success of AF2 and AF3 for ordered proteins, their performance on IDR complexes remains limited, as shown in benchmarks (Alderson et al., 2022; Bret et al., 2024; C. Y. Lee et al., 2024; Omidi et al., 2024). This could potentially be due to several reasons. First, IDRs show low sequence conservation, resulting in poor quality MSAs (Holehouse & Kragelund, 2024). Second, IDRs are best described by an ensemble of conformations which capture their dynamic behaviour. AF2 and AF3, however, predict a single best structure, which may not be representative of IDRs (Ruff & Pappu, 2021). Further, the diversity of binding modes of IDRs in complexes, e.g., DOR and DDR, poses a significant challenge. In particular, AlphaFold predictions are less accurate for fuzzy complexes or DDR complexes, compared to DOR complexes (Alderson et al., 2022; Omidi et al., 2024). Third, the confidence metrics of AF2 and AF3 may not reliably reflect structural accuracy (Elofsson, 2025; Guan & Keating, 2025). Particularly, the presence of IDRs is known to negatively impact the confidence metrics (Dunbrack, 2025; Magana & Kovalevskiy, 2024; Omidi et al., 2024; Varga et al., 2024).", "On our OOD test set, most of the predictions from AF2 and AF3 had low confidence, with an ipTM score lower than 0.75. Notably, compared to AF2, AF3 predictions were less accurate and had lower confidence. It is possible that the confidence cutoffs used for ordered proteins do not apply to IDRs and predictions involving the latter might require different cutoffs. A recent benchmark suggested a lower ipTM score of 0.4 for assessing interfaces involving IDRs (Omidi et al., 2024). Further, the per-residue metrics such as pLDDT and PAE may be more relevant for assessing AF2 and AF3 predictions than the global metrics such as the ipTM and pTM, as is also suggested by a recent benchmark and the AF3 guide (Abramson et al., 2024; Omidi et al., 2024). Thus, AlphaFold confidence metrics might need to be interpreted differently for IDRs, and this requires rigorous benchmarking (Chakravarty et al., 2025; Kim et al., 2024; Varga et al., 2024).", "AF2 has been shown to successfully predict the structures of protein-peptide complexes and domain-motif complexes, although the results could be sensitive to the sequence fragment size, fragment delimitation, and the MSA alignment mode (Bret et al., 2024; B.-G. Lee et al., 2020). Some studies show that MSA subsampling (del Alamo et al., 2022), clustering the sequences in the MSA (Wayment-Steele et al., 2024), and using sliding fragments as input to AF2 (Bret et al., 2024) result in better predictions and can be used to predict multiple conformations. In our comparison, we did not explore these strategies, though they may improve the model predictions."], "subsections": []}, {"title": "Future Directions", "content": ["One of the major roadblocks in training methods such as Disobind is the lack of data. More experimental structures of IDRs in complexes would be valuable (Jahn et al., 2024; J. Zhang et al., 2024). Whereas IDR ensembles derived from MD can be used, generating MD ensembles for IDR complexes is computationally expensive and challenging, and the existing databases like PED contain very few such ensembles for IDR complexes (Ghafouri et al., 2024; Majila et al., 2024). Alternatively, deep generative models can be used to generate ensembles for IDRs in complexes (Janson & Feig, 2024; Majila et al., 2024; Mansoor et al., 2024). However, the current methods are limited to generating ensembles for monomers.", "Disobind and similar methods can be further enhanced by improving the existing pLMs to provide better representations for IDRs plausibly by incorporating physical priors and/or structural information (Majila et al., 2024; Rogers et al., 2023; Wang et al., 2024). Additionally, these methods could be extended to predict protein-protein interactions (PPIs) involving IDRs, and aid in the design of IDR binders. The behaviour of IDRs within cells remains largely unexplored. Methods like Disobind are expected to facilitate an improved understanding of the interactions, function, and modulation of IDRs."], "subsections": []}]}, {"title": "STAR★Methods", "content": [], "subsections": [{"title": "Experimental model and study participant details", "content": ["Not applicable"], "subsections": []}, {"title": "Method details", "content": [], "subsections": [{"title": "Creating the dataset for Disobind", "content": [], "subsections": [{"title": "Gathering PDB structures of IDRs in complexes", "content": ["First, we gathered the available PDB entries for protein complexes containing IDRs from databases on structures of disordered proteins: DIBS (Schad et al., 2018), MFIB (Fichó et al., 2017), and FuzDB (Miskei et al., 2017) (Fig 1). We also included PDB entries from the PDBtot and PDBcdr datasets (Miskei et al., 2020). PDBtot consists of IDRs that either undergo a DOR or DDR transition, whereas PDBcdr consists of IDRs that undergo both DOR and DDR transitions depending upon the partner. We considered all PDB entries associated with each database. This resulted in a set of 4252 PDB entries. PEDS is another source of conformational ensembles of IDRs (Ghafouri et al., 2024). However, PEDS entries were not used: several PED entries were already incorporated into our pipeline and the remaining entries correspond to monomers.", "We then supplemented this set by querying the PDB for additional complexes containing IDRs. For this, we first obtained Uniprot identifiers (IDs) of proteins containing IDRs from DisProt (Aspromonte et al., 2024), IDEAL (Fukuchi et al., 2014), and MobiDB (Piovesan et al., 2024) (Fig 1). Specifically, from Disprot, we obtained all UniProt IDs corresponding to each DisProt entry. From IDEAL, we obtained UniProt IDs for entries annotated as “verified ProS”. These sequences are experimentally verified to be disordered in isolation and ordered upon binding. From MobiDB, we obtained UniProt IDs for entries annotated “curated-disorder-priority”, “homology-disorder-priority”, “curated-lip-priority”, and “homology-lip-priority”. These correspond to sequences with curated experimental evidence for disorder (“curated-disorder”), their homologs (“homology-disorder”), sequences with experimental evidence for disorder with binding motifs such as SLIMs (“curated-lip”), and their homologs (“homology-lip”), respectively. Homology-based annotations were incorporated due to the lack of sufficient experimentally curated data available for IDRs. Importantly, the annotations were based on the source with the highest confidence (“priority”), and annotations based on indirect sources of evidence or predictions were not considered. Choosing the highest confidence annotation may reduce false positive “disorder” or “lip” annotations that may arise from homology-based evidence.", "A total of 357734 unique Uniprot IDs across MobiDB, Disprot, and IDEAL were obtained. The PDB REST API was queried to obtain all the PDB entries associated with a given Uniprot ID, resulting in 48534 PDB entries (Table S1) (Rose et al., 2021). Combining these with the PDB entries obtained in the previous step, we obtained 50294 unique PDB entries. As an aside, we note that MobiDB now provides the PDB IDs for IDR-containing complexes, however, this functionality was not available when this project started."], "subsections": []}, {"title": "Defining binary complexes containing IDRs", "content": ["The above PDB entries were downloaded using the Python requests library and mapped to the sequences in UniProt using the SIFTS (Structure Integration with Function, Taxonomy, and Sequence) tool (Velankar et al., 2013). Entries with an obsolete PDB ID and those lacking a SIFTS mapping were removed, and deprecated PDB IDs were replaced with superseding PDB IDs. Further, entities corresponding to chimeric proteins or non-protein molecules, those having an obsolete UniProt ID, and those containing more than 10000 residues were removed.", "Next, for each PDB entry, we obtained sequence fragments by removing missing residues from the sequence of each chain. The length of these fragments was kept to between 20 and 100 residues. Fragments longer than 100 residues were further divided due to constraints on GPU memory, that limited the scalability of the model architecture. We then identified the disordered residues in these fragments (Fig 1). This was achieved by cross-referencing the UniProt mapping of the fragment with the disorder annotations from DisProt, IDEAL, and MobiDB as obtained above; a residue was considered disordered if it was annotated as such in any of these three databases. Fragments comprising at least 20% disordered residues were considered as IDR fragments whereas the others were considered as non-IDR fragments. This cutoff balances an over-representation of ordered residues while ensuring enough training data. This is also consistent with proteome-wide analyses which indicate that proteins with IDRs have 13–28% disordered residues. Specifically, annotations based on experiments indicate a 17% disorder fraction (Piovesan et al., 2024).", "Subsequently, a set of binary complexes was constructed from each PDB entry. Each binary complex consisted of an IDR fragment paired with another IDR or non-IDR fragment. Binary complexes comprising solely of intra-chain fragments or non-IDR fragments were excluded (Fig 1). This resulted in a total of 2369712 binary complexes."], "subsections": []}, {"title": "Creating merged binary complexes", "content": ["We then created contact maps from these binary complexes. A contact map is a binary matrix of zeros (non-contacts) and ones (contacts); two residues are in contact if the distance between their Cα atoms is less than 8 Å. We merged the contact maps across different binary complexes of a sequence fragment pair, to create “merged contact maps” (Fig 1). This allows us to account for contacts from all available complexes. Further, a merged contact map is less sparse and therefore easier to learn compared to an ensemble of contact maps. It also mitigates the bias caused by having different numbers of structures for different sequence fragment pairs.", "We first grouped the binary complexes across all the PDB entries by the UniProt ID of the constituent proteins in the sequence fragment pair, resulting in 14599 UniProt ID pairs/groups. For computational efficiency, all binary complexes with no contacts were eliminated. The remaining binary complexes in a group were sorted based on the UniProt start residue positions of the protein fragments in the pair and further partitioned into sets. Each set comprised binary complexes containing overlapping sequences (more than one residue overlap) for both fragments such that the sequence length for each fragment, merged across the overlapping sequences, does not exceed the maximum fragment length (100 residues). Contact maps for all binary complexes in a set were aggregated using a logical OR operation to form “merged contact maps”. A sequence fragment pair with the corresponding merged contact map comprises a merged binary complex. We obtained 24442 merged binary complexes. Finally, we filtered merged binary complexes for which the contact density was very high (>5%) or very low (<0.5%), resulting in 6018 merged binary complexes (Fig 1)."], "subsections": []}, {"title": "Creating the training set and OOD test set", "content": ["We created an out-of-distribution (OOD) test set that is sequence non-redundant with our training dataset and the PDB70 dataset used for training AlphaFold. We first removed merged binary complexes whose sequence fragments were derived from a PDB chain in PDB70. We then obtained a non-redundant set of sequences by clustering our dataset of sequence fragments and the PDB70 cluster representatives using MMSeqs2 (Steinegger & Söding, 2017). MMSeqs2 was used at a 20% sequence identity threshold and cluster mode 1. Since the PDB70 dataset is already clustered at 40% sequence similarity, we consider only the representative cluster members for sequence identity comparisons.", "For the OOD test set, we selected clusters without any chain from PDB70 that are singleton (only one cluster member i.e. self) or doublet (two member sequences, both belonging to the same merged binary complex) clusters. This resulted in 53 merged binary complexes for the OOD test set. We removed an OOD test set entry for which AF2 predictions could not be obtained, resulting in 52 OOD test set complexes. The remaining dataset was split into a training (train), development (dev), and an in-distribution (ID) test set in the ratio 0.9:0.05:0.05, resulting in 5362 train, 298 dev, and 297 ID test set merged complexes."], "subsections": []}]}]}, {"title": "Disobind Model Architecture and Training", "content": [], "subsections": [{"title": "Notations", "content": ["Here, we define the notation used in this paper.", "RA×B×C×D: 4D tensor with the axes i, j, k, l corresponding to the four dimensions of size A, B, C, D respectively", "N: batch size used for training.", "L1, L2: lengths of protein sequence fragments 1 and 2 indicating the number of residues in each protein.", "C, C1, C2, C3: the sizes of embedding or feature dimensions.", "E1, E2: embeddings for protein sequence fragments 1 and 2.", "O: interaction tensor obtained post-concatenation.", "Ored: output from the multi-layer perceptron block, reduced from the interaction tensor.", "I1, I2 interface tensors for proteins 1 and 2.", "I: concatenated interface tensors.", "Linear: applies a linear transformation, Linear(X)=XWT+b.", "ELU: exponential linear unit, ELU(x)=maxx,αex−1.", "Sigmoid(x):11+e−x.", "Dropout: regularization technique that randomly sets weights to 0.", "LNorm: layer normalization.", "Concat([A, B]i): concatenate tensors A, B along the axis i.", "Aijkl→ikjl: permute tensor A along the specified axes."], "subsections": []}, {"title": "Inputs and outputs for training", "content": ["We train six separate models, corresponding to six prediction tasks. These correspond to two types of predictions: a) inter-protein contact map and b) interface residue predictions at three coarse-grained (CG) resolutions (1, 5, and 10 sequence-contiguous residues). Coarse-graining at resolution 1 is equivalent to no coarse-graining, i.e., residue-level predictions. The input to the model is the embedding for the two sequence fragments from the ProtT5 pLM E1∈RL1×C,E2∈RL2×C (Elnaggar et al., 2022). The outputs for the two types of predictions respectively are a) a binary contact map, ContactMap∈RL1/k×L2/k and b) a binary interface residue vector, Interface∈RL1/k+L2/k at a particular CG resolution, where k∈[1, 5,10] represents the CG resolution.", "For coarse-grained predictions, the input embeddings were coarse-grained using an AvgPool1d whereas the corresponding contact maps were coarse-grained using a MaxPool2d; the kernel size and stride were set to the CG resolution in both cases.", "The target vectors for interface residue predictions were obtained by reducing the rows and columns of the corresponding contact maps by an OR operation. The input embeddings, output contact maps, and interface residue vectors were zero-padded to the maximum length (100 residues)."], "subsections": []}, {"title": "Model architecture", "content": ["Disobind comprises a projection block, an interface block, a multi-layer perceptron (MLP) block, and an output block (Fig 2). The models for contact map and interface residue prediction employ a contact map block and interface block respectively."], "subsections": []}, {"title": "Projection block", "content": ["The input embeddings E1, E2 are projected to a lower dimension, C1 using a linear layer.\n\n(1)\nZ1=Dropout(ELU(Linear(E1)));Z1∈RN×L1×C1\n\n\n(2)\nZ2=Dropout(ELU(Linear(E2)));Z2∈RN×L2×C1"], "subsections": []}, {"title": "Interaction block", "content": ["The model captures complementarity between the projected embeddings Z1,Z2 by concatenating the outer product and the absolute value of the outer difference between the projected embeddings along the feature dimension.\n\n(3)\nO=Concat([OuterProduct(Z1,Z2),∣OuterDifference(Z1,Z2)∣]l);O∈RN×L1×L2×C2"], "subsections": []}, {"title": "Interface block", "content": ["To capture the interface residues for both proteins I, we compute row-wise and column-wise averages followed by concatenating along the length.\n\n(4)\nI1=Mean(Oijkl→iljk);I1∈RN×C2×L1\n\n\n(5)\nI2=Mean(Oijkl→ilkj);I2∈RN×C2×L2\n\n\n(6)\nI=Concat([I1,I2]k);I2∈RN×C2×(L1+L2)\n\n\n(7)\nI=Iijk→ikj;I∈RN×(L1+L2)×C2"], "subsections": []}, {"title": "Multi-layer perceptron (MLP) block", "content": ["The feature dimension of the interaction tensor O is progressively decreased using a multilayer perceptron (MLP) to obtain Ored.\n\n(8)\nOred=m×{LNorm(ELU(Linear(x)))};Ored∈RN×L1×L2×C3", "We used a 3-layer (m = 3) MLP for contact map prediction whereas it was not used for interface residue prediction (m = 0)."], "subsections": []}, {"title": "Output block", "content": ["It comprises a Linear layer followed by a Sigmoid activation.\n\n(9)\nContactMap=Sigmoid(Linear(Ored));ContactMap∈RN×L1×L2\n\n\n(10)\nInterface=Sigmoid(Linear(I));Interface∈RN×(L1+L2)"], "subsections": []}, {"title": "Training", "content": ["The models were trained using PyTorch version 2.0.1 and CUDA version 11.8, using an NVIDIA A6000 GPU. The model was trained using the Singularity Enhanced (SE) loss function (Si & Yan, 2021), a modified form of the binary cross-entropy (BCE) loss used for class-imbalanced training, with α = 0.9, β = 3 and AdamW (amsgrad variant) optimizer with a weight decay of 0.05. We used an exponential learning rate scheduler for slowly decaying the base learning rate.\n\n(10)\nSELoss=∑(α⋅ylog(y^)+(1−α)⋅(1−y)log((1−y^))⋅(1+y^)β)", "A binary mask was used to exclude the padding from the loss calculation. The train and dev set loss were monitored to check for over/under-fitting (Fig S1)."], "subsections": []}, {"title": "Evaluating predictions", "content": ["The Disobind predicted score was converted to a binary output using a threshold of 0.5. The model performance was evaluated on the train, dev, and ID test set using recall, precision, and F1-score. All metrics were calculated using the torchmetrics library.\n\n(11)\nRecall=TPTP+FNPrecision=TPTP+FPF1score=2⋅Precision⋅RecallPrecision+Recall"], "subsections": []}, {"title": "Hyperparameter tuning and ablations", "content": ["Here, we explore the effect of select hyperparameters on the model performances in the dev and ID test sets. Hyperparameters were tuned for the contact map and interface prediction at CG 1 and minimally modified for the CG 5 and CG 10 models. The final set of hyperparameters is provided (Table S1)."], "subsections": [{"title": "Projection dimension", "content": ["We tuned the projection dimension C1 for the projection block (Table S2)."], "subsections": []}, {"title": "Number of layers in the MLP", "content": ["Linear layers in the MLP were used to either upsample (upsampling layers, US) or downsample (downsampling layers, DS) the feature dimension by a factor of 2. We tuned the number of US (DS) layers. Additionally, we removed the MLP for interface residue prediction (Table S3)."], "subsections": []}, {"title": "α, β for SE loss", "content": ["We tuned the α and β parameters for SE loss; these weigh the contribution of the contact and non-contact elements to the loss (Table S4)."], "subsections": []}, {"title": "pLM embeddings", "content": ["We compared the model performance using embeddings from ProtT5 (C = 1024), ProstT5 (C= 1024), ProSE (C = 6165), and ProtBERT (Bepler & Berger, 2019; Elnaggar et al., 2022; Heinzinger et al., 2024)(C = 1024).", "Further, we compared two types of embeddings: global and local embeddings (Fig S4, Table S10). Local embeddings were obtained using the sequence of the fragment as input to the pLM. In contrast, global embeddings were obtained by using the complete UniProt sequence as input to the pLM and extracting fragment embedding from it."], "subsections": []}]}]}, {"title": "Assessment", "content": [], "subsections": [{"title": "Random baseline", "content": ["To compute random baseline predictions, we predicted random contacts and interface residues based on their frequencies in the training set. For the prediction tasks involving coarse-graining, these frequencies were determined from the coarse-grained contact maps and coarse-grained interface residues in the training set."], "subsections": []}, {"title": "AlphaFold predictions on the OOD test set", "content": ["We used a local AF2 installation and the AF3 webserver for obtaining OOD test set predictions (Abramson et al., 2024; Evans et al., 2022). AF2 was run with the default arguments as specified in their GitHub repository. Due to the failure of AF2 to relax several of the OOD test set entries we switch off the Amber relaxation step and consider the unrelaxed models for further evaluation. Additionally, we removed an OOD test set entry for which we were unable to obtain the AF2 prediction. We considered the best models from AF2 (based on the ipTM+pTM score) and AF3 (based on the ranking score) for comparison with Disobind. The ipTM score was used to assess the overall model confidence and the pLDDT and PAE metrics assessed the per-residue confidence. pLDDT scores higher than 70 and PAE values lower than 5 are considered as confident (Edmunds et al., 2024). An ipTM score higher than 0.8 represents a high-confidence prediction whereas an ipTM score lower than 0.6 represents a likely incorrect prediction; predictions with ipTM scores between 0.6-0.8 require further validation (Abramson et al., 2024; Evans et al., 2022; O’Reilly et al., 2023; Yin et al., 2022). Given that only five AF2 predictions and none of the AF3 predictions in the OOD test set have an ipTM score greater than 0.8, we initially applied an ipTM cutoff of 0.75 as recommended in benchmarks on AlphaFold2 (O’Reilly et al., 2023; Yin et al., 2022). Later we relaxed the ipTM cutoffs to 0.4 and 0 (i.e., no ipTM cutoff).", "To compare the AlphaFold outputs to those from our models, we first derive the binary contact maps from the best-predicted model. We then apply binary masks on this output to ignore low-confidence interactions (pLDDT<70 and PAE>5). Both the pLDDT and PAE masks are applied on the CG 1 contact map obtained from AF2 and AF3 before proceeding further. To apply an ipTM cutoff, we zero the contact maps for AlphaFold predictions with an ipTM lower than the cutoff. The binary contact map was zero-padded to the maximum length (100 residues). The interface residues and coarse-grained predictions were derived from the CG 1 contact map as mentioned previously (See Inputs and Outputs for training)."], "subsections": []}, {"title": "Disobind+AF2 predictions", "content": ["The Disobind+AF2 predictions were obtained by combining the corresponding predictions from Disobind and AF2 using a logical OR operation. We ignore low-confidence AF2 predictions having a pLDDT less than 70 and a PAE greater than 5. No ipTM cutoff was used."], "subsections": []}, {"title": "Performance by residue type", "content": ["Given a protein sequence, disordered regions were identified by combining DisProt, IDEAL, and MobiDB annotations as described in Creating dataset for Disobind section. Residues in these regions were annotated as “disordered residues”, the remaining residues in the sequence were considered “ordered”. Linear interacting peptide (LIP) annotations were obtained from MobiDB as described earlier. We categorized residues into disorder-promoting amino acids (R, P, Q, E, G, S, A, K), aromatic amino acids (F, Y, W), hydrophobic amino acids (A, V, L, I, P, M, F, W), and polar amino acids (S, T, C, N, Q, Y, D, E, K, R, H) (Uversky, 2013, 2019). For LIPs, we evaluated contacts between residues in LIPs and any residue on the partner protein. For all other residue types, we evaluated contacts between residues of the same type. To assess the predictions on specific residue types, we applied a binary mask on the predictions and the target to ignore interface residues (contacts) not involving the relevant residues (residue pairs). We only considered confident predictions from AF2 having pLDDT scores higher than 70 and PAE values lower than 5. No ipTM cutoff was used."], "subsections": []}, {"title": "Comparison with interface predictors for IDRs", "content": ["We selected state-of-the-art interface predictors including AIUPred, MORFchibi, and DeepDISOBind for comparison to Disobind on the OOD test set (Erdős et al., 2025; Malhis & Gsponer, 2025; F. Zhang et al., 2022). These methods have been shown to be comparable to the state-of-the-art in CAID2 (Conte et al., 2023). They provide partner-independent interface residue predictions. For comparison to Disobind, we assess their interface residue predictions on the IDR sequence of the binary complexes in the OOD set. AIUPred predictions were obtained using programmatic access as mentioned in their GitHub repository. For MORFchibi and DeepDISOBind, predictions were obtained using their respective webservers, using defaults. DeepDISOBind provides binary interface predictions. AIUPred predictions are converted to binary interfaces using a 0.5 threshold as specified in the paper (Erdős et al., 2025). For MORFchibi, we use a 0.77 threshold for interface residues as recommended; interface residue regions shorter than four residues are eliminated (Malhis & Gsponer, 2025)."], "subsections": []}]}, {"title": "Quantification and statistical analysis", "content": ["Not applicable"], "subsections": []}, {"title": "Additional resources", "content": ["Not applicable"], "subsections": []}]}, {"title": "Supplementary Material", "content": [], "subsections": []}, {"title": "Data and Code availability", "content": ["The data is deposited to Zenodo at https://doi.org/10.5281/zenodo.14504762. The deposition contains all datasets (train, dev, ID, OOD) along with the input sequences and target contact maps, AF2 and AF3 predictions for the OOD test set, and the PDB structures, SIFTS mappings, PDB API files, and Uniprot Sequences used to create the datasets.", "The Disobind model and scripts to create datasets, train the model, use Disobind/Disobind+AF2, and perform analysis are available at GitHub https://github.com/isblab/disobind.", "Any additional information required to reanalyze the data reported in this paper is available from the lead contacts upon request."], "subsections": []}, {"title": "Acknowledgement", "content": ["We thank ISB Lab members Shreyas Arvindekar, Muskaan Jindal, Omkar Golatkar, and Mubashira KP for their useful comments on the manuscript. Molecular graphics images were produced using the UCSF Chimera and UCSF ChimeraX packages from the Resource for Biocomputing, Visualization, and Informatics at the University of California, San Francisco (supported by NIH P41 RR001081, NIH R01-GM129325, and National Institute of Allergy and Infectious Diseases).", "This work has been supported by the following grants: Department of Atomic Energy (DAE) TIFR grant RTI 4006, Department of Science and Technology (DST) SERB grant SPG/2020/000475, and Department of Biotechnology (DBT) grant BT/PR40323/BTIS/137/78/2023 from the Government of India to S.V."], "subsections": []}]}}
{"pmid": "40286477", "meta": {"content": {"abstract": "Biologically traditional methods, such as the Uversky plot, which rely on hydrophobicity and net charge, have inherent limitations in accurately distinguishing intrinsically disordered regions (IDRs) from ordered protein regions. To overcome these constraints, we propose a novel ensemble framework integrating Machine Learning (ML), Deep Neural Networks (DNN), and Quantum Neural Networks (QNN) to enhance IDR classification accuracy. Notably, this study is the first to employ QNNs for IDR classification, leveraging quantum entanglement to model intricate feature interactions. Amino acid sequences were analyzed to extract biophysical features, including charge distribution, hydrophobicity, and structural properties, which served as inputs for the predictive models. ML was utilized for independent feature learning, DNN for hierarchical interaction modeling, and QNN for capturing high-order dependencies. Our meta-model demonstrated an accuracy of 0.85, surpassing individual classifiers and highlighting the importance of buried amino acids and feature interactions between scaled hydrophobicity and large, buried, and charged residues. This study advances computational protein science by demonstrating the applicability of QNNs in bioinformatics and establishing a robust framework for IDR classification.", "keywords": ["Deep neural network", "Intrinsically disordered region", "Machine learning", "Quantum neural network"], "mesh_terms": ["*Neural Networks, Computer", "*Machine Learning", "*Quantum Theory", "Amino Acid Sequence", "*Intrinsically Disordered Proteins/chemistry/classification", "Hydrophobic and Hydrophilic Interactions", "Computational Biology"], "pub_types": ["Journal Article"]}, "contributors": {"medline": {"affiliations": ["Department of Biotechnology, College of Life Sciences and Biotechnology, Korea University, Seoul 02841, Republic of Korea.", "Department of Biotechnology, College of Life Sciences and Biotechnology, Korea University, Seoul 02841, Republic of Korea; Department of Mechanical Engineering, Korea University, Seoul 02841, Republic of Korea. Electronic address: saekomi5@korea.ac.kr."], "auids": [], "full_names": ["Kang, Seok-Jin", "Shin, Hongchul"], "short_names": ["Kang SJ", "Shin H"]}, "xml": [{"affiliations": ["Department of Biotechnology, College of Life Sciences and Biotechnology, Korea University, Seoul 02841, Republic of Korea."], "full_name": "Kang, Seok-Jin", "identifiers": [], "short_name": "Kang SJ"}, {"affiliations": ["Department of Biotechnology, College of Life Sciences and Biotechnology, Korea University, Seoul 02841, Republic of Korea; Department of Mechanical Engineering, Korea University, Seoul 02841, Republic of Korea. Electronic address: saekomi5@korea.ac.kr."], "full_name": "Shin, Hongchul", "identifiers": [], "short_name": "Shin H"}]}, "identity": {"doi": "10.1016/j.compbiolchem.2025.108480", "pmid": "40286477", "title": "Amino acid sequence-based IDR classification using ensemble machine learning and quantum neural networks."}, "links": {"cites": [], "entrez": {}, "external": [{"attribute": "subscription/membership/fee required", "category": "Full Text Sources", "linkname": "", "provider": "Elsevier Science", "url": "https://linkinghub.elsevier.com/retrieve/pii/S1476-9271(25)00140-9"}], "pmc": [], "refs": [], "review": ["40286477", "36959451", "39093570", "34418423", "36851914", "30657889", "36832654", "34390736", "33248138", "35356546", "31441158"], "similar": ["40286477", "32535960", "40024047", "34830151", "38275091", "36959451", "30866736", "35767567", "32019410", "39433833", "37498545", "35469832", "36272675", "38781245", "39446390", "40131954", "35724478", "39093570", "33810176", "33616531", "34905768", "34836494", "37674132", "34418423", "26555596", "31233120", "29791766", "35094056", "38987470", "37883249", "39383002", "30878482", "33076213", "38048895", "32696354", "38808365", "34487138", "39612690", "36851914", "40270304", "30657889", "38242678", "35325031", "30135358", "36832654", "39593847", "34390736", "38941236", "40054813", "36908253", "27587688", "37487411", "28365882", "35922751", "27807972", "36291695", "21110985", "32823616", "31724233", "38459060", "33248138", "38701796", "32173600", "35356546", "31881961", "40238838", "31441158", "39236563", "38252962", "39539017", "33139563", "36215254", "32722039", "38297184", "37437364", "32352283", "40096220", "27282356", "40278669", "32696395", "35594093", "36210722", "30071670", "31756456", "31999450", "39907269", "26277609", "34992646", "37730717", "32808039", "32896633", "29734867", "35016607", "30939415", "35831369", "34290286", "33895950", "33267376", "39862160", "27140628"], "text_mined": []}, "metadata": {"entrez_date": "2025/04/27 03:14", "fetched_at": "2026-05-02 18:19:10"}, "source": {"journal_abbrev": ["Comput Biol Chem"], "journal_title": ["Computational biology and chemistry"], "pub_date": "2025 Oct", "pub_types": ["Journal Article"], "pub_year": "2025"}}, "content": {}}
{"pmid": "41378882", "meta": {"content": {"abstract": "A ubiquitous and reversible phosphorylation is important for molecular signaling cascades, regulated by the transient interaction of protein kinases. The coupled folding and phosphorylation determining substrate specificity re-calibrates the interactive environment of intrinsically disordered regions (IDRs). There are over 50 computational methods for predicting IDRs in the proteome, yet achieving an accurate depiction remains an ongoing challenge. In this study, we present a standardized and kinase-centric approach for IDR prediction within the human kinome, employing a long short-term memory deep learning framework that achieves a high predictive performance (AUC = 0.97). The web server is now publicly accessible at: https://ciods.in/kindisorder. Our workflow begins with proteome-wide IDR prediction and proceeds with the categorization of short and long IDR segments, followed by an in-depth analysis of their distribution relative to the kinase domain regulatory core. We evaluated the conservation of these IDRs across all 137 human kinase families, computing a trend-setting conservation index to identify both conserved and variable disorder patterns. Through this framework, we uncovered 1039 functional disorder region hotspots that correlate with dynamic conformational shifts, phosphorylation sites, functional motif enrichment, and mutation impact embedded within IDRs. To further validate their regulatory significance, we conducted biophysical profiling of conserved and variable IDRs. Finally, we developed a structural integrity framework to link these IDRs to their influence on intrinsic signaling cascades and substrate specificity. This study offers a comprehensive functional characterization of IDRs in the human kinome, providing a valuable resource for exploring kinase regulation and opportunities in drug repurposing.", "keywords": ["IDR conformation map", "functional disorder hotspots", "human kinome", "long short-term memory"], "mesh_terms": ["Humans", "*Intrinsically Disordered Proteins/chemistry/metabolism/genetics", "*Protein Kinases/chemistry/metabolism/genetics", "Phosphorylation", "Deep Learning", "Computational Biology/methods", "*Proteome/metabolism/chemistry"], "pub_types": ["Journal Article"]}, "contributors": {"medline": {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Department of Zoology, College of Science, King Saud University, P. O. Box 2455, Riyadh Province, Riyadh 11451, Kingdom of Saudi Arabia.", "School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India.", "School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Institute for Regeneration and Repair, University of Edinburgh, Edinburgh EH16 4UU, Scotland.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India.", "School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India.", "School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India.", "Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "auids": ["ORCID: 0000-0003-2319-121X"], "full_names": ["Thomas, Sonet Daniel", "Rajan, Aparna", "Mahin, Althaf", "Ahmed, Mukthar", "Pavithra, S", "Vignesh, U", "Joy, Naveen", "John, Levin", "Varghese, Lijin", "Sambreena, Alimath", "Codi, Jalaluddin Akbar Kandel", "Prasad, Thottethodi Subrahmanya Keshava", "Vijayakumar, Manavalan", "Geetha, S", "Parvathi, R", "Ganesan, R", "Raju, Rajesh"], "short_names": ["Thomas SD", "Rajan A", "Mahin A", "Ahmed M", "Pavithra S", "Vignesh U", "Joy N", "John L", "Varghese L", "Sambreena A", "Codi JAK", "Prasad TSK", "Vijayakumar M", "Geetha S", "Parvathi R", "Ganesan R", "Raju R"]}, "xml": [{"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Thomas, Sonet Daniel", "identifiers": [], "short_name": "Thomas SD"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Rajan, Aparna", "identifiers": [], "short_name": "Rajan A"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Mahin, Althaf", "identifiers": [], "short_name": "Mahin A"}, {"affiliations": ["Department of Zoology, College of Science, King Saud University, P. O. Box 2455, Riyadh Province, Riyadh 11451, Kingdom of Saudi Arabia."], "full_name": "Ahmed, Mukthar", "identifiers": [], "short_name": "Ahmed M"}, {"affiliations": ["School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India."], "full_name": "Pavithra, S", "identifiers": [], "short_name": "Pavithra S"}, {"affiliations": ["School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India."], "full_name": "Vignesh, U", "identifiers": [], "short_name": "Vignesh U"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Joy, Naveen", "identifiers": [], "short_name": "Joy N"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India.", "Institute for Regeneration and Repair, University of Edinburgh, Edinburgh EH16 4UU, Scotland."], "full_name": "John, Levin", "identifiers": [], "short_name": "John L"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Varghese, Lijin", "identifiers": [], "short_name": "Varghese L"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Sambreena, Alimath", "identifiers": [], "short_name": "Sambreena A"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Codi, Jalaluddin Akbar Kandel", "identifiers": [], "short_name": "Codi JAK"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Prasad, Thottethodi Subrahmanya Keshava", "identifiers": [], "short_name": "Prasad TSK"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Vijayakumar, Manavalan", "identifiers": [], "short_name": "Vijayakumar M"}, {"affiliations": ["School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India."], "full_name": "Geetha, S", "identifiers": [], "short_name": "Geetha S"}, {"affiliations": ["School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India."], "full_name": "Parvathi, R", "identifiers": [], "short_name": "Parvathi R"}, {"affiliations": ["School of Computer Science and Engineering, Vellore Institute of Technology, Chennai 600127, Tamil Nadu, India."], "full_name": "Ganesan, R", "identifiers": [], "short_name": "Ganesan R"}, {"affiliations": ["Centre for Integrative Omics Data Science (CIODS), Centre for Systems Biology and Molecular Medicine, Yenepoya (Deemed to be University), Deralakatte, Manglore 575018, Karnataka, India."], "full_name": "Raju, Rajesh", "identifiers": ["0000-0003-2319-121X"], "short_name": "Raju R"}]}, "identity": {"doi": "10.1093/bib/bbaf662", "pmid": "41378882", "title": "Impact of intrinsically disordered regions and functional disorder hotspots in the human kinome."}, "links": {"cites": [], "entrez": {}, "external": [{"attribute": "subscription/membership/fee required", "category": "Full Text Sources", "linkname": "", "provider": "Ovid Technologies, Inc.", "url": "https://www.ovid.com/41378882.pmid"}, {"attribute": "free resource", "category": "Full Text Sources", "linkname": "", "provider": "PubMed Central", "url": "https://pmc.ncbi.nlm.nih.gov/articles/pmid/41378882/"}, {"attribute": "subscription/membership/fee required", "category": "Full Text Sources", "linkname": "", "provider": "Silverchair Information Systems", "url": "https://academic.oup.com/bib/article-lookup/doi/10.1093/bib/bbaf662"}], "pmc": ["12696717"], "refs": ["40054813", "39237195", "39095350", "39052211", "38998919", "38747347", "38662765", "38218213", "38166858", "37904585", "37470698", "37138579", "36927031", "36781849", "36631611", "36631581", "36416266", "36008930", "35830914", "35725977", "35177069", "35163518", "35020807", "34290238", "33875885", "33137204", "33079988", "32722039", "32034199", "31425549", "30670635", "30611608", "28056780", "27911701", "26230689", "26207888", "25531225", "25514926", "25486330", "25420233", "25038412", "23203878", "23055912", "22080206", "20971646", "20179338", "20100603", "18366622", "17522630", "17227859", "16939197", "15955779", "15947016", "15310560", "15044227", "14960716", "14604521", "12381310", "8612268", "3291115"], "review": ["41378882", "39093570", "30611608", "34418423", "31441158", "33248138", "40650026", "31521235", "33243935", "39142260", "36959451", "33567297", "38588835", "39205298", "35882686", "38705823", "24532081", "39914051", "23466039", "25752799", "37657994"], "similar": ["41378882", "31724233", "27718363", "40530859", "41337585", "40221640", "38297118", "39093570", "25099472", "40828855", "33866372", "32900925", "30611608", "34418423", "39933697", "30878482", "35094056", "41276815", "31441158", "41033554", "33248138", "29066345", "40133781", "38662765", "40133791", "35767567", "32553192", "40244295", "27062995", "41053908", "27807972", "33191750", "36291695", "35061901", "32503351", "33719450", "40100159", "29550429", "39446390", "38747347", "29945970", "28934129", "40650026", "26699268", "30841624", "36908253", "31521235", "38297184", "32535960", "33616531", "38701796", "33627133", "33243935", "39576583", "31207240", "39539017", "30096183", "38941236", "34626492", "33036302", "34830151", "40944448", "39142260", "36959451", "31175370", "41365991", "33567297", "29763584", "38588835", "34905768", "26149687", "39453744", "33406091", "39205298", "35882686", "38705823", "36210722", "24532081", "40133775", "39914051", "26307970", "33936509", "40574713", "35275927", "36272675", "40403066", "28167781", "34825861", "23466039", "24115198", "34710113", "25752799", "37657994", "30308971", "35883444", "28612216", "39383002", "34020549", "35290420", "31756456"], "text_mined": [{"category": "General", "source": "full_text", "url": "https://ciods.in/kindisorder"}, {"category": "GitHub", "source": "full_text", "url": "https://github.com/naveen-joy-18/Impact-of-Intrinsically-Disordered-Regions-and-Functional-Disorder-Hotspots-in-the-Human-Kinome"}]}, "metadata": {"entrez_date": "2025/12/11 13:04", "fetched_at": "2026-05-02 18:19:10"}, "source": {"journal_abbrev": ["Brief Bioinform"], "journal_title": ["Briefings in bioinformatics"], "pub_date": "2025 Nov 1", "pub_types": ["Journal Article"], "pub_year": "2025"}}, "content": {"title": "Impact of intrinsically disordered regions and functional disorder hotspots in the human kinome", "body": [{"title": "Abstract", "content": ["A ubiquitous and reversible phosphorylation is important for molecular signaling cascades, regulated by the transient interaction of protein kinases. The coupled folding and phosphorylation determining substrate specificity re-calibrates the interactive environment of intrinsically disordered regions (IDRs). There are over 50 computational methods for predicting IDRs in the proteome, yet achieving an accurate depiction remains an ongoing challenge. In this study, we present a standardized and kinase-centric approach for IDR prediction within the human kinome, employing a long short-term memory deep learning framework that achieves a high predictive performance (AUC = 0.97). The web server is now publicly accessible at: https://ciods.in/kindisorder. Our workflow begins with proteome-wide IDR prediction and proceeds with the categorization of short and long IDR segments, followed by an in-depth analysis of their distribution relative to the kinase domain regulatory core. We evaluated the conservation of these IDRs across all 137 human kinase families, computing a trend-setting conservation index to identify both conserved and variable disorder patterns. Through this framework, we uncovered 1039 functional disorder region hotspots that correlate with dynamic conformational shifts, phosphorylation sites, functional motif enrichment, and mutation impact embedded within IDRs. To further validate their regulatory significance, we conducted biophysical profiling of conserved and variable IDRs. Finally, we developed a structural integrity framework to link these IDRs to their influence on intrinsic signaling cascades and substrate specificity. This study offers a comprehensive functional characterization of IDRs in the human kinome, providing a valuable resource for exploring kinase regulation and opportunities in drug repurposing."], "subsections": []}, {"title": "Introduction", "content": ["Protein phosphorylation is a key determinant of cellular signaling and function, with several therapeutic strategies primarily focusing on phospho-signaling cascades [1–4]. Despite the extensive catalog of over 500 known kinases and their phosphorylation attributes, one-third of kinases remain uncharacterized, highlighting the understudied nature of these kinases [5, 6]. The current structural ensemble represents only half of the known kinases, primarily capturing their domains [7]. The predominance of unstable conformations over stable states highlights the dynamic necessity of intrinsically disordered regions (IDRs), which play a crucial role in kinase-substrate interactions [8–10], kinase heterodimer assembly [11], flexible transition between active and inactive states [12, 13], and phosphorylation events [14]. The activation loop in the protein kinase frame exhibits intrinsic flexibility, making it partially disordered.", "Studies estimate that the human proteome contains approximately 132000 putative binding motifs within the IDRs [15, 16]. The short linear motifs (SLiMs) enable IDRs to adopt diverse conformations that complement binding targets. The SLiMs within IDRs are more conserved than those in structured regions [17]. For instance, the intrinsically disordered kinase (IDK) insertion domain in RTK KIT is critical for recruiting downstream signaling proteins, highlighting the functional significance of IDRs [18]. This structural flexibility is particularly vital for signaling proteins, where IDRs flank kinase domains and integrate their activity into complex regulatory networks [19].", "In addition to that, missense mutations affecting IDRs raise a question about their structural integrity, with disease-associated mutations disproportionately enriched in these regions [20, 21]. Specific substitutions, such as R → W, R → C, and E → K, often drive structural transitions that disrupt IDR functionality, underscoring the need for comprehensive studies on IDR mutations in kinase signaling pathways. IDR dysfunction has been implicated in cancer [22] and cardiovascular diseases [23].", "In the ERK-RSK-PDK complex, IDRs facilitate communication between distinct kinase cores, thereby ensuring precise signaling specificity [24]. Similarly, Src family tyrosine kinases contain an N-terminal unique domain, i.e. intrinsically disordered and subject to modulation by lipidation, phosphorylation, and alternative splicing [25]. In the AGC-family kinases (e.g. PKA, RSK), the long C-terminal tail is largely disordered and carries multiple SLiMs, specifically the hydrophobic motif, that dock onto the kinase core and stabilize the active conformation [24, 26].", "Several computational tools have been developed to predict disorder from sequence features, often validated using experimental datasets. Early predictors like DISPHOS [27], IUPred [28], PONDR-FIT [29], RONN [30], DISpro [31], DISOPRED [32], and DisEMBL [33] laid the foundation for disorder prediction based on physicochemical properties and statistical patterns. More recent advancements include PredIDR2 [34], flDPnn [35], and DeepCNF-D [36], which leverage deep learning frameworks to improve predictive performance. Additionally, the emergence of protein language models, such as ESM [37] and ProteinBERT [38], has revolutionized the field by capturing contextual residue-level information from massive protein sequence corpora. Community benchmarks (CAID experiment) show that top-performing predictors now rely on deep learning and outshine older biophysical methods, but even these have variable accuracy and run times [39]. Existing tools are primarily designed to predict intrinsic disorder across the entire human proteome, which can introduce considerable noise in both data generation and biological interpretation. To accurately identify relevant human kinome regulatory IDRs, it is essential to incorporate context-specific predictive strategies. Within the human kinome, inconsistent disorder predictions across existing tools may obscure the identification of functionally relevant regulatory IDRs, due to variability in prediction outputs, differences in input parameters (e.g. minimum and maximum region length), underlying algorithmic assumptions, and biases in training data composition. Experimental databases such as DisProt [40] and MobiDB [41] provide valuable annotations of IDRs; however, they still cover only a fraction of the disorder landscape, leaving many regions without experimental validation. Consequently, the aforementioned prediction tools rely heavily on these limited datasets as ground truth references. Despite this, IDRs frequently act as functional hotspots, influencing substrate specificity and kinase activation. Exploring these hotspots can reveal new regulatory motifs and support drug repurposing in kinase signaling pathways."], "subsections": []}, {"title": "Methods", "content": [], "subsections": [{"title": "Data collection and processing", "content": ["The human kinome disorder region predictor was developed, trained, and evaluated using the DisProt 9.7 dataset [42]. The dataset comprises 12790 disordered regions across 3114 proteins from various species. After removing duplicates, 5524 entries correspond to Homo sapiens, from which we extracted 1271 disordered regions specific to human proteins. To create a bias, we retrieved 523 human protein kinome disorder region entries from the UniProt database (Supplementary Table 1)."], "subsections": []}, {"title": "Feature selection and preprocessing", "content": ["From the training data, it is evident that a subset of kinases remains deficient in disorder content annotation. To tackle this, we explored certain physicochemical and thermodynamic parameters and their association with the kinase disorder regions. The selection criteria and features are listed in (Supplementary Table 2 & Supplementary File)."], "subsections": []}, {"title": "Model architecture", "content": ["The training data is a combined version of Disprot and UniProt. The dataset contains protein sequences along with their binary labels indicating 1 for disorder and 0 for order. Given the varying sequence lengths, we applied zero padding to standardize input dimensions, a crucial step for long short-term memory (LSTM) network processing. Each amino acid in a sequence was transformed into a feature vector comprising hydrophobicity, disorder propensity, and net charge computed using the Henderson–Hasselbalch equation, accounting for pKa values of ionizable residues at physiological pH (7.4). A binary indicator for residues associated with disorder is also given as a feature.", "We designed a deep bidirectional LSTM network optimized for sequential prediction tasks. LSTMs are well-suited for learning long-range dependencies in protein sequences due to their ability to retain past information through gated memory units. The model comprises an initial bidirectional LSTM layer consisting of 256 units for capturing forward and reverse dependencies within the sequence. The dropout layer has a 40% chance to mitigate overfitting by randomly deactivating neurons. The third layer refines the learned sequence representation using 128 nodes. From that, 64 neurons have fully connected dense layers with a rectified linear unit activation function. The LSTM framework incorporates a similarity-aware representation layer that reduces over-reliance on homologous sequence fragments, thereby mitigating but not fully eliminating the risk of inflated performance due to sequence similarity. The output layer with sigmoid activation function produces per-residue disorder probabilities constrained between 0 and 1. The model is trained using Binary cross-entropy loss, which quantifies the deviation between predicted probabilities (ŷi) and the validation disorder labels (yi).", "After training for 100 epochs with a batch size of 16, we evaluate the model on a held-out test using receiver operating characteristic analysis. The receiver operating characteristic-area under the curve (ROC-AUC) metric is calculated as:", "We compare our model’s ROC curve against multiple other deep learning architectures, such as the Gated Recurrent Unit (GRU), Temporal Convolutional Network (TCN), Bidirectional Recurrent Neural Network (BiRNN), and Convolutional Neural Network (CNN). To optimize classification, we introduce a dynamic threshold mechanism. Instead of a fixed 0.5 cutoff for disorder classification, we calibrate an empirical threshold T by maximizing Youden’s J-statistic:", "To ensure robustness, we compare the bit-wise predictions of our model with the flDPnn [35], flDPnn2 [43], ADOPT [44], DisoFLAG [45], and AIUPred [46] tools. The individual sequence accuracy and overall mean accuracy are calculated."], "subsections": []}, {"title": "Functional significance", "content": ["To evaluate the role of IDRs in kinase function, we generated high-confidence predictions using our model and mapped them to functional domains, including active sites, binding sites, and conserved motifs. We found that IDRs often overlap with these domains and are linked to dynamic conformational states, highlighting their regulatory significance. Particular attention was given to conserved kinase motifs such as the DFG motif [47], which is critical for ATP binding and activation loop dynamics, and the APE motif [48], which contributes to activation segment stabilization. Additionally, we examined the HRD motif, located within the catalytic loop, and the DLG motif, assessing both its canonical and reverse orientations, which may influence conformational flexibility and regulatory interactions [49]. We dynamically assess the mapping of domains, active sites (\\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${S}_A)$\\end{document}, binding interfaces (\\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${S}_B)$\\end{document}, regions involved in functional motifs (\\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${S}_R$\\end{document}) such as HRD, DFG, APE, EPA, DLG, GLD, GFD, DRH. To evaluate whether a given disorder region overlapped with or was proximal to functional motifs, we defined a buffer window of +10, –10 amino acids around the motif positions. A motif or functional site was considered associated with disorder, the score computed using the following conditions:", "where d represents the distance between the disorder region and the functional motif, and \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n$\\beta$\\end{document} is an empirical weight based on motif importance.", "To establish the disorder region overlapping with a kinase domain was considered functionally relevant, so a base score was assigned using a dynamic function:", "where \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n$\\alpha$\\end{document} is an empirically determined weight, \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${L}_{overlap}$\\end{document} represents the length of the overlapping region, and \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${L}_{domain}$\\end{document} is the total domain length.", "To assess how phosphorylation and mutation influence disorder regions, phosphorylation data were extracted from PhosphoSitePlus [50] and mapped to the disorder region. If phosphorylation was detected within a disorder region, it contributed positively based on its regulatory role:", "where \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n$\\gamma$\\end{document} is an empirical weight, and \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${f}_{phos}$\\end{document}is a function that accounts for phosphorylation events at a site. Similarly, mutation is computed as follows: mutation data extracted from KinaseMD [51].", "\\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n$\\delta$\\end{document}\n is the empirical weight, and \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${f}_{mut}$\\end{document} captures the severity of the mutation’s effect on the disorder. The empirical weights α and β were set to reflect the relative structural importance of domain overlap and proximity to conserved kinase motifs (e.g. DFG, HRD), which are known to play critical roles in kinase activation and regulation. The weights γ and δ were assigned to phosphorylation and mutation features, respectively, based on their potential regulatory impact and disruption of disorder regions. These weights were calibrated for balance interpretability and biological significance rather than being strictly optimized."], "subsections": []}, {"title": "Integrated scoring and normalization", "content": ["To quantify the functional significance of each disordered region, we assigned a total feature score C using a weighted sum of contributing features:", "The final normalized score was computed using dynamic scaling:", "where \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${C}_{max}$\\end{document} and \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${C}_{min}$\\end{document} represent the maximum and minimum feature scores across all disorder regions. This ensures a robust 0–1 scale for comparative ranking."], "subsections": []}, {"title": "Family-based analysis", "content": ["Recognizing the evolutionary diversity of kinase families, we performed a family-specific analysis of IDRs using proximity-based clustering. Disorder regions within 10 residues of each other were grouped into the same cluster, provided they appeared in at least two different kinases. For each valid cluster, we calculated a disorder score based on three metrics, the first being the Sequence Similarity Score (SSS), which quantifies conservation within clustered IDRs.", "where \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${L}_{max}$\\end{document} is the length of the longest disorder sequence in the cluster, and \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n$I\\left({R}_i\\right)$\\end{document} is an indicator function that counts the number of completely conserved residues across all disorder regions in the cluster.", "The second metric is the positional similarity score (PSS), the overlap between disorder regions within a cluster is assumed to be complete, yielding:", "This simplification ensures that disorder regions in close proximity are treated as functionally relevant. The third metric is the disorder region consistency (DRC), the consistency of disorder region length across kinases in the cluster is determined as:", "where \\documentclass[12pt]{minimal}\n\\usepackage{amsmath}\n\\usepackage{wasysym}\n\\usepackage{amsfonts}\n\\usepackage{amssymb}\n\\usepackage{amsbsy}\n\\usepackage{upgreek}\n\\usepackage{mathrsfs}\n\\setlength{\\oddsidemargin}{-69pt}\n\\begin{document}\n${L}_{min}$\\end{document} is the shortest disorder region length within the cluster. Considering all three metrics, each cluster is assigned a disorder score as a weighted combination of these metrics:", "The disorder score for each kinase is determined by identifying the highest cluster score in which it appears. This ensures that kinases involved in highly conserved disorder regions obtain a higher disorder score."], "subsections": []}, {"title": "Disorder to order transition", "content": ["To evaluate the propensity of IDRs in kinases to undergo disorder-to-order transitions, we developed a biophysically informed scoring framework. For each disordered segment, we computed a set of sequence-derived features using a sliding window approach. These features included secondary structure propensities based on Chou–Fasman parameters, normalized amino acid composition, hydrophobicity profiles using the Kyte–Doolittle scale, and net charge distributions calculated from the presence of charged residues. Additionally, we incorporated side chain volume and flexibility indices from literature, sequence complexity quantified via Shannon entropy, and contact potential estimated using a simplified Miyazawa–Jernigan interaction matrix. These features were integrated into an empirical free energy model, where the weighted sum of these properties yielded a predicted free energy change (ΔG) for the region. This ΔG was then used to estimate the folding rate using an exponential kinetic model that accounts for sequence length and thermal conditions. To make the output biologically interpretable, a logistic transformation was applied to the free energy values, producing a transition probability score between 0 and 1, indicating the likelihood that a disordered region may adopt an ordered structure. Furthermore, to assess the functional impact of mutations within IDRs, we introduced point mutations into the wild-type sequences and recalculated the ΔG and transition probabilities. The difference in energy (ΔΔG) between wild-type and mutant was used to infer whether the mutation stabilizes or destabilizes the region. All equations, parameter details, and model validations are provided in the Supplementary File."], "subsections": []}]}, {"title": "Results", "content": [], "subsections": [{"title": "High confidence prediction of human kinome intrinsically disordered regions", "content": ["Disordered protein regions are amino acid stretches with a high composition of polar or charged amino acids, which lack sufficient hydrophobic amino acids for cooperative folding and definite structure formation. These regions generally contribute to the conformational heterogeneity of proteins by connecting structurally solid regions and carrying linear peptide motifs and PTM sites to enable interactions with different proteins [52]. This is imperative to predict the IDRs in human kinome to streamline the understanding of dynamic, adaptable, and malleable molecular communication [53].", "To study the impact of disordered regions in the human kinome, we compiled a kinase list from KinMap and KinBase. Due to inconsistencies between sources, we normalized the data by mapping entries to official gene symbols. We then filtered kinases based on their domains using UniProt and InterPro, resulting in 500 protein kinases with defined kinase domains and 23 atypical kinases without such domains. Other kinase types (e.g. lipid, nucleic acid, carbohydrate) were excluded, focusing the analysis on these 523 human protein kinases.", "We predicted the disorder status of 523 kinases, ensuring region-wise disorder prediction without any protein length limitations (Supplementary Table 3). The LSTM-based architecture transformed amino acid sequences into a feature matrix, padded to a uniform length and scaled via StandardScaler to fuel a sequential model comprising dual LSTM layers (100 and 50 units) interspersed with dropout (0.2) for regularization, culminating in a dense sigmoid output layer. Trained over 100 epochs with the Adam optimizer. The model achieved high-quality predictions with an validation accuracy of 0.89 and an AUC of 0.97 precision of 0.95, recall of 0.96, and an F1-score of 0.95 (Fig. 1). To evaluate the competitiveness of our proposed architecture against modern deep learning frameworks, we conducted a comparative performance analysis. The LSTM-based model achieved an AUC-ROC of 0.97 and a testing accuracy of 0.98, indicating strong predictive capability. Among the compared models, the BiRNN achieved the highest AUC-ROC of 0.99, while the TCN demonstrated the highest accuracy (0.98). Overall, the LSTM framework remains highly competitive, exhibiting an excellent balance between accuracy and generalization across different architectures (Table 1). The ROC curve demonstrates the model’s success in identifying patterns within kinase proteins. For comparative validation, we applied existing disorder prediction tools to the kinase dataset, yielding the following AUC values: flDPnn (0.83), ADOPT (0.93), DisoFLAG (0.88), AIUPred (0.77), and flDPnn2 (0.84) (Fig. 2, Table 2). This validation supports our claim that instead of predicting disorder across the entire human proteome, developing a specialized approach tailored to the human kinome and its characteristic disorder region continuum is crucial, which serves as an adaptive hub in regulatory circuits. After all, it imposes no restrictions on protein length, allowing for comprehensive analysis of both short and full-length protein sequences, an area where many current tools fall short. Second, it provides high-resolution predictions at the single amino acid residue level, enabling fine-grained mapping of intrinsic disorder with unprecedented detail.", "Disordered segments >10–15 amino acids within proteins are generally considered relevant, with short segments often overlooked unless experimentally validated. However, SLiMs (3–10 aa) when conserved across orthologs or protein families, potentially carry heavily post-translationally modified sites, increasing protein complexity. These SLiMs are particularly relevant for human kinases, considering motifs such as DFG (Asp-Phe-Gly) and HRD (His-Arg-Asp) being functionally irreplaceable for kinase activity regardless of their short length. So, to extrapolate regulatory information from these short stretches, we considered predicted disordered regions ≥3 amino acids relevant in the current study.", "Our model has predicted a total of 2863 disorder regions across 518 human kinases, with an average disorder content of 17% per kinase. No disordered regions were identified for kinases GUCY2F, TXK, ADCK5, BCKDK, and PDK2. To estimate the dynamics of disordered regions in human kinases, we calculated the ratio of disordered region composition within each kinase based on the ratio of disordered residues to the total sequence length. The results revealed a wide range of disorder distribution among kinases, ranging from very low to heavily disordered composition. Kinases with heavily disordered composition included CDK11B (52.96%), GTF2F1 (45.84%), CDK11A (44.95%), and PRP4K (44.19%). CDK11B and CDK11A are cyclin-dependent kinases with 795 and 783 aa lengths, respectively. Uniprot has annotated two disordered regions in these kinases, specifically within amino acid ranges 17–412 and 733–795 for CDK11B, and within 18–398 and 721–783 for CDK11A. Our prediction model has mapped seven disordered regions precisely within the ranges 1–16, 20–36, 38–61, 77–274, 293–424, 753–775, and 784–794 for CDK11B and 1–16, 20–36, 38–59, 125–262, 281–405, 741–763, and 772–782 for CDK11A with superior AUC and ROC scores compared to flDPnn2, highlighting the predictive robustness. Only the kinase domain structure of CDK11B [PDB: 7UKZ (428–736 aa)] is currently available crystalized and no crystal structure is available for CDK11A. Considering Deiana et al. (2019) have suggested intrinsically disordered proteins as proteins with more than 30% disordered residues, we categorize these kinases with more than 40% disordered composition to be IDKs with short half-lives [54].", "Low disordered composition kinases included EPHA7, GUCY2C, MUSK, NPR2, and MYO3A, with <0.5% disordered regions. EPHA7 (EPH Receptor A7), MUSK (Muscle Associated Receptor Tyrosine Kinase), and NPR2 (Natriuretic Peptide Receptor 2) are tyrosine kinases, MYO3A (Myosin IIIA) is a Serine/Threonine kinase, and GUCY2C (Guanylate Cyclase 2C) is a dual specificity kinase. These are highly stable and rigid kinases with a larger half-life. They potentially have constitutively active (always on) conformation or are regulated by binding to other proteins or cofactors, as their conformational flexibility could be very limited due to low root mean square deviations reported for alpha helices and beta sheets that make up structured domains [55, 56]."], "subsections": []}, {"title": "Distribution of disorder scores across kinase families", "content": ["To assess the functional relevance of disordered regions across kinase families, we evaluated the conservation of DRs across kinase families. Three metric parameters were used as features for conservation analysis. Sequence similarity—to capture residue-level alignment, Positional similarity—to assess overlaps in disorder locations, and regional consistency—to reflect the recurrence of disordered segments across the sequences (see Methods). The FASTA sequences of the entire human protein kinome were assembled, and their corresponding family information was fetched from KinHub [57]. Further to better comprehend the conservation of disordered regions across families, we classified disordered regions based on their length into short (1–20 aa) and long (>20 aa) disordered regions as previously described by Oates et al. (2013) [58].", "As overall conservation of proteins across a kinase family varied, with proteins mostly showing a high level of conservation only within the kinase domain. However, disordered region conservation was noted in the internal and N- & C-terminal segments of many kinase families. For instance, considering short disordered regions, AKT kinases exhibit a conserved mean disorder score of 0.50 ± 0.02 (n = 17 regions across three kinases), predominantly located in the N- and C-terminal regions (82% of IDRs), and show minimal intra-family variance (σ2 = 0.001), suggesting consistent structural flexibility conducive to dynamic signaling interactions. SRC kinases present a moderate and uniform disorder score (0.50 ± 0.00, n = 1 region), primarily at the N-terminus. In contrast, CDK family members display a broader disorder range (mean = 0.32 ± 0.07, n = 36 regions across 18 kinases), with substantial diversity across N-terminal and internal IDRs (Kolmogorov–Smirnov test, P < .01). Notable outliers, such as CDK9 with a uniquely short IDR (length = 1, Z-score = −2.3), may indicate functional divergence or unique regulatory mechanisms. On the other hand, for long disordered regions, the overall mean disorder score is 0.35 ± 0.22, with families like Aurora (Aur), NAK, DCAMKL, MLCK, and NEK exhibiting the highest average scores (>0.55). In contrast, families such as LISK, JAK, Insulin receptor family, IRAK, and YANK display a complete absence of long disordered segments (mean = 0.0). The data are listed in Supplementary Tables 4 & 5.", "In the comparative context of conservation, short and long stretch IDRs in the human kinome reveal distinct patterns of disorder conservation. Short stretch IDRs display a wide range of mean disorder scores (0.0 to 1.0), indicating variable conservation across kinase families, where higher disorder scores may reflect less conserved, rapidly evolving regions adapted for diverse signaling roles. In contrast, long stretch IDRs exhibit a more constrained range of mean disorder scores (0.0 to 0.55), suggesting greater conservation, likely due to their structural roles within kinase domains that require balanced flexibility for function. The disorder score distribution is displayed in Fig. 3. To analyze the relationship between evolutionary conservation of long and short IDRs, Spearman’s rank correlation analysis was implemented. Low Spearman correlation highlights that conservation mechanisms differ between short and long IDRs, with short stretches showing greater evolutionary divergence and long stretches maintaining higher conservation, possibly to preserve essential kinase functionality across species."], "subsections": []}, {"title": "Analysis of functional hotspot disorder regions", "content": ["To identify functionally significant DRs within protein kinases, we sought to define DR functional hotspots in human protein kinases, based on its domain overlap, proximity to functional key motifs, presence of phosphorylation sites and impact of mutations.", "A total of 125 short-stretch disorder regions were identified within kinase domains, each potentially contributing to kinase-specific functions (Supplementary Tables 6 & 7). Using a functional hotspot prediction threshold of 0.12, we categorized these regions based on their disorder intensity and spatial distribution. Among these, 40 were classified as long-stretch disorder regions. Functional hotspots were confidently assigned to 25 of these, spanning across 20 different kinases. Strikingly, a pronounced N-terminal localization bias was observed, with 42.5% of functional hotspots situated within the first 50 amino acid residues. This spatial preference suggests a potential regulatory role for N-terminal intrinsic disorder, possibly mediating dynamic interactions with substrates, cofactors, or regulatory partners. Normalized disorder scores for these hotspots predominantly clustered below 0.2, indicating moderate disorder, although a subset displayed higher disorder intensities, reaching up to 0.32.", "In the kinase divergence, TAF1 exhibited multiple functional hotspots within the domain, suggesting involvement in processes requiring significant conformational flexibility. The N-terminal disorder region of TP53RK demonstrated the highest disorder intensity (0.32) within the domain, marked by the presence of APE and EPA motifs. This region also contained a predominant phosphosite and a novel mutation, underscoring its likely regulatory importance.", "Beyond the kinase domain, we identified 635 short-stretch disorder hotspots and 345 long-stretch disorder hotspots (Supplementary Tables 8 & 9). These exhibited normalized scores above the 0.12 threshold. Kinases, including TTN, OBSCN, WNK1, and ALPK3, harbored multiple high-scoring hotspots, reinforcing the notion that disordered regions outside canonical domains also serve as hubs of regulatory flexibility and functional adaptation. Figure 4 portrays the important functional disorder hotspots in terms of kinase domain.", "In long-stretched disorder scores ranged from 0.04 to 0.38, with most regions exhibiting moderate disorder (mean score: 0.14), suggesting that flexible but not fully unstructured segments dominate kinase regulation beyond the catalytic core. Several kinases, including TTN (five hotspots), WNK1 and CDK12 (4 each), as well as ABL1, TRIO, SPEG, and others with three hotspots each, showed a pronounced enrichment of high-scoring disordered segments. Among all analyzed kinases, ABL1 emerged as a top candidate with a prominent long-stretch disorder hotspot located outside its kinase domain. This region is functionally enriched by the presence of hallmark kinase motifs—HRD, and nearby APE and EPA sequences—typically associated with catalytic regulation and structural stabilization within the kinase domain. Remarkably, this disordered segment also harbors four experimentally validated phosphosites: T532-p, S535-p, S559-p, and S569-p. The co-occurrence of these motifs and phosphosites within a disordered context suggests that this region in ABL1 acts as a unique region.", "Functional disorder hotspot scores for short disordered regions in kinases range from 0.04 to 0.28 (average 0.09). Notably, 106 kinases have only one DR region, often outside the kinase domain, indicating flexible but partly structured roles in regulation. Kinases like TTN, OBSCN, ALPK3, and MAST4 showed high enrichment, with TTN alone having over 100 segments, 6.3% of which are hotspots containing key motifs (GFD, APE) and validated phosphosites, highlighting their role as multifunctional regulatory modules."], "subsections": []}, {"title": "Structural integrity of intrinsically disordered regions in human kinome", "content": ["The initial consideration ended up with how these IDRs are localized in kinase, which we have addressed with the hydrophobic-based solvent exposure index; hydrophobic concentrated IDRs are confined to the inside, to attain a stable energy propagation. Some kinases, such as AKT/PKB, are regulated by a disorder-to-order transition of their activation loop upon its phosphorylation [59]. Under this interpretation, we tried to calibrate the wiring and rewiring events to decipher unique transition properties of the IDRs, which further vary with their functional relevance. The N-terminal region of AKT1 (residues 1–6: MSDVAI) is identified as intrinsically disordered, aligning with established regulatory mechanisms in kinases like AKT/PKB, where phosphorylation induces structural transitions. Phosphorylation at Serine-2 (S2) within this disordered segment leads to a ΔΔG of −0.500 kcal/mol, indicating a significant gain in structural stability. Additionally, the predicted transition probability of 77.02% strongly suggests a disorder-to-order transition upon phosphorylation. These findings support the notion that phosphorylation at S2 acts as a molecular switch, promoting structural ordering of the disordered region and potentially enabling AKT1 activation and downstream signaling. When a mutation is introduced, the severity of the kinases may change, further affecting the continuity level. The proposed framework allows users to input a kinase, its disordered region, and a missense mutation, and returns a probability score indicating the likelihood of structural integrity. Looking at this specific case for AKT1, where the wild-type sequence MSDVAI mutates to MSDLAI at position 4 (V to L), the ΔG values tell us the wild-type is −0.274 (kcal/mol, I assume), and the mutant is −0.208, giving a ΔΔG of 0.066 (see Fig. 5). A positive ΔΔG suggests the mutation slightly destabilizes the protein, as it indicates a less favorable folding energy. The small changes in ΔΔG (like 0.5–1 kcal/mol) can hint at stability shifts, and 0.066 is pretty minor, so the effect might be subtle."], "subsections": []}]}, {"title": "Discussion and conclusions", "content": ["This study presents a high-confidence prediction of IDR in the human kinome. Current prediction tools heavily compress the proteome-level insights. They also struggle with accurately identifying short disorder regions, which play crucial roles in molecular recognition [60]. Our model outperforms flDPnn, flDPnn2, ADOPT, DisoFLAG, AIUPred in AUC, and ROC metrics, highlighting its predictive strength. While flDPnn2 surpasses many existing tools, most models struggle with long sequences, introducing noise at the proteome scale. Many cannot handle proteins over 5000 amino acids, limiting their use in large kinases like OBSCN and TTK. In contrast, our approach provides a scalable, accurate solution for disorder prediction across diverse kinases.", "To further probe the functional significance of kinase IDRs, we categorized the IDRs in domain-specific categorization involving a logical analysis bound to short and long stretch IDRs. It is imperative that molecular recognition property identification and interpretation of flexibility segments in understudied kinases. Using a combination of functional motifs, distance metrics, mutation probability, MultiPTM context, and reported phosphosites evaluated how functional DR hotspots alter kinase modularity. This allowed us to explore how sequence variations within IDRs contribute to disease-associated dysfunctions or regulatory modifications.", "Statistical comparisons between short and long disorder scores revealed distinct distribution patterns, suggesting that long disordered stretches are more conserved and potentially more functionally relevant. Detailed mapping across kinase domains further highlighted that long disorder regions often lie outside the canonical kinase domains, frequently overlapping with hotspots of post-translational modifications and functional motifs, thereby implying regulatory roles. Conversely, short disorder regions, although present both within and outside domains, showed less pronounced conservation. Finally, mutation analysis of a representative AKT kinase demonstrated the structural and energetic impact of a point mutation within a disordered region, with a moderate shift in ΔG and significant transition probability, underscoring the potential of disorder-associated residues to influence protein dynamics. Together, these findings underscore the critical regulatory and functional roles of IDRs in kinase biology and provide a valuable framework for further exploration of disorder-mediated signaling and disease-associated mutations. While our free energy-based model effectively explains the impact of many site-specific mutations by integrating sequence, secondary structure, and side chain information, it is not without limitations. In particular, the model may underperform for mutations involving long-range allosteric effects, regions with sparse structural or evolutionary data, or those influenced by post-translational modifications. These cases highlight areas for future refinement and the potential integration of additional contextual or structural dynamics data."], "subsections": []}, {"title": "Supplementary Material", "content": [], "subsections": []}, {"title": "Research support", "content": ["We acknowledge the financial support for the computational facility from Yenepoya (Deemed to be University). We also thank Dr. Mukhtar Ahmed researcher’s support (ORF-2025-984), King Saud University, Riyadh, Saudi Arabia."], "subsections": []}, {"title": "N/A", "content": ["Conflict of interest: None declared."], "subsections": []}, {"title": "Funding", "content": ["None declared."], "subsections": []}, {"title": "Code and data availability", "content": ["All code and datasets are publicly available at: https://github.com/naveen-joy-18/Impact-of-Intrinsically-Disordered-Regions-and-Functional-Disorder-Hotspots-in-the-Human-Kinome."], "subsections": []}]}}
{"pmid": "41534519", "meta": {"content": {"abstract": "Intrinsically disordered proteins or regions (IDPs or IDRs) adopt diverse binding modes with different partners, ranging from coupled folding and binding to fuzzy binding and fully disordered binding. Characterizing IDR interfaces is challenging both experimentally and computationally. State-of-the-art tools such as AlphaFold multimer and AlphaFold3 can be used to predict IDR binding sites, although they are less accurate at their benchmarked confidence cutoffs. Here, we developed Disobind, a deep-learning method that predicts inter-protein contact maps and interface residues for an IDR and its partner, given their sequences. It uses sequence embeddings from the ProtT5 protein language model. Disobind outperforms state-of-the-art interface predictors for IDRs. It also outperforms AlphaFold multimer and AlphaFold3 at multiple confidence cutoffs. Combining Disobind and AlphaFold-multimer predictions further improves performance. In contrast to current methods, Disobind considers the context of the binding partner and does not depend on structures and multiple sequence alignments. Its predictions can be used to localize IDRs in large assemblies and characterize IDR-mediated interactions.", "keywords": ["DL", "IDP", "IDR", "deep learning", "intrinsically disordered proteins", "intrinsically disordered regions", "pLMs", "protein language model", "protein structure"], "mesh_terms": ["*Intrinsically Disordered Proteins/chemistry/metabolism", "Protein Binding", "Binding Sites", "*Computational Biology/methods", "Deep Learning", "Protein Folding", "Humans", "Amino Acid Sequence"], "pub_types": ["Journal Article"]}, "contributors": {"medline": {"affiliations": ["National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore 560065, Karnataka, India. Electronic address: kartikm@ncbs.res.in.", "National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore 560065, Karnataka, India.", "National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore 560065, Karnataka, India. Electronic address: shruthiv@ncbs.res.in."], "auids": [], "full_names": ["Majila, Kartik", "Ullanat, Varun", "Viswanath, Shruthi"], "short_names": ["Majila K", "Ullanat V", "Viswanath S"]}, "xml": [{"affiliations": ["National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore 560065, Karnataka, India. Electronic address: kartikm@ncbs.res.in."], "full_name": "Majila, Kartik", "identifiers": [], "short_name": "Majila K"}, {"affiliations": ["National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore 560065, Karnataka, India."], "full_name": "Ullanat, Varun", "identifiers": [], "short_name": "Ullanat V"}, {"affiliations": ["National Center for Biological Sciences, Tata Institute of Fundamental Research, Bangalore 560065, Karnataka, India. Electronic address: shruthiv@ncbs.res.in."], "full_name": "Viswanath, Shruthi", "identifiers": [], "short_name": "Viswanath S"}]}, "identity": {"doi": "10.1016/j.cels.2025.101486", "pmid": "41534519", "title": "Disobind: A sequence-based, partner-dependent contact map and interface residue predictor for intrinsically disordered regions."}, "links": {"cites": [], "entrez": {}, "external": [{"attribute": "subscription/membership/fee required", "category": "Full Text Sources", "linkname": "", "provider": "Elsevier Science", "url": "https://linkinghub.elsevier.com/retrieve/pii/S2405-4712(25)00319-9"}], "pmc": [], "refs": [], "review": ["41534519", "31441158", "34418423", "40650026", "33248138", "35139468", "34390736", "38296708", "39093570", "36959451", "36851914", "36833360", "34358545", "33567297", "34610267", "39740355", "31003202", "24930020"], "similar": ["41534519", "39763873", "39446390", "31441158", "40944448", "34418423", "31207240", "33087759", "38866950", "41174305", "30058229", "38079339", "28780862", "32696395", "38701796", "29763584", "40650026", "34905768", "30878482", "40254833", "25391399", "40403066", "32535960", "33248138", "38987470", "35139468", "34390736", "32553192", "32702119", "37878721", "37883249", "38296708", "30366362", "36272675", "41378882", "40221640", "35609776", "31724233", "40441416", "29716994", "34626492", "28394890", "28381244", "39093570", "34461937", "28365882", "36304335", "29066345", "29990433", "36959451", "40133775", "29042212", "36851914", "39933697", "32722039", "36833360", "29228892", "35149212", "30135358", "34358545", "31797594", "27787828", "30841624", "40244295", "39246251", "31756456", "35436286", "33567297", "34610267", "38166858", "25424537", "36153968", "41218076", "41454828", "33936509", "39740355", "39032884", "32824743", "41309611", "36458437", "23142703", "31003202", "32112804", "39614773", "18831774", "26149687", "41337585", "25286318", "26109352", "29309620", "24930020", "39586094", "31804089", "38459060", "40828855", "24115198", "41053908", "39510344", "33875885", "30866736"], "text_mined": []}, "metadata": {"entrez_date": "2026/01/15 16:13", "fetched_at": "2026-05-02 18:19:10"}, "source": {"journal_abbrev": ["Cell Syst"], "journal_title": ["Cell systems"], "pub_date": "2026 Jan 21", "pub_types": ["Journal Article"], "pub_year": "2026"}}, "content": {}}