diff --git a/CITATION.cff b/CITATION.cff index 980524240..1b49c7d12 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -1,7 +1,7 @@ # YAML 1.2 # Metadata for citation of this software according to the CFF format (https://citation-file-format.github.io/) cff-version: 1.2.0 -message: If you use this software in a publication, please cite the ChemRxiv preprint and/or source code archive on Zenodo. +message: If you use this software in a publication, please cite the JOSS paper and/or source code archive on Zenodo. title: datalab abstract: datalab is a place to store experimental data and the connections between them. repository-code: https://github.com/datalab-org/datalab @@ -12,11 +12,12 @@ authors: - given-names: The datalab development team identifiers: - type: doi - value: 10.5281/zenodo.14719467 + value: 10.5281/zenodo.8127782 preferred-citation: type: article title: 'datalab: federated data management infrastructure for materials chemistry and beyond' - doi: 10.26434/chemrxiv.15001945/v1 + doi: + journal: Journal of Open Source Software authors: - given-names: Matthew L. family-names: Evans diff --git a/paper/paper.bib b/paper/paper.bib new file mode 100644 index 000000000..bc50e086f --- /dev/null +++ b/paper/paper.bib @@ -0,0 +1,333 @@ +@article{Barillari2016, + title = {{{openBIS ELN-LIMS}}: an open-source database for academic laboratories}, + shorttitle = {{{openBIS ELN-LIMS}}}, + author = {Barillari, Caterina and Ottoz, Diana S. M. and {Fuentes-Serna}, Juan Mariano and Ramakrishnan, Chandrasekhar and Rinn, Bernd and Rudolf, Fabian}, + year = {2016}, + month = feb, + journal = {Bioinformatics}, + volume = {32}, + number = {4}, + pages = {638--640}, + doi = {10.1093/bioinformatics/btv606}, + urldate = {2025-01-28} +} + +@article{Brandt2021, + title = {{{Kadi4Mat}}: {{A Research Data Infrastructure}} for {{Materials Science}}}, + shorttitle = {{{Kadi4Mat}}}, + author = {Brandt, Nico and Griem, Lars and Herrmann, Christoph and Schoof, Ephraim and Tosato, Giovanna and Zhao, Yinghan and Zschumme, Philipp and Selzer, Michael}, + year = {2021}, + month = feb, + journal = {Data Science Journal}, + volume = {20}, + number = {1}, + doi = {10.5334/dsj-2021-008}, + urldate = {2025-06-08}, + langid = {american} +} + +@article{Ghiringhelli2023, + title = {Shared metadata for data-centric materials science}, + author = {Ghiringhelli, Luca M. and Baldauf, Carsten and Bereau, Tristan and Brockhauser, Sandor and Carbogno, Christian and Chamanara, Javad and Cozzini, Stefano and Curtarolo, Stefano and Draxl, Claudia and Dwaraknath, Shyam and Fekete, {\'A}d{\'a}m and Kermode, James and Koch, Christoph T. and K{\"u}hbach, Markus and Ladines, Alvin Noe and Lambrix, Patrick and Himmer, Maja-Olivia and Levchenko, Sergey V. and Oliveira, Micael and Michalchuk, Adam and Miller, Ronald E. and Onat, Berk and Pavone, Pasquale and Pizzi, Giovanni and Regler, Benjamin and Rignanese, Gian-Marco and Schaarschmidt, J{\"o}rg and Scheidgen, Markus and Schneidewind, Astrid and Sheveleva, Tatyana and Su, Chuanxun and Usvyat, Denis and Valsson, Omar and W{\"o}ll, Christof and Scheffler, Matthias}, + year = {2023}, + month = sep, + journal = {Scientific Data}, + volume = {10}, + number = {1}, + pages = {626}, + doi = {10.1038/s41597-023-02501-8}, + urldate = {2023-09-26}, + copyright = {2023 Springer Nature Limited}, + langid = {english} +} + +@article{Herrmann2025, + title = {Enhancing {{FAIRdata}} by providing digital workflows from data generation to the publication of data: an open source approach described for cyclic voltammetry}, + shorttitle = {Enhancing {{FAIRdata}} by providing digital workflows from data generation to the publication of data}, + author = {Herrmann, David and Hodapp, Patrick and Starman, Martin and Huang, Pei-Chi and Lin, Chia-Lin and Le, Lan B. and Fischer, Tillmann and Bizzarri, Claudia and R{\"o}se, Philipp and Oppel, Niklas and Klar, Jochen and Tremouilhac, Pierre and Holzhauer, Laura and {Herres-Pawlis}, Sonja and Hoffmann, Alexander and Seitz, Tobias and Dorn, Alrik and Zeitler, Kirsten and Jung, Nicole and Br{\"a}se, Stefan}, + year = {2025}, + journal = {Chemical Science}, + volume = {16}, + number = {10}, + pages = {4430--4441}, + publisher = {Royal Society of Chemistry}, + doi = {10.1039/D4SC08620A}, + urldate = {2025-06-08}, + langid = {english} +} + +@article{Higgins2022, + title = {Considerations for implementing electronic laboratory notebooks in an academic research environment}, + author = {Higgins, Stuart G. and {Nogiwa-Valdez}, Akemi A. and Stevens, Molly M.}, + year = {2022}, + month = feb, + journal = {Nature Protocols}, + volume = {17}, + number = {2}, + pages = {179--189}, + publisher = {Nature Publishing Group}, + doi = {10.1038/s41596-021-00645-8}, + urldate = {2025-06-08}, + copyright = {2021 Springer Nature Limited}, + langid = {english} +} + +@article{Lam2025, + title = {General data management workflow to process tabular data in automated and high-throughput heterogeneous catalysis research}, + author = {Lam, Erwin and Maury, Tanguy and Preiss, Sebastian and Hou, Yuhui and Frey, Hannes and Barillari, Caterina and Laveille, Paco}, + year = {2025}, + journal = {Digital Discovery}, + volume = {4}, + number = {2}, + pages = {539--547}, + publisher = {Royal Society of Chemistry}, + doi = {10.1039/D4DD00350K}, + urldate = {2025-06-08}, + langid = {english} +} + +@article{Rhiem2021, + title = {{{SampleDB}}: {{A}} sample and measurement metadata database}, + shorttitle = {{{SampleDB}}}, + author = {Rhiem, Florian}, + year = {2021}, + month = feb, + journal = {Journal of Open Source Software}, + volume = {6}, + number = {58}, + pages = {2107}, + doi = {10.21105/joss.02107}, + urldate = {2025-06-08}, + langid = {english} +} + +@article{Scheffler2022a, + title = {{{FAIR}} data enabling new horizons for materials research}, + author = {Scheffler, Matthias and Aeschlimann, Martin and Albrecht, Martin and Bereau, Tristan and Bungartz, Hans-Joachim and Felser, Claudia and Greiner, Mark and Gro{\ss}, Axel and Koch, Christoph T. and Kremer, Kurt and Nagel, Wolfgang E. and Scheidgen, Markus and W{\"o}ll, Christof and Draxl, Claudia}, + year = {2022}, + month = apr, + journal = {Nature}, + volume = {604}, + number = {7907}, + pages = {635--642}, + publisher = {Nature Publishing Group}, + doi = {10.1038/s41586-022-04501-x}, + urldate = {2024-11-27}, + copyright = {2022 Springer Nature Limited}, + langid = {english} +} + +@article{Scheidgen2023, + title = {{{NOMAD}}: {{A}} distributed web-based platform for managing materials science research data}, + shorttitle = {{{NOMAD}}}, + author = {Scheidgen, Markus and Himanen, Lauri and Ladines, Alvin Noe and Sikter, David and Nakhaee, Mohammad and Fekete, {\'A}d{\'a}m and Chang, Theodore and Golparvar, Amir and M{\'a}rquez, Jos{\'e} A. and Brockhauser, Sandor and Br{\"u}ckner, Sebastian and Ghiringhelli, Luca M. and Dietrich, Felix and Lehmberg, Daniel and Denell, Thea and Albino, Andrea and N{\"a}sstr{\"o}m, Hampus and Shabih, Sherjeel and Dobener, Florian and K{\"u}hbach, Markus and Mozumder, Rubel and Rudzinski, Joseph F. and Daelman, Nathan and Pizarro, Jos{\'e} M. and Kuban, Martin and Salazar, Cuauhtemoc and Ondra{\v c}ka, Pavel and Bungartz, Hans-Joachim and Draxl, Claudia}, + year = {2023}, + month = oct, + journal = {Journal of Open Source Software}, + volume = {8}, + number = {90}, + pages = {5388}, + doi = {10.21105/joss.05388}, + urldate = {2024-11-27}, + langid = {english} +} + +@article{Schlabach2024, + title = {Using {{ELN Functionality}} of {{Kadi4Mat}} ({{KadiWeb}}) in a {{Materials Science Case Study}} of a {{User Facility}}}, + author = {Schlabach, Sabine and Wild, Johannes and Petkau, Oliver and Selzer, Michael and Szab{\'o}, Doroth{\'e}e Vinga}, + year = {2024}, + month = oct, + journal = {Data Science Journal}, + volume = {23}, + number = {1}, + doi = {10.5334/dsj-2024-050}, + urldate = {2025-06-08}, + langid = {american} +} + +@article{Tremouilhac2017a, + title = {Chemotion {{ELN}}: an {{Open Source}} electronic lab notebook for chemists in academia}, + shorttitle = {Chemotion {{ELN}}}, + author = {Tremouilhac, Pierre and Nguyen, An and Huang, Yu-Chieh and Kotov, Serhii and L{\"u}tjohann, Dominic Sebastian and H{\"u}bsch, Florian and Jung, Nicole and Br{\"a}se, Stefan}, + year = {2017}, + month = sep, + journal = {Journal of Cheminformatics}, + volume = {9}, + number = {1}, + pages = {54}, + doi = {10.1186/s13321-017-0240-0}, + urldate = {2025-06-08} +} + +@article{Kanza2023, + title = {Digital research environments: a requirements analysis}, + shorttitle = {Digital research environments}, + author = {Kanza, Samantha and Willoughby, Cerys and Knight, Nicola J. and Bird, Colin L. and Frey, Jeremy G. and Coles, Simon J.}, + year = {2023}, + month = jun, + journal = {Digital Discovery}, + volume = {2}, + number = {3}, + pages = {602--617}, + publisher = {RSC}, + doi = {10.1039/D2DD00121G}, + urldate = {2025-06-08}, + langid = {english} +} + +@article{Mroz2025, + title = {Cross-disciplinary perspectives on the potential for artificial intelligence across chemistry}, + author = {Mroz, Austin and Basford, Annabel and Hastedt, Friedrich and Jayasekera, Isuru and {Mosquera-Lois}, Irea and Sedgwick, Ruby and Ballester, Pedro and Bocarsly, Joshua and Chanona, Ehecatl Antonio del R{\'i}o and Evans, Matthew and Frost, Jarvist and Ganose, Alex and Greenaway, Rebecca and Hii, King Kuok (Mimi) and Li, Yingzhen and Misener, Ruth and Walsh, Aron and Zhang, Dandan and Jelfs, Kim}, + year = {2025}, + journal = {Chemical Society Reviews}, + volume = {54}, + number = {11}, + pages = {5433--5469}, + publisher = {Royal Society of Chemistry}, + doi = {10.1039/D5CS00146C}, + urldate = {2025-06-08}, + langid = {english} +} + + +@article{Wilkinson2016, + title = {The {{FAIR Guiding Principles}} for scientific data management and stewardship}, + author = {Wilkinson, Mark D. and Dumontier, Michel and Aalbersberg, IJsbrand Jan and Appleton, Gabrielle and Axton, Myles and Baak, Arie and Blomberg, Niklas and Boiten, Jan-Willem and {da Silva Santos}, Luiz Bonino and Bourne, Philip E. and Bouwman, Jildau and Brookes, Anthony J. and Clark, Tim and Crosas, Merc{\`e} and Dillo, Ingrid and Dumon, Olivier and Edmunds, Scott and Evelo, Chris T. and Finkers, Richard and {Gonzalez-Beltran}, Alejandra and Gray, Alasdair J. G. and Groth, Paul and Goble, Carole and Grethe, Jeffrey S. and Heringa, Jaap and {'t Hoen}, Peter A. C. and Hooft, Rob and Kuhn, Tobias and Kok, Ruben and Kok, Joost and Lusher, Scott J. and Martone, Maryann E. and Mons, Albert and Packer, Abel L. and Persson, Bengt and {Rocca-Serra}, Philippe and Roos, Marco and {van Schaik}, Rene and Sansone, Susanna-Assunta and Schultes, Erik and Sengstag, Thierry and Slater, Ted and Strawn, George and Swertz, Morris A. and Thompson, Mark and {van der Lei}, Johan and {van Mulligen}, Erik and Velterop, Jan and Waagmeester, Andra and Wittenburg, Peter and Wolstencroft, Katherine and Zhao, Jun and Mons, Barend}, + year = {2016}, + month = mar, + journal = {Scientific Data}, + volume = {3}, + number = {1}, + pages = {160018}, + publisher = {Nature Publishing Group}, + doi = {10.1038/sdata.2016.18}, + urldate = {2022-02-06}, + copyright = {2016 The Author(s)}, + langid = {english} +} + +@online{ELNFileFormat, + title = {The ELN File Format}, + author = {{The ELN Consortium}}, + year = 2026, + url = {https://github.com/TheELNConsortium/TheELNFileFormat}, + doi = {10.5281/zenodo.22915091} +} + +@article{Sefton2025, + title = {{{RO-Crate Metadata Specification}} 1.2.0}, + author = {Sefton, Peter and Ó Carragáin, Eoghan and {Soiland-Reyes}, Stian and Corcho, Oscar and Garijo, Daniel and Palma, Raul and Coppens, Frederik and Goble, Carole and Fernández, José M. and Chard, Kyle and {Gomez-Perez}, Jose Manuel and Crusoe, Michael R. and Eguinoa, Ignacio and Juty, Nick and Holmes, Kristi and Clark, Jason A. and {Capella-Gutierrez}, Salvador and Gray, Alasdair J. G. and Owen, Stuart and Williams, Alan R. and Tartari, Giacomo and Bacall, Finn and Thelen, Thomas and Ménager, Hervé and {Rodríguez-Navas}, Laura and Walk, Paul and {whitehead}, brandon and Wilkinson, Mark and Groth, Paul and Bremer, Erich and Castro, Leyla Jael and Sebby, Karl and Kanitz, Alexander and Trisovic, Ana and Kennedy, Gavin and Graves, Mark and Koehorst, Jasper and Leo, Simone and Portier, Marc and Brack, Paul and Ojsteršek, Milan and Droesbeke, Bert and Niu, Chenxu and Tanabe, Kosuke and Miksa, Tomasz and La Rosa, Marco and Decruw, Cedric and Czerniak, Andreas and Jay, Jeremy and Serra, Sergio and Siebes, Ronald and {de Witt}, Shaun and El Damaty, Shady and Lowe, Douglas and Li, Xuanqi and Gundersen, Sveinung and Radifar, Muhammad and Wittner, Rudolf and Woolland, Oliver and De Geest, Paul and Fils, Douglas and Wetzels, Florian and Sirvent, Raül and Miller, Abigail and Emerson, Jake and Fucci, Davide and Kinoshita, Bruno P. and Bąk, Maciek and Hollunder, Jens and Weise, Martin and Bisht, Vartika and Hiraki, Toshiyuki Nishiyama and Ulrichts, Bram and Falk, Michael and Chadwick, Eli and Bauer, Daniel and Love, James and Adamidi, Eleni and Moore, Josh and Schöbitz, Lars and Meier, Andreas and Fuentes, Juan and Bainglass, Edan and Pataki, Balázs E.}, + year = 2025, + publisher = {researchobject.org}, + doi = {10.5281/zenodo.13751027}, + langid = {english}, + keywords = {data packaging,JSON-LD,linked data,metadata,research object,schema.org} +} + +@article{Soiland-Reyes2022, + title = {Packaging research artefacts with {{RO-Crate}}}, + author = {{Soiland-Reyes}, Stian and Sefton, Peter and Crosas, Mercè and Castro, Leyla Jael and Coppens, Frederik and Fernández, José M. and Garijo, Daniel and Grüning, Björn and La Rosa, Marco and Leo, Simone and Ó Carragáin, Eoghan and Portier, Marc and Trisovic, Ana and Community, RO-Crate and Groth, Paul and Goble, Carole}, + year = 2022, + journal = {Data Science}, + volume = {5}, + number = {2}, + pages = {97--138}, + publisher = {SAGE Publications}, + doi = {10.3233/DS-210053}, + langid = {english} +} + +@article{Evans2025, + title = {Datatractor: {{Metadata}}, automation, and registries for extractor interoperability in the chemical and materials sciences}, + author = {Evans, Matthew L. and Rignanese, Gian-Marco and Elbert, David and Kraus, Peter}, + year = 2025, + journal = {MRS Bulletin}, + volume = {50}, + number = {7}, + pages = {838--845}, + doi = {10.1557/s43577-025-00925-8}, + langid = {english}, + keywords = {Computation/computing,Data/database,Informatics} +} + +@article{Jablonka2023, + title = {14 examples of how {{LLMs}} can transform materials science and chemistry: a reflection on a large language model hackathon}, + author = {Jablonka, Kevin Maik and Ai, Qianxiang and {Al-Feghali}, Alexander and Badhwar, Shruti and Bocarsly, Joshua D. and Bran, Andres M. and Bringuier, Stefan and Brinson, L. Catherine and Choudhary, Kamal and Circi, Defne and Cox, Sam and de Jong, Wibe A. and Evans, Matthew L. and Gastellu, Nicolas and Genzling, Jerome and Gil, María Victoria and Gupta, Ankur K. and Hong, Zhi and Imran, Alishba and Kruschwitz, Sabine and Labarre, Anne and Lála, Jakub and Liu, Tao and Ma, Steven and Majumdar, Sauradeep and Merz, Garrett W. and Moitessier, Nicolas and Moubarak, Elias and Mouriño, Beatriz and Pelkie, Brenden and Pieler, Michael and Ramos, Mayk Caldas and Ranković, Bojana and Rodriques, Samuel G. and Sanders, Jacob N. and Schwaller, Philippe and Schwarting, Marcus and Shi, Jiale and Smit, Berend and Smith, Ben E. and Herck, Joren Van and Völker, Christoph and Ward, Logan and Warren, Sean and Weiser, Benjamin and Zhang, Sylvester and Zhang, Xiaoqi and Zia, Ghezal Ahmad and Scourtas, Aristana and Schmidt, K. J. and Foster, Ian and White, Andrew D. and Blaiszik, Ben}, + year = 2023, + journal = {Digital Discovery}, + volume = {2}, + number = {5}, + pages = {1233--1250}, + publisher = {RSC}, + doi = {10.1039/d3dd00113j}, + langid = {english} +} + +@article{Helmus2013, + title = {Nmrglue: an open source {{Python}} package for the analysis of multidimensional {{NMR}} data}, + author = {Helmus, Jonathan J. and Jaroniec, Christopher P.}, + year = 2013, + journal = {Journal of Biomolecular NMR}, + volume = {55}, + number = {4}, + pages = {355--367}, + doi = {10.1007/s10858-013-9718-x}, + langid = {english}, + keywords = {Data analysis,Data processing,Data visualization,Nuclear magnetic resonance,Open source,Python,Solid-state NMR} +} + +@article{Kerr2017, + title = {{{NMR}} and neutron total scattering studies of silicon-based anode materials for lithium-ion batteries}, + author = {Kerr, Christopher James}, + year = 2017, + doi = {10.17863/CAM.17707}, + langid = {english} +} + +@article{Zimmermann2025, + title = {32 examples of {{LLM}} applications in materials science and chemistry: towards automation, assistants, agents, and accelerated scientific discovery}, + author = {Zimmermann, Yoel and Bazgir, Adib and {Al-Feghali}, Alexander and Ansari, Mehrad and Bocarsly, Joshua and Brinson, L Catherine and Chiang, Yuan and Circi, Defne and Chiu, Min-Hsueh and Daelman, Nathan and Evans, Matthew L and Gangan, Abhijeet S and George, Janine and Harb, Hassan and Khalighinejad, Ghazal and Takrim Khan, Sartaaj and Klawohn, Sascha and Lederbauer, Magdalena and Mahjoubi, Soroush and Mohr, Bernadette and Mohamad Moosavi, Seyed and Naik, Aakash and Beste Ozhan, Aleyna and Plessers, Dieter and Roy, Aritra and Schöppach, Fabian and Schwaller, Philippe and Terboven, Carla and Ueltzen, Katharina and Wu, Yue and Zhu, Shang and Janssen, Jan and Li, Calvin and Foster, Ian and Blaiszik, Ben}, + year = 2025, + journal = {Machine Learning: Science and Technology}, + volume = {6}, + number = {3}, + pages = {030701}, + doi = {10.1088/2632-2153/ae011a}, + langid = {english} +} + +@article{Raccuglia2016, + title = {Machine-learning-assisted materials discovery using failed experiments}, + author = {Raccuglia, Paul and Elbert, Katherine C. and Adler, Philip D. F. and Falk, Casey and Wenny, Malia B. and Mollo, Aurelio and Zeller, Matthias and Friedler, Sorelle A. and Schrier, Joshua and Norquist, Alexander J.}, + year = 2016, + journal = {Nature}, + volume = {533}, + number = {7601}, + pages = {73--76}, + publisher = {Nature Publishing Group}, + doi = {10/f8nd3d}, + langid = {english} +} + +@article{Moxon2026, + title = {{{LinkML}}: an open data modeling framework}, + shorttitle = {{{LinkML}}}, + author = {Moxon, Sierra A T and Solbrig, Harold and Harris, Nomi L and Kalita, Patrick and Miller, Mark A and Patil, Sujay and Schaper, Kevin and Bizon, Chris and Caufield, J Harry and Cuesta, Silvano Cirujano and Cox, Corey and Dekervel, Frank and Dooley, Damion M and Duncan, William D and Fliss, Tim and Gehrke, Sarah and Graefe, Adam S L and Hegde, Harshad and Ireland, A J and Jacobsen, Julius O B and Krishnamurthy, Madan and Kroll, Carlo and Linke, David and Ly, Ryan and Matentzoglu, Nicolas and Overton, James A and Saunders, Jonny L and Unni, Deepak R and Vaidya, Gaurav and Vierdag, Wouter-Michiel A M and {LinkML~Community~Contributors} and Ruebel, Oliver and Chute, Christopher G and Brush, Matthew H and Haendel, Melissa A and Mungall, Christopher J}, + year = {2026}, + month = jan, + journal = {GigaScience}, + volume = {15}, + pages = {giaf152}, + doi = {10.1093/gigascience/giaf152}, + urldate = {2026-05-28} +} + +@misc{Roy2026, + title = {From Knowledge to Action: Outcomes of the 2025 Large Language Model (LLM) Hackathon for Applications in Materials Science and Chemistry}, + author = {Roy, Aritra and Shen, Kevin and MacBride, Andrew and Oladipupo, Awwal and Taskeen, Mudassra and Treyde, Wojtek and Abakar, Ruaa A. E. A. and Abbas, Ahmad D. and Abdelfatah, Elsayed and Abdullahi, Abbas A. and Abyah, Seham S. and Adjmi, Chahd Rahyl and Agbere, Fariha and Aggarwal, Savyasanchi and Ahmed, Muhammad and Ahmed, Tasnim and Ajlouni, Motasem and Akke, Mattias and AlAdwan, Hussein and Alazani, Anwaar S. and Alharbi, Zahra A. and Aljulyhi, Wajd A. and AlKubaish, Mohammed A. and Almahri, Fatima A. and Almohri, Sayed A. and Alobo, David Obeh and Alouni, Mohammed and Alqahtani, Azizah S. and Alsaigh, Omar and Althagafi, Husain and Aman, Md Aqib and Ara, Lena and Arifin and Arretche, Ignacio and Ashy, Abdulaziz and Asim, Syeda A. and Aswad, Amro and Atta, Adeel and Auer, Sören and al Azmi, Abdullah and Balogun, Toheeb and Banik, Suvo and Baibakova, Viktoriia and Baksh, Shakira A. and Bastús, Neus G. and Bayard, Christina J. and Bazgir, Adib and Beal, Louis and Biberić, Lejla and Billah, Wahid and Biswas, Ankita and Bocarsly, Joshua and Bouzidi, Montassar T. and Boydas, Esma B. and Briki, Youssef and Buchanan, Cailin and Cafiero, Mauricio and Caliste, Damien and Cao, Yi and Castañeda, Rafael E. and Chandy, Sruthy K. and Charmes, Benjamin and Chaudhuri, Shayantan and Chen, Yiming and Chen, Alexander and Chen, Jieneng and Chiu, Min-Hsueh and Circi, Defne and Contreras, Cinthya H. and Cure, Yoann and Daelman, Nathan and Dantuluri, Roshini and Davy, Thomas and Dawson, William and Didukh, Leonid and Ding, Rui and Doguwa, Aminu R. and Draxl, Claudia and Edamadaka, Sathya and Elargab, Oulaya and Ertural, Christina and Evans, Matthew L. and Fako, Edvin and Farag, Hossam and Fathurrahman, Nur A. and Fedai, Merve and Ferreira, Rodrigo P. and Fisicaro, Giuseppe and Frank, Thomas and Gaddipati, Sasi K. and Gangan, Abhijeet and Garland, Jennifer and Garrick, James and Genovese, Luigi and Ghadrdran, Maryam and Giri, Sandip and Goulet, Maxime and Goumaz, Jeremy and Gracia, Sara U. and Graham, Jacob and Graves, Gabriel and Greenman, Kevin P. and Greitemeier, Tim and Gruich, Cameron and Gu, Sophie and Guilbert, Salomé and Gundlach, Hans and Gusta, Muriel F. and Haddaoui, Mourad El and Haibel, Alexander J. and Haldar, Anubhab and Handa, Vehaan and Harb, Hassan and Harms, Nathan D. and Hasan, Abdullah Al and Hassan, Abir and He, Qiyao and {Henao-Aristizábal}, Andrés and Hoex, Bram and Hong, Sungil and Horvath, Alexander J. and Hossain, Md Shaib and Huang, Yanqi and Huang, Yuqing and Hubaiev, Kostiantyn and Intal, Donald and Inzani, Katherine and Ishimwe, Kevin and Isik, Tugba and Iyer, Gopal R. and Jager, Katharina and Janssen, Jan and Jeong, Hyewon and Jirasek, Michael and Josephson, Tyler R. and Joshi, Nisarg and Kacem, Yassir Ben and Kalapurakal, Remya A. M. and Kamath, Rakesh R. and Kanagasenthinathan, Sugan and Kang, Dohun and Kantorow, Jason and Kaygisiz, Kübra and Keceli, Murat and Keya, Farhana and Khan, Muhammad U. and Khan, Sartaaj Takrim and Kim, Hyungjun and Kister, Alexander and Klawohn, Sascha and Kovacs, Collin and Krishnan, Pranav and Kryzanowski, Maurycy and Kumar, Ritesh and Kumari, Suman and Kumbhojkar, Gourav and Kuroki, Ryo and Kushwaha, Shashank and Lederbauer, Magdalena and Lee, Jaejun and Lee, Seunghan and Lee, Jeonghwan and Li, Bingcan and Li, Calvin and Li, Zhanzhao and Li, Shi and Li, Shicheng and Liu, Chengyan and Liu, Hao and Liu, Tung Yan and Liu, Yutong and {Vina-Lopez}, Lucia and Lortaraparsert, Chayaphol and Low, Andre K. Y. and Luxford, Saffron and Madariaga, Carlos and Magar, Rishikesh and Maharana, Piyush R. and Mallela, Rahul and Mahmud, Shoaib and Mani, Natesan and Mansoor, Umair and Mansour, Omar B. and Masschelein, Cassandra and Mastej, Kinga O. and Mathanker, Ankit and Meng, Jeffrey and Mezghani, Omran and Ming, Yidong and Mitra, Rishav and Mitsakis, Michail and Miyagishima, Matthew and Mohan, Ravikumar and Mohanraj, Naveen R. and Mohanty, Trupti and Mohr, Bernadette and {Molina-Bakhos}, Francisco A. and Monat, Jeremy and Moosavi, Seyed Mohamad and Mousavi, Shayan and Moussavi, Arman and Mozumber, Rubel and Mufti, Muhammad J. and Muhammed, Diyana and Munde, Ram and Munjal, Mrigi and Márquez, José A. and Nag, Shankha and Nagaro, Giacomo and Nam, Juno and {Napoles-Duarte}, Jose M. and Nduma, Ry and Nguyen, Xuan-Vu and Norouzi, Ebrahim and Ohiro, Oluwatosin and Okabe, Ryotaro and Ordillo, Viejay and Ozawa, Shuichiro and Pagel, Sebastian and Palmer, Daniel and Pan, Angela and Pandey, Akash and Pandit, Vivek and Pandit, Prakul and Parida, Chiku and Park, Jaehee and Park, Hyunsoo and Patel, Hemangi and Pathak, Shakul and Pattnaik, Taradutt and Patyukova, Elena and Paulson, Noah and Pendyala, Deepak S. and Pepek, Erick S. and Petersen, Martin H. and Pham, Thang D. and Phutane, Aniket and Pinky, Sabila K. and Polack, Étienne and Polasik, Alison and Politi, Maria and Pongratz, Tim and Ponugoti, Akhila and Priante, Fabio and Pruyn, Thomas Michael and Puppala, Sai S. and Qazi, Mohammad A. and Quosdorf, Heike and Rabby, Gollam and Raei, Mohammad J. and Rahman, Md Habibur and Rahman, A. B. M. Ashikur and Rajasekaran, Subhashree and Rakib, Tawfiqur and Ramesh, Hemanth N. and Ranadive, Vrushali and Ranka, Karnamohit and Rankovic, Bojana and Ravichandran, Adwaith and Rašović, Ilija and Rigin, Sergei and Rios, Tatem and Rishi, Varun and Robinson, Victor Naden and Rodrigues, Lucas S. and Rodriguez, Oswaldo and Roy, Mahule and Roy, Diptendu and Roy, Subhas and M, Arokia Anto Royan and Rudzinski, Joseph F. and Sabih, Muhammad and Sahoo, Subramanyam and Sain, Srusti Bheem and Saliya, Thahira and Sampath, Vignesh and Sanchez, Jesus Diaz and Santos, Arthur S. S. and Satria, Muliady and Sayeed, Hasan M. and Schaarschmidt, Jörg and Schwaller, Philippe and Segal, Nofit and Senthilvel, Abhishec and Shabih, Sherjeel and Shah, Devanshu and Shahmoradi, Faezeh and Sharlin, Samiha and Sheriff, Killian and Shi, Qiuyu and Shuaibu, Abubakar D. and Siddiqua, Ayesha and Siddiqui, M. A. Shadab and Smalley, Darian and Smith, Benjamin and Sparks, Taylor D. and Speckhard, Daniel T. and Stojanovska, Elena and Subramanian, Akshay and Sun, Jiwon and Sun, Yunkai and Syed, Abdul W. and Ta, Souvik and Takahara, Izumi and Tallau, Kelly and Tang, Guannan and Tariq, Ans B. and Tay, Sui X. and Temirbay, Nurlybek and Tiwari, Surya P. and Tom, Febin and Trapier, Tajah and Trerayapiwat, Kasidet J. and Tripathi, Samanvya and Tuhaifa, Hawra H. and Unal, Mustafa and Uzair, Mohammad and Vasudevan, Vallabh and Vazquez, Estefania and Venturi, Victor and Verma, Rahul and Verma, Ashwini and {Vazquez-Mayagoitia}, Alvaro and Wagner, Nicholas and Wakiuchi, Araki and Wan, Hao and Wang, Liaoyaqi and Wenzel, Wolfgang and Wieczorek, Alexander and Wong, Sze H. and Wu, Yue and Xie, Tong and Yi, Andrew and Yin, Ziqi and Yuwono, Jodie A. and Zaid, Nahed A. and Zaki, Mohd and Zaman, Shehtab and Zarewa, Maimuna U. and Zehtab, Mahtab and Zhang, Baosen and Zhang, Wenyu and Zhang, Melody and Zhang, Yangfan and Zhang, Yuwen and Zhang, Runze and Zhang, Zongmin and Zhao, Huanhuan and Zheng, Yuanlong Bill and Zidani, Ramzi and Zong, Xue and Foster, Ian and Blaiszik, Ben}, + year = {2026}, + journal = {arXiv.org}, + howpublished = {https://arxiv.org/abs/2605.03205v1}, + doi = {10.48550/arXiv.2605.03205}, + langid = {english} +} diff --git a/paper/paper.md b/paper/paper.md new file mode 100644 index 000000000..66250d08c --- /dev/null +++ b/paper/paper.md @@ -0,0 +1,179 @@ +--- +title: 'datalab: federated data management infrastructure for materials chemistry and beyond' +authors: + - name: Matthew L. Evans + orcid: 0000-0002-1182-9098 + affiliation: "1, 2, 3, 4" + equal-contrib: true + corresponding: true + - name: Joshua D. Bocarsly + orcid: 0000-0002-7523-152X + affiliation: "5, 6" + equal-contrib: true + corresponding: true + - name: Benjamin Charmes + orcid: 0009-0007-9474-8632 + affiliation: "1, 4" + - name: Ben E. Smith + orcid: 0000-0001-9673-2449 + affiliation: "4" + - name: Gian-Marco Rignanese + affiliation: "2, 7" + orcid: 0000-0002-1422-1205 + - name: David Waroquiers + affiliation: "3" + orcid: 0000-0001-8943-9762 + - name: Clare P. Grey + orcid: 0000-0001-5572-192X + affiliation: "1" +affiliations: + - name: Yusuf Hamied Department of Chemistry, University of Cambridge, Cambridgeshire, United Kingdom + index: 1 + - name: Institute of Condensed Matter and Nanosciences, Université catholique de Louvain, Chemin des Étoiles 8, Louvain-la-Neuve 1348, Belgium + index: 2 + - name: Matgenix SRL, Rue Armand Bury 185, 6534 Gozée, Belgium + index: 3 + - name: datalab industries ltd., King's Lynn, Norfolk, United Kingdom + index: 4 + - name: Department of Chemistry, University of Houston, Houston, Texas, United States of America + index: 5 + - name: Texas Center for Superconductivity, University of Houston, Houston, Texas, United States of America + index: 6 + - name: WEL Research Institute, Avenue Pasteur, 6, 1300 Wavre, Belgium + index: 7 + +date: March 2026 +bibliography: paper.bib +--- + +# Summary + +*datalab* is an open-source laboratory data management platform for chemical and material sciences, consisting of a Python web server, user-friendly Vue.js web app, and an associated ecosystem of plugins and tools. +It is designed to be deployed at the level of a research group or consortium, providing functionality to track samples and their connections through the entire research lifecycle. In doing so, *datalab* seeks to enable FAIR (Findable, Accessible, Interoperable, and Reusable) [@Wilkinson2016; @Scheffler2022a] data workflows while simultaneously saving researchers time and effort in managing and analyzing experiments. + +# Statement of need + +As the quantity and diversity of scientific data collected in research laboratories grows, data management becomes an increasingly important challenge. +Research organizations, funders, and publishers have recently emphasized the importance of FAIR data. Despite this recognized need, the majority of lab data is not managed in such a way that it could be accessed and reused, even within individual research groups. +For example, a recent survey by the UK’s Physical Sciences Data Infrastructure (PSDI) found that fewer than 20% of respondents digitally managed all of their laboratory data and experiments [@Kanza2023]. +One reason for this poor adoption rate is the challenges involved in practical data management for experimental labs: diverse data types in a variety of formats, complex interconnected experiments, and constantly evolving objectives. +Therefore, data management platforms that can enable user-friendly recording of diverse data and metadata have the potential to enable greater reproducibility, accelerated data analysis, and improved data sharing. +The recent growth in data-driven research and artificial intelligence (AI) for chemistry and materials underscores the importance of effective research data management [@Mroz2025]. + +# State of the field + +Currently, there are several open-source electronic lab notebooks (ELNs) or Laboratory Information Management Systems (LIMSs) aimed at performing FAIR research data management in the experimental chemical and materials sciences. Each package draws a different boundary around which types of data and aspects of the data lifecycle they intend to cover. Importantly, these frameworks are often relied on not just for data recording, but also robust backed-up file storage, syncing from remote scientific instruments, data sharing and collaboration, and data analysis and visualization [@Higgins2022]. +Exemplary open-source frameworks include: [NOMAD](https://nomad-lab.eu/nomad-lab) [@Scheidgen2023; @Ghiringhelli2023], [openBIS](https://openbis.ch) [@Barillari2016; @Lam2025], [eLabFTW](https://www.elabftw.net), [Chemotion](https://chemotion.net) [@Tremouilhac2017a; @Herrmann2025], [Kadi4Mat](https://kadi.iam.kit.edu) [@Brandt2021; @Schlabach2024], and [SampleDB](https://scientific-it-systems.iffgit.fz-juelich.de/SampleDB) [@Rhiem2021]. + +One dividing line between the various existing approaches is the balance between extensibility and ease-of-use, with some platforms being highly customizable at the expense of requiring substantial technical expertise. Another differentiator is in the manner in which raw data makes its way into the platform, i.e., whether the user must pre-process data into a generic format or if the platform can directly ingest data from instruments. + +# Software design + +*datalab* consists of a server (using [Flask](https://flask.palletsproject.com)), a database ([MongoDB](https://mongodb.com)), and a web frontend (written in [Vue.js](https://vuejs.org)), alongside a growing ecosystem of plugins and tools. +This stack is designed to be deployed and configured for a single research group or research consortium, allowing users to record all the data involved in their research projects. + +The core data model is inspired by how researchers traditionally record data in physical lab notebooks. Each sample or device (generically, `Item`) studied in a lab is stored as a document in the database. Unique identifying information for each item is recorded along with information such as its chemical formula, synthesis procedure, chemical hazard statements, qualitative notes, etc., along with data files and visualizations from any measurements that were performed on that sample. Similar interfaces are also provided for managing a chemical inventory and lab equipment. + +Data models for `Item`s are described with [Pydantic](https://pydantic.dev) models and are kept relatively lightweight: enforcing critical information (e.g., a unique id and date), specifying useful optional fields (e.g., chemical formula), and always including arbitrary free-text fields to allow for the flexibility required by the often-unpredictable nature of laboratory research. If needed, the base schemas can also be extended to specify additional fields for specialized `Item` types studied in a given lab (e.g., battery cells). The user is provided with a simple web interface to input, edit, and view data, including interactive views of their measurement data. A Python client library, [datalab-org/datalab-api](https://github.com/datalab-org/datalab-api) is also available to streamline access and perform more complicated tasks, such as aggregated searches, or downstream tasks like syncing with a chemical inventory system ([datalab-industries/datalab-cheminventory-plugin](https://github.com/datalab-industries/datalab-cheminventory-plugin)) or other remote filesystem/API ([datalab-industries/datalab-beholder-plugin](https://github.com/datalab-industries/datalab-beholder-plugin)). + +In addition to data and metadata, *datalab* also emphasizes recording connections between items, such as linking samples to the starting materials (or other samples) that were used in their synthesis, or linking battery test cells to the electrode materials that were used in the cell construction. This creates an evolving graph of connected research items that is stored in the database and displayed in the GUI, allowing researchers to explore measurements from associated items. + +The measurement data, which are quite variable and diverse in typical experimental labs, are handled via modular "data blocks" that consist of raw data parsers, validators, and visualizers. Data blocks can be written as applications in Python, generally using [Bokeh](https://bokeh.org) for interactive visualizations. A core set of commonly used blocks is included in the core of *datalab*, while others can be added using a plugin system. + +These data blocks vary in complexity, from simple CSV parsing and visualization, up to two-way reactive components that can perform automated analysis (normalization, baseline corrections), capture additional out-of-band metadata from the user (e.g., the wavelength used in an X-ray diffraction experiment, where not provided in the file), and index particular properties or metadata in the database for future search (e.g., peak positions or $d$-spacings from an X-ray diffraction experiment). +Data blocks can be rendered either synchronously or asynchronously, via simple scheduled background tasks, depending on the application. Table 1 provides a non-exhaustive summary of the support for different characterization techniques and file types in the current version of the *datalab* core. + +Much of this support is provided by other third-party libraries (many of which the *datalab* community helps to maintain), for example, `galvani` [@Kerr2017] for electrochemical data and `nmrglue` [@Helmus2013] for NMR data, which are wrapped in *datalab* data blocks to provide a user-friendly experience and indexing of metadata for use in the rest of the system. + +```{=latex} +\begin{table}[ht] +\caption{Non-exhaustive summary of supported characterization techniques and file formats in version 0.7. Some are included in the \emph{datalab} core, while others are open-source plugins built using the extensible plugin system (denoted by $\star$).} +\label{tbl:formats} +\small +\begin{tabular}{|l|l|} +\hline +\textbf{Technique} & \textbf{File formats \& vendors} \\ +\hline +X-ray diffraction (XRD) & \begin{tabular}[t]{@{}l@{}} - Plain text semi-standardized \texttt{.xy}, \texttt{.xye}, \texttt{.dat} \\ - Bruker \texttt{.raw} \\ - Panalytical XRDML \\ - Rigaku \texttt{.rasx} \end{tabular} \\ +\hline +Nuclear magnetic resonance (NMR) & \begin{tabular}[t]{@{}l@{}} - JCAMP-DX \\ - Bruker project folders (zipped) \\ - JEOL \texttt{.jdf} \end{tabular} \\ +\hline +Electrochemical cycling and cyclic voltammetry & \begin{tabular}[t]{@{}l@{}} - Battery Data Format \texttt{.bdf} \\ - BioLogic \texttt{.mpr} \\ - Arbin \texttt{.res} \\ - Neware \texttt{.nda}, \texttt{.ndax} \\ - Plain text and Excel exports from various \\ \quad vendor software packages (e.g., Landt, Ivium, \\ \quad Arbin, CH Instruments) \end{tabular} \\ +\hline +Electrochemical impedance spectroscopy (EIS) & \begin{tabular}[t]{@{}l@{}} - BioLogic \texttt{.mpr} \\ - Ivium-exported \texttt{.txt} \end{tabular} \\ +\hline +Ultraviolet-visible (UV-Vis) spectroscopy & \begin{tabular}[t]{@{}l@{}} - Plain text and Excel exports \\ \quad from various vendor software packages \end{tabular} \\ +\hline +Raman spectroscopy and microscopy & - Renishaw \texttt{.wdf} \\ +\hline +Fourier-transform infrared (FTIR) spectroscopy & - Agilent \texttt{.asc} \\ +\hline +Mass spectrometry (MS) & - Mettler-Toledo \texttt{.asc} \\ +\hline +${\star}$ Differential scanning calorimetry (DSC) & - TA Instruments text exports \\ +\hline +${\star}$ Online mass spectrometry (OMS) & - Pfeiffer binary and plain text exports \\ +\hline +${\star}$ X-ray photoelectron spectroscopy (XPS) & - Thermo Scientific VGD (\texttt{.vgd}) \\ +\hline +${\star}$ In situ XRD, NMR \& UV-Vis & \begin{tabular}[t]{@{}l@{}} - Semi-standardized file hierarchies combining \\ \quad \emph{operando} electrochemical or temperature \\ \quad data alongside characterization \end{tabular} \\ +\hline +\end{tabular} +\end{table} +``` + +Data and metadata can be readily exported from *datalab* via the GUI or API. +Where available, *datalab* also aims to export in community-accepted standardized formats, such as the [Battery Data Format (BDF)](https://battery-data-alliance.github.io/battery-data-format/) for electrochemical cycling data, as well as exporting in generic container formats with well-reported schemas, such as JSON, CSV, or HDF5. +Export to the recently standardized [ELNFileFormat](https://github.com/TheELNConsortium/TheELNFileFormat) [@ELNFileFormat] is also supported, allowing for metadata pertaining to multiple items to be provided and combined with any uploaded files in a single archive. *datalab* allows ELN exports not just of individual items or user-defined collections of items, but also of entire subgraphs of related entries to an item. + +We found that one of the major barriers is actually the deployment of a system such as *datalab*; this makes adoption of any self-hosted system (such as those listed above) difficult without significant institutional support, and provides another source of vendor lock-in, even for otherwise open-source projects. +To combat this, *datalab* is accompanied by a series of automated deployment rules, written as [Ansible playbooks](https://ansible.com), that can be used alongside [Terraform](https://developer.hashicorp.com/terraform)/[OpenTofu](https://opentofu.org/) to (optionally) provision a cloud server and deploy a robust *datalab* instance with encrypted offsite backups (using [Borg](https://www.borgbackup.org/)) and a full monitoring stack (using the open-source [Grafana](https://grafana.com/) stack). + +# Research impact statement + +*datalab* is in use in a variety of academic research labs, consortia, and companies across the world. +There exists an opt-in federation, where each individual deployment is encouraged to register a (mutable) canonical URL and a prefix [datalab-org/datalab-federation](https://github.com/datalab-org/datalab-federation) in order to ensure item IDs are globally unique and to provide persistent URLs for physical labelling and data sharing via the resolver service at [purl.datalab-org.io](https://purl.datalab-org.io). +At the time of writing, there are 12 registered *datalab* instances with at least 3 others that remain unregistered, accounting for around 250 users. +Levels of engagement vary among the users, but many are using *datalab* on a near-daily basis to manage all of the data associated with their research projects. +Across the federation, we estimate that *datalab* is being used to track over ten thousand physical research objects. + +In addition to the *datalab* core, a growing plugin ecosystem has developed, including plugins authored by developers outside the core team. +Digital data management with *datalab* has also enabled novel research directions, such as the creation of LLM-based AI agents to accelerate research tasks [@Jablonka2023; @Zimmermann2025]. +Two examples of this, [yellowhammer](https://github.com/datalab-org/yellowhammer) [@Zimmermann2025] and [guillemot](https://github.com/datalab-org/guillemot) [@Roy2026] make use of the *datalab* API to pull in data and perform automated data curation or analysis on the behalf of a user. + +# Future + +While there have already been several stable releases of *datalab* over the last 5 years, *datalab* development is still active across many fronts. +We expect *datalab* to continue to scale horizontally to new domains and measurement techniques via the growing userbase and plugin ecosystem. Additionally, some broader technical changes are planned to broaden the applicability and maximize user friendliness going forward. + +The technical roadmap for a *datalab* v1.0 release includes: + +- A rework of the schema system for easier customizability, sharing, and extension by deployments, as well as the ability to provide semantic annotations via LinkML [@Moxon2026]; this will be accommodated by a rework of the user interface to allow custom schemas to use the same user-friendly web components that exist in the core *datalab* models for rich text input and relationship tracking. +- Further improvements to the *datalab* plugin ecosystem, including enhancements of the base data block with features such as caching, offloading compute, and UI generation, providing clean interfaces to make it easier for contributors to build powerful extensions to handle arbitrary data types. +- An expansion of existing prototypes for AI-driven user interfaces, building on existing work on conversational interfaces [@Jablonka2023] and coding agents ([datalab-org/yellowhammer](https://github.com/datalab-org/yellowhammer)) [@Zimmermann2025], with the aim of allowing users to create rich and expressive pipelines via end user programming. + +# AI usage disclosure + +While the initial development of *datalab* (architecture, proof-of-concept) was performed without the use of AI, recent development (approximately v0.6.3 onwards) has made use of LLM-based AI coding harnesses (e.g., OpenAI's Codex, Anthropic's Claude Code) in various parts of *datalab* and related development, including: code generation for prototyping new features and interfaces, refactoring, code review (usually initial reviews for PRs from external authors), and generation of test cases. +Models used include OpenAI's GPT-5.x series, Anthropic's Claude Sonnet and Opus series from versions 3.7 and above, and open weights models such as Qwen3.6. + +Every pull request is still thoroughly reviewed by a human and we maintain an extensive test suite that runs on each pull request to catch regressions across the project; the human authors and reviewers are ultimately responsible for the code that is merged. + +AI tools were used in a limited way in the preparation of this manuscript, for proofreading and formatting only. + +# Acknowledgements + +M.L.E. thanks the Leverhulme Trust for funding via an Early Career Fellowship, as well as the BEWARE scheme of the Wallonia-Brussels Federation for previous funding under the European Commission's Marie Curie-Skłodowska Action (COFUND 847587). +M.L.E., J.D.B., and C.P.G. acknowledge funding from the European Union's Horizon 2020 research and innovation programme under grant agreement 957189 (DOI: 10.3030/957189), the Battery Interface Genome – Materials Acceleration Platform (BIG-MAP), where *datalab* was prototyped as an external stakeholder project. +J.D.B. was supported by the Faraday Institution CATMAT project (FIRG016) during initial development of *datalab* and is currently supported by the Welch Foundation (E-2179-20240404). + +# Conflict of interest + +M.L.E. is the founder and director of datalab industries ltd. + +# Author contributions + +M.L.E. and J.D.B. conceived the project and designed the architecture. +M.L.E. and J.D.B. implemented the first release of the software. +M.L.E., B.C., B.E.S., and J.D.B. developed and maintain the software. +M.L.E., J.D.B., G-M.R., D.W., and C.P.G. acquired funding and supervised the project.