Conference paper Open Access

I'll take that to go: Big data bags and minimal identifiers for exchange of large, complex datasets

Chard, Kyle; D'Arcy, Mike; Heavner, Ben; Foster, Ian; Kesselman, Carl; Madduri, Ravi; Rodriguez, Alexis; Soiland-Reyes, Stian; Goble, Carole; Clark, Kristi; Deutsch, Eric W.; Dinov, Ivo; Price, Nathan; Toga, Arthur


JSON Export

{
  "files": [
    {
      "links": {
        "self": "https://zenodo.org/api/files/1b457874-0375-4648-aad1-12f61bd424ed/bagminid.pdf"
      }, 
      "checksum": "md5:91195ab648922564b86d629e83ea88d8", 
      "bucket": "1b457874-0375-4648-aad1-12f61bd424ed", 
      "key": "bagminid.pdf", 
      "type": "pdf", 
      "size": 713184
    }
  ], 
  "owners": [
    4363
  ], 
  "doi": "10.1109/BigData.2016.7840618", 
  "stats": {
    "version_unique_downloads": 135.0, 
    "unique_views": 291.0, 
    "views": 305.0, 
    "downloads": 148.0, 
    "unique_downloads": 135.0, 
    "version_unique_views": 290.0, 
    "volume": 105551232.0, 
    "version_downloads": 148.0, 
    "version_views": 304.0, 
    "version_volume": 105551232.0
  }, 
  "links": {
    "doi": "https://doi.org/10.1109/BigData.2016.7840618", 
    "latest_html": "https://zenodo.org/record/820878", 
    "bucket": "https://zenodo.org/api/files/1b457874-0375-4648-aad1-12f61bd424ed", 
    "badge": "https://zenodo.org/badge/doi/10.1109/BigData.2016.7840618.svg", 
    "html": "https://zenodo.org/record/820878", 
    "latest": "https://zenodo.org/api/records/820878"
  }, 
  "created": "2017-06-29T10:23:44.346770+00:00", 
  "updated": "2019-04-10T04:15:03.185382+00:00", 
  "conceptrecid": "820877", 
  "revision": 8, 
  "id": 820878, 
  "metadata": {
    "access_right_category": "success", 
    "part_of": {
      "pages": "319-328", 
      "title": "2016 IEEE International Conference on Big Data (Big Data)"
    }, 
    "doi": "10.1109/BigData.2016.7840618", 
    "description": "<p><em>Big data workflows</em> often require the assembly and exchange of complex, multi-element datasets. For example, in biomedical applications, the input to an analytic pipeline can be a dataset consisting thousands of images and genome sequences assembled from diverse repositories, requiring a description of the contents of the dataset in a concise and unambiguous form. Typical approaches to creating datasets for big data workflows assume that all data reside in a single location, requiring costly data marshaling and permitting errors of omission and commission because dataset members are not explicitly specified.</p>\n\n<p>We address these issues by proposing simple methods and tools for assembling, sharing, and analyzing large and complex datasets that scientists can easily integrate into their daily workflows. These tools combine a simple and robust method for describing data collections (<strong>BDBags</strong>), data descriptions (<strong>Research Objects</strong>), and simple persistent identifiers (<strong>Minids</strong>) to create a powerful ecosystem of tools and services for big data analysis and sharing.</p>\n\n<p>We present these tools and use biomedical case studies to illustrate their use for the rapid assembly, sharing, and analysis of large datasets.</p>", 
    "contributors": [
      {
        "affiliation": "The University of Chicago", 
        "type": "Other", 
        "name": "Jung, Segun"
      }
    ], 
    "title": "I'll take that to go: Big data bags and minimal identifiers for exchange of large, complex datasets", 
    "license": {
      "id": "CC-BY-4.0"
    }, 
    "relations": {
      "version": [
        {
          "count": 1, 
          "index": 0, 
          "parent": {
            "pid_type": "recid", 
            "pid_value": "820877"
          }, 
          "is_last": true, 
          "last_child": {
            "pid_type": "recid", 
            "pid_value": "820878"
          }
        }
      ]
    }, 
    "imprint": {
      "publisher": "IEEE", 
      "isbn": "978-1-4673-9005-7"
    }, 
    "communities": [
      {
        "id": "bioexcel"
      }, 
      {
        "id": "linkeddata"
      }
    ], 
    "grants": [
      {
        "code": "675728", 
        "links": {
          "self": "https://zenodo.org/api/grants/10.13039/501100000780::675728"
        }, 
        "title": "Centre of Excellence for Biomolecular Research", 
        "acronym": "BioExcel", 
        "program": "H2020", 
        "funder": {
          "doi": "10.13039/501100000780", 
          "acronyms": [
            "EC"
          ], 
          "name": "European Commission", 
          "links": {
            "self": "https://zenodo.org/api/funders/10.13039/501100000780"
          }
        }
      }
    ], 
    "keywords": [
      "Big Data", 
      "data analysis", 
      "BDBags", 
      "Big Data analysis", 
      "Big Data bags", 
      "Big Data sharing", 
      "Minid", 
      "data assembling", 
      "data collections", 
      "data descriptions", 
      "datasets", 
      "identifiers", 
      "research objects", 
      "Encoding", 
      "Metadata", 
      "Payloads", 
      "Robustness", 
      "Software", 
      "Uniform resource locators", 
      "bdbag"
    ], 
    "publication_date": "2016-12-05", 
    "creators": [
      {
        "affiliation": "The University of Chicago and Argonne National Laboratory, Chicago IL, USA", 
        "name": "Chard, Kyle"
      }, 
      {
        "affiliation": "University of Southern California, Los Angeles, CA, USA", 
        "name": "D'Arcy, Mike"
      }, 
      {
        "affiliation": "Institute for Systems Biology, Seattle, WA, USA", 
        "name": "Heavner, Ben"
      }, 
      {
        "affiliation": "The University of Chicago and Argonne National Laboratory, Chicago IL, USA", 
        "name": "Foster, Ian"
      }, 
      {
        "affiliation": "University of Southern California, Los Angeles, CA, USA", 
        "name": "Kesselman, Carl"
      }, 
      {
        "affiliation": "The University of Chicago and Argonne National Laboratory, Chicago IL, USA", 
        "name": "Madduri, Ravi"
      }, 
      {
        "affiliation": "The University of Chicago and Argonne National Laboratory, Chicago IL, USA", 
        "name": "Rodriguez, Alexis"
      }, 
      {
        "affiliation": "The University of Manchester, Manchester, UK", 
        "name": "Soiland-Reyes, Stian"
      }, 
      {
        "affiliation": "The University of Manchester, Manchester, UK", 
        "name": "Goble, Carole"
      }, 
      {
        "affiliation": "University of Southern California, Los Angeles, CA, USA", 
        "name": "Clark, Kristi"
      }, 
      {
        "affiliation": "Institute for Systems Biology, Seattle, WA, USA", 
        "name": "Deutsch, Eric W."
      }, 
      {
        "affiliation": "The University of Michigan, Ann Arbor, MI, USA", 
        "name": "Dinov, Ivo"
      }, 
      {
        "affiliation": "Institute for Systems Biology, Seattle, WA, USA", 
        "name": "Price, Nathan"
      }, 
      {
        "affiliation": "University of Southern California, Los Angeles, CA, USA", 
        "name": "Toga, Arthur"
      }
    ], 
    "meeting": {
      "dates": "2016-12-05 / 2016-12-08", 
      "title": "2016 IEEE International Conference on Big Data", 
      "acronym": "Big Data", 
      "url": "http://cci.drexel.edu/bigdata/bigdata2016/", 
      "session": "8", 
      "place": "Washington, DC, USA", 
      "session_part": "BigD491"
    }, 
    "access_right": "open", 
    "resource_type": {
      "subtype": "conferencepaper", 
      "type": "publication", 
      "title": "Conference paper"
    }, 
    "related_identifiers": [
      {
        "scheme": "url", 
        "identifier": "https://static.aminer.org/pdf/fa/bigdata2016/BigD418.pdf", 
        "relation": "isIdenticalTo"
      }, 
      {
        "scheme": "url", 
        "identifier": "https://www.research.manchester.ac.uk/portal/files/45989205/bagminid.pdf", 
        "relation": "isIdenticalTo"
      }, 
      {
        "scheme": "url", 
        "identifier": "http://bd2k.ini.usc.edu/tools/", 
        "relation": "isSupplementedBy"
      }, 
      {
        "scheme": "url", 
        "identifier": "https://github.com/ini-bdds/bdbag", 
        "relation": "isSupplementedBy"
      }, 
      {
        "scheme": "url", 
        "identifier": "https://www.research.manchester.ac.uk/portal/en/publications/ill-take-that-to-go(8335e672-1d85-4649-a245-56fbdb1bd423).html", 
        "relation": "isPartOf"
      }, 
      {
        "scheme": "url", 
        "identifier": "https://w3id.org/ro/bagit", 
        "relation": "cites"
      }
    ]
  }
}
305
148
views
downloads
Views 305
Downloads 148
Data volume 105.6 MB
Unique views 291
Unique downloads 135

Share

Cite as