Dataset Open Access

Resistance web archive collection derivatives

Ruest, Nick; Thurman, Alex


DCAT Export

<?xml version='1.0' encoding='utf-8'?>
<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns:adms="http://www.w3.org/ns/adms#" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:dct="http://purl.org/dc/terms/" xmlns:dctype="http://purl.org/dc/dcmitype/" xmlns:dcat="http://www.w3.org/ns/dcat#" xmlns:duv="http://www.w3.org/ns/duv#" xmlns:foaf="http://xmlns.com/foaf/0.1/" xmlns:frapo="http://purl.org/cerif/frapo/" xmlns:geo="http://www.w3.org/2003/01/geo/wgs84_pos#" xmlns:gsp="http://www.opengis.net/ont/geosparql#" xmlns:locn="http://www.w3.org/ns/locn#" xmlns:org="http://www.w3.org/ns/org#" xmlns:owl="http://www.w3.org/2002/07/owl#" xmlns:prov="http://www.w3.org/ns/prov#" xmlns:rdfs="http://www.w3.org/2000/01/rdf-schema#" xmlns:schema="http://schema.org/" xmlns:skos="http://www.w3.org/2004/02/skos/core#" xmlns:vcard="http://www.w3.org/2006/vcard/ns#" xmlns:wdrs="http://www.w3.org/2007/05/powder-s#">
  <rdf:Description rdf:about="https://doi.org/10.5281/zenodo.3660457">
    <rdf:type rdf:resource="http://www.w3.org/ns/dcat#Dataset"/>
    <dct:type rdf:resource="http://purl.org/dc/dcmitype/Dataset"/>
    <dct:identifier rdf:datatype="http://www.w3.org/2001/XMLSchema#anyURI">https://doi.org/10.5281/zenodo.3660457</dct:identifier>
    <foaf:page rdf:resource="https://doi.org/10.5281/zenodo.3660457"/>
    <dct:creator>
      <rdf:Description rdf:about="http://orcid.org/0000-0003-1891-1112">
        <rdf:type rdf:resource="http://xmlns.com/foaf/0.1/Agent"/>
        <dct:identifier rdf:datatype="http://www.w3.org/2001/XMLSchema#string">0000-0003-1891-1112</dct:identifier>
        <foaf:name>Ruest, Nick</foaf:name>
        <foaf:givenName>Nick</foaf:givenName>
        <foaf:familyName>Ruest</foaf:familyName>
        <org:memberOf>
          <foaf:Organization>
            <foaf:name>York University</foaf:name>
          </foaf:Organization>
        </org:memberOf>
      </rdf:Description>
    </dct:creator>
    <dct:creator>
      <rdf:Description>
        <rdf:type rdf:resource="http://xmlns.com/foaf/0.1/Agent"/>
        <foaf:name>Thurman, Alex</foaf:name>
        <foaf:givenName>Alex</foaf:givenName>
        <foaf:familyName>Thurman</foaf:familyName>
        <org:memberOf>
          <foaf:Organization>
            <foaf:name>Columbia University</foaf:name>
          </foaf:Organization>
        </org:memberOf>
      </rdf:Description>
    </dct:creator>
    <dct:title>Resistance web archive collection derivatives</dct:title>
    <dct:publisher>
      <foaf:Agent>
        <foaf:name>Zenodo</foaf:name>
      </foaf:Agent>
    </dct:publisher>
    <dct:issued rdf:datatype="http://www.w3.org/2001/XMLSchema#gYear">2020</dct:issued>
    <dcat:keyword>web archives</dcat:keyword>
    <dcat:keyword>parquet</dcat:keyword>
    <dcat:keyword>dataframes</dcat:keyword>
    <dcat:keyword>Politics &amp; Elections</dcat:keyword>
    <dcat:keyword>Spontaneous Events</dcat:keyword>
    <dcat:keyword>Political participation</dcat:keyword>
    <dcat:keyword>Opposition (Political philosophy)</dcat:keyword>
    <dcat:keyword>Demonstrations</dcat:keyword>
    <dcat:keyword>Government, Resistance to</dcat:keyword>
    <dcat:keyword>Trump, Donald, 1946-</dcat:keyword>
    <dcat:keyword>Trump, Donald, 1946- --Impeachment</dcat:keyword>
    <dcat:keyword>United States</dcat:keyword>
    <dct:issued rdf:datatype="http://www.w3.org/2001/XMLSchema#date">2020-02-09</dct:issued>
    <owl:sameAs rdf:resource="https://zenodo.org/record/3660457"/>
    <adms:identifier>
      <adms:Identifier>
        <skos:notation rdf:datatype="http://www.w3.org/2001/XMLSchema#anyURI">https://zenodo.org/record/3660457</skos:notation>
        <adms:schemeAgency>url</adms:schemeAgency>
      </adms:Identifier>
    </adms:identifier>
    <dct:isVersionOf rdf:resource="https://doi.org/10.5281/zenodo.3660456"/>
    <dct:isPartOf rdf:resource="https://zenodo.org/communities/wahr"/>
    <dct:description>&lt;p&gt;Web archive derivatives of the &lt;a href="https://archive-it.org/collections/8752"&gt;Resistance&lt;/a&gt; collection from &lt;a href="https://archive-it.org/home/Columbia"&gt;Columbia University Libraries&lt;/a&gt;. The derivatives were created with the &lt;a href="https://github.com/archivesunleashed/aut/"&gt;Archives Unleashed Toolkit&lt;/a&gt; and &lt;a href="https://cloud.archivesunleashed.org/"&gt;Archives Unleashed Cloud&lt;/a&gt;.&lt;/p&gt; &lt;p&gt;The&amp;nbsp;&lt;strong&gt;cul-8752-parquet.tar.gz&lt;/strong&gt; derivatives&amp;nbsp;are&amp;nbsp;in&amp;nbsp;the &lt;a href="https://parquet.apache.org/"&gt;Apache&amp;nbsp;Parquet format&lt;/a&gt;,&amp;nbsp;which&amp;nbsp;is&amp;nbsp;a &lt;a href="http://en.wikipedia.org/wiki/Column-oriented_DBMS"&gt;columnar&amp;nbsp;storage&lt;/a&gt; format. These derivatives are generally small enough to work with on your local machine, and can be easily converted to Pandas DataFrames. See &lt;a href="https://github.com/archivesunleashed/notebooks/blob/master/datathon-nyc/parquet_pandas_stonewall.ipynb"&gt;this&lt;/a&gt; notebook for examples.&lt;/p&gt; &lt;p&gt;&lt;strong&gt;Domains&lt;/strong&gt;&lt;/p&gt; &lt;pre&gt;&lt;code class="language-java"&gt;.webpages().groupBy(ExtractDomainDF($"url").alias("url")).count().sort($"count".desc)&lt;/code&gt;&lt;/pre&gt; &lt;p&gt;Produces&amp;nbsp;a&amp;nbsp;DataFrame&amp;nbsp;with&amp;nbsp;the&amp;nbsp;following&amp;nbsp;columns:&lt;/p&gt; &lt;ul&gt; &lt;li&gt;domain&lt;/li&gt; &lt;li&gt;count&lt;/li&gt; &lt;/ul&gt; &lt;p&gt;&lt;strong&gt;Web&amp;nbsp;Pages&lt;/strong&gt;&lt;/p&gt; &lt;pre&gt;&lt;code class="language-java"&gt;.webpages().select($"crawl_date", $"url", $"mime_type_web_server", $"mime_type_tika", RemoveHTMLDF(RemoveHTTPHeaderDF(($"content"))).alias("content"))&lt;/code&gt;&lt;/pre&gt; &lt;p&gt;Produces&amp;nbsp;a&amp;nbsp;DataFrame&amp;nbsp;with&amp;nbsp;the&amp;nbsp;following&amp;nbsp;columns:&lt;/p&gt; &lt;ul&gt; &lt;li&gt;crawl_date&lt;/li&gt; &lt;li&gt;url&lt;/li&gt; &lt;li&gt;mime_type_web_server&lt;/li&gt; &lt;li&gt;mime_type_tika&lt;/li&gt; &lt;li&gt;content&lt;/li&gt; &lt;/ul&gt; &lt;p&gt;&lt;strong&gt;Web&amp;nbsp;Graph&lt;/strong&gt;&lt;/p&gt; &lt;pre&gt;&lt;code class="language-java"&gt;.webgraph()&lt;/code&gt;&lt;/pre&gt; &lt;p&gt;Produces&amp;nbsp;a&amp;nbsp;DataFrame&amp;nbsp;with&amp;nbsp;the&amp;nbsp;following&amp;nbsp;columns:&lt;/p&gt; &lt;ul&gt; &lt;li&gt;crawl_date&lt;/li&gt; &lt;li&gt;src&lt;/li&gt; &lt;li&gt;dest&lt;/li&gt; &lt;li&gt;anchor&lt;/li&gt; &lt;/ul&gt; &lt;p&gt;&lt;strong&gt;Image&amp;nbsp;Links&lt;/strong&gt;&lt;/p&gt; &lt;pre&gt;&lt;code class="language-java"&gt;.imageLinks()&lt;/code&gt;&lt;/pre&gt; &lt;p&gt;Produces&amp;nbsp;a&amp;nbsp;DataFrame&amp;nbsp;with&amp;nbsp;the&amp;nbsp;following&amp;nbsp;columns:&lt;/p&gt; &lt;ul&gt; &lt;li&gt;src&lt;/li&gt; &lt;li&gt;image_url&lt;/li&gt; &lt;/ul&gt; &lt;p&gt;&lt;a href="https://github.com/archivesunleashed/aut-docs/blob/master/current/binary-analysis.md#binary-analysis"&gt;&lt;strong&gt;Binary&amp;nbsp;Analysis&lt;/strong&gt;&lt;/a&gt;&lt;/p&gt; &lt;ul&gt; &lt;li&gt;PDFs&lt;/li&gt; &lt;li&gt;Spreadsheets&lt;/li&gt; &lt;li&gt;Text&amp;nbsp;files&lt;/li&gt; &lt;li&gt;Word&amp;nbsp;processor&amp;nbsp;files&lt;br&gt; &amp;nbsp;&lt;/li&gt; &lt;/ul&gt; &lt;p&gt;The &lt;strong&gt;cul-8752-auk.tar.gz &lt;/strong&gt;derivatives&lt;strong&gt; &lt;/strong&gt;are the &lt;a href="https://cloud.archivesunleashed.org/derivatives"&gt;standard set of web archive derivatives&lt;/a&gt; produced by the Archives Unleashed Cloud.&lt;/p&gt; &lt;ul&gt; &lt;li&gt;&lt;strong&gt;Gephi &lt;/strong&gt;file, which can be loaded into &lt;a href="https://gephi.org/"&gt;Gephi&lt;/a&gt;. It will have basic characteristics already computed and a basic layout.&lt;/li&gt; &lt;li&gt;&lt;strong&gt;Raw Network&lt;/strong&gt; file, which can also be loaded into &lt;a href="https://gephi.org/"&gt;Gephi&lt;/a&gt;. You will have to use that network program to lay it out yourself.&lt;/li&gt; &lt;li&gt;&lt;strong&gt;Full text&lt;/strong&gt; file. In it, each website within the web archive collection will have its full text presented on one line, along with information around when it was crawled, the name of the domain, and the full URL of the content.&lt;/li&gt; &lt;li&gt;&lt;strong&gt;Domains count&lt;/strong&gt; file. A text file containing the frequency count of domains captured within your web archive.&lt;/li&gt; &lt;/ul&gt;</dct:description>
    <dct:accessRights rdf:resource="http://publications.europa.eu/resource/authority/access-right/PUBLIC"/>
    <dct:accessRights>
      <dct:RightsStatement rdf:about="info:eu-repo/semantics/openAccess">
        <rdfs:label>Open Access</rdfs:label>
      </dct:RightsStatement>
    </dct:accessRights>
    <dcat:distribution>
      <dcat:Distribution>
        <dct:license rdf:resource="https://creativecommons.org/licenses/by/4.0/legalcode"/>
        <dcat:accessURL rdf:resource="https://doi.org/10.5281/zenodo.3660457"/>
      </dcat:Distribution>
    </dcat:distribution>
    <dcat:distribution>
      <dcat:Distribution>
        <dcat:accessURL>https://doi.org/10.5281/zenodo.3660457</dcat:accessURL>
        <dcat:byteSize>4912786968</dcat:byteSize>
        <dcat:downloadURL>https://zenodo.org/record/3660457/files/cul-8752-auk.tar.gz</dcat:downloadURL>
        <dcat:mediaType>application/x-tar</dcat:mediaType>
      </dcat:Distribution>
    </dcat:distribution>
    <dcat:distribution>
      <dcat:Distribution>
        <dcat:accessURL>https://doi.org/10.5281/zenodo.3660457</dcat:accessURL>
        <dcat:byteSize>12806887691</dcat:byteSize>
        <dcat:downloadURL>https://zenodo.org/record/3660457/files/cul-8752-parquet.tar.gz</dcat:downloadURL>
        <dcat:mediaType>application/x-tar</dcat:mediaType>
      </dcat:Distribution>
    </dcat:distribution>
  </rdf:Description>
</rdf:RDF>
47
9
views
downloads
All versions This version
Views 4747
Downloads 99
Data volume 75.8 GB75.8 GB
Unique views 3939
Unique downloads 44

Share

Cite as