Install
Stable version from CRAN
install.packages("ckanr")Development version from GitHub
remotes::install_github("ropensci/ckanr")The default base CKAN URL is https://demo.ckan.org/. To change it,
use ckanr_setup(), or use the url parameter in
each function call.
To set one or both, run:
# restores default CKAN url to http://data.techno-science.ca/
ckanr_setup()
# Just set url
ckanr_setup(url = "http://data.techno-science.ca/")
# set url and key
ckanr_setup(url = "http://data.techno-science.ca/", key = "my-ckan-api-key")Changes
changes(limit = 2, as = "table")[, 1:4]
#> user_id timestamp
#> 1 27778230-2e90-4818-9f00-bbf778c8fa09 2016-06-14T21:31:28.306231
#> 2 27778230-2e90-4818-9f00-bbf778c8fa09 2016-06-14T21:30:26.594125
#> object_id revision_id
#> 1 99f457c9-ea24-48a1-87be-b52385825b6a e2d9463d-e97c-48f5-a816-7fe26ee60dcd
#> 2 99f457c9-ea24-48a1-87be-b52385825b6a 9d846213-1389-4dab-bfe5-77dd3256995aList datasets
package_list(as = "table")
#> [1] "artifact-data-agriculture"
#> [2] "artifact-data-aviation"
#> [3] "artifact-data-bookbinding"
#> [4] "artifact-data-chemistry"
#> [5] "artifact-data-communications"
#> [6] "artifact-data-computing-technology"
#> [7] "artifact-data-domestic-technology"
#> [8] "artifact-data-energy-electric"
#> [9] "artifact-data-exploration-and-survey"
#> [10] "artifact-data-fisheries"
...List tags
tag_list('aviation', as = 'table')
#> vocabulary_id display_name
#> 1 NA Aviation
#> 2 NA Canada Aviation and Space Museum
#> id name
#> 1 cc1db2db-b08b-4888-897f-a17eade2461b Aviation
#> 2 8d05a650-bc7b-4b89-bcc8-c10177e60119 Canada Aviation and Space MuseumShow tags
Subset for readme brevity
tag_show('Aviation')
#> <CKAN Tag> cc1db2db-b08b-4888-897f-a17eade2461b
#> Name: Aviation
#> Display name: Aviation
#> Vocabulary id:
#> No. Packages: 2
#> Packages (up to 5): artifact-data-aviation, cstmc-smstc-artifacts-artefactList groups
group_list(as = 'table')[, 1:3]
#> display_name description
#> 1 Communications
#> 2 Domestic and Industrial Technology
#> 3 Everything
#> 4 Location
#> 5 Resources
#> 6 Scientific Instrumentation
#> 7 Transportation
#> title
#> 1 Communications
#> 2 Domestic and Industrial Technology
#> 3 Everything
#> 4 Location
#> 5 Resources
#> 6 Scientific Instrumentation
#> 7 TransportationShow groups
Subset for readme brevity
group_show('communications', as = 'table')$users
#> openid about capacity name created
#> 1 NA <NA> admin marc 2014-10-24T14:44:29.885262
#> 2 NA admin sepandar 2014-10-23T19:40:42.056418
#> email_hash sysadmin
#> 1 a32002c960476614370a16e9fb81f436 FALSE
#> 2 10b930a228afd1da2647d62e70b71bf8 TRUE
#> activity_streams_email_notifications state number_of_edits
#> 1 FALSE active 516
#> 2 FALSE active 44
#> number_administered_packages display_name fullname
#> 1 40 marc <NA>
#> 2 1 sepandar
#> id
#> 1 27778230-2e90-4818-9f00-bbf778c8fa09
#> 2 b50449ea-1dcc-4d52-b620-fc95bf56034bShow a package
package_show('34d60b13-1fd5-430e-b0ec-c8bc7f4841cf', as = 'table')$resources[, 1:10]
#> resource_group_id cache_last_updated
#> 1 ea8533d9-cdc6-4e0e-97b9-894e06d50b92 NA
#> 2 ea8533d9-cdc6-4e0e-97b9-894e06d50b92 NA
#> 3 ea8533d9-cdc6-4e0e-97b9-894e06d50b92 NA
#> 4 ea8533d9-cdc6-4e0e-97b9-894e06d50b92 NA
#> 5 ea8533d9-cdc6-4e0e-97b9-894e06d50b92 NA
#> revision_timestamp webstore_last_updated
#> 1 2016-06-13T20:05:16.818800 NA
#> 2 2014-11-04T02:59:50.567068 NA
#> 3 2014-11-05T21:23:58.533397 NA
#> 4 2014-11-05T21:25:16.848423 NA
#> 5 2016-06-13T20:06:50.013746 NA
#> id size state hash
#> 1 be2b0af8-24a8-4a55-8b30-89f5459b713a NA active
#> 2 7d65910e-4bdc-4f06-a213-e24e36762767 NA active
#> 3 97622ad7-1507-4f6a-8acb-14e826447389 NA active
#> 4 7a72498a-c49c-4e84-8b10-58991de10df6 NA active
#> 5 7e2cb5de-550d-41a8-ab9d-b2ec35b6671a NA active
#> description format
#> 1 XML Dataset XML
#> 2 Data dictionary for CSTMC artifact datasets. XLS
#> 3 Tips for using the artifacts datasets. .php
#> 4 Tips for using the artifacts datasets. .php
#> 5 Jeux de données XML XMLSearch for packages
out <- package_search(q = '*:*', rows = 2, as = "table")$results
out[, !names(out) %in% 'resources'][, 1:10]
#> license_title maintainer relationships_as_object private
#> 1 Open Government Licence - Canada NULL FALSE
#> 2 Open Government Licence - Canada NULL FALSE
#> maintainer_email revision_timestamp
#> 1 2014-10-28T21:27:57.475091
#> 2 2014-10-28T20:40:55.803602
#> id metadata_created
#> 1 99f457c9-ea24-48a1-87be-b52385825b6a 2014-10-24T17:39:06.411039
#> 2 443cb020-f2ae-48b1-be67-90df1abd298e 2014-10-28T20:39:23.561940
#> metadata_modified author
#> 1 2016-06-14T21:31:27.983485
#> 2 2016-06-14T18:59:17.786219List all datasets of one organization
organization_show() with
include_datasets = TRUE returns only the first 10 datasets.
The server caps that embedded list, and the endpoint takes no
limit parameter. To get every dataset of an organization,
search for them with package_search() and page through the
results. Each request returns at most 1000 rows, so repeat the search
with start until you hold them all.
org <- "my-organization-name"
page <- package_search(q = "*:*", fq = sprintf("organization:%s", org), rows = 1000)
total <- page$count
results <- page$results
start <- 1000
while (length(results) < total) {
page <- package_search(q = "*:*", fq = sprintf("organization:%s", org),
rows = 1000, start = start)
results <- c(results, page$results)
start <- start + 1000
}
length(results)Search for resources
resource_search(q = 'name:data', limit = 2, as = 'table')
#> $count
#> [1] 74
#>
#> $results
#> resource_group_id cache_last_updated webstore_last_updated
#> 1 01a82e52-01bf-4a9c-9b45-c4f9b92529fa NA NA
#> 2 01a82e52-01bf-4a9c-9b45-c4f9b92529fa NA NA
#> id size state last_modified hash
#> 1 e179e910-27fb-44f4-a627-99822af49ffa NA active NA
#> 2 ba84e8b7-b388-4d2a-873a-7b107eb7f135 NA active NA
#> description format mimetype_inner url_type
#> 1 XML Dataset XML NA NA
#> 2 Data dictionary for CSTMC artifact datasets. XLS NA NA
#> mimetype cache_url name
#> 1 NA NA Artifact Data - Exploration and Survey (XML)
#> 2 NA NA Data Dictionary
#> created
#> 1 2014-10-28T15:50:35.374303
#> 2 2014-11-03T18:01:02.094210
#> url
#> 1 http://source.techno-science.ca/datasets-donn%C3%A9es/artifacts-artefacts/groups-groupes/exploration-and-survey-exploration-et-leve.xml
#> 2 http://source.techno-science.ca/datasets-donn%C3%A9es/artifacts-artefacts/cstmc-artifact-data-dictionary-dictionnaire-de-donnees-artefacts-smstc.xls
#> webstore_url position revision_id resource_type
#> 1 NA 0 a22e6741-3e89-4db0-a802-ba594b1c1fad NA
#> 2 NA 1 da1f8585-521d-47ef-8ead-7832474a3421 NAckanr’s dplyr interface
ckanr provides a dplyr SQL interface to
CKAN’s datastore. The primary interface uses the DBI connection
directly. You can access any resource in the datastore with its CKAN
resource ID.
This works only for resources in the datastore. They show the green “Data API” button in CKAN.
ckan <- ckanr::src_ckan("https://my.ckan.org/")
con <- ckan$con
res_id <- "my-ckan-resource-id"
dplyr::tbl(con, res_id) |> dplyr::collect()The interface is read-only. Filter or limit large resources before
you call collect(), because the DataStore returns the query
result to R.
Worked dbplyr example
This example builds a query for a resource and then sends it to CKAN. Replace the URL, key, resource ID, and column names with values from your CKAN site.
First create a connection and list the resources that have a DataStore table.
ckan <- ckanr::src_ckan(
url = "https://my.ckan.org/",
key = Sys.getenv("CKAN_API_KEY")
)
con <- ckan$con
DBI::dbListTables(con)Choose a resource ID from that list. The query stays in R until
dplyr::collect() runs.
resource_id <- "my-resource-id"
data <- dplyr::tbl(con, resource_id)
query <- data |>
dplyr::filter(status == "active") |>
dplyr::select(name, amount) |>
dplyr::mutate(amount_thousands = amount / 1000) |>
dplyr::arrange(dplyr::desc(amount)) |>
head(10)
dplyr::show_query(query)
result <- dplyr::collect(query)
resultThe verbs build SQL on the server. collect() runs the
SQL and returns the rows as an R tibble. Do not use
copy_to(), compute(), or write functions with
this connection. The CKAN DataStore interface is read-only.
Example of using a different CKAN API
See ckanr::servers() for a list of CKAN servers. There
are 127 as of 2020-07-29.
The UK Natural History Museum
Website: https://data.nhm.ac.uk/
List datasets
ckanr_setup(url = "https://data.nhm.ac.uk")
package_list(as = "table")
#> [1] "3d-cetacean-scanning"
#> [2] "3d-laser-scan-of-nhm-pv-r-9372-palaeosauropus-sp"
#> [3] "abyssline"
#> [4] "african-spiny-solanum"
#> [5] "aleyrodoidea-slide-collection"
#> [6] "alice-test-images"
#> [7] "alignments-of-co1-nd1-and-16s-rrna-for-the-land-snail-corilla"
#> [8] "al-sabouni-et-al-reproducibility"
#> [9] "american-phlebotominae-nhm"
#> [10] "an-influence-of-environmental-variability-on-insects-wing-shape-a-case-study-of-british-odonata"
...Tags
list
head(tag_list(as = "table"))
#> vocabulary_id display_name id name
#> 1 NA 16S a2fabd81-89df-4520-96d2-d5cd7bdcf094 16S
#> 2 NA 18S 1f9670fe-c4c5-40bb-a02a-8889a11788e7 18S
#> 3 NA 28S 71e10e81-c643-4b67-af02-4f9516b4238b 28S
#> 4 NA 3d 098ccfee-a0fe-451d-b748-b617086d146c 3d
#> 5 NA 3D b26c5cce-dc41-40c6-9cbf-f966aa75d458 3D
#> 6 NA 3D modelling 119868d7-1753-4c35-8641-14681dc472a7 3D modellingshow
tag_show('arthropods', as = 'table')
#> $vocabulary_id
#> NULL
#>
#> $display_name
#> [1] "arthropods"
#>
#> $id
#> [1] "f9245868-f4cb-4c85-a59d-11692db19e86"
#>
#> $name
#> [1] "arthropods"Packages
search
out <- package_search(q = '*:*', rows = 2, as = 'table')
out$results[, 1:10]
#> license_title maintainer contributors relationships_as_object
#> 1 Creative Commons Attribution NA NULL
#> 2 CC0-1.0 NA <NA> NULL
#> private maintainer_email num_tags affiliation update_frequency
#> 1 FALSE NA 1 Natural History Museum
#> 2 FALSE NA 1 <NA> weekly
#> id
#> 1 d68e20f4-a56d-4a8a-a8d7-dc478ba64c76
#> 2 56e711e6-c847-4f99-915a-6894bb5c5deashow
package_show(id = "56e711e6-c847-4f99-915a-6894bb5c5dea", as = "table")
#> $domain
#> [1] "data.nhm.ac.uk"
#>
#> $license_title
#> [1] "CC0-1.0"
#>
#> $maintainer
#> NULL
#>
#> $relationships_as_object
...The National Geothermal Data System
Website: http://geothermaldata.org/
ckanr_setup("http://search.geothermaldata.org")
x <- package_search(q = '*:*', rows = 1)
x$results
#> [[1]]
#> <CKAN Package> 787986f3-fd65-4e99-b28e-077c69933c76
#> Title: Hawthorne Nevada Deep Direct-Use Feasibility Study - Data Used for Geothermal Resource Conceptual Modeling and Power Capacity Estimates Hawthorne_fracture_data_HWAAD-2A_HWAAD-3.xlsx
#> Creator/Modified: 2020-05-07T23:35:47.237011 / 2020-05-07T23:35:47.328456
#> Resources (up to 5): Hawthorne_fracture_data_HWAAD-2A_HWAAD-3.xlsx
#> Tags (up to 5): 2-meter temperatures, DDU, HWAAD-2, HWAAD-2A, HWAAD-3
#> Groups (up to 5):
NA