In this report, we extract information about published JOSS papers
and generate
graphics as well as a summary table that can be downloaded and used for
further analyses.
suppressPackageStartupMessages({
library(tibble)
library(rcrossref)
library(dplyr)
library(tidyr)
library(ggplot2)
library(lubridate)
library(gh)
library(purrr)
library(jsonlite)
library(DT)
library(plotly)
library(citecorp)
library(readr)
library(rworldmap)
library(gt)
library(stringr)
library(openalexR)
library(arrow)
library(jsonlite)
})
## Keep track of the source of each column
source_track <- c()
## Determine whether to add a caption with today's date to the (non-interactive) plots
add_date_caption <- TRUE
if (add_date_caption) {
dcap <- lubridate::today()
} else {
dcap <- ""
}
## Get list of countries and populations (2022) from the rworldmap/gt packages
data("countrySynonyms")
country_names <- countrySynonyms |>
select(-ID) |>
pivot_longer(names_to = "tmp", values_to = "name", -ISO3) |>
filter(name != "") |>
select(-tmp)
## Country population data from the World Bank (https://data.worldbank.org/indicator/SP.POP.TOTL),
## distributed via the gt R package
country_populations <- countrypops |>
filter(year == 2022)
## Read archived version of summary data frame, to use for filling in
## information about software repositories (due to limit on API requests)
## Sort by the date when software repo info was last obtained
papers_archive <- readRDS(gzcon(url("https://github.com/openjournals/joss-analytics/blob/gh-pages/joss_submission_analytics.rds?raw=true"))) %>%
dplyr::arrange(!is.na(repo_info_obtained), repo_info_obtained)
## Similarly for citation analysis, to avoid having to pull down the
## same information multiple times
citations_archive <- readr::read_delim(
url("https://github.com/openjournals/joss-analytics/blob/gh-pages/joss_submission_citations.tsv?raw=true"),
col_types = cols(.default = "c"), col_names = TRUE,
delim = "\t")
We get the information about published JOSS papers from Crossref,
using the rcrossref R package. The openalexR R
package is used to extract citation counts from OpenAlex.
## First check how many records there are in Crossref
## See ?rcrossref for information about how to provide your email address to get
## improved performance when using the crossref API
issn <- "2475-9066"
joss_details <- rcrossref::cr_journals(issn, works = FALSE) %>%
pluck("data")
(total_dois <- joss_details$total_dois)
## [1] 3700
## Pull down all records from Crossref
papers <- rcrossref::cr_journals(issn, works = TRUE, cursor = "*",
cursor_max = joss_details$total_dois * 2) %>%
pluck("data")
## Only keep articles
papers <- papers %>%
dplyr::filter(type == "journal-article")
dim(papers)
## [1] 3700 28
dim(papers %>% distinct())
## [1] 3700 28
## Check that all papers were pulled down and stop otherwise
if (!(nrow(papers %>% distinct()) >= total_dois)) {
stop("Not all papers were pulled down from Crossref!")
}
## A few papers don't have alternative.ids - generate them from the DOI
noaltid <- which(is.na(papers$alternative.id))
papers$alternative.id[noaltid] <- papers$doi[noaltid]
## Get citation info from Crossref and merge with paper details
# cit <- rcrossref::cr_citation_count(doi = papers$alternative.id)
# papers <- papers %>% dplyr::left_join(
# cit %>% dplyr::rename(citation_count = count),
# by = c("alternative.id" = "doi")
# )
## Remove one duplicated paper
papers <- papers %>% dplyr::filter(alternative.id != "10.21105/joss.00688")
dim(papers)
## [1] 3699 28
dim(papers %>% distinct())
## [1] 3699 28
papers$alternative.id[duplicated(papers$alternative.id)]
## character(0)
source_track <- c(source_track,
structure(rep("crossref", ncol(papers)),
names = colnames(papers)))
## Get info from openalexR and merge with paper details
## Helper function to extract countries from affiliations. Note that this
## information is not available for all papers.
.get_countries <- function(df, wh = "first") {
if ((length(df) == 1 && is.na(df)) || is.null(df$affiliations)) {
""
} else {
if (wh == "first") {
## Only first affiliation for each author
tmp <- unnest(df, cols = c(affiliations), names_sep = "_") |>
dplyr::filter(!duplicated(id) & !is.na(affiliations_country_code)) |>
pull(affiliations_country_code)
} else {
## All affiliations
tmp <- unnest(df, cols = c(affiliations), names_sep = "_") |>
dplyr::filter(!is.na(affiliations_country_code)) |>
pull(affiliations_country_code)
}
if (length(tmp) > 0) {
tmp |>
unique() |>
paste(collapse = ";")
} else {
""
}
}
}
oa <- oa_fetch(entity = "works",
primary_location.source.id = "s4210214273") |>
mutate(affil_countries_all = vapply(authorships, .get_countries, "", wh = "all"),
affil_countries_first = vapply(authorships, .get_countries, "", wh = "first"))
dim(oa)
## [1] 3707 45
length(unique(oa$doi))
## [1] 3704
papers <- papers %>% dplyr::left_join(
oa %>% dplyr::mutate(alternative.id = sub("https://doi.org/", "", doi)) %>%
dplyr::select(alternative.id, cited_by_count, id,
affil_countries_all, affil_countries_first) %>%
dplyr::rename(citation_count = cited_by_count,
openalex_id = id),
by = "alternative.id"
)
dim(papers)
## [1] 3702 32
dim(papers %>% distinct())
## [1] 3702 32
source_track <- c(source_track,
structure(rep("OpenAlex", length(setdiff(colnames(papers),
names(source_track)))),
names = setdiff(colnames(papers), names(source_track))))
For each published paper, we use the JOSS API to get information about pre-review and review issue numbers, corresponding software repository etc.
joss_api <- list()
p <- 1
a0 <- NULL
a <- jsonlite::fromJSON(
url(paste0("https://joss.theoj.org/papers/published.json?page=", p)),
simplifyDataFrame = FALSE
)
while (length(a) > 0 && !identical(a, a0)) {
joss_api <- c(joss_api, a)
p <- p + 1
a0 <- a
a <- tryCatch({
jsonlite::fromJSON(
url(paste0("https://joss.theoj.org/papers/published.json?page=", p)),
simplifyDataFrame = FALSE
)},
error = function(e) return(numeric(0))
)
}
joss_api <- do.call(dplyr::bind_rows, lapply(joss_api, function(w) {
data.frame(api_title = w$title,
api_state = w$state,
author_affiliations = paste(unique(unlist(lapply(w$authors, "[[", "affiliation"))), collapse = ";"),
editor = paste(w$editor, collapse = ","),
reviewers = paste(w$reviewers, collapse = ","),
nbr_reviewers = length(w$reviewers),
repo_url = w$software_repository,
review_issue_id = sub("https://github.com/openjournals/joss-reviews/issues/",
"", w$paper_review),
doi = w$doi,
prereview_issue_id = ifelse(!is.null(w$meta_review_issue_id),
w$meta_review_issue_id, NA_integer_),
languages = gsub(", ", ",", w$languages),
archive_doi = w$software_archive)
}))
dim(joss_api)
## [1] 3700 12
dim(joss_api %>% distinct())
## [1] 3700 12
## Check that all papers were pulled down and stop otherwise
if (!(nrow(joss_api %>% distinct()) >= total_dois)) {
stop("Not all papers were pulled down from the JOSS API!")
}
joss_api$repo_url[duplicated(joss_api$repo_url)]
## [1] "https://github.com/alan-turing-institute/autoemulate"
## [2] "https://gitlab.com/mauricemolli/petitRADTRANS"
## [3] "https://github.com/nomad-coe/greenX"
## [4] "https://github.com/mdhaber/scipy"
## [5] "https://gitlab.com/fame-framework/fame-io"
## [6] "https://github.com/idaholab/moose"
## [7] "https://gitlab.com/libreumg/dataquier.git"
## [8] "https://github.com/idaholab/moose"
## [9] "https://github.com/dynamicslab/pysindy"
## [10] "https://github.com/landlab/landlab"
## [11] "https://github.com/landlab/landlab"
## [12] "https://github.com/symmy596/SurfinPy"
## [13] "https://github.com/arviz-devs/arviz"
## [14] "https://github.com/bcgov/ssdtools"
## [15] "https://github.com/landlab/landlab"
## [16] "https://github.com/pvlib/pvlib-python"
## [17] "https://github.com/mlpack/mlpack"
## [18] "https://github.com/julia-wrobel/registr"
## [19] "https://github.com/barbagroup/pygbe"
papers <- papers %>% dplyr::left_join(joss_api, by = c("alternative.id" = "doi"))
dim(papers)
## [1] 3702 43
dim(papers %>% distinct())
## [1] 3702 43
papers$repo_url[duplicated(papers$repo_url)]
## [1] "https://github.com/barbagroup/pygbe"
## [2] "https://github.com/landlab/landlab"
## [3] "https://github.com/landlab/landlab"
## [4] "https://github.com/landlab/landlab"
## [5] "https://github.com/julia-wrobel/registr"
## [6] "https://github.com/idaholab/moose"
## [7] "https://github.com/dynamicslab/pysindy"
## [8] "https://github.com/symmy596/SurfinPy"
## [9] "https://github.com/mlpack/mlpack"
## [10] "https://github.com/pvlib/pvlib-python"
## [11] "https://github.com/idaholab/moose"
## [12] "https://gitlab.com/libreumg/dataquier.git"
## [13] "https://gitlab.com/mauricemolli/petitRADTRANS"
## [14] "https://github.com/bcgov/ssdtools"
## [15] "https://github.com/nomad-coe/greenX"
## [16] "https://github.com/fluhus/blini"
## [17] "https://github.com/QTC-UMD/rydiqule"
## [18] "https://github.com/arviz-devs/arviz"
## [19] "https://github.com/mdhaber/scipy"
## [20] "https://github.com/adityapt/deepcausalmmm/"
## [21] "https://gitlab.com/fame-framework/fame-io"
## [22] "https://github.com/alan-turing-institute/autoemulate"
source_track <- c(source_track,
structure(rep("JOSS_API", length(setdiff(colnames(papers),
names(source_track)))),
names = setdiff(colnames(papers), names(source_track))))
From each pre-review and review issue, we extract information about review times and assigned labels.
## Pull down info on all issues in the joss-reviews repository
## See ?gh_token for information about GitHub Personal Access Tokens
issues <- gh("/repos/openjournals/joss-reviews/issues",
.limit = 15000, state = "all")
## From each issue, extract required information
iss <- do.call(dplyr::bind_rows, lapply(issues, function(i) {
data.frame(title = i$title,
number = i$number,
state = i$state,
opened = i$created_at,
closed = ifelse(!is.null(i$closed_at),
i$closed_at, NA_character_),
ncomments = i$comments,
labels = paste(setdiff(
vapply(i$labels, getElement,
name = "name", character(1L)),
c("review", "pre-review", "query-scope", "paused")),
collapse = ","))
}))
## Split into REVIEW, PRE-REVIEW, and other issues (the latter category
## is discarded)
issother <- iss %>% dplyr::filter(!grepl("\\[PRE REVIEW\\]", title) &
!grepl("\\[REVIEW\\]", title))
dim(issother)
## [1] 208 7
head(issother)
## title
## 1 Clarify data-sharing checklist item: is bundled compound classification database 'original data'
## 2 Need Software Design (required section)
## 3 [JOSS] gitdealflow-signal-engine: deterministic classification of startup engineering acceleration from public GitHub activity
## 4 ...
## 5 [joss][question] Clarify review target: paper branch (joss) diverges from and lags the v0.7.0 release
## 6 Question About JOSS Reviewer Selection and Participation
## number state opened closed ncomments labels
## 1 11236 closed 2026-09-01T09:00:52Z 2026-09-01T09:00:55Z 1
## 2 11194 closed 2026-08-24T00:22:08Z 2026-08-24T00:22:11Z 1
## 3 11168 closed 2026-08-18T20:17:11Z 2026-08-18T20:17:14Z 1
## 4 10901 closed 2026-07-07T09:47:42Z 2026-07-07T09:47:45Z 1
## 5 10776 closed 2026-06-21T15:23:11Z 2026-06-21T15:23:13Z 1
## 6 10655 closed 2026-06-06T20:41:42Z 2026-06-06T20:41:44Z 1
## For REVIEW issues, generate the DOI of the paper from the issue number
getnbrzeros <- function(s) {
paste(rep(0, 5 - nchar(s)), collapse = "")
}
issrev <- iss %>% dplyr::filter(grepl("\\[REVIEW\\]", title)) %>%
dplyr::mutate(nbrzeros = purrr::map_chr(number, getnbrzeros)) %>%
dplyr::mutate(alternative.id = paste0("10.21105/joss.",
nbrzeros,
number)) %>%
dplyr::select(-nbrzeros) %>%
dplyr::mutate(title = gsub("\\[REVIEW\\]: ", "", title)) %>%
dplyr::rename_at(vars(-alternative.id), ~ paste0("review_", .))
## For pre-review and review issues, respectively, get the number of
## issues closed each month, and the number of those that have the
## 'rejected' label
review_rejected <- iss %>%
dplyr::filter(grepl("\\[REVIEW\\]", title)) %>%
dplyr::filter(!is.na(closed)) %>%
dplyr::mutate(closedmonth = lubridate::floor_date(as.Date(closed), "month")) %>%
dplyr::group_by(closedmonth) %>%
dplyr::summarize(nbr_issues_closed = length(labels),
nbr_rejections = sum(grepl("rejected", labels))) %>%
dplyr::mutate(itype = "review")
prereview_rejected <- iss %>%
dplyr::filter(grepl("\\[PRE REVIEW\\]", title)) %>%
dplyr::filter(!is.na(closed)) %>%
dplyr::mutate(closedmonth = lubridate::floor_date(as.Date(closed), "month")) %>%
dplyr::group_by(closedmonth) %>%
dplyr::summarize(nbr_issues_closed = length(labels),
nbr_rejections = sum(grepl("rejected", labels))) %>%
dplyr::mutate(itype = "pre-review")
all_rejected <- dplyr::bind_rows(review_rejected, prereview_rejected)
## Get only pre-review issues plus review issues opened before 2016-09-18,
## will use these as a proxy for the number of submissions
pi1 <- iss |>
dplyr::filter(grepl("\\[PRE REVIEW\\]", title)) |>
dplyr::mutate(opened = as.Date(opened))
dim(pi1)
## [1] 6852 7
pi2 <- iss |>
dplyr::filter(grepl("\\[REVIEW\\]", title)) |>
dplyr::mutate(opened = as.Date(opened)) |>
dplyr::filter(opened <= as.Date("2016-09-18"))
dim(pi2)
## [1] 49 7
prereview_issues <- dplyr::bind_rows(pi1, pi2)
## For PRE-REVIEW issues, add information about the corresponding REVIEW
## issue number
isspre <- iss %>% dplyr::filter(grepl("\\[PRE REVIEW\\]", title)) %>%
dplyr::filter(!grepl("withdrawn", labels)) %>%
dplyr::filter(!grepl("rejected", labels))
## Some titles have multiple pre-review issues. In these cases, keep the latest
isspre <- isspre %>% dplyr::arrange(desc(number)) %>%
dplyr::filter(!duplicated(title)) %>%
dplyr::mutate(title = gsub("\\[PRE REVIEW\\]: ", "", title)) %>%
dplyr::rename_all(~ paste0("prerev_", .))
papers <- papers %>% dplyr::left_join(issrev, by = "alternative.id") %>%
dplyr::left_join(isspre, by = c("prereview_issue_id" = "prerev_number")) %>%
dplyr::mutate(prerev_opened = as.Date(prerev_opened),
prerev_closed = as.Date(prerev_closed),
review_opened = as.Date(review_opened),
review_closed = as.Date(review_closed)) %>%
dplyr::mutate(days_in_pre = prerev_closed - prerev_opened,
days_in_rev = review_closed - review_opened,
to_review = !is.na(review_opened))
dim(papers)
## [1] 3702 59
dim(papers %>% distinct())
## [1] 3702 59
source_track <- c(source_track,
structure(rep("joss-github", length(setdiff(colnames(papers),
names(source_track)))),
names = setdiff(colnames(papers), names(source_track))))
## Reorder so that software repositories that were interrogated longest
## ago are checked first
tmporder <- order(match(papers$alternative.id, papers_archive$alternative.id),
na.last = FALSE)
software_urls <- papers$repo_url[tmporder]
software_urls[duplicated(software_urls)]
## [1] "https://gitlab.com/libreumg/dataquier.git"
## [2] "https://gitlab.com/mauricemolli/petitRADTRANS"
## [3] "https://gitlab.com/fame-framework/fame-io"
## [4] "https://github.com/arviz-devs/arviz"
## [5] "https://github.com/adityapt/deepcausalmmm/"
## [6] "https://github.com/mlpack/mlpack"
## [7] "https://github.com/pvlib/pvlib-python"
## [8] "https://github.com/alan-turing-institute/autoemulate"
## [9] "https://github.com/barbagroup/pygbe"
## [10] "https://github.com/landlab/landlab"
## [11] "https://github.com/landlab/landlab"
## [12] "https://github.com/landlab/landlab"
## [13] "https://github.com/julia-wrobel/registr"
## [14] "https://github.com/idaholab/moose"
## [15] "https://github.com/dynamicslab/pysindy"
## [16] "https://github.com/symmy596/SurfinPy"
## [17] "https://github.com/idaholab/moose"
## [18] "https://github.com/bcgov/ssdtools"
## [19] "https://github.com/nomad-coe/greenX"
## [20] "https://github.com/fluhus/blini"
## [21] "https://github.com/QTC-UMD/rydiqule"
## [22] "https://github.com/mdhaber/scipy"
is_github <- grepl("github", software_urls)
length(is_github)
## [1] 3702
sum(is_github)
## [1] 3513
software_urls[!is_github]
## [1] "https://bitbucket.org/dghoshal/frieda"
## [2] "https://bitbucket.org/cmutel/brightway2"
## [3] "https://bitbucket.org/meg/cbcbeat"
## [4] "https://bitbucket.org/cloopsy/android/"
## [5] "https://savannah.nongnu.org/projects/complot/"
## [6] "http://mutabit.com/repos.fossil/grafoscopio/"
## [7] "https://www.idpoisson.fr/fullswof/"
## [8] "https://gitlab.com/cerfacs/batman"
## [9] "https://bitbucket.org/cardosan/brightway2-temporalis"
## [10] "https://sourceforge.net/p/mcapl/mcapl_code/ci/master/tree/"
## [11] "https://bitbucket.org/glotzer/rowan"
## [12] "https://gitlab.com/costrouc/pysrim"
## [13] "https://gitlab.com/moorepants/skijumpdesign"
## [14] "https://bitbucket.org/cdegroot/wediff"
## [15] "https://bitbucket.org/ocellarisproject/ocellaris"
## [16] "https://bitbucket.org/mpi4py/mpi4py-fft"
## [17] "https://gitlab.com/celliern/scikit-fdiff/"
## [18] "https://bitbucket.org/manuela_s/hcp/"
## [19] "https://gitlab.com/QComms/cqptoolkit"
## [20] "https://bitbucket.org/dolfin-adjoint/pyadjoint"
## [21] "https://gitlab.inria.fr/azais/treex"
## [22] "https://bitbucket.org/basicsums/basicsums"
## [23] "https://gitlab.com/toposens/public/ros-packages"
## [24] "https://gitlab.com/dlr-dw/ontocode"
## [25] "https://gitlab.com/eidheim/Simple-Web-Server"
## [26] "https://gitlab.com/tesch1/cppduals"
## [27] "https://doi.org/10.17605/OSF.IO/3DS6A"
## [28] "https://gitlab.com/gdetor/genetic_alg"
## [29] "https://gitlab.com/materials-modeling/wulffpack"
## [30] "https://gitlab.com/myqueue/myqueue"
## [31] "https://bitbucket.org/likask/mofem-cephas"
## [32] "https://gitlab.inria.fr/miet/miet"
## [33] "https://gitlab.com/sails-dev/sails"
## [34] "https://bitbucket.org/hammurabicode/hamx"
## [35] "https://gricad-gitlab.univ-grenoble-alpes.fr/ttk/spam/"
## [36] "https://gitlab.com/datafold-dev/datafold/"
## [37] "https://gitlab.com/tamaas/tamaas"
## [38] "https://gitlab.com/utopia-project/dantro"
## [39] "https://gitlab.dune-project.org/dorie/dorie"
## [40] "https://bitbucket.org/rram/dvrlib/src/joss/"
## [41] "https://gitlab.com/davidtourigny/dynamic-fba"
## [42] "https://gitlab.com/utopia-project/utopia"
## [43] "https://bitbucket.org/miketuri/perl-spice-sim-seus/"
## [44] "https://gitlab.com/cosmograil/PyCS3"
## [45] "https://gitlab.com/ampere2/metalwalls"
## [46] "https://gitlab.com/project-dare/dare-platform"
## [47] "https://gitlab.gwdg.de/mpievolbio-it/crbhits"
## [48] "https://gitlab.com/LMSAL_HUB/aia_hub/aiapy"
## [49] "https://bitbucket.org/clhaley/Multitaper.jl"
## [50] "https://gitlab.com/geekysquirrel/bigx"
## [51] "https://gitlab.com/vibes-developers/vibes"
## [52] "https://gitlab.inria.fr/bramas/tbfmm"
## [53] "https://git.iws.uni-stuttgart.de/tools/frackit"
## [54] "https://gitlab.com/gims-developers/gims"
## [55] "https://framagit.org/GustaveCoste/eldam"
## [56] "https://gitlab.com/ffaucher/hawen"
## [57] "https://gitlab.com/mmartin-lagarde/exonoodle-exoplanets/-/tree/master/"
## [58] "https://bitbucket.org/berkeleylab/esdr-pygdh/"
## [59] "https://earth.bsc.es/gitlab/wuruchi/autosubmitreact"
## [60] "https://gitlab.inria.fr/mosaic/bvpy"
## [61] "https://gitlab.com/emd-dev/emd"
## [62] "https://git.rwth-aachen.de/ants/sensorlab/imea"
## [63] "https://gitlab.com/energyincities/besos/"
## [64] "https://gitlab.com/fduchate/predihood"
## [65] "https://gitlab.com/libreumg/dataquier.git"
## [66] "https://gitlab.ethz.ch/holukas/dyco-dynamic-lag-compensation"
## [67] "https://bitbucket.org/sciencecapsule/sciencecapsule"
## [68] "https://gitlab.com/dlr-ve/autumn/"
## [69] "https://gitlab.com/marinvaders/marinvaders"
## [70] "https://gitlab.com/manchester_qbi/manchester_qbi_public/madym_cxx/"
## [71] "https://gitlab.com/jason-rumengan/pyarma"
## [72] "https://bitbucket.org/mituq/muq2.git"
## [73] "https://gitlab.com/remram44/taguette"
## [74] "https://bitbucket.org/orionmhdteam/orion2_release1/src/master/"
## [75] "https://gitlab.com/cracklet/cracklet.git"
## [76] "https://gitlab.com/pyFBS/pyFBS"
## [77] "https://gitlab.kitware.com/LBM/lattice-boltzmann-solver"
## [78] "https://gitlab.com/picos-api/picos"
## [79] "https://gitlab.com/sissopp_developers/sissopp"
## [80] "https://code.usgs.gov/umesc/quant-ecology/fishstan/"
## [81] "https://bitbucket.org/berkeleylab/hardware-control/src/main/"
## [82] "https://gitlab.com/culturalcartography/text2map"
## [83] "https://gitlab.uliege.be/smart_grids/public/gboml"
## [84] "https://gitlab.pasteur.fr/vlegrand/ROCK"
## [85] "https://framagit.org/GustaveCoste/off-product-environmental-impact/"
## [86] "https://gitlab.ruhr-uni-bochum.de/reichp2y/proppy"
## [87] "https://gitlab.com/moerman1/fhi-cc4s"
## [88] "https://gitlab.com/thartwig/asloth"
## [89] "https://gitlab.com/dmt-development/dmt-core"
## [90] "https://gitlab.com/dsbowen/conditional-inference"
## [91] "https://gitlab.inria.fr/bcoye/game-engine-scheduling-simulation"
## [92] "https://jugit.fz-juelich.de/compflu/swalbe.jl/"
## [93] "https://gitlab.kuleuven.be/ITSCreaLab/public-toolboxes/dyntapy"
## [94] "https://gitlab.com/permafrostnet/teaspoon"
## [95] "https://git.geomar.de/digital-earth/dasf/dasf-messaging-python"
## [96] "https://gitlab.com/ags-data-format-wg/ags-python-library"
## [97] "https://gitlab.com/petsc/petsc"
## [98] "https://bitbucket.org/bmskinner/nuclear_morphology"
## [99] "https://git.mpib-berlin.mpg.de/castellum/castellum"
## [100] "https://gitlab.mpikg.mpg.de/curcuraci/bmiptools"
## [101] "https://gitlab.com/wpettersson/kep_solver"
## [102] "https://gitlab.com/InspectorCell/inspectorcell"
## [103] "https://gite.lirmm.fr/doccy/RedOak"
## [104] "https://gitlab.com/programgreg/tagginglatencyestimator"
## [105] "https://gitlab.com/ProjectRHEA/flowsolverrhea"
## [106] "https://gitlab.com/fibreglass/pivc"
## [107] "https://gitlab.com/dglaeser/fieldcompare"
## [108] "https://gitlab.awi.de/sicopolis/sicopolis"
## [109] "https://gitlab.com/jesseds/apav"
## [110] "https://gitlab.ifremer.fr/resourcecode/resourcecode"
## [111] "https://gitlab.com/dlr-ve/esy/amiris/amiris"
## [112] "https://gitlab.com/fame-framework/fame-io"
## [113] "https://gitlab.com/fame-framework/fame-core"
## [114] "https://git.ligo.org/asimov/asimov"
## [115] "https://gitlab.com/cosmograil/starred"
## [116] "https://plmlab.math.cnrs.fr/lmrs/statistique/smmR"
## [117] "https://gitlab.com/pvst/asi"
## [118] "https://gitlab.com/binary_c/binary_c-python/"
## [119] "https://gitlab.inria.fr/melissa/melissa"
## [120] "https://gitlab.com/drti/basic-tools"
## [121] "https://gitlab.com/ENKI-portal/ThermoCodegen"
## [122] "https://gitlab.com/sigcorr/sigcorr"
## [123] "https://gitlab.com/dlr-ve/esy/sfctools/framework/"
## [124] "https://gitlab.com/chaver/choco-mining"
## [125] "https://gitlab.com/pythia-uq/pythia"
## [126] "https://gitlab.com/bioeconomy/forobs/biotrade/"
## [127] "https://gitlab.com/soleil-data-treatment/soleil-software-projects/remote-desktop"
## [128] "https://bitbucket.org/sbarbot/motorcycle/src/master/"
## [129] "https://gitlab.com/habermann_lab/phasik"
## [130] "https://gitlab.com/robizzard/libcdict"
## [131] "https://gitlab.com/jtagusari/hrisk-noisemodelling"
## [132] "https://gitlab.com/tum-ciip/elsa"
## [133] "https://gitlab.com/tue-umphy/software/parmesan"
## [134] "https://gitlab.com/akantu/akantu"
## [135] "https://gitlab.com/cosapp/cosapp"
## [136] "https://gitlab.com/materials-modeling/calorine"
## [137] "https://codebase.helmholtz.cloud/mussel/netlogo-northsea-species.git"
## [138] "https://gitlab.com/bonsamurais/bonsai/util/ipcc"
## [139] "https://gitlab.com/mauricemolli/petitRADTRANS"
## [140] "https://bitbucket.org/robmoss/particle-filter-for-python/"
## [141] "https://gitlab.eudat.eu/coccon-kit/proffastpylot"
## [142] "https://gitlab.com/cmbm-ethz/pourbaix-diagrams"
## [143] "https://gitlab.com/mantik-ai/mantik"
## [144] "https://gitlab.com/dlr-ve/esy/remix/framework"
## [145] "https://gitlab.com/libreumg/dataquier.git"
## [146] "https://gitlab.com/open-darts/open-darts"
## [147] "https://gricad-gitlab.univ-grenoble-alpes.fr/deformvis/insarviz"
## [148] "https://forgemia.inra.fr/pherosensor/pherosensor-toolbox"
## [149] "https://gitlab.com/qc-devs/aqcnes"
## [150] "https://gitlab.com/MartinBeseda/sa-oo-vqe-qiskit.git"
## [151] "https://gitlab.com/mauricemolli/petitRADTRANS"
## [152] "https://gitlab.com/lheea/CN-AeroModels"
## [153] "https://gitlab.com/free-astro/siril"
## [154] "https://forgemia.inra.fr/migale/easy16s"
## [155] "https://gitlab.com/davidwoodburn/itrm"
## [156] "https://git.ufz.de/despot/pysewer/"
## [157] "https://gitlab.dune-project.org/copasi/dune-copasi"
## [158] "https://gitlab.com/davidwoodburn/r3f"
## [159] "https://gitlab.ruhr-uni-bochum.de/ee/cd2es"
## [160] "https://gitlab.com/morikawa-lab-osakau/vibir-parallel-compute"
## [161] "https://gitlab.com/dlr-ve/esy/vencopy/vencopy"
## [162] "https://gitlab.com/grogra/groimp-plugins/Pointcloud"
## [163] "https://zivgitlab.uni-muenster.de/ag-salinga/fastatomstruct"
## [164] "https://gitlab.com/ComputationalScience/idinn"
## [165] "https://codeberg.org/JPHackstein/GREOPy"
## [166] "https://gitlab.com/EliseLei/easychem"
## [167] "https://gitlab.com/djsmithbham/cnearest"
## [168] "https://codeberg.org/benmagill/deflake.rs"
## [169] "https://gitlab.kuleuven.be/gelenslab/publications/pycline"
## [170] "https://gitlab.com/cosmology-ethz/ufig"
## [171] "https://gitlab.com/cmbm-ethz/miop"
## [172] "https://code.europa.eu/kada/mafw"
## [173] "https://gitlab.eclipse.org/eclipse/comma/comma"
## [174] "https://codeberg.org/cepsInria/ceps"
## [175] "https://codebase.helmholtz.cloud/taimur.khan/DeepTrees"
## [176] "https://gitlab.com/cosmology-ethz/galsbi"
## [177] "https://gitlab.com/grogra/groimp-plugins/api"
## [178] "https://gitlab.com/oali/dxtr"
## [179] "https://gitlab.com/micromorph/ratel"
## [180] "https://gitlab.com/uniluxembourg/hpc/research/cadom/serializable-simpy"
## [181] "https://gitlab.com/sunpeek/sunpeek/"
## [182] "https://bitbucket.org/brunopostle/homemaker"
## [183] "https://gitlab.fysik.su.se/operando-catalysis-spectroscopy/polariseval/"
## [184] "https://gitlab.in2p3.fr/lemaitre/cosmologix"
## [185] "https://gitlab.com/bioeconomy/cobwood/cobwood"
## [186] "https://codeberg.org/sarah-quinones/faer/"
## [187] "https://gitlab.com/fame-framework/fame-io"
## [188] "https://gitlab.com/links_and_nodes/links_and_nodes"
## [189] "https://gitlab.com/felics-group/FELiCS"
df <- do.call(dplyr::bind_rows, lapply(unique(software_urls[is_github]), function(u) {
u0 <- gsub("^http://", "https://", gsub("\\.git$", "", gsub("/$", "", u)))
if (grepl("/tree/", u0)) {
u0 <- strsplit(u0, "/tree/")[[1]][1]
}
if (grepl("/blob/", u0)) {
u0 <- strsplit(u0, "/blob/")[[1]][1]
}
info <- try({
gh(gsub("(https://)?(www.)?github.com/", "/repos/", u0))
})
languages <- try({
gh(paste0(gsub("(https://)?(www.)?github.com/", "/repos/", u0), "/languages"),
.limit = 500)
})
topics <- try({
gh(paste0(gsub("(https://)?(www.)?github.com/", "/repos/", u0), "/topics"),
.accept = "application/vnd.github.mercy-preview+json", .limit = 500)
})
contribs <- try({
gh(paste0(gsub("(https://)?(www.)?github.com/", "/repos/", u0), "/contributors"),
.limit = 500)
})
if (!is(info, "try-error") && length(info) > 1) {
if (!is(contribs, "try-error")) {
if (length(contribs) == 0) {
repo_nbr_contribs <- repo_nbr_contribs_2ormore <- NA_integer_
} else {
repo_nbr_contribs <- length(contribs)
repo_nbr_contribs_2ormore <- sum(vapply(contribs, function(x) x$contributions >= 2, NA_integer_))
if (is.na(repo_nbr_contribs_2ormore)) {
print(contribs)
}
}
} else {
repo_nbr_contribs <- repo_nbr_contribs_2ormore <- NA_integer_
}
if (!is(languages, "try-error")) {
if (length(languages) == 0) {
repolang <- ""
} else {
repolang <- paste(paste(names(unlist(languages)),
unlist(languages), sep = ":"), collapse = ",")
}
} else {
repolang <- ""
}
if (!is(topics, "try-error")) {
if (length(topics$names) == 0) {
repotopics <- ""
} else {
repotopics <- paste(unlist(topics$names), collapse = ",")
}
} else {
repotopics <- ""
}
data.frame(repo_url = u,
repo_created = info$created_at,
repo_updated = info$updated_at,
repo_pushed = info$pushed_at,
repo_nbr_stars = info$stargazers_count,
repo_language = ifelse(!is.null(info$language),
info$language, NA_character_),
repo_languages_bytes = repolang,
repo_topics = repotopics,
repo_license = ifelse(!is.null(info$license),
info$license$key, NA_character_),
repo_nbr_contribs = repo_nbr_contribs,
repo_nbr_contribs_2ormore = repo_nbr_contribs_2ormore
)
} else {
NULL
}
})) %>%
dplyr::mutate(repo_created = as.Date(repo_created),
repo_updated = as.Date(repo_updated),
repo_pushed = as.Date(repo_pushed)) %>%
dplyr::distinct() %>%
dplyr::mutate(repo_info_obtained = lubridate::today())
if (length(unique(df$repo_url)) != length(df$repo_url)) {
print(length(unique(df$repo_url)))
print(length(df$repo_url))
print(df$repo_url[duplicated(df$repo_url)])
}
stopifnot(length(unique(df$repo_url)) == length(df$repo_url))
dim(df)
## [1] 2297 12
## For papers not in df (i.e., for which we didn't get a valid response
## from the GitHub API query), use information from the archived data frame
dfarchive <- papers_archive %>%
dplyr::select(colnames(df)[colnames(df) %in% colnames(papers_archive)]) %>%
dplyr::filter(!(repo_url %in% df$repo_url)) %>%
dplyr::arrange(desc(repo_info_obtained)) %>%
dplyr::filter(!duplicated(repo_url))
head(dfarchive)
## # A tibble: 6 × 12
## repo_url repo_created repo_updated repo_pushed repo_nbr_stars repo_language
## <chr> <date> <date> <date> <int> <chr>
## 1 https://gi… 2015-11-23 2025-06-08 2017-05-04 56 Python
## 2 https://gi… 2014-07-30 2026-07-26 2017-01-11 14 TeX
## 3 http://git… 2015-10-28 2026-08-22 2016-05-16 88 Jupyter Note…
## 4 https://gi… 2012-06-21 2026-07-12 2026-07-25 226 Python
## 5 https://gi… 2015-12-16 2025-02-21 2019-08-29 6 Jupyter Note…
## 6 https://gi… 2017-09-21 2026-05-08 2026-05-08 5 R
## # ℹ 6 more variables: repo_languages_bytes <chr>, repo_topics <chr>,
## # repo_license <chr>, repo_nbr_contribs <int>,
## # repo_nbr_contribs_2ormore <int>, repo_info_obtained <date>
dim(dfarchive)
## [1] 1383 12
df <- dplyr::bind_rows(df, dfarchive)
stopifnot(length(unique(df$repo_url)) == length(df$repo_url))
dim(df)
## [1] 3680 12
papers <- papers %>% dplyr::left_join(df, by = "repo_url")
dim(papers)
## [1] 3702 70
source_track <- c(source_track,
structure(rep("sw-github", length(setdiff(colnames(papers),
names(source_track)))),
names = setdiff(colnames(papers), names(source_track))))
## Convert publication date to Date format
## Add information about the half year (H1, H2) of publication
## Count number of authors
papers <- papers %>% dplyr::select(-reference, -license, -link) %>%
dplyr::mutate(published.date = as.Date(published.print)) %>%
dplyr::mutate(
halfyear = paste0(year(published.date),
ifelse(month(published.date) <= 6, "H1", "H2"))
) %>% dplyr::mutate(
halfyear = factor(halfyear,
levels = paste0(rep(sort(unique(year(published.date))),
each = 2), c("H1", "H2")))
) %>% dplyr::mutate(nbr_authors = vapply(author, function(a) nrow(a), NA_integer_))
dim(papers)
## [1] 3702 70
dupidx <- which(papers$alternative.id %in% papers$alternative.id[duplicated(papers)])
papers[dupidx, ] %>% arrange(alternative.id) %>% head(n = 10)
## # A tibble: 0 × 70
## # ℹ 70 variables: created <chr>, deposited <chr>, doi <chr>, indexed <chr>,
## # issn <chr>, member <chr>, prefix <chr>, publisher <chr>, score <chr>,
## # source <chr>, reference.count <chr>, references.count <chr>,
## # is.referenced.by.count <chr>, title <chr>, type <chr>, url <chr>,
## # alternative.id <chr>, container.title <chr>, published.print <chr>,
## # issue <chr>, issued <chr>, page <chr>, volume <chr>,
## # short.container.title <chr>, author <list>, citation_count <int>, …
papers <- papers %>% dplyr::distinct()
dim(papers)
## [1] 3702 70
source_track <- c(source_track,
structure(rep("cleanup", length(setdiff(colnames(papers),
names(source_track)))),
names = setdiff(colnames(papers), names(source_track))))
In some cases, fetching information from (e.g.) the GitHub API fails for a subset of the publications. There are also other reasons for missing values (for example, the earliest submissions do not have an associated pre-review issue). The table below lists the number of missing values for each of the variables in the data frame.
DT::datatable(
data.frame(variable = colnames(papers),
nbr_missing = colSums(is.na(papers))) %>%
dplyr::mutate(source = source_track[variable]),
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
monthly_pubs <- papers %>%
dplyr::mutate(pubmonth = lubridate::floor_date(published.date, "month")) %>%
dplyr::group_by(pubmonth) %>%
dplyr::summarize(npub = n())
ggplot(monthly_pubs,
aes(x = factor(pubmonth), y = npub)) +
geom_bar(stat = "identity") + theme_minimal() +
labs(x = "", y = "Number of published papers per month", caption = dcap) +
theme(axis.title = element_text(size = 15),
axis.text.x = element_text(angle = 90, hjust = 1, vjust = 0.5))
DT::datatable(
monthly_pubs %>%
dplyr::rename("Number of papers" = "npub",
"Month of publication" = "pubmonth"),
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
yearly_pubs <- papers %>%
dplyr::mutate(pubyear = lubridate::year(published.date)) %>%
dplyr::group_by(pubyear) %>%
dplyr::summarize(npub = n())
ggplot(yearly_pubs,
aes(x = factor(pubyear), y = npub)) +
geom_bar(stat = "identity") + theme_minimal() +
labs(x = "", y = "Number of published papers per year", caption = dcap) +
theme(axis.title = element_text(size = 15),
axis.text.x = element_text(angle = 90, hjust = 1, vjust = 0.5))
DT::datatable(
yearly_pubs %>%
dplyr::rename("Number of papers" = "npub",
"Year of publication" = "pubyear"),
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
We use the number of opened pre-review issues in a month as a proxy for the number of submissions.
monthly_subs <- prereview_issues |>
dplyr::mutate(submonth = lubridate::floor_date(opened, "month")) |>
dplyr::group_by(submonth) |>
dplyr::summarize(nsub = n()) |>
# fill in missing months (with 0 submissions)
tidyr::complete(
submonth = seq(min(submonth, na.rm = TRUE),
max(submonth, na.rm = TRUE), by = "month"),
fill = list(nsub = 0)
)
ggplot(monthly_subs,
aes(x = factor(submonth), y = nsub)) +
geom_bar(stat = "identity") + theme_minimal() +
labs(x = "", y = "Number of submissions per month", caption = dcap) +
theme(axis.title = element_text(size = 15),
axis.text.x = element_text(angle = 90, hjust = 1, vjust = 0.5))
DT::datatable(
monthly_subs |>
dplyr::rename("Number of submissions" = "nsub",
"Month of submission" = "submonth"),
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
The plots below illustrate the fraction of pre-review and review issues closed during each month that have the ‘rejected’ label attached.
ggplot(all_rejected,
aes(x = factor(closedmonth), y = nbr_rejections/nbr_issues_closed)) +
geom_bar(stat = "identity") +
theme_minimal() +
facet_wrap(~ itype, ncol = 1) +
labs(x = "Month of issue closing", y = "Fraction of issues rejected",
caption = dcap) +
theme(axis.title = element_text(size = 15),
axis.text.x = element_text(angle = 90, hjust = 1, vjust = 0.5))
Papers with 20 or more citations are grouped in the “>=20” category.
ggplot(papers %>%
dplyr::mutate(citation_count = replace(citation_count,
citation_count >= 20, ">=20")) %>%
dplyr::mutate(citation_count = factor(citation_count,
levels = c(0:20, ">=20"))) %>%
dplyr::group_by(citation_count) %>%
dplyr::tally(),
aes(x = citation_count, y = n)) +
geom_bar(stat = "identity") +
theme_minimal() +
labs(x = "OpenAlex citation count", y = "Number of publications", caption = dcap)
The table below sorts the JOSS papers in decreasing order by the number of citations in OpenAlex.
DT::datatable(
papers %>%
dplyr::mutate(url = paste0("<a href='", url, "' target='_blank'>",
url,"</a>")) %>%
dplyr::arrange(desc(citation_count)) %>%
dplyr::select(title, url, published.date, citation_count),
escape = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
## Warning in instance$preRenderHook(instance): It seems your data is too big for
## client-side DataTables. You may consider server-side processing:
## https://rstudio.github.io/DT/server.html
plotly::ggplotly(
ggplot(papers, aes(x = published.date, y = citation_count, label = title)) +
geom_point(alpha = 0.5) + theme_bw() + scale_y_sqrt() +
geom_smooth() +
labs(x = "Date of publication", y = "OpenAlex citation count", caption = dcap) +
theme(axis.title = element_text(size = 15)),
tooltip = c("label", "x", "y")
)
## Warning: Removed 6 rows containing non-finite outside the scale range
## (`stat_smooth()`).
## Warning: The following aesthetics were dropped during statistical transformation: label.
## ℹ This can happen when ggplot fails to infer the correct grouping structure in
## the data.
## ℹ Did you forget to specify a `group` aesthetic or to convert a numerical
## variable into a factor?
Here, we plot the citation count for all papers published within each half year, sorted in decreasing order.
ggplot(papers %>% dplyr::group_by(halfyear) %>%
dplyr::arrange(desc(citation_count)) %>%
dplyr::mutate(idx = seq_along(citation_count)),
aes(x = idx, y = citation_count)) +
geom_point(alpha = 0.5) +
facet_wrap(~ halfyear, scales = "free") +
theme_bw() +
labs(x = "Index", y = "OpenAlex citation count", caption = dcap)
## Warning: Removed 6 rows containing missing values or values outside the scale range
## (`geom_point()`).
In these plots we investigate whether the time a submission spends in the pre-review or review stage (or their sum) has changed over time. The blue curve corresponds to a rolling median for submissions over 120 days.
## Helper functions (modified from https://stackoverflow.com/questions/65147186/geom-smooth-with-median-instead-of-mean)
rolling_median <- function(formula, data, xwindow = 120, ...) {
## Get order of x-values and sort x/y
ordr <- order(data$x)
x <- data$x[ordr]
y <- data$y[ordr]
## Initialize vector for smoothed y-values
ys <- rep(NA, length(x))
## Calculate median y-value for each unique x-value
for (xs in setdiff(unique(x), NA)) {
## Get x-values in the window, and calculate median of corresponding y
j <- ((xs - xwindow/2) < x) & (x < (xs + xwindow/2))
ys[x == xs] <- median(y[j], na.rm = TRUE)
}
y <- ys
structure(list(x = x, y = y, f = approxfun(x, y)), class = "rollmed")
}
predict.rollmed <- function(mod, newdata, ...) {
setNames(mod$f(newdata$x), newdata$x)
}
ggplot(papers, aes(x = prerev_opened, y = as.numeric(days_in_pre))) +
geom_point() +
geom_smooth(formula = y ~ x, method = "rolling_median",
se = FALSE, method.args = list(xwindow = 120)) +
theme_bw() +
labs(x = "Date of pre-review opening", y = "Number of days in pre-review",
caption = dcap) +
theme(axis.title = element_text(size = 15))
ggplot(papers, aes(x = review_opened, y = as.numeric(days_in_rev))) +
geom_point() +
geom_smooth(formula = y ~ x, method = "rolling_median",
se = FALSE, method.args = list(xwindow = 120)) +
theme_bw() +
labs(x = "Date of review opening", y = "Number of days in review",
caption = dcap) +
theme(axis.title = element_text(size = 15))
ggplot(papers, aes(x = prerev_opened,
y = as.numeric(days_in_pre) + as.numeric(days_in_rev))) +
geom_point() +
geom_smooth(formula = y ~ x, method = "rolling_median",
se = FALSE, method.args = list(xwindow = 120)) +
theme_bw() +
labs(x = "Date of pre-review opening", y = "Number of days in pre-review + review",
caption = dcap) +
theme(axis.title = element_text(size = 15))
Next, we consider the languages used by the submissions, both as reported by JOSS and based on the information encoded in available GitHub repositories (for the latter, we also record the number of bytes of code written in each language). Note that a given submission can use multiple languages.
## Language information from JOSS
sspl <- strsplit(papers$languages, ",")
all_languages <- unique(unlist(sspl))
langs <- do.call(dplyr::bind_rows, lapply(all_languages, function(l) {
data.frame(language = l,
nbr_submissions_JOSS_API = sum(vapply(sspl, function(v) l %in% v, 0)))
}))
## Language information from GitHub software repos
a <- lapply(strsplit(papers$repo_languages_bytes, ","), function(w) strsplit(w, ":"))
a <- a[sapply(a, length) > 0]
langbytes <- as.data.frame(t(as.data.frame(a))) %>%
setNames(c("language", "bytes")) %>%
dplyr::mutate(bytes = as.numeric(bytes)) %>%
dplyr::filter(!is.na(language)) %>%
dplyr::group_by(language) %>%
dplyr::summarize(nbr_bytes_GitHub = sum(bytes),
nbr_repos_GitHub = length(bytes)) %>%
dplyr::arrange(desc(nbr_bytes_GitHub))
langs <- dplyr::full_join(langs, langbytes, by = "language")
ggplot(langs %>% dplyr::arrange(desc(nbr_submissions_JOSS_API)) %>%
dplyr::filter(nbr_submissions_JOSS_API > 10) %>%
dplyr::mutate(language = factor(language, levels = language)),
aes(x = language, y = nbr_submissions_JOSS_API)) +
geom_bar(stat = "identity") +
theme_bw() +
theme(axis.text.x = element_text(angle = 90, hjust = 1, vjust = 0.5)) +
labs(x = "", y = "Number of submissions", caption = dcap) +
theme(axis.title = element_text(size = 15))
DT::datatable(
langs %>% dplyr::arrange(desc(nbr_bytes_GitHub)),
escape = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
ggplot(langs, aes(x = nbr_repos_GitHub, y = nbr_bytes_GitHub)) +
geom_point() + scale_x_log10() + scale_y_log10() + geom_smooth() +
theme_bw() +
labs(x = "Number of repos using the language",
y = "Total number of bytes of code\nwritten in the language",
caption = dcap) +
theme(axis.title = element_text(size = 15))
ggplotly(
ggplot(papers, aes(x = citation_count, y = repo_nbr_stars,
label = title)) +
geom_point(alpha = 0.5) + scale_x_sqrt() + scale_y_sqrt() +
theme_bw() +
labs(x = "OpenAlex citation count", y = "Number of stars, GitHub repo",
caption = dcap) +
theme(axis.title = element_text(size = 15)),
tooltip = c("label", "x", "y")
)
ggplot(papers, aes(x = as.numeric(prerev_opened - repo_created))) +
geom_histogram(bins = 50) +
theme_bw() +
labs(x = "Time (days) from repo creation to JOSS pre-review start",
caption = dcap) +
theme(axis.title = element_text(size = 15))
ggplot(papers, aes(x = as.numeric(repo_pushed - review_closed))) +
geom_histogram(bins = 50) +
theme_bw() +
labs(x = "Time (days) from closure of JOSS review to most recent commit in repo",
caption = dcap) +
theme(axis.title = element_text(size = 15)) +
facet_wrap(~ year(published.date), scales = "free_y")
Submissions associated with rOpenSci and pyOpenSci are not considered here, since they are not explicitly reviewed at JOSS.
ggplot(papers %>%
dplyr::filter(!grepl("rOpenSci|pyOpenSci", prerev_labels)) %>%
dplyr::mutate(year = year(published.date)),
aes(x = nbr_reviewers)) + geom_bar() +
facet_wrap(~ year) + theme_bw() +
labs(x = "Number of reviewers", y = "Number of submissions", caption = dcap)
Submissions associated with rOpenSci and pyOpenSci are not considered here, since they are not explicitly reviewed at JOSS.
reviewers <- papers %>%
dplyr::filter(!grepl("rOpenSci|pyOpenSci", prerev_labels)) %>%
dplyr::mutate(year = year(published.date)) %>%
dplyr::select(reviewers, year) %>%
tidyr::separate_rows(reviewers, sep = ",")
## Most active reviewers
DT::datatable(
reviewers %>% dplyr::group_by(reviewers) %>%
dplyr::summarize(nbr_reviews = length(year),
timespan = paste(unique(c(min(year), max(year))),
collapse = " - ")) %>%
dplyr::arrange(desc(nbr_reviews)),
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
reviewers <- papers %>%
dplyr::filter(!grepl("rOpenSci|pyOpenSci", prerev_labels)) %>%
dplyr::mutate(year = year(published.date)) %>%
dplyr::filter(as.Date(published.date) >= (lubridate::today() - 5 * 365.25)) %>%
dplyr::select(reviewers, year) %>%
tidyr::separate_rows(reviewers, sep = ",")
## Most active reviewers
DT::datatable(
reviewers %>% dplyr::group_by(reviewers) %>%
dplyr::summarize(nbr_reviews = length(year),
timespan = paste(unique(c(min(year), max(year))),
collapse = " - ")) %>%
dplyr::arrange(desc(nbr_reviews)),
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
reviewers <- papers %>%
dplyr::filter(!grepl("rOpenSci|pyOpenSci", prerev_labels)) %>%
dplyr::mutate(year = year(published.date)) %>%
dplyr::filter(as.Date(published.date) >= (lubridate::today() - 365.25)) %>%
dplyr::select(reviewers, year) %>%
tidyr::separate_rows(reviewers, sep = ",")
## Most active reviewers
DT::datatable(
reviewers %>% dplyr::group_by(reviewers) %>%
dplyr::summarize(nbr_reviews = length(year),
timespan = paste(unique(c(min(year), max(year))),
collapse = " - ")) %>%
dplyr::arrange(desc(nbr_reviews)),
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE)
)
ggplot(papers %>%
dplyr::mutate(year = year(published.date),
`r/pyOpenSci` = factor(
grepl("rOpenSci|pyOpenSci", prerev_labels),
levels = c("TRUE", "FALSE"))),
aes(x = editor)) + geom_bar(aes(fill = `r/pyOpenSci`)) +
theme_bw() + facet_wrap(~ year, ncol = 1) +
scale_fill_manual(values = c(`TRUE` = "grey65", `FALSE` = "grey35")) +
theme(axis.text.x = element_text(angle = 90, hjust = 1, vjust = 0.5)) +
labs(x = "Editor", y = "Number of submissions", caption = dcap)
all_licenses <- sort(unique(papers$repo_license))
license_levels = c(grep("apache", all_licenses, value = TRUE),
grep("bsd", all_licenses, value = TRUE),
grep("mit", all_licenses, value = TRUE),
grep("gpl", all_licenses, value = TRUE),
grep("mpl", all_licenses, value = TRUE))
license_levels <- c(license_levels, setdiff(all_licenses, license_levels))
ggplot(papers %>%
dplyr::mutate(repo_license = factor(repo_license,
levels = license_levels)),
aes(x = repo_license)) +
geom_bar() +
theme_bw() +
labs(x = "Software license", y = "Number of submissions", caption = dcap) +
theme(axis.title = element_text(size = 15),
axis.text.x = element_text(angle = 90, hjust = 1, vjust = 0.5)) +
facet_wrap(~ year(published.date), scales = "free_y")
## For plots below, replace licenses present in less
## than 2.5% of the submissions by 'other'
tbl <- table(papers$repo_license)
to_replace <- names(tbl[tbl <= 0.025 * nrow(papers)])
ggplot(papers %>%
dplyr::mutate(year = year(published.date)) %>%
dplyr::mutate(repo_license = replace(repo_license,
repo_license %in% to_replace,
"other")) %>%
dplyr::mutate(year = factor(year),
repo_license = factor(
repo_license,
levels = license_levels[license_levels %in% repo_license]
)) %>%
dplyr::group_by(year, repo_license, .drop = FALSE) %>%
dplyr::count() %>%
dplyr::mutate(year = as.integer(as.character(year))),
aes(x = year, y = n, fill = repo_license)) + geom_area() +
theme_minimal() +
scale_fill_brewer(palette = "Set1", name = "Software\nlicense",
na.value = "grey") +
theme(axis.title = element_text(size = 15)) +
labs(x = "Year", y = "Number of submissions", caption = dcap)
ggplot(papers %>%
dplyr::mutate(year = year(published.date)) %>%
dplyr::mutate(repo_license = replace(repo_license,
repo_license %in% to_replace,
"other")) %>%
dplyr::mutate(year = factor(year),
repo_license = factor(
repo_license,
levels = license_levels[license_levels %in% repo_license]
)) %>%
dplyr::group_by(year, repo_license, .drop = FALSE) %>%
dplyr::summarize(n = n()) %>%
dplyr::mutate(freq = n/sum(n)) %>%
dplyr::mutate(year = as.integer(as.character(year))),
aes(x = year, y = freq, fill = repo_license)) + geom_area() +
theme_minimal() +
scale_fill_brewer(palette = "Set1", name = "Software\nlicense",
na.value = "grey") +
theme(axis.title = element_text(size = 15)) +
labs(x = "Year", y = "Fraction of submissions", caption = dcap)
a <- unlist(strsplit(papers$repo_topics, ","))
a <- a[!is.na(a)]
topicfreq <- table(a)
colors <- viridis::viridis(100)
set.seed(1234)
wordcloud::wordcloud(
names(topicfreq), sqrt(topicfreq), min.freq = 1, max.words = 300,
random.order = FALSE, rot.per = 0.05, use.r.layout = FALSE,
colors = colors, scale = c(10, 0.1), random.color = TRUE,
ordered.colors = FALSE, vfont = c("serif", "plain")
)
DT::datatable(as.data.frame(topicfreq) %>%
dplyr::rename(topic = a, nbr_repos = Freq) %>%
dplyr::arrange(desc(nbr_repos)),
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE))
Here, we take a more detailed look at the papers that cite JOSS papers, using data from the Open Citations Corpus.
## Split into several queries
## Randomize the splitting since a whole query may fail if one ID is not recognized
papidx <- seq_len(nrow(papers))
idxL <- split(sample(papidx, length(papidx), replace = FALSE), ceiling(papidx / 50))
citationsL <- lapply(idxL, function(idx) {
tryCatch({
citecorp::oc_coci_cites(doi = papers$alternative.id[idx]) %>%
dplyr::distinct() %>%
dplyr::mutate(citation_info_obtained = as.character(lubridate::today()))
}, error = function(e) {
NULL
})
})
citationsL <- citationsL[vapply(citationsL, function(df) !is.null(df) && nrow(df) > 0, FALSE)]
if (length(citationsL) > 0) {
citations <- do.call(dplyr::bind_rows, citationsL)
} else {
citations <- NULL
}
dim(citations)
## [1] 87897 8
if (!is.null(citations) && is.data.frame(citations) && "oci" %in% colnames(citations)) {
citations <- citations %>%
dplyr::filter(!(oci %in% citations_archive$oci) &
citing != "")
tmpj <- rcrossref::cr_works(dois = unique(citations$citing))$data %>%
dplyr::select(contains("doi"), contains("container.title"), contains("issn"),
contains("type"), contains("publisher"), contains("prefix"))
citations <- citations %>% dplyr::left_join(tmpj, by = c("citing" = "doi"))
## bioRxiv preprints don't have a 'container.title' or 'issn', but we'll assume
## that they can be
## identified from the prefix 10.1101 - set the container.title
## for these records manually; we may or may not want to count these
## (would it count citations twice, both preprint and publication?)
citations$container.title[citations$prefix == "10.1101"] <- "bioRxiv"
## JOSS is represented by 'The Journal of Open Source Software' as well as
## 'Journal of Open Source Software'
citations$container.title[citations$container.title ==
"Journal of Open Source Software"] <-
"The Journal of Open Source Software"
## Remove real self citations (cited DOI = citing DOI)
citations <- citations %>% dplyr::filter(cited != citing)
## Merge with the archive
citations <- dplyr::bind_rows(citations, citations_archive)
} else {
citations <- citations_archive
if (is.null(citations[["citation_info_obtained"]])) {
citations$citation_info_obtained <- NA_character_
}
}
citations$citation_info_obtained[is.na(citations$citation_info_obtained)] <-
"2021-08-11"
write.table(citations, file = "joss_submission_citations.tsv",
row.names = FALSE, col.names = TRUE, sep = "\t", quote = FALSE)
## Latest successful update of new citation data
max(as.Date(citations$citation_info_obtained))
## [1] "2026-08-19"
## Number of JOSS papers with >0 citations included in this collection
length(unique(citations$cited))
## [1] 2615
## Number of JOSS papers with >0 citations according to OpenAlex
length(which(papers$citation_count > 0))
## [1] 2821
## Number of citations from Open Citations Corpus vs OpenAlex
df0 <- papers %>% dplyr::select(doi, citation_count) %>%
dplyr::full_join(citations %>% dplyr::group_by(cited) %>%
dplyr::tally() %>%
dplyr::mutate(n = replace(n, is.na(n), 0)),
by = c("doi" = "cited"))
## Total citation count OpenAlex
sum(df0$citation_count, na.rm = TRUE)
## [1] 133272
## Total citation count Open Citations Corpus
sum(df0$n, na.rm = TRUE)
## [1] 144224
## Ratio of total citation count Open Citations Corpus/OpenAlex
sum(df0$n, na.rm = TRUE)/sum(df0$citation_count, na.rm = TRUE)
## [1] 1.082178
ggplot(df0, aes(x = citation_count, y = n)) +
geom_abline(slope = 1, intercept = 0) +
geom_point(size = 3, alpha = 0.5) +
labs(x = "OpenAlex citation count", y = "Open Citations Corpus citation count",
caption = dcap) +
theme_bw()
## Zoom in
ggplot(df0, aes(x = citation_count, y = n)) +
geom_abline(slope = 1, intercept = 0) +
geom_point(size = 3, alpha = 0.5) +
labs(x = "OpenAlex citation count", y = "Open Citations Corpus citation count",
caption = dcap) +
theme_bw() +
coord_cartesian(xlim = c(0, 75), ylim = c(0, 75))
## Number of journals citing JOSS papers
length(unique(citations$container.title))
## [1] 15213
length(unique(citations$issn))
## [1] 10762
topcit <- citations %>% dplyr::group_by(container.title) %>%
dplyr::summarize(nbr_citations_of_joss_papers = length(cited),
nbr_cited_joss_papers = length(unique(cited)),
nbr_citing_papers = length(unique(citing)),
nbr_selfcitations_of_joss_papers = sum(author_sc == "yes"),
fraction_selfcitations = signif(nbr_selfcitations_of_joss_papers /
nbr_citations_of_joss_papers, digits = 3)) %>%
dplyr::arrange(desc(nbr_cited_joss_papers))
DT::datatable(topcit,
escape = FALSE, rownames = FALSE,
filter = list(position = 'top', clear = FALSE),
options = list(scrollX = TRUE))
## Warning in instance$preRenderHook(instance): It seems your data is too big for
## client-side DataTables. You may consider server-side processing:
## https://rstudio.github.io/DT/server.html
plotly::ggplotly(
ggplot(topcit, aes(x = nbr_citations_of_joss_papers, y = nbr_cited_joss_papers,
label = container.title)) +
geom_abline(slope = 1, intercept = 0, linetype = "dashed", color = "grey") +
geom_point(size = 3, alpha = 0.5) +
theme_bw() +
labs(caption = dcap, x = "Number of citations of JOSS papers",
y = "Number of cited JOSS papers")
)
plotly::ggplotly(
ggplot(topcit, aes(x = nbr_citations_of_joss_papers, y = nbr_cited_joss_papers,
label = container.title)) +
geom_abline(slope = 1, intercept = 0, linetype = "dashed", color = "grey") +
geom_point(size = 3, alpha = 0.5) +
theme_bw() +
coord_cartesian(xlim = c(0, 100), ylim = c(0, 50)) +
labs(caption = dcap, x = "Number of citations of JOSS papers",
y = "Number of cited JOSS papers")
)
write.table(topcit, file = "joss_submission_citations_byjournal.tsv",
row.names = FALSE, col.names = TRUE, sep = "\t", quote = FALSE)
The tibble object with all data collected above is serialized to a file that can be downloaded and reused.
head(papers) %>% as.data.frame()
## created deposited doi indexed issn member
## 1 2016-05-11 2017-10-24 10.21105/joss.00011 2025-02-21 2475-9066 8722
## 2 2016-05-11 2017-10-23 10.21105/joss.00017 2025-02-21 2475-9066 8722
## 3 2016-05-16 2017-10-24 10.21105/joss.00012 2025-02-21 2475-9066 8722
## 4 2016-05-18 2017-10-23 10.21105/joss.00021 2025-02-21 2475-9066 8722
## 5 2016-05-23 2017-10-23 10.21105/joss.00018 2026-08-04 2475-9066 8722
## 6 2016-05-27 2017-10-23 10.21105/joss.00016 2025-02-21 2475-9066 8722
## prefix publisher score source reference.count references.count
## 1 10.21105 The Open Journal 0 Crossref 1 1
## 2 10.21105 The Open Journal 0 Crossref 6 6
## 3 10.21105 The Open Journal 0 Crossref 4 4
## 4 10.21105 The Open Journal 0 Crossref 1 1
## 5 10.21105 The Open Journal 0 Crossref 3 3
## 6 10.21105 The Open Journal 0 Crossref 3 3
## is.referenced.by.count
## 1 8
## 2 1
## 3 5
## 4 0
## 5 17
## 6 1
## title
## 1 carl: a likelihood-free inference toolbox
## 2 Application Skeleton: Generating Synthetic Applications for Infrastructure Research
## 3 mst_clustering: Clustering via Euclidean Minimum Spanning Trees
## 4 pyuca: a Python implementation of the Unicode Collation Algorithm
## 5 Xenomapper: Mapping reads in a mixed species context
## 6 R3D2: Relativistic Reactive Riemann problem solver for Deflagrations and Detonations
## type url alternative.id
## 1 journal-article https://doi.org/10.21105/joss.00011 10.21105/joss.00011
## 2 journal-article https://doi.org/10.21105/joss.00017 10.21105/joss.00017
## 3 journal-article https://doi.org/10.21105/joss.00012 10.21105/joss.00012
## 4 journal-article https://doi.org/10.21105/joss.00021 10.21105/joss.00021
## 5 journal-article https://doi.org/10.21105/joss.00018 10.21105/joss.00018
## 6 journal-article https://doi.org/10.21105/joss.00016 10.21105/joss.00016
## container.title published.print issue issued page
## 1 The Journal of Open Source Software 2016-05-11 1 2016-05-11 11
## 2 The Journal of Open Source Software 2016-05-11 1 2016-05-11 17
## 3 The Journal of Open Source Software 2016-05-16 1 2016-05-16 12
## 4 The Journal of Open Source Software 2016-05-18 1 2016-05-18 21
## 5 The Journal of Open Source Software 2016-05-22 1 2016-05-22 18
## 6 The Journal of Open Source Software 2016-05-27 1 2016-05-27 16
## volume short.container.title
## 1 1 JOSS
## 2 1 JOSS
## 3 1 JOSS
## 4 1 JOSS
## 5 1 JOSS
## 6 1 JOSS
## author
## 1 https://orcid.org/0000-0002-2082-3106, https://orcid.org/0000-0002-5769-7094, https://orcid.org/0000-0002-7205-0053, FALSE, FALSE, FALSE, Gilles, Kyle, Juan, Louppe, Cranmer, Pavez, first, additional, additional, author, author, author, crossref, crossref, crossref
## 2 https://orcid.org/0000-0001-5921-0035, https://orcid.org/0000-0001-5934-7525, https://orcid.org/0000-0002-7228-4327, https://orcid.org/0000-0003-0527-1435, https://orcid.org/0000-0002-5040-026X, https://orcid.org/0000-0002-9162-6003, FALSE, FALSE, FALSE, FALSE, FALSE, FALSE, Zhao, Daniel, Andre, Matteo, Shantenu, Yadu, Zhang, S. Katz, Merzky, Turilli, Jha, Nand, first, additional, additional, additional, additional, additional, author, author, author, author, author, author, crossref, crossref, crossref, crossref, crossref, crossref
## 3 https://orcid.org/0000-0002-9623-3401, FALSE, Jake, VanderPlas, first, author, crossref
## 4 https://orcid.org/0000-0001-6534-8866, FALSE, J., K. Tauber, first, author, crossref
## 5 https://orcid.org/0000-0001-6624-4698, FALSE, Matthew, J. Wakefield, first, crossref, author
## 6 https://orcid.org/0000-0002-1530-781X, https://orcid.org/0000-0003-4805-0309, FALSE, FALSE, Alice, Ian, Harpole, Hawke, first, additional, author, author, crossref, crossref
## citation_count openalex_id affil_countries_all
## 1 16 https://openalex.org/W2346823692
## 2 6 https://openalex.org/W2369753323 US;NL;CN
## 3 6 https://openalex.org/W2407867073 US
## 4 1 https://openalex.org/W2404749572 FR
## 5 22 https://openalex.org/W2400204549 AU
## 6 1 https://openalex.org/W2402211085 GB
## affil_countries_first
## 1
## 2 US;NL
## 3 US
## 4 FR
## 5 AU
## 6 GB
## api_title
## 1 carl: a likelihood-free inference toolbox
## 2 Application Skeleton: Generating Synthetic Applications for Infrastructure Research
## 3 mst_clustering: Clustering via Euclidean Minimum Spanning Trees
## 4 pyuca: a Python implementation of the Unicode Collation Algorithm
## 5 Xenomapper: Mapping reads in a mixed species context
## 6 R3D2: Relativistic Reactive Riemann problem solver for Deflagrations and Detonations
## api_state
## 1 accepted
## 2 accepted
## 3 accepted
## 4 accepted
## 5 accepted
## 6 accepted
## author_affiliations
## 1 New York University;Federico Santa María University
## 2 AMPLab and BIDS, University of California, Berkeley;National Center for Supercomputing Applications, University of Illinois Urbana-Champaign;RADICAL Laboratory, Rutgers University;Computation Institute, University of Chicago
## 3 University of Washington eScience Institute
## 4 None
## 5 The Walter and Eliza Hall Institute, The University of Melbourne
## 6 University of Southampton
## editor reviewers nbr_reviewers
## 1 @arfon @betatim 1
## 2 @arfon @krother 1
## 3 @arfon @nicoguaro 1
## 4 @arfon @luizirber 1
## 5 @arfon @xuanxu 1
## 6 @arfon @kyleniemeyer 1
## repo_url review_issue_id
## 1 https://github.com/diana-hep/carl 11
## 2 https://github.com/applicationskeleton/Skeleton 17
## 3 http://github.com/jakevdp/mst_clustering 12
## 4 https://github.com/jtauber/pyuca 21
## 5 http://github.com/genomematt/xenomapper/ 18
## 6 https://github.com/harpolea/r3d2 16
## prereview_issue_id languages
## 1 NA Python,Mako
## 2 NA Python,C
## 3 NA Jupyter Notebook,Python
## 4 NA Python
## 5 NA Python
## 6 NA Python,Jupyter Notebook
## archive_doi
## 1 https://doi.org/10.5281/zenodo.47798
## 2 https://doi.org/10.5281/zenodo.13750
## 3 https://doi.org/10.5281/zenodo.50995
## 4 https://doi.org/10.5281/zenodo.51622
## 5 https://doi.org/10.5281/zenodo.51798
## 6 https://doi.org/10.5281/zenodo.51096
## review_title
## 1 carl: a likelihood-free inference toolbox
## 2 Application Skeleton: Generating Synthetic Applications for Infrastructure Research
## 3 mst_clustering: Clustering via Minimum Spanning Trees
## 4 pyuca: a Python implementation of the Unicode Collation Algorithm
## 5 Xenomapper - short read mapping in a mixed species context
## 6 R3D2: Relativistic Reactive Riemann problem solver for Deflagrations and Detonations
## review_number review_state review_opened review_closed review_ncomments
## 1 11 closed 2016-05-04 2016-05-11 36
## 2 17 closed 2016-05-11 2016-05-12 24
## 3 12 closed 2016-05-05 2016-05-16 18
## 4 21 closed 2016-05-17 2016-05-18 15
## 5 18 closed 2016-05-12 2016-05-22 14
## 6 16 closed 2016-05-09 2016-05-27 15
## review_labels prerev_title prerev_state prerev_opened
## 1 accepted,recommend-accept,published <NA> <NA> <NA>
## 2 accepted,recommend-accept,published <NA> <NA> <NA>
## 3 accepted,recommend-accept,published <NA> <NA> <NA>
## 4 accepted,recommend-accept,published <NA> <NA> <NA>
## 5 accepted,recommend-accept,published <NA> <NA> <NA>
## 6 accepted,recommend-accept,published <NA> <NA> <NA>
## prerev_closed prerev_ncomments prerev_labels days_in_pre days_in_rev
## 1 <NA> NA <NA> NA days 7 days
## 2 <NA> NA <NA> NA days 1 days
## 3 <NA> NA <NA> NA days 11 days
## 4 <NA> NA <NA> NA days 1 days
## 5 <NA> NA <NA> NA days 10 days
## 6 <NA> NA <NA> NA days 18 days
## to_review repo_created repo_updated repo_pushed repo_nbr_stars
## 1 TRUE 2015-11-23 2025-06-08 2017-05-04 56
## 2 TRUE 2014-07-30 2026-07-26 2017-01-11 14
## 3 TRUE 2015-10-28 2026-08-22 2016-05-16 88
## 4 TRUE 2012-06-21 2026-07-12 2026-07-25 226
## 5 TRUE 2015-03-23 2023-09-04 2021-06-01 10
## 6 TRUE 2015-12-16 2025-02-21 2019-08-29 6
## repo_language repo_languages_bytes
## 1 Python Python:142285,Mako:33700,Shell:2653,TeX:433
## 2 TeX TeX:1013812,Python:156571,C:20118,Shell:3689,Makefile:1476
## 3 Jupyter Notebook Jupyter Notebook:416862,Python:18113,TeX:1418,Makefile:249
## 4 Python Python:21319
## 5 Python Python:957782,Dockerfile:1756,TeX:1258
## 6 Jupyter Notebook Jupyter Notebook:2691815,Python:95428,TeX:3989
## repo_topics repo_license repo_nbr_contribs
## 1 bsd-3-clause 5
## 2 mit 7
## 3 bsd-2-clause 1
## 4 unicode,unicode-collation-algorithm mit 5
## 5 gpl-3.0 2
## 6 mit 3
## repo_nbr_contribs_2ormore repo_info_obtained published.date halfyear
## 1 4 2026-08-26 2016-05-11 2016H1
## 2 6 2026-08-26 2016-05-11 2016H1
## 3 1 2026-08-26 2016-05-16 2016H1
## 4 5 2026-08-26 2016-05-18 2016H1
## 5 1 2026-09-02 2016-05-22 2016H1
## 6 2 2026-08-26 2016-05-27 2016H1
## nbr_authors
## 1 3
## 2 6
## 3 1
## 4 1
## 5 1
## 6 2
saveRDS(papers, file = "joss_submission_analytics.rds")
To read the current version of this file directly from GitHub, use the following code:
papers <- readRDS(gzcon(url("https://github.com/openjournals/joss-analytics/blob/gh-pages/joss_submission_analytics.rds?raw=true")))
For non-R users, we also save a parquet file. For this, we convert any list column to a JSON string, and the difftime columns to numeric values.
papers_parquet <- papers |>
mutate(across(where(~ inherits(.x, "difftime")), as.numeric)) |>
mutate(across(where(is.list), ~ map_chr(.x, ~ jsonlite::toJSON(.x))))
write_parquet(papers_parquet, "joss_submission_analytics.parquet")
This can be read into python as follows (assuming that
pandas and pyarrow or fastparquet
are available):
import pandas as pd
papers = pd.read_parquet("https://github.com/openjournals/joss-analytics/blob/gh-pages/joss_submission_analytics.parquet?raw=true")
sessionInfo()
## R version 4.6.1 (2026-06-24)
## Platform: aarch64-apple-darwin23
## Running under: macOS Tahoe 26.5.2
##
## Matrix products: default
## BLAS: /Library/Frameworks/R.framework/Versions/4.6/Resources/lib/libRblas.0.dylib
## LAPACK: /Library/Frameworks/R.framework/Versions/4.6/Resources/lib/libRlapack.dylib; LAPACK version 3.12.1
##
## locale:
## [1] en_US.UTF-8/en_US.UTF-8/en_US.UTF-8/C/en_US.UTF-8/en_US.UTF-8
##
## time zone: UTC
## tzcode source: internal
##
## attached base packages:
## [1] stats graphics grDevices utils datasets methods base
##
## other attached packages:
## [1] arrow_25.0.1 openalexR_3.1.0 stringr_1.6.0 gt_1.3.0
## [5] rworldmap_1.3-8 sp_2.2-3 readr_2.2.0 citecorp_0.3.0
## [9] plotly_4.12.1 DT_0.34.0 jsonlite_2.0.0 purrr_1.2.2
## [13] gh_1.6.1 lubridate_1.9.5 ggplot2_4.0.3 tidyr_1.3.2
## [17] dplyr_1.2.1 rcrossref_1.2.1 tibble_3.3.1
##
## loaded via a namespace (and not attached):
## [1] tidyselect_1.2.1 viridisLite_0.4.3 farver_2.1.2
## [4] viridis_0.6.5 urltools_1.7.3.1 fields_17.3
## [7] S7_0.2.2 fastmap_1.2.0 promises_1.5.0
## [10] digest_0.6.39 dotCall64_1.2 timechange_0.4.0
## [13] mime_0.13 lifecycle_1.0.5 terra_1.9-46
## [16] magrittr_2.0.5 compiler_4.6.1 rlang_1.3.0
## [19] sass_0.4.10 tools_4.6.1 wordcloud_2.6
## [22] utf8_1.2.6 yaml_2.3.12 data.table_1.18.6.1
## [25] knitr_1.51 labeling_0.4.3 fauxpas_0.6.0
## [28] htmlwidgets_1.6.4 bit_4.6.0 curl_8.0.0
## [31] reticulate_1.46.0 plyr_1.8.9 xml2_1.6.0
## [34] RColorBrewer_1.1-3 httpcode_0.3.0 miniUI_0.1.2
## [37] withr_3.0.3 triebeard_0.4.1 grid_4.6.1
## [40] xtable_1.8-8 gitcreds_0.1.2 scales_1.4.0
## [43] crul_1.6.0 cli_3.6.6 rmarkdown_2.32
## [46] crayon_1.5.3 generics_0.1.4 otel_0.2.0
## [49] httr_1.4.9 tzdb_0.5.0 cachem_1.1.0
## [52] splines_4.6.1 maps_3.4.3 parallel_4.6.1
## [55] assertthat_0.2.1 vctrs_0.7.3 Matrix_1.7-5
## [58] hms_1.1.4 bit64_4.8.6 crosstalk_1.2.2
## [61] jquerylib_0.1.4 glue_1.8.1 spam_2.11-4
## [64] codetools_0.2-20 stringi_1.8.9 gtable_0.3.6
## [67] later_1.4.8 raster_3.6-32 pillar_1.11.1
## [70] htmltools_0.5.9 httr2_1.3.0 R6_2.6.1
## [73] vroom_1.7.1 evaluate_1.0.5 shiny_1.14.0
## [76] lattice_0.22-9 png_0.1-9 httpuv_1.6.17
## [79] bslib_0.12.0 Rcpp_1.1.2 gridExtra_2.3.1
## [82] nlme_3.1-169 mgcv_1.9-4 whisker_0.4.1
## [85] xfun_0.60 fs_2.1.0 pkgconfig_2.0.3