# Create bibliographies from Zotero collections
library(jsonlite)
library(fs)
library(janitor)
library(tidyverse)
library(knitr)
library(httr2)
library(yaml)
# Input and output paths for attached PDFs
PATH_ZOTERO <- Sys.getenv("PATH_ZOTERO")
stopifnot(PATH_ZOTERO != "") # Hack to require a proper path
PATH_OUT <- "bibliography/files"
# Get publications as jzon (BetterBibTeX JSON debug format) from local Zotero API
# https://retorque.re/zotero-better-bibtex/exporting/pull/index.html
dat <- request(
"http://127.0.0.1:23119/better-bibtex/export?/library;name:My%20Library/collection/cv.jzon"
) |>
req_perform() |>
resp_body_string() |>
fromJSON() |>
pluck("items") |>
tibble() |>
arrange(desc(date)) |>
select(-c(relations))
# Copy files from Zotero storage to repaired paths here
dat <- dat |>
mutate(
file = map_chr(attachments, ~ pluck(.x, "path", .default = NA_character_)),
path = str_replace(file, PATH_ZOTERO, PATH_OUT)
)
walk2(
dat$file,
dat$path,
~ if (!is.na(.y)) {
dir_create(dirname(.y))
file_copy(.x, .y, overwrite = TRUE)
}
)
# Wrangle data
# Choose only specific item types
dat <- dat |>
clean_names() |>
filter(
item_type %in%
c("journalArticle", "preprint", "presentation", "computerProgram", "book")
)
# Different types have different source names
dat <- dat |>
mutate(
source = coalesce(publication_title, repository, publisher, place),
.keep = "unused"
)
# Create a string of authors' last names
dat <- dat |>
mutate(
authors = map_chr(
creators,
~ combine_words(.x$lastName, and = " & ", oxford_comma = FALSE)
),
.keep = "unused"
)
# Use doi (preferred) or url
dat <- dat |>
mutate(
url = coalesce(str_c("https://doi.org/", doi), url)
)
# Write a YAML file for Quarto listing
# Drop empty vectors and NA scalars so they are YAML nulls
na_null <- function(x) if (length(x) == 0 || anyNA(x)) NULL else x
# Build one named list per row
build_item <- function(...) {
row <- list(...)
item <- list(
title = row$title,
url = row$url,
date = row$date,
type = row$item_type,
authors = row$authors,
doi = row$doi,
source = row$source,
file = row$path,
meeting = row$meeting_name,
categories = str_to_lower(pluck(row$tags, "tag", .default = character(0))),
abstract = row$abstract_note
) |>
map(na_null)
if (!is.null(row$extra) && !is.na(row$extra)) {
extra <- yaml.load(row$extra)
# Drop unwanted fields written by Zotero plugins
extra[c("Read_Status", "Read_Status_Date")] <- NULL
item <- modifyList(item, extra)
}
item
}
items <- dat |>
pmap(build_item)
writeLines(as.yaml(items), "bibliography/items.yml")