Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
befb20c
Add `retirement_date` field to personnel data and documentation, upda…
ifeanyi588 Sep 14, 2026
5b7f85e
first attempt at growth decomposition
galileukim Sep 14, 2026
487774e
keep at it with the wage bill
galileukim Sep 15, 2026
e4a86bc
incorporate entry and exit effects
galileukim Sep 15, 2026
502affb
relax github workflows
galileukim Sep 15, 2026
2361812
update renv.lock
galileukim Sep 15, 2026
07ac5e2
fix checks and tests
galileukim Sep 15, 2026
96cd1aa
Add actuarial decrement rate estimation and service table functions
ifeanyi588 Sep 16, 2026
4877e0b
Export `compute_service_table` and `smooth_decrement_rates`, rename `…
ifeanyi588 Sep 16, 2026
7886982
add package dependencies
galileukim Sep 16, 2026
a35cab1
Merge remote-tracking branch 'origin/dev-demographics' into wagebill-…
galileukim Sep 16, 2026
7ceedc6
fix conflicts
galileukim Sep 16, 2026
36fe6ca
fix documentation
galileukim Sep 16, 2026
2f764f1
rename scripts
galileukim Sep 16, 2026
6015449
modify names of scripts
galileukim Sep 17, 2026
15d86b1
fix invalid and dead roxygen import tags
galileukim Sep 17, 2026
ca14a3d
enable roxygen markdown and fix inverted @param arity docs
galileukim Sep 17, 2026
9d850d3
rename the .data argument to data
galileukim Sep 17, 2026
e13cd56
standardise grouping and data argument names
galileukim Sep 17, 2026
2ac3229
update articles for the renamed arguments
galileukim Sep 17, 2026
5c2f1d1
bump to 0.4.0 and document the changes
galileukim Sep 17, 2026
4584de5
match release notes to the existing NEWS style
galileukim Sep 17, 2026
4aab8ae
redact news
galileukim Sep 17, 2026
fe6496d
append news
galileukim Sep 17, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/R-CMD-check.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@
# Need help debugging build failures? Start at https://github.com/r-lib/actions#where-to-find-help
on:
push:
branches: [main, master]
pull_request:

name: R-CMD-check.yaml
Expand Down
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -4,3 +4,4 @@
.Ruserdata
*.rds
docs
spielplatz/
9 changes: 6 additions & 3 deletions DESCRIPTION
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
Package: govhr
Type: Package
Title: Clean and analyze data that informs human resource management
Version: 0.3.5
Version: 0.4.0
Description: This package is a toolkit developed by the World Bank. It cleans, harmonizes, and analyzes human resource data with a standardized methodology.
License: MIT + file LICENSE
Encoding: UTF-8
Expand Down Expand Up @@ -35,9 +35,11 @@ Imports:
tidygraph,
ggraph,
ggiraph,
igraph
Suggests:
igraph,
fixest,
forcats,
ggstats
Suggests:
readr,
testthat (>= 3.0.0),
reactable
Expand All @@ -50,4 +52,5 @@ Authors@R: c(
person("Galileu", "Kim", , "galileukim@worldbank.org", role = c("aut", "cre")),
person("Ifeanyi", "Edochie", , "iedochie@worldbank.org", role = c("aut"))
)
Roxygen: list(markdown = TRUE)
Config/roxygen2/version: 8.1.0
34 changes: 18 additions & 16 deletions NAMESPACE
Original file line number Diff line number Diff line change
Expand Up @@ -19,15 +19,18 @@ export(compute_fastsummary)
export(compute_global_consistency)
export(compute_global_coverage)
export(compute_growth)
export(compute_growth_decomposition)
export(compute_growth_summary)
export(compute_movement_cost)
export(compute_pension_ratio)
export(compute_percentile)
export(compute_quantile)
export(compute_record_consistency)
export(compute_service_table)
export(compute_time_trend)
export(compute_trend_summary)
export(compute_value_consistency)
export(compute_wage_decomposition)
export(compute_wagebill)
export(compute_workforce_movement)
export(convert_constant_ppp)
Expand All @@ -39,11 +42,11 @@ export(dedup_values)
export(define_fns)
export(deflate_to_real)
export(detect_career_transition)
export(detect_career_transitions)
export(detect_inconsistent_cols)
export(detect_personnel_event)
export(detect_reallocation)
export(detect_retirement)
export(estimate_decrement_rates)
export(estimate_exit_rates)
export(fastcount)
export(fastprop)
Expand All @@ -64,6 +67,7 @@ export(ggplot_point_line)
export(ggplot_segment)
export(guess_date_frequency)
export(harmonize_columns)
export(model_wage)
export(pivot_data360)
export(plot_bar_growth)
export(plot_bar_total)
Expand All @@ -75,6 +79,8 @@ export(plot_coverage_heatmap)
export(plot_coverage_trend)
export(plot_decile)
export(plot_histogram)
export(plot_model_wage)
export(plot_model_wage_fes)
export(plot_movement)
export(plot_movement_cost)
export(plot_segment)
Expand All @@ -86,14 +92,16 @@ export(remove_duplicate_personnel)
export(rescale_baseline)
export(sample_group)
export(scale_plot_height)
export(smooth_decrement_rates)
export(validate_data)
import(dplyr)
import(ggplot2)
import(glue)
import(httr)
importFrom(broom,tidy)
importFrom(collapse,fquantile)
importFrom(data.table,
":=",
.N,
CJ,
as.data.table,
copy,
Expand All @@ -120,7 +128,6 @@ importFrom(dplyr,
any_of,
arrange,
case_when,
collect,
count,
desc,
distinct,
Expand Down Expand Up @@ -150,6 +157,8 @@ importFrom(dplyr,
union
)
importFrom(dtplyr,lazy_dt)
importFrom(fixest,fixef)
importFrom(forcats,fct_reorder)
importFrom(ggiraph,
geom_point_interactive,
girafe,
Expand All @@ -158,8 +167,8 @@ importFrom(ggiraph,
)
importFrom(ggplot2,
aes,
arrow,
coord_cartesian,
element_text,
expansion,
facet_wrap,
geom_col,
Expand All @@ -172,9 +181,9 @@ importFrom(ggplot2,
geom_vline,
ggplot,
guide_axis,
label_wrap_gen,
labs,
margin,
position_jitter,
scale_color_manual,
scale_size_identity,
scale_x_continuous,
Expand All @@ -195,10 +204,10 @@ importFrom(ggraph,
scale_edge_width_continuous
)
importFrom(ggrepel,geom_text_repel)
importFrom(ggthemes,scale_color_few)
importFrom(ggstats,ggcoef_model)
importFrom(ggthemes,scale_colour_few)
importFrom(glue,glue)
importFrom(grDevices,colorRampPalette)
importFrom(grid,unit)
importFrom(httr,
GET,
content,
Expand Down Expand Up @@ -262,16 +271,9 @@ importFrom(tibble,
)
importFrom(tidygraph,as_tbl_graph)
importFrom(tidyr,
complete,
nest,
pivot_longer,
nesting,
pivot_wider
)
importFrom(tidyselect,eval_select)
importFrom(validate,
confront,
description,
label,
summary,
validator,
values
)
8 changes: 8 additions & 0 deletions NEWS.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,11 @@
# govhr 0.4.0
This release:
- Introduces novel wage bill modelling and demographic analysis.
- Renames the function arguments, ensuring consistency.
- Warns on deprecated argument names, to be removed in the next relase.
- Fixes documentation and their rendering.
- Fixes invalid and unused imports.

# govhr 0.3.5
This release:
- Ports data transformation and plotting functions from govhrapp to govhr.
Expand Down
18 changes: 9 additions & 9 deletions R/qcheck_consistency.R → R/consistency.R
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
#' Compute the proportion of consistent records and values in a data frame.
#'
#' @param data A data frame.
#' @param id_col A character string specifying the name of the column that uniquely identifies records.
#' @param id_col A string specifying the name of the column that uniquely identifies records.
#' @param value_cols A character vector specifying the name(s) of columns whose values
#' are to be checked for consistency. Value consistency is computed separately for
#' each column and averaged across columns before being combined with record consistency.
Expand All @@ -10,7 +10,7 @@
#' @import dplyr
#' @importFrom purrr map_dbl
#'
#' @return A numeric value representing the proportion of consistent records and values in the data frame.
#' @returns A numeric value representing the proportion of consistent records and values in the data frame.
#' @details Consistency is defined as the proportion of records and values that are consistent
#' across the dataset. A record is considered consistent if it has a unique identifier and all
#' its associated values are consistent. A value is considered consistent if it does not
Expand Down Expand Up @@ -57,18 +57,18 @@ compute_global_consistency <- function(data, id_col, value_cols, digits = 2) {
round(digits)
}

#' Compute the proportion of consistent records in a data frame.
#' Compute the proportion of consistent records in a data frame
#'
#' @param data A data frame.
#' @param id_col A character string specifying the name of the column that uniquely identifies records (e.g., "personnel_id" or "contract_id").
#' @param id_col A string specifying the name of the column that uniquely identifies records (e.g., "personnel_id" or "contract_id").
#' @param group_cols A character vector specifying the names of the columns to group by. Default is NULL, which means no grouping.
#' @param digits An integer specifying the number of decimal places to round the result to. Default is 2.
#'
#' @import dplyr
#' @importFrom data.table as.data.table fifelse
#' @importFrom tibble as_tibble
#'
#' @return A data frame with the proportion of consistent records in the data frame, optionally by group.
#' @returns A data frame with the proportion of consistent records in the data frame, optionally by group.
#'
#' @details A record is considered consistent if it has a unique identifier and all its associated values are consistent.
#' The function computes the proportion of consistent records in the data frame, optionally grouped by specified columns.
Expand Down Expand Up @@ -130,18 +130,18 @@ compute_record_consistency <- function(
tibble::as_tibble(result)
}

#' Compute the proportion of consistent values in a data frame.
#' Compute the proportion of consistent values in a data frame
#'
#' @param data A data frame.
#' @param id_col A character string specifying the name of the column that uniquely identifies records.
#' @param value_col A character string specifying the name of the column whose values are to be checked for consistency.
#' @param id_col A string specifying the name of the column that uniquely identifies records.
#' @param value_col A string specifying the name of the column whose values are to be checked for consistency.
#' @param group_cols A character vector specifying the names of the columns to group by. Default is no grouping.
#' @param digits An integer specifying the number of decimal places to round the result to. Default is 2.
#'
#' @importFrom data.table as.data.table
#' @importFrom tibble as_tibble
#'
#' @return A data frame with the proportion of consistent values in the data frame, optionally by group.
#' @returns A data frame with the proportion of consistent values in the data frame, optionally by group.
#'
#' @details Consistency is broadly defined as the proportion of records and values that are consistent
#' across the dataset. A value is considered consistent if it does not differ from
Expand Down
31 changes: 17 additions & 14 deletions R/qcheck_coverage.R → R/coverage.R
Original file line number Diff line number Diff line change
@@ -1,49 +1,52 @@
#' Compute coverage of non-missing values in a dataset.
#'
#' @param .data A data frame.
#' @param group A character string specifying the column name to group by. If NULL, coverage is computed for the entire data set.
#' @param data A data frame.
#' @param group_cols A string specifying the column name to group by. If NULL, coverage is computed for the entire data set.
#' @param include_ref_date A logical value indicating whether to include the `ref_date` column in the grouping.
#' @param aggregate A logical value indicating whether to aggregate coverage values by the `group`.
#' @param group Deprecated. Use `group_cols` instead.
#'
#' @importFrom data.table as.data.table
#' @importFrom tibble as_tibble
#'
#' @return A data frame with coverage values for each column, optionally grouped by the specified `group`.
#' @returns A data frame with coverage values for each column, optionally grouped by the specified `group`.
#'
#' @export
compute_coverage <- function(
.data,
group = NULL,
data,
group_cols = NULL,
include_ref_date = FALSE,
aggregate = FALSE
aggregate = FALSE,
group = NULL
) {
dt <- data.table::as.data.table(.data)
group_cols <- resolve_renamed_arg(group_cols, group, "group", "group_cols")
dt <- data.table::as.data.table(data)
data_cols <- colnames(dt)

if (include_ref_date) {
group <- unique(c("ref_date", group))
group_cols <- unique(c("ref_date", group_cols))
}

summary_cols <- setdiff(data_cols, group)
summary_cols <- setdiff(data_cols, group_cols)

# wide: one coverage value per summary column, one row per group
if (is.null(group)) {
if (is.null(group_cols)) {
coverage_wide <- dt[,
lapply(.SD, \(col) (sum(!is.na(col)) / length(col)) * 100),
.SDcols = summary_cols
]
} else {
coverage_wide <- dt[,
lapply(.SD, \(col) (sum(!is.na(col)) / length(col)) * 100),
by = c(group),
by = c(group_cols),
.SDcols = summary_cols
]
}

# long: pivot summary_cols into variable/coverage pairs
coverage_data <- data.table::melt(
coverage_wide,
id.vars = group,
id.vars = group_cols,
measure.vars = summary_cols,
variable.name = "variable",
value.name = "coverage",
Expand All @@ -53,7 +56,7 @@ compute_coverage <- function(
if (aggregate) {
coverage_data <- coverage_data[,
.(coverage = mean(coverage, na.rm = TRUE)),
by = c(group)
by = c(group_cols)
]
}

Expand All @@ -65,7 +68,7 @@ compute_coverage <- function(
#' @param data A data frame.
#' @param digits An integer specifying the number of decimal places to round the result to. Default is 2.
#'
#' @return A numeric value representing the proportion of missing values in the data frame.
#' @returns A numeric value representing the proportion of missing values in the data frame.
#'
#' @export
compute_global_coverage <- function(data, digits = 2) {
Expand Down
1 change: 1 addition & 0 deletions R/data.R
Original file line number Diff line number Diff line change
Expand Up @@ -367,6 +367,7 @@
#' \item{race}{Worker's race or broad ethnic classification, where available.}
#' \item{tribe}{Worker's ethnic or tribal affiliation, where available.}
#' \item{first_employment_date}{Date the worker first entered the public service.}
#' \item{retirement_date}{The date of retirement for those whose \code{employment_status} is \code{"pensioner"}}
#' }
#'
#' @details
Expand Down
10 changes: 5 additions & 5 deletions R/data360api.R
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
#' This function retrieves data from the Data360 API.
#'
#' @return A tibble containing data, including country codes, country names,
#' @returns A tibble containing data, including country codes, country names,
#' years, and values.
#' @examples
#' \dontrun{
Expand All @@ -16,7 +16,7 @@
#' @import dplyr
#' @importFrom janitor clean_names
#' @importFrom jsonlite fromJSON
#' @importFrom tibble as_tibble tibble
#' @importFrom tibble tibble
#' @export
get_data360_api <- function(dataset_id, indicator_id, pivot = TRUE) {
base_url <- "https://data360api.worldbank.org/data360/data"
Expand Down Expand Up @@ -89,7 +89,7 @@ get_data360_api <- function(dataset_id, indicator_id, pivot = TRUE) {
#' - `INDICATOR`: Indicator code or name
#' - `OBS_VALUE`: Observation value for the indicator
#'
#' @return A tibble in wide format with columns:
#' @returns A tibble in wide format with columns:
#' - `country_code`: The country or region code (from `REF_AREA`)
#' - `year`: The year or time period (from `TIME_PERIOD`)
#' - One column per unique `INDICATOR`, containing corresponding values from `OBS_VALUE`
Expand Down Expand Up @@ -143,10 +143,10 @@ pivot_data360 <- function(data) {
#' URL, retrieves the metadata in JSON format, and parses it into an R list or
#' data frame.
#'
#' @param dataset_id A character string or numeric identifier specifying the
#' @param dataset_id A string or numeric identifier specifying the
#' dataset for which metadata should be retrieved.
#'
#' @return A list (or data frame) containing the metadata associated with the
#' @returns A list (or data frame) containing the metadata associated with the
#' requested dataset, as returned by the Data360 API.
#'
#' @examples
Expand Down
Loading
Loading