Coverage and analysis samples

This document uses the one-minute aligned recordings to identify interpretable days and hours. Coverage is evaluated on a complete 24-hour local wall-clock grid. A usable day requires at least 80% finite melEDI minutes across that day, including diary-defined sleep. An hour requires at least 50% coverage for hourly outcomes. An unusable hour does not remove its finite minutes from daily calculations. Days containing only zero melEDI values are excluded.

Apply the coverage rules

The explicit wall-clock grid keeps missing minutes in the denominator. The output retains observations and validity indicators, allowing each later metric to apply its own temporal-support rule.

sample_flows <- list()
settings <- list()
for (placement in c("glasses", "chest")) {
  aligned <- read_rds_artifact(file.path(paths$aligned,
    paste0("light_", placement, "_aligned.rds")), "data.frame")
  validate_aligned_coverage_input(aligned, placement)
  coverage <- apply_wall_clock_coverage_rules(
    minute_data = aligned, value_cols = c("MEDI", "LIGHT"),
    coverage_signal = "MEDI", id_cols = c("site", "Id", "position"),
    participant_days = NULL, minimum_hour_coverage = 0.5,
    minimum_day_coverage = 0.8, exclude_all_zero_days = TRUE
  )
  eligible <- add_eligible_signal_channels(aligned, coverage)
  hourly <- dplyr::arrange(coverage$hourly, .data$site, .data$Id,
    .data$position, .data$local_date, .data$clock_hour)
  daily <- dplyr::arrange(coverage$daily, .data$site, .data$Id,
    .data$position, .data$local_date)
  gaps <- coverage_gap_runs(coverage$wall_grid, placement)
  sample_flows[[placement]] <- coverage_sample_flow(eligible, placement)
  validate_coverage_outputs(aligned, eligible, hourly, daily, placement)
  settings[[placement]] <- tibble::tibble(
    placement = placement, minimum_hour_coverage = 0.5,
    minimum_day_coverage = 0.8, expected_wall_minutes_per_day = 1440L,
    diary_sleep_excluded_from_denominator = FALSE,
    minute_values_masked_by_hour_threshold = FALSE,
    all_zero_medi_exclusion_applied = TRUE,
    eligible_hours = sum(hourly$hour_eligible),
    eligible_days = sum(daily$day_eligible),
    ineligible_days = sum(!daily$day_eligible)
  )
  write_rds_artifact(eligible, file.path(paths$coverage,
    paste0("light_", placement, "_coverage.rds")), "preparation/02-coverage-and-samples.qmd")
  for (name in c("hourly", "daily")) {
    write_csv_artifact(get(name), file.path(paths$coverage,
      paste0("light_", placement, "_", name, "_coverage.csv")), "preparation/02-coverage-and-samples.qmd")
  }
  write_csv_artifact(gaps, file.path(paths$coverage,
    paste0("light_", placement, "_gap_runs.csv")), "preparation/02-coverage-and-samples.qmd")
  rm(aligned, coverage, eligible)
  invisible(gc())
}
dplyr::bind_rows(settings)
# A tibble: 2 × 10
  placement minimum_hour_coverage minimum_day_coverage expected_wall_minutes_p…¹
  <chr>                     <dbl>                <dbl>                     <int>
1 glasses                     0.5                  0.8                      1440
2 chest                       0.5                  0.8                      1440
# ℹ abbreviated name: ¹​expected_wall_minutes_per_day
# ℹ 6 more variables: diary_sleep_excluded_from_denominator <lgl>,
#   minute_values_masked_by_hour_threshold <lgl>,
#   all_zero_medi_exclusion_applied <lgl>, eligible_hours <int>,
#   eligible_days <int>, ineligible_days <int>

Sample flow

Participant and day counts below show how the coverage rules determine the available analysis sample. Metric-specific missingness is handled in the metric document rather than removing an otherwise usable day.

sample_flow <- dplyr::bind_rows(sample_flows) |>
  dplyr::arrange(.data$placement, .data$scope, .data$site, .data$stage_order)
write_csv_artifact(sample_flow, file.path(paths$coverage, "sample_flow.csv"), "preparation/02-coverage-and-samples.qmd")
write_csv_artifact(dplyr::bind_rows(settings), file.path(paths$coverage, "coverage_settings.csv"), "preparation/02-coverage-and-samples.qmd")
dplyr::filter(sample_flow, .data$scope == "overall")
# A tibble: 18 × 12
   placement scope   site  stage_order stage_branch         stage   participants
   <chr>     <chr>   <chr>       <int> <chr>                <chr>          <int>
 1 chest     overall ALL             1 common               aligne…          157
 2 chest     overall ALL             2 MEDI                 precov…          157
 3 chest     overall ALL             3 LIGHT                precov…          157
 4 chest     overall ALL             4 coverage_sensitivity all_ze…          154
 5 chest     overall ALL             5 coverage             primar…          154
 6 chest     overall ALL             6 MEDI                 medi_e…          154
 7 chest     overall ALL             7 LIGHT                light_…          154
 8 chest     overall ALL             8 hourly_metric        hourly…          154
 9 chest     overall ALL             9 coverage_sensitivity hour_s…          154
10 glasses   overall ALL             1 common               aligne…          143
11 glasses   overall ALL             2 MEDI                 precov…          143
12 glasses   overall ALL             3 LIGHT                precov…          143
13 glasses   overall ALL             4 coverage_sensitivity all_ze…          141
14 glasses   overall ALL             5 coverage             primar…          141
15 glasses   overall ALL             6 MEDI                 medi_e…          141
16 glasses   overall ALL             7 LIGHT                light_…          141
17 glasses   overall ALL             8 hourly_metric        hourly…          141
18 glasses   overall ALL             9 coverage_sensitivity hour_s…          141
# ℹ 5 more variables: participant_days <int>, true_utc_minutes <int>,
#   wall_minutes <int>, finite_medi_minutes <int>, finite_light_minutes <int>