## ----setup, include = FALSE---------------------------------------------------
knitr::opts_chunk$set(
  collapse = TRUE,
  comment = "#>"
)
library(tidycreel)

## ----complete_example, eval=FALSE---------------------------------------------
# library(tidycreel)
# 
# # Standard workflow using complete trips (default)
# data(example_calendar)
# data(example_counts)
# data(example_interviews)
# 
# design <- creel_design(example_calendar, date = date, strata = day_type) |>
#   add_counts(example_counts) |>
#   add_interviews(example_interviews,
#     catch = catch_total,
#     effort = hours_fished,
#     harvest = catch_kept,
#     trip_status = trip_status,
#     trip_duration = trip_duration
#   )
# 
# # Estimate CPUE using complete trips only (default)
# cpue <- estimate_catch_rate(design)
# print(cpue)
# 
# # Estimate total catch
# total_catch <- estimate_total_catch(design)
# print(total_catch)

## ----low_complete_warning, eval=FALSE-----------------------------------------
# # If complete trip percentage drops below 10%, you'll see:
# # Warning: Only 8% of interviews (n=5) are complete trips
# # Best practice: ≥10% of interviews should be complete trips
# # Consider extending survey hours or sampling more trips to completion

## ----anti_pattern, eval=FALSE-------------------------------------------------
# # WRONG: Do not manually pool trip types
# all_interviews <- rbind(complete_data, incomplete_data)
# estimate_catch_rate(design_with_all_data) # INVALID — always, even after validation passes
# 
# # WRONG: Do not use custom weights to combine
# weighted_mean(c(complete_cpue, incomplete_cpue)) # INVALID!

## ----roving-default-----------------------------------------------------------
roving_design <- creel_design(
  example_calendar,
  date = date, strata = day_type,
  survey_type = "instantaneous", h_open = 14
) |>
  add_counts(example_counts, count_col = effort_hours) |>
  add_interviews(
    example_interviews,
    catch = catch_total, effort = hours_fished, n_anglers = n_anglers,
    harvest = catch_kept, trip_status = trip_status,
    trip_duration = trip_duration,
    interview_type = "roving"
  )

# Unspecified: routed to all-trip mean-of-ratios
estimate_catch_rate(roving_design)$method

# Named explicitly: complete trips, ratio-of-means
estimate_catch_rate(roving_design, use_trips = "complete")$method

## ----validation_setup, eval=FALSE---------------------------------------------
# library(tidycreel)
# 
# # Your survey data should have trip_status field
# # Check trip type distribution
# table(your_interviews$trip_status)
# 
# # Ensure you have:
# # - At least 10% complete trips (Pollock et al. 1994)
# # - At least 30 incomplete trips (for stable estimates)
# # - At least 10 complete trips (for ratio estimation)

## ----validation_design, eval=FALSE--------------------------------------------
# design <- creel_design(your_calendar, date = date, strata = day_type) |>
#   add_counts(your_counts) |>
#   add_interviews(your_interviews,
#     catch = catch_total,
#     effort = hours_fished,
#     trip_status = trip_status,
#     trip_duration = trip_duration
#   )

## ----run_validation, eval=FALSE-----------------------------------------------
# # Validate incomplete trips using TOST
# validation <- validate_incomplete_trips(design,
#   catch = catch_total,
#   effort = hours_fished
# )
# 
# print(validation)

## ----validation_plot, eval=FALSE----------------------------------------------
# # Print method automatically shows plot
# print(validation) # Plot appears after text output
# 
# # Or explicitly plot
# plot(validation)

## ----example_pass-------------------------------------------------------------
library(tidycreel)

# Simulate data where catch rates are similar for complete vs incomplete
set.seed(42)

# Create calendar
calendar <- data.frame(
  date = seq.Date(as.Date("2024-06-01"), as.Date("2024-06-14"), by = "day"),
  day_type = rep(c("weekday", "weekend"), length.out = 14)
)

# Create counts
counts <- data.frame(
  date = calendar$date,
  day_type = calendar$day_type,
  effort_hours = round(runif(14, min = 50, max = 150))
)

# Simulate interviews with SIMILAR catch rates for both trip types
# (stationary catch rate throughout day)
n_complete <- 50
n_incomplete <- 120

# Base CPUE around 2.4 fish/hour for both groups (PASSING scenario)
complete_interviews <- data.frame(
  date = sample(calendar$date, n_complete, replace = TRUE),
  hours_fished = runif(n_complete, min = 2, max = 8),
  trip_status = "complete",
  trip_duration = runif(n_complete, min = 2, max = 8)
)
complete_interviews$catch_total <- rpois(n_complete,
  lambda = complete_interviews$hours_fished * 2.4
)

incomplete_interviews <- data.frame(
  date = sample(calendar$date, n_incomplete, replace = TRUE),
  hours_fished = runif(n_incomplete, min = 1, max = 6),
  trip_status = "incomplete"
)
# For incomplete trips, trip_duration = hours_fished (time interviewed, not total trip)
incomplete_interviews$trip_duration <- incomplete_interviews$hours_fished
# Similar CPUE for incomplete trips (2.3-2.5 range)
incomplete_interviews$catch_total <- rpois(n_incomplete,
  lambda = incomplete_interviews$hours_fished * 2.35
)

interviews <- rbind(complete_interviews, incomplete_interviews)

# Create design
design <- creel_design(calendar, date = date, strata = day_type) |>
  add_counts(counts) |>
  add_interviews(interviews,
    catch = catch_total,
    effort = hours_fished,
    trip_status = trip_status,
    trip_duration = trip_duration
  )

# Run validation
validation_pass <- validate_incomplete_trips(design,
  catch = catch_total,
  effort = hours_fished
)

print(validation_pass)

## ----after_pass, eval=FALSE---------------------------------------------------
# # Now safe to use incomplete trips
# cpue_incomplete <- estimate_catch_rate(design, use_trips = "incomplete")
# print(cpue_incomplete)
# 
# # Or use diagnostic mode to compare
# cpue_diagnostic <- estimate_catch_rate(design, use_trips = "diagnostic")
# print(cpue_diagnostic)

## ----example_fail-------------------------------------------------------------
library(tidycreel)
set.seed(123)

# Same calendar and counts as before
calendar <- data.frame(
  date = seq.Date(as.Date("2024-06-01"), as.Date("2024-06-14"), by = "day"),
  day_type = rep(c("weekday", "weekend"), length.out = 14)
)

counts <- data.frame(
  date = calendar$date,
  day_type = calendar$day_type,
  effort_hours = round(runif(14, min = 50, max = 150))
)

# Simulate interviews with DIFFERENT catch rates (FAILING scenario)
# Complete trips average full day (includes productive morning hours)
# Incomplete trips are mostly afternoon interviews (lower catch rates)

n_complete <- 45
n_incomplete <- 110

# Complete trips: higher CPUE (includes morning fishing, ~3.0 fish/hour)
complete_interviews <- data.frame(
  date = sample(calendar$date, n_complete, replace = TRUE),
  hours_fished = runif(n_complete, min = 3, max = 8),
  trip_status = "complete",
  trip_duration = runif(n_complete, min = 3, max = 8)
)
complete_interviews$catch_total <- rpois(n_complete,
  lambda = complete_interviews$hours_fished * 3.0
)

# Incomplete trips: lower CPUE (afternoon interviews, ~2.0 fish/hour)
incomplete_interviews <- data.frame(
  date = sample(calendar$date, n_incomplete, replace = TRUE),
  hours_fished = runif(n_incomplete, min = 1, max = 5),
  trip_status = "incomplete"
)
# For incomplete trips, trip_duration = hours_fished (time interviewed, not total trip)
incomplete_interviews$trip_duration <- incomplete_interviews$hours_fished
incomplete_interviews$catch_total <- rpois(n_incomplete,
  lambda = incomplete_interviews$hours_fished * 2.0
)

interviews_biased <- rbind(complete_interviews, incomplete_interviews)

# Create design
design_biased <- creel_design(calendar, date = date, strata = day_type) |>
  add_counts(counts) |>
  add_interviews(interviews_biased,
    catch = catch_total,
    effort = hours_fished,
    trip_status = trip_status,
    trip_duration = trip_duration
  )

# Run validation
validation_fail <- validate_incomplete_trips(design_biased,
  catch = catch_total,
  effort = hours_fished
)

print(validation_fail)

## ----after_fail, eval=FALSE---------------------------------------------------
# # DO NOT use incomplete trips
# # Stick with default complete trip estimation
# # `design_biased` is an access-point design, so an unspecified `use_trips`
# # resolves to complete trips. On a roving design it would not -- name it.
# cpue_complete <- estimate_catch_rate(design_biased, use_trips = "complete")
# print(cpue_complete)
# 
# # Investigate why estimates differ
# # Possible reasons:
# # - Time-of-day effects (morning vs afternoon catch rates)
# # - Trip length correlates with skill level
# # - Fish behavior changes throughout day (feeding windows)
# # - Different angler types (early vs late arrivals)

## ----diagnostic_mode, eval=FALSE----------------------------------------------
# # Compare complete vs incomplete estimates
# cpue_diagnostic <- estimate_catch_rate(design,
#   catch = catch_total,
#   effort = hours_fished,
#   use_trips = "diagnostic"
# )
# 
# print(cpue_diagnostic)

## ----grouped_validation, eval=FALSE-------------------------------------------
# # Validate with grouping
# validation_grouped <- validate_incomplete_trips(design,
#   catch = catch_total,
#   effort = hours_fished,
#   by = day_type
# )
# 
# print(validation_grouped)

## ----custom_threshold, eval=FALSE---------------------------------------------
# # Use stricter threshold (±15%), keeping the previous setting to restore later
# old_opts <- options(tidycreel.equivalence_threshold = 0.15)
# validation_strict <- validate_incomplete_trips(design,
#   catch = catch_total,
#   effort = hours_fished
# )
# 
# # Use more permissive threshold (±25%)
# options(tidycreel.equivalence_threshold = 0.25)
# validation_permissive <- validate_incomplete_trips(design,
#   catch = catch_total,
#   effort = hours_fished
# )
# 
# # Restore the threshold that was in effect before
# options(old_opts)

## ----truncation, eval=FALSE---------------------------------------------------
# # Default: truncate incomplete trips <0.5 hours (30 minutes)
# cpue_incomplete <- estimate_catch_rate(design,
#   use_trips = "incomplete",
#   estimator = "mor",
#   truncate_at = 0.5 # Default
# )
# 
# # Custom truncation threshold
# cpue_truncated <- estimate_catch_rate(design,
#   use_trips = "incomplete",
#   estimator = "mor",
#   truncate_at = 1.0 # More conservative: only trips ≥1 hour
# )
# 
# # Disable truncation (not recommended)
# cpue_no_truncation <- estimate_catch_rate(design,
#   use_trips = "incomplete",
#   estimator = "mor",
#   truncate_at = 0 # Includes all incomplete trips
# )

