Select filtered occurrences based on number of records and spatial autocorrelation (see details)
Arguments
- occ_list
list. A list with filtered specie occurrences testing several values (see
occfilt_envandoccfilt_geo)- x
character. Column name with longitude data
- y
character. Column name with latitude data
- env_layer
SpatRaster. Raster variables that will be used to fit the model. Factor variables will be removed.
- filter_prop
logical. If TRUE, the function will return a list with the filtered occurrences and a tibble with the spatial autocorrelation and number of occurrence values
Value
If filter_prop = FALSE, a tibble with selected filtered occurrences. If filter_prop = TRUE, a list with following objects:
A tibble with selected filtered occurrences
A tibble with filter properties with columns:
filt_value: values used for filtering, the value with an asterisk will denote the one selected
n_records: number of occurrence
mean_autocorr: mean spatial autocorrelation.
the remaining columns have the spatial autocorrelation values for each variable.
Details
The function implement the approach used in Velazco et al. (2020) which consists in calculating for each filtered dataset:
1- the number of occurrence.
2- the spatial autocorrelation based on Morans'I for each variable
3- the mean spatial autocorrelation among variables
Then function will select those dataset with average spatial autocorrelation lower than the mean of all dataset, and from this subset will select the one with the highest number occurrences.
If use occfilt_select cite Velazco et al. (2020) as reference.
References
Velazco, S. J. E., Svenning, J-C., Ribeiro, B. R., & Laureto, L. M. O. (2020). On opportunities and threats to conserve the phylogenetic diversity of Neotropical palms. Diversity and Distributions, 27, 512–523. https://doi.org/10.1111/ddi.13215
Examples
# \donttest{
require(terra)
require(dplyr)
# Environmental variables
somevar <- system.file("external/somevar.tif", package = "flexsdm")
somevar <- terra::rast(somevar)
plot(somevar)
# Species occurrences
data("spp")
spp
#> # A tibble: 1,150 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -5541. -145138. 0
#> 2 sp1 -51981. 16322. 0
#> 3 sp1 -269871. 69512. 1
#> 4 sp1 -96261. -32008. 0
#> 5 sp1 269589. -566338. 0
#> 6 sp1 29829. -328468. 0
#> 7 sp1 -152691. 393782. 0
#> 8 sp1 -195081. 253652. 0
#> 9 sp1 -951. -277978. 0
#> 10 sp1 145929. -271498. 0
#> # ℹ 1,140 more rows
spp1 <- spp %>% dplyr::filter(species == "sp1", pr_ab == 1)
## %######################################################%##
#### Cellsize method ####
## %######################################################%##
# Using cellsize method
filtered_occ <- occfilt_geo(
data = spp1,
x = "x",
y = "y",
env_layer = somevar,
method = c("cellsize", factor = c(1, 4, 8, 12, 16, 20)),
prj = crs(somevar)
)
#> Extracting values from raster ...
#> 16 records were removed because they have NAs for some variables
#> Number of unfiltered records: 234
#> Factor: x1
#> Distance threshold (km): 1.539
#> Number of filtered records: 233
#> Factor: x4
#> Distance threshold (km): 6.157
#> Number of filtered records: 203
#> Factor: x8
#> Distance threshold (km): 12.313
#> Number of filtered records: 156
#> Factor: x12
#> Distance threshold (km): 18.47
#> Number of filtered records: 118
#> Factor: x16
#> Distance threshold (km): 24.626
#> Number of filtered records: 96
#> Factor: x20
#> Distance threshold (km): 30.783
#> Number of filtered records: 78
filtered_occ
#> $`1`
#> # A tibble: 233 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -269871. 69512. 1
#> 2 sp1 -149991. 267962. 1
#> 3 sp1 -126231. 196142. 1
#> 4 sp1 91659. -156748. 1
#> 5 sp1 -210471. 326282. 1
#> 6 sp1 -140541. 284972. 1
#> 7 sp1 -217491. 65732. 1
#> 8 sp1 -178611. 225032. 1
#> 9 sp1 -92481. 155642. 1
#> 10 sp1 -367611. 266072. 1
#> # ℹ 223 more rows
#>
#> $`4`
#> # A tibble: 203 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -269871. 69512. 1
#> 2 sp1 -149991. 267962. 1
#> 3 sp1 -126231. 196142. 1
#> 4 sp1 -210471. 326282. 1
#> 5 sp1 -140541. 284972. 1
#> 6 sp1 -217491. 65732. 1
#> 7 sp1 -178611. 225032. 1
#> 8 sp1 -92481. 155642. 1
#> 9 sp1 -367611. 266072. 1
#> 10 sp1 -109491. 46292. 1
#> # ℹ 193 more rows
#>
#> $`8`
#> # A tibble: 156 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -269871. 69512. 1
#> 2 sp1 -149991. 267962. 1
#> 3 sp1 91659. -156748. 1
#> 4 sp1 -210471. 326282. 1
#> 5 sp1 -140541. 284972. 1
#> 6 sp1 -217491. 65732. 1
#> 7 sp1 -92481. 155642. 1
#> 8 sp1 -367611. 266072. 1
#> 9 sp1 -184551. 425372. 1
#> 10 sp1 -260151. 57632. 1
#> # ℹ 146 more rows
#>
#> $`12`
#> # A tibble: 118 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -269871. 69512. 1
#> 2 sp1 -149991. 267962. 1
#> 3 sp1 -140541. 284972. 1
#> 4 sp1 -178611. 225032. 1
#> 5 sp1 -34431. 212072. 1
#> 6 sp1 -331701. 304412. 1
#> 7 sp1 -215871. -51178. 1
#> 8 sp1 -60351. -307138. 1
#> 9 sp1 -331161. 387842. 1
#> 10 sp1 -88971. 137282. 1
#> # ℹ 108 more rows
#>
#> $`16`
#> # A tibble: 96 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -210471. 326282. 1
#> 2 sp1 -313611. 400802. 1
#> 3 sp1 -202911. -64138. 1
#> 4 sp1 -34431. 212072. 1
#> 5 sp1 -172941. -113008. 1
#> 6 sp1 -60351. -307138. 1
#> 7 sp1 -304971. 219632. 1
#> 8 sp1 -195351. 85982. 1
#> 9 sp1 -23091. 56012. 1
#> 10 sp1 -114621. 54122. 1
#> # ℹ 86 more rows
#>
#> $`20`
#> # A tibble: 78 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -126231. 196142. 1
#> 2 sp1 -367611. 266072. 1
#> 3 sp1 -109491. 46292. 1
#> 4 sp1 -34431. 212072. 1
#> 5 sp1 -143781. 233942. 1
#> 6 sp1 -215871. -51178. 1
#> 7 sp1 -60351. -307138. 1
#> 8 sp1 -306321. 134852. 1
#> 9 sp1 -195351. 85982. 1
#> 10 sp1 -328731. 405392. 1
#> # ℹ 68 more rows
#>
# Select filtered occurrences based on
# number of records and spatial autocorrelation
occ_selected <- occfilt_select(
occ_list = filtered_occ,
x = "x",
y = "y",
env_layer = somevar,
filter_prop = FALSE
)
#> Dataset with filtered value 12 was selected
occ_selected
#> # A tibble: 118 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -269871. 69512. 1
#> 2 sp1 -149991. 267962. 1
#> 3 sp1 -140541. 284972. 1
#> 4 sp1 -178611. 225032. 1
#> 5 sp1 -34431. 212072. 1
#> 6 sp1 -331701. 304412. 1
#> 7 sp1 -215871. -51178. 1
#> 8 sp1 -60351. -307138. 1
#> 9 sp1 -331161. 387842. 1
#> 10 sp1 -88971. 137282. 1
#> # ℹ 108 more rows
occ_selected <- occfilt_select(
occ_list = filtered_occ,
x = "x",
y = "y",
env_layer = somevar,
filter_prop = TRUE
)
#> Dataset with filtered value 12 was selected
occ_selected$occ
#> # A tibble: 118 × 4
#> species x y pr_ab
#> <chr> <dbl> <dbl> <dbl>
#> 1 sp1 -269871. 69512. 1
#> 2 sp1 -149991. 267962. 1
#> 3 sp1 -140541. 284972. 1
#> 4 sp1 -178611. 225032. 1
#> 5 sp1 -34431. 212072. 1
#> 6 sp1 -331701. 304412. 1
#> 7 sp1 -215871. -51178. 1
#> 8 sp1 -60351. -307138. 1
#> 9 sp1 -331161. 387842. 1
#> 10 sp1 -88971. 137282. 1
#> # ℹ 108 more rows
occ_selected$filter_prop
#> filt_value mean_autocorr n_records CFP_1 CFP_2 CFP_3 CFP_4
#> 1 1 0.3219265 233 0.3534014 0.20178498 0.3352363 0.3972832
#> 2 4 0.2860045 203 0.3181670 0.15331928 0.3015718 0.3709598
#> 3 8 0.2602486 156 0.3055744 0.12870505 0.2635844 0.3431307
#> 4 * 12 0.2291901 118 0.2568785 0.10808796 0.2311247 0.3206691
#> 5 16 0.2136998 96 0.2664798 0.07228080 0.2457754 0.2702633
#> 6 20 0.2045660 78 0.2539334 0.07758901 0.1968988 0.2898429
# }