diff --git a/pyproject.toml b/pyproject.toml index bc61932..302fde4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -17,7 +17,7 @@ [project] name = "pysits" -version = "1.5.4" +version = "2.0.0.dev2" description = "Python wrapper for the sits R package" readme = "README.md" requires-python = ">=3.10,<4" diff --git a/pysits/__init__.py b/pysits/__init__.py index e0294cd..5355f73 100644 --- a/pysits/__init__.py +++ b/pysits/__init__.py @@ -20,7 +20,6 @@ from .conversions.dsl.mask import MaskValue from .conversions.dsl.tuning import hparam from .settings import __version__ -from .sits.classification import sits_classify, sits_label_classification, sits_smooth from .sits.colors import ( sits_colors, sits_colors_reset, @@ -62,6 +61,9 @@ sits_apply, sits_bands, sits_bbox, + sits_classify, + sits_encode, + sits_label_classification, sits_labels, sits_labels_summary, sits_list_collections, @@ -69,10 +71,17 @@ sits_mixture_model, sits_reduce, sits_select, + sits_smooth, sits_timeline, ) from .sits.data import sits_summary as summary -from .sits.exporters import sits_as_geopandas, sits_as_xarray, sits_to_csv, sits_to_xlsx +from .sits.exporters import ( + sits_as_geopandas, + sits_as_xarray, + sits_timeseries_to_csv, + sits_to_csv, + sits_to_xlsx, +) from .sits.impute import ( impute_linear, impute_mean, @@ -87,6 +96,7 @@ sits_kfold_validate, sits_lightgbm, sits_lighttae, + sits_lstm_fcn, sits_mlp, sits_model_export, sits_pre_train, @@ -107,7 +117,6 @@ sits_cluster_clean, sits_cluster_dendro, sits_cluster_frequency, - sits_encode, sits_geo_dist, sits_get_class, sits_get_data, @@ -118,6 +127,7 @@ sits_pred_references, sits_pred_sample, sits_predictors, + sits_random_sampling, sits_reduce_imbalance, sits_sample, sits_sampling_design, @@ -125,6 +135,7 @@ sits_som_clean_samples, sits_som_evaluate_cluster, sits_som_map, + sits_som_remove_samples, sits_stats, sits_stratified_sampling, sits_validate, @@ -137,13 +148,21 @@ read_rds, ) from .sits.visualization import sits_plot as plot -from .sits.visualization import sits_view +from .sits.visualization import sits_sankey, sits_view __all__ = ( # Classification "sits_classify", "sits_smooth", "sits_label_classification", + # Embeddings + "sits_pre_train", + "sits_encode", + "sits_ssl_mae", + "sits_ssl_lejepa", + "sits_ssl_vicreg", + "sits_barlow_twins", + "sits_contrastive_learning", # Cube "sits_cube", "sits_clean", @@ -170,7 +189,7 @@ "sits_config_user_file", "sits_config_value", "sits_parallel", - # Data management + # Data "sits_bands", "sits_timeline", "sits_labels", @@ -194,6 +213,7 @@ "sits_resnet", "sits_tempcnn", "sits_lighttae", + "sits_lstm_fcn", "sits_svm", "sits_xgboost", "sits_lightgbm", @@ -202,12 +222,6 @@ "sits_model_export", "sits_formula_linear", "sits_formula_logref", - "sits_ssl_mae", - "sits_ssl_lejepa", - "sits_ssl_vicreg", - "sits_barlow_twins", - "sits_contrastive_learning", - "sits_pre_train", # Impute "impute_linear", "impute_mean", @@ -228,13 +242,14 @@ "sits_som_map", "sits_som_evaluate_cluster", "sits_som_clean_samples", + "sits_som_remove_samples", "sits_geo_dist", "sits_patterns", "sits_sample", "sits_reduce_imbalance", "sits_sampling_design", "sits_stratified_sampling", - "sits_encode", + "sits_random_sampling", # Tiles "sits_tiles_to_roi", "sits_roi_to_tiles", @@ -249,12 +264,14 @@ "sits_as_xarray", "sits_as_geopandas", "sits_to_xlsx", + "sits_timeseries_to_csv", # Tuning "sits_tuning_hparams", "sits_tuning", # Visualization "plot", "sits_view", + "sits_sankey", # DSL Variables "MaskValue", "hparam", diff --git a/pysits/conversions/tibble.py b/pysits/conversions/tibble.py index 19e4d2b..db1c5a2 100644 --- a/pysits/conversions/tibble.py +++ b/pysits/conversions/tibble.py @@ -480,7 +480,8 @@ def geopandas_to_tibble(data: GeoPandasDataFrame) -> RDataFrame: f"Warning: Dropping columns with embedded DataFrames: {dropped_columns}" ) - data_safe = data[safe_columns].copy() + # Geometries are transferred as WKT, so the result is a plain data frame + data_safe = PandasDataFrame(data[safe_columns].copy()) if isinstance(data, GeoPandasDataFrame): geom_col = data.geometry.name diff --git a/pysits/docs/content/impute_linear.md b/pysits/docs/content/impute_linear.md index f98f927..4bff3e5 100644 --- a/pysits/docs/content/impute_linear.md +++ b/pysits/docs/content/impute_linear.md @@ -3,7 +3,8 @@ Replace NA values by linear interpolation Remove NA by linear interpolation Args: - data (list | pandas.DataFrame): A time series vector or matrix. + data (list): A time series vector or matrix. Returns: R: A set of filtered time series using the imputation function. + diff --git a/pysits/docs/content/impute_mean.md b/pysits/docs/content/impute_mean.md index c798bfe..30dda98 100644 --- a/pysits/docs/content/impute_mean.md +++ b/pysits/docs/content/impute_mean.md @@ -3,7 +3,7 @@ Remove NA using mean Remove NA using mean Args: - data (list[float] | pandas.DataFrame): A time series or matrix. + data (list[float] | SITSMatrix): A time series vector or matrix. Returns: R: A set of filtered time series using the imputation function. diff --git a/pysits/docs/content/impute_mean_window.md b/pysits/docs/content/impute_mean_window.md index f995737..a043bc2 100644 --- a/pysits/docs/content/impute_mean_window.md +++ b/pysits/docs/content/impute_mean_window.md @@ -3,14 +3,14 @@ Remove NA using weighted moving average Remove NA using weighted moving average Args: - data (list): A time series vector or matrix. + data (list[float]): A time series vector or matrix. k (int): Width of the moving average window. Expands to both sides of the center element e.g. k = 2 means 4 observations (2 left, 2 right) are taken into account. If all observations in the current window are NA, the window size is automatically increased until there are at least 2 non-NA values present. - weighting (str): The weighting strategy to be used. More details - below (default is "simple"). + weighting (str): Weighting strategy to be used. More details below + (default is "simple"). Returns: R: A set of filtered time series using the imputation function. diff --git a/pysits/docs/content/impute_median.md b/pysits/docs/content/impute_median.md index 062680a..0294334 100644 --- a/pysits/docs/content/impute_median.md +++ b/pysits/docs/content/impute_median.md @@ -3,7 +3,7 @@ Remove NA using median Remove NA using median Args: - data (list[float] | SITSMatrix): A time series vector or matrix. + data (list | SITSMatrix): A time series vector or matrix. Returns: R: A set of filtered time series using the imputation function. diff --git a/pysits/docs/content/plot.md b/pysits/docs/content/plot.md index 8bdb850..58d7f52 100644 --- a/pysits/docs/content/plot.md +++ b/pysits/docs/content/plot.md @@ -1,112 +1,117 @@ Plot sits objects. -Unified plotting function that dispatches on the type of the object passed -as `x`. It mirrors the many `plot` methods of the R `sits` package, -covering data cubes (raster, SAR, DEM, vector, RGB), probability and -uncertainty cubes, variance cubes, classified images, time series patterns -and predictions, machine learning / deep learning models, clustering and -self-organizing map (SOM) results, accuracy tables, and t-SNE / embedding -visualizations. The set of accepted keyword arguments depends on the type -of object being plotted. +A single dispatching function that produces a plot appropriate to the type +of the object passed as `x`. It covers data cubes (raster, SAR, DEM, +vector), probability and uncertainty products, variance cubes, patterns, +time-series predictions, embeddings, clustering and SOM outputs, accuracy +tables, and trained models. Depending on the object type, the plot is +rendered as a map, a chart, or a raster image. + +The accepted keyword arguments depend on the type of `x`. The sections +below group the parameters by the kind of object being plotted. Args: - x (SITSCubeModel | SITSTimeSeriesModel | SITSTimeSeriesPatternsModel | SITSMachineLearningMethod | SITSConfusionMatrix): Object to be - plotted. Supported objects include classified raster images, - classified segments, digital elevation model cubes, multi-year - land use/cover embedding predictions, sample distances, class - temporal patterns, probability cubes, raster, SAR, and vector - data cubes, confusion matrices / accuracy metrics, dendrograms, - trained models, time series predictions, t-SNE projections, SOM - results, uncertainty cubes, and variance cubes. + x (SITSCubeModel | SITSTimeSeriesModel | SITSTimeSeriesPatternsModel | SITSMachineLearningMethod | SITSConfusionMatrix): + Object to be plotted. Supported kinds include raster, SAR, DEM, + and vector cubes; classified, probability, uncertainty, and + variance cubes; patterns; time-series and embedding predictions; + geographic distances; clustering and SOM outputs; accuracy + tables; t-SNE projections; and trained models. y: Ignored. Present for compatibility with the generic `plot`. - band (str): Band used for plotting a single-band (grey scale) image. - Applies to raster, SAR, DEM, and vector cubes, and to SOM maps. - red (str): Band assigned to the red channel of an RGB composite - (raster, SAR, and vector cubes). - green (str): Band assigned to the green channel of an RGB composite. - blue (str): Band assigned to the blue channel of an RGB composite. - tile (str): Tile to be plotted (data cubes, probability, uncertainty, - and variance cubes). + band (str): For raster, SAR, DEM, and vector cubes, the band used to + plot a grey (B/W) image. For SOM maps, the band to be plotted. + red (str): Band assigned to the red channel for RGB plots of raster, + SAR, and vector cubes. + green (str): Band assigned to the green channel for RGB plots. + blue (str): Band assigned to the blue channel for RGB plots. + tile (str): Tile to be plotted (for cube objects). dates (list[str]): Dates to be plotted (raster, SAR, and vector cubes). - roi (dict | geopandas.GeoDataFrame): Spatial extent (region of - interest) to plot, in WGS 84. - labels (list[str]): Labels to plot (probability and variance cubes). - bands (list[str]): Bands to be viewed (patterns and time series - predictions). - legend (dict): Associates labels to colors, or a legend specification - for SOM plots. - legend_position (str): Where to place the legend (typically "inside" - or "outside", with defaults varying by plot type). - legend_title (str): Title of the legend (probability and variance + roi (dict): Spatial extent (region of interest) to plot, in WGS 84. + See notes. + labels (list[str]): Labels to plot (probability, variance, and vector cubes). - palette (str): An RColorBrewer or "cols4all" (or HCL) palette used - for color mapping. + bands (list[str]): Bands to be viewed (for patterns and time-series + predictions). + legend (dict): Maps labels to colors (class cubes, SOM maps, and + cluster confusion plots). + legend_position (str): Where to place the legend. Typical default is + `"inside"` for RGB/grey plots and `"outside"` for classified and + probability maps. + legend_title (str): Title of the legend for probability and variance + cubes (for example `"probs"` or `"logvar"`). + palette (str): An `RColorBrewer` or `cols4all` palette. For + chart-based plots (predictions, embeddings, clusters, t-SNE), an + HCL palette name. rev (bool): Whether to reverse the color order in the palette. - scale (float): Relative scale of plot text and map (typically 0.4 to - 1.0). + scale (float): Relative scale (roughly 0.4 to 1.0) of the plot text + and map. quantile (float): Minimum quantile to plot (probability and variance cubes). - first_quantile (float): First quantile for stretching images. - last_quantile (float): Last quantile for stretching images. + first_quantile (float): First quantile used for stretching images. + last_quantile (float): Last quantile used for stretching images. max_cog_size (int): Maximum size of COG (Cloud Optimized GeoTIFF) - overviews, in lines/columns or pixels. - seg_color (str): Color used to draw segment boundaries (vector cubes). - line_width (float): Line width used to draw segment boundaries - (vector cubes). - type (str): Type of plot; meaning depends on the object. For accuracy - objects it is "confusion_matrix" or "metrics"; for variance cubes - it is "map" or "hist"; for SOM maps it is "codes" or "mapping". - cluster: Cluster object produced by `sits_cluster_dendro`, used when - plotting a dendrogram. - cutree_height (float): Height at which to draw a dashed horizontal - line indicating where the dendrogram is cut. - name_cluster (str): Cluster to plot (SOM cluster evaluation). - title (str): Title of the plot (SOM cluster evaluation). - year_grid (bool): Whether to plot patterns as a grid of panels with - labels as columns and years as rows. Defaults to False. - tree_idx (int): Index of the tree to be plotted for an XGBoost model. - plot_embedding (str): For embedding predictions, either "none" (plot - only predicted class intervals) or "area" (overlay a smoothed - vertical embedding profile per year). - stretch (tuple[float, float]): For embedding plots, lower/upper - quantiles used to stretch embedding values before plotting. - class_alpha (float): Transparency of class polygons in embedding plots - (0-1). - area_alpha (float): Transparency of the embedding area in embedding - plots (0-1). - area_width (float): Horizontal width fraction of the embedding area. - area_spar (float): Smoothing parameter for the embedding area spline. - **kwargs (dict): Further specifications passed to the underlying plot. + overviews, in lines/columns (pixels). + seg_color (str): Color used for segment borders in vector cubes. + line_width (float): Line width used for segment borders in vector + cubes. + type (str): Type of plot. For accuracy objects, either + `"confusion_matrix"` or `"metrics"`. For variance cubes, `"map"` + or `"hist"`. For SOM maps, `"codes"` (neuron weight time series) + or `"mapping"` (number of samples per neuron). + year_grid (bool): For patterns, whether to plot a grid of panels + using labels as columns and years as rows (default `False`). + cluster: For clustering plots, the cluster object produced by + `sits_cluster_dendro`. + cutree_height (float): For clustering plots, the height at which to + draw a dashed horizontal line indicating where the dendrogram is + cut. + name_cluster (str): For SOM cluster evaluation, the cluster to plot. + title (str): For SOM cluster evaluation, the title of the plot. + tree_idx (int): For XGBoost models, the index of the tree to be + plotted. + plot_embedding (str): For embedding predictions, either `"none"` (plot + only the predicted class intervals) or `"area"` (overlay a + smoothed vertical embedding profile per year). + stretch (list[float]): For embedding predictions, the lower and upper + quantiles used to stretch embedding values before plotting + (default `[0.02, 0.98]`). + class_alpha (float): For embedding predictions, transparency of the + class polygons in `[0, 1]` (default `0.7`). + area_alpha (float): For embedding predictions, transparency of the + embedding area in `[0, 1]` (default `0.25`). + area_width (float): For embedding predictions, the horizontal width + fraction of the embedding area along the time axis. + area_spar (float): For embedding predictions, the smoothing parameter + passed to the spline fit (default `0.6`); higher values produce + smoother profiles. + **kwargs (dict): Further specifications for the plot. The keywords + understood depend on the type of `x` (see below). Returns: - None: A plot is produced. Depending on the input type this may be a - color map of classified pixels, an RGB or grey-scale image, a - probability or uncertainty map, a variance map (optionally with - segment overlays), a dendrogram, a confusion matrix, a SOM map, a - model diagnostic plot, or a plot for patterns, predictions, - embeddings, and t-SNE projections. Some methods are called only for - their side effect of drawing the plot. + None: A plot appropriate to the type of `x` is drawn. Maps of cubes + yield color or B/W raster images (optionally overlaid with segment + boundaries for vector cubes); probability, uncertainty, and + variance cubes yield per-class or per-pixel maps; classified cubes + yield color maps where each pixel is colored by its label. + Chart-based plots (patterns, predictions, embeddings, clusters, + t-SNE, model diagnostics) render the corresponding plot. Some + methods (accuracy tables, SOM diagnostics, model summaries) are + called only for their side effect of drawing the plot. Notes: - The `roi` argument can be defined as a `dict` giving the spatial - extent (for example with `lon_min`, `lon_max`, `lat_min`, `lat_max`), - a `geopandas.GeoDataFrame`, or another spatial specification accepted - by `sits`. Vector cube plots overlay the segments produced by - `sits_segment` on top of the raster image; their appearance is - controlled by `seg_color` and `line_width`. + The set of valid keyword arguments depends on the type of `x`; + passing arguments that do not apply to a given object type has no + effect. When a region of interest (`roi`) is supported, it defines + the spatial extent to plot in WGS 84. Examples: from pysits import * - # Plot a set of time series patterns - patterns = sits_patterns(cerrado_2classes) + # Plot a set of time-series patterns (one average pattern per label) + patterns = sits_patterns(samples_modis_ndvi) plot(patterns) - # Train a random forest model and plot variable importance - rfor_model = sits_train(samples_modis_ndvi, ml_method=sits_rfor()) - plot(rfor_model) - - # Plot a SOM map produced from a set of samples - som_map = sits_som_map(samples_modis_ndvi) - plot(som_map) + # Train a random forest model and plot its important variables + rf_model = sits_train(samples_modis_ndvi, ml_method=sits_rfor()) + plot(rf_model) diff --git a/pysits/docs/content/sits_accuracy.md b/pysits/docs/content/sits_accuracy.md index 38dc63e..f0767b0 100644 --- a/pysits/docs/content/sits_accuracy.md +++ b/pysits/docs/content/sits_accuracy.md @@ -1,24 +1,25 @@ Assess classification accuracy -This function calculates the accuracy of the classification result. The input -is either a set of classified time series or a classified data cube. Classified -time series are produced by `sits_classify`. Classified images are generated -using `sits_classify` followed by `sits_label_classification`. +This function calculates the accuracy of the classification result. The +input is either a set of classified time series or a classified data cube. +Classified time series are produced by `sits_classify`. Classified images +are generated using `sits_classify` followed by +`sits_label_classification`. For a set of time series, `sits_accuracy` creates a confusion matrix and -calculates the resulting statistics using package `caret`. For a classified -image, the function uses an area-weighted technique proposed by Olofsson et al. -according to references [1-3] to produce reliable accuracy estimates at 95% -confidence level. In both cases, it provides an accuracy assessment of the +calculates the resulting statistics. For a classified image, the function +uses an area-weighted technique proposed by Olofsson et al. according to +references [1-3] to produce reliable accuracy estimates at 95% confidence +level. In both cases, it provides an accuracy assessment of the classified, including Overall Accuracy, Kappa, User's Accuracy, Producer's Accuracy and error matrix (confusion matrix). Args: data (SITSCubeModel | SITSTimeSeriesModel): Either a data cube with classified images or a set of time series. - prediction_attr (str): Name of the column of the segments object that - contains the predicted values (only for vector class cubes). - reference_attr (str): Name of the column of the segments object that - contains the reference values (only for vector class cubes). + prediction_attr (str): Name of the column of the segments that contains + the predicted values (only for vector class cubes). + reference_attr (str): Name of the column of the segments that contains + the reference values (only for vector class cubes). validation (str | pathlib.Path | pandas.DataFrame | geopandas.GeoDataFrame | SITSTimeSeriesModel): Samples for validation (see below). Only required when data is a raster class cube. @@ -28,15 +29,15 @@ Args: Returns: SITSData: The error_matrix, the class_areas, the unbiased estimated - areas, the standard error areas, confidence interval 95 and the accuracy - (user, producer, and overall), or `None` if the data is empty. The result - can be visualized directly on the screen. + areas, the standard error areas, confidence interval 95 and the + accuracy (user, producer, and overall), or `None` if the data is + empty. The result can be visualized directly on the screen. Notes: - The `validation` data needs to contain the following columns: "latitude", - "longitude", "start_date", "end_date", and "label". It can be either a path - to a CSV file, a `SITSTimeSeriesModel`, a `pandas.DataFrame`, or a - `geopandas.GeoDataFrame`. + The `validation` data needs to contain the following columns: + "latitude", "longitude", "start_date", "end_date", and "label". It can + be either a path to a CSV file, a `SITSTimeSeriesModel`, a + `pandas.DataFrame`, or a `geopandas.GeoDataFrame`. When `validation` is a `geopandas.GeoDataFrame`, the columns "latitude" and "longitude" are not required as the locations are extracted from the geometry column. The `centroid` is calculated before extracting the @@ -44,6 +45,7 @@ Notes: Examples: from pysits import * + import tempfile # show accuracy for a set of samples train_data = sits_sample(samples_modis_ndvi, frac=0.5) @@ -65,7 +67,6 @@ Examples: data_dir=data_dir ) # classify a data cube - import tempfile probs_cube = sits_classify( data=cube, ml_model=rfor_model, output_dir=tempfile.gettempdir() ) diff --git a/pysits/docs/content/sits_accuracy_summary.md b/pysits/docs/content/sits_accuracy_summary.md index 877c47e..671e4ae 100644 --- a/pysits/docs/content/sits_accuracy_summary.md +++ b/pysits/docs/content/sits_accuracy_summary.md @@ -1,11 +1,11 @@ Print accuracy summary -Adaptation of the caret::print.confusionMatrix method for the more common -usage in Earth Observation. +Adaptation of the caret::print.confusionMatrix method for the more +common usage in Earth Observation. Args: - x (SITSConfusionMatrix): accuracy object to summarize. - digits (int): number of significant digits when printed. + x (SITSConfusionMatrix): Accuracy assessment object. + digits (int): Number of significant digits when printed. Returns: - SITSData: called for side effects. + SITSData: Called for side effects. diff --git a/pysits/docs/content/sits_add_base_cube.md b/pysits/docs/content/sits_add_base_cube.md index a3b3bd2..b534f59 100644 --- a/pysits/docs/content/sits_add_base_cube.md +++ b/pysits/docs/content/sits_add_base_cube.md @@ -8,7 +8,7 @@ sensor, resolution, bounding box, timeline, and have different bands. Args: cube1 (SITSCubeModel): Data cube. - cube2 (SITSCubeModel): Data cube with base information. + cube2 (SITSCubeModel): Base data cube (e.g., DEM). Returns: SITSCubeModel: a merged data cube with the inclusion of base diff --git a/pysits/docs/content/sits_apply.md b/pysits/docs/content/sits_apply.md index 473066d..25b3c85 100644 --- a/pysits/docs/content/sits_apply.md +++ b/pysits/docs/content/sits_apply.md @@ -1,25 +1,25 @@ Apply a function on a set of time series Apply a named expression to a set of time series or a data cube to be -evaluated and generate new bands (indices). In the case of data cubes, -it creates a new band in `output_dir`. +evaluated and generate new bands (indices). In the case of data cubes, it +creates a new band in `output_dir`. Args: - data (SITSTimeSeriesModel | SITSCubeModel): valid time series or data + data (SITSTimeSeriesModel | SITSCubeModel): Valid time series or data cube. - window_size (int): an odd number representing the size of the sliding + window_size (int): An odd number representing the size of the sliding window of kernel functions used in expressions (for a list of supported kernel functions, please see details). - memsize (int): memory available for classification (in GB). - multicores (int): number of cores to be used for classification. - normalized (bool): does the expression produce a normalized band? - output_dir (str | pathlib.Path): directory where files will be saved. - progress (bool): show progress bar? - **kwargs (dict): named expressions to be evaluated (see details). + memsize (int): Memory available for classification (in GB). + multicores (int): Number of cores to be used for classification. + normalized (bool): Does the expression produces a normalized band? + output_dir (str | pathlib.Path): Directory where files will be saved. + progress (bool): Show progress bar? + **kwargs (dict): Named expressions to be evaluated (see details). Returns: - SITSFrame: time series or data cube with new bands, produced according - to the requested expression. + SITSFrame: A set of time series or a data cube with new bands, produced + according to the requested expression. Notes: The main `sits` classification workflow has the following steps: @@ -39,20 +39,19 @@ Notes: to remove outliers and increase spatial consistency. 9. `sits_label_classification`: produce a classified map by selecting the label with the highest probability from a smoothed cube. - `sits_apply()` allows any valid R expression to compute new bands. Use R - syntax to pass an expression to this function. Besides arithmetic - operators, you can use virtually any R function that can be applied to - elements of a matrix (functions that are unaware of matrix sizes, e.g. - `sqrt()`, `sin()`, `log()`). + `sits_apply()` allows any valid expression to compute new bands. Besides + arithmetic operators, you can use virtually any function that can be + applied to elements of a matrix (functions that are unaware of matrix + sizes, e.g. `sqrt()`, `sin()`, `log()`). Examples of valid expressions: 1. `NDVI = (B08 - B04) / (B08 + B04)` for Sentinel-2 images. - 2. `EVI = 2.5 * (B05 \04) / (B05 + 6 * B04 \7.5 * B02 + 1)` for + 2. `EVI = 2.5 * (B05 ook04) / (B05 + 6 * B04 7.5 * B02 + 1)` for Landsat-8/9 images. 3. `VV_VH_RATIO = VH/VV` for Sentinel-1 images. In this case, set the - `normalized` parameter to `False`. + `normalized` parameter to False. 4. `VV_DB = 10 * log10(VV)` to convert Sentinel-1 RTC images available in Planetary Computer to decibels. Also, set the `normalized` parameter to - `False`. + False. `sits_apply()` also accepts a predefined set of kernel functions (see below) that can be applied to pixels considering its neighborhood. The function considers a neighborhood of a pixel as a set of pixels equidistant @@ -66,14 +65,13 @@ Notes: to a corresponding central pixel on a new matrix. The kernel slides throughout the input image and this process generates an entire new matrix, which is returned as a new band to the cube. The kernel functions ignores - any `NA` values inside the kernel window. If all pixels in the window are - `NA` the result will be `NA`. + any `None` values inside the kernel window. If all pixels in the window are + `None` the result will be `None`. By default, the indexes generated by `sits_apply()` function are normalized between -1 and 1, scaled by a factor of 0.0001. Normalized indexes are - saved as INT2S (Integer with sign). If the `normalized` parameter is - `False`, no scaling factor will be applied and the index will be saved as - FLT4S (signed float) and the values will vary between -3.4e+38 and - 3.4e+38. + saved as INT2S (Integer with sign). If the `normalized` parameter is False, + no scaling factor will be applied and the index will be saved as FLT4S + (signed float) and the values will vary between -3.4e+38 and 3.4e+38. Examples: from pysits import * diff --git a/pysits/docs/content/sits_as_geopandas.md b/pysits/docs/content/sits_as_geopandas.md index 15466e7..3f17e00 100644 --- a/pysits/docs/content/sits_as_geopandas.md +++ b/pysits/docs/content/sits_as_geopandas.md @@ -1,24 +1,24 @@ -Return time series or a data cube as a `geopandas.GeoDataFrame`. +Return a set of time series or a data cube as a `geopandas.GeoDataFrame`. -Converts time series or a data cube to a `geopandas.GeoDataFrame`. +Converts a set of time series or a data cube to a `geopandas.GeoDataFrame`. Args: - data (SITSTimeSeriesModel | SITSCubeModel): time series or data - cube. - crs (str): input coordinate reference system. - as_crs (str): output coordinate reference system. + data (SITSTimeSeriesModel | SITSCubeModel): a set of time series or + a data cube. + crs (CRS): input coordinate reference system. + as_crs (CRS): output coordinate reference system. **kwargs (dict): additional parameters. Returns: - SITSFrame: point or polygon geometry. + SITSFrame: a point or polygon geometry object. Examples: from pysits import * - # convert sits tibble to a geopandas object (point) + # convert sits tibble to a GeoPandas object (point) geo_object = sits_as_geopandas(cerrado_2classes) - # convert sits cube to a geopandas object (polygon) + # convert sits cube to a GeoPandas object (polygon) data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") cube = sits_cube( source="BDC", diff --git a/pysits/docs/content/sits_barlow_twins.md b/pysits/docs/content/sits_barlow_twins.md index dccb95a..e86486d 100644 --- a/pysits/docs/content/sits_barlow_twins.md +++ b/pysits/docs/content/sits_barlow_twins.md @@ -7,34 +7,32 @@ views' embeddings close to the identity: the diagonal -> 1 (invariance) and the off-diagonal -> 0 (redundancy reduction). No negatives are required. The function can be used in two ways: - If `samples` is provided, it trains immediately and returns an encoder-ready - model object (see Value). -- If `samples = None`, it returns a training function with signature + model object (see Returns). +- If `samples` is `None`, it returns a training function with signature `function(samples)` that can be passed to `sits_pre_train` or called later. Args: - samples (SITSTimeSeriesModel): A set of sample time series. If `None` - (default), returns a training function. If provided, triggers - immediate training. Base data samples (e.g., `sits_base`) are not - supported. + samples (SITSTimeSeriesModel): Sample time series. If `None` (default), + returns a training function. If provided, triggers immediate + training. Base data samples (e.g., `sits_base`) are not supported. embedding_dim (int): Dimensionality of the encoder embedding (exported features). Default: 64. proj_dim (int): Dimensionality of the projector head used only during pre-training. Default: 256. - bt_lambda (float): Weight of the redundancy-reduction (off-diagonal) - term in the Barlow Twins loss. Default: 5e-3. + bt_lambda (float): Weight of the redundancy-reduction (off-diagonal) term + in the Barlow Twins loss. Default: 5e-3. num_pairs (int | None): Total number of pairs to form per epoch. When - `None` (default), one pair is formed for every sample in the - training split. - encoder_model (SITSMachineLearningMethod): Deep learning method that - takes time series as input and produces latent representations that - are used to compute the loss function (suggested options: - `sits_tempcnn()`, `sits_lighttae()`, `sits_resnet()`). Default: - `sits_tempcnn()`. + `None` (default), one pair is formed for every sample in the training + split. + encoder_model (SITSMachineLearningMethod): Deep learning method that takes + time series as input and produces latent representations that are used + to compute the loss function (suggested options: `sits_tempcnn()`, + `sits_lighttae()`, `sits_resnet()`). Default: `sits_tempcnn()`. epochs (int): Maximum number of training epochs. batch_size (int): Batch size for training. Larger values improve the Barlow Twins cross-correlation estimate. Default: 128. - validation_split (float): Fraction of samples held out for validation - loss monitoring, in the range (0, 1). + validation_split (float): Fraction of samples held out for validation loss + monitoring, in (0, 1). optimizer: A `torch` optimizer constructor (default: `torch::optim_adamw`). opt_hparams (dict): Optimizer hyperparameters. Common entries: `lr`, @@ -44,11 +42,11 @@ Args: patience (int): Early-stopping patience (epochs without improvement). min_delta (float): Minimum improvement required to reset the patience counter. - verbose (bool): Print training progress? + verbose (bool): Whether to print training progress. seed (int): Random seed for reproducibility. Returns: - R: If `samples = None`, a training function with signature + R: If `samples` is `None`, a training function with signature `function(samples)` that trains a Barlow Twins model and returns a pretrained encoder. If `samples` is provided, the result of applying the training function to `samples` directly. diff --git a/pysits/docs/content/sits_bbox.md b/pysits/docs/content/sits_bbox.md index 9f24243..ade964f 100644 --- a/pysits/docs/content/sits_bbox.md +++ b/pysits/docs/content/sits_bbox.md @@ -5,18 +5,18 @@ projection coordinates in the case of cubes) Args: data (SITSTimeSeriesModel | SITSCubeModel): samples or data cube. - crs (str): CRS of the time series. - as_crs (str): CRS to project the resulting bounding box. + crs (CRS): CRS of the time series. + as_crs (CRS): CRS to project the resulting bounding box. **kwargs (dict): parameters for specific types. Returns: SITSFrame: the bounding box. Notes: - Time series in `sits` are associated with lat/long values in WGS84, - while each data cube is associated to a cartographic projection. To - obtain the bounding box of a data cube in a different projection than - the original, use the `as_crs` parameter. + Time series are associated with lat/long values in WGS84, while + each data cube is associated to a cartographic projection. To + obtain the bounding box of a data cube in a different projection + than the original, use the `as_crs` parameter. Examples: from pysits import * diff --git a/pysits/docs/content/sits_classify.md b/pysits/docs/content/sits_classify.md index 566cd8f..2c5ed56 100644 --- a/pysits/docs/content/sits_classify.md +++ b/pysits/docs/content/sits_classify.md @@ -1,107 +1,88 @@ -Classify a set of time series or a data cube. +Classify time series or data cubes using a trained machine learning model. -This function applies a machine learning model (trained by `sits_train`) -to classify time series or data cubes. Its behavior depends on the type -of the input `data`: +This function applies a model trained by `sits_train` to classify the input +data. Its behavior depends on the type of input provided: -- Set of time series (`SITSTimeSeriesModel`): the output is the same set - of time series with an additional `predicted` column containing the - assigned labels for each point. -- Regular raster cube (`SITSCubeModel`): the output is a probability cube - with the same tiles as the input. Each tile is a multiband image where - each band contains the probability that a pixel belongs to a given - class. -- Segmented (vector) data cube (produced by `sits_segment`): the temporal - model is applied to produce pixel-level probabilities, and the - associated vector support (`vector_info`) is preserved in the output. - The result is a probability cube with vector support, which can then be - passed to `sits_label_classification` for segment-based labeling. - Segment-level aggregation is no longer performed by `sits_classify`; - use `sits_label_classification` to aggregate pixel probabilities inside +- Set of time series (`SITSTimeSeriesModel`): the output is the same set of + time series with an additional `predicted` column containing the labels + assigned to each point. +- Regular raster data cube (`SITSCubeModel`): the output is a probability + cube with the same tiles as the input. Each tile contains a multiband + image in which each band holds the probability that a pixel belongs to a + given class. +- Segmented (vector) data cube (`SITSCubeModel`, produced by + `sits_segment`): the temporal model is applied to produce pixel-level + probabilities and the associated vector support (`vector_info`) is + preserved. The result is a data cube that can be passed to + `sits_label_classification` for segment-based labeling. Segment-level + aggregation is no longer performed by `sits_classify`; use + `sits_label_classification` to aggregate pixel probabilities inside segments and assign classes. Args: - data (SITSTimeSeriesModel | SITSCubeModel): input to classify. Either - a set of time series, a regular raster data cube, or a segmented - vector data cube. - ml_model (SITSMachineLearningMethod): model trained by `sits_train`. - roi (dict | str | pathlib.Path | geopandas.GeoDataFrame): region of - interest, either a `geopandas.GeoDataFrame`, a shapefile, or a - `dict` in WGS 84 with named XY values (`xmin`, `xmax`, `ymin`, - `ymax`) or named lat/long values (`lon_min`, `lat_min`, - `lon_max`, `lat_max`). Applies to raster and vector cubes. - exclusion_mask (geopandas.GeoDataFrame | str | pathlib.Path): areas - to be excluded from the classification process, defined by a - `geopandas.GeoDataFrame` or a shapefile. Applies to raster and - vector cubes. - impute_fn: imputation function to remove NA. - start_date (str): starting date for the classification (in YYYY-MM-DD - format). Applies to raster and vector cubes. - end_date (str): ending date for the classification (in YYYY-MM-DD - format). Applies to raster and vector cubes. - memsize (int): memory available for classification in GB (min = 1, - max = 16384). Applies to raster and vector cubes. - multicores (int): number of cores to be used for classification - (min = 1, max = 2048). - gpu_memory (int): memory available in GPU in GB (default = 4). - batch_size (int): batch size for GPU classification. - block_size (dict): size of the block read and written by each worker, - with `[nrows, ncols]`. Default is `None`, which computes an - optimal block size from `memsize`, `multicores` and the internal - block size of the raster files. Applies to raster and vector - cubes. - output_dir (str | pathlib.Path): directory for output file. Applies - to raster and vector cubes. - version (str): version of the output. Applies to raster and vector - cubes. - n_sam_pol (int): deprecated. Segment-level classification is no longer - performed by `sits_classify`. Use `sits_label_classification` for - segment-based labeling. Applies to vector cubes. - verbose (bool): print information about processing time? Applies to - raster and vector cubes. - progress (bool): show progress bar? - **kwargs (dict): other parameters for specific functions. + data (SITSTimeSeriesModel | SITSCubeModel): Input to classify. Either a + set of time series, a regular raster data cube, or a segmented + data cube. + ml_model (SITSMachineLearningMethod): Model trained by `sits_train`. + roi (geopandas.GeoDataFrame | str | pathlib.Path | dict): Region of + interest, used for raster and vector cubes. Either a + `geopandas.GeoDataFrame`, a shapefile, or a `dict` in WGS 84 + with named XY values ("xmin", "xmax", "ymin", "ymax") or named + lat/long values ("lon_min", "lat_min", "lon_max", "lat_max"). + exclusion_mask (geopandas.GeoDataFrame | str | pathlib.Path): Areas to + be excluded from the classification process, for raster and + vector cubes. Can be defined by a `geopandas.GeoDataFrame` or by + a shapefile. + impute_fn: Imputation function to remove NA values. + start_date (str): Starting date for the classification (YYYY-MM-DD + format), for raster and vector cubes. + end_date (str): Ending date for the classification (YYYY-MM-DD format), + for raster and vector cubes. + memsize (int): Memory available for classification in GB (min = 1, + max = 16384), for raster and vector cubes. + multicores (int): Number of cores to be used for classification + (min = 1, max = 2048). + gpu_memory (int): Memory available in GPU in GB (default = 4). + batch_size (int): Batch size for GPU classification. + block_size (dict): Size of the block read and written by each worker, + for raster and vector cubes. A `dict` with `nrows` and `ncols`. + Default is `None`, which computes an optimal block size from + `memsize`, `multicores` and the internal block size of the + raster files. + output_dir (str | pathlib.Path): Directory for the output file, for + raster and vector cubes. + version (str): Version of the output, for raster and vector cubes. + n_sam_pol (int): Deprecated (vector cubes only). Segment-level + classification is no longer performed by `sits_classify`; use + `sits_label_classification` for segment-based labeling. + verbose (bool): Whether to print information about processing time + (raster and vector cubes). + progress (bool): Whether to show a progress bar. + **kwargs (dict): Other parameters for specific functions. Returns: - SITSCubeModel: for a set of time series, a `SITSTimeSeriesModel` with - predicted labels for each point. For a regular raster cube, a data - cube with probabilities for each class. For a segmented vector - cube, a probability cube with associated vector support that - contains pixel-level probabilities and preserves `vector_info` for - segment-based labeling. + SITSCubeModel | SITSTimeSeriesModel: For a set of time series, a + `SITSTimeSeriesModel` with predicted labels for each point. For a + regular raster cube, a `SITSCubeModel` with probabilities for each + class. For a segmented cube, a probability cube with associated vector + support that contains pixel-level probabilities and preserves + `vector_info` for segment-based labeling. + +Notes: + This function collapses the R S3 methods `sits_classify.sits`, + `sits_classify.raster_cube`, and `sits_classify.vector_cube` into a + single Python entry point; the appropriate behavior is selected based on + the type of `data`. Examples: from pysits import * - # Example 1: classify a set of time series - # create a random forest model - rfor_model = sits_train(samples_modis_ndvi, sits_rfor()) + # Train a random forest model on a set of time series + rfor_model = sits_train(samples_modis_ndvi, ml_model=sits_rfor()) - # classify a point + # Classify a set of time series point_ndvi = sits_select(point_mt_6bands, bands=["NDVI"]) - point_class = sits_classify(data=point_ndvi, ml_model=rfor_model) - plot(point_class) - - # Example 2: classify a raster cube - # create a data cube from local files - data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") - cube = sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=data_dir - ) + point_class = sits_classify(point_ndvi, ml_model=rfor_model) - # classify a data cube - probs_cube = sits_classify( - data=cube, - ml_model=rfor_model, - output_dir="./tempdir" - ) - plot(probs_cube) - - # label the probability cube - label_cube = sits_label_classification( - probs_cube, - output_dir="./tempdir" - ) - plot(label_cube) + # Show the predicted labels + plot(point_class) diff --git a/pysits/docs/content/sits_cluster_dendro.md b/pysits/docs/content/sits_cluster_dendro.md index db7e4fe..3536930 100644 --- a/pysits/docs/content/sits_cluster_dendro.md +++ b/pysits/docs/content/sits_cluster_dendro.md @@ -2,35 +2,36 @@ Find clusters in time series samples These functions support hierarchical agglomerative clustering in sits. They provide support from creating a dendrogram and using it for cleaning samples. -`sits_cluster_dendro()` takes time series and produces a `SITSTimeSeriesModel` -with an added "cluster" column. The function first calculates a dendrogram and -obtains a validity index for best clustering using the adjusted Rand Index. -After cutting the dendrogram using the chosen validity index, it assigns a -cluster to each sample. +`sits_cluster_dendro()` takes a set of time series and produces the same +data with an added "cluster" column. The function first calculates a +dendrogram and obtains a validity index for best clustering using the adjusted +Rand Index. After cutting the dendrogram using the chosen validity index, it +assigns a cluster to each sample. `sits_cluster_frequency()` computes the contingency table between labels and clusters and produces a matrix. Its input is produced by `sits_cluster_dendro()`. -`sits_cluster_clean()` takes time series that have an additional `cluster` -produced by `sits_cluster_dendro()` and removes labels that are minority in -each cluster. +`sits_cluster_clean()` takes time series data that has an additional +`cluster` column produced by `sits_cluster_dendro()` and removes labels that +are minority in each cluster. Args: samples (SITSTimeSeriesModel): input set of time series. bands (list[str]): bands to be used in the clustering. - dist_method (str): one of the supported distances. "dtw": DTW with a + dist_method (str): one of the supported distances "dtw": DTW with a Sakoe-Chiba constraint. "dtw2": DTW with L2 norm and Sakoe-Chiba - constraint. "dtw_basic": A faster DTW with less functionality. "lbk": - Keogh's lower bound for DTW. "lbi": Lemire's lower bound for DTW. + constraint. "dtw_basic": A faster DTW with less functionality. + "lbk": Keogh's lower bound for DTW. "lbi": Lemire's lower bound for + DTW. linkage (str): agglomeration method to be used. One of "ward.D", - "ward.D2", "single", "complete", "average", "mcquitty", "median" or - "centroid". + "ward.D2", "single", "complete", "average", "mcquitty", "median" + or "centroid". k (int): desired number of clusters (overrides default value). palette (str): color palette as per `grDevices::hcl.pals()` function. **kwargs (dict): additional parameters to be passed to dtwclust::tsclust() function. Returns: - SITSTimeSeriesModel: time series with an added "cluster" column. + SITSTimeSeriesModel: time series with a "cluster" column. Notes: Please refer to the sits documentation available in diff --git a/pysits/docs/content/sits_cluster_frequency.md b/pysits/docs/content/sits_cluster_frequency.md index 68a17fc..e80f782 100644 --- a/pysits/docs/content/sits_cluster_frequency.md +++ b/pysits/docs/content/sits_cluster_frequency.md @@ -15,4 +15,4 @@ Examples: clusters = sits_cluster_dendro(cerrado_2classes) freq = sits_cluster_frequency(clusters) - print(freq) + freq diff --git a/pysits/docs/content/sits_colors_qgis.md b/pysits/docs/content/sits_colors_qgis.md index 68d1132..ccc2328 100644 --- a/pysits/docs/content/sits_colors_qgis.md +++ b/pysits/docs/content/sits_colors_qgis.md @@ -12,6 +12,8 @@ Returns: Examples: from pysits import * + import os + import tempfile data_dir = r_package_dir("extdata/raster/classif", package="sits") ro_class = sits_cube( @@ -30,5 +32,5 @@ Examples: "4": "Forest" } ) - qml_file = "/tmp/qgis.qml" + qml_file = os.path.join(tempfile.gettempdir(), "qgis.qml") sits_colors_qgis(ro_class, qml_file) diff --git a/pysits/docs/content/sits_colors_show.md b/pysits/docs/content/sits_colors_show.md index 7e0a298..ff23720 100644 --- a/pysits/docs/content/sits_colors_show.md +++ b/pysits/docs/content/sits_colors_show.md @@ -10,5 +10,7 @@ Returns: None: called for side effects. Examples: + from pysits import * + # show the colors supported by SITS sits_colors_show() diff --git a/pysits/docs/content/sits_combine_predictions.md b/pysits/docs/content/sits_combine_predictions.md index 308acdf..700472c 100644 --- a/pysits/docs/content/sits_combine_predictions.md +++ b/pysits/docs/content/sits_combine_predictions.md @@ -38,10 +38,11 @@ Notes: Examples: from pysits import * + import os import tempfile # create a data cube from local files - data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") + data_dir = os.path.join(r_package_dir("extdata/raster/mod13q1", package="sits")) cube = sits_cube( source="BDC", collection="MOD13Q1-6.1", diff --git a/pysits/docs/content/sits_confidence_sampling.md b/pysits/docs/content/sits_confidence_sampling.md index e8c73c6..0687321 100644 --- a/pysits/docs/content/sits_confidence_sampling.md +++ b/pysits/docs/content/sits_confidence_sampling.md @@ -35,6 +35,7 @@ Returns: Examples: from pysits import * + import tempfile # create a data cube data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") @@ -47,7 +48,7 @@ Examples: rfor_model = sits_train(samples_modis_ndvi, ml_method=sits_rfor()) # classify the cube probs_cube = sits_classify( - data=cube, ml_model=rfor_model, output_dir=tempdir() + data=cube, ml_model=rfor_model, output_dir=tempfile.gettempdir() ) # obtain a new set of samples for active learning # the samples are located in uncertain places diff --git a/pysits/docs/content/sits_config.md b/pysits/docs/content/sits_config.md index 0ddcd82..00c04cd 100644 --- a/pysits/docs/content/sits_config.md +++ b/pysits/docs/content/sits_config.md @@ -1,28 +1,29 @@ Configure parameters for sits package These functions load and show sits configurations. -The `sits` package uses a configuration file that contains information -on parameters required by different functions. This includes -information about the image collections handled by `sits`. -`sits_config()` loads the default configuration file and the user -provided configuration file. The final configuration is obtained by -overriding the options by the values provided by the user. +The `sits` package uses a configuration file that contains information on +parameters required by different functions. This includes information about the +image collections handled by `sits`. +`sits_config()` loads the default configuration file and the user provided +configuration file. The final configuration is obtained by overriding the +options by the values provided by the user. -Users can provide additional configuration files, by specifying the -location of their file in the environmental variable -`SITS_CONFIG_USER_FILE` or as parameter to this function. -To see the key entries and contents of the current configuration -values, use `sits_config_show()`. +Users can provide additional configuration files, by specifying the location of +their file in the environmental variable `SITS_CONFIG_USER_FILE` or as +parameter to this function. +To see the key entries and contents of the current configuration values, use +`sits_config_show()`. Args: - config_user_file (str | pathlib.Path): YAML user configuration file - (a file with a "yml" extension). + config_user_file (str | pathlib.Path): YAML user configuration file (a + file with a "yml" extension). Returns: SITStructureData: Called for side effects. Examples: from pysits import * + import os - yaml_user_file = r_package_dir("extdata/config_user_example.yml", package="sits") + yaml_user_file = os.path.join(r_package_dir("extdata/config_user_example.yml", package="sits")) sits_config(config_user_file=yaml_user_file) diff --git a/pysits/docs/content/sits_config_user_file.md b/pysits/docs/content/sits_config_user_file.md index 6708937..e928038 100644 --- a/pysits/docs/content/sits_config_user_file.md +++ b/pysits/docs/content/sits_config_user_file.md @@ -13,7 +13,6 @@ Returns: Examples: from pysits import * import tempfile - import os - user_file = os.path.join(tempfile.gettempdir(), "my_config_file.yml") + user_file = tempfile.gettempdir() + "/my_config_file.yml" sits_config_user_file(user_file) diff --git a/pysits/docs/content/sits_contrastive_learning.md b/pysits/docs/content/sits_contrastive_learning.md index 6eb1b5b..2ae784e 100644 --- a/pysits/docs/content/sits_contrastive_learning.md +++ b/pysits/docs/content/sits_contrastive_learning.md @@ -14,7 +14,7 @@ After pre-training, the projection head is discarded and only the encoder is kept for downstream use via `sits_encode`. The function can be used in two ways: - If `samples` is provided, it trains immediately and returns an encoder-ready - model object (see Value). + model object (see Returns). - If `samples = None`, it returns a training function with signature `function(samples)` that can be passed to `sits_pre_train` or called later. @@ -33,9 +33,10 @@ et al. (2020). Larger batches also help, since they expose more positives and negatives per anchor and yield a stronger contrastive signal. Args: - samples (SITSTimeSeriesModel): Samples object. If `None` (default), - returns a training function. If provided, triggers immediate - training. Base data samples are not supported. + samples (SITSTimeSeriesModel | None): Sample time series. If `None` + (default), returns a training function. If provided, triggers + immediate training. Base data samples (e.g., `sits_base`) are not + supported. embedding_dim (int): Dimensionality of the encoder embedding (exported features). Default: 64. proj_dim (int): Dimensionality of the projection head output used only @@ -44,16 +45,15 @@ Args: the similarity distribution. Default: 0.07. num_pairs (int | None): Total number of pairs to form. When `None` (default), one pair is formed per sample. - encoder_model (SITSMachineLearningMethod): Deep learning method that - takes time series as input and produces latent representations that - are used to compute the loss function (suggested options: - `sits_tempcnn()`, `sits_lighttae()`, `sits_resnet()`). Default: - `sits_tempcnn()`. + encoder_model (SITSMachineLearningMethod): Deep learning method that takes + time series as input and produces latent representations that are used + to compute the loss function (suggested options: `sits_tempcnn()`, + `sits_lighttae()`, `sits_resnet()`). Default: `sits_tempcnn()`. epochs (int): Maximum number of training epochs. batch_size (int): Batch size for training. Larger batches provide more positives/negatives per sample. Default: 128. - validation_split (float): Fraction of samples held out for validation - loss monitoring, in the range (0, 1). + validation_split (float): Fraction of samples in (0, 1) held out for + validation loss monitoring. optimizer: A `torch` optimizer constructor (default: `torch::optim_adamw`). opt_hparams (dict): Optimizer hyperparameters. Common entries: `lr`, diff --git a/pysits/docs/content/sits_cube.md b/pysits/docs/content/sits_cube.md index dd88222..5a604f2 100644 --- a/pysits/docs/content/sits_cube.md +++ b/pysits/docs/content/sits_cube.md @@ -1,78 +1,71 @@ -Create sits data cubes from local files or cloud-based image collections. +Create data cubes from image collections or local files. -Builds a data cube -- a `pandas.DataFrame` describing spatial and temporal -image data -- from one of several sources. This single function handles four -distinct cases, selected by the combination of arguments you provide: +This function creates data cubes from a variety of sources. Depending on the +parameters provided, it dispatches to one of several behaviors: -- Local raster cube: images already downloaded from a known cloud collection - or created by `sits`, read from a local directory (`data_dir`). -- STAC cube: image collections accessible through the STAC protocol, selected - by spatial (`roi`, `tiles`) and temporal (`start_date`, `end_date`) - restrictions. -- Results cube: local files produced by `sits` operations that generate - results (for example probability cubes and class cubes). -- Vector cube: local files that include a vector file produced by a - segmentation algorithm, merged with a raster cube. +- STAC cube: create a data cube based on spatial and temporal restrictions + in collections accessible by the STAC protocol (pass `roi`/`tiles`, + `start_date`/`end_date`, etc., without `data_dir`). +- Local cube: create a data cube from files stored on a local directory, + assuming users have downloaded the data from a known cloud collection or + the data has been created by `sits` (pass `data_dir`). +- Results cube: create a data cube from local files produced by `sits` + operations that generate results (such as probability cubes and class + cubes) by passing results `bands` such as `"probs"`, `"bayes"`, + `"variance"`, `"class"`, or `"uncertainty"`. +- Vector cube: create a data cube from local files which include a vector + file produced by a segmentation algorithm (pass `raster_cube` and + `vector_dir`). Args: source (str): Data source: one of `"AWS"`, `"BDC"`, `"CDSE"`, - `"DEAFRICA"`, `"DEAUSTRALIA"`, `"HLS"`, `"PLANETSCOPE"`, - `"MPC"`, `"SDC"` or `"USGS"`. For local, results and vector - cubes this is the source from which the original data was - downloaded. - collection (str): Image collection in the data source. Use - `sits_list_collections()` to find supported collections. - bands (list[str]): Spectral bands and indices to include in the cube - (optional). For a results cube these are results bands to be - retrieved (`"probs"`, `"bayes"`, `"variance"`, `"class"`, - `"uncertainty"`). Use `sits_list_collections()` to find the - bands available for each STAC collection. - tiles (list[str]): Tiles from the collection to include in the cube. - roi (dict): Region of interest for STAC cubes. - crs (str): The Coordinate Reference System (CRS) of the `roi`. + `"DEAFRICA"`, `"DEAUSTRALIA"`, `"HLS"`, `"PLANETSCOPE"`, `"MPC"`, + `"SDC"` or `"USGS"`. For local, results, and vector cubes, this is + the source from which the original data was downloaded. + collection (str): Image collection in the data source. To find out the + supported collections, use `sits_list_collections()`. + bands (list[str]): Spectral bands and indices to be included in the cube + (optional). For results cubes, the results bands to be retrieved + (`"probs"`, `"bayes"`, `"variance"`, `"class"`, `"uncertainty"`). + tiles (list[str]): Tiles from the collection to be included in the cube. + roi (dict | geopandas.GeoDataFrame): Region of interest (for STAC + cubes). + crs (str | int): The Coordinate Reference System (CRS) of the `roi`. start_date (str): Initial date to include images from the collection in - the cube, in `YYYY-MM-DD` format (optional). + the cube (optional), in YYYY-MM-DD format. end_date (str): Final date to include images from the collection in the - cube, in `YYYY-MM-DD` format (optional). + cube (optional), in YYYY-MM-DD format. orbit (str): Orbit name (`"ascending"`, `"descending"`) for SAR cubes. - platform (str): Optional parameter specifying the platform for the - `"LANDSAT"` collection. Options: `Landsat-5`, `Landsat-7`, - `Landsat-8`, `Landsat-9`. + platform (str): Optional parameter specifying the platform in case of + the `"LANDSAT"` collection. Options: `Landsat-5`, `Landsat-7`, + `Landsat-8`, `Landsat-9`. data_dir (str | pathlib.Path): Local directory where images are stored - (local, results and vector cubes). - labels (dict): Labels associated to the classes (results cube). + (for local and results cubes). raster_cube (SITSCubeModel): Raster cube to be merged with vector data - (vector cube). + (for vector cubes). vector_dir (str | pathlib.Path): Local directory where vector files are - stored (vector cube). - vector_band (str): Band for the vector cube (`"segments"`, `"probs"`, - `"class"`). Deprecated and will be removed in future versions; - the vector data cube type is now defined from the `raster_cube` - object. + stored (for vector cubes). + vector_band (str): Band for vector cube (`"segments"`, `"probs"`, + `"class"`). This parameter is deprecated and will be removed in + future versions; the type of vector data cube loaded is now defined + based on the `raster_cube` object. + labels (dict): Labels associated with the classes (for results cubes). parse_info (list[str]): Parsing information for local files. - version (str): Version of the classified and/or labelled files (results - and vector cubes). + version (str): Version of the classified and/or labelled files. delim (str): Delimiter for parsing local files (default `"_"`). - multicores (int): Number of workers for parallel processing - (min = 1, max = 2048). - memsize (int): Memory available in GB (results cube). + multicores (int): Number of workers for parallel processing (min = 1, + max = 2048). + memsize (int): Memory available (in GB) (for results cubes). progress (bool): Whether to show a progress bar. - **kwargs (dict): Other parameters passed to specific cube types. + **kwargs (dict): Other parameters to be passed for specific cube types. Returns: - SITSCubeModel: A data cube describing the contents of the images. - -Notes: - The specific cube type is inferred from the combination of arguments. - Local raster cubes use `data_dir`; STAC cubes use spatial and temporal - restrictions such as `roi`, `tiles`, `start_date` and `end_date`; - results cubes use `data_dir` together with results `bands` and - `labels`; vector cubes use `raster_cube` together with `vector_dir`. + SITSCubeModel: A description of the contents of a data cube. Examples: from pysits import * - # Create a local data cube from MODIS files bundled with sits + # Create a data cube from a local directory of files data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") cube = sits_cube( source="BDC", @@ -80,18 +73,8 @@ Examples: data_dir=data_dir ) - # Inspect the cube + # Show the bands available in the cube print(sits_bands(cube)) - print(sits_timeline(cube)) - plot(cube) - # Create a STAC-based data cube restricted by tiles and dates - s2_cube = sits_cube( - source="MPC", - collection="SENTINEL-2-L2A", - tiles="20LKP", - bands=["B05", "CLOUD"], - start_date="2018-07-18", - end_date="2018-08-23" - ) - print(sits_bands(s2_cube)) + # Plot the cube + plot(cube) diff --git a/pysits/docs/content/sits_cube_copy.md b/pysits/docs/content/sits_cube_copy.md index 8e94425..677dcdf 100644 --- a/pysits/docs/content/sits_cube_copy.md +++ b/pysits/docs/content/sits_cube_copy.md @@ -13,49 +13,52 @@ Args: `"lon_max"`, `"lat_max"`) in WGS84; 4. A `dict` (`"xmin"`, `"xmax"`, `"ymin"`, `"ymax"`) with XY coordinates in the projection of the input cube. - res (int): Output spatial resolution of the images. Default is None. + res (int): Output spatial resolution of the images. Default is + `None`. crs (str): The Coordinate Reference System (CRS) of the roi. (see details below). - n_tries (int): Number of attempts to download the same image. Default - is 3. + n_tries (int): Number of attempts to download the same image. + Default is 3. multicores (int): Number of cores for parallel downloading (min = 1, max = 2048). - output_dir (str | pathlib.Path): Output directory where images will be - saved. + output_dir (str | pathlib.Path): Output directory where images will + be saved. progress (bool): Show progress bar? - **kwargs (dict): Additional arguments. + **kwargs (dict): Additional parameters. Returns: SITSCubeModel: Copy of input data cube. The main `sits` classification workflow has the following steps: - 1. `sits_cube`: selects a ARD image collection from a cloud provider. + 1. `sits_cube`: selects a ARD image collection from a cloud + provider. 2. `sits_cube_copy`: copies an ARD image collection from a cloud provider to a local directory for faster processing. 3. `sits_regularize`: create a regular data cube from an ARD image collection. 4. `sits_apply`: create new indices by combining bands of a regular data cube (optional). - 5. `sits_get_data`: extract time series from a regular data cube based - on user-provided labelled samples. + 5. `sits_get_data`: extract time series from a regular data cube + based on user-provided labelled samples. 6. `sits_train`: train a machine learning model based on image time series. 7. `sits_classify`: classify a data cube using a machine learning model and obtain a probability cube. 8. `sits_smooth`: post-process a probability cube using a spatial smoother to remove outliers and increase spatial consistency. - 9. `sits_label_classification`: produce a classified map by selecting - the label with the highest probability from a smoothed cube. + 9. `sits_label_classification`: produce a classified map by + selecting the label with the highest probability from a smoothed + cube. The `roi` parameter is used to crop cube images. To define a `roi` use one of: - A path to a shapefile with polygons; - A `geopandas.GeoDataFrame`; - - A spatial extent object; + - A `SpatExtent` object; - A `dict` (`"lon_min"`, `"lat_min"`, `"lon_max"`, `"lat_max"`) in WGS84; - A `dict` (`"xmin"`, `"xmax"`, `"ymin"`, `"ymax"`) with XY coordinates. - Defining a region of interest using a spatial extent or XY values not - in WGS84 requires the `crs` parameter to be specified. + Defining a region of interest using `SpatExtent` or XY values not in + WGS84 requires the `crs` parameter to be specified. Examples: from pysits import * diff --git a/pysits/docs/content/sits_encode.md b/pysits/docs/content/sits_encode.md index 6ab1c6c..c9bc50b 100644 --- a/pysits/docs/content/sits_encode.md +++ b/pysits/docs/content/sits_encode.md @@ -1,17 +1,15 @@ Encode data using a pre-trained deep learning encoder. -Encodes either a regular raster data cube or a set of time series using a -pre-trained deep learning encoder returned by `sits_pre_train`. The -behavior depends on the type of `data` provided. +Encodes data using a pre-trained deep learning encoder returned by +`sits_pre_train`. The behavior depends on the type of `data` provided: -When `data` is a raster data cube, the output is an embeddings cube with -the same tiling as the input cube. Each tile is written as a multiband -raster, where each band corresponds to one embedding dimension produced -by the encoder. - -When `data` is a set of time series, the output preserves the input -structure and replaces the `time_series` column with the corresponding -embeddings. +- When `data` is a regular raster data cube, the output is an embeddings + cube with the same tiling as the input cube. Each tile is written as a + multiband raster, where each band corresponds to one embedding dimension + produced by the encoder. +- When `data` is a set of time series, the output preserves the input + structure and replaces the `time_series` column with the corresponding + embeddings. Args: data (SITSCubeModel | SITSTimeSeriesModel): Data to be encoded. Either @@ -21,10 +19,10 @@ Args: roi (str | pathlib.Path | geopandas.GeoDataFrame | dict): Optional region of interest used to restrict processing (raster cube only). It may be provided as: (1) a path to a polygon shapefile; (2) a - `geopandas.GeoDataFrame` with `POLYGON` or `MULTIPOLYGON` - geometry; (3) a named bounding box in WGS84 with `xmin`, `xmax`, - `ymin`, `ymax`; or (4) a named lon/lat bounding box with - `lon_min`, `lon_max`, `lat_min`, `lat_max`. + `geopandas.GeoDataFrame` with `POLYGON` or `MULTIPOLYGON` geometry; + (3) a bounding box `dict` in WGS84 with `xmin`, `xmax`, `ymin`, + `ymax`; or (4) a lon/lat bounding box `dict` with `lon_min`, + `lon_max`, `lat_min`, `lat_max`. impute_fn: Imputation function used to interpolate missing values in each pixel time series (default: `impute_linear`). start_date (str): Optional start date for temporal filtering @@ -34,14 +32,13 @@ Args: memsize (int): Memory available for processing in GB (minimum 1), raster cube only. multicores (int): Number of CPU cores used for processing (minimum 1; - maximum 2048 for time series). + for time series, maximum 2048). gpu_memory (int): GPU memory available for encoding in GB (minimum 1). batch_size (int): Batch size used when encoding on GPU. block_size (dict): Size of the block read and written by each worker - (raster cube only). A named vector with `nrows` and `ncols`. - Default is `None`, which computes an optimal block size from - `memsize`, `multicores` and the internal block size of the raster - files. + (raster cube only). A `dict` with `(nrows, ncols)`. Default is + `None`, which computes an optimal block size from `memsize`, + `multicores` and the internal block size of the raster files. output_dir (str | pathlib.Path): Directory where output files will be written (raster cube only). verbose (bool): If `True`, print processing time information (raster @@ -51,7 +48,7 @@ Args: routines. Returns: - SITSTimeSeriesModel: For a raster cube, an embeddings cube written to - `output_dir`, with one multiband raster per tile date. For a set of - time series, a `SITSTimeSeriesModel` with `time_series` containing the - embeddings produced by `encoder`. + SITSTimeSeriesModel | SITSCubeModel: For a raster cube, an embeddings + cube written to `output_dir`, with one multiband raster per tile date. + For a set of time series, a `SITSTimeSeriesModel` with `time_series` + containing the embeddings produced by the encoder. diff --git a/pysits/docs/content/sits_formula_logref.md b/pysits/docs/content/sits_formula_logref.md index e2dba76..b677360 100644 --- a/pysits/docs/content/sits_formula_logref.md +++ b/pysits/docs/content/sits_formula_logref.md @@ -1,11 +1,11 @@ Define a loglinear formula for classification models -A function to be used as a symbolic description of some fitting models such -as svm and random forest. This function tells the models to do a log +A function to be used as a symbolic description of some fitting models such as +svm and random forest. This function tells the models to do a log transformation of the inputs. The `predictors_index` parameter informs the -positions of `tb` fields corresponding to formula independent variables. If -no value is given, the default is `None`, a value indicating that all fields -will be used as predictors. +positions of `tb` fields corresponding to formula independent variables. If no +value is given, the default is `None`, a value indicating that all fields will +be used as predictors. Args: predictors_index (list[int]): Index of the valid columns to compose diff --git a/pysits/docs/content/sits_geo_dist.md b/pysits/docs/content/sits_geo_dist.md index 7d5a88c..cd0c813 100644 --- a/pysits/docs/content/sits_geo_dist.md +++ b/pysits/docs/content/sits_geo_dist.md @@ -12,7 +12,8 @@ Args: crs (str): CRS of the `samples`. Returns: - SITSFrame: sample-to-sample and sample-to-prediction distances. + SITSFrame: A table with sample-to-sample and sample-to-prediction + distances. Notes: As pointed out by Meyer and Pebesma, many classifications using machine @@ -31,7 +32,7 @@ Examples: # read a shapefile for the state of Mato Grosso, Brazil mt_shp = r_package_dir("extdata/shapefiles/mato_grosso/mt.shp", package="sits") - # convert to a geopandas object + # convert to a GeoDataFrame mt_sf = gpd.read_file(mt_shp) # calculate sample-to-sample and sample-to-prediction distances distances = sits_geo_dist( diff --git a/pysits/docs/content/sits_get_class.md b/pysits/docs/content/sits_get_class.md index fd50d13..052ac58 100644 --- a/pysits/docs/content/sits_get_class.md +++ b/pysits/docs/content/sits_get_class.md @@ -12,7 +12,7 @@ Args: `pandas.DataFrame` with columns "longitude" and "latitude". Returns: - SITSFrame: With columns . Notes: @@ -24,7 +24,6 @@ Notes: Examples: from pysits import * - import tempfile # create a random forest model rfor_model = sits_train(samples_modis_ndvi, sits_rfor()) @@ -37,18 +36,18 @@ Examples: ) # classify a data cube probs_cube = sits_classify( - data=cube, ml_model=rfor_model, output_dir=tempfile.gettempdir() + data=cube, ml_model=rfor_model, output_dir=tempdir() ) # plot the probability cube plot(probs_cube) # smooth the probability cube using Bayesian statistics - bayes_cube = sits_smooth(probs_cube, output_dir=tempfile.gettempdir()) + bayes_cube = sits_smooth(probs_cube, output_dir=tempdir()) # plot the smoothed cube plot(bayes_cube) # label the probability cube label_cube = sits_label_classification( bayes_cube, - output_dir=tempfile.gettempdir() + output_dir=tempdir() ) # obtain the a set of points for sampling ground_truth = r_package_dir("extdata/samples/samples_sinop_crop.csv", package="sits") diff --git a/pysits/docs/content/sits_get_data.md b/pysits/docs/content/sits_get_data.md index 7ce84b8..db3f6b6 100644 --- a/pysits/docs/content/sits_get_data.md +++ b/pysits/docs/content/sits_get_data.md @@ -2,79 +2,56 @@ Get time series from a data cube. Retrieve a set of time series from a data cube and put the result in a `SITSTimeSeriesModel`, which contains both the satellite image time -series and their metadata. The samples to be retrieved can be provided -in several forms, and the accepted parameters vary slightly depending -on the type of `samples`. +series and their metadata. The type of retrieval depends on what is +passed as the `samples` argument, which may be a CSV file, a shapefile, +a `geopandas.GeoDataFrame`, a `pandas.DataFrame`, or an existing +`SITSTimeSeriesModel`. -The `samples` parameter may be one of the following: +The `samples` argument can be one of the following: -- A path to a CSV file (extension ".csv") with mandatory columns - `longitude`, `latitude`, `label`, `start_date` and `end_date`. +- A CSV file (extension ".csv") with mandatory columns `longitude`, + `latitude`, `label`, `start_date` and `end_date`. - A `pandas.DataFrame` with mandatory columns `longitude` and `latitude`, and optional columns `start_date`, `end_date` and `label`. - A `geopandas.GeoDataFrame` in POINT or POLYGON geometry. -- A path to a shapefile (extension ".shp") that is a valid POINT or - POLYGON shapefile. +- A shapefile (extension ".shp") which should be a valid shapefile in + POINT or POLYGON geometry. - A valid `SITSTimeSeriesModel` with columns `longitude`, `latitude`, `start_date`, `end_date` and `label`. -For spatial inputs (`geopandas.GeoDataFrame` objects and shapefiles), -if `start_date` and `end_date` are not informed, the function uses -these dates from the cube. +When samples are provided via `geopandas.GeoDataFrame` objects or +shapefiles and `start_date` and `end_date` are not informed, the +function uses these dates from the cube. Args: cube (SITSCubeModel): Data cube from where data is to be retrieved. - samples (SITSTimeSeriesModel | geopandas.GeoDataFrame | pandas.DataFrame | str | pathlib.Path): Location of the samples - to be retrieved. Either a `SITSTimeSeriesModel`, a - `geopandas.GeoDataFrame`, the name of a shapefile or csv file, - or a `pandas.DataFrame` with columns "longitude" and - "latitude". - bands (list[str]): Bands to be retrieved - optional. + samples (SITSTimeSeriesModel | geopandas.GeoDataFrame | pandas.DataFrame | str | pathlib.Path): + Location of the samples to be retrieved. Either a + `SITSTimeSeriesModel`, a `geopandas.GeoDataFrame`, the name of a + shapefile or csv file, or a `pandas.DataFrame` with columns + `longitude` and `latitude`. start_date (str): Start of the interval for the time series - - optional (date in "YYYY-MM-DD" format). Applies to - `pandas.DataFrame`, `geopandas.GeoDataFrame` and shapefile - inputs. + optional (date in "YYYY-MM-DD" format). end_date (str): End of the interval for the time series - optional - (date in "YYYY-MM-DD" format). Applies to `pandas.DataFrame`, - `geopandas.GeoDataFrame` and shapefile inputs. + (date in "YYYY-MM-DD" format). + bands (list[str]): Bands to be retrieved - optional. + impute_fn: Imputation function to remove NA. label (str): Label to be assigned to all time series if a `label` - column is not provided in the input. + column is not provided in the samples - optional. label_attr (str): Attribute in the `geopandas.GeoDataFrame` or shapefile to be used as a polygon label. - n_sam_pol (int): Number of samples per polygon to be read for - POLYGON or MULTIPOLYGON inputs. + n_sam_pol (int): Number of samples per polygon to be read for POLYGON + or MULTIPOLYGON objects. pol_avg (bool): Summarize samples for each polygon? sampling_type (str): Spatial sampling type: random, hexagonal, regular, or Fibonacci. crs (str): The samples CRS. Default is "EPSG:4326". - impute_fn: Imputation function to remove NA. multicores (int): Number of threads to process the time series - (with min = 1 and max = 2048). + (min = 1 and max = 2048). progress (bool): Show progress bar? **kwargs (dict): Specific parameters for each kind of input. Returns: - SITSTimeSeriesModel: The set of time series and metadata: - . - -Examples: - from pysits import * - import pandas as pd - - # reading a lat/long from a local cube - data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") - raster_cube = sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=data_dir - ) - - # obtain a set of samples defined by a lat/long point - samples = pd.DataFrame({ - "longitude": [-55.66738], - "latitude": [-11.76990] - }) - points = sits_get_data(cube=raster_cube, samples=samples) - print(points) + SITSTimeSeriesModel: The set of time series and metadata with columns + . diff --git a/pysits/docs/content/sits_get_probs.md b/pysits/docs/content/sits_get_probs.md index c7ab5a9..891a37d 100644 --- a/pysits/docs/content/sits_get_probs.md +++ b/pysits/docs/content/sits_get_probs.md @@ -10,13 +10,13 @@ Args: Location of the samples to be retrieved. Either a `SITSTimeSeriesModel`, a `geopandas.GeoDataFrame` with POINT geometry, the location of a POINT shapefile, the location of a - CSV file with columns `longitude` and `latitude`, or a - `pandas.DataFrame` with columns `longitude` and `latitude`. + CSV file with columns "longitude" and "latitude", or a + `pandas.DataFrame` with columns "longitude" and "latitude". window_size (int): Size of window around pixel (optional). **kwargs (dict): Additional arguments. Returns: - SITSFrameNested: A table with columns in + SITSFrameNested: table with columns in case no windows are requested and in case windows are requested. @@ -27,7 +27,7 @@ Notes: - SHP: a shapefile in POINT geometry. - `geopandas.GeoDataFrame`: an object with POINT geometry. - `SITSTimeSeriesModel`: a valid set of time series. - - `pandas.DataFrame`: a data frame with `longitude` and `latitude`. + - `pandas.DataFrame`: with `longitude` and `latitude`. Examples: from pysits import * @@ -46,8 +46,6 @@ Examples: data=cube, ml_model=rfor_model, output_dir=tempdir() ) # obtain the a set of points for sampling - ground_truth = r_package_dir("extdata/samples/samples_sinop_crop.csv", - package="sits" - ) + ground_truth = r_package_dir("extdata/samples/samples_sinop_crop.csv", package="sits") # get the classification values for a selected set of locations probs_samples = sits_get_probs(probs_cube, ground_truth) diff --git a/pysits/docs/content/sits_kfold_validate.md b/pysits/docs/content/sits_kfold_validate.md index f0e515b..4a4ed26 100644 --- a/pysits/docs/content/sits_kfold_validate.md +++ b/pysits/docs/content/sits_kfold_validate.md @@ -9,14 +9,13 @@ Args: ml_method (SITSMachineLearningMethod): Machine learning method. impute_fn: Imputation function to remove NA. multicores (int): Number of cores to process in parallel. - gpu_memory (int): Memory available in GPU in GB (default = 4). + gpu_memory (int): Memory available in GPU in GB (default = 4) batch_size (int): Batch size for GPU classification. progress (bool): Show progress bar? - **kwargs (dict): Additional arguments. Returns: - resolve_and_invoke_accuracy_class: A confusion matrix object to be used - for validation assessment. + resolve_and_invoke_accuracy_class: A confusion matrix to be used for + validation assessment. Notes: Cross-validation is a technique for assessing how the results of a @@ -36,12 +35,13 @@ Notes: Examples: from pysits import * import tempfile + import os # A dataset containing a tibble with time series samples # for the Mato Grosso state in Brasil # create a list to store the results results = [] - # accuracy assessment random forest + # accuracy assessment lightTAE acc_rfor = sits_kfold_validate( samples_modis_ndvi, folds=5, @@ -52,7 +52,8 @@ Examples: # put the result in a list results.append(acc_rfor) # save to xlsx file + output_file = os.path.join(tempfile.gettempdir(), "accuracy_mato_grosso_dl_.xlsx") sits_to_xlsx( results, - file=tempfile.NamedTemporaryFile(prefix="accuracy_mato_grosso_dl_", suffix=".xlsx").name + file=output_file ) diff --git a/pysits/docs/content/sits_label_classification.md b/pysits/docs/content/sits_label_classification.md index af94f53..a44a9e4 100644 --- a/pysits/docs/content/sits_label_classification.md +++ b/pysits/docs/content/sits_label_classification.md @@ -11,26 +11,25 @@ class label in the output raster. A GPKG file with segment summaries (including a `class` column) is written automatically. Args: - cube (SITSCubeModel): Classified probability data cube (pixel-based or + cube (SITSCubeModel): classified probability data cube (pixel-based or segment-based). - memsize (int): Maximum overall memory (in GB) to label the + memsize (int): maximum overall memory (in GB) to label the classification. - multicores (int): Number of workers to label the classification in + multicores (int): number of workers to label the classification in parallel. - output_dir (str | pathlib.Path): Output directory for classified files. - version (str): Version of resulting image (in the case of multiple - runs). - progress (bool): Show progress bar? - label_method (str): Decision method for segment-based labeling. One of + output_dir (str | pathlib.Path): output directory for classified files. + version (str): version of resulting image (in the case of multiple runs). + progress (bool): show progress bar? + label_method (str): decision method for segment-based labeling. One of "mean" (default), "median", or "majority". Only used when input is a segment-based probability cube. - **kwargs (dict): Configuration parameters for exact extraction of - segment values. + **kwargs (dict): configuration parameters for + `exactextractr::exact_extract`. Returns: - SITSCubeModel: A data cube with an image with the classified map. When - input is a segment-based probability cube, the output preserves - vector information. + SITSCubeModel: a data cube with an image with the classified map. When + input is a segment-based probability cube, the output is a + segment-based classified cube with vector information preserved. Notes: The main `sits` classification workflow has the following steps: @@ -51,7 +50,7 @@ Notes: 9. `sits_label_classification`: produce a classified map by selecting the label with the highest probability from a smoothed cube. The OBIA workflow adds segmentation before classification: - 1. `sits_segment`: segment the raster cube to produce a vector_cube. + 1. `sits_segment`: segment the raster cube to produce a vector cube. 2. `sits_classify`: classify pixel-level probabilities, preserving vector support. 3. `sits_label_classification`: aggregate probabilities per segment and diff --git a/pysits/docs/content/sits_labels.md b/pysits/docs/content/sits_labels.md index 0c3f6ce..a5da2ec 100644 --- a/pysits/docs/content/sits_labels.md +++ b/pysits/docs/content/sits_labels.md @@ -1,12 +1,12 @@ Get labels associated to a data set -Finds labels in a time series set or data cube +Finds labels in a set of time series or a data cube Args: data (SITSTimeSeriesModel | SITSTimeSeriesPatternsModel | SITSCubeModel | SITSMachineLearningMethod): time series, patterns, data cube, or trained model. Returns: - list: the labels of the input data. + list: The labels of the input data. Examples: from pysits import * @@ -18,7 +18,7 @@ Examples: labels_pat = sits_labels(sits_patterns(samples_modis_ndvi)) # create a random forest model rfor_model = sits_train(samples_modis_ndvi, sits_rfor()) - # get labels for the model + # get lables for the model labels_mod = sits_labels(rfor_model) # create a data cube from local files data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") @@ -29,7 +29,7 @@ Examples: ) # classify a data cube probs_cube = sits_classify( - data=cube, ml_model=rfor_model, output_dir=tempfile.gettempdir() + data=cube, ml_model=rfor_model, output_dir=tempfile.mkdtemp() ) # get the labels for a probs cube labels_probs = sits_labels(probs_cube) diff --git a/pysits/docs/content/sits_labels_summary.md b/pysits/docs/content/sits_labels_summary.md index 3ceec10..5abcbba 100644 --- a/pysits/docs/content/sits_labels_summary.md +++ b/pysits/docs/content/sits_labels_summary.md @@ -3,10 +3,10 @@ Inform label distribution of a set of time series Describes labels in a set of time series. Args: - data (SITSTimeSeriesModel): Valid time series. + data (SITSTimeSeriesModel): Set of time series. Returns: - SITSFrame: A table with the frequency of each label. + SITSFrame: The frequency of each label. Examples: from pysits import * diff --git a/pysits/docs/content/sits_lightgbm.md b/pysits/docs/content/sits_lightgbm.md index 2370aef..f309d6d 100644 --- a/pysits/docs/content/sits_lightgbm.md +++ b/pysits/docs/content/sits_lightgbm.md @@ -1,42 +1,48 @@ Train light gradient boosting model -Use LightGBM algorithm to classify samples. This function is a front-end to -the `lightgbm` package. LightGBM (short for Light Gradient Boosting Machine) -is a gradient boosting framework developed by Microsoft that's designed for -fast, scalable, and efficient training of decision tree-based models. It is -widely used in machine learning for classification, regression, ranking, and -other tasks, especially with large-scale data. +Use LightGBM algorithm to classify samples. This function is a front-end +to the `lightgbm` package. LightGBM (short for Light Gradient Boosting +Machine) is a gradient boosting framework developed by Microsoft that's +designed for fast, scalable, and efficient training of decision +tree-based models. It is widely used in machine learning for +classification, regression, ranking, and other tasks, especially with +large-scale data. Args: samples (SITSTimeSeriesModel): Time series with the training samples. boosting_type (str): Type of boosting algorithm (default = "gbdt"). objective (str): Aim of the classifier (default = "multiclass"). - min_samples_leaf (int): Minimal number of data in one leaf. Can be used - to deal with over-fitting. + min_samples_leaf (int): Minimal number of data in one leaf. Can be + used to deal with over-fitting. max_depth (int): Limit the max depth for tree model. learning_rate (float): Shrinkage rate for leaf-based algorithm. num_iterations (int): Number of iterations to train the model. - n_iter_no_change (int): Number of iterations without improvements until - training stops. - validation_split (float): Fraction of the training data for validation. - The model will set apart this fraction and will evaluate the loss and - any model metrics on this data at the end of each epoch. - **kwargs (dict): Other parameters to be passed to `lightgbm::lightgbm` - function. + n_iter_no_change (int): Number of iterations without improvements + until training stops. + validation_split (float): Fraction of the training data for + validation. The model will set apart this fraction and will + evaluate the loss and any model metrics on this data at the end + of each epoch. + **kwargs (dict): Other parameters to be passed to + `lightgbm::lightgbm` function. Returns: - SITSMachineLearningMethod: Model fitted to input data (to be passed to - `sits_classify`). + SITSMachineLearningMethod: Model fitted to input data (to be passed + to `sits_classify`). Examples: from pysits import * # Example of training a model for time series classification # Retrieve the samples for Mato Grosso - # train a lightgbm model - lgb_model = sits_train(samples_modis_ndvi, ml_method=sits_lightgbm) - # select the NDVI band + # train a random forest model + lgb_model = sits_train(samples_modis_ndvi, + ml_method=sits_lightgbm + ) + # classify the point point_ndvi = sits_select(point_mt_6bands, bands="NDVI") # classify the point - point_class = sits_classify(data=point_ndvi, ml_model=lgb_model) + point_class = sits_classify( + data=point_ndvi, ml_model=lgb_model + ) plot(point_class) diff --git a/pysits/docs/content/sits_lighttae.md b/pysits/docs/content/sits_lighttae.md index 9cb56c8..b5de98f 100644 --- a/pysits/docs/content/sits_lighttae.md +++ b/pysits/docs/content/sits_lighttae.md @@ -17,26 +17,27 @@ by the larger number of available heads. Args: samples (SITSTimeSeriesModel): Time series with the training samples. - samples_validation (SITSTimeSeriesModel): Time series with the validation - samples. If `samples_validation` parameter is provided, + samples_validation (SITSTimeSeriesModel): Time series with the + validation samples. If `samples_validation` parameter is provided, `validation_split` is ignored. epochs (int): Number of iterations to train the model (min = 1, max = 20000). - batch_size (int): Number of samples per gradient update (min = 16, max = - 2048). + batch_size (int): Number of samples per gradient update (min = 16, max + = 2048). validation_split (float): Fraction of training data to be used as validation data. optimizer: Optimizer function to be used. - opt_hparams (dict): Hyperparameters for optimizer: `lr` : Learning rate of - the optimizer `eps`: Term added to the denominator to improve + opt_hparams (dict): Hyperparameters for optimizer: `lr` : Learning rate + of the optimizer `eps`: Term added to the denominator to improve numerical stability. `weight_decay`: L2 regularization rate. lr_decay_epochs (int): Number of epochs to reduce learning rate. lr_decay_rate (float): Decay factor for reducing learning rate. - patience (int): Number of epochs without improvements until training stops. + patience (int): Number of epochs without improvements until training + stops. min_delta (float): Minimum improvement in loss function to reset the patience counter. seed (int): Seed for random values. - verbose (bool): Verbosity mode. Default is False. + verbose (bool): Verbosity mode. Default is `False`. Returns: R: A fitted model to be used for classification of data cubes. @@ -52,7 +53,7 @@ Notes: temporal-attention-pytorch If you use this method, please cite the original TAE and the LTAE paper. We also used the code made available by Maja Schneider in her work with - Marco K\u00f6rner referenced below and available at + Marco Körner referenced below and available at https://github.com/maja601/RC2020-psetae. Examples: @@ -72,18 +73,18 @@ Examples: ) # classify a data cube probs_cube = sits_classify( - data=cube, ml_model=torch_model, output_dir=tempfile.gettempdir() + data=cube, ml_model=torch_model, output_dir=tempfile.mkdtemp() ) # plot the probability cube plot(probs_cube) # smooth the probability cube using Bayesian statistics - bayes_cube = sits_smooth(probs_cube, output_dir=tempfile.gettempdir()) + bayes_cube = sits_smooth(probs_cube, output_dir=tempfile.mkdtemp()) # plot the smoothed cube plot(bayes_cube) # label the probability cube label_cube = sits_label_classification( bayes_cube, - output_dir=tempfile.gettempdir() + output_dir=tempfile.mkdtemp() ) # plot the labelled cube plot(label_cube) diff --git a/pysits/docs/content/sits_list_collections.md b/pysits/docs/content/sits_list_collections.md index 5def432..56fa7c3 100644 --- a/pysits/docs/content/sits_list_collections.md +++ b/pysits/docs/content/sits_list_collections.md @@ -14,5 +14,5 @@ Returns: Examples: from pysits import * - # show the names of the collections supported by SITS + # show the names of the colors supported by SITS sits_list_collections() diff --git a/pysits/docs/content/sits_lstm_fcn.md b/pysits/docs/content/sits_lstm_fcn.md new file mode 100644 index 0000000..bbffb21 --- /dev/null +++ b/pysits/docs/content/sits_lstm_fcn.md @@ -0,0 +1,76 @@ +Train a Long Short Term Memory Fully Convolutional Network + +Uses a branched neural network consisting of a lstm (long short term memory) +branch and a three-layer fully convolutional branch (FCN) followed by +concatenation to classify time series data. +This function is based on the paper by Fazle Karim, Somshubra Majumdar, and +Houshang Darabi. If you use this method, please cite the original LSTM with FCN +paper. +The original python code is available at the website +https://github.com/titu1994/LSTM-FCN. This code is licensed as GPL-3. + +Args: + samples (SITSTimeSeriesModel): Time series with the training samples. + samples_validation (SITSTimeSeriesModel): Time series with the + validation samples. If the `samples_validation` parameter is + provided, the `validation_split` parameter is ignored. + cnn_layers (list[int]): Number of 1D convolutional filters per layer. + cnn_kernels (list[int]): Size of the 1D convolutional kernels. + lstm_width (int): Number of neurons in the lstm hidden layer. + lstm_dropout (float): Dropout rate of the lstm layer. + epochs (int): Number of iterations to train the model. + batch_size (int): Number of samples per gradient update. + validation_split (float): Fraction of training data to be used for + validation. + optimizer: Optimizer function to be used. + opt_hparams (dict): Hyperparameters for optimizer: lr : Learning rate + of the optimizer eps: Term added to the denominator to improve + numerical stability. weight_decay: L2 regularization. + lr_decay_epochs (int): Number of epochs to reduce learning rate. + lr_decay_rate (float): Decay factor for reducing learning rate. + patience (int): Number of epochs without improvements until training + stops. + min_delta (float): Minimum improvement in loss function to reset the + patience counter. + seed (int): Seed for random values. + verbose (bool): Verbosity mode. Default is `False`. + +Returns: + SITSMachineLearningMethod: A fitted model to be used for + classification. + +Examples: + from pysits import * + import tempfile + + # create an LSTM model + torch_model = sits_train( + samples_modis_ndvi, + sits_lstm_fcn(epochs=20, verbose=True) + ) + # plot the model + plot(torch_model) + # create a data cube from local files + data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") + cube = sits_cube( + source="BDC", + collection="MOD13Q1-6.1", + data_dir=data_dir + ) + # classify a data cube + probs_cube = sits_classify( + data=cube, ml_model=torch_model, output_dir=tempfile.gettempdir() + ) + # plot the probability cube + plot(probs_cube) + # smooth the probability cube using Bayesian statistics + bayes_cube = sits_smooth(probs_cube, output_dir=tempfile.gettempdir()) + # plot the smoothed cube + plot(bayes_cube) + # label the probability cube + label_cube = sits_label_classification( + bayes_cube, + output_dir=tempfile.gettempdir() + ) + # plot the labelled cube + plot(label_cube) diff --git a/pysits/docs/content/sits_merge.md b/pysits/docs/content/sits_merge.md index 1cd32c4..1ced361 100644 --- a/pysits/docs/content/sits_merge.md +++ b/pysits/docs/content/sits_merge.md @@ -4,7 +4,7 @@ To merge two series, we consider that they contain different attributes but refer to the same data cube and spatiotemporal location. This function is useful for merging different bands of the same location. For example, one may want to put the raw and smoothed bands for the same set of locations in the -same data set. +same table. In the case of data cubes, the function merges the images based on the following conditions: 1. If the two cubes have different bands but compatible timelines, the bands @@ -21,14 +21,14 @@ following conditions: 3. otherwise, the function will produce an error. Args: - data1 (SITSTimeSeriesModel | SITSCubeModel): Time series or data cube. - data2 (SITSTimeSeriesModel | SITSCubeModel): Time series or data cube. - suffix (list[str]): If data1 and data2 have duplicate bands, this - suffix will be added. - **kwargs (dict): Additional parameters. + data1 (SITSTimeSeriesModel | SITSCubeModel): time series or data cube. + data2 (SITSTimeSeriesModel | SITSCubeModel): time series or data cube. + suffix (list[str]): if data1 and data2 have duplicate bands, this suffix + will be added. + **kwargs (dict): additional parameters. Returns: - SITSFrame: merged data sets (time series or data cube). + SITSFrame: merged data sets. Examples: from pysits import * diff --git a/pysits/docs/content/sits_mixture_model.md b/pysits/docs/content/sits_mixture_model.md index 4022a58..941f5da 100644 --- a/pysits/docs/content/sits_mixture_model.md +++ b/pysits/docs/content/sits_mixture_model.md @@ -11,9 +11,9 @@ Args: endmembers (pandas.DataFrame | str | pathlib.Path): Reference spectral endmembers (see details below). rmse_band (bool): Whether the error associated with the linear model - should be generated. If `True`, a new band with errors for each - pixel is generated using the root mean square measure (RMSE). - Default is `True`. + should be generated. If `True`, a new band with errors for each pixel + is generated using the root mean square measure (RMSE). Default is + `True`. multicores (int): Number of cores to be used for generate the mixture model. progress (bool): Show progress bar? Default is `True`. @@ -23,10 +23,10 @@ Args: Returns: SITSFrame: In case of a cube, a data cube with the fractions of each - endmember. The sum of all fractions is restricted to 1 (scaled from - 0 to 10000), corresponding to the abundance of the endmembers in the - pixels. In case of a set of sample time series, the time series with - the values corresponding to each fraction. + endmember. The sum of all fractions is restricted to 1 (scaled from 0 + to 10000), corresponding to the abundance of the endmembers in the + pixels. In case of sample time series, the time series is returned + with the values corresponding to each fraction. Notes: Many pixels in images of medium-resolution satellites such as Landsat or @@ -49,8 +49,8 @@ Notes: Examples: from pysits import * import pandas as pd - import tempfile import os + import tempfile # Create a sentinel-2 cube s2_cube = sits_cube( diff --git a/pysits/docs/content/sits_mlp.md b/pysits/docs/content/sits_mlp.md index f1ac6b3..8274415 100644 --- a/pysits/docs/content/sits_mlp.md +++ b/pysits/docs/content/sits_mlp.md @@ -1,8 +1,8 @@ Train multi-layer perceptron models using torch -Use a multi-layer perceptron algorithm to classify data. This function uses -the R "torch" and "luz" packages. Please refer to the documentation of those -package for more details. +Use a multi-layer perceptron algorithm to classify data. This function +uses deep learning routines for training and classification. Please refer +to the underlying documentation for more details. Args: samples (SITSTimeSeriesModel): Time series with the training samples. @@ -12,25 +12,25 @@ Args: layers (list[int]): Number of hidden nodes in each layer. dropout_rates (list[float]): Dropout rates (0,1) for each layer. optimizer: Optimizer function to be used. - opt_hparams (dict): Hyperparameters for optimizer: lr : Learning rate - of the optimizer eps: Term added to the denominator to improve - numerical stability.. weight_decay: L2 regularization + opt_hparams (dict): Hyperparameters for optimizer: lr: Learning rate + of the optimizer; eps: Term added to the denominator to improve + numerical stability; weight_decay: L2 regularization. epochs (int): Number of iterations to train the model. batch_size (int): Number of samples per gradient update. validation_split (float): Number between 0 and 1. Fraction of the training data for validation. The model will set apart this - fraction and will evaluate the loss and any model metrics on this - data at the end of each epoch. + fraction and will evaluate the loss and any model metrics on + this data at the end of each epoch. patience (int): Number of epochs without improvements until training stops. min_delta (float): Minimum improvement in loss function to reset the patience counter. seed (int): Seed for random values. - verbose (bool): Verbosity mode (True/False). Default is False. + verbose (bool): Verbosity mode. Default is `False`. Returns: - SITSMachineLearningMethod: A torch mlp model to be used for - classification. + SITSMachineLearningMethod: An MLP model to be used for + classification. Notes: `sits` provides a set of default values for all classification models. @@ -38,15 +38,15 @@ Notes: Nevertheless, users can control all parameters for each model. Novice users can rely on the default values, while experienced ones can fine-tune deep learning models using `sits_tuning`. - The default parameters for the MLP have been chosen based on the work by - Wang et al. 2017 that takes multilayer perceptrons as the baseline for - time series classifications: (a) Three layers with 512 neurons each, - specified by the parameter `layers`; (b) dropout rates of 10 (c) the - "optimizer_adam" as optimizer (default value); (d) a number of training - steps (`epochs`) of 100; (e) a `batch_size` of 64, which indicates how - many time series are used for input at a given steps; (f) a validation - percentage of 20 will be randomly set side for validation. (g) The - "relu" activation function. + The default parameters for the MLP have been chosen based on the work + by Wang et al. 2017 that takes multilayer perceptrons as the baseline + for time series classifications: (a) Three layers with 512 neurons + each, specified by the parameter `layers`; (b) dropout rates of 10 (c) + the "optimizer_adam" as optimizer (default value); (d) a number of + training steps (`epochs`) of 100; (e) a `batch_size` of 64, which + indicates how many time series are used for input at a given steps; + (f) a validation percentage of 20 will be randomly set side for + validation. (g) The "relu" activation function. Examples: from pysits import * diff --git a/pysits/docs/content/sits_model_export.md b/pysits/docs/content/sits_model_export.md index 1139189..80b2870 100644 --- a/pysits/docs/content/sits_model_export.md +++ b/pysits/docs/content/sits_model_export.md @@ -1,14 +1,14 @@ Export classification models Given a trained machine learning or deep learning model, exports the -model as an object for further exploration outside the `sits` package. +model as an object for further exploration outside the package. Args: ml_model (SITSMachineLearningMethod): A trained machine learning - model. + model. Returns: - None: The model in the original format of the machine learning or + None: the model in the original format of the machine learning or deep learning package. Examples: diff --git a/pysits/docs/content/sits_mosaic.md b/pysits/docs/content/sits_mosaic.md index 5f4ca0b..804d38d 100644 --- a/pysits/docs/content/sits_mosaic.md +++ b/pysits/docs/content/sits_mosaic.md @@ -1,20 +1,21 @@ Mosaic classified cubes -Creates a mosaic of all tiles of a data cube. Mosaics can be created from -both regularized ARD images or from classified maps. In the case of ARD -images, a mosaic will be produce for each band/date combination. It is -better to first regularize the data cubes and then use `sits_mosaic`. +Creates a mosaic of all tiles of a data cube. Mosaics can be created +from both regularized ARD images or from classified maps. In the case +of ARD images, a mosaic will be produced for each band/date +combination. It is better to first regularize the data cubes and then +use `sits_mosaic`. Args: cube (SITSCubeModel): A data cube. - crs (str | int): A target coordinate reference system of raster - mosaic. The provided crs could be a string (e.g, "EPSG:4326" or a - proj4string), or an EPSG code number (e.g. 4326). Default is - "EPSG:3857" - WGS 84 / Pseudo-Mercator. - roi (dict | str | pathlib.Path | geopandas.GeoDataFrame): Region of + crs (str | int): A target coordinate reference system of the raster + mosaic. The provided crs could be a string (e.g, "EPSG:4326" + or a proj4string), or an EPSG code number (e.g. 4326). Default + is "EPSG:3857" - WGS 84 / Pseudo-Mercator. + roi (str | pathlib.Path | geopandas.GeoDataFrame | dict): Region of interest (see below). - multicores (int): Number of cores that will be used to crop the images - in parallel. + multicores (int): Number of cores that will be used to crop the + images in parallel. output_dir (str | pathlib.Path): Directory for output images. res (float): Spatial resolution of the mosaic. Default is None. version (str): Version of resulting image (in the case of multiple @@ -28,9 +29,10 @@ Notes: To define a `roi` use one of: - A path to a shapefile with polygons; - A `geopandas.GeoDataFrame`; - - A named `dict` (`"lon_min"`, `"lat_min"`, `"lon_max"`, - `"lat_max"`) in WGS84; - - A named `dict` (`"xmin"`, `"xmax"`, `"ymin"`, `"ymax"`) with XY + - A `SpatExtent` object from `terra` package; + - A `dict` (`"lon_min"`, `"lat_min"`, `"lon_max"`, `"lat_max"`) in + WGS84; + - A `dict` (`"xmin"`, `"xmax"`, `"ymin"`, `"ymax"`) with XY coordinates. The user should specify the CRS of the mosaic. We use "EPSG:3857" (Pseudo-Mercator) as the default. @@ -38,8 +40,8 @@ Notes: Examples: from pysits import * import tempfile - from shapely.geometry import Polygon import geopandas as gpd + from shapely.geometry import Polygon # create a random forest model rfor_model = sits_train(samples_modis_ndvi, sits_rfor()) @@ -52,23 +54,25 @@ Examples: ) # classify a data cube probs_cube = sits_classify( - data=cube, ml_model=rfor_model, output_dir=tempfile.mkdtemp() + data=cube, ml_model=rfor_model, output_dir=tempfile.gettempdir() ) # smooth the probability cube using Bayesian statistics - bayes_cube = sits_smooth(probs_cube, output_dir=tempfile.mkdtemp()) + bayes_cube = sits_smooth(probs_cube, output_dir=tempfile.gettempdir()) # label the probability cube label_cube = sits_label_classification( bayes_cube, - output_dir=tempfile.mkdtemp() + output_dir=tempfile.gettempdir() ) # create roi roi = gpd.GeoSeries( - [Polygon([ - (-55.64768, -11.68649), - (-55.69654, -11.66455), - (-55.62973, -11.61519), - (-55.64768, -11.68649) - ])], + [ + Polygon([ + (-55.64768, -11.68649), + (-55.69654, -11.66455), + (-55.62973, -11.61519), + (-55.64768, -11.68649) + ]) + ], crs="EPSG:4326" ) # crop and mosaic classified image @@ -76,5 +80,5 @@ Examples: cube=label_cube, roi=roi, crs="EPSG:4326", - output_dir=tempfile.mkdtemp() + output_dir=tempfile.gettempdir() ) diff --git a/pysits/docs/content/sits_parallel.md b/pysits/docs/content/sits_parallel.md index d3ef150..d7bfbe0 100644 --- a/pysits/docs/content/sits_parallel.md +++ b/pysits/docs/content/sits_parallel.md @@ -1,38 +1,52 @@ Configure or query sits parallel processing -`sits_parallel` starts, restarts, stops, or gets a persistent `PSOCK` -cluster used by `sits` functions that support parallel processing. +`sits_parallel` starts, restarts, stops, or gets a persistent `PSOCK` cluster +used by `sits` functions that support parallel processing. To stop the cluster and free resources, call `sits_parallel(workers = 0)` (recommended). Calling with `workers = 1` has the same effect (parallel disabled). -When called with no arguments, `sits_parallel()` returns the current -cluster object. If no cluster is active, it returns `None`. +When called with no arguments, `sits_parallel()` returns the current cluster +object. If no cluster is active, it returns `None`. Args: workers (int): Number of workers to use. - `workers >= 2`: start or - restart a `PSOCK` cluster with `workers` workers. - - `workers <= 1`: stop any active cluster. + restart a `PSOCK` cluster with `workers` workers. - `workers <= 1`: + stop any active cluster. log (bool): If `True`, enables worker log/debug mode. output_dir (str | pathlib.Path): Output directory where log files are written when `log = True`. Returns: - None: If called with no arguments, returns the current `parallel` - cluster object or `None` if no cluster is active. If called with - `workers`, returns `None`. + None: If called with no arguments, returns the current `parallel` cluster + object or `None` if no cluster is active. If called with `workers`, + returns nothing. Notes: - This function is intended for long pipelines and production - environments where repeatedly creating and stopping clusters inside - each `sits` call is expensive. After `sits_parallel(workers = N)` is - called, `sits` functions can reuse the same cluster across multiple - calls. - When `workers >= 2`, worker processes inherit the current library - paths (`.libPaths()`) and selected environment variables required for - data access (for example, variables starting with `AWS_`). - When the streaming GPU pipeline is enabled - (`SITS_GPU_PIPELINE=stream`), `sits` functions attach the pipeline's - read and write stages to this cluster instead of creating a private - worker pool. Reads use at most `workers - 1` nodes and writes use one - node, so for a classification with `multicores` read slots start the - cluster with `workers = multicores + 1`. + This function is intended for long pipelines and production environments + where repeatedly creating and stopping clusters inside each `sits` call is + expensive. After `sits_parallel(workers = N)` is called, `sits` functions + can reuse the same cluster across multiple calls. + When `workers >= 2`, worker processes inherit the current library paths + (`.libPaths()`) and selected environment variables required for data access + (for example, variables starting with `AWS_`). + When the streaming GPU pipeline is enabled (`SITS_GPU_PIPELINE=stream`), + `sits` functions attach the pipeline's read and write stages to this + cluster instead of creating a private worker pool. Reads use at most + `workers - 1` nodes and writes use one node, so for a classification with + `multicores` read slots start the cluster with `workers = multicores + 1`. + +Examples: + from pysits import * + + # Start a persistent cluster + sits_parallel(workers=32) + + # Query the current cluster + cl = sits_parallel() + cl + + # Stop the cluster (recommended) + sits_parallel(workers=0) + + # Alternative stop (also disables parallel) + sits_parallel(workers=1) diff --git a/pysits/docs/content/sits_patterns.md b/pysits/docs/content/sits_patterns.md index 9634921..8d15f14 100644 --- a/pysits/docs/content/sits_patterns.md +++ b/pysits/docs/content/sits_patterns.md @@ -12,7 +12,7 @@ which is also described in the reference paper. Args: data (SITSTimeSeriesModel): Time series. freq (int): Interval in days for estimates. - formula: Formula to be applied in the estimate. + formula (str): Formula to be applied in the estimate. **kwargs (dict): Any additional parameters. Returns: diff --git a/pysits/docs/content/sits_pre_train.md b/pysits/docs/content/sits_pre_train.md index c6e3001..3a0a4df 100644 --- a/pysits/docs/content/sits_pre_train.md +++ b/pysits/docs/content/sits_pre_train.md @@ -30,14 +30,14 @@ Args: rl_method (SITSRepresentationLearningMethod): A pre-training representation learning method used to build an encoder that generates embeddings (e.g., `sits_ssl_lejepa()` or - `sits_contrastive_learning()`). It must be a function that - takes `samples` and returns an encoder. + `sits_contrastive_learning()`). It must be a function that takes + `samples` and returns an encoder. Returns: SITSRepresentationLearningMethod: A pre-trained deep learning - encoder and the metadata required for subsequent encoding (e.g., - band order, feature naming, and normalization statistics, when - applicable). + encoder containing the metadata required for subsequent encoding + (e.g., band order, feature naming, and normalization statistics, + when applicable). Examples: from pysits import * diff --git a/pysits/docs/content/sits_pred_features.md b/pysits/docs/content/sits_pred_features.md index e563c84..a971c8f 100644 --- a/pysits/docs/content/sits_pred_features.md +++ b/pysits/docs/content/sits_pred_features.md @@ -1,17 +1,17 @@ Obtain numerical values of predictors for time series samples -Predictors are X-Y values required for machine learning algorithms, -organized as a data table where each row corresponds to a training -sample. The first two columns of the predictors table are categorical -("label_id" and "label"). The other columns are the values of each band -and time, organized first by band and then by time. This function -returns the numeric values associated to each sample. +Predictors are X-Y values required for machine learning algorithms, organized +as a data table where each row corresponds to a training sample. The first two +columns of the predictors table are categorical ("label_id" and "label"). The +other columns are the values of each band and time, organized first by band and +then by time. This function returns the numeric values associated to each +sample. Args: pred (pandas.DataFrame): X-Y predictors, with one row per sample. Returns: - SITSFrame: The Y predictors for the sample, with one row per + SITSFrame: the Y predictors for the sample, with one row per sample. Examples: diff --git a/pysits/docs/content/sits_pred_normalize.md b/pysits/docs/content/sits_pred_normalize.md index 49fcd21..ec62e4f 100644 --- a/pysits/docs/content/sits_pred_normalize.md +++ b/pysits/docs/content/sits_pred_normalize.md @@ -3,12 +3,12 @@ Normalize predictor values Most machine learning algorithms require data to be normalized. This applies to the "SVM" method and to all deep learning ones. To normalize the predictors, it is required that the statistics per band for each -sample have been obtained by the `sits_stats` function. +sample have been obtained by the "sits_stats" function. Args: pred (pandas.DataFrame): X-Y predictors, with one row per sample. - stats (dict): Values of time series for Q02 and Q98 of the data - (two elements). + stats (dict): Values of time series for Q02 and Q98 of the data, + with two elements. Returns: SITSFrame: Normalized predictor values. diff --git a/pysits/docs/content/sits_pred_sample.md b/pysits/docs/content/sits_pred_sample.md index 08b11d7..e05cd7a 100644 --- a/pysits/docs/content/sits_pred_sample.md +++ b/pysits/docs/content/sits_pred_sample.md @@ -7,7 +7,7 @@ fraction of the predictors to serve as test values for the deep learning algorithm. Args: - pred (pandas.DataFrame): X-Y predictors, one row per sample. + pred (pandas.DataFrame): X-Y predictors with one row per sample. frac (float): Fraction of the X-Y predictors to be extracted. Returns: diff --git a/pysits/docs/content/sits_predictors.md b/pysits/docs/content/sits_predictors.md index e8f634c..8eb7971 100644 --- a/pysits/docs/content/sits_predictors.md +++ b/pysits/docs/content/sits_predictors.md @@ -10,4 +10,4 @@ Args: samples (SITSTimeSeriesModel): Time series samples. Returns: - SITSFrame: The predictors for the sample, with one row per sample. + SITSFrame: The predictors for the samples, with one row per sample. diff --git a/pysits/docs/content/sits_random_sampling.md b/pysits/docs/content/sits_random_sampling.md new file mode 100644 index 0000000..81fbb34 --- /dev/null +++ b/pysits/docs/content/sits_random_sampling.md @@ -0,0 +1,30 @@ +Sampling random points in a data cube + +Takes a random sample of locations in a data cube + +Args: + cube (SITSCubeModel): Data cube. + n_samples (int): Number of points to be sampled. + multicores (int): Number of cores used to sample the images in + parallel. + memsize (int): Memory available for sampling. + progress (bool): Show progress bar? Default is `True`. + +Returns: + SITSFrameSF: sample locations. + +Examples: + from pysits import * + + # create a data cube from local files + data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") + cube = sits_cube( + source="BDC", + collection="MOD13Q1-6.1", + data_dir=data_dir + ) + # sample for data cube + samples = sits_random_sampling( + cube=cube, + n_samples=100 + ) diff --git a/pysits/docs/content/sits_reclassify.md b/pysits/docs/content/sits_reclassify.md index c07a943..f435d9e 100644 --- a/pysits/docs/content/sits_reclassify.md +++ b/pysits/docs/content/sits_reclassify.md @@ -1,57 +1,53 @@ Reclassify a classified cube Reclassify a classification cube using a set of named expressions. -For classified cubes, expressions relabel pixels based on logical -conditions that may combine information from the classified cube and -an optional mask cube. -For probability cubes and probability vector cubes, expressions are -used to group input labels into new labels by aggregating -probabilities (summing probabilities of the selected input labels). +For classified label cubes, expressions relabel pixels based on logical +conditions that may combine information from the classified cube and an +optional mask cube. +For probability cubes and probability vector cubes, expressions are used +to group input labels into new labels by aggregating probabilities +(summing probabilities of the selected input labels). Args: - cube (SITSCubeModel): Image cube to be reclassified (a classified + cube (SITSCubeModel): Image cube to be reclassified (classified label cube, probability cube, or probability vector cube). mask (SITSCubeModel): Image cube with additional information to be - used in expressions. Used only for classified cube - reclassification. - rules (dict): Expressions to be evaluated. For classified cubes, - expressions must evaluate to a boolean and may refer to `cube` - and `mask`. For probability cubes and probability vector - cubes, each named rule selects one or more input labels (for - example using `cube %in% c(...)`). The probabilities of the - selected labels are summed to produce the new label given by - the rule name. - exclude_mask_na (bool): Should cube pixels be set to `None` when - `None` values are found in mask pixels? (default `True`). - Used only for classified cubes. - memsize (int): Memory available for processing in GB (min = 1, - max = 16384). - multicores (int): Number of cores to be used for processing - (min = 1, max = 2048). - output_dir (str | pathlib.Path): Directory where files will be - saved. + used in expressions. Used only for classified label cubes. + rules (dict): Expressions to be evaluated. For classified label cubes, + expressions must evaluate to `bool` and may refer to `cube` and + `mask`. For probability cubes and probability vector cubes, each + named rule selects one or more input labels (for example using + `cube %in% c(...)`). The probabilities of the selected labels are + summed to produce the new label given by the rule name. + exclude_mask_na (bool): Should cube pixels be set to `None` when `None` + values are found in mask pixels? (default `True`). Used only for + classified label cubes. + memsize (int): Memory available for processing in GB (min = 1, max = + 16384). + multicores (int): Number of cores to be used for processing (min = 1, + max = 2048). + output_dir (str | pathlib.Path): Directory where files will be saved. version (str): Version of resulting image. progress (bool): Show progress bar? **kwargs (dict): Other parameters for specific methods. Returns: SITSCubeModel: An object of the same type as `cube`: a classified - cube for label cubes, or a probability cube and probability - vector cube for probability cubes and probability vector - cubes, respectively. + label cube for label cubes, or a probability cube and probability + vector cube for probability cubes and probability vector cubes, + respectively. Notes: - For classified cubes, reclassification changes the class assigned - to each pixel based on user-defined rules. Users should refer to - `cube` and `mask` to construct logical expressions. Expressions - are evaluated sequentially on the original classified values; - later rules override earlier ones. - For probability cubes and probability vector cubes, - reclassification is intended to group classes by combining - probabilities. Each named rule defines a new output label. For - each pixel, the probabilities of the selected input labels are - summed and assigned to the corresponding output label. Rules are - evaluated on the original probability layers. + For classified label cubes, reclassification changes the class assigned + to each pixel based on user-defined rules. Users should refer to `cube` + and `mask` to construct logical expressions. Expressions are evaluated + sequentially on the original classified values; later rules override + earlier ones. + For probability cubes and probability vector cubes, reclassification is + intended to group classes by combining probabilities. Each named rule + defines a new output label. For each pixel, the probabilities of the + selected input labels are summed and assigned to the corresponding + output label. Rules are evaluated on the original probability layers. Examples: from pysits import * @@ -117,7 +113,6 @@ Examples: }, progress=False ) - # Open classification map data_dir = r_package_dir("extdata/raster/classif", package="sits") ro_class = sits_cube( @@ -135,7 +130,6 @@ Examples: }, progress=False ) - # Reclassify cube ro_mask = sits_reclassify( cube=ro_class, diff --git a/pysits/docs/content/sits_reduce.md b/pysits/docs/content/sits_reduce.md index f493340..cc52b79 100644 --- a/pysits/docs/content/sits_reduce.md +++ b/pysits/docs/content/sits_reduce.md @@ -1,48 +1,47 @@ Reduces a cube or samples from a summarization function -Apply a temporal reduction from a named expression in a data cube or set -of sample time series. In the case of data cubes, it materializes a new -band in `output_dir`. The result will be a cube with only one date with -the raster reduced from the function. - -`sits_reduce()` allows valid R expression to compute new bands. Use R -syntax to pass an expression to this function. Besides arithmetic -operators, you can use virtually any R function that can be applied to -elements of a matrix. The provided functions must operate at line level -in order to perform temporal reduction on a pixel. -`sits_reduce()` Applies a function to each row of a matrix. In this -matrix, each row represents a pixel and each column represents a single -date. We provide some operations already implemented in the package to -perform the reduce operation. See the list of available functions below: +Apply a temporal reduction from a named expression in a data cube or set of +time series. In the case of cubes, it materializes a new band in +`output_dir`. The result will be a cube with only one date with the raster +reduced from the function. + +`sits_reduce()` allows valid R expression to compute new bands. Use R syntax +to pass an expression to this function. Besides arithmetic operators, you can +use virtually any R function that can be applied to elements of a matrix. The +provided functions must operate at line level in order to perform temporal +reduction on a pixel. +`sits_reduce()` Applies a function to each row of a matrix. In this matrix, +each row represents a pixel and each column represents a single date. We +provide some operations already implemented in the package to perform the +reduce operation. See the list of available functions below: Args: - data (SITSTimeSeriesModel | SITSCubeModel): set of sample time - series or data cube. - impute_fn: Imputation function to remove NA values. - memsize (int): Memory available for classification (in GB). - multicores (int): Number of cores to be used for classification. - output_dir (str | pathlib.Path): Directory where files will be - saved. - progress (bool): Show progress bar? - **kwargs (dict): Named expressions to be evaluated (see details). + data (SITSTimeSeriesModel | SITSCubeModel): valid set of time series or + data cube. + impute_fn: imputation function to remove NA values. + memsize (int): memory available for classification (in GB). + multicores (int): number of cores to be used for classification. + output_dir (str | pathlib.Path): directory where files will be saved. + progress (bool): show progress bar? + **kwargs (dict): named expressions to be evaluated (see details). Returns: - SITSFrame: sample time series or data cube with new bands, produced + SITSFrame: set of time series or data cube with new bands, produced according to the requested expression. Notes: - The `t_sum()`, `t_std()`, `t_skewness()`, `t_kurtosis`, `t_mse` - indexes generate values greater than the limit of a two-byte - integer. Therefore, we save the images generated by these as - Float-32 with no scale. + The `t_sum()`, `t_std()`, `t_skewness()`, `t_kurtosis`, `t_mse` indexes + generate values greater than the limit of a two-byte integer. Therefore, we + save the images generated by these as Float-32 with no scale. Examples: from pysits import * import tempfile # Reduce summarization function + point2 = sits_reduce( - sits_select(point_mt_6bands, bands="NDVI"), + sits_select(point_mt_6bands, "NDVI"), NDVI_MEDIAN="t_median(NDVI)" ) diff --git a/pysits/docs/content/sits_reduce_imbalance.md b/pysits/docs/content/sits_reduce_imbalance.md index 8efc09f..fc54ed4 100644 --- a/pysits/docs/content/sits_reduce_imbalance.md +++ b/pysits/docs/content/sits_reduce_imbalance.md @@ -3,7 +3,7 @@ Reduce imbalance in a set of samples Takes a set of samples with different labels and returns a new set. Deals with class imbalance using the synthetic minority oversampling technique (SMOTE) for oversampling. Undersampling is done using the SOM methods -available in the `sits` package. +available in the sits package. Args: samples (SITSTimeSeriesModel): Sample set to rebalance. @@ -15,7 +15,7 @@ Args: multicores (int): Number of cores to process the data (default 2). Returns: - SITSTimeSeriesModel: A set of samples with reduced imbalance. + SITSTimeSeriesModel: samples with reduced imbalance. Notes: Many training samples for Earth observation data analysis are diff --git a/pysits/docs/content/sits_regularize.md b/pysits/docs/content/sits_regularize.md index cb85b94..e0652b9 100644 --- a/pysits/docs/content/sits_regularize.md +++ b/pysits/docs/content/sits_regularize.md @@ -19,10 +19,10 @@ Args: output_dir (str | pathlib.Path): Valid directory for storing regularized images. timeline (list[str]): User-defined timeline for regularized cube. - roi (dict | geopandas.GeoDataFrame | str | pathlib.Path): Region of + roi (dict | str | pathlib.Path | geopandas.GeoDataFrame): Region of interest (see notes below). - crs (str | int): Coordinate Reference System (CRS) of the roi. (see - details below). + crs (str): Coordinate Reference System (CRS) of the roi. (see details + below). tiles (list[str]): Tiles to be produced. grid_system (str): Grid system to be used for the output images. multicores (int): Number of cores used for regularization; used for @@ -72,16 +72,16 @@ Notes: To define a `roi` use one of: - A path to a shapefile with polygons; - A `geopandas.GeoDataFrame`; - - A `SpatExtent` object from `terra` package; + - A spatial extent object; - A `dict` (`"lon_min"`, `"lat_min"`, `"lon_max"`, `"lat_max"`) in WGS84; - A `dict` (`"xmin"`, `"xmax"`, `"ymin"`, `"ymax"`) with XY coordinates. - Defining a region of interest using `SpatExtent` or XY values not in WGS84 - requires the `crs` parameter to be specified. `sits_regularize()` function - will crop the images that contain the region of interest(). NOTE: Make sure - to inform `roi` with valid geometries. `sits` will drop the use of `roi` if - it contains invalid geometries. + Defining a region of interest using a spatial extent or XY values not in + WGS84 requires the `crs` parameter to be specified. `sits_regularize()` + function will crop the images that contain the region of interest(). NOTE: + Make sure to inform `roi` with valid geometries. `sits` will drop the use + of `roi` if it contains invalid geometries. The optional `tiles` parameter indicates which tiles of the input cube will be used for regularization. When `grid_system` is informed, `tiles` may be combined with `roi` to further restrict which target grid tiles are @@ -112,7 +112,7 @@ Examples: period="P16D", res=60, multicores=2, - output_dir=tempfile.gettempdir() + output_dir=tempfile.mkdtemp() ) ## Sentinel-1 SAR @@ -136,5 +136,5 @@ Examples: res=60, roi=roi, multicores=2, - output_dir=tempfile.gettempdir() + output_dir=tempfile.mkdtemp() ) diff --git a/pysits/docs/content/sits_resnet.md b/pysits/docs/content/sits_resnet.md index cb1527e..e9dce24 100644 --- a/pysits/docs/content/sits_resnet.md +++ b/pysits/docs/content/sits_resnet.md @@ -30,9 +30,9 @@ Args: validation_split (float): Fraction of training data to be used as validation data. optimizer: Optimizer function to be used. - opt_hparams (dict): Hyperparameters for optimizer: lr : Learning rate - of the optimizer eps: Term added to the denominator to improve - numerical stability. weight_decay: L2 regularization + opt_hparams (dict): Hyperparameters for optimizer: lr: Learning rate + of the optimizer; eps: Term added to the denominator to improve + numerical stability; weight_decay: L2 regularization. lr_decay_epochs (int): Number of epochs to reduce learning rate. lr_decay_rate (float): Decay factor for reducing learning rate. patience (int): Number of epochs without improvements until training @@ -43,7 +43,7 @@ Args: verbose (bool): Verbosity mode. Default is `False`. Returns: - SITSMachineLearningMethod: A fitted model to be used for classification. + R: A fitted model to be used for classification. Examples: from pysits import * diff --git a/pysits/docs/content/sits_roi_to_tiles.md b/pysits/docs/content/sits_roi_to_tiles.md index 6450552..1eb1487 100644 --- a/pysits/docs/content/sits_roi_to_tiles.md +++ b/pysits/docs/content/sits_roi_to_tiles.md @@ -6,14 +6,14 @@ returns them as a `geopandas.GeoDataFrame`. Args: roi (dict | str | pathlib.Path | geopandas.GeoDataFrame): Region of interest (see notes below). - crs (str): Coordinate Reference System (CRS) of the roi. (see details + crs (str): Coordinate Reference System (CRS) of the roi (see details below). grid_system (str): Grid system to be used for the output images. (Default is "MGRS") Returns: - SITSFrameSF: A `geopandas.GeoDataFrame` with the intersect tiles with - three columns tile_id, epsg, and the percentage of coverage area. + SITSFrameSF: Intersected tiles with three columns tile_id, epsg, and + the percentage of coverage area. Notes: To define a `roi` use one of: @@ -23,8 +23,8 @@ Notes: WGS84; - A named `dict` (`"xmin"`, `"xmax"`, `"ymin"`, `"ymax"`) with XY coordinates. - Defining a region of interest using XY values not in WGS84 - requires the `crs` parameter to be specified. + Defining a region of interest using XY values not in WGS84 requires the + `crs` parameter to be specified. The `grid_system` parameter allows the user to reproject the files to a grid system which is different from that used in the ARD image collection of the could provider. Currently, the package supports the use of MGRS grid diff --git a/pysits/docs/content/sits_sample.md b/pysits/docs/content/sits_sample.md index f6b82db..28ee7f6 100644 --- a/pysits/docs/content/sits_sample.md +++ b/pysits/docs/content/sits_sample.md @@ -1,27 +1,27 @@ Sample a time series Takes samples from a set of time series and returns a new -`SITSTimeSeriesModel`. For a given field as a group criterion, the -result contains a percentage of the total number of samples per group. -If frac > 1, all sampling will be done with replacement. +`SITSTimeSeriesModel`. For a given field as a group criterion, the result +contains a percentage of the total number of samples per group. If frac > 1, +all sampling will be done with replacement. Args: - data (SITSTimeSeriesModel): Set of time series. - frac (float): Percentage of samples to extract (range: 0.0 to 2.0, + data (SITSTimeSeriesModel): time series. + frac (float): percentage of samples to extract (range: 0.0 to 2.0, default = 0.2). - oversample (bool): Oversample classes with small number of samples? + oversample (bool): oversample classes with small number of samples? Returns: - SITSTimeSeriesModel: A set of time series with a fixed quantity of - samples. + SITSTimeSeriesModel: time series with a fixed quantity of samples. Examples: from pysits import * - # Retrieve a set of time series with 2 classes (cerrado_2classes) + # Retrieve a set of time series with 2 classes + # (cerrado_2classes is available as a sample dataset) # Print the labels of the resulting tibble - print(sits_labels_summary(cerrado_2classes)) + summary(cerrado_2classes) # Sample by fraction data_02 = sits_sample(cerrado_2classes, frac=0.2) # Print the labels - print(sits_labels_summary(data_02)) + summary(data_02) diff --git a/pysits/docs/content/sits_sampling_design.md b/pysits/docs/content/sits_sampling_design.md index b9de38c..00e4ecc 100644 --- a/pysits/docs/content/sits_sampling_design.md +++ b/pysits/docs/content/sits_sampling_design.md @@ -7,15 +7,13 @@ providing five allocation strategies. Args: cube (SITSCubeModel): Classified data cube. expected_ua (dict): Expected values of user's accuracy. - alloc_options (list[float]): Fixed sample allocation for rare - classes. + alloc_options (list[float]): Fixed sample allocation for rare classes. std_err (float): Standard error we would like to achieve. rare_class_prop (float): Proportional area limit for rare classes. Returns: - SITSMatrix: options to decide allocation of sample size to each - class. This uses the same format as Table 5 of - Olofsson et al.(2014). + SITSMatrix: Options to decide allocation of sample size to each class. + This uses the same format as Table 5 of Olofsson et al.(2014). Examples: from pysits import * diff --git a/pysits/docs/content/sits_sankey.md b/pysits/docs/content/sits_sankey.md new file mode 100644 index 0000000..5f7a625 --- /dev/null +++ b/pysits/docs/content/sits_sankey.md @@ -0,0 +1,60 @@ +Plot class trajectories from multi-temporal classified cubes + +Builds a Sankey (alluvial) diagram showing how each pixel changes class +across a sequence of classified cubes (e.g., yearly land-use/land-cover +maps of the same area). It reveals the "from-to" class dynamics over +time, which is useful to inspect transitions and multi-year +classification consistency. +The time steps can be provided in two mutually exclusive ways: as a +single multi-temporal classified cube whose timeline lives in the files +(each file is a step), or as two or more single-step classified cubes. +In both cases the tiles must be aligned across steps. + +Args: + cubes (list[SITSCubeModel]): Alternatively, the same input as a + list: a single multi-temporal cube, or a list of single-step + cubes. Provide either this argument or `**kwargs`, not both. + labels (list[str]): Optional names for each step (one per cube), + shown on the diagram x-axis. Defaults to the start year of each + cube. + roi (dict): Optional region of interest restricting the computation. + legend (dict): Associates labels to colors (overrides the default + sits colors). + palette (str): A "cols4all" palette used for labels without an + assigned color (default = "Set3"). + title (str): Plot title (default = "Class trajectories"). + memsize (int): Maximum memory available (in GB, default = 4). + multicores (int): Number of cores for parallel processing + (default = 2). + progress (bool): Show a progress bar? (default = True). + **kwargs (dict): Classified cubes given as separate arguments: + either a single multi-temporal cube, or two or more single-step + cubes (one per time step). Ignored when `cubes` is supplied. + +Returns: + None + +Examples: + from pysits import * + import tempfile + + # train a random forest model + rfor_model = sits_train(samples_modis_ndvi, sits_rfor()) + # create a data cube from local files + data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") + cube = sits_cube( + source="BDC", + collection="MOD13Q1-6.1", + data_dir=data_dir + ) + # classify and label the cube for two time steps + probs_cube = sits_classify(cube, rfor_model, output_dir=tempfile.gettempdir()) + class_2013 = sits_label_classification( + probs_cube, output_dir=tempfile.gettempdir(), version="v2013" + ) + # a second classified cube for another time step + class_2014 = sits_label_classification( + probs_cube, output_dir=tempfile.gettempdir(), version="v2014" + ) + # plot the Sankey diagram of class trajectories between the two steps + sits_sankey(class_2013, class_2014, labels=["2013", "2014"]) diff --git a/pysits/docs/content/sits_segment.md b/pysits/docs/content/sits_segment.md index d3f7e2b..91b5a1e 100644 --- a/pysits/docs/content/sits_segment.md +++ b/pysits/docs/content/sits_segment.md @@ -8,7 +8,7 @@ additional vector file in "geopackage" format. Args: cube (SITSCubeModel): Regular data cube. seg_fn: Function to apply the segmentation. - roi (dict | str | pathlib.Path | geopandas.GeoDataFrame): Region of + roi (dict | geopandas.GeoDataFrame | str | pathlib.Path): Region of interest (see below). impute_fn: Imputation function to remove NA values. start_date (str): Start date for the segmentation. @@ -18,7 +18,6 @@ Args: output_dir (str | pathlib.Path): Directory for output file. version (str): Version of the output (for multiple segmentations). progress (bool): Show progress bar? - **kwargs (dict): Additional arguments. Returns: SITSCubeModel: The segmentation as a vector data cube. @@ -47,9 +46,9 @@ Notes: Clustering (SLIC) algorithm as adapted by Nowosad and Stepinski for multispectral and multitemporal imagery, remains available but is now deprecated and will be removed in a future release. SLIC clusters pixels - using spectral similarity and spatial–temporal proximity to produce nearly - uniform superpixels, but its iterative nature makes it less efficient for - large-scale Earth observation workflows. + using spectral similarity and spatial\\uniform superpixels, but its + iterative nature makes it less efficient for large-scale Earth observation + workflows. The result of `sits_segment` is a data cube with an additional vector file in the `geopackage` format. The location of the vector file is included in the data cube in a new column, called `vector_info`. diff --git a/pysits/docs/content/sits_select.md b/pysits/docs/content/sits_select.md index cc6641f..e9c396e 100644 --- a/pysits/docs/content/sits_select.md +++ b/pysits/docs/content/sits_select.md @@ -1,19 +1,17 @@ -Filter a data set for bands, tiles, and dates +Filter a data set (time series or cube) for bands, tiles, and dates Filter the bands, tiles, dates and labels from a set of time series or from a data cube. Args: - data (SITSTimeSeriesModel | SITSCubeModel): time series or data - cube. + data (SITSTimeSeriesModel | SITSCubeModel): time series or data cube. bands (list[str]): names of the bands. start_date (str): date in YYYY-MM-DD format: start date to be filtered. - end_date (str): date in YYYY-MM-DD format: end date to be - filtered. + end_date (str): date in YYYY-MM-DD format: end date to be filtered. dates (list[str]): sparse dates to be selected. - labels (list[str]): sparse labels to be selected (only applied - for time series data). + labels (list[str]): sparse labels to be selected (only applied for + time series data). tiles (list[str]): names of the tiles. **kwargs (dict): additional parameters to be provided. @@ -24,13 +22,12 @@ Examples: from pysits import * # Retrieve a set of time series with 2 classes - # (cerrado_2classes is available directly) # Print the original bands - print(sits_bands(cerrado_2classes)) + sits_bands(cerrado_2classes) # Select only the NDVI band data = sits_select(cerrado_2classes, bands=["NDVI"]) # Print the labels of the resulting tibble - print(sits_bands(data)) + sits_bands(data) # select start and end date point_2010 = sits_select(point_mt_6bands, start_date="2000-09-13", diff --git a/pysits/docs/content/sits_show_prediction.md b/pysits/docs/content/sits_show_prediction.md index 24b3636..b136e8d 100644 --- a/pysits/docs/content/sits_show_prediction.md +++ b/pysits/docs/content/sits_show_prediction.md @@ -4,7 +4,7 @@ This function takes a classified time series produced by a machine learning method and displays the result. Args: - class (SITSTimeSeriesModel): A time series that has been classified. + class (SITSTimeSeriesModel): A classified time series. Returns: SITSFrame: Table with the columns "from", "to", "class". diff --git a/pysits/docs/content/sits_slic.md b/pysits/docs/content/sits_slic.md index 5397b23..c2dd194 100644 --- a/pysits/docs/content/sits_slic.md +++ b/pysits/docs/content/sits_slic.md @@ -11,24 +11,24 @@ following Achanta et al. (2012). This SLIC variant is deprecated and will be removed in a future release. See references for more details. Args: - data (SITSMatrix): time series values. - step (int): distance (in number of cells) between initial supercells' + data (pandas.DataFrame): Time series values. + step (int): Distance (in number of cells) between initial supercells' centers. - compactness (float): compactness value. Larger values cause clusters to - be more compact/even (square). - dist_fun (str): distance function. Currently implemented: `euclidean, + compactness (float): Larger values cause clusters to be more + compact/even (square). + dist_fun (str): Distance function. Currently implemented: `euclidean, jsd, dtw`, and any distance function from the `philentropy` package. See `philentropy::getDistMethods()`. - avg_fun (str): averaging function to calculate the values of the - supercells' centers. Accepts any fitting function (e.g., mean or - median) or one of internally implemented "mean" and "median". - Default: "median". - iter (int): number of iterations to create the output. - minarea (int): minimal size of a supercell (in cells). - verbose (bool): show the progress bar? + avg_fun (str): Averaging function to calculate the values of the + supercells' centers. Accepts any fitting R function (e.g., + base::mean() or stats::median()) or one of internally implemented + "mean" and "median". Default: "median". + iter (int): Number of iterations to create the output. + minarea (int): Specifies the minimal size of a supercell (in cells). + verbose (bool): Show the progress bar? Returns: - R: set of segments for a single tile. + R: Set of segments for a single tile. Examples: from pysits import * diff --git a/pysits/docs/content/sits_smooth.md b/pysits/docs/content/sits_smooth.md index a9a756f..80a7e68 100644 --- a/pysits/docs/content/sits_smooth.md +++ b/pysits/docs/content/sits_smooth.md @@ -1,7 +1,7 @@ Smooth probability cubes with spatial predictors -Takes a set of classified raster layers with probabilities and applies a -Bayesian smoothing function. +Takes a set of classified raster layers with probabilities, whose metadata +is created by `sits_cube`, and applies a Bayesian smoothing function. Args: cube (SITSCubeModel): Probability data cube. @@ -43,7 +43,7 @@ Notes: 9. `sits_label_classification`: produce a classified map by selecting the label with the highest probability from a smoothed cube. Machine learning algorithms rely on training samples that are derived from - \--picked by users to represent the desired output + \h-picpicked by users to represent the desired output classes. Given the presence of mixed pixels in images regardless of resolution, and the considerable data variability within each class, these classifiers often produce results with misclassified pixels. diff --git a/pysits/docs/content/sits_snic.md b/pysits/docs/content/sits_snic.md index cc12fc5..306ff43 100644 --- a/pysits/docs/content/sits_snic.md +++ b/pysits/docs/content/sits_snic.md @@ -1,9 +1,9 @@ Segment an image using SNIC -Apply a segmentation on a data cube based on the `snic` package. This is -an adaptation and extension to remote sensing data of the SNIC -superpixels algorithm proposed by Achanta and S\u00fcsstrunk (2017). See -reference for more details. +Apply a segmentation on a data cube based on the `snic` package. This is an +adaptation and extension to remote sensing data of the SNIC superpixels +algorithm proposed by Achanta and S\u00fcsstrunk (2017). See reference for more +details. Args: data (pandas.DataFrame): Time series. @@ -11,13 +11,13 @@ Args: "diamond", "hexagonal", "random"). spacing (int): Distance (in number of cells) between initial supercells' centers. - compactness (float): A compactness value. Larger values cause - clusters to be more compact/even (square). - padding (int): Distance (in pixels) from the image borders within - which no seeds are placed. + compactness (float): A compactness value. Larger values cause clusters + to be more compact/even (square). + padding (int): Distance (in pixels) from the image borders within which + no seeds are placed. Returns: - R: The segmentation function to be applied to a data cube. + R: Segmentation function to be used with `sits_segment`. Examples: from pysits import * diff --git a/pysits/docs/content/sits_som_evaluate_cluster.md b/pysits/docs/content/sits_som_evaluate_cluster.md index da98435..876413f 100644 --- a/pysits/docs/content/sits_som_evaluate_cluster.md +++ b/pysits/docs/content/sits_som_evaluate_cluster.md @@ -5,8 +5,7 @@ found by the SOM map. For each cluster, it provides the percentage of classes inside it. Args: - som_map (SITSFrame): A SOM map produced by the `sits_som_map()` - function. + som_map: A SOM map produced by the `sits_som_map()` function. Returns: SITSFrame: The purity for each cluster. diff --git a/pysits/docs/content/sits_som_map.md b/pysits/docs/content/sits_som_map.md index edf0f2f..2060d11 100644 --- a/pysits/docs/content/sits_som_map.md +++ b/pysits/docs/content/sits_som_map.md @@ -7,31 +7,30 @@ Args: data (SITSTimeSeriesModel): samples to be clustered. grid_xdim (int): X dimension of the SOM grid (default = 25). grid_ydim (int): Y dimension of the SOM grid. - alpha (float): Starting learning rate (decreases according to number - of iterations). + alpha (float): Starting learning rate (decreases according to number of + iterations). rlen (int): Number of iterations to produce the SOM. - distance (str): The type of similarity measure (distance). The - following similarity measurements are supported: `"euclidean"`, - `"dtw"`, and `"cosine"`. The default similarity measure is - `"dtw"`. For single-timestep samples (e.g. embeddings), all bands - are treated as one feature vector. `"dtw"` is not applicable and - falls back to `"euclidean"`, while `"cosine"` compares the angle - between the full embedding vectors. + distance (str): The type of similarity measure (distance). The following + similarity measurements are supported: `"euclidean"`, `"dtw"`, and + `"cosine"`. The default similarity measure is `"dtw"`. For single- + timestep samples (e.g. embeddings), all bands are treated as one + feature vector. `"dtw"` is not applicable and falls back to + `"euclidean"`, while `"cosine"` compares the angle between the full + embedding vectors. som_radius (float): Radius of SOM neighborhood. - mode (str): Type of learning algorithm. The following learning - algorithm are available: `"online"`, `"batch"`, and `"pbatch"`. - The default learning algorithm is `"online"`. + mode (str): Type of learning algorithm. The following learning algorithm + are available: `"online"`, `"batch"`, and `"pbatch"`. The default + learning algorithm is `"online"`. Returns: - SITStructureData: a structure with three members: (1) the samples, - with one additional column indicating to which neuron each sample has - been mapped; (2) the Kohonen map, used for plotting and cluster - quality measures; (3) the labelled neurons, where each class of each - neuron is associated to two values: (a) the prior probability that - this class belongs to a cluster based on the frequency of samples of - this class allocated to the neuron; (b) the posterior probability - that this class belongs to a cluster, using data for the neighbours - on the SOM map. + SITStructureData: produces a structure with three members: (1) the samples, + with one additional column indicating to which neuron each sample has been + mapped; (2) the Kohonen map, used for plotting and cluster quality + measures; (3) the labelled neurons, where each class of each neuron is + associated to two values: (a) the prior probability that this class belongs + to a cluster based on the frequency of samples of this class allocated to + the neuron; (b) the posterior probability that this class belongs to a + cluster, using data for the neighbours on the SOM map. Notes: `sits_som_map` creates a SOM map, where high-dimensional data is mapped diff --git a/pysits/docs/content/sits_som_remove_samples.md b/pysits/docs/content/sits_som_remove_samples.md new file mode 100644 index 0000000..7c61ff4 --- /dev/null +++ b/pysits/docs/content/sits_som_remove_samples.md @@ -0,0 +1,28 @@ +Evaluate cluster + +Remove samples from a given class inside a neuron of another class + +Args: + som_map: A SOM map produced by the `sits_som_map()` function. + som_eval: An evaluation produced by the `sits_som_evaluate_cluster()` + function. + class_cluster (str): Dominant class of a set of neurons. + class_remove (str): Class to be removed from the neurons of + `class_cluster`. + +Returns: + SITSTimeSeriesModel: A new set of samples with the desired class + neurons removed. + +Examples: + from pysits import * + + # create a som map + som_map = sits_som_map(samples_modis_ndvi) + # evaluate the som map and create clusters + som_eval = sits_som_evaluate_cluster(som_map) + # clean the samples + new_samples = sits_som_remove_samples( + som_map, som_eval, + "Pasture", "Cerrado" + ) diff --git a/pysits/docs/content/sits_ssl_lejepa.md b/pysits/docs/content/sits_ssl_lejepa.md index 116951a..5be147a 100644 --- a/pysits/docs/content/sits_ssl_lejepa.md +++ b/pysits/docs/content/sits_ssl_lejepa.md @@ -28,15 +28,15 @@ simpler and more theoretically grounded method with a single trade-off hyperparameter. Args: - samples (SITSTimeSeriesModel): Samples object. If `None` (default), + samples (SITSTimeSeriesModel): samples object. If `None` (default), returns a training function. If provided, triggers immediate training. embedding_dim (int): Dimensionality of the encoder embedding. Default: 64. proj_dim (int): Dimensionality of the projector head used only during pre-training. Default: 128. - lambda (float): Value in (0, 1). Trade-off between invariance - (`1 - lambda`) and SIGReg (`lambda`). Default: 0.02. + lambda (float): Trade-off in (0, 1) between invariance (`1 - lambda`) + and SIGReg (`lambda`). Default: 0.02. num_knots (int): Number of quadrature knots for the SIGReg characteristic function test. Default: 17. num_slices (int): Number of random projection directions for SIGReg. @@ -48,8 +48,8 @@ Args: `sits_tempcnn()`. epochs (int): Maximum number of training epochs. batch_size (int): Batch size for training. Default: 512. - validation_split (float): Value in (0, 1). Fraction of samples held out - for validation loss monitoring. + validation_split (float): Fraction in (0, 1) of samples held out for + validation loss monitoring. optimizer: A `torch` optimizer constructor (default: `torch::optim_adamw`). opt_hparams (dict): Optimizer hyperparameters. @@ -62,7 +62,7 @@ Args: Returns: R: If `samples = None`, a training function. If `samples` is provided, a - pretrained encoder. + pretrained encoder closure. Examples: from pysits import * diff --git a/pysits/docs/content/sits_ssl_mae.md b/pysits/docs/content/sits_ssl_mae.md index 7d772e0..3d15c39 100644 --- a/pysits/docs/content/sits_ssl_mae.md +++ b/pysits/docs/content/sits_ssl_mae.md @@ -35,34 +35,36 @@ When GPU execution is enabled in the environment, training may run on GPU via `luz` accelerators. Otherwise, it runs on CPU. Args: - samples (SITSTimeSeriesModel): Sample time series. If `None` (default), + samples (SITSTimeSeriesModel): Samples object. If `None` (default), returns a training function. If provided, triggers immediate - training. Base data samples (e.g., `sits_base`) are not supported. + training. Base data samples are not supported. embedding_dim (int): Dimensionality of the latent embedding produced by the encoder (Default: 32). - encoder_model (SITSMachineLearningMethod): Deep learning method that takes - time series as input and produces latent representations that are used - to compute the loss function (suggested options: `sits_tempcnn()`, - `sits_lighttae()`, `sits_resnet()`). Default: `sits_tempcnn()`. + encoder_model (SITSMachineLearningMethod): Deep learning method that + takes time series as input and produces latent representations that + are used to compute the loss function (suggested options: + `sits_tempcnn()`, `sits_lighttae()`, `sits_resnet()`). Default: + `sits_tempcnn()`. decoder_width (int): Width of the decoder MLP hidden layer. dropout_rate (float): Dropout rates (0,1) for the linear module of the decoder. - masking_method (str): Mask selection strategy. Options are `"random"` or - `"contiguous"`. + masking_method (str): Mask selection strategy. Options are `"random"` + or `"contiguous"`. mask_ratio (float): Fraction of timesteps to mask, in (0, 1). mask_value (float): Fill value used for masked timesteps. masked_bands (list[str]): Which bands to mask. If `None`, all bands are eligible for masking. epochs (int): Maximum number of training epochs. batch_size (int): Batch size used for training and validation. - validation_split (float): Fraction of samples held out for validation loss - monitoring, in (0, 1). - optimizer: A `torch` optimizer constructor, such as `torch::optim_adamw`. + validation_split (float): Fraction of samples held out for validation + loss monitoring, in (0, 1). + optimizer: A `torch` optimizer constructor, such as + `torch::optim_adamw`. opt_hparams (dict): Optimizer hyperparameters passed to `optimizer`. Common entries include `lr`, `eps`, and `weight_decay`. Only parameters supported by the chosen optimizer are accepted. - lr_decay_epochs (int): Step size (in epochs) for learning-rate decay when - using the step scheduler. + lr_decay_epochs (int): Step size (in epochs) for learning-rate decay + when using the step scheduler. lr_decay_rate (float): Multiplicative decay factor applied by the learning-rate scheduler. patience (int): Number of epochs without improvement in validation loss @@ -73,11 +75,12 @@ Args: seed (int): Random seed used to initialize Torch randomness. Returns: - R: If `samples = None`, a training function with signature + R: If `samples = None`, returns a training function with signature `function(samples)` that trains an MAE and returns a pretrained encoder (a - `torch` module). If `samples` is provided, the result of applying the - training function to `samples` (i.e., a pretrained encoder-ready model - object used by the pretraining pipeline). + `torch` module). + If `samples` is provided, returns the result of applying the training + function to `samples` (i.e., a pretrained encoder-ready model object used + by the pretraining pipeline). Examples: from pysits import * diff --git a/pysits/docs/content/sits_ssl_vicreg.md b/pysits/docs/content/sits_ssl_vicreg.md index b1b785d..4986383 100644 --- a/pysits/docs/content/sits_ssl_vicreg.md +++ b/pysits/docs/content/sits_ssl_vicreg.md @@ -8,22 +8,18 @@ coverage constraint, and each is resampled back to the original length. No labels are required. Both views are passed through a shared encoder and projector. The VICReg loss combines three objectives: -1. Invariance: MSE between the projected representations of the two views - -ce loss combines three objectives: -1. Invariance: MSE between the projected representations of the two views +1. Invariance: MSE between the projected representations of the two views \u2014 pulls paired embeddings together. 2. Variance: a hinge loss that keeps the standard deviation of each embedding - feature above a threshold of 1 across the batch - prevents informational collapse. + feature above a threshold of 1 across the batch \u2014 prevents informational + collapse. 3. Covariance: penalizes off-diagonal entries of the embedding covariance - matrix - decorrelates features. + matrix \u2014 decorrelates features. The function can be used in two ways: - If `samples` is provided, it trains immediately and returns an encoder-ready - model object (see Returns). -- If `samples` is `None`, it returns a training function that can be passed to - `sits_pre_train` or called later. + model object (see Value). +- If `samples = None`, it returns a training function with signature + `function(samples)` that can be passed to `sits_pre_train` or called later. The augmentation strategy creates two views of each sample using the resampling method of Saget et al. (2025). The original time series (length `T`) is @@ -36,30 +32,28 @@ while differing in fine-grained detail. Because the subsampling is random, views differ at every epoch. Args: - samples (SITSTimeSeriesModel): Samples object. If `None` (default), - returns a training function. If provided, triggers immediate - training. Base data samples are not supported. + samples (SITSTimeSeriesModel): A set of sample time series. If `None` + (default), returns a training function. If provided, triggers immediate + training. Base data samples (e.g., `sits_base`) are not supported. embedding_dim (int): Dimensionality of the encoder embedding (exported features). Default: 64. proj_dim (int): Dimensionality of the projector head used only during pre-training. Default: 128. - sim_coeff (float): Weight of the invariance (MSE) term in the VICReg - loss. Default: 25.0. - std_coeff (float): Weight of the variance (hinge) term in the VICReg - loss. Default: 25.0. + sim_coeff (float): Weight of the invariance (MSE) term in the VICReg loss. + Default: 25.0. + std_coeff (float): Weight of the variance (hinge) term in the VICReg loss. + Default: 25.0. cov_coeff (float): Weight of the covariance (off-diagonal) term in the VICReg loss. Default: 1.0. - encoder_model (SITSMachineLearningMethod): Deep learning method that - takes time series as input and produces latent representations that - are used to compute the loss function (suggested options: - `sits_tempcnn()`, `sits_lighttae()`, `sits_resnet()`). Default: - `sits_tempcnn()`. + encoder_model (SITSMachineLearningMethod): Deep learning method that takes + time series as input and produces latent representations that are used + to compute the loss function (suggested options: `sits_tempcnn()`, + `sits_lighttae()`, `sits_resnet()`). Default: `sits_tempcnn()`. epochs (int): Maximum number of training epochs. batch_size (int): Batch size for training. Default: 128. - validation_split (float): Fraction of samples held out for validation - loss monitoring, in (0, 1). - optimizer: A `torch` optimizer constructor (default: - `torch::optim_adamw`). + validation_split (float): Fraction in (0, 1) of samples held out for + validation loss monitoring. + optimizer: A `torch` optimizer constructor (default: `torch::optim_adamw`). opt_hparams (dict): Optimizer hyperparameters. Common entries: `lr`, `eps`, `weight_decay`. lr_decay_epochs (int): Step size (in epochs) for LR decay. @@ -71,9 +65,10 @@ Args: seed (int): Random seed for reproducibility. Returns: - R: If `samples` is `None`, a training function that trains a VICReg model - and returns a pretrained encoder. If `samples` is provided, the result of - applying the training function to `samples` directly. + R: If `samples = None`, a training function with signature + `function(samples)` that trains a VICReg model and returns a pretrained + encoder. If `samples` is provided, the result of applying the training + function to `samples` directly. Examples: from pysits import * diff --git a/pysits/docs/content/sits_stats.md b/pysits/docs/content/sits_stats.md index 68fae17..a5e259c 100644 --- a/pysits/docs/content/sits_stats.md +++ b/pysits/docs/content/sits_stats.md @@ -1,19 +1,18 @@ Obtain statistics for all sample bands Most machine learning algorithms require data to be normalized. This -applies to the "SVM" method and to all deep learning ones. To -normalize the predictors, it is necessary to extract the statistics -of each band of the samples. This function computes the 2 of the -distribution of each band of the samples. This values are used as -minimum and maximum values in the normalization operation performed -by the sits_pred_normalize() function. +applies to the "SVM" method and to all deep learning ones. To normalize +the predictors, it is necessary to extract the statistics of each band of +the samples. This function computes the 2 of the distribution of each band +of the samples. This values are used as minimum and maximum values in the +normalization operation performed by the sits_pred_normalize() function. Args: - samples (SITSTimeSeriesModel): Time series samples used as - training data. + samples (SITSTimeSeriesModel): Time series samples used as training + data. Returns: - SITStructureData: The training data statistics. + SITStructureData: The 2 training data statistics. Examples: from pysits import * diff --git a/pysits/docs/content/sits_stratified_sampling.md b/pysits/docs/content/sits_stratified_sampling.md index 1028f12..eb71bdf 100644 --- a/pysits/docs/content/sits_stratified_sampling.md +++ b/pysits/docs/content/sits_stratified_sampling.md @@ -1,7 +1,7 @@ Retrieval of sample locations by strata for a classified cube This function can be used in two ways: (a) When the parameter -"sampling_design" is available, it takes the cube with different labels +"sampling_design" is available, its takes the cube with different labels and a column for the sampling design table with a number of samples per class and allocates a set of locations for each class. (b) When the parameter "sampling_design" is not provided, the method @@ -12,20 +12,21 @@ the same value is used to retrieve the samples for all classes. Args: cube (SITSCubeModel): Classified cube. - sampling_design (pandas.DataFrame): Result of `sits_sampling_design`. + sampling_design (pandas.DataFrame): Result of sits_sampling_design. alloc (str): Allocation method chosen. samples_per_class (int | dict): Number of samples per class (in case - `sampling_design` is `None`). Either a single integer (applied to - all classes) or a `dict` keyed by class label. When keyed, keys - must be valid class labels of the cube but do not need to cover - all classes — only the listed classes will be sampled. - overhead (float): Additional percentage to account for border points. + sampling_design is None). Either a single integer (applied to + all classes) or a `dict` keyed by class label. When a `dict`, + keys must be valid class labels of the cube but do not need to + cover all classes \u2014 only the listed classes will be sampled. + overhead (float): Additional percentage to account for border + points. multicores (int): Number of cores that will be used to sample the images in parallel. memsize (int): Memory available for sampling. shp_file (str | pathlib.Path): Name of shapefile to be saved (optional). - progress (bool): Show progress bar? Default is `True`. + progress (bool): Show progress bar? Default is True. Returns: SITSFrameSF: Point object with required samples and label. diff --git a/pysits/docs/content/sits_svm.md b/pysits/docs/content/sits_svm.md index d15839a..4d5b256 100644 --- a/pysits/docs/content/sits_svm.md +++ b/pysits/docs/content/sits_svm.md @@ -11,20 +11,21 @@ Please refer to the documentation in that package for more details. Args: samples (SITSTimeSeriesModel): Time series with the training samples. - formula: Symbolic description of the model to be fit. (default: - sits_formula_linear). - scale (list[bool]): Indicates the variables to be scaled. - cachesize (float): Cache memory in MB (default = 1000). - kernel (str): Kernel used in training and predicting. options: - "linear", "polynomial", "radial", "sigmoid" (default: "radial"). + formula (SITSMachineLearningMethod): Symbolic description of the + model to be fit (default: sits_formula_linear). + scale (bool): Indicates whether the variables should be scaled. + cachesize (int): Cache memory in MB (default = 1000). + kernel (str): Kernel used in training and predicting. Options: + "linear", "polynomial", "radial", "sigmoid" (default: + "radial"). degree (int): Exponential of polynomial type kernel (default: 3). coef0 (float): Parameter needed for kernels of type polynomial and sigmoid (default: 0). cost (float): Cost of constraints violation (default: 10). - tolerance (float): Tolerance of termination criterion - (default: 0.001). - epsilon (float): Epsilon in the insensitive-loss function - (default: 0.1). + tolerance (float): Tolerance of termination criterion (default: + 0.001). + epsilon (float): Epsilon in the insensitive-loss function (default: + 0.1). cross (int): Number of cross validation folds applied to assess the quality of the model (default: 10). **kwargs (dict): Other parameters to be passed to e1071::svm diff --git a/pysits/docs/content/sits_tae.md b/pysits/docs/content/sits_tae.md index b384fe6..1e91322 100644 --- a/pysits/docs/content/sits_tae.md +++ b/pysits/docs/content/sits_tae.md @@ -42,8 +42,7 @@ Notes: This function is based on the paper by Vivien Garnot referenced below and code available on github at https://github.com/VSainteuf/pytorch-psetae. We also used the code made available by Maja Schneider in her work with - Marco K - https://github.com/maja601/RC2020-psetae. + Marco K\n https://github.com/maja601/RC2020-psetae. If you use this method, please cite Garnot's and Schneider's work. Examples: diff --git a/pysits/docs/content/sits_tempcnn.md b/pysits/docs/content/sits_tempcnn.md index 27e92df..93a9d19 100644 --- a/pysits/docs/content/sits_tempcnn.md +++ b/pysits/docs/content/sits_tempcnn.md @@ -6,32 +6,36 @@ as the number of perceptron layers. Args: samples (SITSTimeSeriesModel): Time series with the training samples. - samples_validation (SITSTimeSeriesModel): Time series with the validation - samples. If provided, the `validation_split` parameter is ignored. + samples_validation (SITSTimeSeriesModel): Time series with the + validation samples. If the `samples_validation` parameter is + provided, the `validation_split` parameter is ignored. cnn_layers (list[int]): Number of 1D convolutional filters per layer. cnn_kernels (list[int]): Size of the 1D convolutional kernels. cnn_dropout_rates (list[float]): Dropout rates for 1D convolutional filters. dense_layer_nodes (int): Number of nodes in the dense layer. - dense_layer_dropout_rate (float): Dropout rate (0,1) for the dense layer. + dense_layer_dropout_rate (float): Dropout rate (0,1) for the dense + layer. epochs (int): Number of iterations to train the model. batch_size (int): Number of samples per gradient update. validation_split (float): Fraction of training data to be used for validation. optimizer: Optimizer function to be used. - opt_hparams (dict): Hyperparameters for optimizer: lr : Learning rate of - the optimizer eps: Term added to the denominator to improve numerical - stability. weight_decay: L2 regularization + opt_hparams (dict): Hyperparameters for optimizer: lr : Learning rate + of the optimizer eps: Term added to the denominator to improve + numerical stability. weight_decay: L2 regularization. lr_decay_epochs (int): Number of epochs to reduce learning rate. lr_decay_rate (float): Decay factor for reducing learning rate. - patience (int): Number of epochs without improvements until training stops. + patience (int): Number of epochs without improvements until training + stops. min_delta (float): Minimum improvement in loss function to reset the patience counter. seed (int): Seed for random values. - verbose (bool): Verbosity mode. Default is `False`. + verbose (bool): Verbosity mode. Default is False. Returns: - R: A fitted model to be used for classification. + SITSMachineLearningMethod: A fitted model to be used for + classification. Notes: `sits` provides a set of default values for all classification models. diff --git a/pysits/docs/content/sits_texture.md b/pysits/docs/content/sits_texture.md index 02a2b99..0f1c5cf 100644 --- a/pysits/docs/content/sits_texture.md +++ b/pysits/docs/content/sits_texture.md @@ -18,19 +18,21 @@ on. If more than one angle is provided, we compute their average. Args: cube (SITSCubeModel): Valid data cube. - window_size (int): An odd number representing the size of the sliding - window. - angles (list[float]): The direction angles in radians related to the - central pixel and its neighbor (See details). Default is 0. + window_size (int): An odd number representing the size of the + sliding window. + angles (float | list[float]): The direction angles in radians + related to the central pixel and its neighbor (see details). + Default is 0. memsize (int): Memory available for classification (in GB). multicores (int): Number of cores to be used for classification. - output_dir (str | pathlib.Path): Directory where files will be saved. + output_dir (str | pathlib.Path): Directory where files will be + saved. progress (bool): Show progress bar? **kwargs (dict): GLCM function (see details). Returns: - SITSCubeModel: A data cube with new bands, produced according to the - requested measure. + SITSCubeModel: A data cube with new bands, produced according to + the requested measure. Examples: from pysits import * diff --git a/pysits/docs/content/sits_tiles_to_roi.md b/pysits/docs/content/sits_tiles_to_roi.md index acb4398..83d1b1d 100644 --- a/pysits/docs/content/sits_tiles_to_roi.md +++ b/pysits/docs/content/sits_tiles_to_roi.md @@ -5,10 +5,10 @@ Takes a list of tiles from a given grid system and produces a ROI Args: tiles (list[str]): Names of tiles from the selected `grid_system`. - grid_system (str): Grid system that the `tiles` belong to. - Currently supported grid systems are the MGRS grid - (`"MGRS"`, default) and those used by the Brazil Data Cube - (`"BDC_LG_V2"`, `"BDC_MD_V2"` and `"BDC_SM_V2"`). + grid_system (str): Grid system that the `tiles` belong to. Currently + supported grid systems are the MGRS grid (`"MGRS"`, default) and + those used by the Brazil Data Cube (`"BDC_LG_V2"`, `"BDC_MD_V2"` + and `"BDC_SM_V2"`). Returns: SITSNamedVector: Valid ROI to use in other SITS functions. diff --git a/pysits/docs/content/sits_timeline.md b/pysits/docs/content/sits_timeline.md index 3f598e2..2d18ef1 100644 --- a/pysits/docs/content/sits_timeline.md +++ b/pysits/docs/content/sits_timeline.md @@ -1,10 +1,11 @@ Get timeline of a cube or a set of time series -This function returns the timeline for a given data set, either a set of -time series, a data cube, or a trained model. +This function returns the timeline for a given data set, either a set +of time series, a data cube, or a trained model. Args: - data (SITSTimeSeriesModel | SITSCubeModel): time series or data cube. + data (SITSTimeSeriesModel | SITSCubeModel): a set of time series or + a data cube. Returns: list: timeline of samples or data cube. diff --git a/pysits/docs/content/sits_timeseries_to_csv.md b/pysits/docs/content/sits_timeseries_to_csv.md new file mode 100644 index 0000000..c911ffd --- /dev/null +++ b/pysits/docs/content/sits_timeseries_to_csv.md @@ -0,0 +1,23 @@ +Export a full sits time series to the CSV format + +Converts metadata and data from a set of time series to a CSV file. +The CSV file will not contain the actual time series. Its columns will +be the same as those of a CSV file used to retrieve data from ground +information ("latitude", "longitude", "start_date", "end_date", +"cube", "label"), plus all the time series for each data + +Args: + data (SITSTimeSeriesModel): Time series. + file (str | pathlib.Path): Full path of the exported CSV file (valid + file name with extension ".csv"). + +Returns: + None + +Examples: + from pysits import * + import tempfile + + csv_ts = sits_timeseries_to_csv(cerrado_2classes) + csv_file = tempfile.gettempdir() + "/cerrado_2classes_ts.csv" + sits_timeseries_to_csv(cerrado_2classes, file=csv_file) diff --git a/pysits/docs/content/sits_to_csv.md b/pysits/docs/content/sits_to_csv.md index 9f00b62..02b7795 100644 --- a/pysits/docs/content/sits_to_csv.md +++ b/pysits/docs/content/sits_to_csv.md @@ -1,10 +1,10 @@ Export sits time series metadata to the CSV format -Converts metadata from a `SITSTimeSeriesModel` to a CSV file. The CSV -file will not contain the actual time series. Its columns will be the -same as those of a CSV file used to retrieve data from ground -information ("latitude", "longitude", "start_date", "end_date", -"cube", "label"). If the file is `None`, returns a `pandas.DataFrame`. +Converts metadata from a set of time series to a CSV file. The CSV file +will not contain the actual time series. Its columns will be the same as +those of a CSV file used to retrieve data from ground information +("latitude", "longitude", "start_date", "end_date", "cube", "label"). +If the file is `None`, returns a `SITSFrame` as an object. Args: data (SITSTimeSeriesModel): Time series. @@ -15,9 +15,8 @@ Returns: SITSFrame: Data with CSV columns (optional). Examples: - from pysits import * import tempfile - import os + from pysits import * - csv_file = os.path.join(tempfile.gettempdir(), "cerrado_2classes.csv") + csv_file = tempfile.gettempdir() + "/cerrado_2classes.csv" sits_to_csv(cerrado_2classes, file=csv_file) diff --git a/pysits/docs/content/sits_to_xlsx.md b/pysits/docs/content/sits_to_xlsx.md index 7031bac..b481706 100644 --- a/pysits/docs/content/sits_to_xlsx.md +++ b/pysits/docs/content/sits_to_xlsx.md @@ -1,18 +1,17 @@ Save accuracy assessments as Excel files Saves confusion matrices as Excel spreadsheets. This function takes a -list of accuracy assessments generated by `sits_accuracy` and saves -them in an Excel spreadsheet. +list of accuracy assessments generated by `sits_accuracy` and saves them +in an Excel spreadsheet. Args: acc (SITSConfusionMatrix | list[SITSConfusionMatrix]): Accuracy statistics, either an output of `sits_accuracy` or a list of those. - file (str | pathlib.Path): File where the XLSX data is to be - saved. + file (str | pathlib.Path): File where the XLSX data is to be saved. Returns: - None: Called for side effects. + None: Called for its side effects. Examples: from pysits import * @@ -24,7 +23,8 @@ Examples: results = [] # accuracy assessment lightTAE - acc_ltae = sits_kfold_validate(samples_modis_ndvi, + acc_ltae = sits_kfold_validate( + samples_modis_ndvi, folds=5, multicores=1, ml_method=sits_lighttae() @@ -38,5 +38,5 @@ Examples: # save to xlsx file sits_to_xlsx( results, - file=tempfile.NamedTemporaryFile(prefix="accuracy_mato_grosso_dl_", suffix=".xlsx").name + file=tempfile.mktemp(prefix="accuracy_mato_grosso_dl_", suffix=".xlsx") ) diff --git a/pysits/docs/content/sits_tuning.md b/pysits/docs/content/sits_tuning.md index 25eabe8..01d8b73 100644 --- a/pysits/docs/content/sits_tuning.md +++ b/pysits/docs/content/sits_tuning.md @@ -1,8 +1,8 @@ Tuning machine learning models hyper-parameters This function performs a random search on values of selected hyperparameters, -and produces a data frame with the accuracy and kappa values produced by a -validation procedure. The result allows users to select appropriate +and produces a `pandas.DataFrame` with the accuracy and kappa values produced +by a validation procedure. The result allows users to select appropriate hyperparameters for deep learning models. Args: diff --git a/pysits/docs/content/sits_tuning_hparams.md b/pysits/docs/content/sits_tuning_hparams.md index e184ae0..b6aef96 100644 --- a/pysits/docs/content/sits_tuning_hparams.md +++ b/pysits/docs/content/sits_tuning_hparams.md @@ -6,8 +6,8 @@ Users should pass the possible values for hyper-parameters as constants or by calling the following random functions: - `uniform(min = 0, max = 1, n = 1)`: returns random numbers from a uniform distribution with parameters min and max. -- `choice(..., replace = TRUE, n = 1)`: returns random objects passed to `...` - with replacement or not (parameter `replace`). +- `choice(..., replace = True, n = 1)`: returns random objects passed to + `**kwargs` with replacement or not (parameter `replace`). - `randint(min, max, n = 1)`: returns random integers from a uniform distribution with parameters min and max. - `normal(mean = 0, sd = 1, n = 1)`: returns random numbers from a normal diff --git a/pysits/docs/content/sits_uncertainty_sampling.md b/pysits/docs/content/sits_uncertainty_sampling.md index b0e80f5..bc276d0 100644 --- a/pysits/docs/content/sits_uncertainty_sampling.md +++ b/pysits/docs/content/sits_uncertainty_sampling.md @@ -24,17 +24,19 @@ Args: Default is Inf (no upper limit). sampling_window (int): Window size for collecting points (in pixels). The minimum window size is 10. - multicores (int): Number of workers for parallel processing (min = 1, - max = 2048). + multicores (int): Number of workers for parallel processing + (min = 1, max = 2048). memsize (int): Maximum overall memory (in GB) to run the function. progress (bool): Whether to show progress bars. + **kwargs (dict): Additional arguments. Returns: - SITSFrame: Longitude and latitude in WGS84 with locations which have + SITSFrame: Locations with longitude and latitude in WGS84 which have high uncertainty and meet the minimum distance criteria. Examples: from pysits import * + import tempfile # create a data cube data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") @@ -47,12 +49,12 @@ Examples: rfor_model = sits_train(samples_modis_ndvi, ml_method=sits_rfor()) # classify the cube probs_cube = sits_classify( - data=cube, ml_model=rfor_model, output_dir=tempdir() + data=cube, ml_model=rfor_model, output_dir=tempfile.mkdtemp() ) # create an uncertainty cube uncert_cube = sits_uncertainty(probs_cube, type="entropy", - output_dir=tempdir() + output_dir=tempfile.mkdtemp() ) # obtain a new set of samples for active learning # the samples are located in uncertain places diff --git a/pysits/docs/content/sits_validate.md b/pysits/docs/content/sits_validate.md index e80ba5f..29f176e 100644 --- a/pysits/docs/content/sits_validate.md +++ b/pysits/docs/content/sits_validate.md @@ -13,10 +13,10 @@ This function returns the confusion matrix, and Kappa values. Args: samples (SITSTimeSeriesModel): Time series to be validated. - samples_validation (SITSTimeSeriesModel): Optional time series used for + samples_validation (SITSTimeSeriesModel): Optional; time series used for validation. - validation_split (float): Percent of original time series set to be - used for validation if `samples_validation` is `None`. + validation_split (float): Percent of original time series set to be used + for validation if `samples_validation` is `None`. ml_method (SITSMachineLearningMethod): Machine learning method. gpu_memory (int): Memory available in GPU in GB (default = 4). batch_size (int): Batch size for GPU classification. diff --git a/pysits/docs/content/sits_variance.md b/pysits/docs/content/sits_variance.md index abfd7c0..9f6a9f4 100644 --- a/pysits/docs/content/sits_variance.md +++ b/pysits/docs/content/sits_variance.md @@ -1,16 +1,16 @@ Calculate the variance of a probability cube -Takes a probability data cube (either a raster or a segmented -vector cube) and estimates the variance of the logit of the -probability. For a standard raster cube, this is a local sliding-window -variance. For a vector/segmented cube, it calculates the variance of all -pixels inside each segment. This supports the choice of parameters for -Bayesian smoothing. +Takes a probability cube (either a raster or a segmented probability +cube) and estimates the variance of the logit of the probability. For a +standard raster cube, this is a local sliding-window variance. For a +vector/segmented cube, it calculates the variance of all pixels inside +each segment. This supports the choice of parameters for Bayesian +smoothing. Args: cube (SITSCubeModel): Probability data cube. window_size (int): Size of the neighborhood (odd integer). Not used - for a segmented vector cube. + for segmented probability cubes. neigh_fraction (float): Fraction of neighbors with highest probability for Bayesian inference (from 0.0 to 1.0). memsize (int): Maximum overall memory (in GB) to run the smoothing diff --git a/pysits/docs/content/sits_view.md b/pysits/docs/content/sits_view.md index db53524..dc36b76 100644 --- a/pysits/docs/content/sits_view.md +++ b/pysits/docs/content/sits_view.md @@ -3,9 +3,9 @@ View data cubes and samples in leaflet Uses leaflet to visualize time series, raster cube and classified images. Args: - x (SITSTimeSeriesModel | SITSCubeModel | pandas.DataFrame): time - series samples, SOM map, raster cube, probability cube, vector - cube, or classified cube. + x (SITSTimeSeriesModel | pandas.DataFrame | SITSCubeModel): time + series, SOM map, raster cube, probability cube, vector cube, or + classified cube to be visualized. legend (dict): associates labels to colors. palette (str): color palette from RColorBrewer. radius (float): radius of circle markers. @@ -15,27 +15,24 @@ Args: red (str): band for red color. green (str): band for green color. blue (str): band for blue color. - tiles (list[str]): tiles to be plotted (in case of a multi-tile - cube). + tiles (list[str]): tiles to be plotted (in case of a multi-tile cube). dates (list[str]): dates to be plotted. rev (bool): revert color palette? - opacity (float): opacity of segment fill or class cube. - max_cog_size (int): maximum size of COG overviews (lines or - columns). + opacity (float): opacity of segment fill or classified cube. + max_cog_size (int): maximum size of COG overviews (lines or columns). first_quantile (float): first quantile for stretching images. last_quantile (float): last quantile for stretching images. leaflet_megabytes (float): maximum size for leaflet (in MB). version (str): version name (to compare different classifications). - labels (list[str]): labels to be plotted (in case of probs and + labels (list[str]): labels to be plotted (in case of probability and variance cubes). seg_color (str): color for segment boundaries. line_width (float): line width for segments (in pixels). **kwargs (dict): further specifications for sits_view. Returns: - None: a leaflet object containing either samples or data cubes - embedded in a global map that can be visualized directly in a - viewer. + None: a leaflet object containing either samples or data cubes embedded + in a global map that can be visualized directly in a viewer. Notes: To show a false color image, use "band" to chose one of the bands, "tiles" diff --git a/pysits/docs/content/sits_xgboost.md b/pysits/docs/content/sits_xgboost.md index baee0fc..4e2d9d3 100644 --- a/pysits/docs/content/sits_xgboost.md +++ b/pysits/docs/content/sits_xgboost.md @@ -15,22 +15,22 @@ Args: partition of a leaf. Default: 1. max_depth (int): Maximum depth of a tree. Increasing this value makes the model more complex and more likely to overfit. Default: 5. - min_child_weight (float): If the leaf node has a minimum sum of - instance weights lower than min_child_weight, tree splitting stops. - The larger min_child_weight is, the more conservative the algorithm - is. Default: 1. + min_child_weight (float): If the leaf node has a minimum sum of instance + weights lower than min_child_weight, tree splitting stops. The + larger min_child_weight is, the more conservative the algorithm is. + Default: 1. max_delta_step (float): Maximum delta step we allow each leaf output to be. If the value is set to 0, there is no constraint. If it is set to a positive value, it can help making the update step more conservative. Default: 1. - subsample (float): Percentage of samples supplied to a tree. Default: - 0.8. + subsample (float): Percentage of samples supplied to a tree. + Default: 0.8. nfold (int): Number of the subsamples for the cross-validation. nrounds (int): Number of rounds to iterate the cross-validation (default: 100) nthread (int): Number of threads (default = 6) - early_stopping_rounds (int): Training with a validation set will stop - if the performance doesn't improve for k rounds. + early_stopping_rounds (int): Training with a validation set will stop if + the performance doesn't improve for k rounds. verbose (bool): Print information on statistics during the process Returns: diff --git a/pysits/docs/content/summary.md b/pysits/docs/content/summary.md index 64ba928..e5f9a22 100644 --- a/pysits/docs/content/summary.md +++ b/pysits/docs/content/summary.md @@ -1,36 +1,42 @@ -Summarize sits objects and data cubes. +Summarize data cubes, time series, and accuracy objects. -This is a generic function that produces a summary of the input object. -The behavior and available parameters depend on the specific type of -object provided. It can summarize time series, raster cubes, classified -cubes, variance cubes, and accuracy assessments. +This is a generic function. The parameters and the returned summary +depend on the specific type of input object. It handles classified +cubes, raster cubes, variance cubes, time series, and accuracy +assessment objects. Args: - object (SITSTimeSeriesModel | SITSCubeModel | SITSConfusionMatrix): - The object to be summarized. Supported types include: a set - of time series; a raster data cube; a classified data cube; - a variance data cube; an accuracy matrix for training data; - or an accuracy matrix for area data. - **kwargs (dict): Further specifications for the summary. The - accepted keyword arguments depend on the type of `object`: + object (SITSCubeModel | SITSTimeSeriesModel | SITSConfusionMatrix): + The object to be summarized. Supported inputs include a + classified cube, a raster cube, a variance cube, a set of + time series, a sample accuracy object, or an area accuracy + object. + tile (str): (raster cubes only) Tile to be summarized. + date (str): (raster cubes only) Date to be summarized. + intervals (float): (variance cubes only) Intervals to calculate the + quantiles. + sample_size (int): (variance cubes only) The approximate size of + samples to be extracted from the variance cube (by tile). + multicores (int): (variance cubes only) Number of cores to summarize + data (min = 1, max = 2048). + memsize (int): (variance cubes only) Memory in GB available to + summarize data (min = 1, max = 16384). + quantiles (list[str]): (variance cubes only) Quantiles to be shown. + **kwargs (dict): Further specifications for the summary, depending on + the input type. - For a raster data cube: - tile: Tile to be summarized. - date: Date to be summarized. +Returns: + str: A summary appropriate to the input type: a summary of a + classified cube, a data cube, a variance cube, a set of time + series, or a sample/area accuracy assessment. - For a variance data cube: - intervals: Intervals to calculate the quantiles. - sample_size: The approximate size of samples that - will be extracted from the variance cube (by - tile). - multicores: Number of cores to summarize data - (min = 1, max = 2048). - memsize: Memory in GB available to summarize data - (min = 1, max = 16384). - quantiles: Quantiles to be shown. +Examples: + from pysits import * -Returns: - str: A summary of the input object. Depending on the input type, - this is a summary of the time series, the raster data cube, the - classified data cube, the variance data cube, or the - sample/area accuracy. + # Summarize a sits time series tibble + summary(samples_modis_ndvi) + + # Train a model and produce an accuracy assessment, then summarize it + rfor_model = sits_train(samples_modis_ndvi, ml_method=sits_rfor()) + point_class = sits_classify(point_mt_6bands, ml_method=rfor_model) + summary(point_class) diff --git a/pysits/models/data/frame.py b/pysits/models/data/frame.py index 28d51da..2524f24 100644 --- a/pysits/models/data/frame.py +++ b/pysits/models/data/frame.py @@ -22,6 +22,7 @@ from rpy2.robjects.vectors import DataFrame as RDataFrame from pysits.conversions.tibble import ( + geopandas_to_tibble, pandas_to_tibble, tibble_nested_to_pandas, tibble_to_pandas, @@ -116,6 +117,21 @@ def __init__(self, instance, **kwargs): # Initialize super class GeoPandasDataFrame.__init__(self, data=instance, **kwargs) + # + # Data management + # + def _sync_instance(self): + """Sync instance with R.""" + if not self._is_updated: + return + + # ``sf`` objects require a geometry-based conversion. Otherwise the + # spatial classes are lost and R functions dispatch as plain data frames + self._instance = geopandas_to_tibble(self) + + # Update flag + self._is_updated = False + class SITSFrameNested(SITSFrame): """General class for sits frame with embedded data frames.""" diff --git a/pysits/sits/classification.py b/pysits/sits/classification.py deleted file mode 100644 index ddc00e7..0000000 --- a/pysits/sits/classification.py +++ /dev/null @@ -1,42 +0,0 @@ -# -# Copyright (C) 2025 sits developers. -# -# This program is free software; you can redistribute it and/or modify it -# under the terms of the GNU General Public License as published by -# the Free Software Foundation; either version 2 of the License, or -# (at your option) any later version. -# -# This program is distributed in the hope that it will be useful, -# but WITHOUT ANY WARRANTY; without even the implied warranty of -# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -# GNU General Public License for more details. -# -# You should have received a copy of the GNU General Public License -# along with this program; if not, see . -# - -"""Classification operations.""" - -from pysits.backend.pkgs import r_pkg_sits -from pysits.conversions.decorators import function_call -from pysits.docs import attach_doc -from pysits.models.data.cube import SITSCubeModel -from pysits.models.resolver import resolve_and_invoke_content_class - - -@function_call(r_pkg_sits.sits_classify, resolve_and_invoke_content_class) -@attach_doc("sits_classify") -def sits_classify(*args, **kwargs) -> SITSCubeModel: - """Classify data.""" - - -@function_call(r_pkg_sits.sits_smooth, SITSCubeModel) -@attach_doc("sits_smooth") -def sits_smooth(*args, **kwargs) -> SITSCubeModel: - """Smooth classification data.""" - - -@function_call(r_pkg_sits.sits_label_classification, SITSCubeModel) -@attach_doc("sits_label_classification") -def sits_label_classification(*args, **kwargs) -> SITSCubeModel: - """Label probabilities data.""" diff --git a/pysits/sits/data.py b/pysits/sits/data.py index 1a4c9b5..36e5ae1 100644 --- a/pysits/sits/data.py +++ b/pysits/sits/data.py @@ -15,7 +15,7 @@ # along with this program; if not, see . # -"""Data management operations.""" +"""Data operations.""" from datetime import date @@ -28,7 +28,9 @@ from pysits.conversions.decorators import function_call, rpy2_fix_type from pysits.docs import attach_doc from pysits.models.data.base import SITSData +from pysits.models.data.cube import SITSCubeModel from pysits.models.data.frame import SITSFrame +from pysits.models.data.ts import SITSTimeSeriesModel from pysits.models.resolver import ( resolve_and_invoke_accuracy_class, resolve_and_invoke_content_class, @@ -111,6 +113,33 @@ def sits_accuracy_summary(*args, **kwargs) -> SITSData: """Print accuracy summary.""" +@function_call(r_pkg_sits.sits_encode, resolve_and_invoke_content_class) +@attach_doc("sits_encode") +def sits_encode(*args, **kwargs) -> SITSTimeSeriesModel | SITSCubeModel: + """Encode time series or data cubes.""" + + +# +# Classification +# +@function_call(r_pkg_sits.sits_classify, resolve_and_invoke_content_class) +@attach_doc("sits_classify") +def sits_classify(*args, **kwargs) -> SITSCubeModel | SITSTimeSeriesModel: + """Classify data.""" + + +@function_call(r_pkg_sits.sits_smooth, SITSCubeModel) +@attach_doc("sits_smooth") +def sits_smooth(*args, **kwargs) -> SITSCubeModel: + """Smooth classification data.""" + + +@function_call(r_pkg_sits.sits_label_classification, SITSCubeModel) +@attach_doc("sits_label_classification") +def sits_label_classification(*args, **kwargs) -> SITSCubeModel: + """Label probabilities data.""" + + # # Apply operation # diff --git a/pysits/sits/exporters/__init__.py b/pysits/sits/exporters/__init__.py index fc299bf..bd35dfe 100644 --- a/pysits/sits/exporters/__init__.py +++ b/pysits/sits/exporters/__init__.py @@ -17,7 +17,7 @@ """Exporters module.""" -from .files import sits_to_csv, sits_to_xlsx +from .files import sits_timeseries_to_csv, sits_to_csv, sits_to_xlsx from .sf import sits_as_geopandas from .xarray import sits_as_xarray @@ -26,4 +26,5 @@ "sits_as_xarray", "sits_to_csv", "sits_to_xlsx", + "sits_timeseries_to_csv", ) diff --git a/pysits/sits/exporters/files.py b/pysits/sits/exporters/files.py index 77a01f7..552f097 100644 --- a/pysits/sits/exporters/files.py +++ b/pysits/sits/exporters/files.py @@ -35,3 +35,9 @@ def sits_to_csv(*args, **kwargs) -> SITSFrame: def sits_to_xlsx(*args, **kwargs) -> None: """Save accuracy assessments as Excel files.""" ... + + +@function_call(r_pkg_sits.sits_timeseries_to_csv, lambda x: None) +@attach_doc("sits_timeseries_to_csv") +def sits_timeseries_to_csv(*args, **kwargs) -> None: + """Export a full sits timeseries to CSV format.""" diff --git a/pysits/sits/ml.py b/pysits/sits/ml.py index 1bb2f3b..514a92e 100644 --- a/pysits/sits/ml.py +++ b/pysits/sits/ml.py @@ -57,6 +57,7 @@ def convert_opt_hparams(obj: Any) -> Any: "opt_hparams": convert_opt_hparams, } + # # DL Methods # @@ -65,6 +66,7 @@ def convert_opt_hparams(obj: Any) -> Any: sits_lighttae = closure_factory("sits_lighttae", converters=dl_converters) sits_mlp = closure_factory("sits_mlp", converters=dl_converters) sits_resnet = closure_factory("sits_resnet", converters=dl_converters) +sits_lstm_fcn = closure_factory("sits_lstm_fcn", converters=dl_converters) # @@ -77,14 +79,16 @@ def convert_opt_hparams(obj: Any) -> Any: # -# Encoder Methods +# Representation Learning methods # -sits_ssl_mae = closure_factory("sits_ssl_mae") -sits_ssl_lejepa = closure_factory("sits_ssl_lejepa") -sits_ssl_vicreg = closure_factory("sits_ssl_vicreg") +sits_ssl_mae = closure_factory("sits_ssl_mae", converters=dl_converters) +sits_ssl_lejepa = closure_factory("sits_ssl_lejepa", converters=dl_converters) +sits_ssl_vicreg = closure_factory("sits_ssl_vicreg", converters=dl_converters) -sits_barlow_twins = closure_factory("sits_barlow_twins") -sits_contrastive_learning = closure_factory("sits_contrastive_learning") +sits_barlow_twins = closure_factory("sits_barlow_twins", converters=dl_converters) +sits_contrastive_learning = closure_factory( + "sits_contrastive_learning", converters=dl_converters +) # diff --git a/pysits/sits/ts.py b/pysits/sits/ts.py index 660a4ab..b1c593e 100644 --- a/pysits/sits/ts.py +++ b/pysits/sits/ts.py @@ -32,12 +32,6 @@ # # High-level operation # -@function_call(r_pkg_sits.sits_encode, SITSTimeSeriesModel) -@attach_doc("sits_encode") -def sits_encode(*args, **kwargs) -> SITSTimeSeriesModel: - """Encode time series or data cubes.""" - - @function_call(r_pkg_sits.sits_get_data, SITSTimeSeriesModel) @attach_doc("sits_get_data") def sits_get_data(*args, **kwargs) -> SITSTimeSeriesModel: @@ -66,6 +60,12 @@ def sits_stats(*args, **kwargs) -> SITStructureData: """ +@function_call(r_pkg_sits.sits_random_sampling, SITSFrameSF) +@attach_doc("sits_random_sampling") +def sits_random_sampling(*args, **kwargs) -> SITSFrameSF: + """Sampling random points in a data cube.""" + + # # Validation # @@ -164,6 +164,12 @@ def sits_som_clean_samples(*args, **kwargs) -> SITSTimeSeriesModel: """Cleans the samples based on SOM map information.""" +@function_call(r_pkg_sits.sits_som_remove_samples, SITSTimeSeriesModel) +@attach_doc("sits_som_remove_samples") +def sits_som_remove_samples(*args, **kwargs) -> SITSTimeSeriesModel: + """Remove samples confused with a different class in the SOM map.""" + + # # Dendrogram # diff --git a/pysits/sits/visualization.py b/pysits/sits/visualization.py index 77f1c88..0b7b628 100644 --- a/pysits/sits/visualization.py +++ b/pysits/sits/visualization.py @@ -127,3 +127,12 @@ def _(data: SITSMachineLearningMethod, **kwargs) -> None: def _(data: SITSRepresentationLearningMethod, **kwargs) -> None: """Plot representation learning method.""" return plot_base(data, **kwargs) + + +# +# Sankey plot +# +@rpy2_fix_type +def sits_sankey(*args, **kwargs) -> None: + """Plot class trajectories from multi-temporal classified cubes.""" + return plot_base(r_pkg_sits.sits_sankey(*args, **kwargs)) diff --git a/pysits/visualization/base.py b/pysits/visualization/base.py index 7025517..4821c7a 100644 --- a/pysits/visualization/base.py +++ b/pysits/visualization/base.py @@ -80,7 +80,14 @@ def _base_plot( # Handle plots if multiple: - for i, plot in enumerate(plots): + # A single plot object (e.g., a ``ggplot``) is not iterable + try: + figure_groups = list(plots) + + except TypeError: + figure_groups = [plots] + + for i, plot in enumerate(figure_groups): try: figures = list(plot) diff --git a/tests/conftest.py b/tests/conftest.py index 66924f2..f10cd88 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -18,6 +18,7 @@ """Pytest configuration file.""" import webbrowser +from typing import Any import matplotlib @@ -31,13 +32,59 @@ # import matplotlib.pyplot as plt import pytest +import rpy2.robjects as ro + +from pysits.models.data.cube import SITSCubeModel +from pysits.sits.cube import sits_cube +from pysits.sits.utils import r_package_dir + + +# +# Helpers +# +def r_closure_value(method: Any, name: str) -> Any: + """Read a value stored in the R closure of a ml/dl method.""" + return ro.r["environment"](method)[name] + + +def r_opt_hparams(method: Any) -> dict[str, float]: + """Read the optimizer hyperparameters of a ml/dl method.""" + hparams = r_closure_value(method, "opt_hparams") + + return {k: v[0] for k, v in zip(hparams.names, hparams, strict=True)} + + +def r_class(obj: Any) -> list[str]: + """Read the classes of an R object.""" + return list(ro.r["class"](obj)) + + +def r_identical(x: Any, y: Any) -> bool: + """Check if two R objects are identical.""" + return bool(ro.r["identical"](x, y)[0]) + + +# +# Fixtures +# +@pytest.fixture(scope="module") +def local_cube() -> SITSCubeModel: + """Cube created from local files.""" + return sits_cube( + source="BDC", + collection="MOD13Q1-6.1", + data_dir=r_package_dir("extdata/raster/mod13q1", package="sits"), + progress=False, + ) @pytest.fixture def no_plot_window(monkeypatch): """Fixture to prevent matplotlib plot windows from showing during tests.""" monkeypatch.setattr(plt, "show", lambda: None) + yield + plt.close("all") @@ -47,4 +94,5 @@ def no_browser(monkeypatch): monkeypatch.setattr(webbrowser, "open", lambda x: None) monkeypatch.setattr(webbrowser, "open_new", lambda x: None) monkeypatch.setattr(webbrowser, "open_new_tab", lambda x: None) + yield diff --git a/tests/test_classification.py b/tests/test_classification.py index 418b903..aac21e1 100644 --- a/tests/test_classification.py +++ b/tests/test_classification.py @@ -19,33 +19,32 @@ from pathlib import Path -from pysits.sits.classification import ( +from pysits.sits.context import point_mt_6bands, samples_modis_ndvi +from pysits.sits.data import ( sits_classify, sits_label_classification, + sits_labels, + sits_select, sits_smooth, ) -from pysits.sits.context import point_mt_6bands, samples_modis_ndvi -from pysits.sits.cube import sits_cube -from pysits.sits.data import sits_labels, sits_select from pysits.sits.ml import sits_rfor, sits_train -from pysits.sits.utils import r_package_dir, r_set_seed +from pysits.sits.utils import r_set_seed from pysits.sits.visualization import sits_plot -def test_basic_cube_classification(tmp_path: Path, no_plot_window): +def test_basic_cube_classification(tmp_path: Path, local_cube, no_plot_window): """Test basic classification.""" - # Set seed r_set_seed(42) # Train a random forest model rfor_model = sits_train(samples_modis_ndvi, sits_rfor()) - # Create a data cube from local files - data_dir = r_package_dir("extdata/raster/mod13q1", package="sits") - cube = sits_cube(source="BDC", collection="MOD13Q1-6.1", data_dir=data_dir) - # Classify a data cube - probs_cube = sits_classify(data=cube, ml_model=rfor_model, output_dir=tmp_path) + probs_cube = sits_classify( + data=local_cube, + ml_model=rfor_model, + output_dir=tmp_path, + ) # Smooth the probability cube using Bayesian statistics bayes_cube = sits_smooth(probs_cube, output_dir=tmp_path) @@ -74,8 +73,8 @@ def test_basic_ts_classification(tmp_path: Path, no_plot_window): point_ndvi = sits_select(point_mt_6bands, bands=("NDVI")) point_class = sits_classify(data=point_ndvi, ml_model=rf_model) - assert point_class.shape[0] == 1 # noqa: PLR2004 - number of points - assert len(sits_labels(point_class)) == 1 # noqa: PLR2004 - number of labels + assert point_class.shape[0] == 1 # noqa: PLR2004 + assert len(sits_labels(point_class)) == 1 # noqa: PLR2004 # Plot the result sits_plot(point_class) diff --git a/tests/test_cube.py b/tests/test_cube.py index 17bb8a1..86d167e 100644 --- a/tests/test_cube.py +++ b/tests/test_cube.py @@ -22,9 +22,11 @@ import pytest from pandas import DataFrame as PandasDataFrame +from pysits.conversions.common import convert_to_r from pysits.conversions.dsl.mask import MaskValue from pysits.models.data.cube import SITSCubeModel -from pysits.models.data.frame import SITSFrame +from pysits.models.data.frame import SITSFrame, SITSFrameSF +from pysits.models.data.ts import SITSTimeSeriesModel from pysits.sits.cube import ( convert_reclassify_rules, sits_cube, @@ -32,20 +34,10 @@ sits_texture, ) from pysits.sits.data import sits_bands, sits_bbox, sits_labels, sits_timeline +from pysits.sits.ts import sits_get_data, sits_random_sampling from pysits.sits.utils import r_package_dir -@pytest.fixture(scope="module") -def local_cube() -> SITSCubeModel: - """Cube created from local files.""" - return sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=r_package_dir("extdata/raster/mod13q1", package="sits"), - progress=False, - ) - - def test_sits_cube_data_structure(): """Test data structure of sits_cube.""" # Define a region of interest for the city of Sinop @@ -293,6 +285,59 @@ def test_cube_texture(tmp_path: Path, local_cube): assert sits_bands(cube_texture) == ["NDVI", "NDVIVAR"] +def test_cube_random_sampling(local_cube): + """Test random sampling of a data cube.""" + samples = sits_random_sampling( + cube=local_cube, + n_samples=100, + progress=False, + ) + + # Test type + assert isinstance(samples, SITSFrameSF) + + # Columns + assert samples.geometry.name == "geometry" + assert samples.crs is not None + + # Expected samples + assert samples.shape == (100, 1) # number of samples requested + assert set(samples.geometry.geom_type) == {"Point"} + + # The samples must be usable in R as an ``sf`` object + assert "sf" in list(convert_to_r(samples).rclass) + + +def test_cube_random_sampling_get_data(local_cube): + """Test time series extraction using random samples from a data cube.""" + samples = sits_random_sampling( + cube=local_cube, + n_samples=5, + progress=False, + ) + + # get sample time-series + samples_ts = sits_get_data( + cube=local_cube, + samples=samples, + multicores=1, + progress=False, + ) + + # Test type + assert isinstance(samples_ts, SITSTimeSeriesModel) + + # One time series is extracted for each sampled point + assert samples_ts.shape[0] == samples.shape[0] + + # Points without a label are extracted as ``NoClass`` + assert sits_labels(samples_ts) == ["NoClass"] + + # The cube bands and timeline are preserved + assert sits_bands(samples_ts) == sits_bands(local_cube) + assert len(sits_timeline(samples_ts)) == len(sits_timeline(local_cube)) + + def test_cube_sync_instance(local_cube): """Test sync of a modified cube with R.""" cube = local_cube.copy() diff --git a/tests/test_data.py b/tests/test_data.py index 66b87e9..570f909 100644 --- a/tests/test_data.py +++ b/tests/test_data.py @@ -35,7 +35,6 @@ sits_select, sits_timeline, ) -from pysits.sits.utils import r_package_dir def test_cube_select(): @@ -167,21 +166,13 @@ def test_sits_apply(): assert all(band in sits_bands(points_nonnorm) for band in ["NDVI", "NDVI_nonnorm"]) -def test_cube_apply(tmp_path: Path): +def test_cube_apply(tmp_path: Path, local_cube): """Test cube apply.""" - data_dir = r_package_dir( - "extdata/raster/mod13q1", - package="sits", - ) - cube = sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=data_dir, - ) - - # Generate a texture images with variance in NDVI images cube_texture = sits_apply( - data=cube, NDVITEXTURE="w_median(NDVI)", window_size=5, output_dir=tmp_path + data=local_cube, + NDVITEXTURE="w_median(NDVI)", + window_size=5, + output_dir=tmp_path, ) assert all(band in sits_bands(cube_texture) for band in ["NDVI", "NDVITEXTURE"]) @@ -196,21 +187,10 @@ def test_sits_reduce(): assert len(sits_timeline(points_reduced)) == 1 # noqa: PLR2004 - one date -def test_cube_reduce(tmp_path: Path): +def test_cube_reduce(tmp_path: Path, local_cube): """Test reduce in cube.""" - data_dir = r_package_dir( - "extdata/raster/mod13q1", - package="sits", - ) - cube = sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=data_dir, - ) - - # Reduce the cube cube_reduced = sits_reduce( - cube, + local_cube, NDVIMEAN="t_mean(NDVI)", output_dir=tmp_path, progress=False, diff --git a/tests/test_embedding.py b/tests/test_embedding.py new file mode 100644 index 0000000..663a494 --- /dev/null +++ b/tests/test_embedding.py @@ -0,0 +1,251 @@ +# +# Copyright (C) 2025 sits developers. +# +# This program is free software; you can redistribute it and/or modify it +# under the terms of the GNU General Public License as published by +# the Free Software Foundation; either version 2 of the License, or +# (at your option) any later version. +# +# This program is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with this program; if not, see . +# + +"""end-to-end embeddings test.""" + +from pathlib import Path + +import pytest +from rpy2.rinterface_lib.embedded import RRuntimeError + +from pysits.models.data.cube import SITSCubeModel +from pysits.models.data.ts import SITSTimeSeriesModel +from pysits.sits.context import samples_modis_ndvi +from pysits.sits.data import ( + sits_bands, + sits_classify, + sits_encode, + sits_label_classification, + sits_labels, + sits_select, + sits_timeline, +) +from pysits.sits.ml import ( + sits_barlow_twins, + sits_contrastive_learning, + sits_pre_train, + sits_rfor, + sits_ssl_lejepa, + sits_ssl_mae, + sits_ssl_vicreg, + sits_train, +) +from pysits.sits.visualization import sits_plot + +# +# Encoder configuration +# +EMBEDDING_DIM = 8 +EMBEDDING_BANDS = [f"EMB0{i}" for i in range(1, EMBEDDING_DIM + 1)] + +# +# Encoder methods available to test +# +ALL_ENCODER_METHODS = [ + sits_ssl_mae, + sits_ssl_lejepa, + sits_ssl_vicreg, + sits_contrastive_learning, + sits_barlow_twins, +] + + +# +# Fixtures +# +@pytest.fixture(scope="module") +def encoder(): + """Encoder pre-trained with the MODIS NDVI samples.""" + return sits_pre_train( + samples=samples_modis_ndvi, + rl_method=sits_ssl_vicreg( + embedding_dim=EMBEDDING_DIM, + epochs=1, + ), + ) + + +@pytest.fixture(scope="module") +def embeddings(encoder): + """MODIS NDVI samples encoded as embeddings.""" + return sits_encode( + data=samples_modis_ndvi, + encoder=encoder, + ) + + +@pytest.fixture(scope="module") +def embeddings_cube(encoder, local_cube, tmp_path_factory): + """MODIS cube encoded as an embeddings cube.""" + return sits_encode( + data=local_cube, + encoder=encoder, + output_dir=tmp_path_factory.mktemp("embeddings"), + multicores=1, + memsize=4, + progress=False, + ) + + +# +# Test encoding of time series +# +@pytest.mark.parametrize("model_fn", ALL_ENCODER_METHODS) +def test_encode_time_series(model_fn): + """Test time series encoding with all available encoder methods.""" + # Pre-train an encoder + rl_model = sits_pre_train( + samples=samples_modis_ndvi, + rl_method=model_fn( + embedding_dim=EMBEDDING_DIM, + epochs=1, + ), + ) + + # Encode the samples used to pre-train the encoder + embeddings = sits_encode( + data=samples_modis_ndvi, + encoder=rl_model, + ) + + assert isinstance(embeddings, SITSTimeSeriesModel) + + # The original band (``NDVI``) is replaced by the embedding dimensions + assert sits_bands(embeddings) == EMBEDDING_BANDS + + # The samples structure is preserved + assert embeddings.shape[0] == samples_modis_ndvi.shape[0] + assert sits_labels(embeddings) == sits_labels(samples_modis_ndvi) + + # Each embedding is a single point in time (the samples ``start_date``) + assert len(sits_timeline(embeddings)) == 1 + + +# +# Test encoding of data cubes +# +def test_encode_cube(encoder, local_cube, tmp_path: Path): + """Test data cube encoding.""" + embeddings_cube = sits_encode( + data=local_cube, + encoder=encoder, + output_dir=tmp_path, + multicores=1, + memsize=4, + progress=False, + ) + + assert isinstance(embeddings_cube, SITSCubeModel) + + # The original band (``NDVI``) is replaced by the embedding dimensions + assert sits_bands(embeddings_cube) == EMBEDDING_BANDS + + # the cube tiling is preserved + assert embeddings_cube["tile"][0] == local_cube["tile"][0] + + # one file is written for each embedding dimension + assert embeddings_cube["file_info"][0].shape[0] == EMBEDDING_DIM + assert len(list(tmp_path.glob("*EMB*.tif"))) == EMBEDDING_DIM + + # Test recover + recovered_cube = sits_encode( + data=local_cube, + encoder=encoder, + output_dir=tmp_path, + multicores=1, + memsize=4, + progress=False, + ) + + assert sits_bands(recovered_cube) == EMBEDDING_BANDS + assert len(list(tmp_path.glob("*EMB*.tif"))) == EMBEDDING_DIM + + +# +# Test classification based on embeddings +# +def test_encode_cube_classification(embeddings, embeddings_cube, tmp_path: Path): + """Test classification of an embeddings cube.""" + # Train a model using the encoded samples + model = sits_train(embeddings, ml_method=sits_rfor()) + + # Classify the embeddings cube + probs_cube = sits_classify( + data=embeddings_cube, + ml_model=model, + output_dir=tmp_path, + multicores=1, + memsize=4, + progress=False, + ) + + assert isinstance(probs_cube, SITSCubeModel) + assert sits_bands(probs_cube) == ["probs"] + + # Generate labels + label_cube = sits_label_classification(probs_cube, output_dir=tmp_path) + + assert "labels" in label_cube.columns + assert sits_labels(label_cube) == sits_labels(samples_modis_ndvi) + + +# +# Test encoding errors +# +def test_encode_invalid_encoder(): + """Test encoding with a model that is not an encoder.""" + ml_model = sits_train(samples_modis_ndvi, ml_method=sits_rfor()) + + with pytest.raises(RRuntimeError, match="invalid encoder"): + sits_encode(data=samples_modis_ndvi, encoder=ml_model) + + +# +# Test visualization of embeddings +# +@pytest.mark.parametrize( + "plot_args", + [ + {}, + {"mode": "dimensions"}, + {"mode": "PCA"}, + {"mode": "tsne", "perplexity": 30}, + ], +) +def test_encode_time_series_plot(embeddings, plot_args, no_plot_window): + """Test visualization of encoded time series.""" + sits_plot(embeddings, **plot_args) + + +def test_encode_time_series_plot_by_label(embeddings, no_plot_window): + """Test visualization of encoded time series from a single label.""" + label = sits_labels(embeddings)[0] + + sits_plot(sits_select(embeddings, labels=label), mode="dimensions") + + +@pytest.mark.parametrize( + "plot_args", + [ + {}, + {"band": "EMB01"}, + {"red": "EMB04", "green": "EMB02", "blue": "EMB03"}, + ], +) +def test_encode_cube_plot(embeddings_cube, plot_args, no_plot_window): + """Test visualization of an embeddings cube.""" + sits_plot(embeddings_cube, **plot_args) diff --git a/tests/test_exporters.py b/tests/test_exporters.py new file mode 100644 index 0000000..7385bee --- /dev/null +++ b/tests/test_exporters.py @@ -0,0 +1,96 @@ +# +# Copyright (C) 2025 sits developers. +# +# This program is free software; you can redistribute it and/or modify it +# under the terms of the GNU General Public License as published by +# the Free Software Foundation; either version 2 of the License, or +# (at your option) any later version. +# +# This program is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with this program; if not, see . +# + +"""Unit tests for file exporters.""" + +from pathlib import Path +from zipfile import ZipFile + +import pytest + +from pysits.backend.pkgs import r_pkg_sits +from pysits.models.data.frame import SITSFrame +from pysits.models.data.matrix import SITSConfusionMatrix +from pysits.sits.context import cerrado_2classes +from pysits.sits.exporters import sits_timeseries_to_csv, sits_to_csv, sits_to_xlsx +from pysits.sits.ml import sits_rfor +from pysits.sits.ts import sits_sample, sits_validate + + +# +# Helpers +# +def read_xlsx(file: Path) -> dict[str, bytes]: + """Read the reproducible content of an XLSX file.""" + xlsx_metadata = "docProps/core.xml" + + with ZipFile(file) as workbook: + return { + name: workbook.read(name) + for name in workbook.namelist() + if name != xlsx_metadata + } + + +# +# Fixtures +# +@pytest.fixture(scope="module") +def accuracy() -> SITSConfusionMatrix: + """Accuracy assessment of a validated model.""" + return sits_validate( + samples=sits_sample(cerrado_2classes, frac=0.5), + samples_validation=sits_sample(cerrado_2classes, frac=0.5), + ml_method=sits_rfor(), + ) + + +# +# Tests +# +def test_sits_to_csv(tmp_path: Path): + """Test time-series metadata exported as csv.""" + py_file = tmp_path / "python.csv" + r_file = tmp_path / "r.csv" + + result = sits_to_csv(cerrado_2classes, file=py_file) + r_pkg_sits.sits_to_csv(cerrado_2classes._instance, file=str(r_file)) + + assert isinstance(result, SITSFrame) + assert py_file.read_text() == r_file.read_text() + + +def test_sits_timeseries_to_csv(tmp_path: Path): + """Test time-series values exported as csv.""" + py_file = tmp_path / "python.csv" + r_file = tmp_path / "r.csv" + + sits_timeseries_to_csv(cerrado_2classes, file=py_file) + r_pkg_sits.sits_timeseries_to_csv(cerrado_2classes._instance, file=str(r_file)) + + assert py_file.read_text() == r_file.read_text() + + +def test_sits_to_xlsx(tmp_path: Path, accuracy): + """Test accuracy assessment exported as xlsx.""" + py_file = tmp_path / "python.xlsx" + r_file = tmp_path / "r.xlsx" + + sits_to_xlsx(accuracy, file=py_file) + r_pkg_sits.sits_to_xlsx(accuracy._instance, file=str(r_file)) + + assert read_xlsx(py_file) == read_xlsx(r_file) diff --git a/tests/test_geopandas.py b/tests/test_exporters_geopandas.py similarity index 78% rename from tests/test_geopandas.py rename to tests/test_exporters_geopandas.py index 548d269..06d8d0a 100644 --- a/tests/test_geopandas.py +++ b/tests/test_exporters_geopandas.py @@ -15,33 +15,19 @@ # along with this program; if not, see . # -"""Unit tests for geopandas operations.""" +"""Unit tests for geopandas export operations.""" from geopandas import GeoDataFrame as GeoPandasDataFrame from pysits.models.data.frame import SITSFrameSF from pysits.models.data.ts import SITSTimeSeriesSFModel from pysits.sits.context import samples_l8_rondonia_2bands -from pysits.sits.cube import sits_cube from pysits.sits.exporters import sits_as_geopandas -from pysits.sits.utils import r_package_dir -def test_cube_geopandas_export(): +def test_cube_geopandas_export(local_cube): """Test cube as geopandas.""" - # Load cube - data_dir = r_package_dir( - "extdata/raster/mod13q1", - package="sits", - ) - cube = sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=data_dir, - ) - - # Export data - cube_gdf = sits_as_geopandas(cube) + cube_gdf = sits_as_geopandas(local_cube) # Check properties assert cube_gdf.crs is not None diff --git a/tests/test_xarray.py b/tests/test_exporters_xarray.py similarity index 93% rename from tests/test_xarray.py rename to tests/test_exporters_xarray.py index 4ef7419..23c4b93 100644 --- a/tests/test_xarray.py +++ b/tests/test_exporters_xarray.py @@ -63,16 +63,6 @@ # # Auxiliary functions # -def _mod13q1_cube(): - """Create a cube from the local MOD13Q1 files.""" - return sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=r_package_dir("extdata/raster/mod13q1", package="sits"), - progress=False, - ) - - def _class_cube(): """Create a cube from the local classification files.""" return sits_cube( @@ -194,9 +184,9 @@ def cog_cube(cog_cube_dir): # # Data cubes # -def test_xarray_cube_dimensions(): +def test_xarray_cube_dimensions(local_cube): """Test cube conversion dimensions and coordinates.""" - data = sits_as_xarray(_mod13q1_cube()) + data = sits_as_xarray(local_cube) assert isinstance(data, xr.DataArray) assert data.dims == ("band", "time", "y", "x") @@ -214,12 +204,11 @@ def test_xarray_cube_dimensions(): assert "spatial_ref" in data.coords -def test_xarray_cube_values(): +def test_xarray_cube_values(local_cube): """Test cube conversion values, against the values in the files.""" - cube = _mod13q1_cube() - data = sits_as_xarray(cube) + data = sits_as_xarray(local_cube) - with rasterio.open(_first_cube_file(cube)) as dataset: + with rasterio.open(_first_cube_file(local_cube)) as dataset: expected = dataset.read(1).astype("float32") expected_transform = dataset.transform @@ -241,12 +230,11 @@ def test_xarray_cube_values(): assert data.rio.transform().f == pytest.approx(expected_transform.f) -def test_xarray_cube_without_scale(): +def test_xarray_cube_without_scale(local_cube): """Test cube conversion without scaling.""" - cube = _mod13q1_cube() - data = sits_as_xarray(cube, scale=False) + data = sits_as_xarray(local_cube, scale=False) - with rasterio.open(_first_cube_file(cube)) as dataset: + with rasterio.open(_first_cube_file(local_cube)) as dataset: expected = dataset.read(1) # Test with NDVI @@ -257,15 +245,14 @@ def test_xarray_cube_without_scale(): assert np.array_equal(values.values, expected) -def test_xarray_cube_bands(): +def test_xarray_cube_bands(local_cube): """Test cube conversion with band selection.""" - cube = _mod13q1_cube() - data = sits_as_xarray(cube, bands=["NDVI"]) + data = sits_as_xarray(local_cube, bands=["NDVI"]) assert list(data["band"].values) == ["NDVI"] with pytest.raises(ValueError, match="Bands not available"): - sits_as_xarray(cube, bands=["EVI"]) + sits_as_xarray(local_cube, bands=["EVI"]) # @@ -332,8 +319,8 @@ def test_xarray_cube_chunks(cog_cube): # def test_xarray_class_cube(): """Test class cube conversion.""" - cube = _class_cube() - data = sits_as_xarray(cube) + local_cube = _class_cube() + data = sits_as_xarray(local_cube) # Test metadata assert isinstance(data, xr.DataArray) @@ -349,7 +336,7 @@ def test_xarray_class_cube(): } # Load data - with rasterio.open(_first_cube_file(cube)) as dataset: + with rasterio.open(_first_cube_file(local_cube)) as dataset: expected = dataset.read(1) # Values must be the same diff --git a/tests/test_ml_models.py b/tests/test_ml_models.py index 0476c15..7da8ed6 100644 --- a/tests/test_ml_models.py +++ b/tests/test_ml_models.py @@ -19,20 +19,21 @@ import cloudpickle import pytest +from conftest import r_class, r_closure_value, r_opt_hparams from pysits.models.ml import SITSMachineLearningMethod -from pysits.sits.classification import sits_classify from pysits.sits.context import ( point_mt_6bands, samples_l8_rondonia_2bands, samples_modis_ndvi, ) -from pysits.sits.data import sits_labels, sits_select +from pysits.sits.data import sits_classify, sits_labels, sits_select from pysits.sits.ml import ( sits_formula_linear, sits_formula_logref, sits_lightgbm, sits_lighttae, + sits_lstm_fcn, sits_mlp, sits_model_export, sits_resnet, @@ -57,6 +58,19 @@ sits_svm, sits_xgboost, sits_lightgbm, + sits_lstm_fcn, +] + +# +# Models with dl-specific converters +# +DL_MODELS = [ + sits_tae, + sits_tempcnn, + sits_lighttae, + sits_mlp, + sits_resnet, + sits_lstm_fcn, ] @@ -129,6 +143,31 @@ def test_model_svm_params(): assert isinstance(model_logref, SITSMachineLearningMethod) +# +# Test dl-specific converters +# +@pytest.mark.parametrize("model_fn", DL_MODELS) +def test_model_converters(model_fn): + """Test conversion of dl-specific parameters.""" + ml_method = model_fn( + optimizer="torch::optim_adam", + opt_hparams={"lr": 0.01, "eps": 1e-07}, + ) + + # The optimizer is loaded from R (`optim_adam` is not the default optimizer) + assert "optim_adam" in r_class(r_closure_value(ml_method, "optimizer")) + + # The hyperparameters are converted to an R list + assert r_opt_hparams(ml_method) == {"lr": 0.01, "eps": 1e-07} + + +@pytest.mark.parametrize("model_fn", DL_MODELS) +def test_model_converters_invalid_optimizer(model_fn): + """Test conversion of an optimizer defined in an invalid format.""" + with pytest.raises(ValueError, match="Invalid optimizer format"): + model_fn(optimizer=sits_formula_linear) + + # # Test model export # diff --git a/tests/test_rl_models.py b/tests/test_rl_models.py index 955cff1..ecf56f1 100644 --- a/tests/test_rl_models.py +++ b/tests/test_rl_models.py @@ -18,6 +18,7 @@ """Unit tests for representation learning models.""" import pytest +from conftest import r_class, r_closure_value, r_identical, r_opt_hparams from pysits.models.ml import SITSRepresentationLearningMethod from pysits.sits.context import samples_modis_ndvi @@ -79,3 +80,38 @@ def test_model_pre_training_with_encoder_model(model_fn): ) assert isinstance(model, SITSRepresentationLearningMethod) + + # The model was pre-trained with the encoder defined by the user + # (`model_ltae`), and not with the default one (`model_tcnn`) + assert "model_ltae" in r_class(r_closure_value(model._instance, "encoder")) + + +# +# Test encoder methods parameters +# +@pytest.mark.parametrize("model_fn", ALL_ENCODER_METHODS) +def test_encoder_method_encoder_model(model_fn): + """Test encoder model defined by the user is used as is.""" + encoder_model = sits_lighttae(opt_hparams={"lr": 0.02}) + rl_method = model_fn(encoder_model=encoder_model) + + # The encoder model is passed to R untouched + assert r_identical(r_closure_value(rl_method, "encoder_model"), encoder_model) + + # Including its own converted parameters + assert r_opt_hparams(encoder_model) == {"lr": 0.02} + + +@pytest.mark.parametrize("model_fn", ALL_ENCODER_METHODS) +def test_encoder_method_converters(model_fn): + """Test conversion of dl-specific parameters.""" + rl_method = model_fn( + optimizer="torch::optim_adam", + opt_hparams={"lr": 0.01, "eps": 1e-07}, + ) + + # The optimizer is loaded from R (`optim_adam` is not the default optimizer) + assert "optim_adam" in r_class(r_closure_value(rl_method, "optimizer")) + + # The hyperparameters are converted to an R list + assert r_opt_hparams(rl_method) == {"lr": 0.01, "eps": 1e-07} diff --git a/tests/test_som.py b/tests/test_som.py new file mode 100644 index 0000000..09ec876 --- /dev/null +++ b/tests/test_som.py @@ -0,0 +1,103 @@ +# +# Copyright (C) 2025 sits developers. +# +# This program is free software; you can redistribute it and/or modify it +# under the terms of the GNU General Public License as published by +# the Free Software Foundation; either version 2 of the License, or +# (at your option) any later version. +# +# This program is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with this program; if not, see . +# + +"""Unit tests for SOM operations.""" + +import pytest + +from pysits.models.data.ts import SITSTimeSeriesModel +from pysits.sits.context import samples_modis_ndvi +from pysits.sits.ts import ( + sits_som_evaluate_cluster, + sits_som_map, + sits_som_remove_samples, +) +from pysits.sits.utils import r_set_seed + + +# +# Fixtures +# +@pytest.fixture(scope="module") +def som_data(): + """SOM map and cluster evaluation of the MODIS NDVI samples.""" + r_set_seed(42) + + # Build a small SOM map, enough to expose class confusion + som_map = sits_som_map(samples_modis_ndvi, grid_xdim=4, grid_ydim=4) + + # Return objects + return som_map, sits_som_evaluate_cluster(som_map) + + +def test_som_remove_samples(som_data): + """Test removal of samples confused with a different class.""" + som_map, som_eval = som_data + + # Remove! + new_samples = sits_som_remove_samples( + som_map, + som_eval, + "Pasture", + "Cerrado", + ) + + # Check output type + assert isinstance(new_samples, SITSTimeSeriesModel) + + # Only ``Cerrado`` samples are removed + labels = samples_modis_ndvi["label"].value_counts() + new_labels = new_samples["label"].value_counts() + + assert new_labels["Cerrado"] < labels["Cerrado"] + + # Check all other labels are there + for label in ("Pasture", "Soy_Corn", "Forest"): + assert new_labels[label] == labels[label] + + +def test_som_remove_samples_named_arguments(som_data): + """Test removal of samples using named arguments.""" + som_map, som_eval = som_data + + # Remove samples! + new_samples = sits_som_remove_samples( + som_map=som_map, + som_eval=som_eval, + class_cluster="Pasture", + class_remove="Cerrado", + ) + + # Same result as the positional form + assert new_samples.shape == ( + sits_som_remove_samples(som_map, som_eval, "Pasture", "Cerrado").shape + ) + + +def test_som_remove_samples_unknown_label(som_data): + """Test removal of samples using a label not available in the samples.""" + som_map, som_eval = som_data + + new_samples = sits_som_remove_samples( + som_map, + som_eval, + "Pasture", + "NotALabel", + ) + + # No samples are removed + assert len(new_samples) == len(samples_modis_ndvi) diff --git a/tests/test_visualization.py b/tests/test_visualization.py index e11d8e7..da55a9e 100644 --- a/tests/test_visualization.py +++ b/tests/test_visualization.py @@ -17,12 +17,45 @@ """Unit tests for visualization operations.""" -from pysits.sits.context import samples_l8_rondonia_2bands +import pytest +from rpy2.rinterface_lib.embedded import RRuntimeError + +from pysits.sits.context import samples_l8_rondonia_2bands, samples_modis_ndvi from pysits.sits.cube import sits_cube +from pysits.sits.data import sits_classify, sits_label_classification from pysits.sits.ml import sits_pre_train, sits_rfor, sits_ssl_mae, sits_train from pysits.sits.ts import sits_patterns, sits_som_map -from pysits.sits.utils import r_package_dir -from pysits.sits.visualization import sits_plot, sits_view +from pysits.sits.utils import r_package_dir, r_set_seed +from pysits.sits.visualization import sits_plot, sits_sankey, sits_view + + +# +# Fixtures +# +@pytest.fixture(scope="module") +def class_cubes(local_cube, tmp_path_factory): + """Two single-step classified cubes.""" + r_set_seed(42) + output_dir = tmp_path_factory.mktemp("sankey") + + # Train a random forest model + rfor_model = sits_train(samples_modis_ndvi, sits_rfor()) + + # Classify the cube + probs_cube = sits_classify( + local_cube, + rfor_model, + output_dir=output_dir, + progress=False, + ) + + # Prepare cubes + return tuple( + sits_label_classification( + probs_cube, output_dir=output_dir, version=version, progress=False + ) + for version in ("v2013", "v2014") + ) def test_sits_visualization(no_plot_window): @@ -54,15 +87,9 @@ def test_som_visualization(no_plot_window): sits_plot(som) -def test_cube_visualization(no_plot_window): +def test_cube_visualization(local_cube, no_plot_window): """Test cube visualization.""" - cube = sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=r_package_dir("extdata/raster/mod13q1", package="sits"), - ) - - sits_plot(cube) + sits_plot(local_cube) def test_classified_cube_visualization(no_plot_window): @@ -100,14 +127,55 @@ def test_sits_visualization_leaflet(no_browser): sits_view(samples_l8_rondonia_2bands) -def test_cube_visualization_leaflet(no_browser): +def test_cube_visualization_leaflet(local_cube, no_browser): """Test cube visualization.""" - # Create a cube - cube = sits_cube( - source="BDC", - collection="MOD13Q1-6.1", - data_dir=r_package_dir("extdata/raster/mod13q1", package="sits"), + sits_view(local_cube) + + +def test_sankey_visualization(class_cubes, no_plot_window): + """Test sankey visualization of class trajectories.""" + class_2013, class_2014 = class_cubes + + # The diagram is displayed, not returned + assert ( + sits_sankey( + class_2013, + class_2014, + labels=["2013", "2014"], + multicores=1, + progress=False, + ) + is None ) - # Plot the cube - sits_view(cube) + +def test_sankey_visualization_cubes_argument(class_cubes, no_plot_window): + """Test sankey visualization using the ``cubes`` argument.""" + class_2013, class_2014 = class_cubes + + # Cubes can also be given as a list, instead of separate arguments + assert ( + sits_sankey( + cubes=[class_2013, class_2014], + labels=["2013", "2014"], + title="Trajectories", + palette="Set2", + multicores=1, + progress=False, + ) + is None + ) + + +def test_sankey_visualization_invalid_input(class_cubes): + """Test sankey visualization with cubes defined in an invalid format.""" + class_2013, class_2014 = class_cubes + + # Cubes must be given as separate arguments or via ``cubes``, not both + with pytest.raises(RRuntimeError, match="not both"): + sits_sankey( + class_2013, + cubes=[class_2013, class_2014], + multicores=1, + progress=False, + ) diff --git a/uv.lock b/uv.lock index 537ad4f..b8590dd 100644 --- a/uv.lock +++ b/uv.lock @@ -1421,7 +1421,7 @@ wheels = [ [[package]] name = "pysits" -version = "1.5.4" +version = "2.0.0.dev2" source = { editable = "." } dependencies = [ { name = "geopandas" },