This commit is contained in:
Joaquin Gottlebe
2025-07-30 15:57:48 +02:00
parent e49fcfac46
commit a639c34cee
273 changed files with 20151 additions and 0 deletions
@@ -0,0 +1,67 @@
"""
Calculate and compare yearly energy sums from actual photovoltaic and theoretical solar datasets.
This script reads two CSV datasets containing monthly energy values per year — one representing
actual photovoltaic energy measurements, the other representing theoretical solar radiation
estimates. It sums the monthly values per year, and merges the results into a single output file
for comparison or further analysis.
Usage:
python calculate_difference.py <actual_energy_path> <theoretical_energy_path>
<output_path>
Arguments:
actual_energy_path (str or Path): path to CSV file with actual photovoltaic energy data.
theoretical_energy_path (str or Path): path to CSV file with theoretical solar radiation
data.
output_path (str or Path): destination path for the merged yearly sums output
CSV.
"""
import os
import sys
import pandas as pd
import utils
import checks
import logs
def calc_energy_difference(df_actual_path, df_theoretical_path, output_path):
"""
Calculate the absolute yearly energy difference between actual and theoretical datasets.
Reads two CSV files with yearly energy data, computes the absolute difference
in energy values for each common year, and saves the result as a CSV.
Args:
df_actual_path (str or Path): Path to the CSV file with actual photovoltaic energy
data.
df_theoretical_path (str or Path): Path to the CSV file with theoretical solar radiation
data.
output_path (str or Path): Path where the output CSV with yearly energy differences
will be saved.
Returns:
None: The result is saved to `output_path`.
"""
logs.log_processing(os.path.basename(__file__))
checks.check_path(df_actual_path)
checks.check_path(df_theoretical_path)
checks.check_dir(output_path)
df_actual = utils.read_df(df_actual_path, ",", 0, None)
df_theoretical = utils.read_df(df_theoretical_path, ",", 0, None)
diff_difference = abs(
df_actual[df_actual.columns[1]] - df_theoretical[df_theoretical.columns[1]])
df_diff = pd.DataFrame({
'year': df_actual['year'],
'Energy Difference [kWh]': diff_difference})
utils.save_df(df_diff, output_path, False)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(4, __doc__)
calc_energy_difference(sys.argv[1], sys.argv[2], sys.argv[3])
@@ -0,0 +1,64 @@
"""
Calculate theoretical energy generation of solar parks.
This script reads the area of solar parks (in m²) and sunshine duration (in hours),
then calculates the theoretical energy produced (in kWh) using a given efficiency.
Usage:
python calculate_energy.py <area_data_path> <sunshine_data_path> <output_data_path> <efficiency>
Arguments:
area_data_path (str or Path): Path to the input file with solar park area data.
sunshine_data_path (str or Path): Path to the input file with sunshine duration data.
output_data_path (str or Path): Path where the output file will be saved.
efficiency (float): Average efficiency of PV panels (as a decimal, e.g. 0.15
for 15%)
"""
import os
import sys
import utils
import checks
import logs
def calculate_energy(area_path, sunshine_path, output_path, efficiency):
"""
Calculate the theoretical energy production of solar parks in kWh.
Reads input files with solar park area and sunshine duration, then computes
energy using the formula: energy = area * sunshine_duration * power_per_m2 * efficiency.
Args:
area_path (str or Path): Path to the input file with area data (m²).
sunshine_path (str or Path): Path to the input file with sunshine duration data (hours).
output_path (str or Path): Path to save the calculated energy output CSV.
efficiency (float): Average efficiency of PV panels (decimal).
Returns:
None: Saves the calculated energy DataFrame to the specified output path.
"""
logs.log_processing(os.path.basename(__file__))
logs.log_parameter('efficiency', str(efficiency))
checks.check_path(area_path)
checks.check_path(sunshine_path)
checks.check_dir(output_path)
checks.check_empty(efficiency)
content = utils.read_file(area_path)
area = utils.parse_number(content)
sunshine_data = utils.read_df(sunshine_path, ",", 0, 0)
efficiency_number = float(efficiency)
power_per_m2 = 0.1
# main calculation
factor = power_per_m2 * area * efficiency_number
energy = sunshine_data * factor
utils.save_df(energy, output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(5, __doc__)
calculate_energy(sys.argv[1], sys.argv[2], sys.argv[3], sys.argv[4])
@@ -0,0 +1,176 @@
"""
Basic input validation checks.
Functions to check for :
- non-empty variables,
- existing file or directory paths
- correct command-line argument count
- two CRS's if they are equal
- dataframe valid and not empty
- if GeoDataFrame is valid and not empty
- if variable is member of a list
"""
import os
import sys
from pathlib import Path
import pandas as pd
import geopandas as gpd
def check_empty(variable):
"""
Checks if variable is not empty.
Arguments:
variable: variable to check.
Returns:
None
Raises:
ValueError: If the variable is empty.
"""
if not variable:
raise ValueError("Variable is empty")
def check_path(path):
"""
Checks if path exists.
Arguments:
path (str): path to check.
Returns:
None
Raises:
FileNotFoundError: If the path does not exist.
"""
if not os.path.exists(path):
raise FileNotFoundError(f"Invalid path: {path}")
def check_dir(path):
"""
Checks if directory exists.
Arguments:
path (str): path to check.
Returns:
None
Raises:
FileNotFoundError: If the directory's parent path does not exist.
"""
if not Path(path).parent.exists():
raise FileNotFoundError(f"Invalid directory path: {path}")
def check_args(num_arg, output):
"""
Checks sys arguments length against expected.
Args:
num_arg (int): expected number of arguments.
output (str): string to be printed in case of check failure.
Returns:
None
Raises:
RuntimeError: If the number of arguments does not match `num_arg`.
"""
if len(sys.argv) != num_arg:
raise RuntimeError(output)
def check_crs(crs1, crs2):
"""
Checks two CRS's if they are equal.
Arguments:
crs1: first CRS.
crs2: second CRS.
Returns:
None
Raises:
ValueError: If CRS's are not equal.
"""
if crs1 != crs2:
raise ValueError("CRS missmatch.")
def check_df(df):
"""
Checks if DataFrame is valid and not empty.
Arguments:
df: DataFrame to check.
Returns:
None
Raises:
TypeError: If not a DataFrame
ValueError: If DataFrame is empty
ValueError: If all values in the DataFrame are NaN
"""
if not isinstance(df, pd.DataFrame):
raise TypeError("Not a DataFrame.")
if df.empty:
raise ValueError("DataFrame is empty.")
if df.isna().all().all():
raise ValueError("DataFrame contains only NaN values.")
def check_gdf(gdf):
"""
Checks if GeoDataFrame is valid and not empty.
Arguments:
gdf: GeoDataFrame to check.
Returns:
None
Raises:
TypeError: if not an GeoDataFrame
ValueError: If GeoDataFrame is empty
ValueError: If includes empty geometries
"""
if not isinstance(gdf, gpd.GeoDataFrame):
raise TypeError("Not a GeoDataFrame.")
if gdf.empty:
raise ValueError("GeoDataFrame is empty.")
if gdf.geometry.isna().all():
raise ValueError("GeoDataFrame includes empty geometries.")
def check_member(variable, members):
"""
Check if variable is member of a list.
Arguments:
variable: to check
members (list): to check against.
Returns:
None
Raises:
ValueError: If variable is not member of allowed varibales.
"""
if variable not in members:
raise ValueError(f"Invalid value: {
variable}. Must be one of: {members}")
@@ -0,0 +1,85 @@
"""
Process and clean photovoltaic dataset from Destatis.
This script reads a CSV dataset (exported from the Destatis web portal), removes
unnecessary table structures caused by web-to-CSV conversion, and saves a cleaned version.
It also allows extracting a subset of the data, e.g. "Electricity feed-in systems",
"Net nominal capacity", or "Electricity feed-in".
Original dataset source:
https://www-genesis.destatis.de/datenbank/online/statistic/43312/table/43312-0001
Usage:
python clean_pv_data.py <dataset_path> <save_path> <extracted_type>
Arguments:
dataset_path (str or Path): Path to the raw CSV dataset file.
save_path (str or Path): Path where the cleaned CSV will be saved.
extracted_type (str): Type of data to extract (e.g., "Electricity feed-in").
"""
import os
import sys
import pandas as pd
import utils
import checks
import logs
def clean_pv_data(input_path, output_path, extracted_type):
"""
Clean and preprocess the photovoltaic dataset from Destatis.
This function selects relevant columns, renames them for clarity,
defines categorical ordering for months and data types, sorts the data,
and extracts the specified subset. The result is saved as a pivot table CSV.
Args:
input_path (str or Path): Path to the raw input CSV file.
output_path (str or Path): Path to save the cleaned and extracted CSV.
extracted_type (str): The category of data to extract (e.g., "Electricity feed-in").
Returns:
None: The processed DataFrame is saved to `output_path`.
"""
logs.log_processing(os.path.basename(__file__))
checks.check_path(input_path)
checks.check_dir(output_path)
checks.check_empty(extracted_type)
df = utils.read_df(input_path, seperator=";", header_line=0, icol=None)
df_relevant = df[["time", "1_variable_attribute_label",
"value", "value_variable_label"]]
df_cleaned = df_relevant.rename(columns={
'time': 'year',
'1_variable_attribute_label': 'month',
'value_variable_label': 'type'
})
monthly_order = ["January", "February", "March", "April", "May", "June",
"July", "August", "September", "October", "November", "December"]
# Units: [number, MW, Mwh]
type_order = ["Electricity feed-in systems",
"Net nominal capacity", "Electricity feed-in"]
df_cleaned['month'] = pd.Categorical(
df_cleaned['month'], categories=monthly_order, ordered=True)
df_cleaned['type'] = pd.Categorical(
df_cleaned['type'], categories=type_order, ordered=True)
df_cleaned = df_cleaned.sort_values(
by=['year', 'month', 'type'], ascending=True)
extracted_df = df_cleaned[df_cleaned["type"] == extracted_type]
pivot_df = extracted_df.pivot(
index="year", columns="month", values="value")
checks.check_df(pivot_df)
utils.save_df(pivot_df, output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(4, __doc__)
clean_pv_data(sys.argv[1], sys.argv[2], sys.argv[3])
@@ -0,0 +1,59 @@
"""
Perform spatial clipping of geospatial data using GeoPandas.
This script reads an input geospatial file and an overlay file,
clips the input to the overlay boundaries, and saves the clipped
result as a GeoPackage file.
Usage:
python clip.py <input_path> <overlay_path> <output_path>
Arguments:
input_path (str or Path): Path to the input geospatial file (e.g., GeoPackage, shapefile).
overlay_path (str or Path): Path to the overlay geospatial file used for clipping.
output_path (str or Path): Path to save the clipped output as a GeoPackage (.gpkg).
"""
import os
import sys
import geopandas as gpd
import utils
import checks
import logs
def clip(input_path, overlay_path, output_path):
"""
Clips input to overlay and exports result. Includes Errorhandling.
Performs spatial clipping of input data using the geometry boundaries
from the overlay data, and exports the clipped result to a GeoPackage.
Arguments:
input_path (str or Path): Path to the input file (.gpkg).
overlay_path (str or Path): Path to the clipping layer (.gpkg).
output_path (str or Path): Path to save the clipped output (.gpkg).
Returns:
None: The processed clipped output is saved to `output_path`.
"""
logs.log_processing(os.path.basename(__file__))
logs.log_intensive()
checks.check_path(input_path)
checks.check_path(overlay_path)
checks.check_dir(output_path)
input_data = utils.read_gdf(input_path)
overlay_data = utils.read_gdf(overlay_path)
checks.check_crs(input_data.crs, overlay_data.crs)
clipped_data = gpd.clip(input_data, overlay_data)
checks.check_gdf(clipped_data)
utils.save_gdf(clipped_data, output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(4, __doc__)
clip(sys.argv[1], sys.argv[2], sys.argv[3])
@@ -0,0 +1,46 @@
"""
This script reads a tabular CSV dataset, sums the values across all columns for each row,
and writes the resulting single-column DataFrame to an output file.
Usage:
python collapse_columns.py <input_path> <output_path> <column_name>
Arguments:
input_path (str or Path): Path to the input CSV file containing the original tabular data.
output_path (str or Path): Path where the collapsed output CSV file will be saved.
column_name (str): Name of the column in the output file representing the summed values.
"""
import os
import sys
import utils
import checks
import logs
def collapse_columns(input_path, output_path, column_name):
"""
Reads a tabular dataset and collapses each row into a single value by summing across columns.
The result is written as a single-column DataFrame to the output path.
Args:
input_path (str or Path): Path to the input CSV file.
output_path (str or Path): Path where the output CSV file will be saved.
column_name (str): Name of the resulting summed column.
Returns:
None: The resulting summed dataframe is saved to `output_path`
"""
logs.log_processing(os.path.basename(__file__))
logs.log_parameter("column_name", str(column_name))
checks.check_path(input_path)
checks.check_dir(output_path)
data = utils.read_df(input_path, ",", 0, 0)
data = data.sum(axis=1).to_frame(name=column_name)
utils.save_df(data, output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(4, __doc__)
collapse_columns(sys.argv[1], sys.argv[2], sys.argv[3])
@@ -0,0 +1,131 @@
"""
Logging setup and helper functions for processing events.
Configures logging to file and console, with functions
to log start, completion, and saving steps of processing.
Containing logs:
- processing
- processed
- read
- read error
- saved
- saved error
- parameter
- intensive
"""
import logging
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(levelname)s - %(message)s",
handlers=[
logging.FileHandler("log.txt"),
logging.StreamHandler()
]
)
def log_processing(tool_name):
"""
Logs the start of processing for a given tool.
Args:
tool_name (str): Name of the tool being processed.
Returns:
None
"""
logging.info("Processing %s:", tool_name)
def log_processed(tool_name):
"""
Logs the successful completion of processing for a given tool.
Args:
tool_name (str): Name of the tool that has been processed.
Returns:
None
"""
logging.info("Processed: %s", tool_name)
def log_read(path):
"""
Logs the reading of a file from a given path.
Arguments:
path (str): Path to file which were read.
Returns:
None
"""
logging.info("Read: %s", path)
def log_read_error(path):
"""
Logs the reading error of a given path.
Arguments:
path (str): Path were the file should be read.
Returns:
None
"""
logging.error("Failed to read: %s", path)
def log_saved(path):
"""
Logs the saving of output to a given path.
Args:
path (str): Path where the output was saved.
Returns:
None
"""
logging.info("Saved: %s", path)
def log_saved_error(path):
"""
Logs the saving error of output to a given path.
Arguments:
path (str): Path were the output should be saved.
Returns:
None
"""
logging.error("Failed to save: %s", path)
def log_parameter(parameter_name, parameter_value):
"""
Logs a parameter.
Arguments:
parameter_name (str): Parameter name with which the tool is called.
parameter_value (str): Parameter value with which the tool is called.
Returns:
None
"""
logging.info("Tool runs with %s: %s", parameter_name, parameter_value)
def log_intensive():
"""
Logs warning for resource intensiveness.
Arguments:
None
Returns:
None
"""
logging.warning("Tool is resource intensive")
@@ -0,0 +1,65 @@
"""
This script combines monthly sunshine duration data from twelve CSV files
in a given directory by extracting the German average values and merging
them into a single DataFrame. The combined data is then saved as a CSV file.
Original datasets from:
https://opendata.dwd.de/climate_environment/CDC/regional_averages_DE/monthly/sunshine_duration/
Usage:
python merge_series.py <input_directory> <output_path>
Arguments:
input_directory (Path or str): Directory containing the twelve monthly CSV input files.
output_path (Path or str): Path where the combined CSV output file will be saved.
"""
import os
import sys
import pandas as pd
import utils
import checks
import logs
def merge_series(input_directory, output_path):
"""
Extracts the German average sunshine duration series from twelve monthly CSV files
in the specified directory and combines them into a single pandas DataFrame.
The DataFrame columns represent months, and rows represent years.
Args:
input_directory (Path or str): Directory containing the input CSV files.
save_path (Path or str): File path to save the combined CSV output.
Returns:
None: The resulting combined dataframe is saved to `output_path`
"""
logs.log_processing(os.path.basename(__file__))
checks.check_path(input_directory)
checks.check_dir(output_path)
files = os.listdir(input_directory)
checks.check_empty(files)
txt_files = sorted([f for f in files if f.lower().endswith('.txt')])
series_list = []
month_list = ["January", "February", "March", "April", "May", "June",
"July", "August", "September", "October", "November", "December"]
for file, month in zip(txt_files, month_list):
data = utils.read_df(input_directory+"/"+file, ";", 1, 0)
data = data["Deutschland"]
data.name = month
series_list.append(data)
df = pd.concat(series_list, axis=1)
df = df[df.index != 2025]
utils.save_df(df, output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(3, __doc__)
merge_series(sys.argv[1], sys.argv[2])
@@ -0,0 +1,83 @@
'''
This script visualizes changes in energy production or sunshine duration,
either monthly or yearly, using line plots based on a CSV input file.
Depending on the structure of the input data, the script creates:
- A monthly line plot for a selected year if the CSV contains one row per year
and one column per month.
- A yearly line plot if the CSV contains only two columns: year and value.
Usage:
python plot_change.py <input_path> <output_path> <title> <year_filter>
Arguments:
input_path (Path or str): Path to the input CSV file containing energy or sunshine data.
output_path (Path or str): File path where the generated plot (e.g., PNG) will be saved.
title (str): Title of the plot.
year_filter (str or int): Year to select from the dataset for monthly plots.
Required if the input contains monthly values.
'''
import os
import sys
import matplotlib.pyplot as plt
import utils
import checks
import logs
def plot_change(input_path, output_path, title, year_filter=None):
"""
Generate and save a line plot showing energy or sunshine changes over time.
If the input contains multiple columns (i.e., monthly values), this function will
extract the row corresponding to `year_filter` and plot the monthly trend.
Otherwise, it assumes the data is already in a year-to-value format.
Args:
input_path (Path or str): Path to the input CSV file.
output_path (Path or str): Path to save the generated plot image.
title (str): Title of the plot.
year_filter (str or int, optional): Year to filter for when plotting monthly data.
Returns:
None: The generated diagram is saved as an image file at output_path.
"""
logs.log_processing(os.path.basename(__file__))
checks.check_path(input_path)
checks.check_dir(output_path)
checks.check_empty(title)
data = utils.read_df(input_path, ",", 0, None)
if data.shape[1] > 2:
x_label = "Month"
y_label = "Energy [kWh]"
data = data.set_index("year")
x_ticks = data.columns
x_values = range(len(data.columns))
y_value = data.transpose()[int(year_filter)]
else:
x_label = data.columns[0]
y_label = data.columns[-1]
x_ticks = data.iloc[:, 0].values.astype(str)
x_values = range(len(x_ticks))
y_value = data.iloc[:, 1].values
figure, ax = plt.subplots()
ax.set_title(title)
ax.set_xlabel(x_label)
ax.set_ylabel(y_label)
ax.set_xticks(range(len(x_ticks)))
ax.set_xticklabels(x_ticks, rotation=45)
ax.plot(x_values, y_value, 'o:')
figure.tight_layout()
plt.savefig(output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(5, __doc__)
plot_change(sys.argv[1], sys.argv[2], sys.argv[3], sys.argv[4])
@@ -0,0 +1,61 @@
"""
This script reads geographic data files, creates a map visualization by
plotting the data on top of a base map, and saves the resulting map as
an image file.
Usage:
python plot_geo.py <input_path> <base_path> <output_path> <title>
Arguments:
input_path (str or Path): Path to the geographic data file to be plotted.
base_path (str or Path): Path to the base map geographic data file.
output_path (str or Path): Path where the generated map image will be saved.
title (str): Title for the map.
"""
import os
import sys
import matplotlib.pyplot as plt
import utils
import checks
import logs
def plot_geo(input_path, base_path, output_path, title):
"""
Generate a map image by plotting geographic data over a base map.
Arguments:
input_path (str): Path to the geographic data file to be plotted.
base_path (str): Path to the base map geographic data file.
output_path (str): Path where the generated map image will be saved.
title (str): Title of the map for the plot.
Returns:
None: The generated map is saved as an image file at output_path.
"""
logs.log_processing(os.path.basename(__file__))
logs.log_parameter("title", str(title))
checks.check_path(input_path)
checks.check_path(base_path)
checks.check_dir(output_path)
checks.check_empty(title)
data = utils.read_gdf(input_path)
base = utils.read_gdf(base_path)
_, ax = plt.subplots()
ax.set_title(title)
ax.set_xlabel("Longitude")
ax.set_ylabel("Latitude")
base.plot(ax=ax, color="white", edgecolor="black")
data.plot(ax=ax, color="blue", edgecolor="blue")
plt.savefig(output_path)
logs.log_saved(output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(5, __doc__)
plot_geo(sys.argv[1], sys.argv[2], sys.argv[3], sys.argv[4])
@@ -0,0 +1,66 @@
"""
This script calculates the total area of all geometries in a geospatial file.
It uses the appropriate UTM projection to ensure accurate area calculation.
The result is written to a plain text file in either square kilometers (default)
or square meters, depending on the specified unit.
Usage:
python polygons2area.py <input_path> <output_path> <unit>
Arguments:
input_path (str or Path): Path to the input geospatial file (e.g., GeoPackage, Shapefile).
output_path (str or Path): Path to the file where the total area will be saved as a string.
unit (str): Unit for output area: 'km' (square kilometers) or
'm' (square meters).
"""
import os
import sys
import utils
import checks
import logs
def polygons2area(input_path, output_path, unit):
"""
Computes the total area of polygons in a geospatial file and saves the result.
The geometry is first projected into the appropriate UTM CRS for accurate area calculation.
The total area is then summed and converted to the specified unit.
Args:
input_path (str or Path): Path to the input geospatial file.
output_path (str or Path): Path to the output text file.
unit (str): Unit for the area value: 'km' for square kilometers or 'm' for square meters.
Returns:
None: The total area is written as a string to the specified output_path.
"""
logs.log_processing(os.path.basename(__file__))
logs.log_parameter("unit", str(unit))
checks.check_path(input_path)
checks.check_dir(output_path)
checks.check_empty(unit)
checks.check_member(unit, ["m", "km"])
data = utils.read_gdf(input_path)
utm_crs = data.estimate_utm_crs()
projected = data.to_crs(utm_crs)
area_m2 = projected.area.sum()
if unit == "km":
area = area_m2 / 1_000_000
else:
area = area_m2
area_str = f"{area:.2f}"
utils.save_file(area_str, output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(4, __doc__)
polygons2area(sys.argv[1], sys.argv[2], sys.argv[3])
@@ -0,0 +1,58 @@
"""
This script trims rows from the beginning and end of a CSV file and saves the result.
It reads a CSV file into a pandas DataFrame, removes a specified number of rows from
the top and bottom, and writes the trimmed DataFrame to a new file.
Usage:
python trim.py <input_path> <output_path> <beginning> <ending>
Arguments:
input_path (str or Path): Path to the input CSV file.
output_path (str or Path): Path where the trimmed CSV will be saved.
beginning (int or str): Number of rows to trim from the beginning (must be >= 0).
ending (int or str): Number of rows to trim from the end (must be >= 0).
"""
import os
import sys
import utils
import checks
import logs
def trim(input_path, output_path, beginning, ending):
"""
Trim rows from the beginning and end of a DataFrame loaded from a CSV file.
The function reads the data, trims the specified number of rows from both
the top and bottom, and saves the result to a CSV file.
Args:
input_path (str): Path to the input CSV file.
output_path (str): Path where the trimmed CSV file will be saved.
beginning (int or str): Number of rows to remove from the start.
ending (int or str): Number of rows to remove from the end.
Returns:
None: The trimmed data is written to the file specified by `output_path`.
"""
logs.log_processing(os.path.basename(__file__))
logs.log_parameter("beginning", str(beginning))
logs.log_parameter("ending", str(ending))
checks.check_path(input_path)
checks.check_dir(output_path)
checks.check_empty(beginning)
checks.check_empty(ending)
beginning = int(beginning)
ending = int(ending)
data = utils.read_df(input_path, ",", 0, 0)
data = data.iloc[beginning:len(data)-ending]
utils.save_df(data, output_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(5, __doc__)
trim(sys.argv[1], sys.argv[2], sys.argv[3], sys.argv[4])
@@ -0,0 +1,216 @@
"""
Utility functions for solalytics.
"""
import requests
import pandas as pd
import geopandas as gpd
from bs4 import BeautifulSoup
import checks
import logs
def is_number(s):
"""
Checks if given string is a number.
Arguemtns:
s (str): String to be checked.
Returns:
Bool: if string is number or not.
"""
try:
float(s)
return True
except ValueError:
return False
def parse_number(content):
"""
Parses number from file content.
Arguments:
content (str): file content to be parsed.
Returns:
float/int: parsed number.
"""
content = content.strip()
if is_number(content):
return float(content) if '.' in content else int(content)
raise ValueError("The file does not contain a valid number.")
def request_response(url):
"""
Request a reponse from a URL.
Arguments:
url (str): URL to be requested.
Returns:
response (str): response from the URL.
"""
checks.check_empty(url)
response = requests.get(url, timeout=2.5)
checks.check_empty(response)
return response
def parse_html(response):
"""
Parses HTML from a web response.
Arguments:
response (str): Response to be parsed.
Returns:
html: parsed HTML.
"""
html = BeautifulSoup(response.text, 'html.parser')
checks.check_empty(html)
return html
def read_file(path):
"""
Read a file and return its contents.
Args:
path (str): Path to the input file.
Returns:
str: The loaded content.
Raises:
FileNotFoundError: If the file does not exist.
OSError: If the file cannot be opened or read.
"""
checks.check_path(path)
try:
with open(path, 'r', encoding="utf-8") as file:
content = file.read().strip()
checks.check_empty(content)
logs.log_read(path)
return content
except OSError:
logs.log_read_error(path)
raise
def save_file(file, path):
"""
Saves fiven file as file to the specified path.
Args:
file: content of the file
path: The target file path where the file will be written.
"""
checks.check_empty(file)
checks.check_dir(path)
try:
with open(path, 'w', encoding="utf-8") as f:
f.write(file)
logs.log_saved(path)
except OSError:
logs.log_saved_error(path)
raise
def read_df(path, seperator, header_line, icol):
"""
Read a csv file and return a DataFrame.
Args:
path (str): Path to the input file.
seperator (str): csv seperator
Returns:
dataframe: The loaded content.
Raises:
FileNotFoundError: If the file does not exist.
ValueError: If dataframe is empty
OSError: If the file cannot be opened or read.
"""
checks.check_path(path)
try:
data = pd.read_csv(path, sep=seperator,
header=header_line, index_col=icol)
checks.check_df(data)
logs.log_read(path)
return data
except OSError:
logs.log_read_error(path)
raise
def save_df(df, path, index_input=True):
"""
Saves the given DataFrame as a CSV file to the specified path.
Args:
df : The DataFrame to save.
path : The target file path where the CSV will be written.
Raises:
FileNotFoundError: If the target directory does not exist.
Exception: If saving the file fails.
"""
checks.check_df(df)
checks.check_dir(path)
try:
df.to_csv(path, index=index_input)
logs.log_saved(path)
except OSError:
logs.log_saved_error(path)
def read_gdf(path):
"""
Read a geopackage and return a GeoDataFrame.
Args:
path (str): Path to the input file.
Returns:
geodataframe: The loaded content.
Raises:
FileNotFoundError: If the file does not exist.
ValueError: If dataframe is empty
OSError: If the file cannot be opened or read.
"""
checks.check_path(path)
try:
data = gpd.read_file(path)
checks.check_gdf(data)
logs.log_read(path)
return data
except OSError:
logs.log_read_error(path)
raise
def save_gdf(gdf, path):
"""
Saves the given GeoDataFrame as a GeoPackage to the specified path.
Arguments:
gdf: The GeoDataFrame to save.
path: The target file path where the .gpkg will be written.
Raises:
FileNotFoundError: If the target directory does not exist.
Exception: If saving the file fails.
"""
checks.check_gdf(gdf)
checks.check_dir(path)
try:
gdf.to_file(path, driver='gpkg')
logs.log_saved(path)
except OSError:
logs.log_saved_error(path)
@@ -0,0 +1,96 @@
"""
This script downloads all files from a given online directory URL and saves them
to a specified local directory.
It parses the provided URL for downloadable files (ignores subdirectories) and writes
the content of each file into the output directory.
Usage:
python wgetdir.py <directory_url> <directory_path>
Arguments:
directory_url (str or Path): URL to an online directory.
directory_path (str or Path): Path to a local directory where files will be saved.
"""
import sys
import os
from urllib.parse import urljoin
import utils
import checks
import logs
def parse_urls(html):
'''
Fetches all valid URLs from anchor tags in an HTML document.
Arguments:
html (BeautifulSoup): The parsed HTML using BeautifulSoup.
Returns:
List[str]: A list of URLs (hrefs) found in the HTML.
'''
urls = []
for anchor in html.find_all('a', href=True):
href = anchor['href']
if href in ('../', '/'):
continue
if href.endswith('/'):
continue
urls.append(href)
return urls
def wget(url, dir_path):
"""
Download a file form a URL and saves it to a local directory.
Arguments:
url (str): The URL of the online file.
dir_path (str): The local directory path where the file will be saved.
Returns:
None: The downloaded content is written to a file in `dir_path`.
"""
checks.check_empty(url)
checks.check_dir(dir_path)
name = url.split('/')[-1]
path = os.path.join(dir_path, name)
content = utils.request_response(url)
utils.save_file(content.text, path)
def wgetdir(directory_url, directory_path):
"""
Downloads all files from an online directory URL and saves them into a local directory.
Arguments:
directory_url (str): The URL of the online directory containing files.
directory_path (str): The local directory path where files will be saved.
Returns:
None: All downloadable files in the directory are saved to `directory_path`.
"""
logs.log_processing(os.path.basename(__file__))
logs.log_parameter("directory_url", str(directory_url))
checks.check_empty(directory_url)
checks.check_dir(directory_path)
os.makedirs(directory_path, exist_ok=True)
response = utils.request_response(directory_url)
html = utils.parse_html(response)
urls = parse_urls(html)
for url in urls:
url = urljoin(directory_url, url)
wget(url, directory_path)
logs.log_processed(os.path.basename(__file__))
if __name__ == "__main__":
checks.check_args(3, __doc__)
wgetdir(sys.argv[1], sys.argv[2])