''' This code uses XGBoost egression model on for Multivariate Inputs (Predicting a Single Target with 5 features). ''' import time import matplotlib.pyplot as plt import numpy as np import pandas as pd from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score from sklearn.model_selection import train_test_split from xgboost import XGBRegressor # Load CSV dataset and define feature columns and target column name start_cpu = time.process_time() df = pd.read_csv("WT_Dummy_Fatigue.csv") features = [ "wind_speed_mps", "turb_intensity", "yaw_misalignment_deg", "design_load_kN", ] target = "fatigue_life_years" X = df[features] y = df[target] # Split data into training (70%) and testing (30%) sets X_train, X_test, y_train, y_test = train_test_split( X, y, test_size=0.3, random_state=42 ) # Initialize and train the XGBoost Regressor model = XGBRegressor( n_estimators=100, learning_rate=0.1, max_depth=5, random_state=42 ) model.fit(X_train, y_train) # Make predictions on the test set y_pred = model.predict(X_test) # Evaluate model performance mae = mean_absolute_error(y_test, y_pred) r2 = r2_score(y_test, y_pred) print("\n===== XGBoost Regressor Performance") print(f"Mean Absolute Error [years]: {mae:,.2f}") print(f"R-squared Score: {r2:.4f}") # Plot Predicted vs. Actual values plt.figure(figsize=(7, 6)) plt.scatter(y_test, y_pred, alpha=0.75, color="b", s=20, label="Predictions") # Perfect predictions line (diagonal) lims = [ min(y_test.min(), y_pred.min()), max(y_test.max(), y_pred.max()), ] plt.plot(lims, lims, "--", color="r", label="Perfect Fit") plt.xlabel("Actual Values") plt.ylabel("Predicted Values") plt.title("XGBoost Regressor: Predicted vs. Actual Values") plt.legend() plt.grid(True, linestyle=":", alpha=0.75) plt.savefig("WT_Fatigue_XGBoost.png") end_cpu = time.process_time() cpu_duration = np.round(end_cpu - start_cpu, 1) print(f"CPU time to execute ML algorithm: {cpu_duration} [s]\n")