import os import random import argparse import numpy as np import matplotlib.pyplot as plt os.environ["WANDB_SILENT"] = "true" # Shut up WandB #-------------------------------------------------- LOSS FUNCTION -------------------------------------------------- def compute_loss(X, y, w, lam=0.0): B = len(X) r = X @ w - y return (r @ r) / (2 * B) + lam * (w @ w) #(||Xw - y||^2)/2B + λ||w||^2 #-------------------------------------------------- SEED EVERYTHING -------------------------------------------------- def seed_everything(seed): '''Set the seed for all random generators to be the provided seed''' random.seed(seed) np.random.seed(seed) #-------------------------------------------------- CREATE DATASET -------------------------------------------------- def create_linear_dataset(): d = 5 w = np.array([4, -3, 2.5, -1, 0.5]) n_train, n_test = 6000, 2000 #------------------------- RANDOMLY GENERATE EXAMPLES ------------------------- X_train = np.random.normal(0.0, 1.0, (n_train, d)) X_test = np.random.normal(0.0, 1.0, (n_test, d)) # Create n_train & n_test number of normally distributed random examples each with d number of features #------------------------- RANDOMLY GENERATE WEIGHTED TARGETS ------------------------- y_train = X_train @ w + np.random.normal(0.0, 1.0, n_train) y_test = X_test @ w + np.random.normal(0.0, 1.0, n_test) # Create n_train & n_test number of normally distributed random examples each with d number of features # this time multiplied with our true weight vector as to create the ground truths. return X_train, X_test, y_train, y_test #-------------------------------------------------- FORMULAS -------------------------------------------------- #------------------------- GRADIENT COMPUTATION ------------------------- def compute_gradient(X, y, w, lam=0.0): B = len(X) r = X @ w - y return (X.T @ r) / B + 2 * lam * w #(X^T)(Xw - y)/(B) + 2λw #-------------------------------------------------- ARGPARSER TYPES -------------------------------------------------- '''I recently watched a Youtube video about what I refer to nowadays as "type-cheating" and as such decided to implement it (even to my own detriment)''' #------------------------- W DECLARATION ------------------------- def get_w(type='0s'): match type: case '0s': return np.zeros(5) case 'gaussian': return np.random.normal(0.0, 0.01, 5) case 'uniform': return np.random.uniform(-0.1, 0.1, 5) #------------------------- ACCEPT BOTH STRING & INTEGER VALUES ------------------------- def int_or_str(x): try: return int(x) except ValueError: return x #-------------------------------------------------- GRADIENT DESCENT -------------------------------------------------- def gradient_descent(X_train, y_train, X_test, y_test, w_init, lr=0.05, batch_size=64, weight_decay=0.0, max_iters=3000, tol=1e-6, shuffle=False, use_wandb=False ): #------------------------- INITIAL VALUE INTIALIZATION ------------------------- batch_size = len(X_train) if batch_size == 'N' else batch_size # this way I never had to declare SGB, mini-BGD, or full-BGD just batch_size. n = len(X_train) w = w_init.copy() train_loss, test_loss, w_norm = [compute_loss(X_train, y_train, w, weight_decay)], [compute_loss(X_test, y_test, w, weight_decay)], [np.linalg.norm(w)] #------------------------- PRE-TRAINING WANDB UPDATE ------------------------- if use_wandb: import wandb wandb.log({"train_loss": train_loss[-1], \ "test_loss": test_loss[-1], \ "w_norm": w_norm[-1]}) #------------------------- EPOCH LOOP ------------------------- iter = 0 while iter < max_iters: if shuffle: perm = np.random.permutation(n) X_epoch, y_epoch = X_train[perm], y_train[perm] else: X_epoch, y_epoch = X_train, y_train for i in range(0, n, batch_size): if iter >= max_iters: return w, train_loss, test_loss, w_norm X_batch = X_epoch[i:i+batch_size] y_batch = y_epoch[i:i+batch_size] grad = compute_gradient(X_batch, y_batch, w, weight_decay) w = w - lr * grad #------------------------- COMPUTE LOSS ------------------------- train_loss.append(compute_loss(X_train, y_train, w, weight_decay)) test_loss.append(compute_loss(X_test, y_test, w, weight_decay)) w_norm.append(np.linalg.norm(w)) #------------------------- POST-ITERATION WANDB UPDATE ------------------------- if use_wandb: wandb.log({"train_loss": train_loss[-1], \ "test_loss": test_loss[-1], \ "w_norm": w_norm[-1]}) iter += 1 #------------------------- POST-ITERATION TOLERANCE CHECK ------------------------- if abs(train_loss[-2] - train_loss[-1]) <= tol: return w, train_loss, test_loss, w_norm return w, train_loss, test_loss, w_norm #-------------------------------------------------- EXPERIMENTS -------------------------------------------------- '''If these experiments look weirdly formatted to you, that would be because I programmed all of this for use with bash scripts becuase why not then realized that would probably not be a good idea for this assignment so I just lazily converted them to Python.''' def run_experiment_1(use_wandb=False): print("========================= EXPERIMENT 1: WEIGHT INITIALIZATION =========================") seed = 2026 lr = 0.05 batch_size = 64 weight_decay = 0.0 w_inits = ["0s", "gaussian", "uniform"] max_iters = 3000 tol = 1e-6 shuffle = True plt.figure() for w_init in w_inits: print("---------------------------------------------------------------------------------------") seed_everything(seed) X_train, X_test, y_train, y_test = create_linear_dataset() w, train_loss, test_loss, w_norm = \ gradient_descent(X_train, y_train, X_test, y_test, get_w(w_init), lr=lr, batch_size=batch_size, weight_decay=weight_decay, max_iters=max_iters, tol=tol, shuffle=shuffle, use_wandb=use_wandb ) plt.plot(range(len(train_loss)), train_loss, label=f"train ({w_init})") plt.plot(range(len(test_loss)), test_loss, label=f"test ({w_init})") plt.xlabel("Iterations") plt.ylabel("MSE") plt.title("Experiment 1: Weight Initialization") plt.legend() plt.tight_layout() plt.show(block=False) print("=======================================================================================") def run_experiment_2(use_wandb=False): print("========================= EXPERIMENT 2: LEARNING RATE SELECTION =========================") seed = 2026 lrs = [0.01, 0.05, 0.1, 0.2] batch_size = 64 weight_decay = 0.0 w_init = "0s" max_iters = 3000 tol = 1e-6 shuffle = True plt.figure() for lr_i in lrs: print("---------------------------------------------------------------------------------------") seed_everything(seed) X_train, X_test, y_train, y_test = create_linear_dataset() w, train_loss, test_loss, w_norm = \ gradient_descent(X_train, y_train, X_test, y_test, get_w(w_init), lr=lr_i, batch_size=batch_size, weight_decay=weight_decay, max_iters=max_iters, tol=tol, shuffle=shuffle, use_wandb=use_wandb ) plt.plot(range(len(train_loss)), train_loss, label=f"lr={lr_i}") plt.xlabel("Iterations") plt.ylabel("Training MSE") plt.title("Experiment 2: Learning Rate Selection") plt.legend() plt.tight_layout() plt.show(block=False) print("========================================================================================") def run_experiment_3(use_wandb=False): print("========================= EXPERIMENT 3: BATCH SIZE COMPARISON =========================") seed = 2026 lr = 0.05 batch_sizes = [1, 16, 64, 256, 1024, 'N'] weight_decay = 0.0 w_init = "0s" max_iters = 3000 tol = 1e-6 shuffle = True plt.figure() for batch_size_i in batch_sizes: print("---------------------------------------------------------------------------------------") seed_everything(seed) X_train, X_test, y_train, y_test = create_linear_dataset() w, train_loss, test_loss, w_norm = \ gradient_descent(X_train, y_train, X_test, y_test, get_w(w_init), lr=lr, batch_size=batch_size_i, weight_decay=weight_decay, max_iters=max_iters, tol=tol, shuffle=shuffle, use_wandb=use_wandb ) plt.plot(range(len(train_loss)), train_loss, label=f"batch_size={batch_size_i}") plt.xlabel("Iterations") plt.ylabel("Training MSE") plt.title("Experiment 3: Batch Size Comparison") plt.legend() plt.tight_layout() plt.show(block=False) print("=======================================================================================") def run_experiment_4(use_wandb=False): print("========================= EXPERIMENT 4: EFFECT OF WEIGHT DECAY REGULARIZATION =========================") seed = 2026 lr = 0.05 batch_size = 64 weight_decays = [0.0, 10e-3] w_init = "0s" max_iters = 3000 tol = 1e-6 shuffle = True plt.figure() w_norm_runs = [] for weight_decay_i in weight_decays: print("---------------------------------------------------------------------------------------") seed_everything(seed) X_train, X_test, y_train, y_test = create_linear_dataset() w, train_loss, test_loss, w_norm = \ gradient_descent(X_train, y_train, X_test, y_test, get_w(w_init), lr=lr, batch_size=batch_size, weight_decay=weight_decay_i, max_iters=max_iters, tol=tol, shuffle=shuffle, use_wandb=use_wandb ) plt.plot(range(len(train_loss)), train_loss, label=f"train (λ={weight_decay_i})") plt.plot(range(len(test_loss)), test_loss, label=f"test (λ={weight_decay_i})") w_norm_runs.append((weight_decay_i, w_norm)) plt.xlabel("Iterations") plt.ylabel("MSE") plt.title("Experiment 4: Weight Decay") plt.legend() plt.tight_layout() plt.show(block=False) plt.figure() for lam, w_norm in w_norm_runs: w_norm2 = (np.array(w_norm) ** 2) plt.plot(range(len(w_norm2)), w_norm2, label=f"||w||^2 (λ={lam})") plt.xlabel("Iterations") plt.ylabel("||w||^2") plt.title("Experiment 4: Weight Decay") plt.legend() plt.tight_layout() plt.show(block=False) print("=======================================================================================================") #-------------------------------------------------- MAIN FUNCTION-------------------------------------------------- def main(): '''The main function here is currently unused seeing as I didn't catch the "submit a single file" until late in the assignment's history... As such I had/have this code implemented in a much more modularized, several file system further akin to modern AI projects.''' #-------------------------------------------------- ARGUMENT PARSER -------------------------------------------------- #------------------------- PRE-PARSE SEED ------------------------- pre = argparse.ArgumentParser(add_help=False) pre.add_argument("--seed", type=int, default=2026) seed_args, _ = pre.parse_known_args() seed_everything(seed_args.seed) # As I had just recently learned about "type-cheating" for argparser and I wanted to be fancy # so I implemented "type-cheating" for my w_init however w_init made use of random variables # and as I discovered, argparser declares all types at the time of argument parsing # as such I needed to seed_everything before declaring the types and the only way # to do so was to create a parent argument parser that only read the seed #------------------------- INITIAL PARSER W/ SEED ------------------------- parser = argparse.ArgumentParser(parents=[pre]) #------------------------- HYPERPARAMETERS ------------------------- parser.add_argument('--lr', type=float, default=0.05, help="Set a fixed learning rate for the model") parser.add_argument('--batch_size', type=int_or_str, default=64, help="Set the batch size for the model; 1 = SGD, (1, N) = mini-BGD, N = BGD") parser.add_argument('--weight_decay', type=float, default=0, help="Set the amount of L2 regularizatin to be used") parser.add_argument('--w_init', type=get_w, default='0s', help="Set the initial weight vector choices=['0s', 'gaussian', 'uniform']") parser.add_argument('--max_iters', type=int, default=3000, help="Set the maximum number of iterations") parser.add_argument('--tol', type=float, default=1e-6, help="Set the lowest tolerable loss difference across iterations") parser.add_argument('--shuffle', action='store_true', default=False, help="Shuffle the dataset before batching") parser.add_argument('--exp', type=str, help="Label which experiment the run belongs to") parser.add_argument('--use_wandb', action='store_true', default=False) # I initially did this assignment with WandB however, while I much prefer WandB I got the feeling # that you would prefer immediate access to plots not through a third party. # The WandB tracking is still fully implemented, just not where the plots are concerned. #------------------------- PARSE ARGS ------------------------- args = parser.parse_args() #------------------------- CREATE DATASET ------------------------- X_train, X_test, y_train, y_test = create_linear_dataset() #------------------------- WANDB INITIALIZATION ------------------------- if args.use_wandb: import wandb from wandb.sdk.wandb_settings import Settings wandb.init( entity="missouristateuniversity", project="csc537hw1", config=vars(args), settings=Settings(silent=True) ) #------------------------- DECALRE ARGS ------------------------- print("Args in Experiment") print(args) #------------------------- PERFORM GRADIENT DESCENT ------------------------- w, train_loss, test_loss, w_norm = \ gradient_descent(X_train, y_train, X_test, y_test, args.w_init, args.lr, args.batch_size, args.weight_decay, args.max_iters, args.tol, args.shuffle, args.use_wandb ) #-------------------------------------------------- MAIN -------------------------------------------------- if __name__ == "__main__": #main() run_experiment_1() run_experiment_2() run_experiment_3() run_experiment_4() plt.show()