diff --git a/DIRECTORY.md b/DIRECTORY.md index d6ceb04a62eb..2182471a5f0d 100644 --- a/DIRECTORY.md +++ b/DIRECTORY.md @@ -62,6 +62,7 @@ * [Generate Parentheses Iterative](backtracking/generate_parentheses_iterative.py) * [Hamiltonian Cycle](backtracking/hamiltonian_cycle.py) * [Knight Tour](backtracking/knight_tour.py) + * [M Coloring Problem](backtracking/m_coloring_problem.py) * [Match Word Pattern](backtracking/match_word_pattern.py) * [Minimax](backtracking/minimax.py) * [N Queens](backtracking/n_queens.py) @@ -108,6 +109,9 @@ ## [Blockchain](blockchain) * [Diophantine Equation](blockchain/diophantine_equation.py) + * [Merkle Tree](blockchain/merkle_tree.py) + * [Simple Blockchain](blockchain/simple_blockchain.py) + * [Simple Proof Of Work](blockchain/simple_proof_of_work.py) ## [Boolean Algebra](boolean_algebra) * [And Gate](boolean_algebra/and_gate.py) @@ -166,6 +170,7 @@ * [Porta Cipher](ciphers/porta_cipher.py) * [Rabin Miller](ciphers/rabin_miller.py) * [Rail Fence Cipher](ciphers/rail_fence_cipher.py) + * [Rc4](ciphers/rc4.py) * [Rot13](ciphers/rot13.py) * [Rsa Cipher](ciphers/rsa_cipher.py) * [Rsa Factorization](ciphers/rsa_factorization.py) @@ -180,6 +185,7 @@ * [Vernam Cipher](ciphers/vernam_cipher.py) * [Vigenere Cipher](ciphers/vigenere_cipher.py) * [Xor Cipher](ciphers/xor_cipher.py) + * [Xtea](ciphers/xtea.py) ## [Computer Vision](computer_vision) * [Cnn Classification](computer_vision/cnn_classification.py) @@ -198,6 +204,9 @@ ## [Conversions](conversions) * [Astronomical Length Scale Conversion](conversions/astronomical_length_scale_conversion.py) * [Binary To Decimal](conversions/binary_to_decimal.py) + * [Binary To Excess3](conversions/binary_to_excess3.py) + * [Binary To Gray](conversions/binary_to_gray.py) + * [Binary To Gray Code](conversions/binary_to_gray_code.py) * [Binary To Hexadecimal](conversions/binary_to_hexadecimal.py) * [Binary To Octal](conversions/binary_to_octal.py) * [Convert Number To Words](conversions/convert_number_to_words.py) @@ -205,6 +214,7 @@ * [Decimal To Binary](conversions/decimal_to_binary.py) * [Decimal To Hexadecimal](conversions/decimal_to_hexadecimal.py) * [Decimal To Octal](conversions/decimal_to_octal.py) + * [Endianness](conversions/endianness.py) * [Energy Conversions](conversions/energy_conversions.py) * [Excel Title To Column](conversions/excel_title_to_column.py) * [Hex To Bin](conversions/hex_to_bin.py) @@ -540,6 +550,7 @@ ## [Geodesy](geodesy) * [Haversine Distance](geodesy/haversine_distance.py) * [Lamberts Ellipsoidal Distance](geodesy/lamberts_ellipsoidal_distance.py) + * [Radar Target Calculation](geodesy/radar_target_calculation.py) ## [Geometry](geometry) * [Geometry](geometry/geometry.py) @@ -689,6 +700,7 @@ * [Astar](machine_learning/astar.py) * [Automatic Differentiation](machine_learning/automatic_differentiation.py) * [Data Transformations](machine_learning/data_transformations.py) + * [Dbscan](machine_learning/dbscan.py) * [Decision Tree](machine_learning/decision_tree.py) * [Dimensionality Reduction](machine_learning/dimensionality_reduction.py) * [Federated Averaging](machine_learning/federated_averaging.py) @@ -712,15 +724,19 @@ * [Loss Functions](machine_learning/loss_functions.py) * Lstm * [Lstm Prediction](machine_learning/lstm/lstm_prediction.py) + * [Mab](machine_learning/mab.py) + * [Mean Shift](machine_learning/mean_shift.py) * [Mfcc](machine_learning/mfcc.py) * [Mini Batch Gradient Descent](machine_learning/mini_batch_gradient_descent.py) * [Multilayer Perceptron Classifier](machine_learning/multilayer_perceptron_classifier.py) + * [Naive Bayes Text Classification](machine_learning/naive_bayes_text_classification.py) * [Ordinary Least Squares Regression](machine_learning/ordinary_least_squares_regression.py) * [Polynomial Regression](machine_learning/polynomial_regression.py) * [Principle Component Analysis](machine_learning/principle_component_analysis.py) * [Q Learning](machine_learning/q_learning.py) * [Random Forest Classifier](machine_learning/random_forest_classifier.py) * [Random Forest Regressor](machine_learning/random_forest_regressor.py) + * [Rmsprop](machine_learning/rmsprop.py) * [Scoring Functions](machine_learning/scoring_functions.py) * [Self Organizing Map](machine_learning/self_organizing_map.py) * [Sequential Minimum Optimization](machine_learning/sequential_minimum_optimization.py) @@ -737,6 +753,7 @@ * [Arc Length](maths/arc_length.py) * [Area](maths/area.py) * [Area Under Curve](maths/area_under_curve.py) + * [Autocorrelation](maths/autocorrelation.py) * [Average Absolute Deviation](maths/average_absolute_deviation.py) * [Average Mean](maths/average_mean.py) * [Average Median](maths/average_median.py) @@ -782,6 +799,7 @@ * [Fibonacci](maths/fibonacci.py) * [Find Max](maths/find_max.py) * [Find Min](maths/find_min.py) + * [First Fundamental Form](maths/first_fundamental_form.py) * [Floor](maths/floor.py) * [Gamma](maths/gamma.py) * [Gaussian](maths/gaussian.py) @@ -844,6 +862,7 @@ * [Square Root](maths/numerical_analysis/square_root.py) * [Weierstrass Method](maths/numerical_analysis/weierstrass_method.py) * [Odd Sieve](maths/odd_sieve.py) + * [Padovan Sequence](maths/padovan_sequence.py) * [Pell Number](maths/pell_number.py) * [Perfect Cube](maths/perfect_cube.py) * [Perfect Number](maths/perfect_number.py) @@ -873,6 +892,8 @@ * [Reverse Factorial Recursive](maths/reverse_factorial_recursive.py) * [Segmented Sieve](maths/segmented_sieve.py) * Series + * [Alternate Harmonic Series](maths/series/alternate_harmonic_series.py) + * [Alternating Harmonic Series](maths/series/alternating_harmonic_series.py) * [Arithmetic](maths/series/arithmetic.py) * [Geometric](maths/series/geometric.py) * [Geometric Series](maths/series/geometric_series.py) @@ -912,6 +933,7 @@ * [Polygonal Numbers](maths/special_numbers/polygonal_numbers.py) * [Pronic Number](maths/special_numbers/pronic_number.py) * [Proth Number](maths/special_numbers/proth_number.py) + * [Spy Number](maths/special_numbers/spy_number.py) * [Triangular Numbers](maths/special_numbers/triangular_numbers.py) * [Trimorphic Number](maths/special_numbers/trimorphic_number.py) * [Ugly Numbers](maths/special_numbers/ugly_numbers.py) diff --git a/machine_learning/mab.py b/machine_learning/mab.py new file mode 100644 index 000000000000..49e0c08872fb --- /dev/null +++ b/machine_learning/mab.py @@ -0,0 +1,481 @@ +""" +Multi-Armed Bandit (MAB) is a problem in reinforcement learning where an agent must +learn to choose the best action from a set of actions to maximize its reward. + +learn more here: https://en.wikipedia.org/wiki/Multi-armed_bandit + + +The MAB problem can be described as follows: +- There are N arms, each with a different probability of giving a reward. +- The agent must learn to choose the best arm to pull in order to maximize its reward. + +Here 3 optimising strategies have been implemented: +- Epsilon-Greedy +- Upper Confidence Bound (UCB) +- Thompson Sampling + +There are two other strategies implemented to show the performance of +the optimising strategies: +- Random strategy (full exploration) +- Greedy strategy (full exploitation) + +The performance of the strategies is evaluated by the cumulative reward +over a number of rounds. + +""" + +from abc import ABC, abstractmethod + +import matplotlib.pyplot as plt +import numpy as np + + +class Bandit: + """ + A class to represent a multi-armed bandit. + """ + + def __init__(self, probabilities: list[float]) -> None: + """ + Initialize the bandit with a list of probabilities for each arm. + + Args: + probabilities: List of probabilities for each arm. + + Example: + >>> bandit = Bandit([0.1, 0.5, 0.9]) + >>> bandit.num_arms + 3 + """ + self.probabilities = probabilities + self.num_arms = len(probabilities) + + def pull(self, arm_index: int) -> int: + """ + Pull an arm of the bandit. + + Args: + arm_index: The arm to pull. + + Returns: + The reward for the arm. + + Example: + >>> bandit = Bandit([0.1, 0.5, 0.9]) + >>> isinstance(bandit.pull(0), int) + True + """ + rng = np.random.default_rng() + return 1 if rng.random() < self.probabilities[arm_index] else 0 + + +# Epsilon-Greedy strategy + + +class Strategy(ABC): + """ + Base class for all strategies. + """ + + @abstractmethod + def select_arm(self) -> int: + """ + Select an arm to pull. + + Returns: + The index of the arm to pull. + """ + + @abstractmethod + def update(self, arm_index: int, reward: int) -> None: + """ + Update the strategy. + + Args: + arm_index: The index of the arm to pull. + reward: The reward for the arm. + """ + + +class EpsilonGreedy(Strategy): + """ + A class for a simple implementation of the Epsilon-Greedy strategy. + Follow this link to learn more: + https://medium.com/analytics-vidhya/the-epsilon-greedy-algorithm-for-reinforcement-learning-5fe6f96dc870 + """ + + def __init__(self, epsilon: float, num_arms: int) -> None: + """ + Initialize the Epsilon-Greedy strategy. + + Args: + epsilon: The probability of exploring new arms. + num_arms: The number of arms. + """ + self.epsilon = epsilon + self.num_arms = num_arms + self.counts = np.zeros(num_arms) + self.values = np.zeros(num_arms) + + def select_arm(self) -> int: + """ + Select an arm to pull. + + Returns: + The index of the arm to pull. + + Example: + >>> strategy = EpsilonGreedy(epsilon=0.1, num_arms=3) + >>> 0 <= strategy.select_arm() < 3 + True + """ + rng = np.random.default_rng() + + if rng.random() < self.epsilon: + return int(rng.integers(self.num_arms)) + else: + return int(np.argmax(self.values)) + + def update(self, arm_index: int, reward: int) -> None: + """ + Update the strategy. + + Args: + arm_index: The index of the arm to pull. + reward: The reward for the arm. + + Example: + >>> strategy = EpsilonGreedy(epsilon=0.1, num_arms=3) + >>> strategy.update(0, 1) + >>> strategy.counts[0] == 1 + np.True_ + """ + self.counts[arm_index] += 1 + n = self.counts[arm_index] + self.values[arm_index] += (reward - self.values[arm_index]) / n + + +# Upper Confidence Bound (UCB) + + +class UCB(Strategy): + """ + A class for the Upper Confidence Bound (UCB) strategy. + Follow this link to learn more: + https://people.maths.bris.ac.uk/~maajg/teaching/stochopt/ucb.pdf + """ + + def __init__(self, num_arms: int) -> None: + """ + Initialize the UCB strategy. + + Args: + num_arms: The number of arms. + """ + self.num_arms = num_arms + self.counts = np.zeros(num_arms) + self.values = np.zeros(num_arms) + self.total_counts = 0 + + def select_arm(self) -> int: + """ + Select an arm to pull. + + Returns: + The index of the arm to pull. + + Example: + >>> strategy = UCB(num_arms=3) + >>> 0 <= strategy.select_arm() < 3 + True + """ + if self.total_counts < self.num_arms: + return self.total_counts + ucb_values = self.values + np.sqrt(2 * np.log(self.total_counts) / self.counts) + return int(np.argmax(ucb_values)) + + def update(self, arm_index: int, reward: int) -> None: + """ + Update the strategy. + + Args: + arm_index: The index of the arm to pull. + reward: The reward for the arm. + + Example: + >>> strategy = UCB(num_arms=3) + >>> strategy.update(0, 1) + >>> strategy.counts[0] == 1 + np.True_ + """ + self.counts[arm_index] += 1 + self.total_counts += 1 + n = self.counts[arm_index] + self.values[arm_index] += (reward - self.values[arm_index]) / n + + +# Thompson Sampling + + +class ThompsonSampling(Strategy): + """ + A class for the Thompson Sampling strategy. + Follow this link to learn more: + https://en.wikipedia.org/wiki/Thompson_sampling + """ + + def __init__(self, num_arms: int) -> None: + """ + Initialize the Thompson Sampling strategy. + + Args: + num_arms: The number of arms. + """ + self.num_arms = num_arms + self.successes = np.zeros(num_arms) + self.failures = np.zeros(num_arms) + + def select_arm(self) -> int: + """ + Select an arm to pull. + + Returns: + The index of the arm to pull based on the Thompson Sampling strategy + which relies on the Beta distribution. + + Example: + >>> strategy = ThompsonSampling(num_arms=3) + >>> 0 <= strategy.select_arm() < 3 + True + """ + rng = np.random.default_rng() + + samples = [ + rng.beta(self.successes[i] + 1, self.failures[i] + 1) + for i in range(self.num_arms) + ] + return int(np.argmax(samples)) + + def update(self, arm_index: int, reward: int) -> None: + """ + Update the strategy. + + Args: + arm_index: The index of the arm to pull. + reward: The reward for the arm. + + Example: + >>> strategy = ThompsonSampling(num_arms=3) + >>> strategy.update(0, 1) + >>> strategy.successes[0] == 1 + np.True_ + """ + if reward == 1: + self.successes[arm_index] += 1 + else: + self.failures[arm_index] += 1 + + +# Random strategy (full exploration) +class RandomStrategy(Strategy): + """ + A class for choosing an arm uniformly at random at each round to give + a better comparison with the other optimised strategies. + """ + + def __init__(self, num_arms: int) -> None: + """ + Initialize the Random strategy. + + Args: + num_arms: The number of arms. + """ + self.num_arms = num_arms + + def select_arm(self) -> int: + """ + Select an arm to pull. + + Returns: + The index of the arm to pull. + + Example: + >>> strategy = RandomStrategy(num_arms=3) + >>> 0 <= strategy.select_arm() < 3 + True + """ + rng = np.random.default_rng() + return int(rng.integers(self.num_arms)) + + def update(self, arm_index: int, reward: int) -> None: + """ + Update the strategy. + + Args: + arm_index: The index of the arm to pull. + reward: The reward for the arm. + + Example: + >>> strategy = RandomStrategy(num_arms=3) + >>> strategy.update(0, 1) + """ + + +# Greedy strategy (full exploitation) + + +class GreedyStrategy(Strategy): + """ + A class for the Greedy strategy to show how full exploitation can be + detrimental to the performance of the strategy. + """ + + def __init__(self, num_arms: int) -> None: + """ + Initialize the Greedy strategy. + + Args: + num_arms: The number of arms. + """ + self.num_arms = num_arms + self.counts = np.zeros(num_arms) + self.values = np.zeros(num_arms) + + def select_arm(self) -> int: + """ + Select an arm to pull. + + Returns: + The index of the arm to pull. + + Example: + >>> strategy = GreedyStrategy(num_arms=3) + >>> 0 <= strategy.select_arm() < 3 + True + """ + return int(np.argmax(self.values)) + + def update(self, arm_index: int, reward: int) -> None: + """ + Update the strategy. + + Args: + arm_index: The index of the arm to pull. + reward: The reward for the arm. + + Example: + >>> strategy = GreedyStrategy(num_arms=3) + >>> strategy.update(0, 1) + >>> strategy.counts[0] == 1 + np.True_ + """ + self.counts[arm_index] += 1 + n = self.counts[arm_index] + self.values[arm_index] += (reward - self.values[arm_index]) / n + + +def test_mab_strategies() -> None: + """ + Deterministic behavioural tests for the MAB strategies. + + These checks feed each strategy a fixed sequence of rewards and assert + on the resulting internal state and arm selection, so a regression in + the update/select logic will fail the suite instead of only being + visible in the (stochastic) plotted demo. + """ + num_arms = 3 + + # After repeatedly rewarding arm 2, a purely greedy strategy must + # settle on arm 2. + greedy = GreedyStrategy(num_arms=num_arms) + for _ in range(10): + greedy.update(2, 1) + greedy.update(0, 0) + greedy.update(1, 0) + assert greedy.select_arm() == 2 + + # Epsilon-Greedy with epsilon=0 behaves like the greedy strategy. + epsilon_greedy = EpsilonGreedy(epsilon=0.0, num_arms=num_arms) + for _ in range(10): + epsilon_greedy.update(1, 1) + epsilon_greedy.update(0, 0) + epsilon_greedy.update(2, 0) + assert epsilon_greedy.select_arm() == 1 + + # UCB must exhaustively try every arm once before repeating any of them. + ucb = UCB(num_arms=num_arms) + first_round_arms = set() + for _ in range(num_arms): + arm = ucb.select_arm() + first_round_arms.add(arm) + ucb.update(arm, 1) + assert first_round_arms == set(range(num_arms)) + + # Thompson Sampling should heavily favor an arm with only successes + # over arms with only failures. + thompson = ThompsonSampling(num_arms=num_arms) + for _ in range(20): + thompson.update(0, 1) + thompson.update(1, 0) + thompson.update(2, 0) + selections = [thompson.select_arm() for _ in range(50)] + assert selections.count(0) > len(selections) // 2 + + # RandomStrategy.update is a no-op and select_arm always returns a + # valid arm index. + random_strategy = RandomStrategy(num_arms=num_arms) + random_strategy.update(0, 1) + assert 0 <= random_strategy.select_arm() < num_arms + + +def demo_mab_strategies() -> None: + """ + Run a stochastic simulation of the MAB strategies and plot their + cumulative reward over time for visual comparison. + """ + # Simulation + num_arms = 4 + arms_probabilities = [0.1, 0.3, 0.5, 0.8] # True probabilities + + bandit = Bandit(arms_probabilities) + strategies: dict[str, Strategy] = { + "Epsilon-Greedy": EpsilonGreedy(epsilon=0.1, num_arms=num_arms), + "UCB": UCB(num_arms=num_arms), + "Thompson Sampling": ThompsonSampling(num_arms=num_arms), + "Full Exploration(Random)": RandomStrategy(num_arms=num_arms), + "Full Exploitation(Greedy)": GreedyStrategy(num_arms=num_arms), + } + + num_rounds = 1000 + results = {} + + for name, strategy in strategies.items(): + rewards = [] + total_reward = 0 + for _ in range(num_rounds): + arm = strategy.select_arm() + current_reward = bandit.pull(arm) + strategy.update(arm, current_reward) + total_reward += current_reward + rewards.append(total_reward) + results[name] = rewards + + # Plotting results + plt.figure(figsize=(12, 6)) + for name, rewards in results.items(): + plt.plot(rewards, label=name) + + plt.title("Cumulative Reward of Multi-Armed Bandit Strategies") + plt.xlabel("Round") + plt.ylabel("Cumulative Reward") + plt.legend() + plt.grid() + plt.show() + + +if __name__ == "__main__": + import doctest + + doctest.testmod() + test_mab_strategies() + demo_mab_strategies()