freqtrade_origin/freqtrade/freqai/prediction_models/ReinforcementLearner.py

import logging
from pathlib import Path
from typing import Any, Dict, List, Optional, Type

import torch as th
from stable_baselines3.common.callbacks import ProgressBarCallback

from freqtrade.freqai.data_kitchen import FreqaiDataKitchen
from freqtrade.freqai.RL.Base5ActionRLEnv import Actions, Base5ActionRLEnv, Positions
from freqtrade.freqai.RL.BaseEnvironment import BaseEnvironment
from freqtrade.freqai.RL.BaseReinforcementLearningModel import BaseReinforcementLearningModel


logger = logging.getLogger(__name__)


class ReinforcementLearner(BaseReinforcementLearningModel):
    """
    Reinforcement Learning Model prediction model.

    Users can inherit from this class to make their own RL model with custom
    environment/training controls. Define the file as follows:

    ```
    from freqtrade.freqai.prediction_models.ReinforcementLearner import ReinforcementLearner

    class MyCoolRLModel(ReinforcementLearner):
    ```

    Save the file to `user_data/freqaimodels`, then run it with:

    freqtrade trade --freqaimodel MyCoolRLModel --config config.json --strategy SomeCoolStrat

    Here the users can override any of the functions
    available in the `IFreqaiModel` inheritance tree. Most importantly for RL, this
    is where the user overrides `MyRLEnv` (see below), to define custom
    `calculate_reward()` function, or to override any other parts of the environment.

    This class also allows users to override any other part of the IFreqaiModel tree.
    For example, the user can override `def fit()` or `def train()` or `def predict()`
    to take fine-tuned control over these processes.

    Another common override may be `def data_cleaning_predict()` where the user can
    take fine-tuned control over the data handling pipeline.
    """

    def fit(self, data_dictionary: Dict[str, Any], dk: FreqaiDataKitchen, **kwargs):
        """
        User customizable fit method
        :param data_dictionary: dict = common data dictionary containing all train/test
            features/labels/weights.
        :param dk: FreqaiDatakitchen = data kitchen for current pair.
        :return:
        model Any = trained model to be used for inference in dry/live/backtesting
        """
        train_df = data_dictionary["train_features"]
        total_timesteps = self.freqai_info["rl_config"]["train_cycles"] * len(train_df)

        policy_kwargs = dict(activation_fn=th.nn.ReLU, net_arch=self.net_arch)

        if self.activate_tensorboard:
            tb_path = Path(dk.full_path / "tensorboard" / dk.pair.split("/")[0])
        else:
            tb_path = None

        if dk.pair not in self.dd.model_dictionary or not self.continual_learning:
            model = self.MODELCLASS(
                self.policy_type,
                self.train_env,
                policy_kwargs=policy_kwargs,
                tensorboard_log=tb_path,
                **self.freqai_info.get("model_training_parameters", {}),
            )
        else:
            logger.info(
                "Continual training activated - starting training from previously " "trained agent."
            )
            model = self.dd.model_dictionary[dk.pair]
            model.set_env(self.train_env)
        callbacks: List[Any] = [self.eval_callback, self.tensorboard_callback]
        progressbar_callback: Optional[ProgressBarCallback] = None
        if self.rl_config.get("progress_bar", False):
            progressbar_callback = ProgressBarCallback()
            callbacks.insert(0, progressbar_callback)

        try:
            model.learn(
                total_timesteps=int(total_timesteps),
                callback=callbacks,
            )
        finally:
            if progressbar_callback:
                progressbar_callback.on_training_end()

        if Path(dk.data_path / "best_model.zip").is_file():
            logger.info("Callback found a best model.")
            best_model = self.MODELCLASS.load(dk.data_path / "best_model")
            return best_model

        logger.info("Couldn't find best model, using final model instead.")

        return model

    MyRLEnv: Type[BaseEnvironment]

    class MyRLEnv(Base5ActionRLEnv):  # type: ignore[no-redef]
        """
        User can override any function in BaseRLEnv and gym.Env. Here the user
        sets a custom reward based on profit and trade duration.
        """

        def calculate_reward(self, action: int) -> float:
            """
            An example reward function. This is the one function that users will likely
            wish to inject their own creativity into.

                        Warning!
            This is function is a showcase of functionality designed to show as many possible
            environment control features as possible. It is also designed to run quickly
            on small computers. This is a benchmark, it is *not* for live production.

            :param action: int = The action made by the agent for the current candle.
            :return:
            float = the reward to give to the agent for current step (used for optimization
                of weights in NN)
            """
            # first, penalize if the action is not valid
            if not self._is_valid(action):
                self.tensorboard_log("invalid", category="actions")
                return -2

            pnl = self.get_unrealized_profit()
            factor = 100.0

            # reward agent for entering trades
            if action == Actions.Long_enter.value and self._position == Positions.Neutral:
                return 25
            if action == Actions.Short_enter.value and self._position == Positions.Neutral:
                return 25
            # discourage agent from not entering trades
            if action == Actions.Neutral.value and self._position == Positions.Neutral:
                return -1

            max_trade_duration = self.rl_config.get("max_trade_duration_candles", 300)
            trade_duration = self._current_tick - self._last_trade_tick  # type: ignore

            if trade_duration <= max_trade_duration:
                factor *= 1.5
            elif trade_duration > max_trade_duration:
                factor *= 0.5

            # discourage sitting in position
            if (
                self._position in (Positions.Short, Positions.Long)
                and action == Actions.Neutral.value
            ):
                return -1 * trade_duration / max_trade_duration

            # close long
            if action == Actions.Long_exit.value and self._position == Positions.Long:
                if pnl > self.profit_aim * self.rr:
                    factor *= self.rl_config["model_reward_parameters"].get("win_reward_factor", 2)
                return float(pnl * factor)

            # close short
            if action == Actions.Short_exit.value and self._position == Positions.Short:
                if pnl > self.profit_aim * self.rr:
                    factor *= self.rl_config["model_reward_parameters"].get("win_reward_factor", 2)
                return float(pnl * factor)

            return 0.0
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00			`import logging`
refactor environment inheritence tree to accommodate flexible action types/counts. fix bug in train profit handling 2022-08-28 17:21:57 +00:00			`from pathlib import Path`
Improve logic for progressbarcallback handling 2023-10-15 09:20:25 +00:00			`from typing import Any, Dict, List, Optional, Type`
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00
			`import torch as th`
Properly close out progressbarCallback based on suggestions provided in https://github.com/DLR-RM/stable-baselines3/issues/1645 2023-10-15 08:41:07 +00:00			`from stable_baselines3.common.callbacks import ProgressBarCallback`
refactor environment inheritence tree to accommodate flexible action types/counts. fix bug in train profit handling 2022-08-28 17:21:57 +00:00
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00			`from freqtrade.freqai.data_kitchen import FreqaiDataKitchen`
refactor environment inheritence tree to accommodate flexible action types/counts. fix bug in train profit handling 2022-08-28 17:21:57 +00:00			`from freqtrade.freqai.RL.Base5ActionRLEnv import Actions, Base5ActionRLEnv, Positions`
Fix mypy typing errors 2023-04-26 17:43:42 +00:00			`from freqtrade.freqai.RL.BaseEnvironment import BaseEnvironment`
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00			`from freqtrade.freqai.RL.BaseReinforcementLearningModel import BaseReinforcementLearningModel`
refactor environment inheritence tree to accommodate flexible action types/counts. fix bug in train profit handling 2022-08-28 17:21:57 +00:00
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00
			`logger = logging.getLogger(__name__)`


			`class ReinforcementLearner(BaseReinforcementLearningModel):`
			`"""`
improve docstring clarity about how to inherit from ReinforcementLearner, demonstrate inherittance with ReinforcementLearner_multiproc 2022-11-26 10:51:08 +00:00			`Reinforcement Learning Model prediction model.`

			`Users can inherit from this class to make their own RL model with custom`
			`environment/training controls. Define the file as follows:`

			```
			`from freqtrade.freqai.prediction_models.ReinforcementLearner import ReinforcementLearner`

			`class MyCoolRLModel(ReinforcementLearner):`
			```

			Save the file to `user_data/freqaimodels`, then run it with:

			`freqtrade trade --freqaimodel MyCoolRLModel --config config.json --strategy SomeCoolStrat`

			`Here the users can override any of the functions`
			available in the `IFreqaiModel` inheritance tree. Most importantly for RL, this
			is where the user overrides `MyRLEnv` (see below), to define custom
			`calculate_reward()` function, or to override any other parts of the environment.

			`This class also allows users to override any other part of the IFreqaiModel tree.`
			For example, the user can override `def fit()` or `def train()` or `def predict()`
			`to take fine-tuned control over these processes.`

			Another common override may be `def data_cleaning_predict()` where the user can
			`take fine-tuned control over the data handling pipeline.`
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00			`"""`

add tests. add guardrails. 2022-09-14 22:46:35 +00:00			`def fit(self, data_dictionary: Dict[str, Any], dk: FreqaiDataKitchen, **kwargs):`
improve typing, improve docstrings, ensure global tests pass 2022-09-23 17:17:27 +00:00			`"""`
			`User customizable fit method`
fix docstrings 2022-11-13 16:43:52 +00:00			`:param data_dictionary: dict = common data dictionary containing all train/test`
improve typing, improve docstrings, ensure global tests pass 2022-09-23 17:17:27 +00:00			`features/labels/weights.`
fix docstrings 2022-11-13 16:43:52 +00:00			`:param dk: FreqaiDatakitchen = data kitchen for current pair.`
			`:return:`
			`model Any = trained model to be used for inference in dry/live/backtesting`
improve typing, improve docstrings, ensure global tests pass 2022-09-23 17:17:27 +00:00			`"""`
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00			`train_df = data_dictionary["train_features"]`
			`total_timesteps = self.freqai_info["rl_config"]["train_cycles"] * len(train_df)`

ruff format: freqai 2024-05-12 15:12:20 +00:00			`policy_kwargs = dict(activation_fn=th.nn.ReLU, net_arch=self.net_arch)`
add continual retraining feature, handly mypy typing reqs, improve docstrings 2022-08-24 10:54:02 +00:00
deactivate tensorboard by default 2023-05-14 14:39:23 +00:00			`if self.activate_tensorboard:`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`tb_path = Path(dk.full_path / "tensorboard" / dk.pair.split("/")[0])`
deactivate tensorboard by default 2023-05-14 14:39:23 +00:00			`else:`
			`tb_path = None`

fix multiproc callback, add continual learning to multiproc, fix totalprofit bug in env, set eval_freq automatically, improve default reward 2022-08-25 09:46:18 +00:00			`if dk.pair not in self.dd.model_dictionary or not self.continual_learning:`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`model = self.MODELCLASS(`
			`self.policy_type,`
			`self.train_env,`
			`policy_kwargs=policy_kwargs,`
			`tensorboard_log=tb_path,`
			`**self.freqai_info.get("model_training_parameters", {}),`
			`)`
add continual retraining feature, handly mypy typing reqs, improve docstrings 2022-08-24 10:54:02 +00:00			`else:`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`logger.info(`
			`"Continual training activated - starting training from previously " "trained agent."`
			`)`
add continual retraining feature, handly mypy typing reqs, improve docstrings 2022-08-24 10:54:02 +00:00			`model = self.dd.model_dictionary[dk.pair]`
			`model.set_env(self.train_env)`
Improve logic for progressbarcallback handling 2023-10-15 09:20:25 +00:00			`callbacks: List[Any] = [self.eval_callback, self.tensorboard_callback]`
			`progressbar_callback: Optional[ProgressBarCallback] = None`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`if self.rl_config.get("progress_bar", False):`
Improve logic for progressbarcallback handling 2023-10-15 09:20:25 +00:00			`progressbar_callback = ProgressBarCallback()`
			`callbacks.insert(0, progressbar_callback)`
Properly close out progressbarCallback based on suggestions provided in https://github.com/DLR-RM/stable-baselines3/issues/1645 2023-10-15 08:41:07 +00:00
			`try:`
			`model.learn(`
			`total_timesteps=int(total_timesteps),`
			`callback=callbacks,`
			`)`
			`finally:`
Improve logic for progressbarcallback handling 2023-10-15 09:20:25 +00:00			`if progressbar_callback:`
			`progressbar_callback.on_training_end()`
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00
			`if Path(dk.data_path / "best_model.zip").is_file():`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`logger.info("Callback found a best model.")`
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00			`best_model = self.MODELCLASS.load(dk.data_path / "best_model")`
			`return best_model`

Fix wording in log msg 2023-10-15 08:40:45 +00:00			`logger.info("Couldn't find best model, using final model instead.")`
reuse callback, allow user to acces all stable_baselines3 agents via config 2022-08-20 14:35:29 +00:00
			`return model`

Fix mypy typing errors 2023-04-26 17:43:42 +00:00			`MyRLEnv: Type[BaseEnvironment]`

			`class MyRLEnv(Base5ActionRLEnv): # type: ignore[no-redef]`
improve default reward, fix bugs in environment 2022-08-24 16:32:40 +00:00			`"""`
reduce code for base use-case, ensure multiproc inherits custom env, add ability to limit ram use. 2022-08-25 17:05:51 +00:00			`User can override any function in BaseRLEnv and gym.Env. Here the user`
			`sets a custom reward based on profit and trade duration.`
improve default reward, fix bugs in environment 2022-08-24 16:32:40 +00:00			`"""`
fix generic reward, add time duration to reward 2022-08-23 12:58:38 +00:00
ensure typing, remove unsued code 2022-11-26 11:11:59 +00:00			`def calculate_reward(self, action: int) -> float:`
improve typing, improve docstrings, ensure global tests pass 2022-09-23 17:17:27 +00:00			`"""`
			`An example reward function. This is the one function that users will likely`
			`wish to inject their own creativity into.`
add disclaimers everywhere about how example strategies are meant as examples 2023-05-12 08:16:48 +00:00
			`Warning!`
			`This is function is a showcase of functionality designed to show as many possible`
			`environment control features as possible. It is also designed to run quickly`
			`on small computers. This is a benchmark, it is not for live production.`

fix docstrings 2022-11-13 16:43:52 +00:00			`:param action: int = The action made by the agent for the current candle.`
			`:return:`
improve typing, improve docstrings, ensure global tests pass 2022-09-23 17:17:27 +00:00			`float = the reward to give to the agent for current step (used for optimization`
			`of weights in NN)`
			`"""`
reduce code for base use-case, ensure multiproc inherits custom env, add ability to limit ram use. 2022-08-25 17:05:51 +00:00			`# first, penalize if the action is not valid`
			`if not self._is_valid(action):`
add tensorboard category 2023-03-11 22:32:55 +00:00			`self.tensorboard_log("invalid", category="actions")`
reduce code for base use-case, ensure multiproc inherits custom env, add ability to limit ram use. 2022-08-25 17:05:51 +00:00			`return -2`

			`pnl = self.get_unrealized_profit()`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`factor = 100.0`
reduce code for base use-case, ensure multiproc inherits custom env, add ability to limit ram use. 2022-08-25 17:05:51 +00:00
			`# reward agent for entering trades`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`if action == Actions.Long_enter.value and self._position == Positions.Neutral:`
add state/action info to callbacks 2022-12-03 10:16:04 +00:00			`return 25`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`if action == Actions.Short_enter.value and self._position == Positions.Neutral:`
reduce code for base use-case, ensure multiproc inherits custom env, add ability to limit ram use. 2022-08-25 17:05:51 +00:00			`return 25`
			`# discourage agent from not entering trades`
			`if action == Actions.Neutral.value and self._position == Positions.Neutral:`
			`return -1`

ruff format: freqai 2024-05-12 15:12:20 +00:00			`max_trade_duration = self.rl_config.get("max_trade_duration_candles", 300)`
ensure typing, remove unsued code 2022-11-26 11:11:59 +00:00			`trade_duration = self._current_tick - self._last_trade_tick # type: ignore`
reduce code for base use-case, ensure multiproc inherits custom env, add ability to limit ram use. 2022-08-25 17:05:51 +00:00
			`if trade_duration <= max_trade_duration:`
			`factor *= 1.5`
			`elif trade_duration > max_trade_duration:`
			`factor *= 0.5`

			`# discourage sitting in position`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`if (`
			`self._position in (Positions.Short, Positions.Long)`
			`and action == Actions.Neutral.value`
			`):`
reduce code for base use-case, ensure multiproc inherits custom env, add ability to limit ram use. 2022-08-25 17:05:51 +00:00			`return -1 * trade_duration / max_trade_duration`

			`# close long`
			`if action == Actions.Long_exit.value and self._position == Positions.Long:`
			`if pnl > self.profit_aim * self.rr:`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`factor *= self.rl_config["model_reward_parameters"].get("win_reward_factor", 2)`
separate RL install from general FAI install, update docs 2022-10-05 13:58:54 +00:00			`return float(pnl * factor)`
reduce code for base use-case, ensure multiproc inherits custom env, add ability to limit ram use. 2022-08-25 17:05:51 +00:00
			`# close short`
			`if action == Actions.Short_exit.value and self._position == Positions.Short:`
			`if pnl > self.profit_aim * self.rr:`
ruff format: freqai 2024-05-12 15:12:20 +00:00			`factor *= self.rl_config["model_reward_parameters"].get("win_reward_factor", 2)`
separate RL install from general FAI install, update docs 2022-10-05 13:58:54 +00:00			`return float(pnl * factor)`
add multiproc fix flake8 2022-12-03 11:30:04 +00:00
ruff format: freqai 2024-05-12 15:12:20 +00:00			`return 0.0`