Compare_agent.py #6

Description

@vicks4u

"""
RL + MetaTrader5 trading bot template

  • Train with historical data (PPO from stable-baselines3)
  • Optionally execute trades via MetaTrader5 (set LIVE=True to enable)
    CAVEAT: This is an educational template. Backtest & paper-trade first.
    """

import time
import numpy as np
import pandas as pd
import gym
from gym import spaces
import MetaTrader5 as mt5
from stable_baselines3 import PPO
from stable_baselines3.common.vec_env import DummyVecEnv
from stable_baselines3.common.callbacks import CheckpointCallback

-------------------------

USER CONFIG

-------------------------

SYMBOL = "EURUSD"
TIMEFRAME = mt5.TIMEFRAME_M5 # 5 minute bars
LOOKBACK = 50 # observation window (bars)
START_POS = 0 # for historical fetch offset
LOT_SIZE = 0.01 # trade lot size
LIVE = False # <-- Set to True only after full testing (demo account first!)
MODEL_PATH = "ppo_mt5_model"
TRAIN_TIMESTEPS = 20000 # adjust as you like

-------------------------

-------------------------

Helper: connect to MT5

-------------------------

def mt5_connect():
if not mt5.initialize():
raise RuntimeError(f"MT5 initialize() failed, error={mt5.last_error()}")
info = mt5.terminal_info()
if info is None:
raise RuntimeError("Failed to get terminal info after initialize()")
print("MT5 terminal initialized:", info.product)
# Ensure symbol is available
if not mt5.symbol_select(SYMBOL, True):
raise RuntimeError(f"Failed to select symbol {SYMBOL}")
return True

def mt5_shutdown():
mt5.shutdown()

-------------------------

Get historical OHLCV

-------------------------

def fetch_bars(symbol, timeframe, n_bars):
# copy_rates_from_pos returns numpy array with fields: time, open, high, low, close, tick_volume, ...
rates = mt5.copy_rates_from_pos(symbol, timeframe, START_POS, n_bars)
if rates is None:
raise RuntimeError(f"Failed to fetch rates for {symbol}: {mt5.last_error()}")
df = pd.DataFrame(rates)
df['time'] = pd.to_datetime(df['time'], unit='s')
return df

-------------------------

Simple trading Gym env

-------------------------

class MT5TradingEnv(gym.Env):
"""
Observation: last LOOKBACK closes normalized + current position (0/1/-1)
Actions: 0=hold, 1=buy (long), 2=sell (short/close long)
Reward: change in account equity approximated by price moves * position
NOTE: Simplified; this is a research template, not production-ready.
"""
def init(self, df: pd.DataFrame, lookback=LOOKBACK):
super(MT5TradingEnv, self).init()
self.df = df.reset_index(drop=True)
self.lookback = lookback
self.ptr = lookback # current index in df
self.position = 0 # -1 short, 0 flat, 1 long
self.entry_price = 0.0
# Observations: lookback closes (normalized) + position
self.observation_space = spaces.Box(low=-np.inf, high=np.inf, shape=(lookback + 1,), dtype=np.float32)
# Actions: hold(0), buy(1), sell(2)
self.action_space = spaces.Discrete(3)

def _get_obs(self):
closes = self.df.loc[self.ptr - self.lookback:self.ptr - 1, "close"].values.astype(np.float32)
# normalize closes by dividing by last close
norm = closes / (closes[-1] + 1e-9) - 1.0
obs = np.concatenate([norm, np.array([float(self.position)])], axis=0)
return obs
def reset(self):
self.ptr = self.lookback
self.position = 0
self.entry_price = 0.0
return self._get_obs()
def step(self, action):
done = False
reward = 0.0
price = float(self.df.loc[self.ptr, "close"])
# Action logic
if action == 1: # buy
if self.position == 0:
self.position = 1
self.entry_price = price
elif self.position == -1:
# close short and go long
reward += (self.entry_price - price) # profit from short
self.position = 1
self.entry_price = price
elif action == 2: # sell
if self.position == 0:
self.position = -1
self.entry_price = price
elif self.position == 1:
reward += (price - self.entry_price) # profit from long
self.position = -1
self.entry_price = price
# Move pointer
self.ptr += 1
if self.ptr >= len(self.df):
done = True
else:
# reward can also be shaped by unrealized pnl:
next_price = float(self.df.loc[self.ptr, "close"])
unrealized = 0.0
if self.position == 1:
unrealized = next_price - self.entry_price
elif self.position == -1:
unrealized = self.entry_price - next_price
# small per-step reward = unrealized PnL scaled
reward += unrealized * 0.1
obs = self._get_obs() if not done else np.zeros(self.observation_space.shape, dtype=np.float32)
info = {"ptr": self.ptr}
return obs, float(reward), done, info

-------------------------

Order helpers

-------------------------

def send_order(symbol, action, lot=LOT_SIZE, deviation=20):
"""
action: 1=buy, 2=sell
This function sends a ORDER_TYPE_BUY / ORDER_TYPE_SELL market order.
Basic error checking included. For production you need more robust code.
"""
price = mt5.symbol_info_tick(symbol).ask if action == 1 else mt5.symbol_info_tick(symbol).bid
request = {
"action": mt5.TRADE_ACTION_DEAL,
"symbol": symbol,
"volume": float(lot),
"type": mt5.ORDER_TYPE_BUY if action == 1 else mt5.ORDER_TYPE_SELL,
"price": float(price),
"deviation": deviation,
"magic": 234000,
"comment": "RL-bot",
"type_filling": mt5.ORDER_FILLING_IOC,
}
result = mt5.order_send(request)
return result

-------------------------

Main: training flow

-------------------------

def train_agent():
mt5_connect()
# fetch historical bars
n_bars = 5000
df = fetch_bars(SYMBOL, TIMEFRAME, n_bars)
print(f"Fetched {len(df)} bars for {SYMBOL}")
# Create env
env = DummyVecEnv([lambda: MT5TradingEnv(df, lookback=LOOKBACK)])
# model
model = PPO("MlpPolicy", env, verbose=1)
# save checkpoints
cb = CheckpointCallback(save_freq=5000, save_path="./logs/", name_prefix="ppo_mt5")
model.learn(total_timesteps=TRAIN_TIMESTEPS, callback=cb)
model.save(MODEL_PATH)
mt5_shutdown()
print("Training complete, model saved to", MODEL_PATH)

-------------------------

Real-time execution loop (paper/live)

-------------------------

def run_live_loop(model_path=MODEL_PATH, poll_seconds=5):
mt5_connect()
model = PPO.load(model_path)
print("Loaded model:", model_path)
# We'll maintain a small in-memory buffer of recent bars
n_history = LOOKBACK + 10
df = fetch_bars(SYMBOL, TIMEFRAME, n_history)
# pointer is at last bar
while True:
try:
latest = fetch_bars(SYMBOL, TIMEFRAME, 1)
if latest['time'].iloc[-1] > df['time'].iloc[-1]:
# append new bar
df = pd.concat([df, latest]).reset_index(drop=True)
if len(df) > n_history:
df = df.iloc[-n_history:].reset_index(drop=True)
# Build an env instance for this single-step decision
env = MT5TradingEnv(df, lookback=LOOKBACK)
obs = env.reset()
action, _states = model.predict(obs, deterministic=True)
print(f"[{pd.to_datetime('now')}] Action: {action} | Price: {df['close'].iloc[-1]}")
# Send order if LIVE
if LIVE:
if int(action) == 1:
res = send_order(SYMBOL, 1)
print("Order send result:", res)
elif int(action) == 2:
res = send_order(SYMBOL, 2)
print("Order send result:", res)
else:
# paper-trade: just log what would happen
print("LIVE=False -> paper trade logged only")
else:
# no new bar yet
pass
time.sleep(poll_seconds)
except KeyboardInterrupt:
print("Stopping live loop (KeyboardInterrupt).")
break
except Exception as e:
print("Exception in live loop:", str(e))
time.sleep(5)
mt5_shutdown()

-------------------------

If run as script

-------------------------

if name == "main":
import argparse
parser = argparse.ArgumentParser()
parser.add_argument("--mode", choices=["train", "run"], default="train")
args = parser.parse_args()
if args.mode == "train":
print("Starting training...")
train_agent()
elif args.mode == "run":
print("Starting live/paper run loop...")
run_live_loop()

Metadata

Metadata

Assignees

No one assigned

    Labels

    No labels
    No labels

    Projects

    No projects

      Milestone

      No milestone

      Relationships

      None yet

      Development

      No branches or pull requests

      Issue actions

      , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
       blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
      }
      } catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
      })();
      (function(){
      try {
      var __m = "github.com";
      var __re = new RegExp('^' + "github\\.com" + '
      
      Skip to content

      Compare_agent.py #6

      Description

      @vicks4u

      """
      RL + MetaTrader5 trading bot template

      • Train with historical data (PPO from stable-baselines3)
      • Optionally execute trades via MetaTrader5 (set LIVE=True to enable)
        CAVEAT: This is an educational template. Backtest & paper-trade first.
        """

      import time
      import numpy as np
      import pandas as pd
      import gym
      from gym import spaces
      import MetaTrader5 as mt5
      from stable_baselines3 import PPO
      from stable_baselines3.common.vec_env import DummyVecEnv
      from stable_baselines3.common.callbacks import CheckpointCallback

      -------------------------

      USER CONFIG

      -------------------------

      SYMBOL = "EURUSD"
      TIMEFRAME = mt5.TIMEFRAME_M5 # 5 minute bars
      LOOKBACK = 50 # observation window (bars)
      START_POS = 0 # for historical fetch offset
      LOT_SIZE = 0.01 # trade lot size
      LIVE = False # <-- Set to True only after full testing (demo account first!)
      MODEL_PATH = "ppo_mt5_model"
      TRAIN_TIMESTEPS = 20000 # adjust as you like

      -------------------------

      -------------------------

      Helper: connect to MT5

      -------------------------

      def mt5_connect():
      if not mt5.initialize():
      raise RuntimeError(f"MT5 initialize() failed, error={mt5.last_error()}")
      info = mt5.terminal_info()
      if info is None:
      raise RuntimeError("Failed to get terminal info after initialize()")
      print("MT5 terminal initialized:", info.product)
      # Ensure symbol is available
      if not mt5.symbol_select(SYMBOL, True):
      raise RuntimeError(f"Failed to select symbol {SYMBOL}")
      return True

      def mt5_shutdown():
      mt5.shutdown()

      -------------------------

      Get historical OHLCV

      -------------------------

      def fetch_bars(symbol, timeframe, n_bars):
      # copy_rates_from_pos returns numpy array with fields: time, open, high, low, close, tick_volume, ...
      rates = mt5.copy_rates_from_pos(symbol, timeframe, START_POS, n_bars)
      if rates is None:
      raise RuntimeError(f"Failed to fetch rates for {symbol}: {mt5.last_error()}")
      df = pd.DataFrame(rates)
      df['time'] = pd.to_datetime(df['time'], unit='s')
      return df

      -------------------------

      Simple trading Gym env

      -------------------------

      class MT5TradingEnv(gym.Env):
      """
      Observation: last LOOKBACK closes normalized + current position (0/1/-1)
      Actions: 0=hold, 1=buy (long), 2=sell (short/close long)
      Reward: change in account equity approximated by price moves * position
      NOTE: Simplified; this is a research template, not production-ready.
      """
      def init(self, df: pd.DataFrame, lookback=LOOKBACK):
      super(MT5TradingEnv, self).init()
      self.df = df.reset_index(drop=True)
      self.lookback = lookback
      self.ptr = lookback # current index in df
      self.position = 0 # -1 short, 0 flat, 1 long
      self.entry_price = 0.0
      # Observations: lookback closes (normalized) + position
      self.observation_space = spaces.Box(low=-np.inf, high=np.inf, shape=(lookback + 1,), dtype=np.float32)
      # Actions: hold(0), buy(1), sell(2)
      self.action_space = spaces.Discrete(3)

      def _get_obs(self):
      closes = self.df.loc[self.ptr - self.lookback:self.ptr - 1, "close"].values.astype(np.float32)
      # normalize closes by dividing by last close
      norm = closes / (closes[-1] + 1e-9) - 1.0
      obs = np.concatenate([norm, np.array([float(self.position)])], axis=0)
      return obs
      def reset(self):
      self.ptr = self.lookback
      self.position = 0
      self.entry_price = 0.0
      return self._get_obs()
      def step(self, action):
      done = False
      reward = 0.0
      price = float(self.df.loc[self.ptr, "close"])
      # Action logic
      if action == 1: # buy
      if self.position == 0:
      self.position = 1
      self.entry_price = price
      elif self.position == -1:
      # close short and go long
      reward += (self.entry_price - price) # profit from short
      self.position = 1
      self.entry_price = price
      elif action == 2: # sell
      if self.position == 0:
      self.position = -1
      self.entry_price = price
      elif self.position == 1:
      reward += (price - self.entry_price) # profit from long
      self.position = -1
      self.entry_price = price
      # Move pointer
      self.ptr += 1
      if self.ptr >= len(self.df):
      done = True
      else:
      # reward can also be shaped by unrealized pnl:
      next_price = float(self.df.loc[self.ptr, "close"])
      unrealized = 0.0
      if self.position == 1:
      unrealized = next_price - self.entry_price
      elif self.position == -1:
      unrealized = self.entry_price - next_price
      # small per-step reward = unrealized PnL scaled
      reward += unrealized * 0.1
      obs = self._get_obs() if not done else np.zeros(self.observation_space.shape, dtype=np.float32)
      info = {"ptr": self.ptr}
      return obs, float(reward), done, info
      

      -------------------------

      Order helpers

      -------------------------

      def send_order(symbol, action, lot=LOT_SIZE, deviation=20):
      """
      action: 1=buy, 2=sell
      This function sends a ORDER_TYPE_BUY / ORDER_TYPE_SELL market order.
      Basic error checking included. For production you need more robust code.
      """
      price = mt5.symbol_info_tick(symbol).ask if action == 1 else mt5.symbol_info_tick(symbol).bid
      request = {
      "action": mt5.TRADE_ACTION_DEAL,
      "symbol": symbol,
      "volume": float(lot),
      "type": mt5.ORDER_TYPE_BUY if action == 1 else mt5.ORDER_TYPE_SELL,
      "price": float(price),
      "deviation": deviation,
      "magic": 234000,
      "comment": "RL-bot",
      "type_filling": mt5.ORDER_FILLING_IOC,
      }
      result = mt5.order_send(request)
      return result

      -------------------------

      Main: training flow

      -------------------------

      def train_agent():
      mt5_connect()
      # fetch historical bars
      n_bars = 5000
      df = fetch_bars(SYMBOL, TIMEFRAME, n_bars)
      print(f"Fetched {len(df)} bars for {SYMBOL}")
      # Create env
      env = DummyVecEnv([lambda: MT5TradingEnv(df, lookback=LOOKBACK)])
      # model
      model = PPO("MlpPolicy", env, verbose=1)
      # save checkpoints
      cb = CheckpointCallback(save_freq=5000, save_path="./logs/", name_prefix="ppo_mt5")
      model.learn(total_timesteps=TRAIN_TIMESTEPS, callback=cb)
      model.save(MODEL_PATH)
      mt5_shutdown()
      print("Training complete, model saved to", MODEL_PATH)

      -------------------------

      Real-time execution loop (paper/live)

      -------------------------

      def run_live_loop(model_path=MODEL_PATH, poll_seconds=5):
      mt5_connect()
      model = PPO.load(model_path)
      print("Loaded model:", model_path)
      # We'll maintain a small in-memory buffer of recent bars
      n_history = LOOKBACK + 10
      df = fetch_bars(SYMBOL, TIMEFRAME, n_history)
      # pointer is at last bar
      while True:
      try:
      latest = fetch_bars(SYMBOL, TIMEFRAME, 1)
      if latest['time'].iloc[-1] > df['time'].iloc[-1]:
      # append new bar
      df = pd.concat([df, latest]).reset_index(drop=True)
      if len(df) > n_history:
      df = df.iloc[-n_history:].reset_index(drop=True)
      # Build an env instance for this single-step decision
      env = MT5TradingEnv(df, lookback=LOOKBACK)
      obs = env.reset()
      action, _states = model.predict(obs, deterministic=True)
      print(f"[{pd.to_datetime('now')}] Action: {action} | Price: {df['close'].iloc[-1]}")
      # Send order if LIVE
      if LIVE:
      if int(action) == 1:
      res = send_order(SYMBOL, 1)
      print("Order send result:", res)
      elif int(action) == 2:
      res = send_order(SYMBOL, 2)
      print("Order send result:", res)
      else:
      # paper-trade: just log what would happen
      print("LIVE=False -> paper trade logged only")
      else:
      # no new bar yet
      pass
      time.sleep(poll_seconds)
      except KeyboardInterrupt:
      print("Stopping live loop (KeyboardInterrupt).")
      break
      except Exception as e:
      print("Exception in live loop:", str(e))
      time.sleep(5)
      mt5_shutdown()

      -------------------------

      If run as script

      -------------------------

      if name == "main":
      import argparse
      parser = argparse.ArgumentParser()
      parser.add_argument("--mode", choices=["train", "run"], default="train")
      args = parser.parse_args()
      if args.mode == "train":
      print("Starting training...")
      train_agent()
      elif args.mode == "run":
      print("Starting live/paper run loop...")
      run_live_loop()

      Metadata

      Metadata

      Assignees

      No one assigned

        Labels

        No labels
        No labels

        Projects

        No projects

          Milestone

          No milestone

          Relationships

          None yet

          Development

          No branches or pull requests

          Issue actions

          , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
          Skip to content

          Compare_agent.py #6

          Description

          @vicks4u

          """
          RL + MetaTrader5 trading bot template

          • Train with historical data (PPO from stable-baselines3)
          • Optionally execute trades via MetaTrader5 (set LIVE=True to enable)
            CAVEAT: This is an educational template. Backtest & paper-trade first.
            """

          import time
          import numpy as np
          import pandas as pd
          import gym
          from gym import spaces
          import MetaTrader5 as mt5
          from stable_baselines3 import PPO
          from stable_baselines3.common.vec_env import DummyVecEnv
          from stable_baselines3.common.callbacks import CheckpointCallback

          -------------------------

          USER CONFIG

          -------------------------

          SYMBOL = "EURUSD"
          TIMEFRAME = mt5.TIMEFRAME_M5 # 5 minute bars
          LOOKBACK = 50 # observation window (bars)
          START_POS = 0 # for historical fetch offset
          LOT_SIZE = 0.01 # trade lot size
          LIVE = False # <-- Set to True only after full testing (demo account first!)
          MODEL_PATH = "ppo_mt5_model"
          TRAIN_TIMESTEPS = 20000 # adjust as you like

          -------------------------

          -------------------------

          Helper: connect to MT5

          -------------------------

          def mt5_connect():
          if not mt5.initialize():
          raise RuntimeError(f"MT5 initialize() failed, error={mt5.last_error()}")
          info = mt5.terminal_info()
          if info is None:
          raise RuntimeError("Failed to get terminal info after initialize()")
          print("MT5 terminal initialized:", info.product)
          # Ensure symbol is available
          if not mt5.symbol_select(SYMBOL, True):
          raise RuntimeError(f"Failed to select symbol {SYMBOL}")
          return True

          def mt5_shutdown():
          mt5.shutdown()

          -------------------------

          Get historical OHLCV

          -------------------------

          def fetch_bars(symbol, timeframe, n_bars):
          # copy_rates_from_pos returns numpy array with fields: time, open, high, low, close, tick_volume, ...
          rates = mt5.copy_rates_from_pos(symbol, timeframe, START_POS, n_bars)
          if rates is None:
          raise RuntimeError(f"Failed to fetch rates for {symbol}: {mt5.last_error()}")
          df = pd.DataFrame(rates)
          df['time'] = pd.to_datetime(df['time'], unit='s')
          return df

          -------------------------

          Simple trading Gym env

          -------------------------

          class MT5TradingEnv(gym.Env):
          """
          Observation: last LOOKBACK closes normalized + current position (0/1/-1)
          Actions: 0=hold, 1=buy (long), 2=sell (short/close long)
          Reward: change in account equity approximated by price moves * position
          NOTE: Simplified; this is a research template, not production-ready.
          """
          def init(self, df: pd.DataFrame, lookback=LOOKBACK):
          super(MT5TradingEnv, self).init()
          self.df = df.reset_index(drop=True)
          self.lookback = lookback
          self.ptr = lookback # current index in df
          self.position = 0 # -1 short, 0 flat, 1 long
          self.entry_price = 0.0
          # Observations: lookback closes (normalized) + position
          self.observation_space = spaces.Box(low=-np.inf, high=np.inf, shape=(lookback + 1,), dtype=np.float32)
          # Actions: hold(0), buy(1), sell(2)
          self.action_space = spaces.Discrete(3)

          def _get_obs(self):
          closes = self.df.loc[self.ptr - self.lookback:self.ptr - 1, "close"].values.astype(np.float32)
          # normalize closes by dividing by last close
          norm = closes / (closes[-1] + 1e-9) - 1.0
          obs = np.concatenate([norm, np.array([float(self.position)])], axis=0)
          return obs
          def reset(self):
          self.ptr = self.lookback
          self.position = 0
          self.entry_price = 0.0
          return self._get_obs()
          def step(self, action):
          done = False
          reward = 0.0
          price = float(self.df.loc[self.ptr, "close"])
          # Action logic
          if action == 1: # buy
          if self.position == 0:
          self.position = 1
          self.entry_price = price
          elif self.position == -1:
          # close short and go long
          reward += (self.entry_price - price) # profit from short
          self.position = 1
          self.entry_price = price
          elif action == 2: # sell
          if self.position == 0:
          self.position = -1
          self.entry_price = price
          elif self.position == 1:
          reward += (price - self.entry_price) # profit from long
          self.position = -1
          self.entry_price = price
          # Move pointer
          self.ptr += 1
          if self.ptr >= len(self.df):
          done = True
          else:
          # reward can also be shaped by unrealized pnl:
          next_price = float(self.df.loc[self.ptr, "close"])
          unrealized = 0.0
          if self.position == 1:
          unrealized = next_price - self.entry_price
          elif self.position == -1:
          unrealized = self.entry_price - next_price
          # small per-step reward = unrealized PnL scaled
          reward += unrealized * 0.1
          obs = self._get_obs() if not done else np.zeros(self.observation_space.shape, dtype=np.float32)
          info = {"ptr": self.ptr}
          return obs, float(reward), done, info
          

          -------------------------

          Order helpers

          -------------------------

          def send_order(symbol, action, lot=LOT_SIZE, deviation=20):
          """
          action: 1=buy, 2=sell
          This function sends a ORDER_TYPE_BUY / ORDER_TYPE_SELL market order.
          Basic error checking included. For production you need more robust code.
          """
          price = mt5.symbol_info_tick(symbol).ask if action == 1 else mt5.symbol_info_tick(symbol).bid
          request = {
          "action": mt5.TRADE_ACTION_DEAL,
          "symbol": symbol,
          "volume": float(lot),
          "type": mt5.ORDER_TYPE_BUY if action == 1 else mt5.ORDER_TYPE_SELL,
          "price": float(price),
          "deviation": deviation,
          "magic": 234000,
          "comment": "RL-bot",
          "type_filling": mt5.ORDER_FILLING_IOC,
          }
          result = mt5.order_send(request)
          return result

          -------------------------

          Main: training flow

          -------------------------

          def train_agent():
          mt5_connect()
          # fetch historical bars
          n_bars = 5000
          df = fetch_bars(SYMBOL, TIMEFRAME, n_bars)
          print(f"Fetched {len(df)} bars for {SYMBOL}")
          # Create env
          env = DummyVecEnv([lambda: MT5TradingEnv(df, lookback=LOOKBACK)])
          # model
          model = PPO("MlpPolicy", env, verbose=1)
          # save checkpoints
          cb = CheckpointCallback(save_freq=5000, save_path="./logs/", name_prefix="ppo_mt5")
          model.learn(total_timesteps=TRAIN_TIMESTEPS, callback=cb)
          model.save(MODEL_PATH)
          mt5_shutdown()
          print("Training complete, model saved to", MODEL_PATH)

          -------------------------

          Real-time execution loop (paper/live)

          -------------------------

          def run_live_loop(model_path=MODEL_PATH, poll_seconds=5):
          mt5_connect()
          model = PPO.load(model_path)
          print("Loaded model:", model_path)
          # We'll maintain a small in-memory buffer of recent bars
          n_history = LOOKBACK + 10
          df = fetch_bars(SYMBOL, TIMEFRAME, n_history)
          # pointer is at last bar
          while True:
          try:
          latest = fetch_bars(SYMBOL, TIMEFRAME, 1)
          if latest['time'].iloc[-1] > df['time'].iloc[-1]:
          # append new bar
          df = pd.concat([df, latest]).reset_index(drop=True)
          if len(df) > n_history:
          df = df.iloc[-n_history:].reset_index(drop=True)
          # Build an env instance for this single-step decision
          env = MT5TradingEnv(df, lookback=LOOKBACK)
          obs = env.reset()
          action, _states = model.predict(obs, deterministic=True)
          print(f"[{pd.to_datetime('now')}] Action: {action} | Price: {df['close'].iloc[-1]}")
          # Send order if LIVE
          if LIVE:
          if int(action) == 1:
          res = send_order(SYMBOL, 1)
          print("Order send result:", res)
          elif int(action) == 2:
          res = send_order(SYMBOL, 2)
          print("Order send result:", res)
          else:
          # paper-trade: just log what would happen
          print("LIVE=False -> paper trade logged only")
          else:
          # no new bar yet
          pass
          time.sleep(poll_seconds)
          except KeyboardInterrupt:
          print("Stopping live loop (KeyboardInterrupt).")
          break
          except Exception as e:
          print("Exception in live loop:", str(e))
          time.sleep(5)
          mt5_shutdown()

          -------------------------

          If run as script

          -------------------------

          if name == "main":
          import argparse
          parser = argparse.ArgumentParser()
          parser.add_argument("--mode", choices=["train", "run"], default="train")
          args = parser.parse_args()
          if args.mode == "train":
          print("Starting training...")
          train_agent()
          elif args.mode == "run":
          print("Starting live/paper run loop...")
          run_live_loop()

          Metadata

          Metadata

          Assignees

          No one assigned

            Labels

            No labels
            No labels

            Projects

            No projects

              Milestone

              No milestone

              Relationships

              None yet

              Development

              No branches or pull requests

              Issue actions

              , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
              Skip to content

              Compare_agent.py #6

              Description

              @vicks4u

              """
              RL + MetaTrader5 trading bot template

              • Train with historical data (PPO from stable-baselines3)
              • Optionally execute trades via MetaTrader5 (set LIVE=True to enable)
                CAVEAT: This is an educational template. Backtest & paper-trade first.
                """

              import time
              import numpy as np
              import pandas as pd
              import gym
              from gym import spaces
              import MetaTrader5 as mt5
              from stable_baselines3 import PPO
              from stable_baselines3.common.vec_env import DummyVecEnv
              from stable_baselines3.common.callbacks import CheckpointCallback

              -------------------------

              USER CONFIG

              -------------------------

              SYMBOL = "EURUSD"
              TIMEFRAME = mt5.TIMEFRAME_M5 # 5 minute bars
              LOOKBACK = 50 # observation window (bars)
              START_POS = 0 # for historical fetch offset
              LOT_SIZE = 0.01 # trade lot size
              LIVE = False # <-- Set to True only after full testing (demo account first!)
              MODEL_PATH = "ppo_mt5_model"
              TRAIN_TIMESTEPS = 20000 # adjust as you like

              -------------------------

              -------------------------

              Helper: connect to MT5

              -------------------------

              def mt5_connect():
              if not mt5.initialize():
              raise RuntimeError(f"MT5 initialize() failed, error={mt5.last_error()}")
              info = mt5.terminal_info()
              if info is None:
              raise RuntimeError("Failed to get terminal info after initialize()")
              print("MT5 terminal initialized:", info.product)
              # Ensure symbol is available
              if not mt5.symbol_select(SYMBOL, True):
              raise RuntimeError(f"Failed to select symbol {SYMBOL}")
              return True

              def mt5_shutdown():
              mt5.shutdown()

              -------------------------

              Get historical OHLCV

              -------------------------

              def fetch_bars(symbol, timeframe, n_bars):
              # copy_rates_from_pos returns numpy array with fields: time, open, high, low, close, tick_volume, ...
              rates = mt5.copy_rates_from_pos(symbol, timeframe, START_POS, n_bars)
              if rates is None:
              raise RuntimeError(f"Failed to fetch rates for {symbol}: {mt5.last_error()}")
              df = pd.DataFrame(rates)
              df['time'] = pd.to_datetime(df['time'], unit='s')
              return df

              -------------------------

              Simple trading Gym env

              -------------------------

              class MT5TradingEnv(gym.Env):
              """
              Observation: last LOOKBACK closes normalized + current position (0/1/-1)
              Actions: 0=hold, 1=buy (long), 2=sell (short/close long)
              Reward: change in account equity approximated by price moves * position
              NOTE: Simplified; this is a research template, not production-ready.
              """
              def init(self, df: pd.DataFrame, lookback=LOOKBACK):
              super(MT5TradingEnv, self).init()
              self.df = df.reset_index(drop=True)
              self.lookback = lookback
              self.ptr = lookback # current index in df
              self.position = 0 # -1 short, 0 flat, 1 long
              self.entry_price = 0.0
              # Observations: lookback closes (normalized) + position
              self.observation_space = spaces.Box(low=-np.inf, high=np.inf, shape=(lookback + 1,), dtype=np.float32)
              # Actions: hold(0), buy(1), sell(2)
              self.action_space = spaces.Discrete(3)

              def _get_obs(self):
              closes = self.df.loc[self.ptr - self.lookback:self.ptr - 1, "close"].values.astype(np.float32)
              # normalize closes by dividing by last close
              norm = closes / (closes[-1] + 1e-9) - 1.0
              obs = np.concatenate([norm, np.array([float(self.position)])], axis=0)
              return obs
              def reset(self):
              self.ptr = self.lookback
              self.position = 0
              self.entry_price = 0.0
              return self._get_obs()
              def step(self, action):
              done = False
              reward = 0.0
              price = float(self.df.loc[self.ptr, "close"])
              # Action logic
              if action == 1: # buy
              if self.position == 0:
              self.position = 1
              self.entry_price = price
              elif self.position == -1:
              # close short and go long
              reward += (self.entry_price - price) # profit from short
              self.position = 1
              self.entry_price = price
              elif action == 2: # sell
              if self.position == 0:
              self.position = -1
              self.entry_price = price
              elif self.position == 1:
              reward += (price - self.entry_price) # profit from long
              self.position = -1
              self.entry_price = price
              # Move pointer
              self.ptr += 1
              if self.ptr >= len(self.df):
              done = True
              else:
              # reward can also be shaped by unrealized pnl:
              next_price = float(self.df.loc[self.ptr, "close"])
              unrealized = 0.0
              if self.position == 1:
              unrealized = next_price - self.entry_price
              elif self.position == -1:
              unrealized = self.entry_price - next_price
              # small per-step reward = unrealized PnL scaled
              reward += unrealized * 0.1
              obs = self._get_obs() if not done else np.zeros(self.observation_space.shape, dtype=np.float32)
              info = {"ptr": self.ptr}
              return obs, float(reward), done, info
              

              -------------------------

              Order helpers

              -------------------------

              def send_order(symbol, action, lot=LOT_SIZE, deviation=20):
              """
              action: 1=buy, 2=sell
              This function sends a ORDER_TYPE_BUY / ORDER_TYPE_SELL market order.
              Basic error checking included. For production you need more robust code.
              """
              price = mt5.symbol_info_tick(symbol).ask if action == 1 else mt5.symbol_info_tick(symbol).bid
              request = {
              "action": mt5.TRADE_ACTION_DEAL,
              "symbol": symbol,
              "volume": float(lot),
              "type": mt5.ORDER_TYPE_BUY if action == 1 else mt5.ORDER_TYPE_SELL,
              "price": float(price),
              "deviation": deviation,
              "magic": 234000,
              "comment": "RL-bot",
              "type_filling": mt5.ORDER_FILLING_IOC,
              }
              result = mt5.order_send(request)
              return result

              -------------------------

              Main: training flow

              -------------------------

              def train_agent():
              mt5_connect()
              # fetch historical bars
              n_bars = 5000
              df = fetch_bars(SYMBOL, TIMEFRAME, n_bars)
              print(f"Fetched {len(df)} bars for {SYMBOL}")
              # Create env
              env = DummyVecEnv([lambda: MT5TradingEnv(df, lookback=LOOKBACK)])
              # model
              model = PPO("MlpPolicy", env, verbose=1)
              # save checkpoints
              cb = CheckpointCallback(save_freq=5000, save_path="./logs/", name_prefix="ppo_mt5")
              model.learn(total_timesteps=TRAIN_TIMESTEPS, callback=cb)
              model.save(MODEL_PATH)
              mt5_shutdown()
              print("Training complete, model saved to", MODEL_PATH)

              -------------------------

              Real-time execution loop (paper/live)

              -------------------------

              def run_live_loop(model_path=MODEL_PATH, poll_seconds=5):
              mt5_connect()
              model = PPO.load(model_path)
              print("Loaded model:", model_path)
              # We'll maintain a small in-memory buffer of recent bars
              n_history = LOOKBACK + 10
              df = fetch_bars(SYMBOL, TIMEFRAME, n_history)
              # pointer is at last bar
              while True:
              try:
              latest = fetch_bars(SYMBOL, TIMEFRAME, 1)
              if latest['time'].iloc[-1] > df['time'].iloc[-1]:
              # append new bar
              df = pd.concat([df, latest]).reset_index(drop=True)
              if len(df) > n_history:
              df = df.iloc[-n_history:].reset_index(drop=True)
              # Build an env instance for this single-step decision
              env = MT5TradingEnv(df, lookback=LOOKBACK)
              obs = env.reset()
              action, _states = model.predict(obs, deterministic=True)
              print(f"[{pd.to_datetime('now')}] Action: {action} | Price: {df['close'].iloc[-1]}")
              # Send order if LIVE
              if LIVE:
              if int(action) == 1:
              res = send_order(SYMBOL, 1)
              print("Order send result:", res)
              elif int(action) == 2:
              res = send_order(SYMBOL, 2)
              print("Order send result:", res)
              else:
              # paper-trade: just log what would happen
              print("LIVE=False -> paper trade logged only")
              else:
              # no new bar yet
              pass
              time.sleep(poll_seconds)
              except KeyboardInterrupt:
              print("Stopping live loop (KeyboardInterrupt).")
              break
              except Exception as e:
              print("Exception in live loop:", str(e))
              time.sleep(5)
              mt5_shutdown()

              -------------------------

              If run as script

              -------------------------

              if name == "main":
              import argparse
              parser = argparse.ArgumentParser()
              parser.add_argument("--mode", choices=["train", "run"], default="train")
              args = parser.parse_args()
              if args.mode == "train":
              print("Starting training...")
              train_agent()
              elif args.mode == "run":
              print("Starting live/paper run loop...")
              run_live_loop()

              Metadata

              Metadata

              Assignees

              No one assigned

                Labels

                No labels
                No labels

                Projects

                No projects

                  Milestone

                  No milestone

                  Relationships

                  None yet

                  Development

                  No branches or pull requests

                  Issue actions

                  , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
                  Skip to content

                  Compare_agent.py #6

                  Description

                  @vicks4u

                  """
                  RL + MetaTrader5 trading bot template

                  • Train with historical data (PPO from stable-baselines3)
                  • Optionally execute trades via MetaTrader5 (set LIVE=True to enable)
                    CAVEAT: This is an educational template. Backtest & paper-trade first.
                    """

                  import time
                  import numpy as np
                  import pandas as pd
                  import gym
                  from gym import spaces
                  import MetaTrader5 as mt5
                  from stable_baselines3 import PPO
                  from stable_baselines3.common.vec_env import DummyVecEnv
                  from stable_baselines3.common.callbacks import CheckpointCallback

                  -------------------------

                  USER CONFIG

                  -------------------------

                  SYMBOL = "EURUSD"
                  TIMEFRAME = mt5.TIMEFRAME_M5 # 5 minute bars
                  LOOKBACK = 50 # observation window (bars)
                  START_POS = 0 # for historical fetch offset
                  LOT_SIZE = 0.01 # trade lot size
                  LIVE = False # <-- Set to True only after full testing (demo account first!)
                  MODEL_PATH = "ppo_mt5_model"
                  TRAIN_TIMESTEPS = 20000 # adjust as you like

                  -------------------------

                  -------------------------

                  Helper: connect to MT5

                  -------------------------

                  def mt5_connect():
                  if not mt5.initialize():
                  raise RuntimeError(f"MT5 initialize() failed, error={mt5.last_error()}")
                  info = mt5.terminal_info()
                  if info is None:
                  raise RuntimeError("Failed to get terminal info after initialize()")
                  print("MT5 terminal initialized:", info.product)
                  # Ensure symbol is available
                  if not mt5.symbol_select(SYMBOL, True):
                  raise RuntimeError(f"Failed to select symbol {SYMBOL}")
                  return True

                  def mt5_shutdown():
                  mt5.shutdown()

                  -------------------------

                  Get historical OHLCV

                  -------------------------

                  def fetch_bars(symbol, timeframe, n_bars):
                  # copy_rates_from_pos returns numpy array with fields: time, open, high, low, close, tick_volume, ...
                  rates = mt5.copy_rates_from_pos(symbol, timeframe, START_POS, n_bars)
                  if rates is None:
                  raise RuntimeError(f"Failed to fetch rates for {symbol}: {mt5.last_error()}")
                  df = pd.DataFrame(rates)
                  df['time'] = pd.to_datetime(df['time'], unit='s')
                  return df

                  -------------------------

                  Simple trading Gym env

                  -------------------------

                  class MT5TradingEnv(gym.Env):
                  """
                  Observation: last LOOKBACK closes normalized + current position (0/1/-1)
                  Actions: 0=hold, 1=buy (long), 2=sell (short/close long)
                  Reward: change in account equity approximated by price moves * position
                  NOTE: Simplified; this is a research template, not production-ready.
                  """
                  def init(self, df: pd.DataFrame, lookback=LOOKBACK):
                  super(MT5TradingEnv, self).init()
                  self.df = df.reset_index(drop=True)
                  self.lookback = lookback
                  self.ptr = lookback # current index in df
                  self.position = 0 # -1 short, 0 flat, 1 long
                  self.entry_price = 0.0
                  # Observations: lookback closes (normalized) + position
                  self.observation_space = spaces.Box(low=-np.inf, high=np.inf, shape=(lookback + 1,), dtype=np.float32)
                  # Actions: hold(0), buy(1), sell(2)
                  self.action_space = spaces.Discrete(3)

                  def _get_obs(self):
                  closes = self.df.loc[self.ptr - self.lookback:self.ptr - 1, "close"].values.astype(np.float32)
                  # normalize closes by dividing by last close
                  norm = closes / (closes[-1] + 1e-9) - 1.0
                  obs = np.concatenate([norm, np.array([float(self.position)])], axis=0)
                  return obs
                  def reset(self):
                  self.ptr = self.lookback
                  self.position = 0
                  self.entry_price = 0.0
                  return self._get_obs()
                  def step(self, action):
                  done = False
                  reward = 0.0
                  price = float(self.df.loc[self.ptr, "close"])
                  # Action logic
                  if action == 1: # buy
                  if self.position == 0:
                  self.position = 1
                  self.entry_price = price
                  elif self.position == -1:
                  # close short and go long
                  reward += (self.entry_price - price) # profit from short
                  self.position = 1
                  self.entry_price = price
                  elif action == 2: # sell
                  if self.position == 0:
                  self.position = -1
                  self.entry_price = price
                  elif self.position == 1:
                  reward += (price - self.entry_price) # profit from long
                  self.position = -1
                  self.entry_price = price
                  # Move pointer
                  self.ptr += 1
                  if self.ptr >= len(self.df):
                  done = True
                  else:
                  # reward can also be shaped by unrealized pnl:
                  next_price = float(self.df.loc[self.ptr, "close"])
                  unrealized = 0.0
                  if self.position == 1:
                  unrealized = next_price - self.entry_price
                  elif self.position == -1:
                  unrealized = self.entry_price - next_price
                  # small per-step reward = unrealized PnL scaled
                  reward += unrealized * 0.1
                  obs = self._get_obs() if not done else np.zeros(self.observation_space.shape, dtype=np.float32)
                  info = {"ptr": self.ptr}
                  return obs, float(reward), done, info
                  

                  -------------------------

                  Order helpers

                  -------------------------

                  def send_order(symbol, action, lot=LOT_SIZE, deviation=20):
                  """
                  action: 1=buy, 2=sell
                  This function sends a ORDER_TYPE_BUY / ORDER_TYPE_SELL market order.
                  Basic error checking included. For production you need more robust code.
                  """
                  price = mt5.symbol_info_tick(symbol).ask if action == 1 else mt5.symbol_info_tick(symbol).bid
                  request = {
                  "action": mt5.TRADE_ACTION_DEAL,
                  "symbol": symbol,
                  "volume": float(lot),
                  "type": mt5.ORDER_TYPE_BUY if action == 1 else mt5.ORDER_TYPE_SELL,
                  "price": float(price),
                  "deviation": deviation,
                  "magic": 234000,
                  "comment": "RL-bot",
                  "type_filling": mt5.ORDER_FILLING_IOC,
                  }
                  result = mt5.order_send(request)
                  return result

                  -------------------------

                  Main: training flow

                  -------------------------

                  def train_agent():
                  mt5_connect()
                  # fetch historical bars
                  n_bars = 5000
                  df = fetch_bars(SYMBOL, TIMEFRAME, n_bars)
                  print(f"Fetched {len(df)} bars for {SYMBOL}")
                  # Create env
                  env = DummyVecEnv([lambda: MT5TradingEnv(df, lookback=LOOKBACK)])
                  # model
                  model = PPO("MlpPolicy", env, verbose=1)
                  # save checkpoints
                  cb = CheckpointCallback(save_freq=5000, save_path="./logs/", name_prefix="ppo_mt5")
                  model.learn(total_timesteps=TRAIN_TIMESTEPS, callback=cb)
                  model.save(MODEL_PATH)
                  mt5_shutdown()
                  print("Training complete, model saved to", MODEL_PATH)

                  -------------------------

                  Real-time execution loop (paper/live)

                  -------------------------

                  def run_live_loop(model_path=MODEL_PATH, poll_seconds=5):
                  mt5_connect()
                  model = PPO.load(model_path)
                  print("Loaded model:", model_path)
                  # We'll maintain a small in-memory buffer of recent bars
                  n_history = LOOKBACK + 10
                  df = fetch_bars(SYMBOL, TIMEFRAME, n_history)
                  # pointer is at last bar
                  while True:
                  try:
                  latest = fetch_bars(SYMBOL, TIMEFRAME, 1)
                  if latest['time'].iloc[-1] > df['time'].iloc[-1]:
                  # append new bar
                  df = pd.concat([df, latest]).reset_index(drop=True)
                  if len(df) > n_history:
                  df = df.iloc[-n_history:].reset_index(drop=True)
                  # Build an env instance for this single-step decision
                  env = MT5TradingEnv(df, lookback=LOOKBACK)
                  obs = env.reset()
                  action, _states = model.predict(obs, deterministic=True)
                  print(f"[{pd.to_datetime('now')}] Action: {action} | Price: {df['close'].iloc[-1]}")
                  # Send order if LIVE
                  if LIVE:
                  if int(action) == 1:
                  res = send_order(SYMBOL, 1)
                  print("Order send result:", res)
                  elif int(action) == 2:
                  res = send_order(SYMBOL, 2)
                  print("Order send result:", res)
                  else:
                  # paper-trade: just log what would happen
                  print("LIVE=False -> paper trade logged only")
                  else:
                  # no new bar yet
                  pass
                  time.sleep(poll_seconds)
                  except KeyboardInterrupt:
                  print("Stopping live loop (KeyboardInterrupt).")
                  break
                  except Exception as e:
                  print("Exception in live loop:", str(e))
                  time.sleep(5)
                  mt5_shutdown()

                  -------------------------

                  If run as script

                  -------------------------

                  if name == "main":
                  import argparse
                  parser = argparse.ArgumentParser()
                  parser.add_argument("--mode", choices=["train", "run"], default="train")
                  args = parser.parse_args()
                  if args.mode == "train":
                  print("Starting training...")
                  train_agent()
                  elif args.mode == "run":
                  print("Starting live/paper run loop...")
                  run_live_loop()

                  Metadata

                  Metadata

                  Assignees

                  No one assigned

                    Labels

                    No labels
                    No labels

                    Projects

                    No projects

                      Milestone

                      No milestone

                      Relationships

                      None yet

                      Development

                      No branches or pull requests

                      Issue actions

                      , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
                      Skip to content

                      Compare_agent.py #6

                      Description

                      @vicks4u

                      """
                      RL + MetaTrader5 trading bot template

                      • Train with historical data (PPO from stable-baselines3)
                      • Optionally execute trades via MetaTrader5 (set LIVE=True to enable)
                        CAVEAT: This is an educational template. Backtest & paper-trade first.
                        """

                      import time
                      import numpy as np
                      import pandas as pd
                      import gym
                      from gym import spaces
                      import MetaTrader5 as mt5
                      from stable_baselines3 import PPO
                      from stable_baselines3.common.vec_env import DummyVecEnv
                      from stable_baselines3.common.callbacks import CheckpointCallback

                      -------------------------

                      USER CONFIG

                      -------------------------

                      SYMBOL = "EURUSD"
                      TIMEFRAME = mt5.TIMEFRAME_M5 # 5 minute bars
                      LOOKBACK = 50 # observation window (bars)
                      START_POS = 0 # for historical fetch offset
                      LOT_SIZE = 0.01 # trade lot size
                      LIVE = False # <-- Set to True only after full testing (demo account first!)
                      MODEL_PATH = "ppo_mt5_model"
                      TRAIN_TIMESTEPS = 20000 # adjust as you like

                      -------------------------

                      -------------------------

                      Helper: connect to MT5

                      -------------------------

                      def mt5_connect():
                      if not mt5.initialize():
                      raise RuntimeError(f"MT5 initialize() failed, error={mt5.last_error()}")
                      info = mt5.terminal_info()
                      if info is None:
                      raise RuntimeError("Failed to get terminal info after initialize()")
                      print("MT5 terminal initialized:", info.product)
                      # Ensure symbol is available
                      if not mt5.symbol_select(SYMBOL, True):
                      raise RuntimeError(f"Failed to select symbol {SYMBOL}")
                      return True

                      def mt5_shutdown():
                      mt5.shutdown()

                      -------------------------

                      Get historical OHLCV

                      -------------------------

                      def fetch_bars(symbol, timeframe, n_bars):
                      # copy_rates_from_pos returns numpy array with fields: time, open, high, low, close, tick_volume, ...
                      rates = mt5.copy_rates_from_pos(symbol, timeframe, START_POS, n_bars)
                      if rates is None:
                      raise RuntimeError(f"Failed to fetch rates for {symbol}: {mt5.last_error()}")
                      df = pd.DataFrame(rates)
                      df['time'] = pd.to_datetime(df['time'], unit='s')
                      return df

                      -------------------------

                      Simple trading Gym env

                      -------------------------

                      class MT5TradingEnv(gym.Env):
                      """
                      Observation: last LOOKBACK closes normalized + current position (0/1/-1)
                      Actions: 0=hold, 1=buy (long), 2=sell (short/close long)
                      Reward: change in account equity approximated by price moves * position
                      NOTE: Simplified; this is a research template, not production-ready.
                      """
                      def init(self, df: pd.DataFrame, lookback=LOOKBACK):
                      super(MT5TradingEnv, self).init()
                      self.df = df.reset_index(drop=True)
                      self.lookback = lookback
                      self.ptr = lookback # current index in df
                      self.position = 0 # -1 short, 0 flat, 1 long
                      self.entry_price = 0.0
                      # Observations: lookback closes (normalized) + position
                      self.observation_space = spaces.Box(low=-np.inf, high=np.inf, shape=(lookback + 1,), dtype=np.float32)
                      # Actions: hold(0), buy(1), sell(2)
                      self.action_space = spaces.Discrete(3)

                      def _get_obs(self):
                      closes = self.df.loc[self.ptr - self.lookback:self.ptr - 1, "close"].values.astype(np.float32)
                      # normalize closes by dividing by last close
                      norm = closes / (closes[-1] + 1e-9) - 1.0
                      obs = np.concatenate([norm, np.array([float(self.position)])], axis=0)
                      return obs
                      def reset(self):
                      self.ptr = self.lookback
                      self.position = 0
                      self.entry_price = 0.0
                      return self._get_obs()
                      def step(self, action):
                      done = False
                      reward = 0.0
                      price = float(self.df.loc[self.ptr, "close"])
                      # Action logic
                      if action == 1: # buy
                      if self.position == 0:
                      self.position = 1
                      self.entry_price = price
                      elif self.position == -1:
                      # close short and go long
                      reward += (self.entry_price - price) # profit from short
                      self.position = 1
                      self.entry_price = price
                      elif action == 2: # sell
                      if self.position == 0:
                      self.position = -1
                      self.entry_price = price
                      elif self.position == 1:
                      reward += (price - self.entry_price) # profit from long
                      self.position = -1
                      self.entry_price = price
                      # Move pointer
                      self.ptr += 1
                      if self.ptr >= len(self.df):
                      done = True
                      else:
                      # reward can also be shaped by unrealized pnl:
                      next_price = float(self.df.loc[self.ptr, "close"])
                      unrealized = 0.0
                      if self.position == 1:
                      unrealized = next_price - self.entry_price
                      elif self.position == -1:
                      unrealized = self.entry_price - next_price
                      # small per-step reward = unrealized PnL scaled
                      reward += unrealized * 0.1
                      obs = self._get_obs() if not done else np.zeros(self.observation_space.shape, dtype=np.float32)
                      info = {"ptr": self.ptr}
                      return obs, float(reward), done, info
                      

                      -------------------------

                      Order helpers

                      -------------------------

                      def send_order(symbol, action, lot=LOT_SIZE, deviation=20):
                      """
                      action: 1=buy, 2=sell
                      This function sends a ORDER_TYPE_BUY / ORDER_TYPE_SELL market order.
                      Basic error checking included. For production you need more robust code.
                      """
                      price = mt5.symbol_info_tick(symbol).ask if action == 1 else mt5.symbol_info_tick(symbol).bid
                      request = {
                      "action": mt5.TRADE_ACTION_DEAL,
                      "symbol": symbol,
                      "volume": float(lot),
                      "type": mt5.ORDER_TYPE_BUY if action == 1 else mt5.ORDER_TYPE_SELL,
                      "price": float(price),
                      "deviation": deviation,
                      "magic": 234000,
                      "comment": "RL-bot",
                      "type_filling": mt5.ORDER_FILLING_IOC,
                      }
                      result = mt5.order_send(request)
                      return result

                      -------------------------

                      Main: training flow

                      -------------------------

                      def train_agent():
                      mt5_connect()
                      # fetch historical bars
                      n_bars = 5000
                      df = fetch_bars(SYMBOL, TIMEFRAME, n_bars)
                      print(f"Fetched {len(df)} bars for {SYMBOL}")
                      # Create env
                      env = DummyVecEnv([lambda: MT5TradingEnv(df, lookback=LOOKBACK)])
                      # model
                      model = PPO("MlpPolicy", env, verbose=1)
                      # save checkpoints
                      cb = CheckpointCallback(save_freq=5000, save_path="./logs/", name_prefix="ppo_mt5")
                      model.learn(total_timesteps=TRAIN_TIMESTEPS, callback=cb)
                      model.save(MODEL_PATH)
                      mt5_shutdown()
                      print("Training complete, model saved to", MODEL_PATH)

                      -------------------------

                      Real-time execution loop (paper/live)

                      -------------------------

                      def run_live_loop(model_path=MODEL_PATH, poll_seconds=5):
                      mt5_connect()
                      model = PPO.load(model_path)
                      print("Loaded model:", model_path)
                      # We'll maintain a small in-memory buffer of recent bars
                      n_history = LOOKBACK + 10
                      df = fetch_bars(SYMBOL, TIMEFRAME, n_history)
                      # pointer is at last bar
                      while True:
                      try:
                      latest = fetch_bars(SYMBOL, TIMEFRAME, 1)
                      if latest['time'].iloc[-1] > df['time'].iloc[-1]:
                      # append new bar
                      df = pd.concat([df, latest]).reset_index(drop=True)
                      if len(df) > n_history:
                      df = df.iloc[-n_history:].reset_index(drop=True)
                      # Build an env instance for this single-step decision
                      env = MT5TradingEnv(df, lookback=LOOKBACK)
                      obs = env.reset()
                      action, _states = model.predict(obs, deterministic=True)
                      print(f"[{pd.to_datetime('now')}] Action: {action} | Price: {df['close'].iloc[-1]}")
                      # Send order if LIVE
                      if LIVE:
                      if int(action) == 1:
                      res = send_order(SYMBOL, 1)
                      print("Order send result:", res)
                      elif int(action) == 2:
                      res = send_order(SYMBOL, 2)
                      print("Order send result:", res)
                      else:
                      # paper-trade: just log what would happen
                      print("LIVE=False -> paper trade logged only")
                      else:
                      # no new bar yet
                      pass
                      time.sleep(poll_seconds)
                      except KeyboardInterrupt:
                      print("Stopping live loop (KeyboardInterrupt).")
                      break
                      except Exception as e:
                      print("Exception in live loop:", str(e))
                      time.sleep(5)
                      mt5_shutdown()

                      -------------------------

                      If run as script

                      -------------------------

                      if name == "main":
                      import argparse
                      parser = argparse.ArgumentParser()
                      parser.add_argument("--mode", choices=["train", "run"], default="train")
                      args = parser.parse_args()
                      if args.mode == "train":
                      print("Starting training...")
                      train_agent()
                      elif args.mode == "run":
                      print("Starting live/paper run loop...")
                      run_live_loop()

                      Metadata

                      Metadata

                      Assignees

                      No one assigned

                        Labels

                        No labels
                        No labels

                        Projects

                        No projects

                          Milestone

                          No milestone

                          Relationships

                          None yet

                          Development

                          No branches or pull requests

                          Issue actions

                          , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
                          Skip to content

                          Compare_agent.py #6

                          Description

                          @vicks4u

                          """
                          RL + MetaTrader5 trading bot template

                          • Train with historical data (PPO from stable-baselines3)
                          • Optionally execute trades via MetaTrader5 (set LIVE=True to enable)
                            CAVEAT: This is an educational template. Backtest & paper-trade first.
                            """

                          import time
                          import numpy as np
                          import pandas as pd
                          import gym
                          from gym import spaces
                          import MetaTrader5 as mt5
                          from stable_baselines3 import PPO
                          from stable_baselines3.common.vec_env import DummyVecEnv
                          from stable_baselines3.common.callbacks import CheckpointCallback

                          -------------------------

                          USER CONFIG

                          -------------------------

                          SYMBOL = "EURUSD"
                          TIMEFRAME = mt5.TIMEFRAME_M5 # 5 minute bars
                          LOOKBACK = 50 # observation window (bars)
                          START_POS = 0 # for historical fetch offset
                          LOT_SIZE = 0.01 # trade lot size
                          LIVE = False # <-- Set to True only after full testing (demo account first!)
                          MODEL_PATH = "ppo_mt5_model"
                          TRAIN_TIMESTEPS = 20000 # adjust as you like

                          -------------------------

                          -------------------------

                          Helper: connect to MT5

                          -------------------------

                          def mt5_connect():
                          if not mt5.initialize():
                          raise RuntimeError(f"MT5 initialize() failed, error={mt5.last_error()}")
                          info = mt5.terminal_info()
                          if info is None:
                          raise RuntimeError("Failed to get terminal info after initialize()")
                          print("MT5 terminal initialized:", info.product)
                          # Ensure symbol is available
                          if not mt5.symbol_select(SYMBOL, True):
                          raise RuntimeError(f"Failed to select symbol {SYMBOL}")
                          return True

                          def mt5_shutdown():
                          mt5.shutdown()

                          -------------------------

                          Get historical OHLCV

                          -------------------------

                          def fetch_bars(symbol, timeframe, n_bars):
                          # copy_rates_from_pos returns numpy array with fields: time, open, high, low, close, tick_volume, ...
                          rates = mt5.copy_rates_from_pos(symbol, timeframe, START_POS, n_bars)
                          if rates is None:
                          raise RuntimeError(f"Failed to fetch rates for {symbol}: {mt5.last_error()}")
                          df = pd.DataFrame(rates)
                          df['time'] = pd.to_datetime(df['time'], unit='s')
                          return df

                          -------------------------

                          Simple trading Gym env

                          -------------------------

                          class MT5TradingEnv(gym.Env):
                          """
                          Observation: last LOOKBACK closes normalized + current position (0/1/-1)
                          Actions: 0=hold, 1=buy (long), 2=sell (short/close long)
                          Reward: change in account equity approximated by price moves * position
                          NOTE: Simplified; this is a research template, not production-ready.
                          """
                          def init(self, df: pd.DataFrame, lookback=LOOKBACK):
                          super(MT5TradingEnv, self).init()
                          self.df = df.reset_index(drop=True)
                          self.lookback = lookback
                          self.ptr = lookback # current index in df
                          self.position = 0 # -1 short, 0 flat, 1 long
                          self.entry_price = 0.0
                          # Observations: lookback closes (normalized) + position
                          self.observation_space = spaces.Box(low=-np.inf, high=np.inf, shape=(lookback + 1,), dtype=np.float32)
                          # Actions: hold(0), buy(1), sell(2)
                          self.action_space = spaces.Discrete(3)

                          def _get_obs(self):
                          closes = self.df.loc[self.ptr - self.lookback:self.ptr - 1, "close"].values.astype(np.float32)
                          # normalize closes by dividing by last close
                          norm = closes / (closes[-1] + 1e-9) - 1.0
                          obs = np.concatenate([norm, np.array([float(self.position)])], axis=0)
                          return obs
                          def reset(self):
                          self.ptr = self.lookback
                          self.position = 0
                          self.entry_price = 0.0
                          return self._get_obs()
                          def step(self, action):
                          done = False
                          reward = 0.0
                          price = float(self.df.loc[self.ptr, "close"])
                          # Action logic
                          if action == 1: # buy
                          if self.position == 0:
                          self.position = 1
                          self.entry_price = price
                          elif self.position == -1:
                          # close short and go long
                          reward += (self.entry_price - price) # profit from short
                          self.position = 1
                          self.entry_price = price
                          elif action == 2: # sell
                          if self.position == 0:
                          self.position = -1
                          self.entry_price = price
                          elif self.position == 1:
                          reward += (price - self.entry_price) # profit from long
                          self.position = -1
                          self.entry_price = price
                          # Move pointer
                          self.ptr += 1
                          if self.ptr >= len(self.df):
                          done = True
                          else:
                          # reward can also be shaped by unrealized pnl:
                          next_price = float(self.df.loc[self.ptr, "close"])
                          unrealized = 0.0
                          if self.position == 1:
                          unrealized = next_price - self.entry_price
                          elif self.position == -1:
                          unrealized = self.entry_price - next_price
                          # small per-step reward = unrealized PnL scaled
                          reward += unrealized * 0.1
                          obs = self._get_obs() if not done else np.zeros(self.observation_space.shape, dtype=np.float32)
                          info = {"ptr": self.ptr}
                          return obs, float(reward), done, info
                          

                          -------------------------

                          Order helpers

                          -------------------------

                          def send_order(symbol, action, lot=LOT_SIZE, deviation=20):
                          """
                          action: 1=buy, 2=sell
                          This function sends a ORDER_TYPE_BUY / ORDER_TYPE_SELL market order.
                          Basic error checking included. For production you need more robust code.
                          """
                          price = mt5.symbol_info_tick(symbol).ask if action == 1 else mt5.symbol_info_tick(symbol).bid
                          request = {
                          "action": mt5.TRADE_ACTION_DEAL,
                          "symbol": symbol,
                          "volume": float(lot),
                          "type": mt5.ORDER_TYPE_BUY if action == 1 else mt5.ORDER_TYPE_SELL,
                          "price": float(price),
                          "deviation": deviation,
                          "magic": 234000,
                          "comment": "RL-bot",
                          "type_filling": mt5.ORDER_FILLING_IOC,
                          }
                          result = mt5.order_send(request)
                          return result

                          -------------------------

                          Main: training flow

                          -------------------------

                          def train_agent():
                          mt5_connect()
                          # fetch historical bars
                          n_bars = 5000
                          df = fetch_bars(SYMBOL, TIMEFRAME, n_bars)
                          print(f"Fetched {len(df)} bars for {SYMBOL}")
                          # Create env
                          env = DummyVecEnv([lambda: MT5TradingEnv(df, lookback=LOOKBACK)])
                          # model
                          model = PPO("MlpPolicy", env, verbose=1)
                          # save checkpoints
                          cb = CheckpointCallback(save_freq=5000, save_path="./logs/", name_prefix="ppo_mt5")
                          model.learn(total_timesteps=TRAIN_TIMESTEPS, callback=cb)
                          model.save(MODEL_PATH)
                          mt5_shutdown()
                          print("Training complete, model saved to", MODEL_PATH)

                          -------------------------

                          Real-time execution loop (paper/live)

                          -------------------------

                          def run_live_loop(model_path=MODEL_PATH, poll_seconds=5):
                          mt5_connect()
                          model = PPO.load(model_path)
                          print("Loaded model:", model_path)
                          # We'll maintain a small in-memory buffer of recent bars
                          n_history = LOOKBACK + 10
                          df = fetch_bars(SYMBOL, TIMEFRAME, n_history)
                          # pointer is at last bar
                          while True:
                          try:
                          latest = fetch_bars(SYMBOL, TIMEFRAME, 1)
                          if latest['time'].iloc[-1] > df['time'].iloc[-1]:
                          # append new bar
                          df = pd.concat([df, latest]).reset_index(drop=True)
                          if len(df) > n_history:
                          df = df.iloc[-n_history:].reset_index(drop=True)
                          # Build an env instance for this single-step decision
                          env = MT5TradingEnv(df, lookback=LOOKBACK)
                          obs = env.reset()
                          action, _states = model.predict(obs, deterministic=True)
                          print(f"[{pd.to_datetime('now')}] Action: {action} | Price: {df['close'].iloc[-1]}")
                          # Send order if LIVE
                          if LIVE:
                          if int(action) == 1:
                          res = send_order(SYMBOL, 1)
                          print("Order send result:", res)
                          elif int(action) == 2:
                          res = send_order(SYMBOL, 2)
                          print("Order send result:", res)
                          else:
                          # paper-trade: just log what would happen
                          print("LIVE=False -> paper trade logged only")
                          else:
                          # no new bar yet
                          pass
                          time.sleep(poll_seconds)
                          except KeyboardInterrupt:
                          print("Stopping live loop (KeyboardInterrupt).")
                          break
                          except Exception as e:
                          print("Exception in live loop:", str(e))
                          time.sleep(5)
                          mt5_shutdown()

                          -------------------------

                          If run as script

                          -------------------------

                          if name == "main":
                          import argparse
                          parser = argparse.ArgumentParser()
                          parser.add_argument("--mode", choices=["train", "run"], default="train")
                          args = parser.parse_args()
                          if args.mode == "train":
                          print("Starting training...")
                          train_agent()
                          elif args.mode == "run":
                          print("Starting live/paper run loop...")
                          run_live_loop()

                          Metadata

                          Metadata

                          Assignees

                          No one assigned

                            Labels

                            No labels
                            No labels

                            Projects

                            No projects

                              Milestone

                              No milestone

                              Relationships

                              None yet

                              Development

                              No branches or pull requests

                              Issue actions

                              , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
                              Skip to content

                              Compare_agent.py #6

                              Description

                              @vicks4u

                              """
                              RL + MetaTrader5 trading bot template

                              • Train with historical data (PPO from stable-baselines3)
                              • Optionally execute trades via MetaTrader5 (set LIVE=True to enable)
                                CAVEAT: This is an educational template. Backtest & paper-trade first.
                                """

                              import time
                              import numpy as np
                              import pandas as pd
                              import gym
                              from gym import spaces
                              import MetaTrader5 as mt5
                              from stable_baselines3 import PPO
                              from stable_baselines3.common.vec_env import DummyVecEnv
                              from stable_baselines3.common.callbacks import CheckpointCallback

                              -------------------------

                              USER CONFIG

                              -------------------------

                              SYMBOL = "EURUSD"
                              TIMEFRAME = mt5.TIMEFRAME_M5 # 5 minute bars
                              LOOKBACK = 50 # observation window (bars)
                              START_POS = 0 # for historical fetch offset
                              LOT_SIZE = 0.01 # trade lot size
                              LIVE = False # <-- Set to True only after full testing (demo account first!)
                              MODEL_PATH = "ppo_mt5_model"
                              TRAIN_TIMESTEPS = 20000 # adjust as you like

                              -------------------------

                              -------------------------

                              Helper: connect to MT5

                              -------------------------

                              def mt5_connect():
                              if not mt5.initialize():
                              raise RuntimeError(f"MT5 initialize() failed, error={mt5.last_error()}")
                              info = mt5.terminal_info()
                              if info is None:
                              raise RuntimeError("Failed to get terminal info after initialize()")
                              print("MT5 terminal initialized:", info.product)
                              # Ensure symbol is available
                              if not mt5.symbol_select(SYMBOL, True):
                              raise RuntimeError(f"Failed to select symbol {SYMBOL}")
                              return True

                              def mt5_shutdown():
                              mt5.shutdown()

                              -------------------------

                              Get historical OHLCV

                              -------------------------

                              def fetch_bars(symbol, timeframe, n_bars):
                              # copy_rates_from_pos returns numpy array with fields: time, open, high, low, close, tick_volume, ...
                              rates = mt5.copy_rates_from_pos(symbol, timeframe, START_POS, n_bars)
                              if rates is None:
                              raise RuntimeError(f"Failed to fetch rates for {symbol}: {mt5.last_error()}")
                              df = pd.DataFrame(rates)
                              df['time'] = pd.to_datetime(df['time'], unit='s')
                              return df

                              -------------------------

                              Simple trading Gym env

                              -------------------------

                              class MT5TradingEnv(gym.Env):
                              """
                              Observation: last LOOKBACK closes normalized + current position (0/1/-1)
                              Actions: 0=hold, 1=buy (long), 2=sell (short/close long)
                              Reward: change in account equity approximated by price moves * position
                              NOTE: Simplified; this is a research template, not production-ready.
                              """
                              def init(self, df: pd.DataFrame, lookback=LOOKBACK):
                              super(MT5TradingEnv, self).init()
                              self.df = df.reset_index(drop=True)
                              self.lookback = lookback
                              self.ptr = lookback # current index in df
                              self.position = 0 # -1 short, 0 flat, 1 long
                              self.entry_price = 0.0
                              # Observations: lookback closes (normalized) + position
                              self.observation_space = spaces.Box(low=-np.inf, high=np.inf, shape=(lookback + 1,), dtype=np.float32)
                              # Actions: hold(0), buy(1), sell(2)
                              self.action_space = spaces.Discrete(3)

                              def _get_obs(self):
                              closes = self.df.loc[self.ptr - self.lookback:self.ptr - 1, "close"].values.astype(np.float32)
                              # normalize closes by dividing by last close
                              norm = closes / (closes[-1] + 1e-9) - 1.0
                              obs = np.concatenate([norm, np.array([float(self.position)])], axis=0)
                              return obs
                              def reset(self):
                              self.ptr = self.lookback
                              self.position = 0
                              self.entry_price = 0.0
                              return self._get_obs()
                              def step(self, action):
                              done = False
                              reward = 0.0
                              price = float(self.df.loc[self.ptr, "close"])
                              # Action logic
                              if action == 1: # buy
                              if self.position == 0:
                              self.position = 1
                              self.entry_price = price
                              elif self.position == -1:
                              # close short and go long
                              reward += (self.entry_price - price) # profit from short
                              self.position = 1
                              self.entry_price = price
                              elif action == 2: # sell
                              if self.position == 0:
                              self.position = -1
                              self.entry_price = price
                              elif self.position == 1:
                              reward += (price - self.entry_price) # profit from long
                              self.position = -1
                              self.entry_price = price
                              # Move pointer
                              self.ptr += 1
                              if self.ptr >= len(self.df):
                              done = True
                              else:
                              # reward can also be shaped by unrealized pnl:
                              next_price = float(self.df.loc[self.ptr, "close"])
                              unrealized = 0.0
                              if self.position == 1:
                              unrealized = next_price - self.entry_price
                              elif self.position == -1:
                              unrealized = self.entry_price - next_price
                              # small per-step reward = unrealized PnL scaled
                              reward += unrealized * 0.1
                              obs = self._get_obs() if not done else np.zeros(self.observation_space.shape, dtype=np.float32)
                              info = {"ptr": self.ptr}
                              return obs, float(reward), done, info
                              

                              -------------------------

                              Order helpers

                              -------------------------

                              def send_order(symbol, action, lot=LOT_SIZE, deviation=20):
                              """
                              action: 1=buy, 2=sell
                              This function sends a ORDER_TYPE_BUY / ORDER_TYPE_SELL market order.
                              Basic error checking included. For production you need more robust code.
                              """
                              price = mt5.symbol_info_tick(symbol).ask if action == 1 else mt5.symbol_info_tick(symbol).bid
                              request = {
                              "action": mt5.TRADE_ACTION_DEAL,
                              "symbol": symbol,
                              "volume": float(lot),
                              "type": mt5.ORDER_TYPE_BUY if action == 1 else mt5.ORDER_TYPE_SELL,
                              "price": float(price),
                              "deviation": deviation,
                              "magic": 234000,
                              "comment": "RL-bot",
                              "type_filling": mt5.ORDER_FILLING_IOC,
                              }
                              result = mt5.order_send(request)
                              return result

                              -------------------------

                              Main: training flow

                              -------------------------

                              def train_agent():
                              mt5_connect()
                              # fetch historical bars
                              n_bars = 5000
                              df = fetch_bars(SYMBOL, TIMEFRAME, n_bars)
                              print(f"Fetched {len(df)} bars for {SYMBOL}")
                              # Create env
                              env = DummyVecEnv([lambda: MT5TradingEnv(df, lookback=LOOKBACK)])
                              # model
                              model = PPO("MlpPolicy", env, verbose=1)
                              # save checkpoints
                              cb = CheckpointCallback(save_freq=5000, save_path="./logs/", name_prefix="ppo_mt5")
                              model.learn(total_timesteps=TRAIN_TIMESTEPS, callback=cb)
                              model.save(MODEL_PATH)
                              mt5_shutdown()
                              print("Training complete, model saved to", MODEL_PATH)

                              -------------------------

                              Real-time execution loop (paper/live)

                              -------------------------

                              def run_live_loop(model_path=MODEL_PATH, poll_seconds=5):
                              mt5_connect()
                              model = PPO.load(model_path)
                              print("Loaded model:", model_path)
                              # We'll maintain a small in-memory buffer of recent bars
                              n_history = LOOKBACK + 10
                              df = fetch_bars(SYMBOL, TIMEFRAME, n_history)
                              # pointer is at last bar
                              while True:
                              try:
                              latest = fetch_bars(SYMBOL, TIMEFRAME, 1)
                              if latest['time'].iloc[-1] > df['time'].iloc[-1]:
                              # append new bar
                              df = pd.concat([df, latest]).reset_index(drop=True)
                              if len(df) > n_history:
                              df = df.iloc[-n_history:].reset_index(drop=True)
                              # Build an env instance for this single-step decision
                              env = MT5TradingEnv(df, lookback=LOOKBACK)
                              obs = env.reset()
                              action, _states = model.predict(obs, deterministic=True)
                              print(f"[{pd.to_datetime('now')}] Action: {action} | Price: {df['close'].iloc[-1]}")
                              # Send order if LIVE
                              if LIVE:
                              if int(action) == 1:
                              res = send_order(SYMBOL, 1)
                              print("Order send result:", res)
                              elif int(action) == 2:
                              res = send_order(SYMBOL, 2)
                              print("Order send result:", res)
                              else:
                              # paper-trade: just log what would happen
                              print("LIVE=False -> paper trade logged only")
                              else:
                              # no new bar yet
                              pass
                              time.sleep(poll_seconds)
                              except KeyboardInterrupt:
                              print("Stopping live loop (KeyboardInterrupt).")
                              break
                              except Exception as e:
                              print("Exception in live loop:", str(e))
                              time.sleep(5)
                              mt5_shutdown()

                              -------------------------

                              If run as script

                              -------------------------

                              if name == "main":
                              import argparse
                              parser = argparse.ArgumentParser()
                              parser.add_argument("--mode", choices=["train", "run"], default="train")
                              args = parser.parse_args()
                              if args.mode == "train":
                              print("Starting training...")
                              train_agent()
                              elif args.mode == "run":
                              print("Starting live/paper run loop...")
                              run_live_loop()

                              Metadata

                              Metadata

                              Assignees

                              No one assigned

                                Labels

                                No labels
                                No labels

                                Projects

                                No projects

                                  Milestone

                                  No milestone

                                  Relationships

                                  None yet

                                  Development

                                  No branches or pull requests

                                  Issue actions