default_settings: null
behaviors:
  Agent:
    trainer_type: ppo
    hyperparameters:
      batch_size: 128
      buffer_size: 2048
      learning_rate: 0.0003
      beta: 0.01
      epsilon: 0.2
      lambd: 0.95
      num_epoch: 3
      shared_critic: false
      learning_rate_schedule: linear
      beta_schedule: linear
      epsilon_schedule: linear
    network_settings:
      normalize: false
      hidden_units: 512
      num_layers: 2
      vis_encode_type: simple
      memory: null
      goal_conditioning_type: hyper
      deterministic: false
    reward_signals:
      curiosity:
        gamma: 0.99
        strength: 0.02
        network_settings:
          normalize: false
          hidden_units: 256
          num_layers: 2
          vis_encode_type: simple
          memory: null
          goal_conditioning_type: hyper
          deterministic: false
        learning_rate: 0.0003
        encoding_size: 256
      extrinsic:
        gamma: 0.99
        strength: 1.0
        network_settings:
          normalize: false
          hidden_units: 128
          num_layers: 2
          vis_encode_type: simple
          memory: null
          goal_conditioning_type: hyper
          deterministic: false
    init_path: null
    keep_checkpoints: 5
    checkpoint_interval: 500000
    max_steps: 450000
    time_horizon: 2048
    summary_freq: 4500
    threaded: true
    self_play: null
    behavioral_cloning: null
env_settings:
  env_path: c:/users/pdsie/documents/hivex/src/hivex/training/baseline/ml_agents/dev_environments/Hivex_WildfireResourceManagement_win
  env_args: null
  base_port: 5006
  num_envs: 1
  num_areas: 1
  seed: 5000
  max_lifetime_restarts: 10
  restarts_rate_limit_n: 1
  restarts_rate_limit_period_s: 60
engine_settings:
  width: 84
  height: 84
  quality_level: 5
  time_scale: 20
  target_frame_rate: -1
  capture_frame_rate: 60
  no_graphics: true
environment_parameters:
  difficulty:
    curriculum:
    - value:
        sampler_type: constant
        sampler_parameters:
          seed: 5000
          value: 2
      name: difficulty
      completion_criteria: null
  task:
    curriculum:
    - value:
        sampler_type: constant
        sampler_parameters:
          seed: 5001
          value: 2
      name: task
      completion_criteria: null
checkpoint_settings:
  run_id: WildfireResourceManagement/train/WildfireResourceManagement_difficulty_2_task_2_run_id_1_train
  initialize_from: null
  load_model: false
  resume: false
  force: false
  train_model: false
  inference: false
  results_dir: results
torch_settings:
  device: null
debug: false