kismet163 commited on
Commit
bbfe2b0
·
verified ·
1 Parent(s): 09d5367

Push agent to the Hub

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ replay.mp4 filter=lfs diff=lfs merge=lfs -text
README.md CHANGED
@@ -1,6 +1,6 @@
1
  ---
2
  tags:
3
- - CartPole-v1
4
  - ppo
5
  - deep-reinforcement-learning
6
  - reinforcement-learning
@@ -13,18 +13,18 @@ model-index:
13
  type: reinforcement-learning
14
  name: reinforcement-learning
15
  dataset:
16
- name: CartPole-v1
17
- type: CartPole-v1
18
  metrics:
19
  - type: mean_reward
20
- value: 191.10 +/- 70.41
21
  name: mean_reward
22
  verified: false
23
  ---
24
 
25
- # PPO Agent Playing CartPole-v1
26
 
27
- This is a trained model of a PPO agent playing CartPole-v1.
28
 
29
  # Hyperparameters
30
  ```python
@@ -32,30 +32,30 @@ model-index:
32
  'seed': 1
33
  'torch_deterministic': True
34
  'cuda': True
35
- 'track': False
36
- 'wandb_project_name': 'cleanRL-PPO'
37
  'wandb_entity': None
38
  'capture_video': False
39
- 'env_id': 'CartPole-v1'
40
- 'total_timesteps': 50000
41
- 'learning_rate': 0.00025
42
- 'num_envs': 4
43
- 'num_steps': 128
44
  'anneal_lr': True
45
  'gae': True
46
  'gamma': 0.99
47
  'gae_lambda': 0.95
48
- 'num_minibatches': 4
49
- 'update_epochs': 4
50
  'norm_adv': True
51
  'clip_coef': 0.2
52
  'clip_vloss': True
53
- 'ent_coef': 0.01
54
  'vf_coef': 0.5
55
  'max_grad_norm': 0.5
56
  'target_kl': 0.02
57
  'repo_id': 'kismet163/ppo-LunarLander-v2'
58
- 'batch_size': 512
59
- 'minibatch_size': 128}
60
  ```
61
 
 
1
  ---
2
  tags:
3
+ - HalfCheetahBulletEnv-v0
4
  - ppo
5
  - deep-reinforcement-learning
6
  - reinforcement-learning
 
13
  type: reinforcement-learning
14
  name: reinforcement-learning
15
  dataset:
16
+ name: HalfCheetahBulletEnv-v0
17
+ type: HalfCheetahBulletEnv-v0
18
  metrics:
19
  - type: mean_reward
20
+ value: -1605.41 +/- 93.38
21
  name: mean_reward
22
  verified: false
23
  ---
24
 
25
+ # PPO Agent Playing HalfCheetahBulletEnv-v0
26
 
27
+ This is a trained model of a PPO agent playing HalfCheetahBulletEnv-v0.
28
 
29
  # Hyperparameters
30
  ```python
 
32
  'seed': 1
33
  'torch_deterministic': True
34
  'cuda': True
35
+ 'track': True
36
+ 'wandb_project_name': 'PPO-RL'
37
  'wandb_entity': None
38
  'capture_video': False
39
+ 'env_id': 'HalfCheetahBulletEnv-v0'
40
+ 'total_timesteps': 2000
41
+ 'learning_rate': 0.0003
42
+ 'num_envs': 1
43
+ 'num_steps': 2048
44
  'anneal_lr': True
45
  'gae': True
46
  'gamma': 0.99
47
  'gae_lambda': 0.95
48
+ 'num_minibatches': 32
49
+ 'update_epochs': 10
50
  'norm_adv': True
51
  'clip_coef': 0.2
52
  'clip_vloss': True
53
+ 'ent_coef': 0.0
54
  'vf_coef': 0.5
55
  'max_grad_norm': 0.5
56
  'target_kl': 0.02
57
  'repo_id': 'kismet163/ppo-LunarLander-v2'
58
+ 'batch_size': 2048
59
+ 'minibatch_size': 64}
60
  ```
61
 
logs/events.out.tfevents.1735013103.yoshii.693.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5a9475387a1f10ad7b8cb4b1c008de9cb3c2d4fd75c762bdd797754fd0a59110
3
+ size 746
model.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8c3faacfb5d84e64cd22a826585cee7f78968ce7d3501f6841958d6deb6580cd
3
- size 40466
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d576ef1926481073887eb62cba4c0011a7bd1181cac1b5dbd3c4667b2377db0f
3
+ size 52941
replay.mp4 CHANGED
Binary files a/replay.mp4 and b/replay.mp4 differ
 
results.json CHANGED
@@ -1 +1 @@
1
- {"env_id": "CartPole-v1", "mean_reward": 191.1, "std_reward": 70.41086563876345, "n_evaluation_episodes": 30, "eval_datetime": "2024-12-23T02:09:26.124704"}
 
1
+ {"env_id": "HalfCheetahBulletEnv-v0", "mean_reward": -1605.4089231533642, "std_reward": 93.37885720626618, "n_evaluation_episodes": 30, "eval_datetime": "2024-12-23T21:06:09.713043"}