jakegrigsby commited on Apr 8

Commit

ad1a770

verified ·

1 Parent(s): 155c5b6

Upload folder using huggingface_hub

Browse files

Files changed (23) hide show

medium-rl-maxq/ckpts/policy_weights/policy_epoch_0.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_10.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_12.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_14.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_16.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_18.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_2.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_20.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_22.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_24.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_26.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_28.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_30.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_32.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_34.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_36.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_38.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_4.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_40.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_42.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_6.pt +3 -0
medium-rl-maxq/ckpts/policy_weights/policy_epoch_8.pt +3 -0
medium-rl-maxq/config.txt +106 -0

medium-rl-maxq/ckpts/policy_weights/policy_epoch_0.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8c86a370761e9a977d4bd06b67bb5acff4e86a8f8bcd9feed53d3272a030bbf4
+size 202329850

medium-rl-maxq/ckpts/policy_weights/policy_epoch_10.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9ff4ec75e03f6a46cb43ace0996ffb93aad277abbf80c3c6701ff121e44a2219
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_12.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:438b228143d8862ed9704ea0af90f084170f7dcfa49e4c3b8197623c638d05e5
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_14.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:fc73f7779c575724a257ee21a5d9a2dce5be1375856ca655f4984d7935941322
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_16.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ce162b68cbe7e1d9983c2eecb5c9d5353b72ef252ad3ed7f33953438dc1d5d55
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_18.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d4c9c1311f69fe681cbbf8f11679d53bcb09bd60c90e0a00e50f65fe9241634e
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_2.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:232a7f79373be8dd052092e4bde5f4c43d96284538e384001c5b863493ddad0f
+size 202329850

medium-rl-maxq/ckpts/policy_weights/policy_epoch_20.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8eda51e03f2185f5b212c388d0b3db28a5c889b886d67cabc3b3839796dc2740
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_22.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b4347d484f991c0a9b29aa65f925b189f1b8c9dcb949d4434ab6622515350f12
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_24.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:34ae3d6158bb34e6af8c05c1d29e3ae0976410f790a976ab3d56474aab20c0a1
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_26.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:cb7e9c78e6b95d5dbefa5b435645ef371c7677f23b2a4e9ff02139cb5ac43bfa
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_28.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ab303f5de376afdfad6e951a02ce146837fd964381e7f734219a6abac9b1c7ae
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_30.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6dbb8825dc8f9fc0defcb0697a3204b12ea4df356e06131055f054d777c54a2e
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_32.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:45431fb51ccbd61e60db39b79215afc929d67e0814355e37c62e5b877f286100
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_34.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ca6c9ad6b65abd18bd475a9c47c9156accc176bf6fa9b0967e30b9451afe950f
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_36.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:896e4c23af40174389576bc8edb9aa5651de850184f31a3bdc261ce3964ba9eb
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_38.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f8527938327238920bab34b66e0164b4feb728139a216e684774a82cb0ad9f20
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_4.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:cb9cf2d60827dbe84b9e8318b30e5ff7f7b59e0c09d1f3961b91c362884e929f
+size 202329850

medium-rl-maxq/ckpts/policy_weights/policy_epoch_40.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1079af52a729228ba8153fcc280ffcd1a696ec3d3ab5d96c551effaa5967388a
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_42.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:379b45580ea21ec9dee7497dfb58a0bf155d23d222ba5a482fa1608875eba3ac
+size 202330106

medium-rl-maxq/ckpts/policy_weights/policy_epoch_6.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:979ce9ba924f49b70f65ac75a1748e9cd016f3b63ee5ca60b3270f90ed8719f0
+size 202329850

medium-rl-maxq/ckpts/policy_weights/policy_epoch_8.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:cfe584c16fce6ff0dd1ae3a228007b32c68d231921d282c046c4d2683a505008
+size 202329850

medium-rl-maxq/config.txt ADDED Viewed

	@@ -0,0 +1,106 @@

+# Parameters for Actor:
+# ==============================================================================
+Actor.activation = 'leaky_relu'
+Actor.cont_dist_kind = 'normal'
+Actor.d_hidden = 400
+Actor.dropout_p = 0.0
+Actor.gmm_modes = 5
+Actor.log_std_high = 2.0
+Actor.log_std_low = -5.0
+Actor.n_layers = 2
+# Parameters for Agent:
+# ==============================================================================
+Agent.fake_filter = False
+Agent.gamma = 0.999
+Agent.num_critics = 4
+Agent.num_critics_td = 2
+Agent.offline_coeff = 1.0
+Agent.online_coeff = 0.0
+Agent.popart = True
+Agent.reward_multiplier = 10.0
+Agent.tau = 0.003
+Agent.use_multigamma = True
+Agent.use_target_actor = True
+# Parameters for Experiment:
+# ==============================================================================
+Experiment.batches_per_update = 1
+Experiment.critic_loss_weight = 10.0
+Experiment.env_mode = 'async'
+Experiment.force_reset_train_envs_every = None
+Experiment.grad_clip = 1.0
+Experiment.has_replay_buffer_rights = True
+Experiment.l2_coeff = 0.001
+Experiment.learning_rate = 0.0001
+Experiment.local_time_optimizer = False
+Experiment.lr_warmup_steps = 500
+Experiment.mixed_precision = 'no'
+Experiment.padded_sampling = 'none'
+Experiment.save_trajs_as = 'npz'
+Experiment.stagger_traj_file_lengths = True
+Experiment.wandb_group_name = None
+# Parameters for FlashAttention:
+# ==============================================================================
+FlashAttention.window_size = (-1, -1)
+# Parameters for MetamonTstepEncoder:
+# ==============================================================================
+MetamonTstepEncoder.d_model = 100
+MetamonTstepEncoder.extra_emb_dim = 18
+MetamonTstepEncoder.n_heads = 5
+MetamonTstepEncoder.n_layers = 3
+MetamonTstepEncoder.scratch_tokens = 6
+MetamonTstepEncoder.token_mask_aug = False
+# Parameters for Multigammas:
+# ==============================================================================
+Multigammas.continuous = [0.1, 0.9, 0.95, 0.97, 0.99, 0.995]
+Multigammas.discrete = [0.1, 0.9, 0.95, 0.97, 0.99, 0.995]
+# Parameters for MultiModalEmbedding:
+# ==============================================================================
+MultiModalEmbedding.dropout = 0.05
+MultiModalEmbedding.numerical_tokens = 6
+# Parameters for NCritics:
+# ==============================================================================
+NCritics.activation = 'leaky_relu'
+NCritics.d_hidden = 400
+NCritics.dropout_p = 0.0
+NCritics.n_layers = 2
+# Parameters for PopArtLayer:
+# ==============================================================================
+PopArtLayer.beta = 0.0005
+PopArtLayer.init_nu = 100.0
+# Parameters for TformerTrajEncoder:
+# ==============================================================================
+TformerTrajEncoder.activation = 'leaky_relu'
+TformerTrajEncoder.causal = True
+TformerTrajEncoder.d_ff = 3072
+TformerTrajEncoder.d_model = 768
+TformerTrajEncoder.dropout_attn = 0.0
+TformerTrajEncoder.dropout_emb = 0.05
+TformerTrajEncoder.dropout_ff = 0.05
+TformerTrajEncoder.dropout_qkv = 0.0
+TformerTrajEncoder.head_scaling = True
+TformerTrajEncoder.n_heads = 8
+TformerTrajEncoder.n_layers = 6
+TformerTrajEncoder.norm = 'layer'
+TformerTrajEncoder.normformer_norms = True
+TformerTrajEncoder.sigma_reparam = True
+# Parameters for TimestepTransformer:
+# ==============================================================================
+# None.
+# Parameters for TokenEmbedding:
+# ==============================================================================
+# None.
+# Parameters for TransformerTurnEmbedding:
+# ==============================================================================
+TransformerTurnEmbedding.dropout = 0.05