Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
26 commits
Select commit Hold shift + click to select a range
56cea78
5.0 add druss
l1onh3art88 Aug 20, 2026
4594a37
Merge branch 'PufferAI:5.0' into 5.0
l1onh3art88 Aug 20, 2026
3a70878
missing bot code
l1onh3art88 Aug 21, 2026
1d5c5de
same DR now for opponents and players
l1onh3art88 Aug 21, 2026
caf5ece
Merge branch 'PufferAI:5.0' into 5.0
l1onh3art88 Aug 24, 2026
9fe2249
bot noise and hist policy noise
l1onh3art88 Aug 25, 2026
82bd869
selfplay with bot eval for sweep
l1onh3art88 Aug 26, 2026
fc21c3b
selfplay fixes
l1onh3art88 Aug 26, 2026
3a34f01
Merge branch '5.0' of https://github.com/l1onh3art88/PufferLib into 5.0
l1onh3art88 Aug 26, 2026
9dedc4c
fixing selfplay sweeps vs bots eval
l1onh3art88 Aug 26, 2026
4adc736
selfplay + constellation fixes
l1onh3art88 Aug 28, 2026
0cd9312
bug fixes
l1onh3art88 Aug 28, 2026
f7e0601
const
l1onh3art88 Aug 28, 2026
9838e78
revert
l1onh3art88 Aug 28, 2026
719dc04
robocode selfplay 93%
l1onh3art88 Aug 31, 2026
47af905
Merge branch 'PufferAI:5.0' into 5.0
l1onh3art88 Sep 2, 2026
f1e9233
system for multiple selfplay envs
l1onh3art88 Sep 2, 2026
f4e3668
Merge branch '5.0' of https://github.com/l1onh3art88/PufferLib into 5.0
l1onh3art88 Sep 2, 2026
760081f
sweep bug
l1onh3art88 Sep 2, 2026
2187be0
games per env
l1onh3art88 Sep 3, 2026
8426a17
new best against adaptive opponents
l1onh3art88 Sep 4, 2026
d681c14
Merge branch 'PufferAI:5.0' into 5.0
l1onh3art88 Sep 6, 2026
88ab972
slimeconfig
l1onh3art88 Sep 6, 2026
f5edfc5
Merge branch '5.0' of https://github.com/l1onh3art88/PufferLib into 5.0
l1onh3art88 Sep 6, 2026
1c3bae5
deletd jar
l1onh3art88 Sep 6, 2026
8495261
score
l1onh3art88 Sep 6, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 13 additions & 3 deletions config/default.ini
Original file line number Diff line number Diff line change
Expand Up @@ -42,8 +42,9 @@ hist_policy_percent = 0
# Selfplay-pool training. When enabled, hist_policy_percent of agent slots (see
# [vec]) play against older checkpoints from the current run. Opponents are
# resampled from the pool every opp_timeout_steps global steps (0 = never).
# At end of train, final policy is matched vs up to eval_pool_size opponents from
# the pool (mean A winrate becomes the sweep score when eval_games > 0).
# At end of train, final policy scoring (first match wins):
# eval_bot_games > 0 → mean env/perf vs the eval_bots ladder (1 Protein point)
# else eval_games > 0 → mean A winrate vs up to eval_pool_size pool opps
[selfplay]
enabled = 0
max_size = 16
Expand All @@ -52,6 +53,15 @@ seed = 42
opp_timeout_steps = 500_000_000
eval_pool_size = 8
eval_games = 0
# 0 = off. >0 = after train, eval final vs each bot in eval_bots for this many
# episodes each; mean env/perf becomes TrainResult.score (points=1 for Protein).
eval_bot_games = 0
# Ladder rungs, weakest first: the [env] bot_policy ids the env's header
# defines. Required when eval_bot_games > 0, e.g. eval_bots = 3,4,5,6.

# Per-rung [env] overrides for the bot ladder, as full section.key = value
# lines (e.g. env.dr = 0). Env-owned so core needs no per-env knowledge.
[bot_eval]

# Args used by your env's binding.c go here
[env]
Expand Down Expand Up @@ -97,7 +107,7 @@ verb_eps_anneal_start = 0.4
verb_eps_anneal_end = 1.0

[sweep]
metric = score
metric = score
metric_distribution = linear
goal = maximize
max_suggestion_cost = 3600
Expand Down
122 changes: 89 additions & 33 deletions config/robocode.ini
Original file line number Diff line number Diff line change
Expand Up @@ -3,56 +3,92 @@ env_name = robocode
checkpoint_interval = 100

[vec]
total_agents = 16384
total_agents = 8192
num_buffers = 4
num_threads = 4
num_policies = 1
hist_policy_percent = 0.0
hist_policy_hidden_size = 256
hist_policy_num_layers = 4
num_policies = 2
hist_policy_hidden_size = 512
hist_policy_num_layers = 7
hist_policy_percent = 0.0645349994

[selfplay]
enabled = 0
enabled = 1
max_size = 100
seed = 42
opp_timeout_steps = 100000000
eval_games = 4096
opp_timeout_steps = 10000000
eval_pool_size = 8
eval_games = 0
# 8192 ladder envs x 10 games each. Scripted bots keep their kNN across
# episodes, so one game per env measures only cold bots.
eval_bot_games = 81920
# wave_surfer, hawk_on_fire, raiko, drussgt (BOT_* in bots.h)
eval_bots = 3,4,5,6

# Ladder rungs run without domain randomization or curriculum noise.
# Keys are bare [env] names; the ladder adds the "env." prefix itself.

[bot_eval]
dr = 0
bot_cl_noise = 0
hist_cl_noise = 0

[env]
num_agents = 1
num_bots = 1
dr = 0.699881017
num_agents = 2
num_bots = 0
width = 800
height = 600
reward_damage = 0.0
reward_spot = 0.0
reward_melee_damage_inflicted = 0.007448376232810955
reward_damage_taken = -0.004918136926337855
reward_range_damage_inflicted = 0.017177545879507577
dr = 0.6
bot_policy = 3
reward_damage = 0.00266758003
reward_spot = 0
reward_melee_damage_inflicted = 0.0030519499
reward_damage_taken = -0.00483758003
reward_range_damage_inflicted = 0.0116466004
bot_policy = 6
max_ticks = 3000
bot_cl_decay = 0.0126646999
hist_cl_decay = 0
bot_cl_noise = 0
hist_cl_noise = 0

[policy]
hidden_size = 256
num_layers = 4
hidden_size = 512
num_layers = 7.04325008

[train]
gpus = 1
# vs wave-surfer confirmation. Do not start from the 14B repro.
total_timesteps = 200_000_000
learning_rate = 0.0072914747025563794
gamma = 0.9996207050319893
gae_lambda = 0.8883938238847433
minibatch_size = 8192
total_timesteps = 2999979980
learning_rate = 0.00123165001
anneal_lr = 1
min_lr_ratio = 0
gamma = 0.998830974
gae_lambda = 0.853752017
replay_ratio = 1.92679
clip_coef = 0.200000003
vf_coef = 1.23714006
vf_clip_coef = 0.221594006
max_grad_norm = 0.878167987
ent_coef = 9.99999975e-06
anneal_ent_coef = 0
min_ent_coef_ratio = 0.100000001
momentum = 0.918410003
minibatch_size = 16384
horizon = 128
vtrace = 0
vtrace_rho_clip = 1
vtrace_c_clip = 1
verb_eps = 0
verb_eps_anneal_start = 0.400000006
verb_eps_anneal_end = 1

[sweep]
gpus = 8

[sweep.train.total_timesteps]
distribution = log_normal
min = 5e8
max = 1e11
min = 2e8
max = 3e9
mean = 5e8
scale = auto
scale = time

[sweep.policy.hidden_size]
distribution = uniform_pow2
Expand All @@ -66,30 +102,50 @@ min = 1
max = 8
scale = auto

[sweep.vec.total_agents]
distribution = uniform_pow2
min = 256
max = 16384
scale = auto

[sweep.vec.hist_policy_percent]
distribution = uniform
min = 0.01
max = 1
mean = 0.25
scale = auto

[sweep.selfplay.opp_timeout_steps]
distribution = log_normal
min = 1e7
max = 1e9
mean = 1e8
scale = auto

[sweep.env.dr]
distribution = uniform
min = 0.0
max = 0.6
min = 0
max = 0.8
mean = 0.3
scale = auto

[sweep.env.reward_melee_damage_inflicted]
distribution = uniform
min = 0.0
min = 0
max = 0.02
mean = 0.005
scale = auto

[sweep.env.reward_range_damage_inflicted]
distribution = uniform
min = 0.0
min = 0
max = 0.02
mean = 0.005
scale = auto

[sweep.env.reward_damage_taken]
distribution = uniform
min = -0.02
max = 0.0
max = 0
mean = -0.005
scale = auto
79 changes: 58 additions & 21 deletions config/slimevolley.ini
Original file line number Diff line number Diff line change
@@ -1,47 +1,70 @@
[base]
env_name = slimevolley
# Same class of 5.0 trainer change that blocked drone: stale actor + RNN carry.
async = 0
reset_every_horizon = 1

[vec]
total_agents = 16384
total_agents = 2048
num_buffers = 4
num_threads = 2
num_policies = 2
hist_policy_percent = 0.171324745
hist_policy_hidden_size = 512
hist_policy_num_layers = 4

# Selfplay: slot 0 is the primary policy, slot 1 is a frozen checkpoint on
# hist_policy_percent of the envs. Final score is mean env/perf vs eval_bots.
[selfplay]
enabled = 1
max_size = 100
seed = 42
opp_timeout_steps = 165567856
eval_pool_size = 8
eval_games = 4096
# Set > 0 to score off the bot ladder instead of the pool eval.
eval_bot_games = 8192
eval_bots = 0

# 1v1: num_agents + num_bots must be 2. Selfplay instead of the bot with
# ./puffer train --env.num_agents=2 --env.num_bots=0 --selfplay.enabled=1
# --vec.num_policies=2
[env]
num_agents = 1
gamma = 0.99
num_agents = 2
num_bots = 0
bot_policy = 0
max_ticks = 3000

[policy]
hidden_size = 128
# Constellation #206 was 3.297; 5.0 truncates to int.
num_layers = 3
hidden_size = 512
num_layers = 4.63476706

[train]
gpus = 1
# 4.0 #206 solved at 115M; 5.0 needs ~250M without prio replay.
total_timesteps = 280000000
learning_rate = 0.00207143
total_timesteps = 151858608
learning_rate = 0.00140897371
anneal_lr = 1
min_lr_ratio = 0
gamma = 0.993389
gae_lambda = 0.984654
replay_ratio = 2.85524
clip_coef = 0.142887
gamma = 0.986717224
gae_lambda = 0.845131516
replay_ratio = 3.59742641
clip_coef = 0.116425134
vf_coef = 5
vf_clip_coef = 0.0465092
max_grad_norm = 0.1
ent_coef = 0.00113432
momentum = 0.981078
vf_clip_coef = 2.56304932
max_grad_norm = 0.100000001
ent_coef = 9.99999975e-06
anneal_ent_coef = 0
min_ent_coef_ratio = 0.1
momentum = 0.932234406
minibatch_size = 4096
horizon = 32
# 4.0 always applied rho/c clips. vtrace=0 made those keys no-ops.
vtrace = 1
vtrace = 0
vtrace_rho_clip = 1.3635
vtrace_c_clip = 2.29725

[sweep]
# Lives margin in [0, 1]; the default cl_perf is a robocode curriculum metric
# this env does not log. Symmetric selfplay pins perf at 0.5, so a selfplay
# sweep scores off the pool eval below; vs the bot perf is the dense signal.
metric = perf
downsample = 5

[sweep.train.total_timesteps]
Expand All @@ -50,3 +73,17 @@ min = 1e8
max = 2e9
mean = 3e8
scale = time

[sweep.vec.hist_policy_percent]
distribution = uniform
min = 0.01
max = 0.5
mean = 0.15
scale = auto

[sweep.selfplay.opp_timeout_steps]
distribution = log_normal
min = 1e7
max = 1e9
mean = 1e8
scale = auto
Loading
Loading