diff --git a/config/default.ini b/config/default.ini index 99181dcaa3..a5eafbe65f 100644 --- a/config/default.ini +++ b/config/default.ini @@ -42,8 +42,9 @@ hist_policy_percent = 0 # Selfplay-pool training. When enabled, hist_policy_percent of agent slots (see # [vec]) play against older checkpoints from the current run. Opponents are # resampled from the pool every opp_timeout_steps global steps (0 = never). -# At end of train, final policy is matched vs up to eval_pool_size opponents from -# the pool (mean A winrate becomes the sweep score when eval_games > 0). +# At end of train, final policy scoring (first match wins): +# eval_bot_games > 0 → mean env/perf vs the eval_bots ladder (1 Protein point) +# else eval_games > 0 → mean A winrate vs up to eval_pool_size pool opps [selfplay] enabled = 0 max_size = 16 @@ -52,6 +53,15 @@ seed = 42 opp_timeout_steps = 500_000_000 eval_pool_size = 8 eval_games = 0 +# 0 = off. >0 = after train, eval final vs each bot in eval_bots for this many +# episodes each; mean env/perf becomes TrainResult.score (points=1 for Protein). +eval_bot_games = 0 +# Ladder rungs, weakest first: the [env] bot_policy ids the env's header +# defines. Required when eval_bot_games > 0, e.g. eval_bots = 3,4,5,6. + +# Per-rung [env] overrides for the bot ladder, as full section.key = value +# lines (e.g. env.dr = 0). Env-owned so core needs no per-env knowledge. +[bot_eval] # Args used by your env's binding.c go here [env] @@ -97,7 +107,7 @@ verb_eps_anneal_start = 0.4 verb_eps_anneal_end = 1.0 [sweep] -metric = score +metric = score metric_distribution = linear goal = maximize max_suggestion_cost = 3600 diff --git a/config/robocode.ini b/config/robocode.ini index 32e9f79bfb..e61f1d949f 100644 --- a/config/robocode.ini +++ b/config/robocode.ini @@ -3,56 +3,92 @@ env_name = robocode checkpoint_interval = 100 [vec] -total_agents = 16384 +total_agents = 8192 num_buffers = 4 num_threads = 4 -num_policies = 1 -hist_policy_percent = 0.0 -hist_policy_hidden_size = 256 -hist_policy_num_layers = 4 +num_policies = 2 +hist_policy_hidden_size = 512 +hist_policy_num_layers = 7 +hist_policy_percent = 0.0645349994 [selfplay] -enabled = 0 +enabled = 1 max_size = 100 seed = 42 -opp_timeout_steps = 100000000 -eval_games = 4096 +opp_timeout_steps = 10000000 eval_pool_size = 8 +eval_games = 0 +# 8192 ladder envs x 10 games each. Scripted bots keep their kNN across +# episodes, so one game per env measures only cold bots. +eval_bot_games = 81920 +# wave_surfer, hawk_on_fire, raiko, drussgt (BOT_* in bots.h) +eval_bots = 3,4,5,6 + +# Ladder rungs run without domain randomization or curriculum noise. +# Keys are bare [env] names; the ladder adds the "env." prefix itself. + +[bot_eval] +dr = 0 +bot_cl_noise = 0 +hist_cl_noise = 0 [env] -num_agents = 1 -num_bots = 1 +dr = 0.699881017 +num_agents = 2 +num_bots = 0 width = 800 height = 600 -reward_damage = 0.0 -reward_spot = 0.0 -reward_melee_damage_inflicted = 0.007448376232810955 -reward_damage_taken = -0.004918136926337855 -reward_range_damage_inflicted = 0.017177545879507577 -dr = 0.6 -bot_policy = 3 +reward_damage = 0.00266758003 +reward_spot = 0 +reward_melee_damage_inflicted = 0.0030519499 +reward_damage_taken = -0.00483758003 +reward_range_damage_inflicted = 0.0116466004 +bot_policy = 6 max_ticks = 3000 +bot_cl_decay = 0.0126646999 +hist_cl_decay = 0 +bot_cl_noise = 0 +hist_cl_noise = 0 [policy] -hidden_size = 256 -num_layers = 4 +hidden_size = 512 +num_layers = 7.04325008 [train] gpus = 1 -# vs wave-surfer confirmation. Do not start from the 14B repro. -total_timesteps = 200_000_000 -learning_rate = 0.0072914747025563794 -gamma = 0.9996207050319893 -gae_lambda = 0.8883938238847433 -minibatch_size = 8192 +total_timesteps = 2999979980 +learning_rate = 0.00123165001 +anneal_lr = 1 +min_lr_ratio = 0 +gamma = 0.998830974 +gae_lambda = 0.853752017 +replay_ratio = 1.92679 +clip_coef = 0.200000003 +vf_coef = 1.23714006 +vf_clip_coef = 0.221594006 +max_grad_norm = 0.878167987 +ent_coef = 9.99999975e-06 +anneal_ent_coef = 0 +min_ent_coef_ratio = 0.100000001 +momentum = 0.918410003 +minibatch_size = 16384 horizon = 128 +vtrace = 0 +vtrace_rho_clip = 1 +vtrace_c_clip = 1 +verb_eps = 0 +verb_eps_anneal_start = 0.400000006 +verb_eps_anneal_end = 1 + +[sweep] +gpus = 8 [sweep.train.total_timesteps] distribution = log_normal -min = 5e8 -max = 1e11 +min = 2e8 +max = 3e9 mean = 5e8 -scale = auto +scale = time [sweep.policy.hidden_size] distribution = uniform_pow2 @@ -66,23 +102,43 @@ min = 1 max = 8 scale = auto +[sweep.vec.total_agents] +distribution = uniform_pow2 +min = 256 +max = 16384 +scale = auto + +[sweep.vec.hist_policy_percent] +distribution = uniform +min = 0.01 +max = 1 +mean = 0.25 +scale = auto + +[sweep.selfplay.opp_timeout_steps] +distribution = log_normal +min = 1e7 +max = 1e9 +mean = 1e8 +scale = auto + [sweep.env.dr] distribution = uniform -min = 0.0 -max = 0.6 +min = 0 +max = 0.8 mean = 0.3 scale = auto [sweep.env.reward_melee_damage_inflicted] distribution = uniform -min = 0.0 +min = 0 max = 0.02 mean = 0.005 scale = auto [sweep.env.reward_range_damage_inflicted] distribution = uniform -min = 0.0 +min = 0 max = 0.02 mean = 0.005 scale = auto @@ -90,6 +146,6 @@ scale = auto [sweep.env.reward_damage_taken] distribution = uniform min = -0.02 -max = 0.0 +max = 0 mean = -0.005 scale = auto diff --git a/config/slimevolley.ini b/config/slimevolley.ini index 487e26e562..499917f13d 100644 --- a/config/slimevolley.ini +++ b/config/slimevolley.ini @@ -1,47 +1,70 @@ [base] env_name = slimevolley # Same class of 5.0 trainer change that blocked drone: stale actor + RNN carry. -async = 0 -reset_every_horizon = 1 [vec] -total_agents = 16384 +total_agents = 2048 num_buffers = 4 num_threads = 2 +num_policies = 2 +hist_policy_percent = 0.171324745 +hist_policy_hidden_size = 512 +hist_policy_num_layers = 4 +# Selfplay: slot 0 is the primary policy, slot 1 is a frozen checkpoint on +# hist_policy_percent of the envs. Final score is mean env/perf vs eval_bots. +[selfplay] +enabled = 1 +max_size = 100 +seed = 42 +opp_timeout_steps = 165567856 +eval_pool_size = 8 +eval_games = 4096 +# Set > 0 to score off the bot ladder instead of the pool eval. +eval_bot_games = 8192 +eval_bots = 0 + +# 1v1: num_agents + num_bots must be 2. Selfplay instead of the bot with +# ./puffer train --env.num_agents=2 --env.num_bots=0 --selfplay.enabled=1 +# --vec.num_policies=2 [env] -num_agents = 1 -gamma = 0.99 +num_agents = 2 +num_bots = 0 +bot_policy = 0 +max_ticks = 3000 [policy] -hidden_size = 128 -# Constellation #206 was 3.297; 5.0 truncates to int. -num_layers = 3 +hidden_size = 512 +num_layers = 4.63476706 [train] gpus = 1 -# 4.0 #206 solved at 115M; 5.0 needs ~250M without prio replay. -total_timesteps = 280000000 -learning_rate = 0.00207143 +total_timesteps = 151858608 +learning_rate = 0.00140897371 anneal_lr = 1 min_lr_ratio = 0 -gamma = 0.993389 -gae_lambda = 0.984654 -replay_ratio = 2.85524 -clip_coef = 0.142887 +gamma = 0.986717224 +gae_lambda = 0.845131516 +replay_ratio = 3.59742641 +clip_coef = 0.116425134 vf_coef = 5 -vf_clip_coef = 0.0465092 -max_grad_norm = 0.1 -ent_coef = 0.00113432 -momentum = 0.981078 +vf_clip_coef = 2.56304932 +max_grad_norm = 0.100000001 +ent_coef = 9.99999975e-06 +anneal_ent_coef = 0 +min_ent_coef_ratio = 0.1 +momentum = 0.932234406 minibatch_size = 4096 horizon = 32 -# 4.0 always applied rho/c clips. vtrace=0 made those keys no-ops. -vtrace = 1 +vtrace = 0 vtrace_rho_clip = 1.3635 vtrace_c_clip = 2.29725 [sweep] +# Lives margin in [0, 1]; the default cl_perf is a robocode curriculum metric +# this env does not log. Symmetric selfplay pins perf at 0.5, so a selfplay +# sweep scores off the pool eval below; vs the bot perf is the dense signal. +metric = perf downsample = 5 [sweep.train.total_timesteps] @@ -50,3 +73,17 @@ min = 1e8 max = 2e9 mean = 3e8 scale = time + +[sweep.vec.hist_policy_percent] +distribution = uniform +min = 0.01 +max = 0.5 +mean = 0.15 +scale = auto + +[sweep.selfplay.opp_timeout_steps] +distribution = log_normal +min = 1e7 +max = 1e9 +mean = 1e8 +scale = auto diff --git a/ocean/robocode/agent_drussgt.h b/ocean/robocode/agent_drussgt.h new file mode 100644 index 0000000000..96b9a30254 --- /dev/null +++ b/ocean/robocode/agent_drussgt.h @@ -0,0 +1,718 @@ +// Structural adaptation of DrussGT 3.1.4159 (jk.mega.DrussGT) +// by Julian Kent / Skilgannon — open source (jar in third_party/). +// +// Core systems (speed-aware, not a full mega port): +// * Go-to wave surfing with SHORT path-sim → wave-hit GF (official idea) +// * One kNN danger profile per wave (not kNN-per-candidate) +// * First + light second wave +// * Visit flattener + hit-weighted surf samples +// * Gunheat-filtered waves + imaginary pre-fire waves +// * Visit-count DC gun (wave-pass GF logs) + cold linear/circular blend +// * Fair last-scan aim only +// +// Intentionally omitted for train SPS: multi-buffer VCS, precise GF ranges, +// dual guns, bullet shadows, shielding. +// +// License: keep open-source; credit Skilgannon / DrussGT. + +#ifndef ROBOCODE_AGENT_DRUSSGT_H +#define ROBOCODE_AGENT_DRUSSGT_H + +#include "agent_common.h" + +#ifndef DGT_KNN_K +#define DGT_KNN_K 6 +#endif +#ifndef DGT_GUN_K +#define DGT_GUN_K 10 +#endif +#ifndef DGT_GOTO_CANDS +#define DGT_GOTO_CANDS 8 +#endif +#ifndef DGT_SIM_STEPS +#define DGT_SIM_STEPS 12 +#endif +#ifndef DGT_BEST_DIST +#define DGT_BEST_DIST 450.0f +#endif +#ifndef DGT_DANGER_BINS +#define DGT_DANGER_BINS 17 +#endif + +// Manhattan weights (official metric). 8 attributes. +static const float DGT_FEAT_W[DGT_FEATS] = { + 5.00f, // distance / 900 + 4.00f, // |lat vel| / 8 + 2.00f, // adv vel / 8 + 3.00f, // 1/(1+k*tsdc) + 2.50f, // 1/(1+k*tsdecel) + 2.00f, // accel + 3.00f, // wall + 2.00f, // dist-last-10 / 80 +}; + +static inline float dgt_norm_time(float t, float k) { + return 1.0f / (1.0f + k * fmaxf(t, 0.0f)); +} + +static inline float dgt_kernel(float dgf, float inv_two_s2) { + float x = dgf * dgf * inv_two_s2; + return 1.0f / (1.0f + x + 0.5f * x * x); +} + +static inline void dgt_features(Robot* bot, float tgt_x, float tgt_y, Robocode* env, + int tick, int last_dir_change, int last_decel, + float last_v, float dist_last10, + float out[DGT_FEATS]) { + float dx = bot->x - tgt_x, dy = bot->y - tgt_y; + float dist = sqrtf(dx * dx + dy * dy); + float inv = (dist > 1e-6f) ? 1.0f / dist : 1.0f; + float ux = dx * inv, uy = dy * inv; + float bvx = cos_deg(bot->heading) * bot->v; + float bvy = sin_deg(bot->heading) * bot->v; + float wall_min = fminf(fminf(bot->x, env->width - bot->x), + fminf(bot->y, env->height - bot->y)); + float wall_half = fmaxf(fminf(env->width, env->height) * 0.5f, 1.0f); + out[0] = rb_clampf(dist * (1.0f / 900.0f), 0.0f, 1.2f); + out[1] = fabsf(-bvx * uy + bvy * ux) * 0.125f; + out[2] = rb_clampf((bvx * ux + bvy * uy) * 0.125f, -1.0f, 1.0f); + out[3] = dgt_norm_time((float)(tick - last_dir_change), 0.08f); + out[4] = dgt_norm_time((float)(tick - last_decel), 0.08f); + out[5] = rb_clampf((bot->v - last_v) * 0.5f, -1.0f, 1.0f); + out[6] = rb_clampf(wall_min / wall_half, 0.0f, 1.0f); + out[7] = rb_clampf(dist_last10 * (1.0f / 80.0f), 0.0f, 1.5f); +} + +static inline float dgt_feat_dist(const float* a, const float* b) { + float d = 0.0f; + for (int f = 0; f < DGT_FEATS; f++) { + d += fabsf((a[f] - b[f]) * DGT_FEAT_W[f]); + } + return d; +} + +static inline void dgt_knn_add(float knn_feats[][DGT_FEATS], float* knn_gf, + int* knn_n, int* knn_head, int cap, + const float feats[DGT_FEATS], float gf) { + int slot = *knn_head; + memcpy(knn_feats[slot], feats, DGT_FEATS * sizeof(float)); + knn_gf[slot] = gf; + *knn_head = (slot + 1) % cap; + if (*knn_n < cap) (*knn_n)++; +} + +static int dgt_knn_query(const float feats[DGT_FEATS], + const float knn_feats[][DGT_FEATS], + int knn_n, int k_max, + float* best_d, int* best_i) { + for (int k = 0; k < k_max; k++) { + best_d[k] = 1e18f; + best_i[k] = -1; + } + if (knn_n <= 0) return 0; + for (int n = 0; n < knn_n; n++) { + float d = dgt_feat_dist(feats, knn_feats[n]); + if (d >= best_d[k_max - 1]) continue; + int k = k_max - 1; + while (k > 0 && d < best_d[k - 1]) { + best_d[k] = best_d[k - 1]; + best_i[k] = best_i[k - 1]; + k--; + } + best_d[k] = d; + best_i[k] = n; + } + int count = 0; + for (int k = 0; k < k_max; k++) if (best_i[k] >= 0) count++; + return count; +} + +static void dgt_build_danger_profile(const float feats[DGT_FEATS], + const float knn_feats[][DGT_FEATS], + const float* knn_gf, int knn_n, + float dens[DGT_DANGER_BINS]) { + for (int b = 0; b < DGT_DANGER_BINS; b++) dens[b] = 0.0f; + if (knn_n <= 0) { + for (int b = 0; b < DGT_DANGER_BINS; b++) { + float gf = -1.0f + 2.0f * (float)b / (float)(DGT_DANGER_BINS - 1); + dens[b] = 0.25f / (1.0f + 8.0f * gf * gf); + } + return; + } + float best_d[DGT_KNN_K]; + int best_i[DGT_KNN_K]; + int nk = dgt_knn_query(feats, knn_feats, knn_n, DGT_KNN_K, best_d, best_i); + const float inv_two_s2 = 1.0f / (2.0f * 0.14f * 0.14f); + for (int k = 0; k < nk; k++) { + float w = 1.0f / (1.0f + best_d[k]); + float g0 = knn_gf[best_i[k]]; + for (int b = 0; b < DGT_DANGER_BINS; b++) { + float gf = -1.0f + 2.0f * (float)b / (float)(DGT_DANGER_BINS - 1); + dens[b] += w * dgt_kernel(gf - g0, inv_two_s2); + } + } +} + +static inline float dgt_profile_danger(const float dens[DGT_DANGER_BINS], float gf) { + gf = rb_clampf(gf, -1.0f, 1.0f); + float t = (gf + 1.0f) * 0.5f * (float)(DGT_DANGER_BINS - 1); + int i0 = (int)t; + if (i0 < 0) i0 = 0; + if (i0 >= DGT_DANGER_BINS - 1) return dens[DGT_DANGER_BINS - 1]; + float f = t - (float)i0; + return dens[i0] * (1.0f - f) + dens[i0 + 1] * f; +} + +static float dgt_gun_best_gf(const float feats[DGT_FEATS], + const float knn_feats[][DGT_FEATS], + const float* knn_gf, int knn_n) { + if (knn_n <= 0) return 0.0f; + float best_d[DGT_GUN_K]; + int best_i[DGT_GUN_K]; + int nk = dgt_knn_query(feats, knn_feats, knn_n, DGT_GUN_K, best_d, best_i); + if (nk <= 0) return 0.0f; + const float inv_two_s2 = 1.0f / (2.0f * 0.10f * 0.10f); + float best_gf = 0.0f; + float best_dens = -1.0f; + for (int pass = 0; pass < 2; pass++) { + int n_eval = (pass == 0) ? nk : 11; + for (int e = 0; e < n_eval; e++) { + float gf = (pass == 0) + ? knn_gf[best_i[e]] + : (-1.0f + 0.2f * (float)e); + float dens = 0.0f; + for (int k = 0; k < nk; k++) { + float w = 1.0f / (1.0f + best_d[k]); + dens += w * dgt_kernel(gf - knn_gf[best_i[k]], inv_two_s2); + } + if (dens > best_dens) { + best_dens = dens; + best_gf = gf; + } + } + } + return rb_clampf(best_gf, -1.0f, 1.0f); +} + +static inline float dgt_bullet_power(Robot* bot, float enemy_energy, float dist, + float enemy_fp, int bullets_hit, + int bullets_fired) { + float base = 1.95f; + float hitrate = (bullets_fired > 4) + ? (float)bullets_hit / (float)bullets_fired : 0.0f; + if (hitrate > 0.33f || dist < 180.0f) base = 2.95f; + if (dist > 600.0f && hitrate < 0.25f) base = 1.95f; + float power = base; + power = fminf(power, fmaxf(0.1f, ((float)bot->energy - 0.1f) * 0.25f)); + power = fminf(power, fmaxf(0.1f, enemy_energy * 0.25f)); + if (enemy_fp > 0.1f && enemy_fp < power && bot->energy < 40.0f) { + power = fmaxf(enemy_fp, 0.15f); + } + if (power >= 2.4f) power = 2.95f; + else if (power >= 1.5f) power = 1.95f; + else if (power >= 0.9f) power = 0.95f; + else if (power >= 0.4f) power = 0.45f; + else power = 0.15f; + return rb_clampf(power, 0.1f, fminf(3.0f, (float)bot->energy - 0.1f)); +} + +// Fair linear lead from last scan only. +static inline float dgt_linear_aim(Robot* bot, BotMem* m, float bspeed) { + float dx = m->last_x - bot->x; + float dy = m->last_y - bot->y; + float dist = fmaxf(sqrtf(dx * dx + dy * dy), 1.0f); + float dt = dist / fmaxf(bspeed, 0.1f); + float tvx = cos_deg(m->last_heading) * m->last_v; + float tvy = sin_deg(m->last_heading) * m->last_v; + float px = m->last_x + tvx * dt; + float py = m->last_y + tvy * dt; + // One refine pass. + dx = px - bot->x; dy = py - bot->y; + dist = fmaxf(sqrtf(dx * dx + dy * dy), 1.0f); + dt = dist / fmaxf(bspeed, 0.1f); + return rb_abs_bearing_deg(bot->x, bot->y, + m->last_x + tvx * dt, m->last_y + tvy * dt); +} + +// Fair circular lead from last scan (constant lateral orbit estimate). +static inline float dgt_circular_aim(Robot* bot, BotMem* m, float bspeed) { + float dx = m->last_x - bot->x; + float dy = m->last_y - bot->y; + float dist = fmaxf(sqrtf(dx * dx + dy * dy), 1.0f); + float dt = dist / fmaxf(bspeed, 0.1f); + float inv = 1.0f / dist; + float ux = dx * inv, uy = dy * inv; + float tvx = cos_deg(m->last_heading) * m->last_v; + float tvy = sin_deg(m->last_heading) * m->last_v; + float lat = -tvx * uy + tvy * ux; + float ang_vel = lat / dist; + float bearing = rb_abs_bearing_deg(bot->x, bot->y, m->last_x, m->last_y); + float new_bearing = bearing + ang_vel * dt * RB_R2D; + return rb_abs_bearing_deg(bot->x, bot->y, + bot->x + dist * cos_deg(new_bearing), + bot->y + dist * sin_deg(new_bearing)); +} + +static inline float dgt_hist_dist10(BotMem* m) { + if (m->dgt_hist_n < 2) return 0.0f; + int span = m->dgt_hist_n < 10 ? m->dgt_hist_n : 10; + int oldest = (m->dgt_hist_head - span + DGT_HIST) % DGT_HIST; + int latest = (m->dgt_hist_head - 1 + DGT_HIST) % DGT_HIST; + return rb_dist(m->dgt_hist_x[oldest], m->dgt_hist_y[oldest], + m->dgt_hist_x[latest], m->dgt_hist_y[latest]); +} + +static void dgt_fallback_move(Robocode* env, Robot* bot, BotMem* m) { + float abs_bearing = rb_abs_bearing_deg(bot->x, bot->y, m->last_x, m->last_y); + float distance = rb_dist(bot->x, bot->y, m->last_x, m->last_y); + float best_x = bot->x, best_y = bot->y; + float best_score = -1e18f; + for (int di = 0; di < 2; di++) { + float dir = (di == 0) ? 1.0f : -1.0f; + for (int t = 0; t < 6; t++) { + float orbit = 60.0f + 16.0f * (float)t; + float ang = (abs_bearing + dir * orbit) * RB_D2R; + float step = rb_clampf(distance * 0.4f + 100.0f, 90.0f, 190.0f); + float x = bot->x + step * cosf(ang); + float y = bot->y + step * sinf(ang); + if (x < 30.0f || x > env->width - 30.0f || + y < 30.0f || y > env->height - 30.0f) continue; + float d = rb_dist(x, y, m->last_x, m->last_y); + float score = -fabsf(d - DGT_BEST_DIST); + if (distance < 220.0f) score += 0.7f * d; + float wall = fminf(fminf(x, env->width - x), fminf(y, env->height - y)); + score += 0.35f * wall; + if (score > best_score) { + best_score = score; + best_x = x; + best_y = y; + } + } + } + rb_drive_to(env, bot, best_x, best_y); +} + +static inline float dgt_point_gf(DGTWave* w, float x, float y) { + float bearing = rb_abs_bearing_deg(w->ox, w->oy, x, y); + float gf_raw = rb_norm_deg(bearing - w->head_on); + float mea_deg = fmaxf(asinf(fminf(8.0f / w->speed, 1.0f)) * RB_R2D, 0.1f); + return rb_clampf((gf_raw / mea_deg) * (float)w->lat_sign, -1.5f, 1.5f); +} + +// Short path-sim: drive toward (tx,ty) until wave hits; return hit GF. +// Uses a cheap constant-speed slide (no full accel model) — captures go-to +// hit location for imminent waves without 120-step turn sim cost. +static float dgt_wave_hit_gf(Robot* bot, DGTWave* w, float tx, float ty, + int tick_now, float* out_hit_dist) { + float x = bot->x, y = bot->y; + float radius = (float)(tick_now - w->fire_tick) * w->speed; + float dx0 = tx - x, dy0 = ty - y; + float travel = sqrtf(dx0 * dx0 + dy0 * dy0); + float spd = 6.5f; // typical go-to cruise (between accel phases) + float ux = 0.0f, uy = 0.0f; + if (travel > 1.0f) { + ux = dx0 / travel; + uy = dy0 / travel; + } + for (int step = 0; step < DGT_SIM_STEPS; step++) { + float ddx = x - w->ox, ddy = y - w->oy; + float dist = sqrtf(ddx * ddx + ddy * ddy); + if (radius + w->speed >= dist - 18.0f) { + if (out_hit_dist) *out_hit_dist = dist; + return dgt_point_gf(w, x, y); + } + // Stop sliding once near destination. + float remx = tx - x, remy = ty - y; + if (remx * remx + remy * remy < 64.0f) { + x = tx; + y = ty; + } else { + x += ux * spd; + y += uy * spd; + } + radius += w->speed; + } + if (out_hit_dist) { + float ddx = x - w->ox, ddy = y - w->oy; + *out_hit_dist = sqrtf(ddx * ddx + ddy * ddy); + } + return dgt_point_gf(w, x, y); +} + +// Go-to surf: path-sim hit GF + profile dens. First wave full sim; second light. +static void dgt_surf_goto(Robocode* env, Robot* bot, BotMem* m) { + float abs_bearing = rb_abs_bearing_deg(bot->x, bot->y, m->last_x, m->last_y); + float distance = fmaxf(rb_dist(bot->x, bot->y, m->last_x, m->last_y), 1.0f); + + // Collect active waves sorted by TTI (soonest first). Prefer real over imag. + int idx[DGT_WAVES]; + float tti_a[DGT_WAVES]; + int n = 0; + for (int wi = 0; wi < DGT_WAVES; wi++) { + DGTWave* w = &m->dgt_waves[wi]; + if (!w->active) continue; + float tti = (rb_dist(bot->x, bot->y, w->ox, w->oy) + - (m->tick - w->fire_tick) * w->speed) / fmaxf(w->speed, 0.1f); + // Prefer real waves: inflate imaginary TTI slightly in sort key only. + float key = tti + (w->imaginary ? 50.0f : 0.0f); + idx[n] = wi; + tti_a[n] = key; + n++; + } + if (n == 0) { + dgt_fallback_move(env, bot, m); + return; + } + for (int a = 0; a < n; a++) { + for (int b = a + 1; b < n; b++) { + if (tti_a[b] < tti_a[a]) { + float tt = tti_a[a]; tti_a[a] = tti_a[b]; tti_a[b] = tt; + int ti = idx[a]; idx[a] = idx[b]; idx[b] = ti; + } + } + } + int surf_n = n > 2 ? 2 : n; + DGTWave* w0 = &m->dgt_waves[idx[0]]; + DGTWave* w1 = (surf_n > 1) ? &m->dgt_waves[idx[1]] : NULL; + + float dens0[DGT_DANGER_BINS]; + dgt_build_danger_profile(w0->feats, m->dgt_surf_feats, m->dgt_surf_gf, + m->dgt_surf_n, dens0); + // Second wave reuses dens0 (enemy gun state usually similar; saves a kNN). + + float best_x = bot->x, best_y = bot->y; + float best_danger = 1e18f; + float p0 = w0->power > 0.0f ? w0->power : ((20.0f - w0->speed) / 3.0f); + float dmg0 = 1.0f + 0.12f * (4.0f * p0); + // Path-sim only for imminent first waves; far waves use point GF. + float tti0 = (rb_dist(bot->x, bot->y, w0->ox, w0->oy) + - (m->tick - w0->fire_tick) * w0->speed) / fmaxf(w0->speed, 0.1f); + int do_sim = (tti0 < 55.0f && !w0->imaginary); + + for (int c = 0; c < DGT_GOTO_CANDS; c++) { + float t = (float)c / (float)(DGT_GOTO_CANDS - 1); + float gf_hint = -1.05f + 2.1f * t; + float ea = abs_bearing + 180.0f + gf_hint * 55.0f; + float ed = rb_clampf(0.5f * distance + 0.5f * DGT_BEST_DIST, 140.0f, 520.0f); + if (distance < 200.0f) ed = fmaxf(ed, 280.0f); + float x = m->last_x + ed * cos_deg(ea); + float y = m->last_y + ed * sin_deg(ea); + if ((c & 1) == 0) { + float orbit = abs_bearing + 90.0f * (gf_hint >= 0.0f ? 1.0f : -1.0f) + + gf_hint * 30.0f; + float rad = rb_clampf(0.4f * distance + 90.0f, 80.0f, 220.0f); + x = bot->x + rad * cos_deg(orbit); + y = bot->y + rad * sin_deg(orbit); + } + if (x < 28.0f || x > env->width - 28.0f || + y < 28.0f || y > env->height - 28.0f) continue; + + float hit_dist = rb_dist(x, y, w0->ox, w0->oy); + float egf0 = do_sim + ? dgt_wave_hit_gf(bot, w0, x, y, m->tick, &hit_dist) + : dgt_point_gf(w0, x, y); + float danger = dmg0 * dgt_profile_danger(dens0, egf0); + if (w0->imaginary) danger *= 0.55f; + + if (w1) { + float egf1 = dgt_point_gf(w1, x, y); + float p1 = w1->power > 0.0f ? w1->power : ((20.0f - w1->speed) / 3.0f); + float dmg1 = 1.0f + 0.12f * (4.0f * p1); + float wgt = 0.32f * dmg1; + if (w1->imaginary) wgt *= 0.5f; + danger += wgt * dgt_profile_danger(dens0, egf1); + } + + float d_en = rb_dist(x, y, m->last_x, m->last_y); + danger += 0.001f * fabsf(d_en - DGT_BEST_DIST); + danger += 0.0012f * fmaxf(0.0f, 300.0f - hit_dist); + if (danger < best_danger) { + best_danger = danger; + best_x = x; + best_y = y; + } + } + m->dgt_goto_x = best_x; + m->dgt_goto_y = best_y; + rb_drive_to(env, bot, best_x, best_y); +} + +static void dgt_update_gun_waves(BotMem* m) { + for (int i = 0; i < DGT_GUN_WAVES; i++) { + DGTGunWave* w = &m->dgt_gun_waves[i]; + if (!w->active) continue; + w->dist_traveled += w->speed; + float d = rb_dist(w->ox, w->oy, m->last_x, m->last_y); + if (w->dist_traveled + w->speed >= d) { + float bearing = rb_abs_bearing_deg(w->ox, w->oy, m->last_x, m->last_y); + float gf = rb_clampf( + (rb_norm_deg(bearing - w->abs_bearing) / fmaxf(w->mea, 0.1f)) * w->lat_dir, + -1.2f, 1.2f); + dgt_knn_add(m->dgt_gun_feats, m->dgt_gun_gf, &m->dgt_gun_n, + &m->dgt_gun_head, DGT_GUN_CAP, w->feats, gf); + m->dgt_current_gf = gf; + w->active = 0; + } else if (w->dist_traveled > 1200.0f) { + w->active = 0; + } + } +} + +static void dgt_expire_and_flatten(BotMem* m, Robot* bot) { + for (int wi = 0; wi < DGT_WAVES; wi++) { + DGTWave* w = &m->dgt_waves[wi]; + if (!w->active) continue; + if (w->imaginary && (m->tick - w->fire_tick) > 14) { + w->active = 0; + continue; + } + float radius = (m->tick - w->fire_tick) * w->speed; + float dist_now = rb_dist(bot->x, bot->y, w->ox, w->oy); + if (radius >= dist_now + 18.0f || m->tick - w->fire_tick > 350) { + if (!w->imaginary) { + float gf = dgt_point_gf(w, bot->x, bot->y); + dgt_knn_add(m->dgt_surf_feats, m->dgt_surf_gf, &m->dgt_surf_n, + &m->dgt_surf_head, DGT_KNN_CAP, w->feats, gf); + m->dgt_waves_passed++; + } + w->active = 0; + } + } +} + +static void dgt_add_enemy_wave(BotMem* m, Robot* bot, Robocode* env, + float power, int imaginary) { + if (power < 0.1f || power > 3.01f) return; + DGTWave* w = &m->dgt_waves[m->dgt_wave_head]; + m->dgt_wave_head = (m->dgt_wave_head + 1) % DGT_WAVES; + w->ox = m->last_x; + w->oy = m->last_y; + w->speed = rb_bullet_speed(power); + w->power = power; + w->fire_tick = imaginary ? m->tick + 1 : m->tick - 1; + if (w->fire_tick < 0) w->fire_tick = 0; + w->head_on = rb_abs_bearing_deg(w->ox, w->oy, bot->x, bot->y); + float lat_sum = 0.0f; + for (int i = 0; i < 10; i++) lat_sum += fabsf(m->dgt_lat_hist[i]); + dgt_features(bot, w->ox, w->oy, env, m->tick, m->last_dir_change_tick, + m->last_dir_change_tick, m->dgt_last_v, lat_sum, w->feats); + float bvx = cos_deg(bot->heading) * bot->v; + float bvy = sin_deg(bot->heading) * bot->v; + float dx = bot->x - w->ox, dy = bot->y - w->oy; + float dist = fmaxf(sqrtf(dx * dx + dy * dy), 1.0f); + float lat = (-bvx * dy + bvy * dx) / dist; + w->lat_sign = (lat >= 0.0f) ? 1 : -1; + w->imaginary = imaginary; + w->active = 1; +} + +static void bot_drussgt_step(Robocode* env, int bot_idx, BotMem* m) { + Robot* bot = &env->robots[bot_idx]; + int target_idx = rb_nearest_agent(env, bot); + if (target_idx < 0) return; + if (!rb_scan_target(env, bot, m, target_idx)) return; + + if (!m->dgt_initialized) { + m->dgt_initialized = 1; + m->dgt_enemy_energy = (float)m->last_energy_seen; + m->dgt_enemy_firepower = 2.0f; + m->dgt_enemy_gunheat = 0.0f; + m->dgt_lat_dir = 1.0f; + m->dgt_enemy_lat_dir = 1.0f; + m->dgt_last_v = bot->v; + m->dgt_enemy_last_v = m->last_v; + m->dgt_enemy_dir_change_tick = m->tick; + m->dgt_enemy_decel_tick = m->tick; + m->dgt_goto_x = bot->x; + m->dgt_goto_y = bot->y; + for (int i = 0; i < DGT_WAVES; i++) m->dgt_waves[i].active = 0; + for (int i = 0; i < DGT_GUN_WAVES; i++) m->dgt_gun_waves[i].active = 0; + } + + m->dgt_hist_x[m->dgt_hist_head] = m->last_x; + m->dgt_hist_y[m->dgt_hist_head] = m->last_y; + m->dgt_hist_head = (m->dgt_hist_head + 1) % DGT_HIST; + if (m->dgt_hist_n < DGT_HIST) m->dgt_hist_n++; + + m->dgt_enemy_gunheat = fmaxf(0.0f, m->dgt_enemy_gunheat - 0.1f); + + float drop = m->dgt_enemy_energy - (float)m->last_energy_seen; + if (drop >= 0.1f && drop <= 3.0f && m->dgt_enemy_gunheat <= 0.15f) { + m->dgt_enemy_firepower = drop; + m->dgt_enemy_gunheat = 1.0f + drop / 5.0f; + int promoted = 0; + for (int wi = 0; wi < DGT_WAVES; wi++) { + DGTWave* w = &m->dgt_waves[wi]; + if (w->active && w->imaginary && fabsf(w->power - drop) < 0.45f) { + w->imaginary = 0; + w->power = drop; + w->speed = rb_bullet_speed(drop); + w->fire_tick = m->tick > 0 ? m->tick - 1 : 0; + promoted = 1; + break; + } + } + if (!promoted) dgt_add_enemy_wave(m, bot, env, drop, 0); + for (int wi = 0; wi < DGT_WAVES; wi++) { + if (m->dgt_waves[wi].active && m->dgt_waves[wi].imaginary) + m->dgt_waves[wi].active = 0; + } + } + m->dgt_enemy_energy = (float)m->last_energy_seen; + + if (m->dgt_enemy_gunheat <= 0.0f) { + int has_imag = 0; + for (int wi = 0; wi < DGT_WAVES; wi++) { + if (m->dgt_waves[wi].active && m->dgt_waves[wi].imaginary) { + has_imag = 1; + break; + } + } + if (!has_imag) { + float pred = m->dgt_enemy_firepower > 0.1f ? m->dgt_enemy_firepower : 2.0f; + dgt_add_enemy_wave(m, bot, env, pred, 1); + } + } + + float abs_bearing = rb_abs_bearing_deg(bot->x, bot->y, m->last_x, m->last_y); + float dxe = m->last_x - bot->x, dye = m->last_y - bot->y; + float dist = fmaxf(sqrtf(dxe * dxe + dye * dye), 1.0f); + float inv = 1.0f / dist; + float ux = dxe * inv, uy = dye * inv; + float bvx = cos_deg(bot->heading) * bot->v; + float bvy = sin_deg(bot->heading) * bot->v; + float lat_v = -bvx * uy + bvy * ux; + m->dgt_lat_hist[m->dgt_lat_hist_i % 10] = lat_v; + m->dgt_lat_hist_i++; + if (fabsf(lat_v) > 0.1f) { + float nd = lat_v > 0.0f ? 1.0f : -1.0f; + if (nd != m->dgt_lat_dir) { + m->dgt_lat_dir = nd; + m->last_dir_change_tick = m->tick; + } + } + float tvx = cos_deg(m->last_heading) * m->last_v; + float tvy = sin_deg(m->last_heading) * m->last_v; + float enemy_lat = -tvx * uy + tvy * ux; + if (fabsf(enemy_lat) > 0.05f) { + float ed = enemy_lat > 0.0f ? 1.0f : -1.0f; + if (ed != m->dgt_enemy_lat_dir) { + m->dgt_enemy_lat_dir = ed; + m->dgt_enemy_dir_change_tick = m->tick; + } + } + if (m->last_v < m->dgt_enemy_last_v - 0.5f) { + m->dgt_enemy_decel_tick = m->tick; + } + + dgt_expire_and_flatten(m, bot); + dgt_update_gun_waves(m); + dgt_surf_goto(env, bot, m); // every tick + + // ---- Gun ---- + int can_fire = (bot->gun_heat <= 0.0f && bot->energy > 1.0f + && m->last_energy_seen > 0.0f); + // Full DC while cooling into a shot so the gun is aligned at fire time. + if (bot->gun_heat <= 0.35f) { + float gfeats[DGT_FEATS]; + Robot fake = *bot; + fake.x = m->last_x; + fake.y = m->last_y; + fake.heading = m->last_heading; + fake.v = m->last_v; + dgt_features(&fake, bot->x, bot->y, env, m->tick, + m->dgt_enemy_dir_change_tick, m->dgt_enemy_decel_tick, + m->dgt_enemy_last_v, dgt_hist_dist10(m), gfeats); + float power = dgt_bullet_power(bot, (float)m->last_energy_seen, dist, + m->dgt_enemy_firepower, + m->dgt_bullets_hit, m->dgt_bullets_fired); + if (power >= bot->energy) power = fmaxf(0.1f, (float)bot->energy - 0.1f); + float bspeed = rb_bullet_speed(power); + float max_escape = asinf(fminf(8.0f / bspeed, 1.0f)) * RB_R2D; + float gf = dgt_gun_best_gf(gfeats, m->dgt_gun_feats, m->dgt_gun_gf, m->dgt_gun_n); + float aim = abs_bearing + m->dgt_enemy_lat_dir * gf * max_escape; + // Cold gun: linear + circular blend from last scan (fair). + if (m->dgt_gun_n < 16) { + float lin = dgt_linear_aim(bot, m, bspeed); + float circ = dgt_circular_aim(bot, m, bspeed); + float blend = 0.55f * lin + 0.45f * circ; + float w = (float)m->dgt_gun_n / 16.0f; + aim = w * aim + (1.0f - w) * blend; + } else if (m->dgt_gun_n < 40) { + float lin = dgt_linear_aim(bot, m, bspeed); + aim = 0.88f * aim + 0.12f * lin; + } + float gun_delta = rb_turn_gun_to(bot, aim); + if (can_fire && fabsf(gun_delta) < 2.0f && power < bot->energy) { + fire(env, bot, bot_idx, power); + m->dgt_bullets_fired++; + DGTGunWave* gw = &m->dgt_gun_waves[m->dgt_gun_wave_head]; + m->dgt_gun_wave_head = (m->dgt_gun_wave_head + 1) % DGT_GUN_WAVES; + gw->ox = bot->x; + gw->oy = bot->y; + gw->abs_bearing = abs_bearing; + gw->lat_dir = m->dgt_enemy_lat_dir; + gw->speed = bspeed; + gw->mea = max_escape; + gw->dist_traveled = 0.0f; + memcpy(gw->feats, gfeats, DGT_FEATS * sizeof(float)); + gw->active = 1; + } + } else { + float aim = abs_bearing + m->dgt_enemy_lat_dir * 0.2f * + (asinf(fminf(8.0f / 14.0f, 1.0f)) * RB_R2D); + rb_turn_gun_to(bot, aim); + } + + rb_turn_radar_to(bot, abs_bearing, 10.0f); + m->dgt_last_v = bot->v; + m->dgt_enemy_last_v = m->last_v; +} + +static void dgt_on_hit_by_bullet(BotMem* m, float bullet_heading, float bullet_power) { + float speed = rb_bullet_speed(bullet_power); + DGTWave* best = NULL; + int best_age = -1; + for (int wi = 0; wi < DGT_WAVES; wi++) { + DGTWave* w = &m->dgt_waves[wi]; + if (!w->active) continue; + if (fabsf(w->speed - speed) > 0.6f) continue; + int age = m->tick - w->fire_tick; + if (age > best_age) { best_age = age; best = w; } + } + if (best == NULL) return; + float gf_raw = rb_norm_deg(bullet_heading - best->head_on); + float mea_deg = fmaxf(asinf(fminf(8.0f / best->speed, 1.0f)) * RB_R2D, 0.1f); + float gf = rb_clampf((gf_raw / mea_deg) * (float)best->lat_sign, -1.5f, 1.5f); + // Triple-weight hits vs flattener visits. + for (int r = 0; r < 3; r++) { + dgt_knn_add(m->dgt_surf_feats, m->dgt_surf_gf, &m->dgt_surf_n, + &m->dgt_surf_head, DGT_KNN_CAP, best->feats, gf); + } + m->dgt_hits_taken++; + best->active = 0; +} + +static void dgt_on_bullet_hit(BotMem* m, float hit_x, float hit_y) { + m->dgt_bullets_hit++; + DGTGunWave* best = NULL; + float best_err = 1e18f; + for (int i = 0; i < DGT_GUN_WAVES; i++) { + DGTGunWave* w = &m->dgt_gun_waves[i]; + if (!w->active) continue; + float err = fabsf(rb_dist(w->ox, w->oy, hit_x, hit_y) - w->dist_traveled); + if (err < best_err) { best_err = err; best = w; } + } + if (best == NULL) return; + float bearing = rb_abs_bearing_deg(best->ox, best->oy, hit_x, hit_y); + float gf = rb_clampf( + (rb_norm_deg(bearing - best->abs_bearing) / fmaxf(best->mea, 0.1f)) * best->lat_dir, + -1.2f, 1.2f); + dgt_knn_add(m->dgt_gun_feats, m->dgt_gun_gf, &m->dgt_gun_n, + &m->dgt_gun_head, DGT_GUN_CAP, best->feats, gf); + dgt_knn_add(m->dgt_gun_feats, m->dgt_gun_gf, &m->dgt_gun_n, + &m->dgt_gun_head, DGT_GUN_CAP, best->feats, gf); + best->active = 0; +} + +#endif // ROBOCODE_AGENT_DRUSSGT_H diff --git a/ocean/robocode/bot_tournament.c b/ocean/robocode/bot_tournament.c new file mode 100644 index 0000000000..8769922fef --- /dev/null +++ b/ocean/robocode/bot_tournament.c @@ -0,0 +1,150 @@ +// Round-robin tournament among scripted robocode bots (3,4,5,6). +// +// Compile from repo root: +// gcc -O2 -Iocean/robocode -Isrc -Iraylib-5.5_linux_amd64/include \ +// -o build/bot_tournament ocean/robocode/bot_tournament.c \ +// -Lraylib-5.5_linux_amd64/lib -lraylib -lm -lpthread -ldl \ +// -Wl,-rpath,$PWD/raylib-5.5_linux_amd64/lib +// +// Usage: +// ./build/bot_tournament [games_per_pair] [seed] +// +// Policies: 3=wave_surfer 4=hawk_on_fire 5=raiko 6=drussgt + +#include +#include +#include + +#include "robocode.h" + +static const char* policy_name(int p) { + switch (p) { + case 3: return "wave_surfer"; + case 4: return "hawk_on_fire"; + case 5: return "raiko"; + case 6: return "drussgt"; + default: return "unknown"; + } +} + +// Returns 0 if policy_a wins, 1 if policy_b wins, -1 draw. +static int play_one(int policy_a, int policy_b, int max_ticks, unsigned int seed) { + Robocode env; + memset(&env, 0, sizeof(env)); + env.num_agents = 0; + env.num_bots = 2; + env.width = 800; + env.height = 600; + env.max_ticks = max_ticks; + env.bot_policy = policy_a; + env.bot_policy_1 = policy_b; + env.bot_match_winner = -2; + env.rng = seed; + env.dr = 0.0f; + env.client = NULL; + + init(&env); + puf_reset(&env); + + for (int step = 0; step < max_ticks + 5; step++) { + env.bot_match_winner = -2; + puf_step(&env); + if (env.bot_match_winner != -2) { + int w = env.bot_match_winner; + puf_close(&env); + return w; + } + } + puf_close(&env); + return -1; +} + +int main(int argc, char** argv) { + int games = (argc > 1) ? atoi(argv[1]) : 100; + unsigned int seed0 = (argc > 2) ? (unsigned)atoi(argv[2]) : 42u; + if (games < 2) games = 2; + if (games % 2) games++; + + int policies[] = {3, 4, 5, 6}; + const int N = 4; + int wins[4][4]; + int draws[4][4]; + memset(wins, 0, sizeof(wins)); + memset(draws, 0, sizeof(draws)); + int total_wins[4] = {0}; + int total_games[4] = {0}; + + printf("Bot tournament: 3,4,5,6 | games/pair=%d (half each side) | seed=%u\n", + games, seed0); + printf("Field 800x600 max_ticks=3000\n\n"); + fflush(stdout); + + unsigned int game_i = 0; + int half = games / 2; + for (int i = 0; i < N; i++) { + for (int j = i + 1; j < N; j++) { + int pa = policies[i], pb = policies[j]; + int w_a = 0, w_b = 0, d = 0; + for (int g = 0; g < half; g++) { + int r = play_one(pa, pb, 3000, seed0 + 10007u * (++game_i)); + if (r == 0) w_a++; + else if (r == 1) w_b++; + else d++; + } + for (int g = 0; g < half; g++) { + int r = play_one(pb, pa, 3000, seed0 + 10007u * (++game_i)); + if (r == 0) w_b++; + else if (r == 1) w_a++; + else d++; + } + wins[i][j] = w_a; + wins[j][i] = w_b; + draws[i][j] = draws[j][i] = d; + total_wins[i] += w_a; + total_wins[j] += w_b; + total_games[i] += games; + total_games[j] += games; + printf("%-14s vs %-14s | wins %3d-%3d draws=%3d | score_wr=%.3f\n", + policy_name(pa), policy_name(pb), w_a, w_b, d, + (w_a + 0.5 * d) / (double)games); + fflush(stdout); + } + } + + printf("\n=== Win matrix (row beat col) ===\n"); + printf("%16s", ""); + for (int j = 0; j < N; j++) printf("%14s", policy_name(policies[j])); + printf("\n"); + for (int i = 0; i < N; i++) { + printf("%16s", policy_name(policies[i])); + for (int j = 0; j < N; j++) { + if (i == j) printf("%14s", "—"); + else printf("%14d", wins[i][j]); + } + printf("\n"); + } + + double score[4]; + int order[4] = {0, 1, 2, 3}; + for (int i = 0; i < N; i++) { + score[i] = (double)total_wins[i]; + for (int j = 0; j < N; j++) if (i != j) score[i] += 0.5 * draws[i][j]; + } + for (int a = 0; a < N; a++) { + for (int b = a + 1; b < N; b++) { + if (score[order[b]] > score[order[a]]) { + int t = order[a]; order[a] = order[b]; order[b] = t; + } + } + } + + printf("\n=== Ranking (score = wins + 0.5*draws) ===\n"); + for (int r = 0; r < N; r++) { + int i = order[r]; + double wr = score[i] / (double)total_games[i]; + printf("#%d %-14s score=%.1f / %d wr=%.3f pure_wins=%d\n", + r + 1, policy_name(policies[i]), score[i], total_games[i], wr, + total_wins[i]); + } + return 0; +} diff --git a/ocean/robocode/bots.h b/ocean/robocode/bots.h index 01b9f8c810..43e9c562fc 100644 --- a/ocean/robocode/bots.h +++ b/ocean/robocode/bots.h @@ -29,6 +29,7 @@ typedef enum { BOT_WAVE_SURFER = 3, BOT_HAWK_ON_FIRE = 4, BOT_RAIKO = 5, + BOT_DRUSSGT = 6, } BotPolicy; #define WS_NUM_WAVES 8 @@ -74,6 +75,38 @@ typedef struct { int active; } RBRaikoWave; +// DrussGT adaptation sizes (agent_drussgt.h). Moderate caps for train SPS. +#ifndef DGT_WAVES +#define DGT_WAVES 10 +#define DGT_GUN_WAVES 14 +#define DGT_KNN_CAP 96 +#define DGT_GUN_CAP 160 +#define DGT_FEATS 8 +#define DGT_HIST 16 +#endif +typedef struct { + float ox, oy; + float head_on; + float speed; + float power; + int fire_tick; + int lat_sign; + float feats[DGT_FEATS]; + int active; + int imaginary; +} DGTWave; + +typedef struct { + float ox, oy; + float abs_bearing; + float lat_dir; + float speed; + float mea; + float dist_traveled; + float feats[DGT_FEATS]; + int active; +} DGTGunWave; + struct BotMem { int tick; int orbit_dir; // -1, 0, +1 @@ -110,10 +143,49 @@ struct BotMem { int raiko_wave_head; int raiko_guess[RAIKO_DIST_BINS][RAIKO_GF_BINS]; RBRaikoWave raiko_waves[RAIKO_WAVES]; + + // DrussGT (agent_drussgt.h). kNN persists across episodes; waves clear on reset. + int dgt_initialized; + float dgt_enemy_gunheat; + float dgt_enemy_energy; + float dgt_enemy_firepower; + float dgt_lat_dir; + float dgt_enemy_lat_dir; + float dgt_last_v; + float dgt_enemy_last_v; + int dgt_enemy_dir_change_tick; + int dgt_enemy_decel_tick; + float dgt_goto_x, dgt_goto_y; + int dgt_wave_head; + DGTWave dgt_waves[DGT_WAVES]; + int dgt_surf_n, dgt_surf_head; + float dgt_surf_feats[DGT_KNN_CAP][DGT_FEATS]; + float dgt_surf_gf[DGT_KNN_CAP]; + int dgt_gun_n, dgt_gun_head; + float dgt_gun_feats[DGT_GUN_CAP][DGT_FEATS]; + float dgt_gun_gf[DGT_GUN_CAP]; + int dgt_gun_wave_head; + DGTGunWave dgt_gun_waves[DGT_GUN_WAVES]; + int dgt_bullets_hit, dgt_bullets_fired; + int dgt_hits_taken, dgt_waves_passed; + int dgt_hist_n, dgt_hist_head; + float dgt_hist_x[DGT_HIST]; + float dgt_hist_y[DGT_HIST]; + float dgt_lat_hist[10]; + int dgt_lat_hist_i; + float dgt_current_gf; }; #include "agent_hawk_on_fire.h" #include "agent_raiko.h" +#include "agent_drussgt.h" + +// Per-bot policy: bot_policy_1 overrides for the second bot in bot-vs-bot. +static inline int bot_policy_for(Robocode* env, int bot_idx) { + int bi = bot_idx - env->num_agents; + if (bi == 1 && env->bot_policy_1 >= 0) return env->bot_policy_1; + return env->bot_policy; +} // ---- Lifetime --------------------------------------------------------------- static inline void bot_mems_alloc(Robocode* env) { @@ -147,6 +219,15 @@ static inline void bot_mems_episode_reset(Robocode* env) { m->raiko_bearing_dir = m->raiko_bearing_dir == 0.0f ? 1.0f : m->raiko_bearing_dir; for (int wi = 0; wi < WS_NUM_WAVES; wi++) m->waves[wi].active = 0; for (int wi = 0; wi < RAIKO_WAVES; wi++) m->raiko_waves[wi].active = 0; + // DrussGT: clear per-round waves; keep kNN across episodes. + m->dgt_initialized = 0; + for (int wi = 0; wi < DGT_WAVES; wi++) m->dgt_waves[wi].active = 0; + for (int wi = 0; wi < DGT_GUN_WAVES; wi++) m->dgt_gun_waves[wi].active = 0; + m->dgt_hist_n = 0; + m->dgt_hist_head = 0; + m->dgt_lat_hist_i = 0; + for (int hi = 0; hi < 10; hi++) m->dgt_lat_hist[hi] = 0.0f; + m->dgt_current_gf = 0.0f; } } @@ -225,9 +306,14 @@ static inline void ws_add_sample(BotMem* m, const float feats[WS_NUM_FEATS], flo // sample to the kNN. static void bot_on_hit_by_bullet(Robocode* env, int bot_idx, float bullet_heading, float bullet_power) { - if (env->bot_policy != BOT_WAVE_SURFER) return; if (env->bot_mems == NULL) return; BotMem* m = &env->bot_mems[bot_idx - env->num_agents]; + int policy = bot_policy_for(env, bot_idx); + if (policy == BOT_DRUSSGT) { + dgt_on_hit_by_bullet(m, bullet_heading, bullet_power); + return; + } + if (policy != BOT_WAVE_SURFER) return; float speed = 20.0f - 3.0f * bullet_power; WSWave* best = NULL; int best_age = -1; @@ -251,6 +337,41 @@ static void bot_on_hit_by_bullet(Robocode* env, int bot_idx, best->active = 0; } +// Curriculum random bot: sample from the same discrete action tables agents use. +// Called when rand_unit(env) < env->bot_cl_noise instead of the scripted policy. +static void bot_random_step(Robocode* env, int bot_idx) { + Robot* bot = &env->robots[bot_idx]; + float move_atn = ACCEL_VALUES[rand_r(&env->rng) % 4]; + move(env, bot, move_atn); + + float turn_atn = TURN_VALUES[rand_r(&env->rng) % 9]; + float max_turn = 10.0f - 0.75f * fabsf(bot->v); + if (max_turn < 0.0f) max_turn = 0.0f; + float body = turn(&bot->heading, turn_atn, max_turn, 0.0f); + + float gun_atn = GUN_TURN_VALUES[rand_r(&env->rng) % 11]; + float gun = turn(&bot->gun_heading, gun_atn, 20.0f, body); + + float radar_atn = RADAR_TURN_VALUES[rand_r(&env->rng) % 11]; + bot->radar_heading_prev = bot->radar_heading; + turn(&bot->radar_heading, radar_atn, 45.0f, body + gun); + + float firepower = FIREPOWER_VALUES[rand_r(&env->rng) % 6]; + if (firepower > 0.0f) { + fire(env, bot, bot_idx, firepower); + } + + float px = bot->x, py = bot->y; + bot->x = fmaxf(16.0f, fminf(bot->x, env->width - 16.0f)); + bot->y = fmaxf(16.0f, fminf(bot->y, env->height - 16.0f)); + if (bot->x != px || bot->y != py) { + float wall_dmg = fabsf(bot->v) * 0.5f - 1.0f; + if (wall_dmg < 0.0f) wall_dmg = 0.0f; + bot->energy -= wall_dmg; + bot->v = 0.0f; + } +} + // ---- Main entry ------------------------------------------------------------- static void bot_step(Robocode* env, int bot_idx) { Robot* bot = &env->robots[bot_idx]; @@ -259,20 +380,31 @@ static void bot_step(Robocode* env, int bot_idx) { if (bot->energy < 0) return; if (bot->energy == 0) { bot->v = 0; return; } if (bot->gun_heat > 0) bot->gun_heat -= 0.1f; - if (env->bot_policy == BOT_STATIONARY) return; + int policy = bot_policy_for(env, bot_idx); + if (policy == BOT_STATIONARY) return; BotMem* m = &env->bot_mems[bot_idx - env->num_agents]; m->tick++; if (m->orbit_dir == 0) m->orbit_dir = 1; - if (env->bot_policy == BOT_HAWK_ON_FIRE) { + // Curriculum: randomly replace scripted policy with discrete noise. + if (env->bot_cl_noise > 0.0f && rand_unit(env) < env->bot_cl_noise) { + bot_random_step(env, bot_idx); + return; + } + + if (policy == BOT_HAWK_ON_FIRE) { bot_hawk_on_fire_step(env, bot_idx, m); return; } - if (env->bot_policy == BOT_RAIKO) { + if (policy == BOT_RAIKO) { bot_raiko_step(env, bot_idx, m); return; } + if (policy == BOT_DRUSSGT) { + bot_drussgt_step(env, bot_idx, m); + return; + } // Pick a target index. Normal training/eval bots target RL agents. // Temporary bot-vs-bot harnesses use num_agents=0, where bots target @@ -342,10 +474,10 @@ static void bot_step(Robocode* env, int bot_idx) { // Detect target fire from energy drop BEFORE overwriting last_energy_seen. float drop = m->last_scan_tick > 0 ? (m->last_energy_seen - tgt->energy) : 0.0f; bool fired = (drop > 0.0f && drop <= 3.0f); - if (env->bot_policy == BOT_SURFER && fired) { + if (policy == BOT_SURFER && fired) { m->orbit_dir = -m->orbit_dir; m->last_dir_change_tick = m->tick; - } else if (env->bot_policy == BOT_WAVE_SURFER && fired) { + } else if (policy == BOT_WAVE_SURFER && fired) { // Wave origin = target's PREVIOUS scanned position (where they // were the tick before they fired). speed inferred from drop. WSWave* w = &m->waves[m->wave_head]; @@ -372,7 +504,7 @@ static void bot_step(Robocode* env, int bot_idx) { if (m->last_scan_tick == 0) return; // still hunting for first contact // ---- Wave-surfer: expire missed waves, then choose orbit direction --- - if (env->bot_policy == BOT_WAVE_SURFER) { + if (policy == BOT_WAVE_SURFER) { for (int wi = 0; wi < WS_NUM_WAVES; wi++) { WSWave* w = &m->waves[wi]; if (!w->active) continue; diff --git a/ocean/robocode/robocode.h b/ocean/robocode/robocode.h index 914aa3a7ce..ff0fd67945 100644 --- a/ocean/robocode/robocode.h +++ b/ocean/robocode/robocode.h @@ -6,6 +6,7 @@ #include #include "raylib.h" typedef float obs_t; +#define PUF_HAS_BOT_POLICY #include "pufferenv.h" #define NUM_ACTIONS 5 @@ -60,6 +61,11 @@ struct Log { float policy_0_score; float policy_1_score; float draw_rate; + // Opponent action-noise curriculum faced this episode (pre-decay). + float bot_cl_noise; + float hist_cl_noise; + // Win credit * (1 - noise). Full credit only vs annealed (noise=0) opps. + float cl_perf; float n; }; @@ -129,8 +135,21 @@ struct Env { float reward_range_damage_inflicted_slot_1; float dr; int bot_policy; + // Optional second-bot policy for bot-vs-bot harnesses (num_agents=0). + // <0 means both bots use bot_policy (normal train/eval). + int bot_policy_1; + // Bot-vs-bot result when num_agents==0: -2 ongoing, 0/1 winner idx, -1 draw. + int bot_match_winner; BotMem* bot_mems; // per-bot scratch (allocated by bots.h) + // Opponent action-noise curriculum. With prob bot_cl_noise, scripted bots + // take uniform discrete actions; with prob hist_cl_noise, hist slot-1 + // actions are overwritten. Primary win decays the active noise by *_decay. + float bot_cl_noise; + float bot_cl_decay; + float hist_cl_noise; + float hist_cl_decay; + // Selfplay-pool tagging. tag = 0 means pure selfplay (both slots = primary // policy). tag > 0 means historical: slot 0 = primary, slot 1 = frozen // historical opponent. boundary_reached is set on game-end so the trainer @@ -141,6 +160,10 @@ struct Env { unsigned int rng; }; +static inline void puf_set_bot_policy(Env* env, int bot_policy) { + env->bot_policy = bot_policy; +} + void init(Robocode* env); static inline float robocode_get_float(Dict* kwargs, const char* key, float default_value) { @@ -177,6 +200,16 @@ void puf_init(Env* env, Dict* kwargs) { "reward_range_damage_inflicted_slot_1", env->reward_range_damage_inflicted); env->dr = robocode_get_float(kwargs, "dr", 0.0f); env->bot_policy = dict_get(kwargs, "bot_policy"); + env->bot_policy_1 = (int)robocode_get_float(kwargs, "bot_policy_1", -1.0f); + env->bot_match_winner = -2; + env->bot_cl_noise = robocode_get_float(kwargs, "bot_cl_noise", 0.0f); + env->bot_cl_decay = robocode_get_float(kwargs, "bot_cl_decay", 0.0f); + env->hist_cl_noise = robocode_get_float(kwargs, "hist_cl_noise", 0.0f); + env->hist_cl_decay = robocode_get_float(kwargs, "hist_cl_decay", 0.0f); + if (env->bot_cl_noise < 0.0f) env->bot_cl_noise = 0.0f; + if (env->bot_cl_noise > 1.0f) env->bot_cl_noise = 1.0f; + if (env->hist_cl_noise < 0.0f) env->hist_cl_noise = 0.0f; + if (env->hist_cl_noise > 1.0f) env->hist_cl_noise = 1.0f; env->agents[0].policy = 0; env->agents[1].policy = 1; env->agents[0].action_mask = NULL; @@ -196,6 +229,9 @@ void puf_log(Log* log, Dict* out) { dict_set(out, "policy_0_score", log->policy_0_score); dict_set(out, "policy_1_score", log->policy_1_score); dict_set(out, "draw_rate", log->draw_rate); + dict_set(out, "bot_cl_noise", log->bot_cl_noise); + dict_set(out, "hist_cl_noise", log->hist_cl_noise); + dict_set(out, "cl_perf", log->cl_perf); dict_set(out, "n", log->n); } @@ -231,6 +267,8 @@ void add_log(Robocode* env) { env->log.melee_damage_inflicted += env->logs[i].melee_damage_inflicted; env->log.damage_taken += env->logs[i].damage_taken; env->log.range_damage_inflicted += env->logs[i].range_damage_inflicted; + env->log.bot_cl_noise += env->logs[i].bot_cl_noise; + env->log.hist_cl_noise += env->logs[i].hist_cl_noise; env->log.n += 1.0f; } } @@ -371,6 +409,17 @@ void move(Robocode* env, Robot* robot, float distance) { robot->energy -= melee_damage; robot->v = 0; target->v = 0; // both robots stop on ramming collision (classic rule) + // Mirror bullet kills: +1/-1 terminal reward, perf = bots killed. + bool s_agent = robot_idx < env->num_agents; + bool t_agent = j < env->num_agents; + bool killed = target->energy <= 0.0f; + if (s_agent && killed) { + add_agent_reward(env, robot_idx, 1.0f); + if (!t_agent) env->logs[robot_idx].perf += 1.0f; + } + if (t_agent && killed) { + add_agent_reward(env, j, -1.0f); + } return; } @@ -444,10 +493,6 @@ static inline void sample_dr_triplet(Robocode* env, float* a, float* b, float* c } } -static inline void sample_agent_multipliers(Robocode* env, Robot* robot) { - sample_dr_triplet(env, &robot->speed_mult, &robot->handling_mult, &robot->power_mult); -} - static inline void assign_agent_reward_coefficients(Robocode* env, Robot* robot, int agent_idx) { if (agent_idx == 0) { robot->reward_melee_damage_inflicted = env->reward_melee_damage_inflicted_slot_0; @@ -568,6 +613,10 @@ void puf_reset(Robocode* env) { // boundary_reached is owned by selfplay alignment; do not clear it here. int total_robots = env->num_agents + env->num_bots; memset(env->bullets, 0, NUM_BULLETS * total_robots * sizeof(Bullet)); + // One DR draw per episode shared by all agents so selfplay slots stay fair + // (independent per-slot draws can make matches unwinnable). + float speed_mult = 1.0f, handling_mult = 1.0f, power_mult = 1.0f; + sample_dr_triplet(env, &speed_mult, &handling_mult, &power_mult); int idx = 0; float x, y; while (idx < total_robots) { @@ -596,7 +645,9 @@ void puf_reset(Robocode* env) { robot->gun_heat = 3; robot->bullet_idx = 0; if (idx < env->num_agents) { - sample_agent_multipliers(env, robot); + robot->speed_mult = speed_mult; + robot->handling_mult = handling_mult; + robot->power_mult = power_mult; assign_agent_reward_coefficients(env, robot, idx); env->logs[idx] = (Log){0}; } else { @@ -640,15 +691,41 @@ static inline int agent_terminal_outcome(Robocode* env) { // 0 draw. Historical accounting only applies when env->tag > 0. static inline void end_episode(Robocode* env, int outcome) { float s0_score = (outcome > 0) ? 1.0f : (outcome < 0) ? 0.0f : 0.5f; + // Noise faced this episode (pre-decay). Bot vs scripted, hist vs frozen, else 0. + float noise = 0.0f; + if (env->num_bots > 0 && env->num_agents == 1) + noise = env->bot_cl_noise; + else if (env->tag > 0) + noise = env->hist_cl_noise; + if (noise < 0.0f) noise = 0.0f; + if (noise > 1.0f) noise = 1.0f; // Scale by num_agents so that (policy_0_score / n) where n increments by // num_agents per episode in add_log gives the win rate directly. match() // reads this from env/policy_0_score after eval_log divides by n. env->log.policy_0_score += s0_score * env->num_agents; env->log.policy_1_score += (1.0f - s0_score) * env->num_agents; + env->log.cl_perf += s0_score * (1.0f - noise) * env->num_agents; if (outcome == 0) env->log.draw_rate += env->num_agents; if (env->tag > 0) { env->boundary_reached = 1; } + // Snapshot pre-decay noise for metrics (what the agent actually faced). + for (int a = 0; a < env->num_agents; a++) { + env->logs[a].bot_cl_noise = (env->num_bots > 0 && env->num_agents == 1) + ? noise : 0.0f; + env->logs[a].hist_cl_noise = (env->tag > 0) ? noise : 0.0f; + } + // Curriculum: primary win → harden bot / frozen opp (lower random rate). + if (outcome > 0 && env->num_bots > 0 && env->bot_cl_decay > 0.0f + && env->bot_cl_noise > 0.0f) { + env->bot_cl_noise -= env->bot_cl_decay; + if (env->bot_cl_noise < 0.0f) env->bot_cl_noise = 0.0f; + } + if (outcome > 0 && env->tag > 0 && env->hist_cl_decay > 0.0f + && env->hist_cl_noise > 0.0f) { + env->hist_cl_noise -= env->hist_cl_decay; + if (env->hist_cl_noise < 0.0f) env->hist_cl_noise = 0.0f; + } for (int a = 0; a < env->num_agents; a++) { *env->agents[a].terminals = 1.0f; } @@ -696,10 +773,27 @@ static void robocode_human_controls(Robocode *env) { } } +static inline void bot_vs_bot_check(Robocode* env) { + if (env->num_agents > 0 || env->num_bots <= 0) return; + int alive = 0, winner = -1; + for (int b = 0; b < env->num_bots; b++) { + if (env->robots[b].energy > 0.0f) { + alive++; + winner = b; + } + } + if (alive == 1) env->bot_match_winner = winner; + else if (alive == 0) env->bot_match_winner = -1; +} + void puf_step(Robocode* env) { // Timeout: all agents step in lockstep, so logs[0].episode_length is shared. env->tick += 1; if (env->tick > env->max_ticks) { + if (env->num_agents <= 0) { + env->bot_match_winner = -1; + return; + } end_episode(env, 0); // draw return; } @@ -788,6 +882,15 @@ void puf_step(Robocode* env) { if (!t_agent && s_agent) { bot_on_hit_by_bullet(env, j, bullet->heading, bullet->firepower); } + // DrussGT gun learning when a bot's bullet hits anyone. + if (!s_agent && env->bot_mems != NULL + && bot_policy_for(env, shooter) == BOT_DRUSSGT) { + int bmem = shooter - env->num_agents; + if (bmem >= 0 && bmem < env->num_bots) { + dgt_on_bullet_hit(&env->bot_mems[bmem], + target->x, target->y); + } + } bool killed = target->energy <= 0.0f; if (s_agent) { record_range_damage_inflicted(env, shooter, damage); @@ -803,9 +906,15 @@ void puf_step(Robocode* env) { } } if (env->num_bots > 0 && !any_bot_alive) { + if (env->num_agents <= 0) { + env->bot_match_winner = -1; + return; + } end_episode(env, +1); // primary wiped all bots return; } + bot_vs_bot_check(env); + if (env->num_agents <= 0 && env->bot_match_winner != -2) return; agent_outcome = agent_terminal_outcome(env); if (agent_outcome != 2) { end_episode(env, agent_outcome); @@ -823,6 +932,16 @@ void puf_step(Robocode* env) { continue; } + // Hist curriculum: with hist_cl_noise, replace frozen slot actions. + if (i > 0 && env->tag > 0 && env->hist_cl_noise > 0.0f + && rand_unit(env) < env->hist_cl_noise) { + atn[0] = (float)(rand_r(&env->rng) % 4); + atn[1] = (float)(rand_r(&env->rng) % 9); + atn[2] = (float)(rand_r(&env->rng) % 11); + atn[3] = (float)(rand_r(&env->rng) % 11); + atn[4] = (float)(rand_r(&env->rng) % 6); + } + // Cool down gun if (robot->gun_heat > 0) { robot->gun_heat -= 0.1f; @@ -889,9 +1008,15 @@ void puf_step(Robocode* env) { } } if (!any_bot_alive) { + if (env->num_agents <= 0) { + env->bot_match_winner = -1; + return; + } end_episode(env, +1); return; } + bot_vs_bot_check(env); + if (env->num_agents <= 0 && env->bot_match_winner != -2) return; } compute_observations(env); } diff --git a/ocean/slimevolley/slimevolley.h b/ocean/slimevolley/slimevolley.h index 6419b6f1fa..41643c8cb7 100644 --- a/ocean/slimevolley/slimevolley.h +++ b/ocean/slimevolley/slimevolley.h @@ -4,6 +4,7 @@ #include #include typedef float obs_t; +#define PUF_HAS_BOT_POLICY #include "pufferenv.h" // CONFIG @@ -13,6 +14,10 @@ typedef float obs_t; typedef Env SlimeVolley; +// Scripted opponents, weakest first. These are the [env] bot_policy ids and the +// rungs of [selfplay] eval_bots. +enum { BOT_ABRANTI = 0 }; + #define REF_W 48 #define REF_H REF_W #define REF_U 1.5 // ground height @@ -30,27 +35,13 @@ typedef Env SlimeVolley; #define WINDOW_WIDTH 1200 #define WINDOW_HEIGHT 500 #define FACTOR (WINDOW_WIDTH / REF_W) -#define PIXEL_MODE false -#define PIXEL_SCALE 4 -#define PIXEL_WIDTH (84*2) -#define PIXEL_HEIGHT 84 -#define MAX_TICKS 3000 - -// Day colors -const Color BALL_COLOR = {255, 200, 20, 255}; -const Color AGENT_LEFT_COLOR = {240, 75, 0, 255}; -const Color AGENT_RIGHT_COLOR = {0, 150, 255, 255}; -const Color PIXEL_AGENT_LEFT_COLOR = {240, 75, 0, 255}; -const Color PIXEL_AGENT_RIGHT_COLOR = {0, 150, 255, 255}; -const Color BACKGROUND_COLOR = {255, 255, 255, 255}; -const Color FENCE_COLOR = {240, 210, 130, 255}; + const Color COIN_COLOR = {240, 210, 130, 255}; +const Color FENCE_COLOR = {240, 210, 130, 255}; const Color GROUND_COLOR = {128, 227, 153, 255}; - -const Color PUFF_RED = (Color){187, 0, 0, 255}; -const Color PUFF_CYAN = (Color){0, 187, 187, 255}; -const Color PUFF_WHITE = (Color){241, 241, 241, 241}; -const Color PUFF_BACKGROUND = (Color){6, 24, 24, 255}; +const Color PUFF_RED = {187, 0, 0, 255}; +const Color PUFF_CYAN = {0, 187, 187, 255}; +const Color PUFF_BACKGROUND = {6, 24, 24, 255}; // UTILS typedef struct { @@ -82,13 +73,10 @@ typedef struct { float vx; float vy; float prev_x; - float prev_y; - Color c; } Ball; void ball_move(Ball* ball){ ball->prev_x = ball->x; - ball->prev_y = ball->y; ball->x += ball->vx * TIMESTEP; ball->y += ball->vy * TIMESTEP; } @@ -98,7 +86,28 @@ void ball_accelerate(Ball* ball, float ax, float ay){ ball->vy += ay * TIMESTEP; } +// Returns 0 while the ball is live, else the side that conceded: -1 left, 1 right. int ball_check_edges(Ball* ball){ + // Solid fence. The old test only fired on the tick the ball crossed a face + // (x inside the band, prev_x outside), so a ball that got inside the band + // any other way - a player bounce shoving it off the wall, a stub bounce + // dropping it in - drifted out the far side and scored for free. Resolve + // against the face the ball came from instead. That also catches a full + // tunnel in one step, and runs after the player bounces so a hit into the + // wall cannot push the ball through it. + float fence = REF_WALL_WIDTH/2 + ball->r; + if (ball->y <= REF_WALL_HEIGHT){ + float side = (ball->prev_x >= fence) - (ball->prev_x <= -fence); + if (side == 0){ + side = (ball->x >= 0) ? 1.0f : -1.0f; // already inside: nearest face + } + if (side*ball->x < fence){ + ball->x = side*(fence + NUDGE*TIMESTEP); + if (side*ball->vx < 0){ + ball->vx *= -FRICTION; + } + } + } if (ball->x <= (ball->r-REF_W/2)){ ball->vx *= -FRICTION; ball->x = ball->r-REF_W/2+NUDGE*TIMESTEP; @@ -107,28 +116,14 @@ int ball_check_edges(Ball* ball){ ball->vx *= -FRICTION; ball->x = REF_W/2-ball->r-NUDGE*TIMESTEP; } - if (ball->y <= (ball->r+REF_U)){ - ball->vy *= -FRICTION; - ball->y = ball->r+REF_U+NUDGE*TIMESTEP; - if (ball->x <= 0){ - return -1; - } - else{ - return 1; - } - } if (ball->y >= (REF_H-ball->r)){ ball->vy *= -FRICTION; ball->y = REF_H-ball->r-NUDGE*TIMESTEP; } - // fence: - if ((ball->x <= (REF_WALL_WIDTH/2+ball->r)) && (ball->prev_x > (REF_WALL_WIDTH/2+ball->r)) && (ball->y <= REF_WALL_HEIGHT)){ - ball->vx *= -FRICTION; - ball->x = REF_WALL_WIDTH/2+ball->r+NUDGE*TIMESTEP; - } - if ((ball->x >= (-REF_WALL_WIDTH/2-ball->r)) && (ball->prev_x < (-REF_WALL_WIDTH/2-ball->r)) && (ball->y <= REF_WALL_HEIGHT)){ - ball->vx *= -FRICTION; - ball->x = -REF_WALL_WIDTH/2-ball->r-NUDGE*TIMESTEP; + if (ball->y <= (ball->r+REF_U)){ + ball->vy *= -FRICTION; + ball->y = ball->r+REF_U+NUDGE*TIMESTEP; + return (ball->x <= 0) ? -1 : 1; } return 0; } @@ -170,51 +165,15 @@ void ball_bounce(Ball* ball, SphericalObject* p){ ball->vy = uy + p->vy; } -void ball_limit_speed(Ball* ball, float minSpeed, float maxSpeed){ +void ball_limit_speed(Ball* ball, float max_speed){ float mag2 = ball->vx*ball->vx+ball->vy*ball->vy; - if (mag2 > (maxSpeed*maxSpeed)){ + if (mag2 > (max_speed*max_speed)){ float mag = sqrt(mag2); - ball->vx /= mag; - ball->vy /= mag; - ball->vx *= maxSpeed; - ball->vy *= maxSpeed; + ball->vx *= max_speed/mag; + ball->vy *= max_speed/mag; } } -// Relative State -typedef struct { - //agent - float x; - float y; - float vx; - float vy; - //ball - float bx; - float by; - float bvx; - float bvy; - //opponent - float ox; - float oy; - float ovx; - float ovy; -} RelativeState; - -// WALL -typedef struct { - float x; - float y; - float w; - float h; - Color c; -} Wall; - -void wall_display(Wall* wall){ - Rectangle rec = {to_x_pixel(wall->x - wall->w/2), to_y_pixel(wall->y + wall->h/2), - to_p(wall->w), to_p(wall->h)}; - DrawRectangleRec(rec, wall->c); -} - // PLAYER (game entity; RL Agent is from pufferenv) typedef struct { float x; @@ -227,12 +186,10 @@ typedef struct { float desired_vx; float desired_vy; float* observations; - RelativeState *state; int lives; } Player; void agent_display(Player *agent, float bx, float by) { - // fprintf(stderr, "agent_display: x=%f, y=%f, r=%f, lives=%d, dir=%d\n", agent->x, agent->y, agent->r, agent->lives, agent->dir); float x = agent->x; float y = agent->y; float r = agent->r; @@ -298,43 +255,29 @@ void agent_display(Player *agent, float bx, float by) { } void agent_set_action(Player* agent, float* action){ - bool forward = false; - bool backward = false; - bool jump = false; - if (action[0] > 0){ - forward = true; - } - if (action[1] > 0){ - backward = true; - } - if (action[2] > 0){ - jump = true; - } agent->desired_vx = 0; agent->desired_vy = 0; + bool forward = action[0] > 0; + bool backward = action[1] > 0; if (forward && !backward){ agent->desired_vx = -PLAYER_SPEED_X; } if (backward && !forward){ agent->desired_vx = PLAYER_SPEED_X; } - if (jump){ + if (action[2] > 0){ agent->desired_vy = PLAYER_SPEED_Y; } } -void agent_move(Player* agent){ - agent->x += agent->vx * TIMESTEP; - agent->y += agent->vy * TIMESTEP; -} - void agent_update(Player* agent){ agent->vy += GRAVITY * TIMESTEP; if (agent->y <= REF_U + NUDGE*TIMESTEP){ // if grounded agent->vy = agent->desired_vy; } agent->vx = agent->desired_vx*agent->dir; - agent_move(agent); + agent->x += agent->vx * TIMESTEP; + agent->y += agent->vy * TIMESTEP; if (agent->y <= REF_U){ agent->y = REF_U; agent->vy = 0; @@ -350,162 +293,175 @@ void agent_update(Player* agent){ } } +// Ego-centric and mirrored by dir, so both sides see the same game. void agent_update_state(Player* agent, Ball* ball, Player* opponent){ - int obs_idx = 0; float* observations = agent->observations; - - // self - observations[obs_idx++] = agent->x*agent->dir / 10.0f; - observations[obs_idx++] = agent->y / 10.0f; - observations[obs_idx++] = agent->vx*agent->dir / 10.0f; - observations[obs_idx++] = agent->vy / 10.0f; - // ball - observations[obs_idx++] = ball->x*agent->dir / 10.0f; - observations[obs_idx++] = ball->y / 10.0f; - observations[obs_idx++] = ball->vx*agent->dir / 10.0f; - observations[obs_idx++] = ball->vy / 10.0f; - // opponent - observations[obs_idx++] = opponent->x*(-agent->dir) / 10.0f; // negate direction for opponent - observations[obs_idx++] = opponent->y / 10.0f; - observations[obs_idx++] = opponent->vx*(-agent->dir) / 10.0f ; // negate direction for opponent - observations[obs_idx++] = opponent->vy / 10.0f; + observations[0] = agent->x*agent->dir / 10.0f; + observations[1] = agent->y / 10.0f; + observations[2] = agent->vx*agent->dir / 10.0f; + observations[3] = agent->vy / 10.0f; + observations[4] = ball->x*agent->dir / 10.0f; + observations[5] = ball->y / 10.0f; + observations[6] = ball->vx*agent->dir / 10.0f; + observations[7] = ball->vy / 10.0f; + observations[8] = opponent->x*(-agent->dir) / 10.0f; + observations[9] = opponent->y / 10.0f; + observations[10] = opponent->vx*(-agent->dir) / 10.0f; + observations[11] = opponent->vy / 10.0f; } // ENV // Required struct. Only use floats! +typedef struct Log Log; struct Log { + // Lives margin from each slot's own view, mapped to [0, 1]. 0.5 is a draw, + // 1.0 a 5-0 sweep. Denser than win credit, so it is the sweep metric. float perf; - float score; + float score; // lives margin from each slot's own view float episode_return; float episode_length; + // Per-slot win credit for match() scoring + a selfplay sanity check. In + // selfplay both average to ~0.5; in match A=policy 0, B=policy 1, so + // policy_0_score is A's win rate. Each game hands out 1.0 of credit total + // (win=1.0, draw=0.5 each). Scaled by num_agents on accumulation so the + // eval_log mean (sum / n, n incremented per slot per episode) is the rate. + float policy_0_score; + float policy_1_score; + float draw_rate; float n; }; struct Env { Log log; - Agent agents[2]; // pufferenv RL agents (learning slots) - Player players[2]; // game entities (left/right) - Wall* ground; - Wall* fence; - Ball* fence_stub; - Ball* ball; + Agent agents[2]; // pufferenv RL agents (learning slots) + Player players[2]; // game entities: 0 = left, 1 = right + Ball ball; + float episode_return[2]; + float bot_observations[OBS_SIZE]; // right side when it is a scripted bot + float bot_actions[NUM_ATNS]; + int num_agents; // 1 (right side is a bot) or 2 (selfplay) + int num_bots; + int bot_policy; // BOT_* id the scripted side runs + int max_ticks; // episode timeout; configured per-run via [env].max_ticks int delay_frames; - int num_agents; // 1 or 2 learning agents; if 1, right side is a bot + int tick; + // Selfplay-pool tagging. tag = 0 means pure selfplay (both slots = primary + // policy). tag > 0 means historical: slot 0 = primary, slot 1 = frozen + // historical opponent. boundary_reached is set on game-end so the trainer + // can swap frozen banks only between games. int tag; int boundary_reached; - float* bot_observations; - float* bot_actions; - int tick; Texture2D puffers; unsigned int rng; }; +static inline void puf_set_bot_policy(Env* env, int bot_policy) { + env->bot_policy = bot_policy; +} + +void puf_init(Env* env, Dict* kwargs) { + env->num_agents = dict_get(kwargs, "num_agents"); + env->num_bots = dict_get(kwargs, "num_bots"); + env->bot_policy = dict_get(kwargs, "bot_policy"); + env->max_ticks = dict_get(kwargs, "max_ticks"); + assert(env->num_agents + env->num_bots == 2 + && "slimevolley is 1v1: env.num_agents + env.num_bots must be 2"); + assert(env->bot_policy == BOT_ABRANTI + && "slimevolley ships one bot: env.bot_policy must be 0"); + for (int i = 0; i < env->num_agents; i++) { + env->agents[i].policy = i; + env->agents[i].action_mask = NULL; + } +} + +void puf_log(Log* log, Dict* out) { + dict_set(out, "perf", log->perf); + dict_set(out, "score", log->score); + dict_set(out, "episode_return", log->episode_return); + dict_set(out, "episode_length", log->episode_length); + dict_set(out, "policy_0_score", log->policy_0_score); + dict_set(out, "policy_1_score", log->policy_1_score); + dict_set(out, "draw_rate", log->draw_rate); + dict_set(out, "n", log->n); +} + float randf(SlimeVolley* env) { return (float)rand_r(&env->rng) / (float)RAND_MAX; } -/* Recommended to have an init function of some kind if you allocate -* extra memory. This should be freed by puf_close. Don't forget to call -* this in binding.c! -*/ -void init(SlimeVolley* env) { - env->ground = (Wall*)malloc(sizeof(Wall)); - *env->ground = (Wall){0}; - env->ground->x = 0; - env->ground->y = REF_U / 2.0f; - env->ground->w = REF_W; - env->ground->h = REF_U; - env->ground->c = GROUND_COLOR; - env->fence = (Wall*)malloc(sizeof(Wall)); - *env->fence = (Wall){0}; - env->fence->x = 0; - env->fence->y = (REF_U + REF_WALL_HEIGHT)/2.0f; - env->fence->w = REF_WALL_WIDTH; - env->fence->h = REF_WALL_HEIGHT - 1.5f; - env->fence->c = FENCE_COLOR; - env->fence_stub = (Ball*)malloc(sizeof(Ball)); - *env->fence_stub = (Ball){0}; - env->fence_stub->x = 0; - env->fence_stub->y = REF_WALL_HEIGHT; - env->fence_stub->r = REF_WALL_WIDTH/2.0f; - env->fence_stub->c = FENCE_COLOR; - env->ball = (Ball*)malloc(sizeof(Ball)); - env->bot_observations = NULL; - env->bot_actions = NULL; - if (env->num_agents == 1) { - env->bot_observations = (float*)calloc(12, sizeof(float)); - env->bot_actions = (float*)calloc(3, sizeof(float)); - } +void new_match(SlimeVolley* env) { + env->ball = (Ball){ + .x = 0, + .y = REF_W/4, + .r = 0.5, + .vx = 40.0f*randf(env) - 20.0f, + .vy = 15.0f*randf(env) + 10.0f, + }; + env->delay_frames = INIT_DELAY_FRAMES; } // Required function void puf_reset(SlimeVolley* env) { env->tick = 0; - env->delay_frames = INIT_DELAY_FRAMES; - float ball_vx = 40.0f*randf(env) - 20.0f; - float ball_vy = 15.0f*randf(env) + 10.0f; - *env->ball = (Ball){0}; - env->ball->x = 0; - env->ball->y = REF_W/4; - env->ball->vx = ball_vx; - env->ball->vy = ball_vy; - env->ball->r = 0.5; - env->ball->c = BALL_COLOR; - for (int i=0; i < 2; i++) { - obs_t* observations; - if (i == 0) { - observations = env->agents[0].observations; - } else if (env->num_agents == 1) { - observations = env->bot_observations; - } else { - observations = env->agents[1].observations; - } - env->players[i] = (Player){0}; - env->players[i].x = i == 0 ? -REF_W/4 : REF_W/4; - env->players[i].y = REF_U; - env->players[i].r = 1.5; - env->players[i].dir = i == 0 ? -1 : 1; - env->players[i].c = i == 0 ? PUFF_RED : PUFF_CYAN; - env->players[i].lives = MAXLIVES; - env->players[i].observations = observations; - } - agent_update_state(&env->players[0], env->ball, &env->players[1]); - agent_update_state(&env->players[1], env->ball, &env->players[0]); + for (int i = 0; i < 2; i++) { + env->players[i] = (Player){ + .x = (i == 0) ? -REF_W/4 : REF_W/4, + .y = REF_U, + .r = 1.5, + .dir = (i == 0) ? -1 : 1, + .c = (i == 0) ? PUFF_RED : PUFF_CYAN, + .lives = MAXLIVES, + }; + env->players[i].observations = (i < env->num_agents) + ? env->agents[i].observations : env->bot_observations; + env->episode_return[i] = 0.0f; + } + new_match(env); + agent_update_state(&env->players[0], &env->ball, &env->players[1]); + agent_update_state(&env->players[1], &env->ball, &env->players[0]); } -float clip(float val, float min, float max) { - if (val < min) { - return min; - } else if (val > max) { - return max; - } - return val; +void add_reward(SlimeVolley* env, int slot, float reward) { + env->agents[slot].rewards[0] += reward; + env->episode_return[slot] += reward; } -void new_match(SlimeVolley* env) { - float ball_vx = 40.0f*randf(env) - 20.0f; - float ball_vy = 15.0f*randf(env) + 10.0f; - *env->ball = (Ball){0}; - env->ball->x = 0; - env->ball->y = REF_W/4; - env->ball->vx = ball_vx; - env->ball->vy = ball_vy; - env->ball->r = 0.5; - env->ball->c = BALL_COLOR; - env->delay_frames = INIT_DELAY_FRAMES; +// Every episode-end path. outcome: 1 slot 0 won, -1 slot 0 lost, 0 draw. +void end_episode(SlimeVolley* env, int outcome) { + float s0_score = (outcome > 0) ? 1.0f : (outcome < 0) ? 0.0f : 0.5f; + env->log.policy_0_score += s0_score * env->num_agents; + env->log.policy_1_score += (1.0f - s0_score) * env->num_agents; + if (outcome == 0) { + env->log.draw_rate += env->num_agents; + } + if (env->tag > 0) { + env->boundary_reached = 1; + } + int margin = env->players[0].lives - env->players[1].lives; + for (int i = 0; i < env->num_agents; i++) { + int slot_margin = (i == 0) ? margin : -margin; + env->log.perf += (slot_margin + MAXLIVES) / (2.0f*MAXLIVES); + env->log.score += slot_margin; + env->log.episode_return += env->episode_return[i]; + env->log.episode_length += env->tick; + env->log.n += 1.0f; + env->agents[i].terminals[0] = 1.0f; + } + puf_reset(env); } -void abranti_simple_bot(float* obs, float* action) { - // the bot policy. just 7 params but hard to beat. +void puf_bot(SlimeVolley* env, int bot_idx) { + // BOT_ABRANTI. just 7 params but hard to beat. + float* obs = env->players[bot_idx].observations; float x_agent = obs[0]; float x_ball = obs[4]; float vx_ball = obs[6]; - float backward = (-23.757145f * x_agent + 23.206863f * x_ball + 0.7943352f * vx_ball) + 1.4617119f; - float forward = -64.6463748f * backward + 22.4668393f; - action[0] = forward; - action[1] = backward; - action[2] = 1.0f; // always jump + float backward = -23.757145f*x_agent + 23.206863f*x_ball + + 0.7943352f*vx_ball + 1.4617119f; + env->bot_actions[0] = -64.6463748f * backward + 22.4668393f; + env->bot_actions[1] = backward; + env->bot_actions[2] = 1.0f; // always jump } // Hold Left Shift + A/D/W or arrows/space. @@ -529,37 +485,31 @@ static void slimevolley_human_controls(SlimeVolley *env) { // Required function void puf_step(SlimeVolley* env) { - env->agents[0].rewards[0] = 0; - env->agents[0].terminals[0] = 0; - if (env->num_agents == 2){ - env->agents[1].rewards[0] = 0; - env->agents[1].terminals[0] = 0; + env->tick++; + for (int i = 0; i < env->num_agents; i++) { + env->agents[i].rewards[0] = 0; + env->agents[i].terminals[0] = 0; } - + Player* left = &env->players[0]; Player* right = &env->players[1]; - Ball* ball = env->ball; + Ball* ball = &env->ball; - env->tick++; agent_set_action(left, env->agents[0].actions); - if (env->num_agents == 1){ - abranti_simple_bot(right->observations, env->bot_actions); + if (env->num_bots == 1){ + puf_bot(env, 1); agent_set_action(right, env->bot_actions); - } - else { + } else { agent_set_action(right, env->agents[1].actions); } - - // Update agent_update(left); agent_update(right); if (env->delay_frames == 0) { ball_accelerate(ball, 0, GRAVITY); - ball_limit_speed(ball, 0, MAX_BALL_SPEED); + ball_limit_speed(ball, MAX_BALL_SPEED); ball_move(ball); - } - else { + } else { env->delay_frames--; } @@ -569,45 +519,28 @@ void puf_step(SlimeVolley* env) { if (ball_is_colliding(ball, (SphericalObject*)right)){ ball_bounce(ball, (SphericalObject*)right); } - if (ball_is_colliding(ball, (SphericalObject*)env->fence_stub)){ - ball_bounce(ball, (SphericalObject*)env->fence_stub); + SphericalObject fence_stub = {.y = REF_WALL_HEIGHT, .r = REF_WALL_WIDTH/2}; + if (ball_is_colliding(ball, &fence_stub)){ + ball_bounce(ball, &fence_stub); } - int right_reward = -ball_check_edges(ball); - - if (right_reward != 0){ - new_match(env); - if (right_reward == -1){ - right->lives--; - env->agents[0].rewards[0] = 1.0f; - if (env->num_agents == 2){ - env->agents[1].rewards[0] = -1.0f; - } - } - else{ - left->lives--; - env->agents[0].rewards[0] = -1.0f; - if (env->num_agents == 2){ - env->agents[1].rewards[0] = 1.0f; - } + int conceded = ball_check_edges(ball); + if (conceded != 0){ + env->players[(conceded < 0) ? 0 : 1].lives--; + float reward = (conceded < 0) ? -1.0f : 1.0f; // slot 0's view + add_reward(env, 0, reward); + if (env->num_agents == 2){ + add_reward(env, 1, -reward); } + new_match(env); } agent_update_state(left, ball, right); agent_update_state(right, ball, left); - if (env->tick > MAX_TICKS || left->lives <= 0 || right->lives <= 0){ - env->agents[0].terminals[0] = 1; - if (env->num_agents == 2){ - env->agents[1].terminals[0] = 1; - } - env->log.perf += (left->lives - right->lives + 5.0f) / 10.0f; - env->log.score += (float)(left->lives - right->lives); - env->log.episode_return += (5.0f - right->lives); - env->log.episode_length += (float)env->tick; - env->log.n += 1; - puf_reset(env); - } - + if (env->tick > env->max_ticks || left->lives <= 0 || right->lives <= 0){ + int outcome = (left->lives > right->lives) - (left->lives < right->lives); + end_episode(env, outcome); + } } // Required function. Should handle creating the client on first call @@ -626,17 +559,16 @@ void puf_render(SlimeVolley* env) { slimevolley_human_controls(env); BeginDrawing(); ClearBackground(PUFF_BACKGROUND); - wall_display(env->ground); - wall_display(env->fence); - - // Fence - Ball* stub = env->fence_stub; - DrawCircleV( - (Vector2){to_x_pixel(stub->x), to_y_pixel(stub->y)}, - to_p(stub->r), stub->c - ); - - Ball* puff = env->ball; + DrawRectangleRec((Rectangle){to_x_pixel(-REF_W/2.0f), to_y_pixel(REF_U), + to_p(REF_W), to_p(REF_U)}, GROUND_COLOR); + float fence_h = REF_WALL_HEIGHT - REF_U; + DrawRectangleRec((Rectangle){to_x_pixel(-REF_WALL_WIDTH/2.0f), + to_y_pixel(REF_U + fence_h), to_p(REF_WALL_WIDTH), to_p(fence_h)}, + FENCE_COLOR); + DrawCircleV((Vector2){to_x_pixel(0), to_y_pixel(REF_WALL_HEIGHT)}, + to_p(REF_WALL_WIDTH/2.0f), FENCE_COLOR); + + Ball* puff = &env->ball; DrawTexturePro( env->puffers, (Rectangle){ @@ -656,7 +588,7 @@ void puf_render(SlimeVolley* env) { ); for (int i=0; i<2; i++) { - agent_display(&env->players[i], env->ball->x, env->ball->y); + agent_display(&env->players[i], env->ball.x, env->ball.y); } EndDrawing(); @@ -666,33 +598,7 @@ void puf_render(SlimeVolley* env) { // Required function. Should clean up anything you allocated // Do not free env->observations, actions, rewards, terminals void puf_close(SlimeVolley* env) { - free(env->ground); - free(env->fence); - free(env->fence_stub); - free(env->ball); - free(env->bot_observations); - free(env->bot_actions); if (IsWindowReady()) { CloseWindow(); } } - -void puf_init(Env* env, Dict* kwargs) { - env->num_agents = dict_get(kwargs, "num_agents"); - if (env->num_agents < 1) env->num_agents = 1; - if (env->num_agents > 2) env->num_agents = 2; - for (int i = 0; i < env->num_agents; i++) { - env->agents[i].policy = 0; - env->agents[i].action_mask = NULL; - } - init(env); -} - -void puf_log(Log* log, Dict* out) { - dict_set(out, "perf", log->perf); - dict_set(out, "score", log->score); - dict_set(out, "episode_return", log->episode_return); - dict_set(out, "episode_length", log->episode_length); - dict_set(out, "n", log->n); -} - diff --git a/src/constellation.c b/src/constellation.c index d401c69343..79b4f0364c 100644 --- a/src/constellation.c +++ b/src/constellation.c @@ -390,7 +390,7 @@ static void write_env(FILE* fp, const char* env, Table* table) { if (r > 0) { fputc(',', fp); } - fprintf(fp, "%.6g", table_get(table, r, c)); + fprintf(fp, "%.9g", table_get(table, r, c)); } fputc('\n', fp); } @@ -1034,7 +1034,7 @@ void copy_hypers_to_clipboard(Table *table, char* buffer, int row) { } else if (val == (long long)val) { buffer += sprintf(buffer, "%s = %lld\n", suffix, (long long)val); } else { - buffer += sprintf(buffer, "%s = %g\n", suffix, val); + buffer += sprintf(buffer, "%s = %.9g\n", suffix, val); } } buffer[0] = '\0'; diff --git a/src/ini.h b/src/ini.h index 6d80261e7e..0a22f241f2 100644 --- a/src/ini.h +++ b/src/ini.h @@ -370,6 +370,26 @@ static inline const char* puf_ini_get_str(Ini* ini, const char* section, return dict_get_str(puf_ini_section(ini, section, 0), key); } +// Reads a key as a list of numbers. A scalar reads as one element and a missing +// key as none, so an optional list needs no placeholder in default.ini. +static inline int puf_ini_get_list(Ini* ini, const char* section, + const char* key, double* out, int max) { + DictItem* item = dict_find(puf_ini_section(ini, section, 0), key); + if (!item) { + return 0; + } + int n = item->len ? item->len : 1; + if (n > max) { + fprintf(stderr, "config error: [%s] %s has %d values, max %d\n", + section, key, n, max); + exit(1); + } + for (int i = 0; i < n; i++) { + out[i] = item->len ? item->values[i] : item->value; + } + return n; +} + static inline void puf_ini_put(Ini* ini, const char* full_key, const char* raw) { const char* split = strrchr(full_key, '.'); if (!split) { diff --git a/src/pufferenv.h b/src/pufferenv.h index e812100099..bf04f2779c 100644 --- a/src/pufferenv.h +++ b/src/pufferenv.h @@ -48,6 +48,13 @@ void puf_step(Env* env); void puf_render(Env* env); void puf_close(Env* env); void puf_log(Log* log, Dict* out); +// Bot ladder writes this between rungs so one PuffeRL can eval a whole ladder. +// Default no-op; envs with scripted opponents #define PUF_HAS_BOT_POLICY and +// assign env->bot_policy. +#ifndef PUF_HAS_BOT_POLICY +static inline void puf_set_bot_policy(Env* env, int bot_policy) { +} +#endif typedef uint16_t bf16; diff --git a/src/pufferl.cu b/src/pufferl.cu index 6dafbf5355..0bf667e328 100644 --- a/src/pufferl.cu +++ b/src/pufferl.cu @@ -1198,24 +1198,39 @@ static void* vec_thread_main(void* arg) { } } +// Fresh episodes + zero RNN carry. Workers sit in BUF_WAITING between rollouts. +static void env_restart(PuffeRL* p) { + VecEnv* vec = p->vec; + if (PUF_BACKEND == PUF_GPU) { + puf_reset(vec->envs); + } else { + #pragma omp parallel for schedule(static) num_threads(vec->num_workers) + for (int i = 0; i < vec->size; i++) { + puf_reset(&vec->envs[i]); + } + cpu_upload(p, 0, vec->total_agents, p->default_stream); + } + for (int b = 0; b < p->num_policies; b++) { + for (int i = 0; i < vec->buffers; i++) { + Prec* st = &p->policies[b].buffer_states[i]; + cudaMemset(st->data, 0, numel(st->shape) * sizeof(precision_t)); + } + } + cudaDeviceSynchronize(); +} + static void env_start(PuffeRL* p) { + VecEnv* vec = p->vec; if (PUF_BACKEND == PUF_GPU) { - puf_reset(p->vec->envs); - cudaDeviceSynchronize(); + env_restart(p); return; } - VecEnv* vec = p->vec; vec->worker_state = (int*)calloc(1, vec->buffers * sizeof(int)); vec->threads = (pthread_t*)calloc(1, vec->buffers * sizeof(pthread_t)); VecThreadArg* args = (VecThreadArg*)calloc(1, vec->buffers * sizeof(VecThreadArg)); vec->accum = (float*)calloc(1, vec->buffers * NUM_PROF * sizeof(float)); - #pragma omp parallel for schedule(static) num_threads(vec->num_workers) - for (int i = 0; i < vec->size; i++) { - puf_reset(&vec->envs[i]); - } - cpu_upload(p, 0, vec->total_agents, p->default_stream); - cudaDeviceSynchronize(); + env_restart(p); for (int i = 0; i < vec->buffers; i++) { args[i].pufferl = p; args[i].buf = i; @@ -2492,7 +2507,13 @@ typedef struct { #define EVAL_MATCH 2 #define SELFPLAY_MAX_HIST 8 +#define SELFPLAY_MAX_LADDER 16 #define SELFPLAY_PATH_MAX 4096 +// Bot-ladder parallelism, held fixed so rung scores stay comparable across +// sweep trials that vary vec.total_agents. selfplay.eval_bot_games / this is +// the games each env plays; scripted bots keep their kNN across episodes, so +// one game per env measures only cold bots. +#define SELFPLAY_LADDER_ENVS 8192 // One historical opponent ↔ policies[policy_idx] (env tag == policy_idx). typedef struct { @@ -2511,7 +2532,6 @@ typedef struct { } Selfplay; void selfplay_add_checkpoint(Selfplay* sp, const char* path) { - while (access(path, R_OK) != 0) usleep(50000); for (int i = 0; i < sp->pool_size; i++) { if (strcmp(sp->pool[i], path) == 0) return; } @@ -2883,7 +2903,11 @@ static PuffeRL* eval_make(Ini* ini, TrainContext* ctx, int mode, int render) { int match = mode == EVAL_MATCH; long eval_agents = puf_ini_get(ini, "base", "eval_agents"); if (render) { - puf_ini_put(ini, "vec.total_agents", "1"); + // One env on screen, whatever its agent count. env_setup allocates whole + // envs, so total_agents must be a multiple of env.num_agents. + char nb[32]; + snprintf(nb, sizeof(nb), "%d", (int)puf_ini_get(ini, "env", "num_agents")); + puf_ini_put(ini, "vec.total_agents", nb); puf_ini_put(ini, "vec.num_buffers", "1"); puf_ini_put(ini, "vec.num_threads", "1"); puf_ini_put(ini, "train.verb_eps", "1"); @@ -2953,16 +2977,29 @@ TrainResult run_train(Ini* ini, TrainContext* ctx) { if (!use_selfplay) { puf_ini_put(ini, "vec.num_policies", "1"); puf_ini_put(ini, "vec.hist_policy_percent", "0"); + } else { + int npol = puf_ini_get(ini, "vec", "num_policies"); + assert(npol >= 2 && npol <= SELFPLAY_MAX_HIST + 1 + && "selfplay requires vec.num_policies in 2..SELFPLAY_MAX_HIST+1"); + assert(puf_ini_get(ini, "vec", "hist_policy_percent") > 0 + && "selfplay requires vec.hist_policy_percent > 0"); + // Pool opps are this run's checkpoints; hist arch must match policy. + char hb[32], lb[32]; + snprintf(hb, sizeof(hb), "%d", + (int)puf_ini_get(ini, "policy", "hidden_size")); + snprintf(lb, sizeof(lb), "%d", + (int)puf_ini_get(ini, "policy", "num_layers")); + puf_ini_put(ini, "vec.hist_policy_hidden_size", hb); + puf_ini_put(ini, "vec.hist_policy_num_layers", lb); + double ladder[SELFPLAY_MAX_LADDER]; + assert((puf_ini_get(ini, "selfplay", "eval_bot_games") <= 0 + || puf_ini_get_list(ini, "selfplay", "eval_bots", ladder, + SELFPLAY_MAX_LADDER) > 0) + && "selfplay.eval_bot_games requires selfplay.eval_bots"); } char run_id[64]; - const char* configured_run_id = puf_ini_get_str(ini, "base", "run_id"); - if (!configured_run_id[0] || strcmp(configured_run_id, "None") == 0) { - snprintf(run_id, sizeof(run_id), "%ld", (long)(1000.0 * wall_clock())); - puf_ini_put(ini, "base.run_id", run_id); - } else { - snprintf(run_id, sizeof(run_id), "%s", configured_run_id); - } + snprintf(run_id, sizeof(run_id), "%s", puf_ini_get_str(ini, "base", "run_id")); char checkpoint_dir[2048]; char log_dir[2048]; @@ -2983,9 +3020,7 @@ TrainResult run_train(Ini* ini, TrainContext* ctx) { char initial_checkpoint[4096]; snprintf(initial_checkpoint, sizeof(initial_checkpoint), "%s/%016ld.bin", checkpoint_dir, pufferl->global_step); - if (ctx->artifact_owner) { - puf_save_weights(pufferl, initial_checkpoint); - } + puf_save_weights(pufferl, initial_checkpoint); selfplay.num_hist = pufferl->num_policies - 1; assert(selfplay.num_hist > 0 && selfplay.num_hist <= SELFPLAY_MAX_HIST && "selfplay requires num_policies in 2..SELFPLAY_MAX_HIST+1"); @@ -3065,8 +3100,10 @@ TrainResult run_train(Ini* ini, TrainContext* ctx) { && (epoch + 1) % checkpoint_interval == 0)) { snprintf(saved_checkpoint, sizeof(saved_checkpoint), "%s/%016ld.bin", checkpoint_dir, pufferl->global_step); - if (ctx->artifact_owner) { + if (ctx->artifact_owner || use_selfplay) { puf_save_weights(pufferl, saved_checkpoint); + } + if (ctx->artifact_owner) { snprintf(final_checkpoint, sizeof(final_checkpoint), "%s", saved_checkpoint); } @@ -3138,25 +3175,28 @@ TrainResult run_train(Ini* ini, TrainContext* ctx) { dict_set(&new_log, "pool/num_hist", selfplay.num_hist); dict_set(&new_log, "pool/num_policies", pufferl->num_policies); } - // Dense keys: replace last_log wholesale. - dict_clear(&last_log); - dict_copy(&last_log, &new_log); - dict_clear(&new_log); - + // n=0 logs omit env/*; keep last complete-episode snapshot. + int episodes = dict_get(&new_log, "env/n") > 0; if (ctx->artifact_owner) { - puf_dashboard_print(ini, pufferl, &last_log, (int)pufferl->epoch); + puf_dashboard_print(ini, pufferl, &new_log, (int)pufferl->epoch); + } + result.cost = dict_get(&new_log, "uptime"); + result.steps = dict_get(&new_log, "agent_steps"); + if (episodes || !last_log.size) { + dict_clear(&last_log); + dict_copy(&last_log, &new_log); + } else { + dict_set(&last_log, "uptime", result.cost); + dict_set(&last_log, "agent_steps", result.steps); } - // Wait until the objective appears; do not treat negative values as missing. - if (!dict_find(&last_log, target_key)) { - continue; + if (episodes && dict_find(&new_log, target_key)) { + puf_log_history_add(&log_history, &last_log); } - puf_log_history_add(&log_history, &last_log); + dict_clear(&new_log); } // TrainResult curve: bin-mean over log_history (same as artifact metrics). - result.cost = dict_get(&last_log, "uptime"); - result.steps = dict_get(&last_log, "agent_steps"); DictItem* target = dict_find(&last_log, target_key); result.score = target ? (float)target->value : 0; @@ -3188,19 +3228,82 @@ TrainResult run_train(Ini* ini, TrainContext* ctx) { result.step_points[points - 1] = result.steps; } + long bot_games = use_selfplay ? puf_ini_get(ini, "selfplay", "eval_bot_games") : 0; int max_opp = use_selfplay ? puf_ini_get(ini, "selfplay", "eval_pool_size") : 0; long pool_games = use_selfplay ? puf_ini_get(ini, "selfplay", "eval_games") : 0; - int pool_eval = use_selfplay && max_opp > 0 && pool_games > 0 && final_checkpoint[0]; + int bot_ladder = use_selfplay && bot_games > 0 && final_checkpoint[0]; + int pool_eval = !bot_ladder && use_selfplay && max_opp > 0 + && pool_games > 0 && final_checkpoint[0]; long eval_episodes = puf_ini_get(ini, "base", "eval_episodes"); - if (ctx->artifact_owner && !pool_eval && eval_episodes > 0) { + if (ctx->artifact_owner && !bot_ladder && !pool_eval && eval_episodes > 0) { EvalResult r = eval_loop(ini, pufferl, EVAL_SCORE, 1, 0, eval_episodes, &last_log, (int)pufferl->epoch); result.score = result.scores[result.points - 1] = r.score; } close_pufferl(pufferl); - if (pool_eval && ctx->artifact_owner) { + char log_path[4096]; + snprintf(log_path, sizeof(log_path), "%s/%s.ini", log_dir, run_id); + if (ctx->artifact_owner) { + FILE* fp = fopen(log_path, "w"); + assert(fp && "failed to open log for writing"); + fprintf(fp, "# PufferLib log v1\n"); + puf_ini_write(fp, ini); + fclose(fp); + } + + // Final Protein point (points=1): bot ladder > pool match > train curve. + if (bot_ladder && ctx->artifact_owner) { + puf_ini_put(ini, "base.load_model_path", final_checkpoint); + puf_ini_put(ini, "env.num_agents", "1"); + puf_ini_put(ini, "env.num_bots", "1"); + puf_ini_put(ini, "selfplay.enabled", "0"); + puf_ini_put(ini, "vec.num_policies", "1"); + puf_ini_put(ini, "vec.hist_policy_percent", "0"); + // [bot_eval] keys are [env] names (dr, not env.dr). Dotted keys break + // sweep argv, which flattens as section.key and splits on the last dot. + Dict* bot_eval = puf_ini_section(ini, "bot_eval", 1); + for (int i = 0; i < bot_eval->size; i++) { + DictItem* over = &bot_eval->items[i]; + char ek[PUF_DICT_MAX_KEY + 8]; + snprintf(ek, sizeof(ek), "env.%s", over->key); + puf_ini_put(ini, ek, over->str); + } + // Fixed parallelism; ignore swept train total_agents. Each env plays + // bot_games / SELFPLAY_LADDER_ENVS games, so bots face a warmed-up kNN + // rather than being re-measured cold once per env. + char nbuf[32]; + snprintf(nbuf, sizeof(nbuf), "%d", SELFPLAY_LADDER_ENVS); + puf_ini_put(ini, "vec.total_agents", nbuf); + double ladder[SELFPLAY_MAX_LADDER]; + int rungs = puf_ini_get_list(ini, "selfplay", "eval_bots", ladder, + SELFPLAY_MAX_LADDER); + // One PuffeRL for the whole ladder. close_pufferl frees nothing, so a + // trainer per rung is a leak. Swap bot_policy and restart instead. + PuffeRL* ep = eval_make(ini, ctx, EVAL_SCORE, 0); + float sum = 0; + for (int i = 0; i < rungs; i++) { + if (PUF_BACKEND != PUF_GPU) { + for (int e = 0; e < ep->vec->size; e++) { + puf_set_bot_policy(&ep->vec->envs[e], (int)ladder[i]); + } + } + env_restart(ep); + EvalResult r = eval_loop(ini, ep, EVAL_SCORE, 0, 0, bot_games, NULL, 0); + sum += r.perf; + printf("bot_eval policy=%d games=%d perf=%.4f\n", + (int)ladder[i], r.games, r.perf); + } + close_pufferl(ep); + result.score = result.scores[0] = sum / rungs; + result.points = 1; + result.costs[0] = result.cost; + result.step_points[0] = result.steps; + dict_set(&last_log, "selfplay/bot_ladder_perf", result.score); + printf("bot_eval mean_perf=%.4f\n", result.score); + } else if (pool_eval && ctx->artifact_owner) { puf_ini_put(ini, "base.load_model_path", final_checkpoint); + TrainContext eval_ctx = {.world_size = 1, .artifact_owner = 1}; int n_opp = 0; float sum = 0; for (int i = 0; i < selfplay.pool_size && n_opp < max_opp; i++) { @@ -3208,7 +3311,7 @@ TrainResult run_train(Ini* ini, TrainContext* ctx) { continue; } puf_ini_put(ini, "base.load_enemy_model_path", selfplay.pool[i]); - PuffeRL* ep = eval_make(ini, ctx, EVAL_MATCH, 0); + PuffeRL* ep = eval_make(ini, &eval_ctx, EVAL_MATCH, 0); EvalResult r = eval_loop(ini, ep, EVAL_MATCH, 0, 0, pool_games, NULL, 0); close_pufferl(ep); sum += r.score; @@ -3227,22 +3330,17 @@ TrainResult run_train(Ini* ini, TrainContext* ctx) { } if (ctx->artifact_owner) { + assert(log_history.size == 0 || dict_find(&last_log, target_key)); puf_log_history_add(&log_history, &last_log); - char log_path[4096]; - snprintf(log_path, sizeof(log_path), "%s/%s.ini", log_dir, run_id); - - FILE* fp = fopen(log_path, "w"); + FILE* fp = fopen(log_path, "a"); assert(fp && "failed to open log for writing"); - - fprintf(fp, "# PufferLib log v1\n"); - puf_ini_write(fp, ini); fprintf(fp, "\n[metrics]\n"); - // Dense keys from first history row; bin-mean same as TrainResult curve. + // Last snapshot keys (train + selfplay eval overlays). if (log_history.size > 0) { int metric_points = points; double* out = (double*)calloc(metric_points, sizeof(double)); - Dict* key_src = &log_history.items[0]; + Dict* key_src = &log_history.items[log_history.size - 1]; for (int k = 0; k < key_src->size; k++) { const char* key = key_src->items[k].key; if (strncmp(key, "loss/", 5) == 0) { @@ -3287,6 +3385,13 @@ TrainResult launch_train(Ini* ini) { assert(horizon % ADV_VEC_WIDTH == 0 && "train.horizon must be a multiple of ADV_VEC_WIDTH (4 float / 8 bf16)"); + const char* run_id = puf_ini_get_str(ini, "base", "run_id"); + if (!run_id[0] || strcmp(run_id, "None") == 0) { + char buf[64]; + snprintf(buf, sizeof(buf), "%ld", (long)(1000.0 * wall_clock())); + puf_ini_put(ini, "base.run_id", buf); + } + int nccl_pipe[2] = {-1, -1}; if (world_size > 1) { assert(pipe(nccl_pipe) == 0 && "pipe failed");