Views
No views yet
ac2. Checkpoint saved
after training step 11 (0-indexed). Strict upstream eval parity:
1100s hard kill, verbatim prompts/entrypoints, group 64x8, T=1.0, kl 0.1.1{
2 "step": 11,
3 "progress/batch": 11,
4 "optim/lr": 4e-05,
5 "progress/done_frac": 0.24,
6 "puct/buffer_size": 184,
7 "puct/sampled_size": 8,
8 "puct/T": 5632,
9 "puct/scale_last": 0.43720195436973397,
10 "puct/buffer_value/mean": 0.9149434334514396,
11 "puct/buffer_value/std": 0.0642731216436444,
12 "puct/buffer_value/min": 0.5041471954736918,
13 "puct/buffer_value/max": 0.9413491498434258,
14 "puct/buffer_timestep/mean": 4.739130434782608,
15 "puct/buffer_timestep/std": 3.326015912853092,
16 "puct/buffer_timestep/min": -1.0,
17 "puct/buffer_timestep/max": 10.0,
18 "puct/buffer_construction_len/mean": 2428.0597826086955,
19 "puct/buffer_construction_len/std": 1495.9568268996866,
20 "puct/buffer_construction_len/min": 1024.0,
21 "puct/buffer_construction_len/max": 16384.0,
22 "puct/sampled_value/mean": 0.939635735720031,
23 "puct/sampled_value/std": 0.00036859767966586105,
24 "puct/sampled_value/min": 0.9390539554415861,
25 "puct/sampled_value/max": 0.9402257471526131,
26 "puct/sampled_timestep/mean": 10.0,
27 "puct/sampled_timestep/std": 0.0,
28 "puct/sampled_timestep/min": 10.0,
29 "puct/sampled_timestep/max": 10.0,
30 "puct/sampled_construction_len/mean": 2560.0,
31 "puct/sampled_construction_len/std": 886.8100134752651,
32 "puct/sampled_construction_len/min": 2048.0,
33 "puct/sampled_construction_len/max": 4096.0,
34 "time/sampling": 5444.957594156265,
35 "env/all/ac_tokens_per_turn": 8546.083984375,
36 "env/all/ob_tokens_per_turn": 3804.25,
37 "env/all/turns_per_episode": 1.0,
38 "env/all/total_episodes": 512,
39 "env/all/total_turns": 512,
40 "env/all/total_ac_tokens": 4375595,
41 "env/all/total_ob_tokens": 1947776,
42 "env/all/time/sampling_mean": 309.8909119348973,
43 "env/all/time/sampling_max": 403.2546110153198,
44 "env/all/time/env_step_mean": 2478.502006866038,
45 "env/all/time/env_step_max": 5030.087179660797,
46 "env/all/reward/mean": 0.3814100951490428,
47 "env/all/reward/max": 0.940276030589368,
48 "env/all/reward/min": 0.0,
49 "env/all/format": 1.0,
50 "env/all/format/min": 1.0,
51 "env/all/format/max": 1.0,
52 "env/all/reward": 0.3814100951490428,
53 "env/all/correctness": 0.423828125,
54 "env/all/correctness/min": 0.0,
55 "env/all/correctness/max": 1.0,
56 "env/all/raw_score": 0.8999169065267738,
57 "env/all/raw_score/min": 0.3610111094273138,
58 "env/all/raw_score/max": 0.940276030589368,
59 "env/all/initial_raw_score": 0.939635735720031,
60 "env/all/initial_raw_score/min": 0.9390539554415861,
61 "env/all/initial_raw_score/max": 0.9402257471526131,
62 "env/all/msg": "Evaluation timed out after 1100 seconds.",
63 "env/all/parsed_code": "```python\nimport numpy as np\nfrom typing import Tuple\nimport time\nimport random\n\ndef _simpson_l2sq(conv: np.ndarray) -> Tuple[float, np.ndarray]:\n \"\"\"Compute ||f*f||_2^2 via Simpson's rule with endpoint zeros and gradient.\"\"\"\n m = conv.size\n if m == 0:\n return 0.0, np.zeros_like(conv)\n\n dx = 1.0 / (m + 1)\n y = np.zeros(m + 2, dtype=conv.dtype)\n y[0] = 0.0\n y[1:-1] = conv\n y[-1] = 0.0\n\n lhs = y[:-1]\n rhs = y[1:]\n l2_sq = (dx / 3.0) * np.sum(lhs * lhs + lhs * rhs + rhs * rhs)\n\n grad_y = (dx / 3.0) * (4.0 * y + np.roll(y, 1) + np.roll(y, -1))\n grad_conv = grad_y[1:-1]\n return float(l2_sq), grad_conv\n\ndef _l1(conv: np.ndarray) -> Tuple[float, np.ndarray]:\n \"\"\"Compute ||f*f||_1 and its gradient.\"\"\"\n m = conv.size\n dx = 1.0 / (m + 1) if m > 0 else 1.0\n val = dx * float(np.sum(conv)) if m > 0 else 0.0\n grad = np.full_like(conv, dx)\n return val, grad\n\ndef _linf(conv: np.ndarray) -> Tuple[float, np.ndarray]:\n \"\"\"Compute ||f*f||_inf and its subgradient.\"\"\"\n if conv.size == 0:\n return 0.0, np.zeros_like(conv)\n m = float(np.max(conv))\n mask = conv == m\n count = int(mask.sum())\n if count == 0 or m <= 0.0:\n return m, np.zeros_like(conv)\n grad = mask.astype(conv.dtype)\n return m, grad\n\ndef _objective_and_grad_conv(conv: np.ndarray) -> Tuple[float, np.ndarray]:\n \"\"\"Compute C = l2_sq / (l1 * linf) and its gradient.\"\"\"\n l2_sq, g_l2 = _simpson_l2sq(conv)\n l1, g_l1 = _l1(conv)\n linf, g_linf = _linf(conv)\n\n if l1 <= 0.0 or linf <= 0.0:\n return 0.0, np.zeros_like(conv)\n\n denom = l1 * linf\n c_value = l2_sq / denom\n\n num_grad = g_l2 * denom - l2_sq * (g_l1 * linf + l1 * g_linf)\n g_conv = num_grad / (denom * denom)\n\n return float(c_value), g_conv\n\ndef _grad_h_from_conv_grad(h: np.ndarray, g_conv: np.ndarray) -> np.ndarray:\n \"\"\"Compute gradient of C w.r.t h from gradient of C w.r.t conv.\"\"\"\n h_rev = h[::-1]\n g_h = np.convolve(g_conv, h_rev, mode=\"valid\")\n return 2.0 * g_h\n\nclass _Adam:\n \"\"\"Lightweight Adam optimizer for numpy arrays (per-candidate).\"\"\"\n def __init__(self, shape, lr=3e-2, beta1=0.9, beta2=0.999, eps=1e-8, dtype=np.float32):\n self.m = np.zeros(shape, dtype=dtype)\n self.v = np.zeros(shape, dtype=dtype)\n self.t = 0\n self.lr = lr\n self.b1 = beta1\n self.b2 = beta2\n self.eps = eps\n\n def step(self, params, grad):\n self.t += 1\n self.m = self.b1 * self.m + (1 - self.b1) * grad\n self.v = self.b2 * self.v + (1 - self.b2) * (grad * grad)\n m_hat = self.m / (1 - self.b1 ** self.t)\n v_hat = self.v / (1 - self.b2 ** self.t)\n return params + self.lr * m_hat / (np.sqrt(v_hat) + self.eps)\n\ndef _batch_objective(h_batch: np.ndarray) -> Tuple[np.ndarray, list[np.ndarray]]:\n \"\"\"Vectorized evaluation of objective and gradient.\"\"\"\n bsz = h_batch.shape[0]\n c_vals = np.zeros(bsz, dtype=np.float32)\n conv_grads = [None] * bsz\n for b in range(bsz):\n h = np.clip(h_batch[b], 0.0, None)\n conv = np.convolve(h, h, mode=\"full\")\n c_val, g_conv = _objective_and_grad_conv(conv)\n c_vals[b] = c_val\n conv_grads[b] = g_conv\n return c_vals, conv_grads\n\ndef _phase_update(h_batch, opt_list, lr, add_noise=False, t=0, eta=1e-2, gamma=0.3):\n \"\"\"Update all candidates in the batch.\"\"\"\n bsz = h_batch.shape[0]\n c_vals, conv_grads = _batch_objective(h_batch)\n grads = np.zeros_like(h_batch, dtype=h_batch.dtype)\n for b in range(bsz):\n clipped = np.clip(h_batch[b], 0.0, None)\n grads[b] = _grad_h_from_conv_grad(clipped, conv_grads[b])\n\n if add_noise:\n sigma = eta / ((t + 1) ** gamma)\n grads = grads + sigma * np.random.normal(size=grads.shape).astype(grads.dtype)\n\n for b in range(bsz):\n opt = opt_list[b]\n opt.lr = lr\n h_new = opt.step(h_batch[b], grads[b].astype(h_batch.dtype))\n h_batch[b] = np.clip(h_new, 0.0, None)\n\n return h_batch, c_vals\n\ndef _elitist_respawn(h_batch, c_vals, keep_frac, init_sampler, opt_list):\n \"\"\"Keep top fraction and respawn the rest.\"\"\"\n bsz = h_batch.shape[0]\n keep_n = max(1, int(bsz * keep_frac))\n idx = np.argsort(c_vals)[-keep_n:]\n survivors = h_batch[idx].copy()\n\n fresh = init_sampler(bsz - keep_n)\n new_batch = np.concatenate([survivors, fresh], axis=0)\n\n new_opts = [opt_list[i] for i in idx]\n for _ in range(bsz - keep_n):\n new_opts.append(_Adam(shape=h_batch.shape[1:], lr=opt_list[0].lr, dtype=h_batch.dtype))\n\n return new_batch, new_opts\n\ndef _upsample_1d(h: np.ndarray) -> np.ndarray:\n \"\"\"Upsample by linear interpolation for structured preservation.\"\"\"\n return np.interp(np.linspace(0, 1, 2 * h.size), np.linspace(0, 1, h.size), h)\n\ndef _single_candidate_finetune(h0: np.ndarray, lr=3e-3, steps=150_000) -> Tuple[np.ndarray, float]:\n \"\"\"Refine a single candidate using Adam with projection.\"\"\"\n h = h0.astype(np.float32).copy()\n opt = _Adam(h.shape, lr=lr, dtype=h.dtype)\n best_c = 0.0\n for _ in range(steps):\n h_clip = np.clip(h, 0.0, None)\n conv = np.convolve(h_clip, h_clip, mode=\"full\")\n c_val, g_conv = _objective_and_grad_conv(conv)\n g_h = _grad_h_from_conv_grad(h_clip, g_conv)\n h = np.clip(opt.step(h, g_h.astype(h.dtype)), 0.0, None)\n best_c = max(best_c, c_val)\n return h, float(best_c)\n\ndef construct_function():\n \"\"\"\n Construct optimized step function sequence to improve C2 lower bound.\n This version enhances exploration of diverse patterns, including uniform distributions\n and periodic structures, while refining with more aggressive Adam optimization.\n \"\"\"\n # Starting with larger sequence length for more detailed patterns\n n_start = 4096 # Increased from 512 for finer resolution\n bsz = 256 # Reduced batch size for more diverse exploration\n total_iter = 150_000 # More iterations for thorough exploration\n explore_steps = 100_000 # Longer exploration phase\n drop_every = 500 # More frequent elitist respawns\n keep_frac = 0.25 # Slightly reduced diversity to focus on better candidates\n\n # Initialize from previous best if available\n if 'height_sequence_1' in globals():\n prev = np.array(height_sequence_1, dtype=np.float32)\n else:\n prev = np.ones(n_start, dtype=np.float32)\n\n # Resample to initial length\n prev = np.clip(prev, 0.0, 1000.0)\n if prev.shape[0] != n_start:\n x_old = np.linspace(-0.5, 0.5, prev.shape[0])\n x_new = np.linspace(-0.5, 0.5, n_start)\n prev = np.interp(x_new, x_old, prev).astype(np.float32)\n\n # Define more diverse initialization, emphasizing uniform and periodic patterns\n def init_sampler(m):\n out = np.random.uniform(0.0, 1000.0, size=(m, n_start)).astype(np.float32)\n out[0] = prev # Include previous best\n\n for i in range(1, m):\n # Randomly select from 7 patterns\n pattern_idx = np.random.randint(0, 7)\n if pattern_idx == 0:\n # Uniform distribution with some variation\n base = np.random.uniform(0.2, 0.8, size=n_start)\n noise = np.random.normal(0, 0.05, size=n_start)\n out[i] = np.clip(base + noise, 0.0, 1000.0).astype(np.float32)\n elif pattern_idx == 1:\n # Periodic sine wave\n freq = np.random.uniform(0.1, 0.3)\n phase = np.random.uniform(0, 2 * np.pi)\n t = np.linspace(-0.5, 0.5, n_start)\n out[i] = 0.5 * (1 + np.sin(2 * np.pi * freq * t + phase)).astype(np.float32)\n elif pattern_idx == 2:\n # Random sparse spikes\n out[i] = np.random.uniform(0.0, 0.1, size=n_start).astype(np.float32)\n spike_pos = np.random.choice(n_start, 10, replace=False)\n out[i][spike_pos] += np.random.uniform(0.4, 1.5, size=10).astype(np.float32)\n elif pattern_idx == 3:\n # Multi-peak pattern\n num_peaks = np.random.randint(5, 8)\n peaks = np.zeros(n_start)\n for _ in range(num_peaks):\n pos = np.random.randint(0, n_start)\n val = np.random.uniform(150, 300)\n peaks[pos] = val\n out[i] = peaks.astype(np.float32)\n elif pattern_idx == 4:\n # Gaussian-like distribution\n mean = np.random.uniform(0.2, 0.8)\n std = np.random.uniform(0.05, 0.2)\n out[i] = np.random.normal(loc=mean, scale=std, size=n_start).astype(np.float32)\n out[i] = np.clip(out[i], 0.0, 1000.0)\n elif pattern_idx == 5:\n # Dense peak clusters\n cluster_count = np.random.randint(2, 5)\n out[i] = np.zeros(n_start)\n for _ in range(cluster_count):\n start = np.random.randint(0, n_start - 10)\n for j in range(start, start + 10):\n out[i][j] += np.random.uniform(50, 150)\n out[i] = np.clip(out[i], 0.0, 1000.0)\n elif pattern_idx == 6:\n # Single peak pattern\n peak_pos = np.random.randint(0, n_start - 1)\n out[i] = np.zeros(n_start)\n out[i][peak_pos] = 1000.0\n\n return out\n\n h_batch = init_sampler(bsz)\n opt_list = [_Adam(shape=(n_start,), lr=0.01, dtype=np.float32) for _ in range(bsz)]\n best_h = h_batch.copy()\n best_c = np.full(bsz, -np.inf, dtype=np.float32)\n\n start_time = time.time()\n\n for t in range(total_iter):\n if t < explore_steps:\n # Explosive phase with moderate learning rate\n h_batch, c_vals = _phase_update(\n h_batch, opt_list, lr=0.01, add_noise=True, t=t, eta=1e-2, gamma=0.3\n )\n else:\n # Exploitation phase with lower learning rate and noise\n h_batch, c_vals = _phase_update(\n h_batch, opt_list, lr=1e-3, add_noise=True, t=t, eta=1e-2, gamma=0.3\n )\n\n # Update best candidates\n improved_idx = c_vals > best_c\n best_c = np.where(improved_idx, c_vals, best_c)\n best_h[improved_idx] = h_batch[improved_idx]\n\n # Periodic elitist respawn for diversity\n if (t + 1) % drop_every == 0:\n h_batch, opt_list = _elitist_respawn(\n h_batch, c_vals, keep_frac=keep_frac, init_sampler=init_sampler, opt_list=opt_list\n )\n\n # Periodic logging and early stopping\n if t % 500 == 0:\n elapsed = time.time() - start_time\n print(f\"Iteration {t} (elapsed: {elapsed:.1f}s) - Best score: {best_c[np.argmax(best_c)]:.6f}\")\n if elapsed > 950:\n print(\"Reached time limit, stopping early.\")\n break\n\n # Refinement process with multi-resolution and enhanced fine-tuning\n idx = np.argmax(best_c)\n h_star = np.clip(best_h[idx].astype(np.float32), 0.0, None)\n h_up1 = _upsample_1d(h_star)\n h_up1, _ = _single_candidate_finetune(h_up1, lr=3e-3, steps=200_000)\n\n h_up2 = _upsample_1d(h_up1)\n h_up2, _ = _single_candidate_finetune(h_up2, lr=3e-3, steps=200_000)\n\n h_final = np.clip(h_up2, 0.0, 1000.0)\n heights = h_final.tolist()\n r_value = evaluate_sequence(heights)\n print(f\"Final C2 lower bound: {r_value:.6f}\")\n return heights\n```",
64 "env/all/time/policy": 309.8909119348973,
65 "env/all/time/policy/min": 164.62883520126343,
66 "env/all/time/policy/max": 403.2546110153198,
67 "env/all/time/env_step": 2478.502006866038,
68 "env/all/time/env_step/min": 0.008361101150512695,
69 "env/all/time/env_step/max": 5030.087179660797,
70 "env/all/time/reward_compute": 3.46451997756958e-07,
71 "env/all/time/reward_compute/min": 1.862645149230957e-07,
72 "env/all/time/reward_compute/max": 9.424984455108643e-07,
73 "env/all/by_group/frac_mixed": 1.0,
74 "env/all/by_group/frac_all_good": 0.0,
75 "env/all/by_group/frac_all_bad": 0.0,
76 "advantage/mean": 0.02230602316558361,
77 "advantage/min": -1.0,
78 "advantage/max": 3.248243808746338,
79 "time/assemble_training_data": 6.838516712188721,
80 "time/kl_vs_base": 87.40019941329956,
81 "kl_policy_base": 0.0006346934824250638,
82 "time/train": 644.7200102806091,
83 "time/save_checkpoint": 12.259060859680176,
84 "time/total": 6199.003251314163
85}[2026-07-09T06:38:34+00:00] job=1812632 node=node-30 ngpu=3 ntrain=1 replicas=2 flash_attn=no
[2026-07-09T06:45:52+00:00] job=1812704 node=node-30 ngpu=3 ntrain=1 replicas=2 flash_attn=no
[2026-07-09T07:00:42+00:00] job=1812735 node=node-1 ngpu=3 ntrain=1 replicas=2 flash_attn=no
[2026-07-09T07:26:33+00:00] job=1812827 node=node-14 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-09T09:21:09+00:00] job=1813131 node=node-1 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-09T14:53:48+00:00] job=1813132 node=node-2 ngpu=6 ntrain=2 replicas=4 flash_attn=yes
[2026-07-10T03:31:51+00:00] job=1816627 node=node-14 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-10T03:54:03+00:00] job=1816628 node=node-29 ngpu=6 ntrain=2 replicas=4 flash_attn=yes