Views
No views yet
ac2. Checkpoint saved
after training step 48 (0-indexed). Strict upstream eval parity:
1100s hard kill, verbatim prompts/entrypoints, group 64x8, T=1.0, kl 0.1.1{
2 "step": 48,
3 "progress/batch": 48,
4 "optim/lr": 4e-05,
5 "progress/done_frac": 0.98,
6 "puct/buffer_size": 773,
7 "puct/sampled_size": 8,
8 "puct/T": 24576,
9 "puct/scale_last": 0.28149319883602064,
10 "puct/buffer_value/mean": 0.9392180411608678,
11 "puct/buffer_value/std": 0.03358943746298418,
12 "puct/buffer_value/min": 0.6666666666666636,
13 "puct/buffer_value/max": 0.9481598655026873,
14 "puct/buffer_timestep/mean": 23.326002587322122,
15 "puct/buffer_timestep/std": 13.97205894774713,
16 "puct/buffer_timestep/min": -1.0,
17 "puct/buffer_timestep/max": 47.0,
18 "puct/buffer_construction_len/mean": 3915.6002587322123,
19 "puct/buffer_construction_len/std": 1460.2627975396513,
20 "puct/buffer_construction_len/min": 1024.0,
21 "puct/buffer_construction_len/max": 32768.0,
22 "puct/sampled_value/mean": 0.948159048438824,
23 "puct/sampled_value/std": 5.551642955285764e-07,
24 "puct/sampled_value/min": 0.9481584780094944,
25 "puct/sampled_value/max": 0.9481598655026873,
26 "puct/sampled_timestep/mean": 47.0,
27 "puct/sampled_timestep/std": 0.0,
28 "puct/sampled_timestep/min": 47.0,
29 "puct/sampled_timestep/max": 47.0,
30 "puct/sampled_construction_len/mean": 4096.0,
31 "puct/sampled_construction_len/std": 0.0,
32 "puct/sampled_construction_len/min": 4096.0,
33 "puct/sampled_construction_len/max": 4096.0,
34 "time/sampling": 6207.640930652618,
35 "env/all/ac_tokens_per_turn": 8444.4296875,
36 "env/all/ob_tokens_per_turn": 4492.5,
37 "env/all/turns_per_episode": 1.0,
38 "env/all/total_episodes": 512,
39 "env/all/total_turns": 512,
40 "env/all/total_ac_tokens": 4323548,
41 "env/all/total_ob_tokens": 2300160,
42 "env/all/time/sampling_mean": 339.5917226425372,
43 "env/all/time/sampling_max": 441.2252719402313,
44 "env/all/time/env_step_mean": 2925.2046835492365,
45 "env/all/time/env_step_max": 5753.392118692398,
46 "env/all/reward/mean": 0.8567431799424731,
47 "env/all/reward/max": 0.9481624473662335,
48 "env/all/reward/min": 0.0,
49 "env/all/format": 1.0,
50 "env/all/format/min": 1.0,
51 "env/all/format/max": 1.0,
52 "env/all/reward": 0.8567431799424731,
53 "env/all/correctness": 0.9296875,
54 "env/all/correctness/min": 0.0,
55 "env/all/correctness/max": 1.0,
56 "env/all/raw_score": 0.9215388826271979,
57 "env/all/raw_score/min": 0.3258811517265268,
58 "env/all/raw_score/max": 0.9481624473662335,
59 "env/all/initial_raw_score": 0.9481590484388238,
60 "env/all/initial_raw_score/min": 0.9481584780094944,
61 "env/all/initial_raw_score/max": 0.9481598655026873,
62 "env/all/msg": "Success; raw_score=0.9481602241616786",
63 "env/all/parsed_code": "```python\nimport numpy as np\nimport time\nfrom typing import List, Tuple\nimport random\n\ndef _simpson_l2sq(conv: np.ndarray) -> Tuple[float, np.ndarray]:\n m = conv.size\n if m == 0:\n return 0.0, np.zeros_like(conv)\n dx = 1.0 / (m + 1)\n y = np.zeros(m + 2, dtype=conv.dtype)\n y[0] = 0.0\n y[1:-1] = conv\n y[-1] = 0.0\n lhs = y[:-1]\n rhs = y[1:]\n l2_sq = (dx / 3.0) * np.sum(lhs * lhs + lhs * rhs + rhs * rhs)\n grad_y = (dx / 3.0) * (4.0 * y + np.roll(y, 1) + np.roll(y, -1))\n grad_conv = grad_y[1:-1]\n return float(l2_sq), grad_conv\n\ndef _l1(conv: np.ndarray) -> Tuple[float, np.ndarray]:\n m = conv.size\n dx = 1.0 / (m + 1) if m > 0 else 1.0\n val = dx * float(np.sum(conv)) if m > 0 else 0.0\n grad = np.full_like(conv, dx)\n return val, grad\n\ndef _linf(conv: np.ndarray) -> Tuple[float, np.ndarray]:\n if conv.size == 0:\n return 0.0, np.zeros_like(conv)\n m = float(np.max(conv))\n if m <= 0.0:\n return m, np.zeros_like(conv)\n mask = conv == m\n count = int(mask.sum())\n if count == 0:\n return m, np.zeros_like(conv)\n grad = mask.astype(conv.dtype) / count\n return m, grad\n\ndef _objective_and_grad_conv(conv: np.ndarray) -> Tuple[float, np.ndarray]:\n l2_sq, g_l2 = _simpson_l2sq(conv)\n l1, g_l1 = _l1(conv)\n linf, g_linf = _linf(conv)\n if l1 <= 0.0 or linf <= 0.0:\n return 0.0, np.zeros_like(conv)\n denom = l1 * linf\n c_value = l2_sq / denom\n num_grad = g_l2 * denom - l2_sq * (g_l1 * linf + l1 * g_linf)\n g_conv = num_grad / (denom ** 2)\n return float(c_value), g_conv\n\ndef _grad_h_from_conv_grad(h: np.ndarray, g_conv: np.ndarray) -> np.ndarray:\n h_rev = h[::-1]\n g_h = np.convolve(g_conv, h_rev, mode=\"valid\")\n return 2.0 * g_h\n\ndef _upsample_1d(h: np.ndarray) -> np.ndarray:\n n = h.shape[0]\n new_h = np.zeros(2 * n, dtype=np.float32)\n new_h[::2] = h\n return new_h\n\ndef construct_function():\n \"\"\"\n This function constructs a sequence of non-negative heights to maximize the evaluation function.\n It uses a combination of enhanced exploration with diverse initial seeds, gradient ascent with adaptive learning, and multi-scale refinement.\n \"\"\"\n np.random.seed(42)\n height_sequence_1 = globals().get(\"height_sequence_1\", None)\n target_length = 4096\n initial_n = 8\n\n # Generate diverse initial sequences, including a single-peak candidate\n if height_sequence_1 is not None:\n initial_h = np.array(height_sequence_1, dtype=np.float32)\n initial_n = min(len(initial_h), target_length)\n x_old = np.linspace(-0.5, 0.5, len(initial_h))\n x_new = np.linspace(-0.5, 0.5, initial_n)\n h = np.interp(x_new, x_old, initial_h)\n else:\n # Generate diverse initial candidates with high-impact single-peak candidate\n candidates = []\n \n # Single-peak candidate with optimized sum\n seq_single_peak = np.zeros(target_length)\n seq_single_peak[target_length // 2] = 0.2 # Increased peak for better exploration\n seq_single_peak = np.clip(seq_single_peak, 0.0, 1000.0)\n candidates.append(seq_single_peak)\n \n # Multi-peak random distribution with dynamic intensity\n seq_multi = np.random.rand(target_length) * 0.5\n seq_multi = np.clip(seq_multi / np.sum(seq_multi) * 0.4, 0.0, 1000.0)\n candidates.append(seq_multi)\n \n # Sinusoidal pattern with adaptive frequencies\n x = np.linspace(-0.5, 0.5, target_length)\n seq_sw = 0.05 * np.sin(2 * np.pi * x * 4 + np.random.rand(10))\n seq_sw = np.clip(seq_sw / np.sum(seq_sw) * 0.4, 0.0, 1000.0)\n candidates.append(seq_sw)\n \n # Multi-scale Gaussian peaks\n seq_g = 0.05 * np.sum([np.exp(-((x - pos) / 0.15)**2) for pos in np.linspace(-0.2, 0.4, 20)])\n seq_g = np.clip(seq_g / np.sum(seq_g) * 0.5, 0.0, 1000.0)\n candidates.append(seq_g)\n \n # Sharp central peaks with exponential decay\n seq_peak = np.zeros(target_length)\n seq_peak[target_length // 2] = 0.2\n seq_peak = np.clip(seq_peak, 0.0, 1000.0)\n candidates.append(seq_peak)\n \n # Uniform distribution with noise\n seq_uniform = np.full(target_length, 0.15 / target_length)\n seq_uniform += np.random.rand(target_length) * 0.1\n seq_uniform = np.clip(seq_uniform, 0.0, 1000.0)\n candidates.append(seq_uniform)\n \n # Periodic sum of sine waves\n seq_sine = 0.05 * np.sin(2 * np.pi * x * 2) + 0.05 * np.sin(2 * np.pi * x * 5) + 0.05 * np.sin(2 * np.pi * x * 8)\n seq_sine = np.clip(seq_sine / np.sum(seq_sine) * 0.4, 0.0, 1000.0)\n candidates.append(seq_sine)\n \n # Comb of spaced peaks\n seq_comb = np.zeros(target_length)\n for i in range(0, target_length, 8):\n seq_comb[i] = 0.15 / (target_length // 8)\n seq_comb = np.clip(seq_comb, 0.0, 1000.0)\n candidates.append(seq_comb)\n \n # Added additional initial sequence: structured Gaussian-like peaks\n seq_gauss = 0.05 * np.sum([np.exp(-((x - pos) / 0.15)**2) for pos in np.linspace(-0.2, 0.4, 40)])\n seq_gauss = np.clip(seq_gauss / np.sum(seq_gauss) * 0.5, 0.0, 1000.0)\n candidates.append(seq_gauss)\n \n # Added random initial guess with higher intensity\n seq_rand = np.random.rand(target_length) * 0.5\n seq_rand = np.clip(seq_rand / np.sum(seq_rand) * 0.5, 0.0, 1000.0)\n candidates.append(seq_rand)\n \n # Added structured spiky initial guess\n seq_peaks = np.zeros(target_length)\n for i in range(0, target_length, 4):\n seq_peaks[i] = 0.15 / (target_length // 4)\n seq_peaks = np.clip(seq_peaks / np.sum(seq_peaks) * 0.5, 0.0, 1000.0)\n candidates.append(seq_peaks)\n \n best_c = -1.0\n best_h = candidates[0]\n for h_candidate in candidates:\n try:\n c = evaluate_sequence(h_candidate.tolist())\n if c > best_c:\n best_c = c\n best_h = h_candidate\n except:\n pass\n h = best_h\n\n # Parameters for enhanced exploration and refinement with adjusted optimization\n learning_rate_initial = 20.0 # Moderated initial learning rate\n noise_scale_initial = 100.0 # Reduced initial noise scale for better exploration\n noise_decay = 0.995 # Slightly slower decay to stabilize\n learning_rate_decay = 0.9999 # Slight decay for convergence\n max_steps = 600000\n upscale_steps = 250000\n refine_steps = 350000\n momentum = 0.95 # Reduced momentum for better exploration\n\n start_time = time.time()\n\n # Multi-scale optimization with enhanced exploration\n current_length = initial_n\n scale_factor = 2\n\n for _ in range(4):\n if current_length < target_length:\n h = _upsample_1d(h)\n current_length *= 2\n h_sum = np.sum(h)\n if h_sum < 0.01:\n # Only scale if sum is too low, but avoid fixed sum\n scale_factor = 1.0 / h_sum\n h = np.clip(h * scale_factor, 0.0, 1000.0)\n\n # Initial optimization with dynamic noise and learning rate\n h_opt = np.copy(h)\n h_best = np.copy(h)\n best_c = -1.0\n prev_update = np.zeros_like(h_opt)\n\n # Increase the number of steps for better exploration\n for step in range(upscale_steps):\n clipped_h = np.clip(h_opt, 0.0, None)\n conv = np.convolve(clipped_h, clipped_h, mode='full')\n obj_val, grad_conv = _objective_and_grad_conv(conv)\n grad_h = _grad_h_from_conv_grad(clipped_h, grad_conv)\n\n noise_scale = noise_scale_initial * (noise_decay ** step)\n learning_rate_current = learning_rate_initial * (learning_rate_decay ** step)\n\n # Lower noise probability in early steps\n if np.random.rand() < 0.3:\n noise = noise_scale * np.random.normal(size=h_opt.shape)\n update = learning_rate_current * grad_h + momentum * prev_update + noise\n else:\n update = learning_rate_current * grad_h + momentum * prev_update\n\n h_opt = np.clip(h_opt + update, 0.0, 1000.0)\n prev_update = update\n\n if step % 300 == 0:\n try:\n current_c = evaluate_sequence(h_opt.tolist())\n if current_c > best_c:\n best_c = current_c\n h_best = h_opt.copy()\n except:\n pass\n\n remaining_time = 1000 - (time.time() - start_time)\n if remaining_time < 3:\n print(f\"Time remaining: {remaining_time} seconds. Performing final refinement.\")\n break\n h = h_best\n else:\n break\n\n # Final refined optimization with adaptive steps\n h_refined = np.copy(h)\n prev_update_refine = np.zeros_like(h_refined)\n\n # Increase refinement steps and learning rate\n for step in range(refine_steps):\n clipped_h = np.clip(h_refined, 0.0, None)\n conv = np.convolve(clipped_h, clipped_h, mode='full')\n obj_val, grad_conv = _objective_and_grad_conv(conv)\n grad_h = _grad_h_from_conv_grad(clipped_h, grad_conv)\n\n learning_rate_current = learning_rate_initial * (learning_rate_decay ** step)\n\n # Adjust learning rate to higher values during early refinement\n if step < refine_steps // 2:\n learning_rate_current *= 1.5\n elif step < refine_steps * 3 // 4:\n learning_rate_current *= 1.2\n update = learning_rate_current * grad_h + momentum * prev_update_refine\n h_refined = np.clip(h_refined + update, 0.0, 1000.0)\n prev_update_refine = update\n\n remaining_time = 1000 - (time.time() - start_time)\n if remaining_time < 3:\n print(f\"Time remaining: {remaining_time} seconds. Final step.\")\n break\n\n h_final = np.clip(h_refined, 0.0, 1000.0)\n heights = h_final.tolist()\n r_value = evaluate_sequence(heights)\n print(f\"Final C2 lower bound: {r_value}\")\n return heights\n```",
64 "env/all/time/policy": 339.5917226425372,
65 "env/all/time/policy/min": 161.65260553359985,
66 "env/all/time/policy/max": 441.2252719402313,
67 "env/all/time/env_step": 2925.2046835492365,
68 "env/all/time/env_step/min": 0.008703231811523438,
69 "env/all/time/env_step/max": 5753.392118692398,
70 "env/all/time/reward_compute": 2.980232238769531e-07,
71 "env/all/time/reward_compute/min": 1.9744038581848145e-07,
72 "env/all/time/reward_compute/max": 7.562339305877686e-07,
73 "env/all/by_group/frac_mixed": 1.0,
74 "env/all/by_group/frac_all_good": 0.0,
75 "env/all/by_group/frac_all_bad": 0.0,
76 "advantage/mean": 0.019606390967965126,
77 "advantage/min": -1.0,
78 "advantage/max": 5.621390342712402,
79 "time/assemble_training_data": 6.128742933273315,
80 "time/kl_vs_base": 94.7595055103302,
81 "kl_policy_base": 0.000796365609858185,
82 "time/train": 625.5753602981567,
83 "time/save_checkpoint": 19.218113899230957,
84 "time/total": 6958.120721817017
85}[2026-07-09T06:42:12+00:00] job=1812634 node=node-30 ngpu=3 ntrain=1 replicas=2 flash_attn=no
[2026-07-09T07:26:33+00:00] job=1812736 node=node-12 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-09T09:27:32+00:00] job=1813133 node=node-14 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-09T16:26:04+00:00] job=1813134 node=node-4 ngpu=6 ntrain=2 replicas=4 flash_attn=yes
[2026-07-11T01:56:04+00:00] job=1821290 node=node-4 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-11T21:10:54+00:00] job=1825946 node=node-20 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-12T11:05:51+00:00] job=1827208 node=node-11 ngpu=6 ntrain=2 replicas=4 flash_attn=yes