Views
No views yet
ac1. Checkpoint saved
after training step 37 (0-indexed). Strict upstream eval parity:
1100s hard kill, verbatim prompts/entrypoints, group 64x8, T=1.0, kl 0.1.1{
2 "step": 37,
3 "progress/batch": 37,
4 "optim/lr": 4e-05,
5 "progress/done_frac": 0.76,
6 "puct/buffer_size": 600,
7 "puct/sampled_size": 8,
8 "puct/T": 18944,
9 "puct/scale_last": 0.49366520167535133,
10 "puct/buffer_value/mean": -1.5188180714349582,
11 "puct/buffer_value/std": 0.06821509825161788,
12 "puct/buffer_value/min": -2.000000000000008,
13 "puct/buffer_value/max": -1.5063347983246491,
14 "puct/buffer_timestep/mean": 17.746666666666666,
15 "puct/buffer_timestep/std": 10.8272413640574,
16 "puct/buffer_timestep/min": -1.0,
17 "puct/buffer_timestep/max": 36.0,
18 "puct/buffer_construction_len/mean": 1072.5516666666667,
19 "puct/buffer_construction_len/std": 548.1077515694843,
20 "puct/buffer_construction_len/min": 1000.0,
21 "puct/buffer_construction_len/max": 7397.0,
22 "puct/sampled_value/mean": -1.5063356138518325,
23 "puct/sampled_value/std": 3.082403020757577e-07,
24 "puct/sampled_value/min": -1.5063357303557188,
25 "puct/sampled_value/max": -1.5063347983246491,
26 "puct/sampled_timestep/mean": 36.0,
27 "puct/sampled_timestep/std": 0.0,
28 "puct/sampled_timestep/min": 36.0,
29 "puct/sampled_timestep/max": 36.0,
30 "puct/sampled_construction_len/mean": 1000.0,
31 "puct/sampled_construction_len/std": 0.0,
32 "puct/sampled_construction_len/min": 1000.0,
33 "puct/sampled_construction_len/max": 1000.0,
34 "time/sampling": 3940.4239585399628,
35 "env/all/ac_tokens_per_turn": 9809.763671875,
36 "env/all/ob_tokens_per_turn": 2992.375,
37 "env/all/turns_per_episode": 1.0,
38 "env/all/total_episodes": 512,
39 "env/all/total_turns": 512,
40 "env/all/total_ac_tokens": 5022599,
41 "env/all/total_ob_tokens": 1532096,
42 "env/all/time/sampling_mean": 638.0919856145047,
43 "env/all/time/sampling_max": 920.0986170768738,
44 "env/all/time/env_step_mean": 1389.3346345517784,
45 "env/all/time/env_step_max": 3016.7719662189484,
46 "env/all/reward/mean": 0.6265261343397244,
47 "env/all/reward/max": 0.6638637299802108,
48 "env/all/reward/min": 0.0,
49 "env/all/format": 1.0,
50 "env/all/format/min": 1.0,
51 "env/all/format/max": 1.0,
52 "env/all/reward": 0.6265261343397244,
53 "env/all/correctness": 0.966796875,
54 "env/all/correctness/min": 0.0,
55 "env/all/correctness/max": 1.0,
56 "env/all/raw_score": 1.5545733291013593,
57 "env/all/raw_score/min": 1.506333225029739,
58 "env/all/raw_score/max": 2.6462567626835267,
59 "env/all/initial_raw_score": -1.506335613851833,
60 "env/all/initial_raw_score/min": -1.5063357303557188,
61 "env/all/initial_raw_score/max": -1.5063347983246491,
62 "env/all/msg": "Success; raw_score=1.5063347615723666",
63 "env/all/parsed_code": "```python\nimport time\nimport numpy as np\nfrom scipy.optimize import linprog\nimport random\nimport copy\n\ndef propose_candidate(seed=42, **kwargs):\n \"\"\"\n Proposes a sequence of non-negative numbers to minimize the evaluation score \n by dynamically selecting constraints, improving perturbation strategies,\n and enhancing the line search to avoid local minima. This approach \n emphasizes broader constraint coverage, balanced perturbation, and\n more efficient convergence during optimization.\n \"\"\"\n np.random.seed(seed)\n deadline = time.time() + 1000 - 10 # Run for up to 1000 seconds\n\n # Use the existing best sequence if available\n if 'height_sequence_1' in globals() and isinstance(height_sequence_1, (list, np.ndarray)):\n best_sequence = np.array(list(height_sequence_1))\n else:\n # Initialize with a structured pattern (sine wave) normalized to sqrt(2n)\n n = 1000\n x = np.linspace(0, 2 * np.pi, n)\n initial_sequence = np.sin(x) * 2 + 2\n initial_sequence = np.clip(initial_sequence, 0, 1000)\n initial_sequence = np.clip(initial_sequence, 0, 1000)\n initial_sequence /= initial_sequence.sum() # Normalize to [0,1]\n initial_sequence = initial_sequence * np.sqrt(2 * n) # Scale to theoretical normalization\n best_sequence = initial_sequence.copy()\n \n curr_sequence = best_sequence.copy()\n best_score = evaluate_sequence(curr_sequence.tolist())\n no_improvement_counter = 0\n iteration_count = 0\n n = len(curr_sequence)\n\n # Parameters for line search and perturbation\n k_values = [500, 800, 1000, 1500] # Larger constraints for better coverage\n perturbation_steps = 250 # Perturbation frequency\n perturbation_scale = 0.01 # Smaller perturbations to avoid divergence\n alpha_search_iterations = 50 # Reduce iterative overhead\n dynamic_k_factor = 0.75 # Factor to adjust k based on iteration\n spread_perturbation = 0.05 # Scale for spreading values\n\n while time.time() < deadline:\n current_time = time.time()\n try:\n # Compute current convolution and max_b\n conv = np.convolve(curr_sequence, curr_sequence)\n max_b = np.max(conv)\n sum_a = np.sum(curr_sequence)\n if sum_a < 0.001:\n raise ValueError(\"Sum too small\")\n\n # Select constraints based on convolution, with more spread coverage\n k = int(k_values[iteration_count % len(k_values)] * dynamic_k_factor)\n sorted_conv = sorted(enumerate(conv), key=lambda x: -x[1])\n top_indices = [i for i, _ in sorted_conv[:k]]\n bottom_k = int(k * 0.5)\n bottom_indices = [i for i, _ in sorted_conv[-bottom_k:]]\n spaced_indices = list(range(2, len(conv), 100))\n all_indices = np.unique(np.concatenate([top_indices, bottom_indices, spaced_indices])).tolist()\n A = []\n for j in all_indices:\n row = np.zeros(n)\n for m in range(n):\n i = j - m\n if 0 <= i < n:\n row[m] = curr_sequence[i]\n A.append(row)\n A = np.array(A)\n b = np.full(len(all_indices), max_b * 0.95) # Dynamic epsilon adjustment\n\n # Define LP problem: maximize sum of g_0 subject to constraints\n c = [-1.0] * n\n bounds = [(0.0, 1000.0) for _ in range(n)]\n\n # Solve the LP using 'highs' for performance\n res = linprog(c=c, A_ub=A, b_ub=b, bounds=bounds, method='highs', options={\"feastol\": 1e-8})\n if res.success:\n g_0 = res.x\n sum_g = np.sum(g_0)\n if sum_g == 0:\n continue # Avoid division by zero\n\n # Normalize g_0 to match the theoretical normalization sum\n g_0_normalized = g_0 / sum_g * np.sqrt(2 * n)\n\n # Perform adaptive line search to minimize max_b\n low, high = 0.0, 1.0\n for _ in range(alpha_search_iterations):\n m1 = low + (high - low)/3\n m2 = high - (high - low)/3\n new_seq1 = (1 - m1) * curr_sequence + m1 * g_0_normalized\n new_seq2 = (1 - m2) * curr_sequence + m2 * g_0_normalized\n conv1 = np.convolve(new_seq1, new_seq1)\n conv2 = np.convolve(new_seq2, new_seq2)\n new_max_b1 = np.max(conv1)\n new_max_b2 = np.max(conv2)\n if new_max_b1 < new_max_b2:\n high = m2\n else:\n low = m1\n alpha_opt = (low + high) / 2\n new_sequence = (1 - alpha_opt) * curr_sequence + alpha_opt * g_0_normalized\n\n # Ensure non-negative and bounded values\n new_sequence = np.clip(new_sequence, 0.0, 1000.0)\n new_sequence = new_sequence / np.sum(new_sequence) * np.sqrt(2 * n)\n curr_sequence = new_sequence.copy()\n\n # Evaluate and update best sequence\n curr_score = evaluate_sequence(curr_sequence.tolist())\n if curr_score < best_score:\n best_score = curr_score\n best_sequence = curr_sequence.copy()\n no_improvement_counter = 0\n print(f\"New best score: {best_score}\")\n else:\n no_improvement_counter += 1\n else:\n print(\"LP solution not found, breaking\")\n break\n\n # Targeted perturbation: reduce peaks and increase valleys\n if iteration_count % perturbation_steps == 0:\n # Identify indices with high contribution to max_b\n conv = np.convolve(curr_sequence, curr_sequence)\n top_k = int(n / 10)\n top_indices = np.argsort(conv)[-top_k:]\n # Collect original indices contributing to these top_k convolution indices\n peak_indices = set()\n for k in top_indices:\n # For each convolution index k, find original indices contributing\n start = max(0, k - (n - 1))\n end = min(n - 1, k)\n for i in range(start, end + 1):\n peak_indices.add(i)\n peak_indices = list(peak_indices)\n if not peak_indices:\n continue\n # Now perturb these indices\n # Reduce peaks\n curr_sequence[peak_indices] -= np.random.uniform(0.005, 0.01, len(peak_indices))\n curr_sequence[peak_indices] = np.maximum(curr_sequence[peak_indices], 0.0)\n # Increase valleys: find indices not in peak_indices and add small values\n non_peak_indices = [i for i in range(n) if i not in peak_indices]\n if non_peak_indices:\n curr_sequence[non_peak_indices] += np.random.uniform(0.001, 0.005, len(non_peak_indices))\n curr_sequence[non_peak_indices] = np.minimum(curr_sequence[non_peak_indices], 1000.0)\n\n # Ensure normalization and non-negativity\n curr_sequence = np.clip(curr_sequence, 0.0, 1000.0)\n curr_sequence = curr_sequence / np.sum(curr_sequence) * np.sqrt(2 * n)\n\n # Early stopping if no improvement for many iterations\n iteration_count += 1\n if iteration_count % 100 == 0:\n print(f\"Iteration {iteration_count}, current best score: {best_score}\")\n if no_improvement_counter > 200: # Increased threshold\n print(\"No improvement for 200 iterations, stopping early\")\n break\n\n except Exception as e:\n print(f\"Error during optimization: {e}. Skipping update.\")\n continue\n\n # Final safety check to ensure non-negative and valid values\n final_sequence = np.maximum(0.0, best_sequence).tolist()\n return [float(x) for x in final_sequence]\n```",
64 "env/all/time/policy": 638.0919856145047,
65 "env/all/time/policy/min": 237.52105617523193,
66 "env/all/time/policy/max": 920.0986170768738,
67 "env/all/time/env_step": 1389.3346345517784,
68 "env/all/time/env_step/min": 2.03932785987854,
69 "env/all/time/env_step/max": 3016.7719662189484,
70 "env/all/time/reward_compute": 3.1013041734695435e-07,
71 "env/all/time/reward_compute/min": 2.2351741790771484e-07,
72 "env/all/time/reward_compute/max": 6.109476089477539e-07,
73 "env/all/by_group/frac_mixed": 1.0,
74 "env/all/by_group/frac_all_good": 0.0,
75 "env/all/by_group/frac_all_bad": 0.0,
76 "advantage/mean": 0.0032395338639616966,
77 "advantage/min": -1.0,
78 "advantage/max": 0.7407029867172241,
79 "time/assemble_training_data": 10.740275859832764,
80 "time/kl_vs_base": 156.6307556629181,
81 "kl_policy_base": 0.0006539640598930418,
82 "time/train": 1251.2690472602844,
83 "time/save_checkpoint": 20.949172496795654,
84 "time/total": 5382.8427748680115
85}[2026-07-09T06:24:18+00:00] job=1812628 node=node-12 ngpu=3 ntrain=1 replicas=2 flash_attn=no
[2026-07-09T08:30:32+00:00] job=1813127 node=node-22 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-09T13:07:08+00:00] job=1813128 node=node-6 ngpu=6 ntrain=2 replicas=4 flash_attn=yes
[2026-07-11T13:13:37+00:00] job=1824165 node=node-6 ngpu=3 ntrain=1 replicas=2 flash_attn=yes