Views
No views yet
ac1. Checkpoint saved
after training step 40 (0-indexed). Strict upstream eval parity:
1100s hard kill, verbatim prompts/entrypoints, group 64x8, T=1.0, kl 0.1.1{
2 "step": 40,
3 "progress/batch": 40,
4 "optim/lr": 4e-05,
5 "progress/done_frac": 0.82,
6 "puct/buffer_size": 648,
7 "puct/sampled_size": 8,
8 "puct/T": 20480,
9 "puct/scale_last": 0.540111866327813,
10 "puct/buffer_value/mean": -1.520990723824579,
11 "puct/buffer_value/std": 0.07273895389881602,
12 "puct/buffer_value/min": -2.0464450913574384,
13 "puct/buffer_value/max": -1.5063332250296253,
14 "puct/buffer_timestep/mean": 19.246913580246915,
15 "puct/buffer_timestep/std": 11.693124353254769,
16 "puct/buffer_timestep/min": -1.0,
17 "puct/buffer_timestep/max": 39.0,
18 "puct/buffer_construction_len/mean": 1067.1774691358025,
19 "puct/buffer_construction_len/std": 527.7590765561706,
20 "puct/buffer_construction_len/min": 1000.0,
21 "puct/buffer_construction_len/max": 7397.0,
22 "puct/sampled_value/mean": -1.5063332250296337,
23 "puct/sampled_value/std": 4.449903517629546e-15,
24 "puct/sampled_value/min": -1.5063332250296375,
25 "puct/sampled_value/max": -1.5063332250296253,
26 "puct/sampled_timestep/mean": 39.0,
27 "puct/sampled_timestep/std": 0.0,
28 "puct/sampled_timestep/min": 39.0,
29 "puct/sampled_timestep/max": 39.0,
30 "puct/sampled_construction_len/mean": 1000.0,
31 "puct/sampled_construction_len/std": 0.0,
32 "puct/sampled_construction_len/min": 1000.0,
33 "puct/sampled_construction_len/max": 1000.0,
34 "time/sampling": 3630.445642709732,
35 "env/all/ac_tokens_per_turn": 8859.541015625,
36 "env/all/ob_tokens_per_turn": 2888.125,
37 "env/all/turns_per_episode": 1.0,
38 "env/all/total_episodes": 512,
39 "env/all/total_turns": 512,
40 "env/all/total_ac_tokens": 4536085,
41 "env/all/total_ob_tokens": 1478720,
42 "env/all/time/sampling_mean": 320.05671777110547,
43 "env/all/time/sampling_max": 442.37985610961914,
44 "env/all/time/env_step_mean": 1503.820031573996,
45 "env/all/time/env_step_max": 3197.7444117069244,
46 "env/all/reward/mean": 0.5839205906312984,
47 "env/all/reward/max": 0.663863729980261,
48 "env/all/reward/min": 0.0,
49 "env/all/format": 1.0,
50 "env/all/format/min": 1.0,
51 "env/all/format/max": 1.0,
52 "env/all/reward": 0.5839205906312984,
53 "env/all/correctness": 0.890625,
54 "env/all/correctness/min": 0.0,
55 "env/all/correctness/max": 1.0,
56 "env/all/raw_score": 1.5314367359740813,
57 "env/all/raw_score/min": 1.506333225029625,
58 "env/all/raw_score/max": 2.196402862462494,
59 "env/all/initial_raw_score": -1.5063332250296337,
60 "env/all/initial_raw_score/min": -1.5063332250296375,
61 "env/all/initial_raw_score/max": -1.5063332250296253,
62 "env/all/msg": "Success; raw_score=1.506333225029625",
63 "env/all/parsed_code": "```python\nimport time\nimport numpy as np\nfrom scipy.optimize import linprog\nimport random\nimport copy\n\ndef propose_candidate(seed=42, **kwargs):\n \"\"\"\n Proposes a sequence of non-negative numbers to minimize the evaluation score \n by adapting the search strategy to find more optimal sequences, focusing on\n diverse perturbation methods and more refined LP constraint selection.\n \"\"\"\n np.random.seed(seed)\n deadline = time.time() + 1000 - 10 # Run for up to 1000 seconds\n\n # Use the existing best sequence if available\n if 'height_sequence_1' in globals() and isinstance(height_sequence_1, (list, np.ndarray)):\n best_sequence = np.array(list(height_sequence_1))\n else:\n # Initialize with a more balanced sequence\n n = 1000\n initial_sequence = np.full(n, 0.01) # Small flat values\n best_sequence = initial_sequence.copy()\n \n curr_sequence = best_sequence.copy()\n best_score = evaluate_sequence(curr_sequence.tolist())\n no_improvement_counter = 0\n iteration_count = 0\n n = len(curr_sequence)\n\n # Parameters for line search and perturbation\n k_values = [100, 300, 500] # More varied constraint coverage\n perturbation_steps = 20 # Reduced frequency for more exploration\n perturbation_scale = 0.02 # Moderate perturbation\n alpha_search_iterations = 150 # More thorough line search\n exploration_ratio = 0.7 # Broader exploration\n\n while time.time() < deadline:\n current_time = time.time()\n try:\n # Compute current convolution and max_b\n conv = np.convolve(curr_sequence, curr_sequence)\n max_b = np.max(conv)\n sum_a = np.sum(curr_sequence)\n if sum_a < 0.001:\n raise ValueError(\"Sum too small\")\n\n # Select dynamic constraints based on current convolution\n k = k_values[iteration_count % len(k_values)]\n sorted_conv = sorted(enumerate(conv), key=lambda x: -x[1])\n top_indices = [i for i, _ in sorted_conv[:k]]\n bottom_k = int(k * 0.4)\n bottom_indices = [i for i, _ in sorted_conv[-bottom_k:]]\n spaced_indices = list(range(2, len(conv), 30)) # More spaced out indexing\n all_indices = np.unique(np.concatenate([top_indices, bottom_indices, spaced_indices])).tolist()\n\n # Construct LP constraints\n A = []\n for j in all_indices:\n row = np.zeros(n)\n for m in range(n):\n i = j - m\n if 0 <= i < n:\n row[m] = curr_sequence[i]\n A.append(row)\n A = np.array(A)\n b = np.full(len(all_indices), max_b * 0.9) # Loosened constraints\n\n # Define LP problem: maximize sum of g_0 under constraints\n c = [-1.0] * n # Minimize -sum(g) -> Maximize sum(g)\n bounds = [(0.0, 1000.0) for _ in range(n)]\n\n # Solve the LP using 'highs' for performance\n res = linprog(c=c, A_ub=A, b_ub=b, bounds=bounds, method='highs', options={\"feastol\": 1e-8})\n if res.success:\n g_0 = res.x\n sum_g = np.sum(g_0)\n if sum_g == 0:\n continue # Avoid division by zero\n\n # Normalize to theoretical normalization without forcing sum\n g_0_normalized = g_0 / sum_g * np.sqrt(2 * n)\n\n # Perform extended line search\n low, high = 0.0, 1.0\n for _ in range(alpha_search_iterations):\n m1 = low + (high - low)/3\n m2 = high - (high - low)/3\n new_seq1 = (1 - m1) * curr_sequence + m1 * g_0_normalized\n new_seq2 = (1 - m2) * curr_sequence + m2 * g_0_normalized\n conv1 = np.convolve(new_seq1, new_seq1)\n conv2 = np.convolve(new_seq2, new_seq2)\n new_max_b1 = np.max(conv1)\n new_max_b2 = np.max(conv2)\n if new_max_b1 < new_max_b2:\n high = m2\n else:\n low = m1\n alpha_opt = (low + high) / 2\n new_sequence = (1 - alpha_opt) * curr_sequence + alpha_opt * g_0_normalized\n\n # Ensure non-negative and bounded values\n new_sequence = np.clip(new_sequence, 0.0, 1000.0)\n new_sequence = new_sequence / np.sum(new_sequence) * np.sqrt(2 * n)\n curr_sequence = new_sequence.copy()\n\n # Evaluate and update best sequence\n curr_score = evaluate_sequence(curr_sequence.tolist())\n if curr_score < best_score:\n best_score = curr_score\n best_sequence = curr_sequence.copy()\n no_improvement_counter = 0\n print(f\"New best score: {best_score}\")\n else:\n no_improvement_counter += 1\n else:\n print(\"LP solution not found, breaking\")\n break\n\n # Targeted and random perturbation for exploration\n if iteration_count % perturbation_steps == 0:\n # Identify indices with high and low contribution to max_b\n conv = np.convolve(curr_sequence, curr_sequence)\n top_k = int(n / 10)\n top_indices = np.argsort(conv)[-top_k:]\n bottom_k = int(n / 20)\n bottom_indices = np.argsort(conv)[:bottom_k]\n peak_indices = list(set(top_indices) | set(bottom_indices))\n if not peak_indices:\n continue\n # Reduce peaks and increase valleys\n curr_sequence[peak_indices] -= np.random.uniform(0.001, 0.003, len(peak_indices))\n curr_sequence[peak_indices] = np.maximum(curr_sequence[peak_indices], 0.0)\n # Add random noise for broader exploration\n curr_sequence += np.random.uniform(-0.003, 0.003, n)\n curr_sequence = np.clip(curr_sequence, 0.0, 1000.0)\n curr_sequence = curr_sequence / np.sum(curr_sequence) * np.sqrt(2 * n)\n\n # Early stopping if no improvement for many iterations\n iteration_count += 1\n if iteration_count % 100 == 0:\n print(f\"Iteration {iteration_count}, current best score: {best_score}\")\n if no_improvement_counter > 250: # Increased threshold\n print(\"No improvement for 250 iterations, stopping early\")\n break\n\n except Exception as e:\n print(f\"Error during optimization: {e}. Skipping update.\")\n continue\n\n # Final safety check to ensure non-negative and valid values\n final_sequence = np.maximum(0.0, best_sequence).tolist()\n return [float(x) for x in final_sequence]\n```",
64 "env/all/time/policy": 320.05671777110547,
65 "env/all/time/policy/min": 55.54224681854248,
66 "env/all/time/policy/max": 442.37985610961914,
67 "env/all/time/env_step": 1503.820031573996,
68 "env/all/time/env_step/min": 0.05613875389099121,
69 "env/all/time/env_step/max": 3197.7444117069244,
70 "env/all/time/reward_compute": 3.4365803003311157e-07,
71 "env/all/time/reward_compute/min": 2.2351741790771484e-07,
72 "env/all/time/reward_compute/max": 5.997717380523682e-07,
73 "env/all/by_group/frac_mixed": 1.0,
74 "env/all/by_group/frac_all_good": 0.0,
75 "env/all/by_group/frac_all_bad": 0.0,
76 "advantage/mean": 0.003977205138653517,
77 "advantage/min": -1.0,
78 "advantage/max": 1.0753402709960938,
79 "time/assemble_training_data": 9.809161186218262,
80 "time/kl_vs_base": 98.09162449836731,
81 "kl_policy_base": 0.0007054390734992921,
82 "time/train": 566.0076968669891,
83 "time/save_checkpoint": 19.663567304611206,
84 "time/total": 4332.16147851944
85}[2026-07-09T06:24:18+00:00] job=1812628 node=node-12 ngpu=3 ntrain=1 replicas=2 flash_attn=no
[2026-07-09T08:30:32+00:00] job=1813127 node=node-22 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-09T13:07:08+00:00] job=1813128 node=node-6 ngpu=6 ntrain=2 replicas=4 flash_attn=yes
[2026-07-11T13:13:37+00:00] job=1824165 node=node-6 ngpu=3 ntrain=1 replicas=2 flash_attn=yes
[2026-07-12T04:58:24+00:00] job=1827204 node=node-4 ngpu=6 ntrain=2 replicas=4 flash_attn=yes