diff --git a/data/MP/sweep_splits/gga+u_split_sweep12k.json b/data/MP/sweep_splits/gga+u_split_sweep12k.json new file mode 100644 index 00000000..838dde7b --- /dev/null +++ b/data/MP/sweep_splits/gga+u_split_sweep12k.json @@ -0,0 +1 @@ +{"train": [13, 18, 38, 52, 56, 72, 86, 89, 94, 108, 118, 173, 176, 185, 198, 205, 217, 228, 240, 256, 258, 261, 271, 273, 310, 312, 314, 323, 340, 382, 384, 409, 424, 458, 466, 471, 473, 493, 496, 512, 527, 543, 549, 556, 576, 609, 626, 633, 635, 650, 653, 655, 682, 683, 695, 699, 705, 715, 734, 777, 793, 818, 830, 837, 838, 860, 863, 866, 868, 879, 880, 888, 892, 904, 910, 926, 945, 952, 964, 969, 972, 975, 1000, 1016, 1029, 1041, 1045, 1054, 1059, 1066, 1076, 1086, 1094, 1097, 1118, 1165, 1202, 1206, 1216, 1225, 1228, 1237, 1239, 1245, 1248, 1253, 1258, 1262, 1279, 1281, 1294, 1301, 1308, 1317, 1328, 1333, 1334, 1345, 1346, 1372, 1373, 1377, 1398, 1401, 1406, 1414, 1415, 1448, 1451, 1454, 1458, 1459, 1463, 1472, 1480, 1482, 1510, 1512, 1513, 1516, 1517, 1519, 1529, 1535, 1542, 1571, 1592, 1594, 1622, 1635, 1661, 1674, 1676, 1678, 1687, 1688, 1689, 1694, 1695, 1707, 1713, 1715, 1717, 1723, 1738, 1741, 1747, 1749, 1752, 1755, 1777, 1789, 1808, 1809, 1819, 1828, 1848, 1865, 1871, 1873, 1896, 1897, 1906, 1912, 1932, 1950, 1955, 1963, 1972, 1988, 1997, 1999, 2007, 2019, 2021, 2029, 2034, 2061, 2065, 2073, 2078, 2096, 2105, 2110, 2120, 2121, 2137, 2143, 2166, 2180, 2191, 2193, 2198, 2208, 2239, 2243, 2248, 2253, 2258, 2268, 2280, 2285, 2288, 2291, 2297, 2302, 2310, 2317, 2322, 2323, 2330, 2335, 2364, 2369, 2375, 2382, 2404, 2420, 2429, 2434, 2443, 2451, 2452, 2459, 2463, 2469, 2470, 2473, 2474, 2494, 2502, 2503, 2509, 2511, 2513, 2514, 2534, 2547, 2559, 2564, 2573, 2583, 2596, 2604, 2622, 2625, 2627, 2629, 2631, 2632, 2634, 2641, 2661, 2670, 2672, 2678, 2682, 2687, 2695, 2709, 2710, 2721, 2724, 2725, 2733, 2734, 2737, 2759, 2800, 2802, 2816, 2829, 2831, 2836, 2844, 2851, 2859, 2860, 2879, 2893, 2896, 2902, 2930, 2939, 2941, 2986, 3005, 3008, 3013, 3014, 3016, 3033, 3034, 3044, 3048, 3060, 3062, 3063, 3083, 3098, 3108, 3113, 3115, 3119, 3120, 3121, 3129, 3149, 3155, 3175, 3185, 3225, 3229, 3230, 3233, 3237, 3244, 3245, 3271, 3287, 3288, 3307, 3309, 3312, 3314, 3318, 3330, 3339, 3346, 3351, 3357, 3358, 3374, 3379, 3406, 3416, 3425, 3428, 3435, 3458, 3461, 3463, 3495, 3526, 3536, 3547, 3554, 3561, 3565, 3579, 3588, 3594, 3609, 3614, 3618, 3622, 3624, 3645, 3651, 3652, 3663, 3664, 3673, 3674, 3704, 3705, 3715, 3720, 3727, 3738, 3749, 3764, 3793, 3802, 3807, 3814, 3816, 3826, 3827, 3831, 3839, 3843, 3853, 3854, 3859, 3862, 3869, 3879, 3887, 3890, 3905, 3929, 3931, 3941, 3957, 3965, 3977, 4002, 4014, 4017, 4039, 4051, 4052, 4062, 4064, 4079, 4080, 4090, 4093, 4102, 4106, 4137, 4156, 4162, 4166, 4172, 4174, 4180, 4191, 4196, 4210, 4217, 4226, 4227, 4228, 4239, 4245, 4263, 4269, 4275, 4280, 4300, 4302, 4311, 4320, 4323, 4328, 4349, 4356, 4360, 4382, 4384, 4386, 4391, 4392, 4404, 4418, 4431, 4436, 4441, 4472, 4474, 4479, 4493, 4502, 4505, 4523, 4550, 4551, 4552, 4579, 4588, 4597, 4603, 4605, 4611, 4614, 4615, 4618, 4631, 4639, 4641, 4642, 4654, 4657, 4679, 4682, 4687, 4700, 4714, 4719, 4723, 4742, 4757, 4770, 4788, 4791, 4799, 4804, 4814, 4817, 4833, 4837, 4850, 4852, 4881, 4896, 4909, 4910, 4939, 4946, 4950, 4953, 4963, 4972, 4982, 4993, 4996, 4998, 5002, 5013, 5024, 5041, 5057, 5059, 5062, 5080, 5089, 5102, 5111, 5119, 5125, 5126, 5127, 5131, 5132, 5137, 5145, 5163, 5170, 5172, 5176, 5181, 5182, 5189, 5191, 5196, 5212, 5213, 5217, 5251, 5268, 5281, 5282, 5285, 5293, 5299, 5300, 5303, 5306, 5313, 5325, 5339, 5340, 5345, 5356, 5358, 5367, 5378, 5394, 5418, 5419, 5433, 5436, 5442, 5451, 5457, 5483, 5495, 5522, 5525, 5526, 5539, 5541, 5555, 5569, 5582, 5584, 5593, 5632, 5643, 5660, 5697, 5707, 5725, 5745, 5747, 5769, 5792, 5795, 5812, 5825, 5835, 5843, 5845, 5851, 5859, 5879, 5880, 5883, 5893, 5899, 5901, 5904, 5907, 5915, 5946, 5950, 5997, 6000, 6025, 6028, 6050, 6056, 6061, 6063, 6075, 6094, 6097, 6104, 6108, 6112, 6122, 6163, 6193, 6206, 6218, 6241, 6248, 6253, 6257, 6273, 6289, 6298, 6320, 6334, 6372, 6373, 6375, 6381, 6383, 6385, 6392, 6398, 6404, 6413, 6415, 6429, 6431, 6439, 6444, 6453, 6464, 6472, 6482, 6483, 6507, 6511, 6514, 6526, 6533, 6588, 6613, 6621, 6639, 6655, 6674, 6676, 6681, 6684, 6686, 6691, 6694, 6695, 6698, 6702, 6719, 6739, 6745, 6751, 6767, 6783, 6786, 6832, 6833, 6844, 6848, 6857, 6862, 6888, 6894, 6943, 6954, 6958, 6960, 6974, 7035, 7053, 7057, 7071, 7077, 7083, 7084, 7092, 7106, 7111, 7122, 7139, 7143, 7156, 7174, 7176, 7182, 7183, 7190, 7192, 7203, 7217, 7221, 7228, 7236, 7239, 7249, 7263, 7293, 7307, 7318, 7332, 7340, 7346, 7352, 7356, 7362, 7367, 7373, 7394, 7401, 7403, 7404, 7414, 7431, 7438, 7439, 7440, 7454, 7467, 7483, 7493, 7527, 7528, 7537, 7541, 7542, 7546, 7548, 7568, 7572, 7574, 7575, 7590, 7603, 7634, 7636, 7637, 7647, 7668, 7685, 7689, 7725, 7739, 7762, 7770, 7794, 7795, 7810, 7815, 7816, 7821, 7835, 7840, 7848, 7856, 7873, 7883, 7884, 7887, 7900, 7910, 7919, 7929, 7932, 7939, 7941, 7943, 7955, 7970, 7982, 7985, 7997, 7999, 8016, 8019, 8024, 8028, 8040, 8042, 8043, 8044, 8049, 8065, 8080, 8089, 8095, 8099, 8100, 8109, 8116, 8129, 8135, 8144, 8153, 8166, 8171, 8181, 8190, 8195, 8199, 8203, 8212, 8215, 8232, 8234, 8235, 8255, 8279, 8290, 8294, 8314, 8315, 8321, 8331, 8336, 8341, 8344, 8345, 8348, 8360, 8379, 8390, 8393, 8399, 8406, 8409, 8419, 8426, 8427, 8428, 8443, 8452, 8453, 8472, 8481, 8485, 8488, 8516, 8527, 8538, 8562, 8568, 8583, 8589, 8605, 8606, 8615, 8622, 8641, 8651, 8686, 8715, 8747, 8752, 8767, 8768, 8782, 8793, 8800, 8806, 8813, 8832, 8853, 8873, 8885, 8888, 8889, 8890, 8891, 8894, 8895, 8902, 8908, 8909, 8914, 8919, 8922, 8924, 8927, 8928, 8964, 8967, 8968, 8973, 8975, 8979, 8988, 8989, 8998, 9020, 9059, 9073, 9078, 9105, 9113, 9129, 9132, 9145, 9149, 9153, 9159, 9168, 9170, 9173, 9181, 9182, 9184, 9189, 9190, 9220, 9225, 9226, 9229, 9231, 9233, 9244, 9245, 9275, 9287, 9295, 9301, 9320, 9321, 9340, 9343, 9349, 9350, 9352, 9360, 9362, 9363, 9364, 9367, 9374, 9377, 9381, 9398, 9400, 9401, 9454, 9468, 9469, 9503, 9507, 9508, 9512, 9520, 9538, 9546, 9549, 9556, 9563, 9577, 9579, 9581, 9588, 9594, 9599, 9604, 9605, 9617, 9620, 9643, 9644, 9649, 9655, 9659, 9668, 9673, 9681, 9683, 9687, 9700, 9701, 9714, 9726, 9728, 9729, 9730, 9735, 9753, 9758, 9761, 9788, 9789, 9808, 9818, 9853, 9861, 9863, 9865, 9874, 9889, 9912, 9916, 9919, 9920, 9923, 9937, 9938, 9942, 9945, 9957, 9972, 9995, 9999, 10005, 10030, 10032, 10034, 10036, 10037, 10055, 10058, 10063, 10065, 10076, 10081, 10083, 10093, 10114, 10121, 10129, 10135, 10136, 10147, 10152, 10155, 10157, 10165, 10175, 10209, 10210, 10217, 10218, 10226, 10256, 10263, 10291, 10301, 10307, 10313, 10337, 10340, 10342, 10362, 10367, 10372, 10383, 10386, 10387, 10398, 10404, 10408, 10425, 10430, 10432, 10436, 10447, 10455, 10457, 10476, 10479, 10487, 10503, 10505, 10506, 10507, 10512, 10525, 10543, 10545, 10546, 10548, 10553, 10563, 10570, 10575, 10582, 10608, 10611, 10613, 10631, 10632, 10639, 10649, 10661, 10670, 10700, 10711, 10716, 10730, 10737, 10746, 10751, 10798, 10803, 10804, 10814, 10823, 10834, 10839, 10841, 10853, 10871, 10874, 10884, 10891, 10903, 10909, 10912, 10925, 10933, 10934, 10938, 10974, 10992, 10997, 11013, 11016, 11032, 11033, 11035, 11038, 11043, 11050, 11053, 11090, 11092, 11096, 11108, 11112, 11116, 11124, 11128, 11131, 11142, 11145, 11168, 11170, 11177, 11185, 11187, 11207, 11215, 11217, 11222, 11223, 11259, 11273, 11280, 11283, 11302, 11312, 11320, 11321, 11327, 11333, 11344, 11357, 11361, 11376, 11386, 11392, 11398, 11406, 11409, 11411, 11418, 11425, 11436, 11438, 11447, 11448, 11450, 11453, 11465, 11472, 11483, 11500, 11520, 11540, 11560, 11571, 11579, 11599, 11608, 11622, 11623, 11630, 11637, 11641, 11653, 11665, 11695, 11702, 11719, 11747, 11755, 11768, 11775, 11777, 11782, 11787, 11792, 11822, 11824, 11826, 11839, 11842, 11846, 11848, 11852, 11864, 11872, 11881, 11883, 11908, 11911, 11928, 11932, 11952, 11971, 12032, 12036, 12060, 12074, 12082, 12089, 12098, 12099, 12117, 12152, 12161, 12181, 12186, 12190, 12192, 12193, 12197, 12199, 12202, 12204, 12216, 12218, 12221, 12231, 12242, 12264, 12278, 12281, 12292, 12298, 12299, 12324, 12329, 12332, 12335, 12348, 12369, 12381, 12382, 12389, 12390, 12399, 12405, 12418, 12425, 12431, 12436, 12437, 12441, 12447, 12457, 12475, 12480, 12491, 12492, 12505, 12507, 12513, 12522, 12541, 12564, 12571, 12582, 12586, 12591, 12597, 12607, 12608, 12620, 12639, 12642, 12655, 12686, 12703, 12704, 12707, 12721, 12767, 12783, 12787, 12788, 12798, 12806, 12814, 12817, 12820, 12823, 12830, 12847, 12852, 12859, 12863, 12873, 12882, 12905, 12916, 12921, 12923, 12940, 12941, 12955, 12960, 12971, 12978, 12981, 12995, 13025, 13027, 13033, 13040, 13044, 13052, 13055, 13070, 13111, 13118, 13140, 13157, 13171, 13172, 13224, 13258, 13267, 13290, 13294, 13311, 13312, 13335, 13348, 13352, 13353, 13375, 13382, 13393, 13396, 13407, 13409, 13427, 13431, 13434, 13441, 13450, 13459, 13472, 13480, 13481, 13493, 13507, 13508, 13512, 13517, 13519, 13521, 13528, 13532, 13539, 13547, 13572, 13576, 13580, 13600, 13618, 13633, 13637, 13640, 13641, 13661, 13673, 13676, 13696, 13700, 13706, 13707, 13724, 13729, 13747, 13761, 13765, 13771, 13772, 13774, 13778, 13786, 13809, 13827, 13840, 13854, 13855, 13857, 13865, 13883, 13886, 13900, 13909, 13930, 13940, 13942, 13960, 13972, 13981, 13995, 13996, 13997, 14015, 14016, 14020, 14025, 14027, 14032, 14038, 14051, 14052, 14089, 14096, 14104, 14122, 14126, 14134, 14137, 14141, 14150, 14160, 14161, 14169, 14173, 14184, 14201, 14220, 14231, 14232, 14241, 14256, 14266, 14269, 14273, 14277, 14283, 14288, 14303, 14312, 14319, 14339, 14340, 14352, 14372, 14378, 14394, 14395, 14428, 14451, 14452, 14458, 14465, 14475, 14479, 14480, 14498, 14501, 14502, 14508, 14512, 14526, 14527, 14528, 14536, 14540, 14548, 14568, 14574, 14594, 14623, 14639, 14659, 14666, 14672, 14691, 14693, 14697, 14702, 14704, 14709, 14715, 14720, 14731, 14739, 14758, 14764, 14767, 14778, 14779, 14781, 14813, 14821, 14827, 14832, 14833, 14836, 14850, 14857, 14861, 14868, 14875, 14899, 14901, 14903, 14911, 14915, 14917, 14922, 14929, 14934, 14946, 14947, 14955, 14961, 14965, 14972, 14974, 14982, 14998, 15013, 15041, 15053, 15062, 15068, 15069, 15081, 15116, 15120, 15130, 15131, 15142, 15143, 15152, 15156, 15166, 15170, 15177, 15178, 15182, 15192, 15198, 15209, 15220, 15232, 15234, 15246, 15249, 15250, 15296, 15299, 15300, 15303, 15304, 15306, 15307, 15316, 15327, 15338, 15339, 15341, 15342, 15347, 15348, 15361, 15366, 15404, 15415, 15417, 15420, 15422, 15426, 15430, 15445, 15458, 15466, 15484, 15486, 15503, 15508, 15511, 15513, 15514, 15525, 15532, 15561, 15572, 15580, 15582, 15590, 15599, 15611, 15614, 15655, 15666, 15671, 15675, 15677, 15678, 15681, 15686, 15691, 15693, 15701, 15706, 15721, 15724, 15743, 15754, 15761, 15774, 15795, 15796, 15806, 15819, 15826, 15829, 15832, 15842, 15845, 15860, 15863, 15867, 15873, 15879, 15882, 15885, 15890, 15912, 15917, 15937, 15943, 15945, 15952, 15962, 15969, 15971, 15976, 15981, 15988, 16005, 16016, 16020, 16027, 16035, 16040, 16051, 16054, 16067, 16072, 16091, 16097, 16113, 16123, 16138, 16145, 16150, 16154, 16156, 16165, 16168, 16192, 16216, 16220, 16246, 16262, 16274, 16284, 16301, 16310, 16321, 16345, 16347, 16348, 16351, 16360, 16364, 16389, 16396, 16397, 16408, 16418, 16429, 16431, 16462, 16475, 16485, 16495, 16496, 16497, 16505, 16511, 16522, 16532, 16539, 16550, 16553, 16571, 16573, 16583, 16601, 16602, 16633, 16662, 16671, 16675, 16679, 16683, 16689, 16691, 16709, 16723, 16725, 16728, 16753, 16770, 16791, 16792, 16796, 16801, 16813, 16821, 16824, 16830, 16839, 16849, 16855, 16875, 16883, 16889, 16896, 16900, 16927, 16943, 16944, 16951, 16961, 16968, 16970, 16979, 16981, 17011, 17024, 17025, 17035, 17042, 17061, 17064, 17065, 17066, 17068, 17079, 17082, 17121, 17123, 17129, 17130, 17132, 17140, 17142, 17144, 17148, 17162, 17170, 17178, 17182, 17190, 17200, 17204, 17210, 17218, 17237, 17261, 17270, 17300, 17325, 17331, 17335, 17361, 17376, 17380, 17390, 17400, 17403, 17404, 17406, 17408, 17421, 17444, 17469, 17476, 17482, 17489, 17494, 17505, 17513, 17520, 17522, 17532, 17534, 17553, 17569, 17601, 17606, 17617, 17645, 17651, 17652, 17676, 17692, 17695, 17701, 17703, 17714, 17741, 17742, 17748, 17757, 17760, 17761, 17772, 17777, 17790, 17810, 17817, 17821, 17835, 17861, 17867, 17879, 17909, 17913, 17916, 17918, 17926, 17929, 17933, 17935, 17936, 17937, 17978, 17986, 17990, 17991, 18005, 18017, 18021, 18036, 18040, 18075, 18099, 18105, 18110, 18111, 18113, 18123, 18127, 18128, 18137, 18155, 18156, 18159, 18177, 18179, 18186, 18191, 18205, 18213, 18215, 18218, 18219, 18231, 18238, 18243, 18245, 18257, 18272, 18310, 18311, 18325, 18326, 18340, 18345, 18346, 18348, 18353, 18374, 18388, 18391, 18403, 18417, 18423, 18426, 18442, 18446, 18451, 18461, 18497, 18504, 18506, 18521, 18536, 18549, 18550, 18568, 18570, 18575, 18578, 18586, 18588, 18602, 18617, 18619, 18620, 18630, 18631, 18639, 18641, 18643, 18650, 18660, 18683, 18687, 18691, 18705, 18709, 18713, 18715, 18719, 18723, 18729, 18731, 18750, 18770, 18771, 18786, 18790, 18802, 18809, 18821, 18824, 18838, 18842, 18845, 18846, 18868, 18874, 18906, 18926, 18939, 18941, 18950, 18954, 18956, 18962, 18964, 18965, 18969, 18976, 19003, 19005, 19008, 19028, 19032, 19041, 19048, 19055, 19059, 19062, 19083, 19106, 19119, 19149, 19151, 19155, 19163, 19170, 19175, 19197, 19213, 19216, 19232, 19237, 19251, 19256, 19264, 19276, 19297, 19301, 19305, 19312, 19317, 19326, 19331, 19337, 19354, 19356, 19371, 19375, 19377, 19378, 19400, 19403, 19408, 19413, 19429, 19430, 19432, 19433, 19447, 19448, 19459, 19465, 19473, 19485, 19494, 19496, 19504, 19522, 19529, 19537, 19545, 19555, 19558, 19571, 19574, 19587, 19595, 19597, 19627, 19638, 19657, 19658, 19666, 19679, 19689, 19694, 19695, 19699, 19700, 19714, 19718, 19730, 19731, 19756, 19757, 19792, 19795, 19807, 19811, 19817, 19824, 19832, 19839, 19857, 19862, 19864, 19875, 19876, 19889, 19899, 19906, 19919, 19929, 19962, 19975, 19978, 19984, 19997, 20018, 20037, 20039, 20042, 20050, 20065, 20084, 20097, 20109, 20112, 20115, 20120, 20121, 20124, 20130, 20135, 20140, 20142, 20146, 20154, 20157, 20162, 20166, 20179, 20184, 20214, 20226, 20228, 20262, 20314, 20328, 20330, 20335, 20338, 20340, 20356, 20372, 20404, 20410, 20415, 20418, 20420, 20429, 20439, 20444, 20455, 20456, 20458, 20461, 20467, 20470, 20483, 20489, 20499, 20523, 20543, 20546, 20547, 20552, 20564, 20570, 20576, 20584, 20585, 20593, 20601, 20613, 20616, 20624, 20630, 20637, 20643, 20646, 20669, 20683, 20700, 20730, 20737, 20743, 20744, 20749, 20752, 20760, 20761, 20766, 20770, 20771, 20779, 20798, 20829, 20842, 20848, 20851, 20861, 20864, 20889, 20893, 20899, 20904, 20905, 20935, 20943, 20947, 20952, 20957, 20970, 20988, 21000, 21007, 21018, 21021, 21056, 21061, 21063, 21071, 21080, 21081, 21089, 21093, 21096, 21137, 21138, 21150, 21152, 21153, 21161, 21177, 21187, 21200, 21214, 21222, 21228, 21239, 21248, 21277, 21284, 21286, 21294, 21305, 21313, 21346, 21350, 21353, 21355, 21373, 21389, 21391, 21408, 21414, 21417, 21425, 21428, 21434, 21437, 21438, 21468, 21485, 21500, 21504, 21506, 21511, 21512, 21514, 21533, 21541, 21548, 21549, 21550, 21564, 21566, 21577, 21584, 21589, 21616, 21624, 21628, 21633, 21634, 21643, 21650, 21666, 21677, 21679, 21681, 21682, 21687, 21697, 21698, 21703, 21742, 21748, 21759, 21768, 21770, 21783, 21787, 21791, 21793, 21798, 21817, 21830, 21838, 21841, 21842, 21844, 21854, 21864, 21896, 21906, 21917, 21919, 21921, 21929, 21940, 21942, 21955, 21962, 21974, 21982, 21986, 21992, 21996, 22001, 22020, 22040, 22047, 22057, 22067, 22083, 22085, 22091, 22111, 22113, 22116, 22121, 22128, 22132, 22149, 22155, 22167, 22175, 22177, 22188, 22193, 22206, 22224, 22228, 22229, 22233, 22240, 22259, 22265, 22268, 22282, 22299, 22303, 22307, 22310, 22318, 22330, 22332, 22338, 22342, 22354, 22360, 22368, 22381, 22390, 22393, 22399, 22402, 22407, 22411, 22416, 22435, 22441, 22448, 22500, 22501, 22503, 22509, 22524, 22531, 22532, 22533, 22539, 22544, 22545, 22566, 22570, 22579, 22585, 22594, 22598, 22600, 22603, 22607, 22609, 22618, 22619, 22625, 22643, 22659, 22679, 22681, 22687, 22696, 22697, 22715, 22730, 22735, 22746, 22748, 22749, 22764, 22768, 22776, 22789, 22797, 22812, 22814, 22822, 22825, 22846, 22847, 22855, 22860, 22871, 22875, 22879, 22882, 22885, 22899, 22904, 22907, 22915, 22919, 22939, 22945, 22947, 22949, 22952, 22960, 22964, 22973, 22975, 22993, 22996, 23000, 23005, 23040, 23055, 23079, 23084, 23091, 23094, 23104, 23119, 23129, 23133, 23142, 23143, 23154, 23166, 23167, 23186, 23189, 23194, 23207, 23211, 23227, 23229, 23253, 23262, 23267, 23270, 23280, 23282, 23284, 23285, 23291, 23294, 23295, 23308, 23333, 23342, 23366, 23368, 23371, 23388, 23393, 23400, 23409, 23410, 23411, 23456, 23460, 23473, 23478, 23482, 23488, 23492, 23495, 23515, 23516, 23522, 23525, 23546, 23550, 23557, 23584, 23592, 23600, 23616, 23626, 23645, 23649, 23652, 23654, 23668, 23670, 23674, 23699, 23703, 23704, 23708, 23726, 23727, 23730, 23734, 23738, 23742, 23762, 23773, 23776, 23784, 23790, 23795, 23799, 23802, 23825, 23844, 23856, 23866, 23873, 23879, 23902, 23904, 23906, 23922, 23927, 23929, 23937, 23991, 24004, 24005, 24015, 24023, 24041, 24055, 24061, 24062, 24080, 24088, 24101, 24112, 24146, 24147, 24152, 24165, 24176, 24177, 24191, 24199, 24201, 24209, 24224, 24225, 24226, 24231, 24233, 24236, 24241, 24255, 24260, 24266, 24269, 24271, 24283, 24303, 24306, 24310, 24313, 24318, 24321, 24323, 24327, 24330, 24337, 24339, 24355, 24367, 24381, 24389, 24396, 24409, 24424, 24426, 24438, 24463, 24488, 24506, 24513, 24522, 24527, 24533, 24534, 24537, 24561, 24571, 24590, 24592, 24608, 24612, 24620, 24640, 24657, 24660, 24686, 24701, 24722, 24752, 24765, 24776, 24780, 24794, 24798, 24814, 24816, 24819, 24823, 24841, 24844, 24846, 24860, 24914, 24915, 24917, 24920, 24925, 24939, 24966, 24986, 24993, 24996, 25004, 25015, 25020, 25028, 25031, 25038, 25043, 25058, 25067, 25073, 25085, 25118, 25122, 25124, 25134, 25140, 25157, 25163, 25167, 25173, 25179, 25190, 25197, 25205, 25216, 25222, 25223, 25230, 25237, 25238, 25243, 25257, 25271, 25273, 25274, 25288, 25303, 25316, 25322, 25336, 25341, 25349, 25371, 25374, 25378, 25392, 25405, 25421, 25427, 25430, 25438, 25452, 25454, 25461, 25471, 25478, 25479, 25490, 25501, 25508, 25512, 25524, 25530, 25538, 25558, 25560, 25589, 25611, 25612, 25624, 25633, 25650, 25654, 25661, 25666, 25671, 25674, 25676, 25696, 25718, 25721, 25730, 25737, 25741, 25745, 25749, 25755, 25762, 25763, 25776, 25788, 25798, 25807, 25809, 25819, 25832, 25848, 25855, 25866, 25872, 25891, 25893, 25906, 25909, 25910, 25922, 25940, 25945, 25949, 25953, 25961, 25967, 25976, 25989, 25991, 25997, 26021, 26026, 26030, 26033, 26034, 26035, 26046, 26063, 26079, 26081, 26086, 26091, 26098, 26107, 26141, 26144, 26149, 26151, 26162, 26182, 26186, 26188, 26196, 26203, 26204, 26205, 26225, 26255, 26259, 26267, 26270, 26276, 26280, 26283, 26287, 26304, 26339, 26344, 26363, 26378, 26381, 26389, 26417, 26426, 26436, 26448, 26453, 26467, 26476, 26486, 26494, 26512], "validation": [197, 338, 420, 644, 649, 676, 852, 894, 899, 951, 962, 1027, 1329, 1355, 1653, 1655, 1714, 2047, 2109, 2254, 2265, 2496, 2544, 2644, 2688, 2758, 2813, 2847, 3355, 3477, 3562, 3771, 3776, 3998, 4008, 4168, 4215, 4410, 4491, 4580, 4994, 5044, 5058, 5064, 5270, 5423, 5467, 5558, 5837, 6026, 6030, 6331, 6570, 6604, 6854, 7016, 7031, 7091, 7224, 7272, 7308, 7432, 7663, 7773, 7913, 7938, 8012, 8062, 8097, 8151, 8208, 8209, 8272, 8373, 8504, 8573, 8601, 8632, 8904, 8960, 9013, 9054, 9109, 9162, 9195, 9227, 9249, 9273, 9472, 9896, 9994, 10306, 10336, 10513, 10594, 10680, 10771, 10831, 10847, 10967, 11132, 11141, 11150, 11279, 11306, 11499, 11593, 11656, 11754, 11998, 12272, 12428, 12546, 12550, 12551, 12590, 12677, 12870, 12890, 12907, 13041, 13127, 13380, 13486, 13723, 13753, 13861, 13896, 13899, 13946, 14117, 14172, 14179, 14341, 14406, 14408, 14417, 14549, 14638, 14748, 14906, 15033, 15090, 15201, 15265, 15360, 15363, 15365, 15370, 15377, 15407, 15453, 15547, 15652, 15690, 15768, 15983, 16019, 16053, 16117, 16139, 16203, 16346, 16697, 16736, 17094, 17139, 17250, 17265, 17577, 17628, 17674, 17806, 17808, 17845, 17868, 17948, 18066, 18124, 18385, 18441, 18490, 18656, 18720, 18849, 19039, 19042, 19138, 19566, 19573, 19586, 19685, 19708, 19850, 19943, 20028, 20073, 20176, 20358, 20535, 20567, 20713, 20728, 21002, 21070, 21240, 21400, 21484, 21495, 21538, 21865, 21913, 22066, 22086, 22110, 22129, 22392, 22452, 22620, 22828, 22938, 22989, 23002, 23044, 23122, 23178, 23354, 23385, 23590, 23638, 23749, 23779, 23805, 23941, 24170, 24262, 24291, 24597, 24727, 24741, 24777, 24824, 24867, 24875, 24923, 24948, 24983, 25187, 25337, 25345, 25382, 25480, 25571, 25646, 25750, 25754, 25869, 25916, 26373, 26429, 26439, 26496], "test": [32, 64, 126, 141, 268, 274, 421, 533, 593, 631, 736, 749, 909, 1171, 1649, 1785, 1928, 2016, 2024, 2031, 2202, 2272, 2647, 2704, 2714, 2715, 2957, 3046, 3052, 3072, 3081, 3165, 3212, 3258, 3263, 3361, 3380, 3399, 3560, 3564, 3569, 3729, 3779, 3796, 3799, 3833, 3845, 3873, 3882, 3936, 3949, 4071, 4084, 4176, 4230, 4250, 4289, 4294, 4331, 4332, 4364, 4406, 4422, 4462, 4464, 4483, 4576, 4733, 4754, 4758, 4840, 4966, 4974, 5076, 5184, 5231, 5234, 5244, 5276, 5355, 5443, 5454, 5508, 5519, 5563, 5613, 5658, 5664, 5686, 5724, 5819, 5823, 5848, 5876, 5895, 5935, 5947, 6003, 6021, 6162, 6191, 6460, 6517, 6662, 6671, 6718, 6754, 6760, 6834, 6921, 6936, 6980, 6998, 7121, 7312, 7363, 7381, 7494, 7677, 7680, 7684, 7707, 7799, 7850, 7852, 7865, 7881, 7888, 7897, 7923, 7925, 7940, 7980, 7990, 7991, 8011, 8057, 8063, 8084, 8286, 8413, 8603, 8640, 8693, 8735, 9042, 9211, 9455, 9470, 9490, 9564, 9567, 9593, 9627, 9691, 9773, 9817, 9819, 9855, 9930, 9985, 9996, 10239, 10242, 10246, 10258, 10271, 10304, 10371, 10381, 10391, 10421, 10518, 10554, 10567, 10576, 10580, 10635, 10749, 10754, 10759, 10801, 10886, 10906, 10956, 10959, 10987, 11066, 11076, 11194, 11204, 11267, 11277, 11377, 11404, 11427, 11469, 11528, 11538, 11541, 11547, 11572, 11578, 11609, 11610, 11613, 11635, 11652, 11684, 11691, 11757, 11886, 11888, 11924, 11937, 11942, 11951, 11958, 11961, 11964, 11967, 11973, 12001, 12005, 12015, 12054, 12206, 12238, 12259, 12263, 12267, 12268, 12274, 12340, 12383, 12416, 12417, 12471, 12525, 12623, 12629, 12663, 12689, 12714, 12750, 12755, 12763, 12775, 12825, 12850, 12867, 12919, 12947, 13014, 13057, 13131, 13188, 13198, 13209, 13235, 13269, 13323, 13515, 13609, 13621, 13695, 13793, 13849, 13887, 13905, 13955, 14043, 14082, 14163, 14240, 14279, 14382, 14486, 14506, 14571, 14653, 14685, 14743, 14856, 14883, 14908, 15010, 15026, 15137, 15139, 15161, 15180, 15196, 15199, 15358, 15376, 15399, 15542, 15562, 15679, 15697, 15709, 15731, 15798, 15818, 15846, 15906, 15914, 16075, 16204, 16212, 16272, 16287, 16293, 16305, 16313, 16330, 16334, 16370, 16372, 16400, 16439, 16445, 16476, 16501, 16592, 16615, 16620, 16682, 16826, 16827, 16847, 16902, 16911, 16928, 16935, 16949, 17063, 17108, 17116, 17147, 17222, 17230, 17321, 17327, 17382, 17384, 17449, 17554, 17563, 17611, 17771, 17792, 17831, 17890, 17917, 17944, 17973, 18002, 18023, 18033, 18060, 18114, 18173, 18204, 18274, 18312, 18418, 18419, 18428, 18520, 18603, 18717, 18724, 18759, 19123, 19161, 19419, 19482, 19570, 19681, 19682, 19716, 19761, 19788, 19810, 19941, 19942, 20076, 20118, 20316, 20378, 20560, 20693, 20865, 21156, 21203, 21206, 21267, 21311, 21361, 21536, 21571, 21617, 21636, 21642, 21653, 21656, 21743, 21810, 21859, 21915, 21924, 21925, 21985, 22033, 22081, 22098, 22104, 22153, 22176, 22181, 22209, 22230, 22256, 22281, 22340, 22377, 22385, 22418, 22497, 22714, 22878, 23068, 23092, 23118, 23226, 23249, 23260, 23348, 23349, 23381, 23564, 23745, 23808, 23831, 24083, 24084, 24104, 24126, 24213, 24274, 24287, 24314, 24319, 24342, 24351, 24412, 24446, 24482, 24530, 24532, 24578, 24618, 24638, 24698, 24712, 24871, 24951, 25072, 25095, 25098, 25114, 25120, 25123, 25130, 25150, 25159, 25188, 25207, 25220, 25266, 25298, 25334, 25340, 25347, 25366, 25431, 25443, 25474, 25491, 25532, 25564, 25610, 25613, 25616, 25643, 25685, 25687, 25707, 25723, 25740, 25801, 25803, 25843, 25911, 25985, 26025, 26053, 26150, 26169, 26202, 26226, 26244, 26246, 26291, 26319, 26320, 26321, 26376, 26383, 26425, 26446, 26518]} diff --git a/data/MP/sweep_splits/gga_split_sweep12k.json b/data/MP/sweep_splits/gga_split_sweep12k.json new file mode 100644 index 00000000..92afa884 --- /dev/null +++ b/data/MP/sweep_splits/gga_split_sweep12k.json @@ -0,0 +1 @@ +{"train": [12, 22, 24, 25, 37, 54, 73, 76, 90, 91, 96, 112, 116, 134, 145, 153, 160, 170, 182, 195, 210, 212, 227, 241, 262, 275, 281, 290, 295, 328, 333, 345, 360, 368, 382, 406, 414, 436, 444, 466, 474, 475, 495, 496, 511, 549, 551, 572, 574, 579, 580, 595, 612, 622, 624, 630, 646, 673, 679, 690, 694, 695, 698, 703, 708, 718, 732, 742, 755, 769, 776, 782, 789, 790, 819, 822, 851, 854, 855, 870, 879, 880, 882, 883, 885, 886, 887, 900, 906, 914, 953, 956, 963, 967, 981, 984, 995, 997, 1031, 1038, 1050, 1056, 1057, 1060, 1069, 1081, 1084, 1086, 1089, 1096, 1100, 1115, 1117, 1128, 1130, 1134, 1135, 1143, 1179, 1192, 1196, 1223, 1224, 1241, 1244, 1246, 1247, 1254, 1258, 1265, 1283, 1284, 1285, 1295, 1296, 1302, 1304, 1324, 1331, 1358, 1374, 1380, 1411, 1412, 1415, 1417, 1420, 1433, 1437, 1444, 1478, 1492, 1512, 1519, 1525, 1530, 1536, 1552, 1558, 1582, 1586, 1587, 1591, 1600, 1614, 1630, 1633, 1640, 1643, 1650, 1653, 1661, 1686, 1691, 1693, 1704, 1706, 1708, 1723, 1732, 1748, 1776, 1826, 1834, 1841, 1852, 1858, 1866, 1879, 1880, 1884, 1888, 1895, 1916, 1935, 1965, 1969, 1983, 2005, 2012, 2037, 2047, 2049, 2052, 2088, 2090, 2092, 2098, 2101, 2108, 2121, 2123, 2128, 2169, 2172, 2176, 2178, 2180, 2195, 2198, 2214, 2224, 2231, 2237, 2258, 2291, 2305, 2322, 2344, 2361, 2374, 2393, 2423, 2436, 2455, 2457, 2487, 2489, 2490, 2495, 2503, 2506, 2509, 2514, 2533, 2537, 2541, 2543, 2551, 2553, 2559, 2593, 2595, 2605, 2612, 2614, 2626, 2638, 2642, 2647, 2654, 2664, 2667, 2684, 2687, 2708, 2722, 2728, 2731, 2747, 2777, 2779, 2787, 2793, 2812, 2819, 2824, 2860, 2861, 2877, 2878, 2879, 2907, 2916, 2920, 2923, 2938, 2941, 2954, 2966, 2980, 3008, 3012, 3022, 3027, 3041, 3046, 3048, 3077, 3094, 3097, 3104, 3113, 3116, 3137, 3169, 3175, 3182, 3202, 3212, 3218, 3223, 3242, 3270, 3272, 3290, 3310, 3311, 3320, 3322, 3323, 3337, 3338, 3340, 3350, 3352, 3385, 3397, 3398, 3410, 3440, 3448, 3454, 3465, 3473, 3479, 3481, 3493, 3495, 3510, 3518, 3519, 3523, 3533, 3556, 3570, 3573, 3579, 3584, 3592, 3593, 3595, 3603, 3605, 3613, 3632, 3641, 3658, 3664, 3678, 3698, 3713, 3736, 3738, 3751, 3773, 3778, 3781, 3802, 3812, 3830, 3847, 3855, 3857, 3863, 3869, 3871, 3875, 3881, 3887, 3891, 3896, 3903, 3906, 3913, 3922, 3927, 3934, 3950, 3952, 3971, 3979, 3984, 3992, 3996, 4028, 4039, 4061, 4064, 4101, 4114, 4137, 4142, 4154, 4161, 4166, 4172, 4180, 4181, 4184, 4186, 4187, 4198, 4216, 4237, 4240, 4244, 4251, 4258, 4264, 4269, 4290, 4302, 4306, 4307, 4322, 4323, 4325, 4337, 4347, 4353, 4372, 4377, 4382, 4389, 4391, 4392, 4403, 4411, 4417, 4439, 4442, 4475, 4478, 4484, 4490, 4497, 4506, 4511, 4512, 4540, 4547, 4569, 4651, 4653, 4656, 4662, 4672, 4673, 4680, 4718, 4740, 4741, 4746, 4771, 4790, 4807, 4808, 4811, 4816, 4827, 4833, 4837, 4876, 4905, 4911, 4913, 4918, 4923, 4928, 4933, 4937, 4948, 4961, 4964, 4967, 4969, 4970, 4973, 4979, 4997, 5009, 5019, 5020, 5021, 5023, 5032, 5051, 5053, 5066, 5068, 5069, 5070, 5074, 5083, 5087, 5089, 5100, 5139, 5143, 5146, 5173, 5202, 5210, 5227, 5243, 5249, 5257, 5263, 5271, 5283, 5292, 5295, 5298, 5303, 5311, 5339, 5340, 5350, 5354, 5356, 5360, 5361, 5366, 5381, 5384, 5404, 5407, 5410, 5417, 5428, 5431, 5454, 5458, 5461, 5467, 5479, 5490, 5510, 5518, 5525, 5527, 5528, 5532, 5534, 5590, 5613, 5617, 5622, 5632, 5641, 5644, 5653, 5669, 5670, 5675, 5679, 5684, 5692, 5700, 5713, 5718, 5734, 5737, 5753, 5754, 5798, 5802, 5820, 5825, 5831, 5841, 5843, 5845, 5847, 5854, 5869, 5872, 5875, 5887, 5891, 5897, 5904, 5922, 5923, 5927, 5929, 5939, 5948, 5950, 5958, 5963, 5969, 6001, 6003, 6009, 6038, 6052, 6066, 6069, 6086, 6088, 6092, 6094, 6099, 6108, 6115, 6118, 6119, 6122, 6132, 6149, 6157, 6160, 6162, 6163, 6171, 6176, 6178, 6182, 6183, 6185, 6204, 6211, 6214, 6244, 6252, 6260, 6275, 6304, 6314, 6317, 6327, 6337, 6364, 6392, 6415, 6417, 6418, 6420, 6428, 6449, 6473, 6485, 6508, 6522, 6530, 6532, 6540, 6541, 6544, 6549, 6575, 6580, 6591, 6593, 6598, 6625, 6626, 6628, 6671, 6673, 6688, 6693, 6716, 6732, 6740, 6751, 6760, 6770, 6779, 6785, 6793, 6796, 6798, 6802, 6803, 6805, 6810, 6822, 6827, 6833, 6840, 6850, 6883, 6904, 6911, 6914, 6923, 6936, 6940, 6946, 6947, 6970, 6980, 6982, 7007, 7011, 7012, 7013, 7015, 7025, 7028, 7050, 7056, 7063, 7068, 7072, 7083, 7091, 7105, 7115, 7118, 7120, 7143, 7145, 7151, 7164, 7169, 7209, 7216, 7225, 7228, 7242, 7246, 7273, 7278, 7286, 7289, 7292, 7296, 7306, 7313, 7315, 7322, 7328, 7332, 7338, 7341, 7343, 7344, 7347, 7352, 7361, 7375, 7376, 7379, 7408, 7409, 7413, 7431, 7452, 7464, 7467, 7480, 7493, 7506, 7513, 7530, 7531, 7532, 7533, 7545, 7548, 7549, 7550, 7558, 7565, 7570, 7602, 7623, 7645, 7660, 7661, 7697, 7717, 7723, 7735, 7736, 7757, 7758, 7772, 7784, 7797, 7800, 7811, 7834, 7849, 7873, 7875, 7877, 7888, 7894, 7895, 7907, 7909, 7926, 7932, 7969, 7983, 7985, 7995, 8023, 8033, 8035, 8051, 8066, 8068, 8070, 8075, 8079, 8088, 8091, 8092, 8098, 8107, 8114, 8117, 8143, 8145, 8148, 8156, 8164, 8167, 8174, 8176, 8192, 8210, 8213, 8214, 8271, 8277, 8279, 8301, 8305, 8316, 8332, 8349, 8350, 8353, 8366, 8376, 8386, 8388, 8403, 8406, 8428, 8460, 8466, 8469, 8480, 8486, 8500, 8501, 8515, 8516, 8518, 8522, 8530, 8531, 8532, 8535, 8554, 8556, 8560, 8563, 8575, 8581, 8588, 8592, 8599, 8604, 8616, 8617, 8625, 8632, 8651, 8677, 8683, 8717, 8740, 8746, 8748, 8760, 8763, 8794, 8801, 8803, 8804, 8806, 8816, 8818, 8822, 8837, 8838, 8846, 8854, 8860, 8876, 8896, 8903, 8907, 8914, 8917, 8922, 8938, 8973, 8983, 8989, 8992, 9027, 9028, 9045, 9052, 9060, 9065, 9067, 9077, 9082, 9100, 9102, 9103, 9104, 9118, 9119, 9134, 9137, 9140, 9156, 9157, 9158, 9186, 9192, 9193, 9201, 9208, 9216, 9217, 9219, 9225, 9234, 9245, 9246, 9268, 9272, 9279, 9283, 9290, 9295, 9303, 9321, 9328, 9354, 9357, 9366, 9374, 9378, 9381, 9407, 9408, 9424, 9429, 9430, 9446, 9483, 9487, 9490, 9524, 9527, 9541, 9545, 9547, 9555, 9564, 9566, 9589, 9594, 9597, 9611, 9617, 9619, 9620, 9652, 9665, 9705, 9711, 9716, 9752, 9773, 9775, 9782, 9807, 9813, 9821, 9825, 9826, 9835, 9839, 9844, 9847, 9859, 9865, 9872, 9874, 9906, 9908, 9911, 9935, 9938, 9941, 9959, 9972, 9976, 9984, 9988, 9999, 10003, 10004, 10022, 10024, 10037, 10040, 10069, 10087, 10088, 10093, 10109, 10111, 10116, 10117, 10119, 10145, 10152, 10156, 10160, 10164, 10166, 10184, 10212, 10234, 10250, 10252, 10254, 10301, 10303, 10312, 10319, 10321, 10339, 10348, 10360, 10361, 10365, 10368, 10374, 10376, 10377, 10393, 10396, 10406, 10410, 10448, 10454, 10484, 10494, 10498, 10501, 10504, 10505, 10511, 10530, 10531, 10534, 10541, 10549, 10551, 10570, 10577, 10579, 10586, 10587, 10595, 10602, 10609, 10623, 10630, 10634, 10638, 10647, 10650, 10652, 10659, 10660, 10665, 10671, 10680, 10684, 10686, 10710, 10721, 10764, 10787, 10797, 10801, 10806, 10808, 10816, 10824, 10827, 10839, 10855, 10857, 10873, 10878, 10892, 10901, 10904, 10923, 10937, 10948, 10953, 10958, 10983, 10996, 11007, 11011, 11032, 11037, 11046, 11051, 11055, 11056, 11060, 11061, 11068, 11094, 11096, 11109, 11118, 11119, 11130, 11136, 11145, 11152, 11154, 11209, 11230, 11236, 11279, 11280, 11312, 11319, 11324, 11326, 11329, 11333, 11340, 11342, 11372, 11376, 11381, 11384, 11393, 11426, 11430, 11434, 11437, 11441, 11442, 11445, 11447, 11457, 11462, 11467, 11469, 11472, 11475, 11485, 11491, 11495, 11498, 11505, 11515, 11522, 11545, 11555, 11561, 11567, 11568, 11572, 11574, 11582, 11588, 11590, 11617, 11619, 11622, 11632, 11635, 11659, 11671, 11675, 11696, 11699, 11706, 11714, 11720, 11733, 11739, 11744, 11751, 11782, 11783, 11792, 11811, 11821, 11825, 11826, 11831, 11835, 11837, 11847, 11848, 11877, 11878, 11881, 11885, 11894, 11895, 11937, 11938, 11948, 11974, 12008, 12011, 12019, 12029, 12036, 12049, 12050, 12052, 12054, 12071, 12074, 12084, 12089, 12092, 12093, 12096, 12105, 12140, 12150, 12155, 12161, 12166, 12168, 12171, 12174, 12179, 12180, 12186, 12187, 12197, 12198, 12202, 12205, 12218, 12221, 12236, 12245, 12252, 12267, 12269, 12273, 12284, 12293, 12295, 12299, 12306, 12307, 12309, 12311, 12313, 12324, 12335, 12350, 12370, 12379, 12385, 12406, 12411, 12413, 12415, 12417, 12453, 12459, 12483, 12485, 12495, 12502, 12521, 12531, 12543, 12547, 12548, 12556, 12560, 12563, 12565, 12569, 12570, 12573, 12575, 12581, 12584, 12586, 12588, 12591, 12601, 12607, 12612, 12616, 12622, 12633, 12648, 12683, 12726, 12728, 12731, 12752, 12755, 12785, 12787, 12823, 12836, 12848, 12854, 12856, 12861, 12872, 12875, 12895, 12899, 12902, 12916, 12918, 12937, 12938, 12948, 12954, 12974, 12994, 13000, 13002, 13004, 13026, 13034, 13050, 13053, 13054, 13056, 13067, 13083, 13091, 13108, 13114, 13118, 13127, 13131, 13138, 13140, 13150, 13174, 13177, 13181, 13185, 13187, 13189, 13191, 13217, 13227, 13228, 13237, 13255, 13257, 13262, 13265, 13283, 13284, 13288, 13310, 13320, 13328, 13329, 13358, 13364, 13373, 13375, 13385, 13392, 13400, 13405, 13408, 13438, 13444, 13458, 13471, 13479, 13491, 13493, 13502, 13503, 13513, 13522, 13532, 13544, 13550, 13559, 13564, 13567, 13569, 13596, 13616, 13617, 13627, 13629, 13632, 13636, 13651, 13658, 13663, 13668, 13670, 13672, 13690, 13701, 13728, 13732, 13742, 13748, 13753, 13754, 13796, 13799, 13805, 13810, 13818, 13823, 13824, 13825, 13833, 13835, 13847, 13848, 13855, 13858, 13861, 13864, 13875, 13876, 13889, 13893, 13903, 13915, 13928, 13931, 13940, 13969, 13976, 13980, 13982, 13990, 13997, 14023, 14025, 14026, 14028, 14048, 14049, 14054, 14078, 14092, 14109, 14116, 14132, 14148, 14157, 14191, 14197, 14198, 14222, 14236, 14240, 14253, 14256, 14283, 14287, 14292, 14294, 14311, 14320, 14326, 14335, 14337, 14346, 14347, 14348, 14358, 14365, 14366, 14376, 14383, 14393, 14398, 14409, 14415, 14434, 14436, 14437, 14441, 14444, 14446, 14479, 14482, 14492, 14505, 14523, 14541, 14547, 14550, 14558, 14567, 14573, 14577, 14585, 14586, 14587, 14601, 14619, 14626, 14630, 14649, 14658, 14672, 14684, 14692, 14703, 14714, 14739, 14747, 14751, 14753, 14759, 14761, 14762, 14793, 14798, 14805, 14831, 14835, 14863, 14887, 14889, 14892, 14903, 14904, 14907, 14946, 14957, 14959, 14961, 14965, 14969, 14970, 14972, 14975, 14994, 15006, 15018, 15025, 15054, 15076, 15097, 15098, 15103, 15107, 15119, 15144, 15148, 15160, 15167, 15183, 15187, 15188, 15203, 15209, 15218, 15231, 15238, 15243, 15244, 15248, 15261, 15277, 15278, 15283, 15300, 15305, 15315, 15317, 15330, 15342, 15366, 15368, 15375, 15381, 15384, 15385, 15386, 15391, 15393, 15401, 15412, 15414, 15418, 15421, 15433, 15444, 15447, 15453, 15454, 15463, 15465, 15474, 15477, 15482, 15487, 15494, 15510, 15536, 15537, 15548, 15549, 15551, 15570, 15571, 15579, 15582, 15592, 15612, 15622, 15631, 15637, 15646, 15649, 15658, 15659, 15663, 15668, 15669, 15676, 15692, 15696, 15722, 15736, 15750, 15751, 15772, 15773, 15778, 15783, 15786, 15787, 15795, 15836, 15841, 15846, 15849, 15856, 15861, 15863, 15868, 15880, 15891, 15892, 15895, 15897, 15905, 15909, 15918, 15935, 15936, 15956, 15958, 15960, 15971, 15974, 15982, 15990, 15996, 15998, 16008, 16009, 16014, 16024, 16032, 16038, 16053, 16076, 16083, 16094, 16111, 16112, 16118, 16123, 16139, 16143, 16151, 16173, 16181, 16182, 16186, 16194, 16195, 16236, 16248, 16272, 16279, 16280, 16285, 16287, 16291, 16298, 16299, 16307, 16316, 16329, 16334, 16340, 16360, 16363, 16364, 16380, 16381, 16390, 16394, 16398, 16429, 16439, 16446, 16450, 16461, 16465, 16469, 16473, 16485, 16487, 16491, 16497, 16504, 16510, 16537, 16553, 16562, 16570, 16598, 16599, 16622, 16623, 16630, 16632, 16658, 16675, 16681, 16684, 16689, 16706, 16707, 16710, 16718, 16757, 16759, 16767, 16783, 16800, 16820, 16827, 16837, 16839, 16848, 16876, 16882, 16893, 16913, 16916, 16944, 16947, 16973, 16983, 16997, 17004, 17007, 17032, 17033, 17037, 17038, 17040, 17044, 17063, 17069, 17071, 17085, 17108, 17117, 17126, 17127, 17132, 17159, 17171, 17181, 17189, 17195, 17200, 17202, 17214, 17216, 17226, 17233, 17249, 17250, 17261, 17263, 17270, 17306, 17308, 17312, 17324, 17325, 17331, 17332, 17369, 17375, 17386, 17405, 17407, 17419, 17427, 17433, 17439, 17457, 17465, 17468, 17508, 17517, 17535, 17539, 17547, 17556, 17557, 17558, 17561, 17564, 17576, 17578, 17593, 17599, 17614, 17620, 17628, 17636, 17643, 17644, 17650, 17654, 17657, 17665, 17683, 17687, 17699, 17702, 17708, 17711, 17717, 17719, 17736, 17754, 17758, 17784, 17787, 17788, 17803, 17807, 17808, 17812, 17818, 17836, 17838, 17840, 17843, 17846, 17856, 17875, 17881, 17882, 17886, 17903, 17906, 17908, 17910, 17911, 17923, 17947, 17967, 17976, 17994, 17998, 18003, 18017, 18026, 18046, 18055, 18067, 18074, 18075, 18094, 18108, 18119, 18120, 18122, 18125, 18133, 18145, 18147, 18155, 18163, 18167, 18178, 18179, 18195, 18197, 18226, 18230, 18249, 18256, 18274, 18283, 18284, 18296, 18304, 18306, 18311, 18315, 18318, 18321, 18339, 18365, 18368, 18370, 18395, 18406, 18409, 18413, 18415, 18420, 18423, 18424, 18427, 18441, 18462, 18482, 18485, 18496, 18510, 18515, 18521, 18522, 18524, 18525, 18529, 18530, 18538, 18542, 18556, 18559, 18563, 18570, 18575, 18585, 18587, 18598, 18611, 18628, 18637, 18639, 18643, 18653, 18673, 18678, 18681, 18682, 18683, 18698, 18705, 18711, 18733, 18740, 18744, 18751, 18755, 18757, 18763, 18771, 18780, 18787, 18794, 18819, 18834, 18838, 18847, 18850, 18866, 18867, 18868, 18873, 18886, 18888, 18889, 18890, 18892, 18896, 18897, 18913, 18919, 18934, 18944, 18954, 18970, 18977, 18981, 18996, 19012, 19061, 19076, 19078, 19080, 19087, 19104, 19108, 19116, 19119, 19136, 19146, 19174, 19177, 19182, 19186, 19195, 19212, 19213, 19214, 19217, 19224, 19225, 19257, 19260, 19271, 19281, 19284, 19313, 19317, 19331, 19335, 19336, 19340, 19350, 19377, 19386, 19389, 19397, 19416, 19418, 19434, 19439, 19445, 19456, 19462, 19472, 19473, 19482, 19483, 19494, 19516, 19521, 19527, 19548, 19556, 19560, 19578, 19589, 19614, 19620, 19621, 19639, 19644, 19646, 19648, 19653, 19678, 19682, 19692, 19693, 19696, 19710, 19716, 19722, 19730, 19736, 19760, 19763, 19767, 19778, 19800, 19818, 19847, 19883, 19886, 19891, 19896, 19910, 19912, 19926, 19928, 19930, 19935, 19939, 19952, 19961, 19962, 19971, 19976, 19981, 19991, 19995, 20003, 20012, 20013, 20014, 20046, 20050, 20052, 20054, 20055, 20058, 20084, 20090, 20095, 20097, 20099, 20101, 20102, 20110, 20123, 20130, 20133, 20137, 20141, 20144, 20146, 20172, 20176, 20178, 20180, 20188, 20216, 20226, 20230, 20246, 20247, 20268, 20285, 20299, 20303, 20314, 20329, 20337, 20341, 20348, 20352, 20366, 20368, 20377, 20390, 20391, 20397, 20409, 20423, 20431, 20456, 20470, 20473, 20504, 20525, 20532, 20565, 20566, 20569, 20571, 20599, 20605, 20613, 20615, 20622, 20626, 20627, 20630, 20646, 20650, 20665, 20674, 20675, 20689, 20707, 20714, 20729, 20743, 20745, 20751, 20754, 20774, 20781, 20788, 20795, 20802, 20803, 20804, 20810, 20814, 20818, 20820, 20821, 20823, 20832, 20853, 20857, 20879, 20881, 20888, 20890, 20891, 20912, 20919, 20923, 20929, 20936, 20956, 20969, 20970, 20974, 20979, 20984, 20997, 21015, 21051, 21059, 21066, 21093, 21094, 21095, 21124, 21158, 21165, 21176, 21187, 21191, 21194, 21200, 21215, 21229, 21237, 21243, 21244, 21248, 21249, 21257, 21258, 21273, 21285, 21292, 21293, 21302, 21316, 21317, 21322, 21323, 21332, 21336, 21347, 21350, 21377, 21385, 21407, 21408, 21426, 21430, 21455, 21456, 21463, 21481, 21486, 21489, 21506, 21509, 21511, 21514, 21546, 21550, 21556, 21558, 21564, 21583, 21594, 21600, 21615, 21627, 21638, 21642, 21646, 21670, 21674, 21694, 21701, 21706, 21717, 21720, 21721, 21723, 21748, 21764, 21765, 21776, 21787, 21804, 21808, 21837, 21838, 21862, 21866, 21870, 21874, 21882, 21899, 21903, 21925, 21931, 21932, 21944, 21952, 21978, 21986, 21990, 22002, 22036, 22041, 22045, 22053, 22069, 22074, 22102, 22104, 22105, 22110, 22133, 22149, 22152, 22154, 22178, 22182, 22185, 22199, 22207, 22208, 22220, 22222, 22237, 22246, 22252, 22265, 22267, 22268, 22273, 22286, 22324, 22340, 22342, 22351, 22352, 22383, 22394, 22401, 22403, 22419, 22446, 22450, 22453, 22470, 22489, 22491, 22509, 22514, 22521, 22523, 22527, 22535, 22542, 22546, 22557, 22561, 22567, 22573, 22577, 22581, 22588, 22591, 22592, 22597, 22608, 22617, 22620, 22637, 22646, 22659, 22672, 22680, 22692, 22695, 22699, 22702, 22705, 22713, 22726, 22758, 22762, 22773, 22778, 22808, 22821, 22836, 22844, 22845, 22875, 22877, 22888, 22903, 22916, 22928, 22937, 22945, 22961, 22963, 22989, 23009, 23015, 23022, 23027, 23039, 23054, 23058, 23065, 23101, 23109, 23118, 23120, 23131, 23135, 23140, 23185, 23190, 23197, 23204, 23249, 23250, 23256, 23292, 23296, 23311, 23326, 23332, 23341, 23344, 23369, 23382, 23384, 23389, 23396, 23412, 23417, 23421, 23423, 23428, 23443, 23451, 23455, 23462, 23465, 23469, 23472, 23495, 23503, 23510, 23519, 23525, 23530, 23552, 23559, 23565, 23566, 23575, 23588, 23590, 23601, 23615, 23631, 23634, 23638, 23641, 23646, 23647, 23649, 23654, 23664, 23665, 23678, 23679, 23684, 23697, 23701, 23705, 23715, 23716, 23717, 23723, 23734, 23736, 23749, 23762, 23774, 23777, 23782, 23788, 23808, 23846, 23847, 23874, 23880, 23904, 23906, 23909, 23918, 23924, 23926, 23957, 23959, 23964, 23982, 23989, 23998, 24013, 24029, 24037, 24040, 24046, 24071, 24098, 24102, 24104, 24111, 24116, 24125, 24127, 24136, 24149, 24156, 24157, 24162, 24170, 24173, 24175, 24189, 24195, 24213, 24217, 24224, 24225, 24226, 24227, 24233, 24236, 24252, 24254, 24255, 24279, 24282, 24286, 24288, 24298, 24300, 24302, 24303, 24310, 24313, 24317, 24320, 24347, 24360, 24362, 24365, 24368, 24375, 24382, 24390, 24394, 24401, 24407, 24422, 24426, 24427, 24464, 24481, 24493, 24500, 24507, 24511, 24526, 24543, 24558, 24563, 24564, 24574, 24589, 24607, 24616, 24633, 24668, 24683, 24695, 24719, 24722, 24723, 24731, 24746, 24760, 24772, 24778, 24812, 24830, 24831, 24832, 24835, 24844, 24846, 24853, 24864, 24880, 24882, 24895, 24907, 24913, 24946, 24948, 24950, 24959, 24964, 24973, 24992, 24997, 25000, 25052, 25079, 25081, 25086, 25090, 25103, 25105, 25114, 25119, 25140, 25155, 25161, 25165, 25169, 25185, 25191, 25198, 25207, 25223, 25236, 25243, 25245, 25261, 25268, 25273, 25289, 25298, 25299, 25320, 25325, 25353, 25361, 25373, 25374, 25382, 25404, 25412, 25413, 25414, 25427, 25429, 25430, 25438, 25439, 25446, 25458, 25469, 25470, 25488, 25506, 25519, 25525, 25530, 25564, 25565, 25569, 25581, 25582, 25606, 25627, 25631, 25634, 25637, 25640, 25644, 25646, 25655, 25660, 25669, 25675, 25682, 25688, 25714, 25723, 25724, 25732, 25736, 25740, 25744, 25765, 25767, 25782, 25794, 25803, 25809, 25865, 25869, 25870, 25876, 25879, 25889, 25909, 25911, 25918, 25919, 25921, 25931, 25957, 25958, 25966, 25969, 25987, 25994, 25998, 26003, 26004, 26008, 26026, 26031, 26038, 26059, 26062, 26075, 26077, 26080, 26081, 26086, 26099, 26132, 26139, 26147, 26161, 26167, 26173, 26186, 26202, 26215, 26220, 26242, 26253, 26258, 26259, 26269, 26272, 26280, 26286, 26299, 26325, 26352, 26355, 26389, 26392, 26408, 26424, 26430, 26457, 26466, 26477, 26480, 26482, 26496, 26511, 26518, 26529, 26534, 26558, 26590, 26610, 26615, 26641, 26644, 26648, 26662, 26709, 26718, 26719, 26757, 26760, 26762, 26779, 26788, 26834, 26836, 26844, 26847, 26862, 26866, 26871, 26877, 26883, 26891, 26895, 26898, 26900, 26901, 26908, 26913, 26919, 26932, 26939, 26945, 26946, 26948, 26960, 26973, 26974, 26976, 26977, 26984, 26991, 27000, 27004, 27011, 27020, 27027, 27040, 27051, 27056, 27057, 27058, 27066, 27090, 27102, 27105, 27108, 27112, 27121, 27129, 27133, 27145, 27170, 27175, 27184, 27194, 27211, 27227, 27239, 27243, 27244, 27251, 27271, 27274, 27276, 27284, 27294, 27295, 27298, 27306, 27313, 27327, 27328, 27332, 27359, 27370, 27380, 27388, 27396, 27399, 27415, 27418, 27430, 27434, 27436, 27438, 27445, 27464, 27480, 27493, 27499, 27512, 27517, 27527, 27539, 27554, 27557, 27581, 27588, 27610, 27622, 27629, 27649, 27706, 27709, 27713, 27714, 27715, 27732, 27742, 27744, 27746, 27750, 27756, 27763, 27768, 27779, 27781, 27795, 27798, 27809, 27811, 27835, 27845, 27887, 27900, 27904, 27906, 27920, 27924, 27926, 27934, 27936, 27938, 27948, 27949, 27975, 27976, 27982, 27983, 27992, 27999, 28000, 28005, 28041, 28048, 28060, 28061, 28106, 28121, 28123, 28126, 28150, 28189, 28210, 28227, 28232, 28238, 28257, 28258, 28278, 28284, 28294, 28304, 28305, 28313, 28318, 28322, 28325, 28333, 28339, 28343, 28344, 28367, 28370, 28384, 28391, 28413, 28432, 28438, 28444, 28445, 28455, 28463, 28467, 28471, 28476, 28477, 28494, 28500, 28506, 28523, 28526, 28528, 28546, 28549, 28552, 28566, 28587, 28590, 28597, 28605, 28607, 28612, 28615, 28617, 28648, 28650, 28663, 28679, 28688, 28718, 28720, 28733, 28735, 28763, 28777, 28788, 28790, 28793, 28800, 28816, 28825, 28866, 28870, 28872, 28895, 28899, 28903, 28907, 28915, 28919, 28931, 28937, 28942, 28948, 28966, 28982, 29001, 29010, 29011, 29015, 29041, 29056, 29058, 29080, 29100, 29113, 29115, 29116, 29117, 29126, 29131, 29137, 29158, 29166, 29191, 29195, 29224, 29232, 29252, 29282, 29290, 29306, 29308, 29331, 29338, 29340, 29341, 29351, 29365, 29389, 29401, 29413, 29419, 29427, 29430, 29439, 29444, 29471, 29494, 29497, 29502, 29515, 29518, 29522, 29529, 29540, 29551, 29553, 29555, 29569, 29577, 29579, 29593, 29613, 29619, 29625, 29630, 29631, 29632, 29633, 29646, 29659, 29670, 29690, 29692, 29704, 29738, 29758, 29760, 29769, 29781, 29783, 29784, 29788, 29791, 29800, 29808, 29818, 29819, 29827, 29833, 29851, 29852, 29861, 29866, 29872, 29882, 29892, 29896, 29916, 29919, 29922, 29926, 29947, 29950, 29966, 29981, 29985, 29992, 29993, 29999, 30014, 30018, 30028, 30029, 30035, 30038, 30054, 30074, 30076, 30102, 30105, 30106, 30119, 30141, 30144, 30148, 30157, 30160, 30218, 30222, 30248, 30262, 30279, 30280, 30281, 30282, 30290, 30292, 30313, 30317, 30318, 30320, 30321, 30335, 30342, 30357, 30361, 30364, 30396, 30429, 30441, 30445, 30448, 30457, 30461, 30465, 30467, 30473, 30475, 30483, 30497, 30498, 30500, 30502, 30541, 30545, 30552, 30555, 30556, 30559, 30591, 30595, 30610, 30614, 30623, 30643, 30670, 30711, 30718, 30719, 30722, 30726, 30727, 30731, 30733, 30740, 30753, 30758, 30774, 30781, 30782, 30806, 30812, 30816, 30841, 30843, 30853, 30862, 30881, 30909, 30912, 30913, 30927, 30948, 30968, 30985, 31008, 31013, 31021, 31022, 31028, 31034, 31044, 31052, 31056, 31061, 31068, 31069, 31076, 31089, 31099, 31125, 31163, 31175, 31188, 31207, 31208, 31214, 31220, 31244, 31249, 31250, 31276, 31279, 31286, 31301, 31308, 31337, 31342, 31347, 31348, 31352, 31366, 31393, 31409, 31414, 31416, 31422, 31427, 31431, 31433, 31435, 31437, 31441, 31456, 31458, 31463, 31473, 31517, 31525, 31535, 31541, 31548, 31557, 31559, 31566, 31570, 31572, 31596, 31598, 31615, 31618, 31631, 31642, 31648, 31662, 31664, 31667, 31680, 31681, 31693, 31699, 31703, 31709, 31711, 31713, 31730, 31757, 31762, 31770, 31792, 31795, 31800, 31815, 31816, 31833, 31837, 31855, 31859, 31864, 31870, 31877, 31879, 31882, 31885, 31890, 31891, 31901, 31905, 31915, 31916, 31931, 31939, 31963, 31964, 31965, 31969, 31986, 31991, 32014, 32017, 32027, 32050, 32070, 32072, 32077, 32080, 32085, 32086, 32117, 32120, 32124, 32126, 32132, 32133, 32143, 32155, 32180, 32190, 32197, 32205, 32209, 32217, 32224, 32238, 32240, 32250, 32270, 32276, 32291, 32302, 32303, 32308, 32310, 32314, 32315, 32330, 32345, 32353, 32360, 32363, 32385, 32413, 32423, 32444, 32447, 32450, 32452, 32454, 32469, 32474, 32481, 32494, 32501, 32502, 32516, 32521, 32524, 32535, 32550, 32555, 32560, 32584, 32587, 32637, 32642, 32644, 32648, 32663, 32674, 32678, 32697, 32713, 32728, 32732, 32748, 32750, 32754, 32761, 32769, 32784, 32785, 32805, 32811, 32818, 32826, 32833, 32844, 32848, 32849, 32883, 32892, 32896, 32899, 32904, 32917, 32922, 32929, 32938, 32966, 32971, 32978, 33004, 33014, 33037, 33039, 33041, 33049, 33051, 33054, 33058, 33060, 33093, 33097, 33106, 33114, 33118, 33137, 33144, 33157, 33176, 33177, 33184, 33204, 33209, 33237, 33239, 33244, 33248, 33258, 33262, 33275, 33284, 33294, 33298, 33302, 33319, 33326, 33367, 33369, 33370, 33379, 33384, 33400, 33409, 33415, 33418, 33421, 33429, 33431, 33446, 33451, 33457, 33461, 33467, 33470, 33487, 33503, 33506, 33510, 33512, 33513, 33515, 33516, 33517, 33523, 33557, 33568, 33577, 33582, 33590, 33609, 33635, 33637, 33643, 33682, 33687, 33688, 33697, 33698, 33706, 33709, 33723, 33729, 33730, 33735, 33746, 33749, 33752, 33759, 33761, 33767, 33799, 33802, 33804, 33824, 33829, 33830, 33833, 33836, 33842, 33843, 33869, 33880, 33899, 33920, 33930, 33937, 33939, 33941, 33963, 33971, 33978, 33993, 34005, 34007, 34019, 34026, 34055, 34071, 34072, 34074, 34082, 34109, 34116, 34117, 34120, 34141, 34149, 34153, 34166, 34194, 34198, 34204, 34237, 34240, 34241, 34257, 34283, 34297, 34324, 34326, 34341, 34346, 34355, 34356, 34381, 34400, 34403, 34418, 34421, 34434, 34440, 34462, 34464, 34479, 34488, 34490, 34493, 34524, 34538, 34542, 34544, 34548, 34582, 34584, 34589, 34610, 34629, 34638, 34678, 34680, 34707, 34719, 34727, 34728, 34774, 34776, 34779, 34792, 34793, 34794, 34802, 34805, 34808, 34824, 34850, 34853, 34865, 34871, 34904, 34905, 34906, 34910, 34916, 34927, 34950, 34975, 34983, 34992, 35007, 35022, 35033, 35038, 35043, 35051, 35053, 35058, 35086, 35107, 35108, 35116, 35124, 35140, 35141, 35145, 35162, 35169, 35177, 35189, 35190, 35191, 35194, 35216, 35219, 35234, 35237, 35238, 35240, 35248, 35253, 35260, 35268, 35293, 35294, 35301, 35307, 35324, 35325, 35326, 35327, 35348, 35353, 35378, 35381, 35403, 35417, 35430, 35439, 35446, 35457, 35466, 35471, 35484, 35488, 35490, 35502, 35519, 35521, 35529, 35532, 35536, 35541, 35542, 35543, 35581, 35590, 35619, 35638, 35639, 35653, 35668, 35669, 35673, 35682, 35683, 35690, 35694, 35706, 35707, 35709, 35738, 35754, 35756, 35757, 35759, 35763, 35766, 35777, 35783, 35784, 35792, 35806, 35811, 35819, 35827, 35838, 35839, 35840, 35865, 35893, 35917, 35924, 35948, 35957, 35960, 35973, 35986, 35997, 36000, 36005, 36014, 36020, 36022, 36031, 36035, 36046, 36066, 36069, 36071, 36082, 36086, 36088, 36122, 36136, 36141, 36147, 36152, 36201, 36205, 36213, 36214, 36221, 36224, 36231, 36234, 36237, 36255, 36261, 36269, 36282, 36303, 36304, 36314, 36318, 36326, 36333, 36337, 36349, 36351, 36364, 36388, 36397, 36400, 36410, 36422, 36426, 36440, 36472, 36486, 36551, 36555, 36569, 36574, 36579, 36595, 36596, 36599, 36603, 36622, 36628, 36644, 36657, 36674, 36678, 36693, 36697, 36715, 36719, 36733, 36748, 36757, 36773, 36781, 36791, 36793, 36798, 36809, 36815, 36828, 36833, 36838, 36839, 36859, 36867, 36875, 36880, 36883, 36888, 36898, 36902, 36907, 36911, 36920, 36926, 36944, 36947, 36959, 36970, 36976, 36978, 37015, 37017, 37031, 37036, 37038, 37040, 37049, 37050, 37056, 37066, 37073, 37074, 37090, 37104, 37107, 37112, 37114, 37130, 37139, 37153, 37160, 37165, 37168, 37186, 37197, 37228, 37229, 37236, 37237, 37239, 37257, 37271, 37277, 37280, 37297, 37302, 37310, 37312, 37313, 37332, 37335, 37336, 37362, 37373, 37396, 37406, 37413, 37415, 37418, 37420, 37423, 37427, 37440, 37475, 37481, 37491, 37494, 37501, 37502, 37507, 37525, 37531, 37536, 37540, 37544, 37557, 37574, 37580, 37584, 37587, 37593, 37605, 37631, 37636, 37646, 37660, 37662, 37668, 37704, 37709, 37731, 37735, 37743, 37746, 37750, 37751, 37757, 37761, 37789, 37805, 37815, 37819, 37825, 37830, 37841, 37860, 37861, 37880, 37887, 37903, 37908, 37912, 37924, 37930, 37955, 37960, 37968, 37982, 37996, 38015, 38017, 38020, 38027, 38028, 38062, 38076, 38083, 38089, 38109, 38138, 38141, 38143, 38157, 38159, 38166, 38171, 38175, 38177, 38181, 38191, 38200, 38215, 38218, 38222, 38244, 38248, 38264, 38266, 38283, 38290, 38294, 38308, 38320, 38324, 38331, 38333, 38337, 38341, 38357, 38364, 38370, 38377, 38383, 38385, 38390, 38394, 38400, 38405, 38409, 38442, 38443, 38444, 38450, 38460, 38461, 38471, 38485, 38486, 38500, 38508, 38530, 38551, 38555, 38577, 38579, 38589, 38600, 38603, 38614, 38651, 38654, 38659, 38664, 38671, 38680, 38690, 38691, 38699, 38707, 38712, 38729, 38733, 38735, 38749, 38756, 38762, 38771, 38774, 38809, 38810, 38816, 38831, 38848, 38854, 38861, 38863, 38873, 38878, 38888, 38895, 38908, 38922, 38939, 38941, 38944, 38949, 38964, 38965, 38971, 38995, 39023, 39028, 39030, 39040, 39047, 39055, 39057, 39058, 39064, 39074, 39084, 39099, 39105, 39106, 39109, 39122, 39154, 39156, 39176, 39191, 39195, 39206, 39207, 39209, 39217, 39224, 39234, 39242, 39264, 39273, 39278, 39289, 39299, 39301, 39309, 39312, 39324, 39332, 39334, 39344, 39353, 39355, 39366, 39375, 39380, 39381, 39385, 39397, 39411, 39416, 39417, 39434, 39464, 39483, 39509, 39521, 39555, 39556, 39581, 39584, 39592, 39599, 39607, 39609, 39635, 39637, 39663, 39673, 39689, 39690, 39691, 39699, 39701, 39724, 39738, 39739, 39743, 39745, 39752, 39765, 39781, 39789, 39791, 39792, 39800, 39806, 39808, 39809, 39818, 39822, 39825, 39830, 39835, 39851, 39854, 39861, 39868, 39876, 39878, 39884, 39889, 39891, 39893, 39900, 39909, 39910, 39920, 39941, 39972, 39997, 40019, 40035, 40119, 40123, 40129, 40132, 40134, 40135, 40136, 40157, 40162, 40170, 40203, 40205, 40211, 40215, 40234, 40255, 40256, 40259, 40266, 40270, 40273, 40280, 40281, 40288, 40312, 40315, 40318, 40320, 40322, 40328, 40362, 40368, 40373, 40375, 40378, 40381, 40387, 40393, 40404, 40405, 40411, 40423, 40426, 40427, 40441, 40444, 40449, 40461, 40464, 40470, 40474, 40477, 40482, 40499, 40503, 40512, 40513, 40514, 40517, 40519, 40521, 40529, 40531, 40534, 40540, 40543, 40561, 40575, 40596, 40600, 40606, 40622, 40627, 40629, 40633, 40639, 40661, 40663, 40671, 40678, 40686, 40690, 40693, 40700, 40702, 40719, 40730, 40731, 40735, 40745, 40751, 40766, 40769, 40772, 40778, 40788, 40795, 40796, 40821, 40822, 40838, 40841, 40849, 40854, 40860, 40863, 40864, 40892, 40894, 40912, 40913, 40919, 40922, 40934, 40964, 40975, 41000, 41005, 41009, 41010, 41014, 41031, 41032, 41033, 41038, 41051, 41065, 41070, 41096, 41097, 41103, 41105, 41109, 41111, 41114, 41115, 41127, 41132, 41134, 41145, 41147, 41167, 41173, 41179, 41182, 41207, 41220, 41221, 41226, 41232, 41235, 41250, 41292, 41308, 41320, 41360, 41361, 41377, 41381, 41400, 41401, 41403, 41439, 41473, 41487, 41498, 41511, 41519, 41526, 41532, 41545, 41568, 41572, 41574, 41575, 41586, 41598, 41599, 41601, 41604, 41610, 41619, 41635, 41652, 41676, 41678, 41687, 41694, 41696, 41703, 41732, 41734, 41735, 41745, 41758, 41766, 41774, 41784, 41790, 41797, 41817, 41841, 41844, 41853, 41861, 41867, 41873, 41879, 41883, 41885, 41887, 41903, 41925, 41932, 41939, 41960, 41964, 41972, 41984, 41994, 42019, 42045, 42053, 42057, 42064, 42087, 42095, 42102, 42113, 42114, 42115, 42120, 42134, 42151, 42160, 42161, 42162, 42165, 42167, 42171, 42174, 42178, 42190, 42198, 42199, 42224, 42227, 42238, 42245, 42247, 42262, 42266, 42273, 42275, 42284, 42292, 42293, 42302, 42318, 42329, 42331, 42335, 42339, 42354, 42360, 42361, 42365, 42376, 42396, 42405, 42411, 42426, 42439, 42441, 42443, 42449, 42452, 42465, 42483, 42496, 42509, 42528, 42567, 42572, 42579, 42585, 42586, 42593, 42601, 42605, 42611, 42628, 42635, 42657, 42667, 42672, 42677, 42682, 42695, 42700, 42702, 42704, 42713, 42715, 42719, 42730, 42734, 42740, 42742, 42750, 42768, 42777, 42787, 42788, 42790, 42808, 42821, 42823, 42826, 42834, 42849, 42863, 42884, 42887, 42897, 42909, 42916, 42923, 42932, 42939, 42944, 42955, 42959, 42980, 42986, 43002, 43006, 43007, 43010, 43013, 43032, 43036, 43037, 43064, 43067, 43068, 43091, 43103, 43119, 43128, 43129, 43141, 43144, 43150, 43176, 43191, 43215, 43217, 43228, 43243, 43247, 43250, 43270, 43271, 43274, 43279, 43284, 43301, 43303, 43311, 43327, 43342, 43347, 43390, 43399, 43401, 43402, 43405, 43425, 43428, 43431, 43433, 43457, 43466, 43468, 43482, 43486, 43495, 43497, 43505, 43510, 43533, 43534, 43537, 43541, 43562, 43568, 43586, 43594, 43609, 43614, 43619, 43623, 43640, 43641, 43646, 43647, 43657, 43664, 43666, 43672, 43676, 43690, 43710, 43731, 43732, 43742, 43751, 43754, 43761, 43773, 43774, 43780, 43791, 43812, 43820, 43829, 43830, 43847, 43850, 43863, 43867, 43873, 43880, 43887, 43889, 43891, 43901, 43912, 43920, 43953, 43978, 43980, 43991, 43993, 43998, 44006, 44011, 44013, 44020, 44031, 44038, 44047, 44048, 44095, 44099, 44102, 44110, 44112, 44126, 44147, 44170, 44171, 44176, 44178, 44186, 44189, 44209, 44222, 44237, 44244, 44255, 44256, 44261, 44266, 44280, 44281, 44291, 44299, 44311, 44327, 44330, 44340, 44341, 44349, 44354, 44355, 44379, 44386, 44429, 44431, 44433, 44439, 44458, 44473, 44485, 44489, 44495, 44510, 44513, 44518, 44564, 44568, 44571, 44573, 44577, 44582, 44595, 44609, 44615, 44620, 44631, 44646, 44655, 44659, 44661, 44673, 44683, 44687, 44689, 44693, 44694, 44696, 44706, 44710, 44711, 44716, 44739, 44742, 44746, 44748, 44753, 44768, 44769, 44770, 44771, 44790, 44791, 44797, 44803, 44806, 44831, 44836, 44853, 44868, 44880, 44895, 44898, 44909, 44927, 44933, 44941, 44951, 44954, 44984, 44987, 44994, 45004, 45013, 45019, 45021, 45024, 45046, 45050, 45058, 45117, 45118, 45132, 45145, 45160, 45166, 45167, 45173, 45177, 45203, 45208, 45214, 45215, 45218, 45227, 45238, 45239, 45252, 45270, 45282, 45300, 45303, 45308, 45315, 45329, 45333, 45334, 45345, 45368, 45373, 45389, 45402, 45405, 45427, 45432, 45455, 45467, 45502, 45533, 45539, 45542, 45554, 45556, 45563, 45567, 45570, 45583, 45585, 45588, 45597, 45598, 45602, 45611, 45612, 45613, 45632, 45662, 45663, 45671, 45673, 45678, 45680, 45696, 45697, 45708, 45716, 45720, 45723, 45749, 45755, 45758, 45763, 45769, 45778, 45781, 45795, 45798, 45825, 45829, 45839, 45845, 45866, 45867, 45874, 45884, 45886, 45899, 45907, 45915, 45918, 45926, 45935, 45936, 45938, 45973, 45981, 45988, 45989, 45994, 46002, 46009, 46014, 46058, 46062, 46071, 46073, 46076, 46102, 46113, 46128, 46130, 46135, 46138, 46152, 46162, 46169, 46191, 46209, 46250, 46255, 46262, 46263, 46272, 46279, 46281, 46285, 46287, 46309, 46310, 46313, 46323, 46337, 46360, 46374, 46377, 46378, 46383, 46384, 46388, 46399, 46401, 46406, 46409, 46413, 46414, 46416, 46418, 46427, 46438, 46453, 46455, 46467, 46469, 46480, 46481, 46489, 46502, 46503, 46504, 46541, 46557, 46563, 46573, 46578, 46585, 46586, 46587, 46591, 46598, 46601, 46607, 46617, 46643, 46659, 46668, 46682, 46702, 46704, 46722, 46750, 46758, 46785, 46794, 46795, 46803, 46806, 46810, 46860, 46864, 46875, 46879, 46894, 46901, 46928, 46941, 46952, 46954, 46964, 46968, 46987, 47002, 47014, 47015, 47018, 47051, 47056, 47061, 47070, 47082, 47091, 47109, 47110, 47159, 47163, 47187, 47191, 47197, 47198, 47206, 47211, 47220, 47241, 47261, 47287, 47290, 47309, 47314, 47328, 47354, 47382, 47391, 47392, 47398, 47403, 47410, 47415, 47426, 47435, 47438, 47448, 47452, 47453, 47457, 47464, 47466, 47471, 47481, 47489, 47495, 47502, 47506, 47512, 47541, 47544, 47549, 47557, 47561, 47593, 47613, 47620, 47627, 47638, 47639, 47642, 47652, 47666, 47669, 47670, 47726, 47734, 47745, 47764, 47768, 47777, 47781, 47812, 47828, 47829, 47830, 47848, 47852, 47878, 47898, 47901, 47925, 47938, 47942, 47956, 47960, 47961, 47964, 47974, 47975, 47984, 47992, 48006, 48011, 48022, 48031, 48054, 48060, 48069, 48126, 48166, 48168, 48188, 48197, 48205, 48210, 48240, 48243, 48244, 48249, 48264, 48288, 48294, 48302, 48304, 48312, 48317, 48323, 48324, 48339, 48351, 48358, 48361, 48367, 48382, 48390, 48395, 48408, 48409, 48420, 48421, 48428, 48434, 48436, 48453, 48458, 48462, 48469, 48470, 48477, 48479, 48496, 48497, 48498, 48499, 48528, 48534, 48537, 48550, 48562, 48585, 48586, 48599, 48615, 48623, 48627, 48629, 48632, 48636, 48639, 48648, 48655, 48656, 48678, 48680, 48685, 48692, 48696, 48705, 48709, 48716, 48718, 48722, 48727, 48742, 48746, 48750, 48755, 48767, 48770, 48796, 48799, 48813, 48818, 48834, 48836, 48837, 48843, 48847, 48854, 48867, 48873, 48887, 48892, 48897, 48904, 48938, 48944, 48949, 48962, 48981, 48986, 48997, 49002, 49004, 49005, 49026, 49044, 49046, 49049, 49051, 49063, 49065, 49100, 49103, 49121, 49125, 49136, 49137, 49151, 49169, 49173, 49180, 49183, 49209, 49230, 49255, 49259, 49262, 49280, 49288, 49314, 49318, 49326, 49334, 49355, 49357, 49359, 49364, 49381, 49382, 49384, 49387, 49392, 49401, 49403, 49455, 49458, 49466, 49469, 49474, 49491, 49529, 49540, 49543, 49558, 49562, 49591, 49592, 49605, 49607, 49614, 49618, 49620, 49621, 49626, 49637, 49644, 49657, 49668, 49681, 49697, 49709, 49714, 49724, 49725, 49753, 49754, 49755, 49763, 49774, 49775, 49787, 49796, 49798, 49799, 49813, 49819, 49839, 49840, 49850, 49855, 49864, 49868, 49880, 49888, 49894, 49895, 49899, 49912, 49935, 49951, 49968, 49972, 49994, 49999, 50007, 50014, 50015, 50019, 50038, 50056, 50064, 50073, 50079, 50080, 50087, 50103, 50105, 50129, 50138, 50139, 50155, 50169, 50174, 50190, 50194, 50198, 50213, 50244, 50253, 50258, 50269, 50270, 50274, 50278, 50296, 50299, 50320, 50321, 50323, 50326, 50334, 50343, 50347, 50352, 50359, 50362, 50371, 50372, 50386, 50401, 50410, 50420, 50424, 50428, 50431, 50454, 50462, 50463, 50466, 50472, 50492, 50494, 50507, 50509, 50512, 50513, 50518, 50540, 50571, 50572, 50591, 50600, 50606, 50615, 50620, 50634, 50637, 50643, 50652, 50657, 50667, 50673, 50680, 50682, 50697, 50716, 50734, 50740, 50753, 50764, 50773, 50784, 50791, 50792, 50797, 50809, 50811, 50829, 50836, 50839, 50844, 50858, 50881, 50893, 50894, 50905, 50909, 50910, 50923, 50925, 50934, 50948, 50970, 50979, 50983, 50987, 51041, 51043, 51063, 51066, 51084, 51086, 51099, 51105, 51109, 51135, 51137, 51139, 51143, 51145, 51152, 51161, 51164, 51167, 51182, 51184, 51204, 51207, 51228, 51239, 51246, 51253, 51259, 51265, 51270, 51280, 51281, 51290, 51297, 51302, 51307, 51309, 51315, 51331, 51332, 51357, 51361, 51369, 51374, 51378, 51396, 51398, 51400, 51405, 51416, 51422, 51434, 51444, 51445, 51457, 51463, 51464, 51466, 51470, 51474, 51476, 51489, 51491, 51493, 51499, 51508, 51521, 51534, 51552, 51561, 51578, 51600, 51614, 51615, 51616, 51622, 51634, 51639, 51646, 51647, 51657, 51658, 51665, 51678, 51680, 51681, 51686, 51689, 51690, 51693, 51706, 51726, 51750, 51757, 51760, 51762, 51764, 51768, 51778, 51792, 51796, 51808, 51811, 51815, 51822, 51830, 51838, 51840, 51861, 51874, 51882, 51897, 51906, 51910, 51913, 51920, 51922, 51924, 51934, 51958, 51978, 51983, 51996, 52013, 52050, 52072, 52073, 52074, 52080, 52100, 52104, 52111, 52136, 52147, 52159, 52165, 52176, 52190, 52201, 52209, 52223, 52239, 52249, 52258, 52273, 52284, 52285, 52286, 52292, 52296, 52300, 52304, 52309, 52341, 52373, 52381, 52382, 52385, 52406, 52420, 52425, 52426, 52431, 52433, 52437, 52449, 52450, 52455, 52456, 52472, 52482, 52485, 52489, 52503, 52515, 52518, 52521, 52525, 52567, 52571, 52580, 52587, 52600, 52618, 52631, 52632, 52636, 52637, 52638, 52650, 52654, 52680, 52689, 52696, 52700, 52704, 52727, 52733, 52739, 52760, 52774, 52786, 52788, 52792, 52795, 52818, 52825, 52854, 52873, 52875, 52889, 52897, 52902, 52919, 52965, 52975, 52981, 52994, 53003, 53014, 53022, 53026, 53029, 53030, 53034, 53037, 53040, 53050, 53052, 53075, 53078, 53096, 53109, 53116, 53138, 53164, 53181, 53189, 53198, 53200, 53213, 53220, 53224, 53233, 53248, 53253, 53269, 53276, 53280, 53283, 53284, 53301, 53311, 53327, 53331, 53347, 53353, 53355, 53365, 53374, 53390, 53392, 53394, 53397, 53407, 53416, 53419, 53420, 53422, 53434, 53440, 53454, 53459, 53481, 53494, 53497, 53509, 53533, 53536, 53538, 53542, 53555, 53570, 53572, 53579, 53585, 53588, 53592, 53595, 53600, 53608, 53635, 53642, 53648, 53649, 53651, 53654, 53662, 53685, 53706, 53707, 53720, 53728, 53740, 53741, 53745, 53755, 53757, 53764, 53778, 53786, 53794, 53821, 53823, 53825, 53840, 53847, 53849, 53857, 53884, 53890, 53904, 53912, 53918, 53945, 53950, 53953, 53969, 53998, 54027, 54029, 54037, 54055, 54056, 54072, 54075, 54076, 54079, 54080, 54083, 54085, 54091, 54100, 54101, 54104, 54105, 54108, 54120, 54122, 54135, 54136, 54152, 54181, 54203, 54212, 54214, 54217, 54221, 54226, 54236, 54263, 54264, 54266, 54298, 54300, 54303, 54316, 54317, 54331, 54346, 54364, 54377, 54394, 54432, 54435, 54449, 54453, 54454, 54457, 54460, 54504, 54509, 54522, 54525, 54528, 54546, 54557, 54559, 54571, 54575, 54585, 54586, 54589, 54593, 54602, 54609, 54614, 54638, 54650, 54674, 54676, 54682, 54687, 54706, 54717, 54724, 54729, 54734, 54736, 54760, 54770, 54775, 54781, 54784, 54805, 54806, 54813, 54815, 54830, 54841, 54852, 54857, 54860, 54871, 54885, 54902, 54919, 54923, 54938, 54945, 54947, 54954, 54959, 54973, 54974, 55003, 55014, 55016, 55021, 55069, 55077, 55086, 55107, 55112, 55119, 55122, 55124, 55141, 55142, 55150, 55154, 55158, 55174, 55184, 55187, 55198, 55223, 55224, 55230, 55233, 55243, 55246, 55265, 55268, 55274, 55290, 55308, 55314, 55338, 55346, 55354, 55361, 55375, 55382, 55390, 55442, 55444, 55446, 55448, 55452, 55455, 55459, 55460, 55486, 55488, 55507, 55512, 55519, 55556, 55563, 55569, 55571, 55572, 55575, 55578, 55600, 55608, 55613, 55616, 55637, 55649, 55655, 55662, 55666, 55668, 55688, 55697, 55700, 55709, 55719, 55730, 55735, 55738, 55746, 55749, 55751, 55757, 55765, 55773, 55776, 55783, 55797, 55801, 55818, 55821, 55844, 55850, 55851, 55859, 55867, 55870, 55875, 55879, 55882, 55903, 55905, 55909, 55921, 55938, 55943, 55947, 55950, 55963, 55965, 55998, 56000, 56002, 56007, 56008, 56017, 56022, 56037, 56041, 56073, 56074, 56082, 56083, 56097, 56099, 56100, 56107, 56116, 56119, 56128, 56130, 56139, 56142, 56145, 56146, 56147, 56152, 56171, 56178, 56181, 56182, 56191, 56204, 56217, 56219, 56221, 56225, 56227, 56253, 56264, 56273, 56278, 56303, 56306, 56318, 56332, 56334, 56351, 56364, 56379, 56391, 56407, 56441, 56453, 56455, 56463, 56467, 56478, 56501, 56509, 56513, 56514, 56518, 56529, 56530, 56535, 56560, 56566, 56570, 56574, 56577, 56588, 56589, 56590, 56599, 56607, 56626, 56627, 56635, 56637, 56645, 56665, 56668, 56669, 56670, 56701, 56702, 56711, 56713, 56726, 56742, 56755, 56759, 56787, 56812, 56813, 56825, 56831, 56832, 56843, 56852, 56899, 56905, 56925, 56928, 56930, 56938, 56946, 56952, 56962, 56964, 56970, 56983, 56993, 57003, 57010, 57013, 57022, 57024, 57032, 57043, 57063, 57079, 57087, 57088, 57097, 57101, 57105, 57114, 57115, 57120, 57130, 57135, 57147, 57151, 57153, 57169, 57172, 57173, 57174, 57190, 57199, 57203, 57207, 57208, 57210, 57226, 57235, 57245, 57249, 57250, 57260, 57275, 57301, 57330, 57339, 57344, 57354, 57358, 57373, 57389, 57390, 57391, 57438, 57440, 57465, 57466, 57468, 57493, 57506, 57508, 57516, 57520, 57521, 57534, 57542, 57556, 57569, 57595, 57605, 57609, 57639, 57648, 57651, 57655, 57656, 57668, 57686, 57716, 57723, 57727, 57733, 57735, 57740, 57749, 57770, 57789, 57791, 57796, 57798, 57817, 57827, 57834, 57844, 57858, 57882, 57886, 57887, 57897, 57904, 57913, 57925, 57941, 57942, 57948, 57953, 57961, 57965, 57982, 57987, 57999, 58020, 58034, 58037, 58058, 58059, 58061, 58076, 58103, 58107, 58109, 58126, 58128, 58130, 58131, 58132, 58136, 58140, 58173, 58180, 58191, 58204, 58217, 58218, 58224, 58225, 58227, 58252, 58281, 58292, 58299, 58300, 58303, 58311, 58313, 58315, 58323, 58331, 58339, 58348, 58350, 58369, 58389, 58394, 58404, 58430, 58435, 58443, 58445, 58458, 58475, 58524, 58544, 58550, 58551, 58552, 58561, 58576, 58584, 58597, 58603, 58610, 58615, 58621, 58629, 58653, 58654, 58664, 58678, 58685, 58686, 58690, 58713, 58718, 58719, 58743, 58747, 58753, 58768, 58773, 58775, 58788, 58800, 58813, 58826, 58828, 58834, 58836, 58838, 58842, 58856, 58864, 58866, 58873, 58874, 58888, 58890, 58912, 58916, 58954, 58957, 58975, 58987, 58996, 59006, 59008, 59046, 59063, 59065, 59067, 59089, 59090, 59096, 59101, 59110, 59132, 59142, 59147, 59149, 59155, 59156, 59157, 59158, 59163, 59190, 59195, 59204, 59222, 59224, 59229, 59247, 59262, 59263, 59277, 59282, 59290, 59309, 59327, 59334, 59345, 59352, 59360, 59371, 59374, 59388, 59392, 59404, 59416, 59418, 59429, 59434, 59445, 59460, 59467, 59468, 59482, 59484, 59498, 59519, 59528, 59548, 59558, 59582, 59592, 59612, 59620, 59634, 59638, 59640, 59648, 59655, 59658, 59660, 59666, 59668, 59683, 59691, 59696, 59698, 59705, 59707, 59727, 59728, 59730, 59743, 59750, 59752, 59753, 59772, 59778, 59803, 59809, 59810, 59817, 59821, 59823, 59826, 59829, 59859, 59861, 59865, 59882, 59893, 59897, 59910, 59920, 59921, 59927, 59932, 59934, 59965, 59967, 59969, 59971, 59982, 60008, 60017, 60023, 60033, 60039, 60050, 60054, 60061, 60084, 60090, 60099, 60104, 60109, 60110, 60122, 60134, 60136, 60139, 60149, 60151, 60156, 60163, 60171, 60177, 60178, 60185, 60191, 60195, 60196, 60199, 60203, 60204, 60208, 60220, 60229, 60232, 60237, 60241, 60243, 60265, 60267, 60279, 60286, 60290, 60293, 60305, 60307, 60332, 60352, 60360, 60364, 60408, 60491, 60501, 60514, 60520, 60534, 60548, 60558, 60559, 60576, 60588, 60591, 60594, 60599, 60602, 60618, 60623, 60624, 60637, 60641, 60642, 60656, 60658, 60666, 60668, 60683, 60690, 60691, 60694, 60713, 60720, 60729, 60738, 60739, 60741, 60744, 60745, 60755, 60757, 60758, 60764, 60787, 60795, 60804, 60816, 60817, 60818, 60822, 60835, 60845, 60855, 60868, 60875, 60880, 60884, 60900, 60902, 60905, 60923, 60950, 60956, 60959, 60966, 60977, 60983, 61005, 61006, 61014, 61016, 61018, 61023, 61024, 61028, 61043, 61066, 61070, 61071, 61076, 61077, 61088, 61096, 61114, 61120, 61122, 61133, 61150, 61166, 61172, 61183, 61187, 61190, 61202, 61207, 61210, 61211, 61215, 61219, 61224, 61237, 61238, 61242, 61246, 61247, 61248, 61250, 61251, 61277, 61290, 61291, 61293, 61297, 61298, 61330, 61331, 61336, 61342, 61350, 61373, 61375, 61377, 61380, 61381, 61382, 61384, 61390, 61393, 61399, 61401, 61419, 61421, 61422, 61423, 61449, 61467, 61468, 61478, 61480, 61500, 61549, 61561, 61563, 61585, 61587, 61588, 61601, 61619, 61621, 61628, 61631, 61632, 61658, 61659, 61663, 61665, 61670, 61679, 61683, 61685, 61691, 61700, 61709, 61712, 61720, 61724, 61725, 61737, 61745, 61766, 61767, 61769, 61783, 61788, 61791, 61794, 61803, 61807, 61809, 61824, 61826, 61831, 61846, 61847, 61863, 61874, 61884, 61887, 61902, 61911, 61921, 61925, 61938, 61946, 61952, 61957, 61971, 61989, 61997, 62008, 62015, 62036, 62038, 62045, 62048, 62049, 62058, 62060, 62066, 62104, 62107, 62117, 62122, 62126, 62135, 62145, 62162, 62179, 62180, 62192, 62206, 62207, 62217, 62220, 62226, 62229, 62235, 62244, 62250, 62262, 62265, 62276, 62277, 62283, 62289, 62313, 62321, 62323, 62329, 62350, 62361, 62369, 62372, 62381, 62388, 62400, 62416, 62422, 62428, 62431, 62450, 62459, 62461, 62468, 62479, 62483, 62490, 62502, 62505, 62510, 62548, 62559, 62562, 62572, 62579, 62594, 62602, 62604, 62625, 62629, 62637, 62656, 62662, 62663, 62664, 62667, 62670, 62671, 62675, 62694, 62708, 62718, 62720, 62721, 62722, 62728, 62729, 62745, 62751, 62753, 62760, 62763, 62766, 62772, 62776, 62780, 62802, 62816, 62822, 62840, 62862, 62882, 62891, 62894, 62905, 62907, 62911, 62937, 62947, 62966, 62974, 62981, 62996, 63011, 63019, 63029, 63038, 63053, 63061, 63075, 63087, 63089, 63108, 63171, 63175, 63176, 63180, 63198, 63206, 63215, 63223, 63232, 63263, 63269, 63272, 63286, 63288, 63293, 63305, 63314, 63318, 63329, 63334, 63335, 63341, 63357, 63372, 63376, 63397, 63400, 63416, 63428, 63441, 63457, 63459, 63461, 63472, 63480, 63505, 63510, 63511, 63516, 63525, 63527, 63530, 63536, 63542, 63546, 63548, 63562, 63563, 63569, 63573, 63584, 63597, 63601, 63619, 63628, 63636, 63638, 63654, 63662, 63664, 63675, 63706, 63715, 63719, 63735, 63750, 63760, 63767, 63779, 63796, 63820, 63839, 63841, 63850, 63861, 63865, 63868, 63878, 63888, 63890, 63896, 63908, 63912, 63917, 63918, 63919, 63931, 63941, 63943, 63956, 63963, 63964, 63974, 63988, 64011, 64029, 64031, 64034, 64042, 64053, 64054, 64057, 64065, 64070, 64079, 64085, 64089, 64091, 64094, 64128, 64138, 64139, 64140, 64145, 64147, 64154, 64176, 64193, 64199, 64202, 64204, 64207, 64208, 64220, 64221, 64224, 64226, 64232, 64244, 64245, 64249, 64256, 64264, 64265, 64274, 64276, 64277, 64278, 64279, 64301, 64313, 64314, 64317, 64323, 64348, 64352, 64375, 64387, 64394, 64396, 64431, 64438, 64460, 64472, 64475, 64511, 64532, 64538, 64546, 64547, 64556, 64562, 64563, 64580, 64585, 64601, 64603, 64611, 64624, 64639, 64648, 64651, 64653, 64654, 64660, 64664, 64681, 64685, 64702, 64714, 64718, 64741, 64754, 64756, 64764, 64777, 64779, 64828, 64835, 64845, 64850, 64863, 64874, 64885, 64909, 64934, 64942, 64950, 64967, 64972, 64986, 64989, 64991, 65008, 65011, 65020, 65035, 65046, 65051, 65064, 65070, 65071, 65080, 65087, 65093, 65105, 65119, 65123, 65127, 65144, 65146, 65154, 65158, 65174, 65179, 65181, 65192, 65199, 65215, 65216, 65223, 65227, 65250, 65253, 65254, 65272, 65276, 65290, 65347, 65348, 65351, 65356, 65357, 65358, 65359, 65370, 65380, 65387, 65396, 65406, 65407, 65414, 65416, 65421, 65427, 65428, 65446, 65449, 65466, 65467, 65487, 65501, 65507, 65511, 65513, 65517, 65539, 65546, 65547, 65549, 65574, 65576, 65586, 65595, 65611, 65617, 65629, 65635, 65648, 65666, 65683, 65684, 65689, 65693, 65696, 65707, 65714, 65728, 65729, 65758, 65759, 65794, 65799, 65800, 65801, 65807, 65811, 65819, 65845, 65848, 65855, 65861, 65874, 65885, 65901, 65907, 65912, 65918, 65940, 65948, 65952, 65953, 65954, 65964, 65976, 65977, 65981, 65992, 66003, 66004, 66006, 66018, 66034, 66036, 66053, 66054, 66069, 66072, 66080, 66084, 66091, 66094, 66123, 66131, 66134, 66135, 66156, 66167, 66170, 66177, 66179, 66191, 66201, 66206, 66217, 66222, 66223, 66224, 66235, 66251, 66253, 66257, 66261, 66270, 66271, 66279, 66304, 66314, 66331, 66337, 66343, 66344, 66362, 66367, 66373, 66376, 66377, 66381, 66399, 66405, 66413, 66418, 66421, 66438, 66439, 66444, 66459, 66469, 66480, 66496, 66501, 66525, 66544, 66545, 66555, 66559, 66564, 66598, 66614, 66617, 66622, 66630, 66648, 66650, 66659, 66661, 66676, 66689, 66690, 66700, 66701, 66702, 66707, 66709, 66726, 66727, 66736, 66738, 66782, 66785, 66794, 66795, 66807, 66808, 66810, 66818, 66860, 66863, 66864, 66865, 66872, 66884, 66898, 66905, 66915, 66929, 66931, 66934, 66947, 66949, 66973, 66976, 66980, 66983, 67000, 67005, 67010, 67012, 67018, 67056, 67085, 67098, 67110, 67127, 67128, 67140, 67142, 67159, 67160, 67162, 67163, 67175, 67180, 67184, 67189, 67210, 67228, 67239, 67242, 67246, 67260, 67275, 67277, 67286, 67291, 67303, 67311, 67326, 67327, 67336, 67346, 67362, 67368, 67378, 67390, 67391, 67394, 67403, 67419, 67441, 67443, 67444, 67454, 67470, 67474, 67478, 67482, 67486, 67504, 67507, 67530, 67532, 67540, 67554, 67559, 67560, 67574, 67581, 67583, 67585, 67602, 67636, 67648, 67666, 67682, 67693, 67732, 67737, 67741, 67744, 67746, 67769, 67773, 67782, 67791, 67797, 67807, 67811, 67820, 67821, 67831, 67837, 67844, 67852, 67870, 67879, 67881, 67883, 67888, 67924, 67945, 67946, 67953, 67962, 67966, 67978, 68002, 68015, 68022, 68051, 68064, 68065, 68067, 68072, 68086, 68091, 68094, 68110, 68111, 68123, 68124, 68131, 68135, 68140, 68142, 68144, 68147, 68159, 68175, 68188, 68190, 68209, 68219, 68226, 68229, 68230, 68231, 68237, 68246, 68248, 68252, 68270, 68277, 68290, 68291, 68299, 68301, 68305, 68314, 68338, 68351, 68387, 68399, 68414, 68433, 68451, 68456, 68459, 68466, 68485, 68486, 68488, 68489, 68499, 68510, 68516, 68527, 68528, 68532, 68534, 68535, 68542, 68554, 68570, 68611, 68614, 68623, 68624, 68644, 68651, 68661, 68662, 68676, 68677, 68679, 68688, 68694, 68698, 68728, 68735, 68768, 68770, 68785, 68812, 68843, 68894, 68900, 68905, 68907, 68909, 68915, 68923, 68938, 68941, 68955, 68958, 68960, 68974, 69002, 69007, 69021, 69025, 69041, 69042, 69054, 69059, 69063, 69074, 69082, 69094, 69159, 69163, 69170, 69183, 69195, 69199, 69211, 69212, 69213, 69231, 69234, 69272, 69277, 69285, 69292, 69293, 69297, 69298, 69306, 69309, 69318, 69324, 69326, 69374, 69377, 69382, 69403, 69413, 69416, 69428, 69436, 69450, 69467, 69475, 69492, 69493, 69496, 69515, 69525, 69564, 69565, 69577, 69578, 69579, 69588, 69595, 69609, 69631, 69636, 69638, 69649, 69660, 69679, 69683, 69684, 69693, 69712, 69724, 69730, 69731, 69745, 69747, 69749, 69752, 69773, 69779, 69790, 69793, 69798, 69825, 69826, 69830, 69835, 69842, 69844, 69873, 69878, 69887, 69895, 69896, 69914, 69923, 69942, 69951, 69953, 69959, 69970, 69990, 70002, 70022, 70025, 70037, 70040, 70050, 70057, 70059, 70078, 70102, 70110, 70125, 70130, 70152, 70158, 70159, 70161, 70163, 70167, 70171, 70172, 70174, 70192, 70196, 70197, 70243, 70245, 70250, 70269, 70284, 70287, 70292, 70293, 70298, 70302, 70329, 70331, 70341, 70350, 70355, 70356, 70359, 70374, 70381, 70386, 70391, 70401, 70441, 70443, 70445, 70470, 70474, 70482, 70511, 70521, 70547, 70550, 70551, 70552, 70564, 70570, 70573, 70575, 70579, 70580, 70609, 70620, 70621, 70623, 70625, 70628, 70635, 70646, 70653, 70658, 70671, 70676, 70694, 70718, 70719, 70724, 70736, 70738, 70741, 70752, 70759, 70774, 70776, 70781, 70785, 70793, 70794, 70795, 70805, 70828, 70829, 70833, 70851, 70888, 70906, 70909, 70911, 70916, 70928, 70939, 70943, 70962, 70968, 70975, 70985, 70987, 70989, 70992, 71004, 71012, 71019, 71023, 71037, 71045, 71050, 71065, 71067, 71070, 71072, 71073, 71083, 71103, 71111, 71119, 71137, 71138, 71153, 71183, 71198, 71208, 71238, 71243, 71251, 71253, 71268, 71301, 71303, 71319, 71322, 71393, 71394, 71407, 71408, 71417, 71430, 71433, 71470, 71472, 71476, 71497, 71502, 71523, 71524, 71526, 71550, 71572, 71578, 71585, 71598, 71599, 71607, 71610, 71615, 71617, 71618, 71625, 71632, 71665, 71675, 71691, 71692, 71696, 71705, 71706, 71707, 71711, 71722, 71727, 71745, 71764, 71777, 71803, 71809, 71813, 71814, 71834, 71850, 71858, 71859, 71866, 71876, 71891, 71896, 71899, 71915, 71917, 71920, 71921, 71935, 71946, 71949, 71971, 71981, 71992, 72015, 72017, 72025, 72035, 72038, 72039, 72044, 72057, 72058, 72074, 72079, 72085, 72088, 72099, 72103, 72105, 72125, 72137, 72138, 72145, 72146, 72154, 72172, 72173, 72187, 72200, 72210, 72218, 72227, 72236, 72243, 72264, 72266, 72286, 72292, 72308, 72310, 72322, 72334, 72338, 72341, 72347, 72364, 72371, 72378, 72385, 72388, 72396, 72404, 72409, 72417, 72434, 72435, 72447, 72452, 72480, 72483, 72491, 72501, 72513, 72516, 72530, 72543, 72546, 72554, 72565, 72567, 72569, 72576, 72577, 72581, 72584, 72595, 72598, 72613, 72619, 72622, 72628, 72632, 72649, 72655, 72661, 72688, 72689, 72692, 72693, 72710, 72711, 72715, 72719, 72737, 72763, 72765, 72794, 72799, 72806, 72807, 72808, 72822, 72824, 72826, 72827, 72857, 72864, 72865, 72871, 72876, 72877, 72891, 72899, 72900, 72902, 72906, 72925, 72928, 72934, 72937, 72939, 72967, 72979, 72982, 72983, 72985, 72987, 72994, 73013, 73016, 73043, 73046, 73062, 73066, 73067, 73093, 73103, 73105, 73109, 73118, 73127, 73151, 73158, 73187, 73198, 73199, 73204, 73207, 73212, 73241, 73248, 73253, 73261, 73264, 73280, 73321, 73327, 73331, 73333, 73342, 73345, 73357, 73374, 73379, 73385, 73401, 73404, 73406, 73413, 73416, 73464, 73470, 73503, 73512, 73515, 73518, 73525, 73526, 73530, 73580, 73603, 73636, 73651, 73664, 73666, 73671, 73684, 73686, 73706, 73723, 73724, 73738, 73740, 73743, 73758, 73769, 73770, 73776, 73778, 73794, 73796, 73801, 73803, 73804, 73808, 73867, 73882, 73899, 73903, 73907, 73932, 73933, 73937, 73943, 73948, 73953, 73958, 73969, 73977, 73986, 73987, 74013, 74015, 74024, 74043, 74054, 74058, 74069, 74072, 74104, 74113, 74117, 74147, 74163, 74165, 74182, 74210, 74212, 74222, 74231, 74242, 74250, 74265, 74268, 74269, 74280, 74282, 74285, 74290, 74300, 74312, 74314, 74315, 74321, 74332, 74335, 74344, 74348, 74356, 74362, 74365, 74382, 74398, 74402, 74418, 74439, 74444, 74448, 74465, 74483, 74491, 74494, 74495, 74509, 74525, 74526, 74536, 74543, 74544, 74548, 74551, 74554, 74570, 74610, 74622, 74633, 74637, 74640, 74650, 74656, 74661, 74678, 74685, 74689, 74703, 74707, 74715, 74721, 74726, 74728, 74737, 74744, 74752, 74754, 74755, 74762, 74778, 74789, 74795, 74796, 74809, 74817, 74822, 74827, 74835, 74836, 74847, 74849, 74861, 74870, 74874, 74876, 74878, 74880, 74886, 74898, 74900, 74904, 74930, 74934, 74951, 74954, 74962, 74966, 74988, 74999, 75033, 75045, 75048, 75051, 75057, 75070, 75097, 75112, 75120, 75125, 75126, 75140, 75158, 75168, 75175, 75177, 75191, 75207, 75212, 75214, 75217, 75231, 75252, 75256, 75259, 75269, 75280, 75288, 75289, 75307, 75309, 75326, 75344, 75348, 75351, 75358, 75360, 75367, 75383, 75397, 75398, 75407, 75410, 75416, 75422, 75435, 75439, 75448, 75456, 75457, 75464, 75474, 75480, 75488, 75489, 75497, 75500, 75521, 75524, 75526, 75533, 75546, 75557, 75560, 75569, 75596, 75599, 75604, 75625, 75637, 75642, 75669, 75677, 75688, 75696, 75714, 75723, 75738, 75742, 75745, 75752, 75761, 75763, 75779, 75790, 75793, 75802, 75806, 75811, 75818, 75835, 75836, 75844, 75847, 75848, 75867, 75877, 75889, 75890, 75894, 75910, 75915, 75926, 75930, 75937, 75944, 75955, 75957, 75962, 75976, 75984, 76005, 76017, 76026, 76035, 76038, 76041, 76046, 76049, 76056, 76059, 76065, 76096, 76109, 76110, 76111, 76123, 76126, 76136, 76142, 76149, 76156, 76157, 76159, 76184, 76194, 76214, 76228, 76239, 76251, 76274, 76283, 76287, 76294, 76296, 76313, 76326, 76327, 76348, 76362, 76371, 76372, 76374, 76377, 76384, 76385, 76392, 76399, 76406, 76411, 76421, 76439, 76443, 76446, 76449, 76473, 76478, 76487, 76512, 76525, 76526, 76539, 76549, 76552, 76573, 76591, 76596, 76600, 76601, 76607, 76615, 76632, 76645, 76647, 76655, 76672, 76673, 76677, 76679, 76689, 76695, 76707, 76720, 76725, 76733, 76751, 76764, 76772, 76782, 76784, 76790, 76791, 76817, 76826, 76829, 76845, 76850, 76857, 76864, 76866, 76889, 76893, 76904, 76926, 76941, 76950, 76957, 76973, 76979, 76992, 77001, 77003, 77008, 77018, 77021, 77023, 77027, 77035, 77056, 77060, 77076, 77102, 77103, 77107, 77111, 77136, 77152, 77154, 77157, 77162, 77166, 77172, 77179, 77189, 77192, 77196, 77210, 77211, 77214, 77217, 77231, 77232, 77236, 77241, 77245, 77247, 77262, 77271, 77278, 77279, 77280, 77281, 77301, 77315, 77319, 77333, 77337, 77338, 77339, 77341, 77345, 77354, 77357, 77360, 77362, 77380, 77384, 77404, 77430, 77441, 77443, 77447, 77475, 77476, 77478, 77498, 77499, 77502, 77504, 77509, 77523, 77526, 77528, 77532, 77535, 77540, 77585, 77586, 77590, 77601, 77606, 77622, 77623, 77637, 77638, 77639, 77646, 77658, 77664, 77671, 77678, 77684, 77685, 77687, 77688, 77743, 77745, 77746, 77755, 77760, 77762, 77772, 77774, 77784, 77790, 77803, 77805, 77806, 77817, 77819, 77821, 77826, 77835, 77842, 77864, 77868, 77876, 77878, 77892, 77902, 77903, 77925, 77970, 77976, 77980, 77986, 77991, 77998, 78008, 78010, 78015, 78026, 78047, 78049, 78050, 78056, 78065, 78070, 78074, 78076, 78077, 78083, 78100, 78113, 78121, 78132, 78142, 78146, 78150, 78152, 78179, 78182, 78184, 78188, 78211, 78216, 78239, 78254, 78255, 78266, 78268, 78277, 78294, 78295, 78316, 78320, 78324, 78332, 78342, 78362, 78368, 78370, 78382, 78391, 78396, 78407, 78414, 78431, 78462, 78465, 78470, 78471, 78486, 78536, 78541, 78550, 78583, 78585, 78587, 78592, 78597, 78619, 78630, 78642, 78648, 78660, 78666, 78684, 78690, 78697, 78698, 78708, 78721, 78725, 78734, 78737, 78754, 78755, 78780, 78792, 78804, 78806, 78808, 78812, 78825, 78826, 78827, 78832, 78866, 78868, 78869, 78875, 78881, 78893, 78894, 78910, 78913, 78934, 78939, 78944, 78947, 78963, 78964, 79018, 79021, 79026, 79033, 79035, 79049, 79053, 79054, 79065, 79081, 79086, 79095, 79097, 79121, 79141, 79143, 79149, 79154, 79161, 79189, 79196, 79200, 79219, 79249, 79250, 79257, 79262, 79265, 79268, 79274, 79278, 79291, 79298, 79302, 79317, 79319, 79322, 79341, 79352, 79353, 79356, 79363, 79366, 79381, 79384, 79399, 79400, 79407, 79433, 79443, 79452, 79462, 79470, 79481, 79489, 79500, 79508, 79519, 79526, 79527, 79528, 79532, 79539, 79553, 79583, 79592, 79601, 79614, 79626, 79628, 79631, 79634, 79643, 79649, 79657, 79691, 79707, 79709, 79711, 79727, 79730, 79733, 79750, 79754, 79759, 79762, 79767, 79778, 79787, 79794, 79799, 79819, 79828, 79829, 79879, 79898, 79900, 79911, 79917, 79927, 79932, 79939, 79943, 79970, 79979, 79984, 79987, 80004, 80021, 80025, 80027, 80029, 80052, 80058, 80076, 80084, 80095, 80104, 80111, 80117, 80141, 80180, 80182, 80184, 80190, 80191, 80196, 80201, 80202, 80203, 80211, 80217, 80218, 80220, 80229, 80231, 80233, 80238, 80247, 80256, 80268, 80282, 80298, 80358, 80359, 80360, 80368, 80369, 80370, 80381, 80386, 80388, 80394, 80424, 80436, 80444, 80464, 80471, 80474, 80482, 80495, 80507, 80513, 80514, 80518, 80525, 80527, 80534, 80546, 80547, 80553, 80555, 80569, 80574, 80586, 80597, 80604, 80613, 80616, 80621, 80623, 80629, 80633, 80646, 80648, 80652, 80658, 80665, 80684, 80694, 80700, 80710, 80717, 80719, 80737, 80748, 80751, 80755, 80757, 80770, 80772, 80798, 80802, 80806, 80859, 80870, 80890, 80898, 80951, 80955, 80958, 80959, 80977, 80983, 80986, 81012, 81018, 81019, 81034, 81038, 81040, 81046, 81049, 81056, 81058, 81073, 81077, 81090, 81097, 81099, 81100, 81118, 81158, 81180, 81198, 81208, 81210, 81221, 81259, 81268, 81286, 81299, 81319, 81320, 81330, 81332, 81355, 81357, 81370, 81371, 81378, 81380, 81388, 81400, 81412, 81422, 81441, 81447, 81451, 81473, 81479, 81484, 81488, 81492, 81495, 81500, 81503, 81514, 81519, 81535, 81545, 81547, 81560, 81588, 81591, 81611, 81612, 81623, 81640, 81643, 81645, 81662, 81664, 81674, 81679, 81682, 81691, 81700, 81702, 81720, 81721, 81725, 81734, 81736, 81743, 81747, 81775, 81778, 81784, 81799, 81808, 81811, 81815, 81822, 81825, 81829, 81830, 81839, 81849, 81857, 81867, 81884, 81891, 81897, 81905, 81911, 81912, 81916, 81933, 81938, 81942, 81946, 81952, 81979, 81990, 82010, 82011, 82035, 82045, 82079, 82082, 82114, 82126, 82133, 82157, 82179, 82185, 82195, 82227, 82245, 82251, 82253, 82257, 82258, 82268, 82269, 82274, 82275, 82278, 82281, 82292, 82328, 82331, 82340, 82360, 82363, 82384, 82420, 82421, 82446, 82452, 82455, 82457, 82466, 82472, 82477, 82484, 82489, 82492, 82511, 82528, 82542, 82554, 82560, 82581, 82588, 82593, 82595, 82599, 82602, 82615, 82617, 82619, 82622, 82642, 82669, 82670, 82679, 82683, 82686, 82693, 82694, 82696, 82726, 82729, 82736, 82746, 82747, 82754, 82774, 82777, 82791, 82808, 82812, 82815, 82824, 82826, 82830, 82832, 82838, 82850, 82857, 82879, 82881, 82899, 82908, 82911, 82916, 82917, 82932, 82937, 82950, 82958, 82959, 82970, 82973, 82995, 83012, 83013, 83017, 83021, 83022, 83041, 83057, 83067, 83069, 83072, 83077, 83082, 83095, 83108, 83119, 83136, 83141, 83146, 83165, 83174, 83179, 83181, 83184, 83188, 83205, 83220, 83232, 83237, 83239, 83243, 83247, 83248, 83254, 83263, 83275, 83279, 83300, 83303, 83324, 83335, 83345, 83355, 83361, 83362, 83363, 83367, 83378, 83410, 83417, 83430, 83439, 83445, 83461, 83463, 83468, 83481, 83482, 83499, 83506, 83509, 83518, 83519, 83521, 83530, 83536, 83537, 83542, 83551, 83556, 83562, 83572, 83575, 83586, 83589, 83591, 83593, 83600, 83632, 83651, 83660, 83666, 83673, 83677, 83693, 83695, 83697, 83712, 83716, 83718, 83730, 83737, 83740, 83741, 83743, 83744, 83746, 83800, 83801, 83803, 83811, 83835, 83849, 83862, 83863, 83864, 83872, 83889, 83892, 83894, 83902, 83910, 83913, 83927, 83932, 83955, 84007, 84019, 84024, 84041, 84081, 84095, 84124, 84133, 84142, 84144, 84149, 84155, 84157, 84176, 84189, 84191, 84201, 84204, 84207, 84210, 84211, 84214, 84215, 84216, 84217, 84262, 84264, 84291, 84305, 84312, 84316, 84337, 84357, 84364, 84379, 84386, 84392, 84414, 84418, 84446, 84453, 84456, 84459, 84463, 84469, 84472, 84482, 84485, 84486, 84493, 84494, 84497, 84500, 84503, 84508, 84515, 84522, 84523, 84534, 84562, 84590, 84598, 84616, 84617, 84634, 84654, 84655, 84658, 84662, 84678, 84679, 84687, 84695, 84697, 84702, 84704, 84720, 84727], "validation": [207, 376, 691, 804, 1013, 1019, 1951, 2075, 2137, 2164, 2397, 2465, 2500, 2680, 2751, 2818, 3405, 3422, 3467, 3471, 3521, 3724, 3748, 3824, 4005, 4056, 4120, 4121, 4284, 4330, 4609, 4848, 4897, 4927, 4982, 5001, 5076, 5188, 5437, 5460, 5554, 5607, 5615, 5748, 5771, 5792, 5806, 5877, 6010, 6019, 6025, 6324, 6331, 6584, 6692, 6811, 6910, 6934, 6939, 6966, 7054, 7255, 7340, 7440, 7536, 7598, 7931, 7941, 8019, 8055, 8194, 8318, 8356, 8371, 8417, 8603, 8701, 8709, 8732, 8892, 8905, 8950, 8966, 9048, 9091, 9221, 9554, 9674, 9743, 9779, 9819, 9834, 9838, 9952, 9995, 10133, 10181, 10202, 10239, 10331, 10435, 10463, 10472, 10512, 10626, 10723, 10781, 10818, 10829, 10832, 10965, 11132, 11193, 11243, 11276, 11331, 11365, 11448, 11656, 11701, 11990, 12111, 12170, 12176, 12181, 12381, 12408, 12412, 12553, 12630, 12827, 12913, 13154, 13163, 13175, 13199, 13435, 13630, 13631, 13659, 13751, 13764, 13773, 14065, 14345, 14380, 14490, 14606, 14731, 14735, 14776, 15078, 15106, 15128, 15156, 15206, 15271, 15310, 15438, 15509, 15655, 15951, 16136, 16198, 16240, 16250, 16303, 16501, 16543, 16569, 16634, 16654, 16733, 16768, 16842, 16981, 17054, 17135, 17146, 17166, 17242, 17272, 17288, 17410, 17509, 17721, 17909, 18016, 18025, 18221, 18265, 18270, 18281, 18376, 18429, 18440, 18561, 18783, 18841, 18960, 19021, 19063, 19210, 19394, 19447, 19567, 19594, 19612, 19695, 19852, 19963, 20027, 20040, 20259, 20289, 20310, 20365, 20467, 20475, 20510, 20539, 20568, 20653, 20892, 20893, 20905, 21068, 21233, 21277, 21495, 21507, 21532, 21854, 21969, 22116, 22316, 22384, 22727, 22815, 22906, 22919, 23031, 23112, 23132, 23168, 23246, 23327, 23517, 23932, 23953, 23958, 24011, 24086, 24159, 24201, 24700, 24800, 24866, 24988, 25196, 25368, 25486, 25745, 25760, 26067, 26197, 26493, 26523, 26587, 26696, 26777, 27161, 27265, 27526, 28014, 28069, 28183, 28265, 28457, 28567, 28581, 28638, 28649, 28686, 28859, 29288, 29402, 29423, 29426, 29772, 29773, 29809, 29930, 30019, 30068, 30323, 30537, 30911, 31038, 31141, 31269, 31298, 31313, 31335, 31396, 31498, 31512, 31551, 31619, 31712, 31894, 31967, 32074, 32254, 32288, 32460, 32834, 33079, 33129, 33211, 33217, 33566, 33611, 33775, 33856, 33908, 33914, 34482, 34675, 35049, 35146, 35322, 35416, 36027, 36115, 36165, 36172, 36178, 36331, 36345, 36361, 36404, 36515, 36703, 36876, 36892, 37033, 37294, 37437, 37479, 37577, 37589, 37685, 37702, 37827, 37917, 37943, 38009, 38069, 38235, 38243, 38263, 38350, 38462, 38737, 38776, 38920, 39014, 39033, 39230, 39546, 39697, 39731, 39759, 39827, 39901, 40065, 40094, 40347, 40588, 40644, 40658, 40696, 40718, 40732, 40809, 40979, 41083, 41092, 41157, 41323, 41474, 41542, 41656, 41872, 41993, 42210, 42252, 42326, 42397, 42472, 42546, 42629, 42681, 42733, 43087, 43098, 43113, 43160, 43170, 43309, 43605, 43680, 43755, 43793, 43813, 43967, 44053, 44246, 44360, 44456, 44480, 44532, 44607, 44649, 44789, 44902, 45122, 45263, 45299, 45346, 45461, 45644, 45898, 45927, 45958, 45997, 46118, 46127, 46185, 46197, 46236, 46284, 46318, 46352, 46449, 46522, 46893, 46909, 47062, 47117, 47369, 47407, 47486, 47529, 47533, 47559, 47572, 47630, 47640, 47722, 47802, 47855, 48120, 48203, 48254, 48255, 48284, 48321, 48345, 48440, 48461, 48610, 48673, 48730, 48736, 48762, 48865, 48946, 48973, 49031, 49116, 49164, 49244, 49408, 49472, 49477, 49932, 50022, 50145, 50273, 50485, 50709, 50758, 50769, 50800, 50813, 50868, 51026, 51257, 51451, 51881, 51946, 52041, 52122, 52250, 52263, 52266, 52275, 52345, 52514, 52568, 52613, 52724, 53088, 53168, 53305, 53320, 53571, 53596, 53657, 53748, 53926, 53928, 53935, 54315, 54359, 54450, 54452, 54592, 54666, 54668, 54811, 54936, 54995, 55088, 55089, 55235, 55487, 55529, 55549, 55628, 55690, 55894, 55916, 55981, 56288, 56305, 56312, 56393, 56414, 56430, 56483, 56489, 56646, 56744, 57045, 57141, 57216, 57281, 57314, 57382, 57409, 57420, 57473, 57636, 57677, 57680, 57790, 57985, 58008, 58124, 58159, 58183, 58228, 58233, 58266, 58332, 58385, 58488, 58534, 58646, 58798, 58841, 58880, 58883, 58936, 58991, 59015, 59023, 59094, 59174, 59230, 59276, 59289, 59294, 59486, 59491, 59504, 59505, 59514, 59647, 59959, 60034, 60082, 60300, 60346, 60556, 60563, 60695, 60859, 60901, 60948, 61107, 61319, 61517, 61530, 61543, 61677, 61983, 62109, 62223, 62403, 62485, 62788, 62805, 62847, 63057, 63266, 63366, 63720, 63802, 63849, 64205, 64327, 64447, 64638, 64820, 64832, 65110, 65263, 65281, 65349, 65481, 65584, 65644, 65787, 65870, 65944, 65983, 65988, 66112, 66205, 66340, 66539, 66574, 66575, 66584, 66613, 66621, 66728, 66764, 66921, 66990, 67237, 67337, 67359, 67374, 67643, 67644, 67717, 67909, 67960, 68007, 68114, 68164, 68457, 68632, 68730, 68795, 68809, 69027, 69075, 69079, 69122, 69140, 69194, 69265, 69307, 69531, 69666, 69676, 69682, 69945, 70065, 70136, 70416, 70615, 70910, 70966, 70969, 71233, 71348, 71486, 71539, 71674, 71687, 71747, 72054, 72114, 72275, 72386, 72449, 72453, 72496, 72820, 72837, 72845, 72860, 72889, 73122, 73136, 73185, 73350, 73454, 73481, 73751, 73835, 73847, 73848, 74112, 74127, 74229, 74240, 74473, 74597, 74644, 75021, 75053, 75187, 75250, 75268, 75418, 75487, 75511, 75616, 75634, 75725, 75861, 76011, 76211, 76322, 76366, 76530, 76641, 76990, 77127, 77138, 77250, 77310, 77451, 77453, 77797, 77848, 78025, 78094, 78149, 78273, 78394, 78425, 78488, 78542, 78555, 78574, 78639, 78829, 78970, 79227, 79234, 79256, 79267, 79529, 79584, 79591, 79629, 79849, 79915, 80009, 80086, 80101, 80209, 80245, 80311, 80481, 80491, 80705, 80820, 80868, 80873, 80895, 80908, 80940, 80981, 81273, 81298, 81309, 81311, 81789, 81932, 82137, 82158, 82165, 82193, 82230, 82520, 82527, 82628, 82667, 83018, 83086, 83134, 83163, 83314, 83327, 83334, 83580, 83612, 83626, 83661, 83905, 83930, 84096, 84101, 84164, 84167, 84249, 84359, 84384, 84420, 84454, 84483, 84539, 84589, 84602, 84630, 84676, 84696], "test": [43, 62, 115, 176, 226, 268, 277, 331, 355, 566, 598, 651, 714, 783, 811, 941, 949, 985, 1032, 1157, 1236, 1278, 1391, 1418, 1442, 1510, 1578, 1781, 1980, 1988, 1993, 2078, 2118, 2144, 2235, 2402, 2479, 2520, 2544, 2545, 2613, 2631, 2699, 2736, 2745, 2795, 2906, 2922, 2929, 2936, 3020, 3135, 3172, 3213, 3246, 3306, 3367, 3608, 3619, 3676, 3712, 3827, 3870, 3940, 3974, 3985, 3987, 4013, 4017, 4089, 4159, 4309, 4326, 4368, 4401, 4502, 4533, 4549, 4570, 4661, 4855, 4957, 4962, 4968, 4994, 5015, 5071, 5099, 5141, 5276, 5327, 5329, 5357, 5495, 5513, 5587, 5614, 5688, 5740, 5761, 5781, 5858, 5916, 5947, 5984, 6113, 6238, 6291, 6315, 6423, 6516, 6637, 6691, 6744, 6754, 6762, 6783, 6801, 6829, 6844, 6856, 6882, 6895, 6991, 7005, 7022, 7042, 7157, 7185, 7232, 7298, 7320, 7395, 7424, 7439, 7442, 7449, 7515, 7649, 7778, 7781, 7794, 7887, 7914, 7936, 7948, 7960, 8010, 8080, 8089, 8283, 8391, 8424, 8464, 8467, 8565, 8593, 8710, 8929, 8946, 8951, 9002, 9038, 9051, 9141, 9198, 9228, 9249, 9253, 9263, 9291, 9347, 9587, 9615, 9631, 9701, 9754, 9757, 9830, 9877, 9901, 9940, 9953, 9957, 10005, 10092, 10128, 10155, 10167, 10188, 10273, 10372, 10476, 10554, 10569, 10573, 10607, 10699, 10719, 10726, 10805, 10837, 10988, 11009, 11038, 11188, 11250, 11274, 11292, 11293, 11311, 11345, 11370, 11378, 11401, 11429, 11557, 11646, 11753, 11781, 11845, 11879, 11882, 11914, 11923, 12133, 12207, 12234, 12247, 12333, 12374, 12375, 12377, 12404, 12464, 12475, 12491, 12557, 12770, 12774, 12786, 12819, 12884, 12896, 12909, 12982, 13070, 13192, 13275, 13297, 13439, 13443, 13541, 13549, 13597, 13640, 13768, 13828, 13849, 13944, 13991, 14072, 14226, 14247, 14295, 14338, 14378, 14384, 14395, 14556, 14635, 14676, 14704, 14742, 14745, 14760, 14855, 14897, 14914, 14938, 14997, 14999, 15093, 15147, 15215, 15235, 15251, 15272, 15357, 15362, 15406, 15437, 15515, 15594, 15597, 15687, 15700, 15775, 15809, 15902, 15911, 15947, 15954, 16028, 16068, 16073, 16214, 16386, 16415, 16523, 16789, 16813, 16918, 16929, 16964, 16988, 17014, 17109, 17248, 17255, 17327, 17333, 17385, 17536, 17559, 17568, 17745, 17826, 17907, 17971, 18136, 18238, 18345, 18359, 18364, 18374, 18397, 18419, 18459, 18500, 18504, 18527, 18568, 18601, 18651, 18746, 18777, 19039, 19123, 19176, 19233, 19240, 19261, 19268, 19282, 19306, 19413, 19457, 19706, 19762, 20016, 20122, 20307, 20355, 20367, 20406, 20446, 20454, 20516, 20530, 20559, 20612, 20666, 20711, 20716, 20734, 20760, 20863, 20973, 21100, 21130, 21186, 21206, 21220, 21247, 21256, 21270, 21330, 21349, 21358, 21459, 21542, 21580, 21605, 21702, 21738, 21744, 21760, 21835, 21841, 22021, 22033, 22066, 22166, 22171, 22184, 22278, 22279, 22318, 22380, 22399, 22468, 22513, 22528, 22585, 22598, 22624, 22691, 22716, 22725, 22754, 22861, 22892, 22943, 23011, 23073, 23083, 23098, 23141, 23144, 23157, 23200, 23248, 23301, 23303, 23353, 23481, 23494, 23520, 23558, 23567, 23583, 23607, 23613, 23667, 23670, 23673, 23725, 23787, 23824, 23890, 23894, 23951, 23977, 23980, 24023, 24079, 24112, 24138, 24160, 24186, 24203, 24246, 24272, 24273, 24354, 24369, 24374, 24527, 24598, 24650, 24742, 24819, 24953, 25030, 25129, 25187, 25226, 25304, 25327, 25393, 25460, 25523, 25570, 25571, 25574, 25577, 25583, 25609, 25622, 25624, 25639, 25662, 25676, 25677, 25795, 25798, 25811, 25823, 25834, 25886, 25940, 26046, 26094, 26106, 26177, 26246, 26293, 26302, 26316, 26349, 26391, 26472, 26487, 26498, 26538, 26651, 26652, 26671, 26691, 26741, 26787, 26798, 26824, 26826, 26830, 26849, 26856, 26861, 26889, 26905, 26962, 27149, 27235, 27246, 27257, 27323, 27372, 27435, 27457, 27497, 27548, 27617, 27656, 27681, 27719, 27724, 27734, 27765, 27829, 28003, 28066, 28099, 28203, 28219, 28235, 28317, 28397, 28399, 28407, 28459, 28478, 28512, 28559, 28623, 28753, 28762, 28766, 28832, 28857, 28863, 28930, 28968, 29046, 29114, 29162, 29179, 29206, 29357, 29359, 29382, 29509, 29524, 29610, 29654, 29757, 29768, 29825, 29894, 29937, 30062, 30093, 30124, 30214, 30239, 30416, 30427, 30449, 30477, 30520, 30532, 30617, 30624, 30691, 30708, 30735, 30805, 30921, 31092, 31239, 31334, 31384, 31390, 31391, 31484, 31579, 31595, 31607, 31677, 31791, 31845, 31860, 31919, 32022, 32053, 32191, 32231, 32260, 32297, 32305, 32337, 32418, 32435, 32455, 32533, 32556, 32712, 32773, 32794, 32817, 32835, 32914, 32967, 33008, 33024, 33068, 33120, 33143, 33150, 33160, 33195, 33235, 33387, 33469, 33491, 33592, 33624, 33633, 33773, 33841, 33936, 33988, 34016, 34133, 34191, 34195, 34239, 34353, 34416, 34499, 34733, 34786, 34826, 34829, 35001, 35071, 35084, 35106, 35128, 35207, 35315, 35354, 35368, 35372, 35447, 35453, 35710, 35739, 35752, 35765, 35791, 35846, 35884, 35971, 36156, 36195, 36281, 36328, 36500, 36625, 36667, 36668, 36681, 36684, 36685, 36710, 36746, 36754, 36762, 36786, 36846, 36864, 36874, 36893, 36980, 37024, 37029, 37076, 37111, 37174, 37181, 37220, 37295, 37547, 37602, 37687, 38012, 38068, 38167, 38183, 38184, 38271, 38356, 38439, 38494, 38532, 38548, 38553, 38634, 38653, 38685, 38696, 38805, 38820, 38835, 38940, 38976, 38990, 39150, 39180, 39182, 39183, 39239, 39288, 39325, 39362, 39372, 39510, 39585, 39620, 39669, 39711, 39732, 39817, 39921, 39959, 39965, 40008, 40009, 40036, 40069, 40337, 40358, 40391, 40434, 40440, 40473, 40567, 40598, 40602, 40630, 40670, 40706, 40720, 40763, 40832, 40899, 40901, 40937, 41049, 41138, 41208, 41212, 41227, 41249, 41343, 41356, 41423, 41459, 41486, 41565, 41712, 41761, 41820, 41834, 41916, 41923, 41946, 41950, 41973, 42013, 42155, 42188, 42189, 42194, 42285, 42356, 42400, 42412, 42447, 42542, 42547, 42552, 42573, 42583, 42588, 42622, 42640, 42684, 42736, 42965, 43023, 43177, 43214, 43275, 43357, 43395, 43424, 43488, 43687, 43715, 43716, 43745, 43810, 43816, 43819, 43834, 43878, 43885, 43995, 44024, 44029, 44044, 44078, 44191, 44305, 44319, 44344, 44402, 44488, 44520, 44546, 44580, 44608, 44643, 44832, 44840, 44848, 44860, 44871, 44875, 44899, 44969, 45090, 45129, 45201, 45231, 45262, 45328, 45336, 45381, 45395, 45423, 45442, 45460, 45540, 45618, 45656, 45684, 45702, 45710, 45744, 45851, 45905, 45917, 45995, 46064, 46090, 46170, 46270, 46524, 46569, 46625, 46700, 46733, 46805, 46815, 46842, 46853, 46927, 46946, 46947, 46982, 47038, 47119, 47161, 47188, 47219, 47225, 47229, 47331, 47445, 47449, 47532, 47534, 47702, 47725, 47776, 47784, 47881, 47895, 48008, 48100, 48171, 48184, 48261, 48392, 48413, 48545, 48595, 48652, 48846, 48899, 48905, 48952, 48961, 49089, 49148, 49149, 49153, 49159, 49195, 49263, 49310, 49321, 49493, 49516, 49789, 49836, 49931, 50053, 50160, 50191, 50238, 50286, 50375, 50526, 50578, 50671, 50699, 50872, 50997, 51160, 51286, 51300, 51425, 51467, 51479, 51498, 51500, 51502, 51652, 51654, 51739, 51798, 51799, 51975, 52024, 52025, 52031, 52056, 52095, 52160, 52204, 52207, 52232, 52247, 52283, 52308, 52315, 52321, 52325, 52326, 52376, 52390, 52432, 52475, 52601, 52641, 52643, 52706, 52723, 52837, 52871, 52916, 52937, 52982, 53028, 53154, 53158, 53295, 53343, 53370, 53398, 53405, 53445, 53480, 53517, 53543, 53551, 53556, 53576, 53597, 53675, 53684, 53701, 53744, 53802, 53818, 53863, 53913, 53915, 54023, 54058, 54068, 54123, 54206, 54295, 54392, 54428, 54439, 54443, 54599, 54621, 54763, 54858, 54926, 55055, 55075, 55203, 55250, 55261, 55305, 55320, 55443, 55508, 55524, 55736, 55814, 55855, 55897, 55899, 55942, 55945, 55976, 56027, 56062, 56192, 56211, 56236, 56271, 56311, 56319, 56389, 56492, 56581, 56643, 56683, 56694, 56740, 56754, 56786, 56789, 56870, 57038, 57055, 57106, 57181, 57192, 57201, 57270, 57309, 57325, 57370, 57401, 57548, 57568, 57607, 57739, 57819, 57879, 57920, 57964, 57983, 57986, 58013, 58029, 58050, 58182, 58287, 58329, 58429, 58600, 58723, 58769, 58770, 58889, 58985, 59020, 59027, 59034, 59052, 59058, 59123, 59287, 59304, 59347, 59362, 59447, 59547, 59579, 59680, 60005, 60053, 60226, 60326, 60336, 60427, 60442, 60466, 60483, 60578, 60605, 60620, 60632, 60655, 60753, 60806, 60865, 60882, 61074, 61105, 61116, 61130, 61253, 61256, 61295, 61311, 61312, 61397, 61440, 61529, 61556, 61582, 61692, 61697, 61886, 61969, 62021, 62024, 62071, 62084, 62138, 62146, 62249, 62299, 62351, 62370, 62374, 62438, 62486, 62487, 62529, 62591, 62674, 62738, 62868, 62877, 63135, 63178, 63200, 63213, 63345, 63354, 63365, 63393, 63571, 63576, 63606, 63645, 63744, 63768, 63798, 63812, 63946, 64016, 64101, 64115, 64170, 64227, 64351, 64373, 64422, 64489, 64503, 64525, 64577, 64596, 64606, 64645, 64731, 64761, 64797, 64867, 64908, 64949, 64980, 65002, 65010, 65029, 65102, 65299, 65302, 65304, 65375, 65404, 65433, 65491, 65504, 65555, 65590, 65606, 65623, 65627, 65679, 65788, 65846, 65852, 65888, 65892, 65946, 66040, 66045, 66048, 66085, 66115, 66155, 66229, 66272, 66335, 66338, 66363, 66378, 66506, 66546, 66609, 66631, 66675, 66758, 66768, 66896, 66940, 67003, 67014, 67036, 67168, 67245, 67281, 67335, 67342, 67354, 67413, 67464, 67529, 67584, 67658, 67679, 67712, 67715, 67776, 67859, 67951, 68032, 68055, 68061, 68198, 68243, 68300, 68326, 68392, 68394, 68411, 68423, 68474, 68482, 68559, 68573, 68595, 68727, 68734, 68740, 68779, 68874, 68882, 68927, 68964, 69018, 69080, 69093, 69116, 69181, 69228, 69260, 69304, 69317, 69379, 69459, 69619, 69713, 69755, 69868, 69943, 69957, 69989, 70080, 70106, 70137, 70138, 70141, 70166, 70217, 70232, 70251, 70259, 70283, 70337, 70352, 70380, 70442, 70472, 70558, 70636, 70640, 70673, 70802, 70986, 71014, 71058, 71062, 71115, 71131, 71188, 71228, 71306, 71345, 71466, 71528, 71536, 71573, 71629, 71650, 71694, 71702, 71749, 71885, 72065, 72110, 72140, 72179, 72220, 72263, 72342, 72429, 72431, 72461, 72481, 72553, 72579, 72597, 72614, 72636, 72683, 72686, 72720, 72725, 72739, 72812, 72817, 72819, 72859, 72956, 72986, 72992, 73004, 73064, 73073, 73087, 73159, 73218, 73226, 73283, 73291, 73382, 73383, 73414, 73428, 73473, 73487, 73642, 73702, 73760, 73820, 73840, 73852, 73863, 73870, 73931, 74001, 74022, 74045, 74065, 74080, 74105, 74135, 74221, 74249, 74320, 74339, 74340, 74434, 74490, 74578, 74594, 74603, 74712, 74733, 74812, 74909, 74914, 74927, 74953, 74970, 75076, 75145, 75156, 75159, 75180, 75320, 75328, 75370, 75378, 75393, 75501, 75582, 75605, 75661, 75679, 75731, 75780, 75823, 75873, 75922, 75980, 75987, 76066, 76090, 76100, 76179, 76314, 76361, 76364, 76468, 76491, 76500, 76510, 76534, 76594, 76738, 76789, 76795, 76797, 76803, 76925, 76946, 77006, 77043, 77085, 77086, 77135, 77194, 77233, 77260, 77412, 77426, 77435, 77467, 77470, 77514, 77549, 77573, 77629, 77630, 77633, 77648, 77679, 77680, 77710, 77749, 77763, 77866, 77965, 77966, 77997, 78016, 78171, 78226, 78280, 78355, 78399, 78532, 78596, 78688, 78689, 78703, 78753, 78783, 78805, 78819, 78823, 78957, 78991, 79007, 79060, 79128, 79285, 79332, 79335, 79380, 79413, 79422, 79432, 79503, 79522, 79540, 79572, 79585, 79646, 79647, 79677, 79708, 79715, 79728, 79788, 79802, 79806, 79913, 79996, 80063, 80114, 80167, 80241, 80259, 80356, 80418, 80533, 80550, 80585, 80618, 80727, 80771, 80796, 80822, 80878, 80889, 80926, 80973, 81000, 81024, 81028, 81078, 81104, 81154, 81161, 81237, 81308, 81356, 81360, 81407, 81490, 81561, 81616, 81635, 81637, 81675, 81676, 81694, 81699, 81709, 81716, 81739, 81758, 81767, 81782, 81785, 81786, 81793, 81949, 81961, 81964, 81984, 82003, 82006, 82025, 82037, 82051, 82136, 82160, 82271, 82337, 82341, 82359, 82367, 82416, 82424, 82447, 82453, 82486, 82548, 82562, 82565, 82576, 82585, 82690, 82711, 82727, 82811, 82846, 82852, 82894, 82966, 82975, 83020, 83061, 83065, 83091, 83129, 83143, 83172, 83189, 83190, 83231, 83246, 83299, 83353, 83432, 83447, 83493, 83500, 83554, 83555, 83608, 83699, 83856, 83878, 83919, 83999, 84012, 84037, 84048, 84080, 84104, 84432, 84477, 84478, 84552, 84553, 84557, 84609, 84628, 84667]} diff --git a/docs/figures/w160_train_val.png b/docs/figures/w160_train_val.png new file mode 100644 index 00000000..da23c3c7 Binary files /dev/null and b/docs/figures/w160_train_val.png differ diff --git a/docs/figures/width_ablation_nmae.png b/docs/figures/width_ablation_nmae.png new file mode 100644 index 00000000..87ebe1ab Binary files /dev/null and b/docs/figures/width_ablation_nmae.png differ diff --git a/docs/lr_wd_width_sweep.md b/docs/lr_wd_width_sweep.md new file mode 100644 index 00000000..3adbe4b9 --- /dev/null +++ b/docs/lr_wd_width_sweep.md @@ -0,0 +1,234 @@ +# LR / weight-decay sweep across width (W32 / W64 / W96) + +Status: stages A+B COMPLETE (launched 2026-07-30/31, finished 2026-07-31 — +16 + 15 trials, all succeeded; results below). Stage C not launched. + +## Results (see W&B `mp-gga-ggau-lrwd`; aggregate with scripts/review_lrwd_sweep.py) + +Best val NMAE% on the 8-epoch/12K proxy. Noise bar from the three verbatim +anchor reruns: **±0.04–0.07 points** (1.914→1.968, 1.592→1.665, 1.524→1.566). + +| lr | W32 | W64 | W96 | +|---------|-------|-------|-------| +| 2.5e-4 | 2.330 | 1.998 | 1.943 | +| 5e-4 | 2.133 | 1.841 | 1.836 | +| 1e-3 | 1.935 | 1.708 | 1.757 | +| 2e-3 | **1.914** | 1.759 | **1.524** | +| 4e-3 | 1.954 | **1.592** | 1.562 | +| 8e-3 | — | 1.580 | — | + +| wd @ lr* (AdamW) | W32 | W64 | W96 | +|---------|-------|-------|-------| +| 0 (+rerun) | 1.914 / 1.968 | 1.592 / 1.665 | 1.524 / 1.566 | +| 1e-4 | 1.968 | 1.631 | 1.535 | +| 1e-3 | 1.995 | 1.610 | 1.640 | +| 1e-2 | 1.970 | 1.654 | 1.549 | +| cross (lr*/2, 1e-3) | 1.904 | 1.638 | 1.731 | + +Conclusions: + +1. **lr\* does not shrink with width** — every width sits in a flat 2–4e-3 + basin (differences at/inside the noise bar). The production lr=1e-3 is + suboptimal at all widths, and the penalty grows with width: ~1% relative + at W32, ~7% at W64, ~13% at W96. Extrapolation to W128/W160: use 2e-3. +2. **Weight decay is neutral across 1e-4–1e-2 at every width** — even on + this subset, where overfitting bites ~9× earlier than at full scale, so + at 111K it has even less room to help. Recommend keeping **wd=0** + (continuity with all incumbents); wd up to 1e-2 is demonstrably safe if + ever wanted for other reasons. (Only outlier: W96 @ 1e-3 = 1.640, + marginally past the bar; with 1e-2 neutral on both sides it reads as an + unlucky draw, not a trend.) +3. W64's stage-A curiosities dissolved: the 8e-3 "win" (1.580 vs 1.592) and + the 2e-3 dip (1.759; cross term at same lr with wd landed 1.638) are + both within run-to-run noise. +4. Width ordering at tuned recipes is unchanged: W96 1.524 < W64 1.592 < + W32 1.914. + +**Stage C recipes**: W32 (2e-3, 0), W64 (4e-3, 0), W96 (2e-3, 0) — with +2e-3 defensible everywhere given the flat basin. + +## Why + +Every rung of the width ladder (`mp-gga-ggau-width`: W64, W96, W128, W160) +trains with the same recipe — `lr: 0.001`, `weight_decay: 0.0` — inherited +from the June W64 run. For Adam in standard parametrization the optimal LR +typically *shrinks* as width grows (roughly ∝ 1/width in the μP limit), so +the wider rungs are plausibly mistuned and the width-scaling conclusions +conflate capacity with recipe mismatch. W96's 17.8% win over W64 survived +that handicap; the question is whether the gaps (and the W128/W160 verdicts) +change once each width gets its own recipe. + +This sweep tunes LR and WD independently at widths 32, 64, 96, then uses the +per-width optima to (a) re-examine the width ranking and (b) extrapolate a +recipe for W128/W160 rather than sweeping at those (much more expensive) +widths. + +## Trial design (proxy task) + +Full production runs are ~5 days; sweep trials compress two axes and change +nothing else: + +- **Data**: same staged capped dataset (111,257 ids). Train is subsampled to + ~12K structures (frac 0.1112 of each functional's capped train split, so + the 76/24 gga/gga+u mix is preserved: 9,139 + 2,862). **Validation is the + untouched production val set** (847 + 262 = 1,109), so trial `val_loss` is + on the same yardstick as the production curves. The subset lives in two + small split JSONs checked into the repo (`data/MP/sweep_splits/`) and + referenced by repo-relative path — they ride along in the submit.sh bundle, + so no bucket upload and no interaction with the `.staged.ok` warm-node + skip. +- **Schedule**: `epochs: 8`, `warmup_length: 1` — each trial runs its own + complete warmup+cosine schedule, compressed. Comparing LRs mid-schedule is + misleading; comparing completed short schedules preserves ranking much + better. 8 epochs × 12K ≈ 24K optimizer steps ≈ 0.9 production epochs. +- **Hardware/batch**: 1 node, GB200x4, DDP, batch 1/GPU — the *same + effective batch (4) as production*, so the tuned LR transfers directly. + (A single-GPU trial would change the effective batch and invalidate it.) +- Everything else identical to `config_gga_gga+u_w96.yaml`: bf16-mixed, no + activation checkpointing, Adam betas (0.9, 0.99), offline W&B + sidecar. + +Per-trial wall-clock (from measured production step times): W96 ≈ 4 h, +W64 ≈ 3 h, W32 ≈ 2–2.5 h. + +## Stage A — LR grid at wd = 0 + +`lr ∈ {2.5e-4, 5e-4, 1e-3, 2e-3, 4e-3}` × `width ∈ {32, 64, 96}` = **15 +trials, ≈ 50 node-hours**. Factor-2 spacing matches the flatness of Adam LR +basins; the 16× range is centered on the incumbent 1e-3 and wide enough for +a ~3× shift across 32→96. + +Decision rule per width: `lr*` = argmin of final-epoch `val_loss` (sanity: +best-epoch val and curve shape agree; a diverged/NaN trial is a top-edge +signal, likely at 4e-3 for the wider models). **If a width's optimum lands +on a grid edge, extend one more factor-2 point before moving on.** The old +Optuna tier-1 result (lr ≈ 3.5e-3 best for W32, different data/precision) +hints W32 may press the top edge. + +## Stage B — WD grid at lr* + +Weight decay uses **AdamW** (`optimizer: adamw`, added to +`lightning.py`) — the incumbent plain Adam applies `weight_decay` as +coupled L2, which gets rescaled per-parameter by the adaptive denominator +and makes values neither interpretable nor comparable across widths. At +wd = 0 the two optimizers are identical, so nothing about existing runs or +Stage A changes. + +Per width, 5 trials: + +- `wd ∈ {1e-4, 1e-3, 1e-2}` at `lr*` +- one cross term `(lr*/2, wd=1e-3)` to catch LR–WD interaction (in AdamW the + effective per-step decay is `lr·wd`, so a strong-WD winner may prefer a + lower LR) +- a **verbatim rerun of the Stage-A anchor** `(lr*, wd=0)` — the run-to-run + noise bar that decides whether Stage B differences are real + +**15 trials, ≈ 50 node-hours.** + +Caveat to carry into analysis: on a 12K subset, overfitting appears ~9× +earlier than on the full set, so the subset will overstate how much WD +helps. Treat Stage B as establishing the *tolerance and trend* (does wd hurt +below some threshold? does λ_eff = lr·wd stay constant across widths?), and +confirm the absolute choice in Stage C. If a large wd wins on the subset, +prefer the largest wd that is *neutral-or-better* rather than the argmin. + +## Stage C — full-scale confirmation + +One run per width at the tuned `(lr*, wd*)`: full 111K dataset, production +config, **10–12 epoch budget** (~45 h W96, ~32 h W64 on GB200x4). Compare +val_loss at matched epochs against the incumbent curves, which already exist +at lr=1e-3/wd=0: + +- W64: `mp-gga-ggau-width` run `zz3oecp7` (16-epoch best 0.008398) +- W96: run `z21di7sl` / `revived-energy-7` (ep10 0.008336 … ep27 0.006901) + +W32 has no incumbent in this campaign; run it only if the trend fit needs +the third full-scale point. + +## Analysis + +- Fit `log lr*` vs `log width` → slope α; predict lr for W128/W160. + (Expected α between 0 and −1.) Same check on wd: is `lr*·wd*` + width-stable? +- Re-examine the width ranking at matched epochs with tuned recipes — this + feeds the width-ablation writeup. +- All sweep runs log to W&B project **`mp-gga-ggau-lrwd`**. Run names are + auto-generated `w{width}_{MMDD-HHMM}` (restart-segment convention), so + aggregate by `config.lr` / `config.weight_decay` / + `config.model.n_channels` — same pattern as + `scripts/review_width_ablation.py`. + +## Runbook + +```bash +# 1. Get the production split files locally (small; either source works) +rclone copy cw:mp/chg_datasets/functionals/gga/split_capped.json /tmp/sweep/gga/ +rclone copy cw:mp/chg_datasets/functionals/gga+u/split_capped.json /tmp/sweep/gga+u/ +# or: scp della:/scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga/split_capped.json ... + +# 2. Generate the sweep splits (checked into the branch for provenance) +uv run python scripts/coreweave/make_sweep_split.py \ + /tmp/sweep/gga/split_capped.json data/MP/sweep_splits/gga_split_sweep12k.json --frac 0.1112 +uv run python scripts/coreweave/make_sweep_split.py \ + "/tmp/sweep/gga+u/split_capped.json" "data/MP/sweep_splits/gga+u_split_sweep12k.json" --frac 0.1112 + +# 3. Stage A configs (already generated in src/electrai/configs/MP/sweep_lrwd/) +uv run python scripts/coreweave/gen_lrwd_sweep.py --stage a +# ... submit the printed submit.sh lines from a clean worktree; jobs run at +# batch priority and won't starve the live W128 run. + +# 4. After Stage A: regenerate for Stage B with the real per-width winners +uv run python scripts/coreweave/gen_lrwd_sweep.py --stage b \ + --best-lr 32= 64= 96= +``` + +Each config filename stem doubles as the checkpoint namespace +(`run_training.sh` derives ckpt dirs from the stem), so trials never +cross-resume; the Stage-B noise rerun gets a `_rep2` stem for the same +reason. + +## Known limitations + +- 24K steps ≈ 0.9 production epochs per trial: LR ranking at short horizon + with a completed cosine is a standard, usually-reliable proxy, but a + near-tie between adjacent LRs at a width should be broken toward the + *lower* LR (long-horizon optima drift down, not up). +- WD conclusions from the subset overstate regularization benefit (see + Stage B caveat); Stage C is the arbiter. +- The scheduler steps per epoch, so an 8-epoch cosine has coarse resolution; + all trials share it, so comparisons are fair. + +## W128 withheld-test-set evaluation (Aug 5–7, 2026) + +Checkpoint `w128_ckpt_epoch55_val0.005785.ckpt` (stage-C W128 at lr 2e-3, +epoch 55) evaluated on the withheld `test` split of `split_capped.json` — +never seen in training or validation. Della jobs 12057752 (GGA) + 12111587 +(GGA+U), fp32, single A100-80GB; config +`src/electrai/configs/MP/config_gga_gga+u_w128_test_della.yaml`; per-sample +CSVs in `/scratch/gpfs/ROSENGROUP/bb9080/w128_test_eval/results{,_padsu}/`. + +| Subset | n | mean NMAE | median | p90 | p99 | max | share <1% | +|---|---|---|---|---|---|---|---| +| GGA | 1700 | 0.480% | 0.373% | 0.900% | 1.842% | 2.95% | 92.0% | +| GGA+U (PADS) | 524 | 0.483% | 0.397% | 0.857% | 1.794% | 3.36% | 94.5% | +| **Combined** | **2224** | **0.481%** | 0.379% | 0.890% | 1.832% | 3.36% | 92.6% | + +**Combined test NMAE 0.481% is under the 0.5% ChargE3Net threshold** and +consistent with the 0.579% val loss (val is bf16-on-GB200; test is fp32). + +Provenance notes uncovered during this eval: + +- The split files existed only in the buckets; they were staged from + `s3://oa-electrai` to `/scratch/gpfs/ROSENGROUP/bb9080/w128_test_eval/` + and their test/validation indices verified byte-identical to the + training-time splits (via `data/MP/sweep_splits/*_sweep12k.json` + pass-through). +- **The training data's `gga+u` inputs are PADS, not SAD.** Della's + `functionals/gga+u_sad` and `gga+u_pads` share identical filelists and + labels; only inputs differ. Evaluating with SAD inputs gives ~10% NMAE + across the whole GGA+U test set (the model degrades SAD inputs below + their own 7–9% input error); PADS inputs give 0.48%. Probe: Della job + 12111541. Any "GGA+U" result from the gga-ggau width campaign is a + PADS-input result. +- bf16-mixed autocast OOMs this checkpoint on A100 (a ~75 GiB allocation + inside a decoder conv on a 108³ sample, Della job 12049722) — evaluate at + fp32 on Della, matching the throughput benchmark (job 12046200). diff --git a/job_w128_test_eval.slurm b/job_w128_test_eval.slurm new file mode 100644 index 00000000..1e24ee9c --- /dev/null +++ b/job_w128_test_eval.slurm @@ -0,0 +1,52 @@ +#!/bin/bash +#SBATCH --job-name=w128-test-eval +#SBATCH --nodes=1 +#SBATCH --ntasks=1 +#SBATCH --cpus-per-task=8 +#SBATCH --mem=64G +#SBATCH --gres=gpu:1 +#SBATCH --constraint=gpu80 +#SBATCH --time=04:00:00 +#SBATCH --output=/scratch/gpfs/ROSENGROUP/bb9080/logs/w128-test-eval-%j.out + +# Evaluate the W128 lr2e-3 checkpoint (epoch 55, val 0.005785) on the withheld +# GGA + GGA+U test set (split_capped.json test keys: 1700 + 524 samples). +# Config: src/electrai/configs/MP/config_gga_gga+u_w128_test_della.yaml + +module purge +module load anaconda3/2025.6 +conda activate electrai +module load proxy/default +export PATH=$HOME/.local/bin:$PATH +# Variable grid sizes fragment the caching allocator; expandable segments +# lets large per-sample activations reuse freed reserved memory. +export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True + +cd /scratch/gpfs/ROSENGROUP/bb9080/electrai + +EVAL=/scratch/gpfs/ROSENGROUP/bb9080/w128_test_eval +CKPT_SRC=/scratch/gpfs/ROSENGROUP/bb9080/checkpoints/w128_ckpt_epoch55_val0.005785.ckpt + +mkdir -p ${EVAL}/ckpt ${EVAL}/results +# test entrypoint expects /last.ckpt +ln -sfn ${CKPT_SRC} ${EVAL}/ckpt/last.ckpt + +echo "=== W128 test-set eval ===" +echo "Checkpoint: ${CKPT_SRC}" +echo "Results: ${EVAL}/results" +echo "Start: $(date)" +echo "" + +srun uv run --no-sync python ./src/electrai/entrypoints/main.py test \ + --config src/electrai/configs/MP/config_gga_gga+u_w128_test_della.yaml + +echo "" +echo "End: $(date)" + +METRICS=${EVAL}/results/metrics.csv +if [ ! -f ${METRICS} ]; then + echo "ERROR: metrics.csv not found at ${METRICS}" + exit 1 +fi +N_ROWS=$(tail -n +2 ${METRICS} | wc -l) +echo "metrics.csv rows: ${N_ROWS} (expected 2224)" diff --git a/modal/benchmark.py b/modal/benchmark.py new file mode 100644 index 00000000..bcf82777 --- /dev/null +++ b/modal/benchmark.py @@ -0,0 +1,278 @@ +"""Modal GPU benchmark for electrai. + +Runs a configurable training benchmark on Modal GPUs with data from the +electrai-data Volume, logs metrics to WandB, and reports wall-clock time. + +Two data sources on the Volume: +- "s3" (default): 205 samples from s3://openathena/electrai/ (≤25MB, matches EC2 benchmark) +- "dataset_4": 2,885 samples from Globus/Della dataset_4 (large grids, needs A100 for prod config) + +Usage: + # Default: 50 samples from S3 set, 5 epochs, 32ch/16blk, L4 (matches EC2 benchmark) + modal run modal/benchmark.py + + # Dataset_4 with tiny model on L4 + modal run modal/benchmark.py --dataset dataset_4 --channels 8 --residual-blocks 2 + + # Production model on A100 with dataset_4 + modal run modal/benchmark.py --dataset dataset_4 --gpu A100 --channels 32 --residual-blocks 16 + + # All S3 samples + modal run modal/benchmark.py --samples 0 + + # Quick smoke test + modal run modal/benchmark.py --samples 10 --epochs 2 --channels 8 --residual-blocks 2 +""" + +from __future__ import annotations + +from pathlib import Path + +import modal + +ROOT = Path(__file__).parent.parent + +data_volume = modal.Volume.from_name("electrai-data") + +image = ( + modal.Image.debian_slim(python_version="3.12") + .apt_install("git") + .pip_install_from_pyproject( + str(ROOT / "pyproject.toml"), optional_dependencies=["dev"] + ) + .add_local_dir(str(ROOT / "src"), remote_path="/root/electrai/src", copy=True) + .add_local_dir( + str(ROOT / "scripts"), remote_path="/root/electrai/scripts", copy=True + ) + .add_local_file( + str(ROOT / "pyproject.toml"), + remote_path="/root/electrai/pyproject.toml", + copy=True, + ) + .run_commands("cd /root/electrai && pip install --no-deps -e .") +) + +app = modal.App("electrai-benchmark", image=image) + +# Data roots on the Volume +DATASETS = { + # Mirrors s3://openathena/electrai/ — same data as EC2 gpu-benchmark.yml + # Note: S3 uses input/ but RhoRead expects data/, so we symlink + "s3": { + "root": "/data/s3/openathena/electrai", + "input_dir": "input", # S3 naming + "max_file_size": 25, # matches EC2 benchmark default + }, + # Globus/Della dataset_4 — 2,885 samples, large grids + "dataset_4": { + "root": "/data/mp/chg_datasets/dataset_4", + "input_dir": "data", # Della naming + "max_file_size": 100, + }, +} + + +@app.function( + gpu="L4", + volumes={"/data": data_volume}, + secrets=[modal.Secret.from_name("wandb-credentials")], + timeout=7200, + retries=0, +) +def run_benchmark( + epochs: int = 5, + channels: int = 32, + residual_blocks: int = 16, + samples: int = 50, + max_file_size: float = -1, + seed: int = 42, + wandb_project: str = "elf-net-ci", + gpu_type: str = "L4", + dataset: str = "s3", + local_copy: bool = False, +): + """Run benchmark and return results.""" + import logging + import sys + + log = logging.getLogger(__name__) + logging.basicConfig(level=logging.INFO) + + sys.path.insert(0, "/root/electrai/scripts") + from e2e_train import run_training + + ds = DATASETS[dataset] + ds_root = Path(ds["root"]) + input_dir = ds["input_dir"] + + # Use dataset-specific default if max_file_size not explicitly set + if max_file_size < 0: + max_file_size = ds["max_file_size"] + + # Build filelist from files on disk (S3 set has no mp_filelist.txt) + data_dir = ds_root / input_dir + all_ids = sorted(p.stem for p in data_dir.glob("*.CHGCAR")) + log.info("Dataset %r: %d total samples in %s", dataset, len(all_ids), data_dir) + + # Filter by file size (avoid OOM on large grids) + if max_file_size > 0: + max_bytes = int(max_file_size * 1024 * 1024) + eligible = [ + sid + for sid in all_ids + if (data_dir / f"{sid}.CHGCAR").stat().st_size <= max_bytes + ] + log.info( + "File size filter: %d/%d eligible (<=%.0fMB)", + len(eligible), + len(all_ids), + max_file_size, + ) + else: + eligible = all_ids + + # Select samples: first N (lexicographic, matching s3_sync.py behavior) + if 0 < samples < len(eligible): + subset = eligible[:samples] + log.info("Selected first %d/%d eligible samples", samples, len(eligible)) + else: + subset = eligible + samples = len(subset) + log.info("Using all %d eligible samples", samples) + + if not subset: + raise ValueError( + f"No eligible samples (dataset={dataset}, total={len(all_ids)}, " + f"max_file_size={max_file_size}MB)" + ) + + data_root = "/tmp/benchmark_data" + Path(data_root).mkdir(parents=True, exist_ok=True) + + if local_copy: + # Copy selected samples to local disk (Volume I/O is ~15x slower) + import shutil + + local_data = Path(data_root) / "data" + local_label = Path(data_root) / "label" + local_data.mkdir(parents=True, exist_ok=True) + local_label.mkdir(parents=True, exist_ok=True) + for sid in subset: + shutil.copy2( + ds_root / input_dir / f"{sid}.CHGCAR", local_data / f"{sid}.CHGCAR" + ) + shutil.copy2( + ds_root / "label" / f"{sid}.CHGCAR", local_label / f"{sid}.CHGCAR" + ) + log.info("Copied %d samples to local disk", len(subset)) + else: + # Symlink to volume (slower I/O but no copy overhead for large datasets) + data_link = Path(data_root) / "data" + if not data_link.exists(): + data_link.symlink_to(ds_root / input_dir) + label_link = Path(data_root) / "label" + if not label_link.exists(): + label_link.symlink_to(ds_root / "label") + log.info("Using volume directly (no local copy)") + + Path(data_root, "mp_filelist.txt").write_text("\n".join(subset) + "\n") + + # Always use gradient checkpointing (32ch/16blk needs it even for ≤25MB files) + use_grad_ckpt = True + + log.info( + "Benchmark: gpu=%s, epochs=%d, channels=%d, blocks=%d, samples=%d, " + "dataset=%s, grad_ckpt=%s", + gpu_type, + epochs, + channels, + residual_blocks, + samples, + dataset, + use_grad_ckpt, + ) + + # Set WandB run name and env vars for platform tagging + import os + import time + + os.environ["INSTANCE_TYPE"] = f"modal-{gpu_type}" + # Set workflow-like name so WandB run name is descriptive. + # GHA sets GITHUB_RUN_NUMBER; for local runs, use timestamp. + os.environ["GITHUB_WORKFLOW"] = "Modal Benchmark" + if "GITHUB_RUN_NUMBER" not in os.environ: + os.environ["GITHUB_RUN_NUMBER"] = time.strftime("%y%m%d-%H%M") + + results = run_training( + channels=channels, + residual_blocks=residual_blocks, + epochs=epochs, + seed=seed, + gpu=True, + gradient_checkpoint=use_grad_ckpt, + data_root=data_root, + max_file_size=0, # already filtered above + wandb_project=wandb_project, + verbose=True, + ) + + log.info("val_loss: %.6f", results["final_val_loss"]) + log.info("train_loss: %.6f", results["final_train_loss"]) + log.info("Wallclock: %.1fs", results["wallclock_s"]) + log.info("GPU: %s (Modal)", gpu_type) + + return results + + +@app.local_entrypoint() +def main( + gpu: str = "L4", + epochs: int = 5, + channels: int = 32, + residual_blocks: int = 16, + samples: int = 50, + max_file_size: float = -1, + seed: int = 42, + wandb_project: str = "elf-net-ci", + dataset: str = "s3", + local_copy: bool = False, +): + import logging + + logging.basicConfig(level=logging.INFO) + log = logging.getLogger(__name__) + + benchmark_fn = run_benchmark + if gpu != "L4": + benchmark_fn = run_benchmark.with_options(gpu=gpu) + + results = benchmark_fn.remote( + epochs=epochs, + channels=channels, + residual_blocks=residual_blocks, + samples=samples, + max_file_size=max_file_size, + seed=seed, + wandb_project=wandb_project, + gpu_type=gpu, + dataset=dataset, + local_copy=local_copy, + ) + + log.info( + "Benchmark complete: val_loss=%.6f, wallclock=%.1fs on %s", + results["final_val_loss"], + results["wallclock_s"], + gpu, + ) + + # Print parseable output for GHA summary + wandb_url = results.get("wandb_run_url") or "" + print(f"BENCHMARK_VAL_LOSS={results['final_val_loss']:.6f}") # noqa: T201 + print(f"BENCHMARK_TRAIN_LOSS={results['final_train_loss']:.6f}") # noqa: T201 + print(f"BENCHMARK_WALLCLOCK={results['wallclock_s']:.0f}") # noqa: T201 + print(f"BENCHMARK_GPU={gpu}") # noqa: T201 + print(f"BENCHMARK_DATASET={dataset}") # noqa: T201 + print(f"BENCHMARK_SAMPLES={samples}") # noqa: T201 + if wandb_url: + print(f"BENCHMARK_WANDB_URL={wandb_url}") # noqa: T201 diff --git a/modal/ci.py b/modal/ci.py new file mode 100644 index 00000000..0cd9da96 --- /dev/null +++ b/modal/ci.py @@ -0,0 +1,93 @@ +"""Modal GPU CI for electrai e2e training test.""" + +from __future__ import annotations + +from pathlib import Path + +import modal + +ROOT = Path(__file__).parent.parent + +# Dependencies read from pyproject.toml (shared with train.py, populate_volume.py) +image = ( + modal.Image.debian_slim(python_version="3.12") + .apt_install("git") + .pip_install_from_pyproject( + str(ROOT / "pyproject.toml"), optional_dependencies=["dev"] + ) + .add_local_dir(str(ROOT / "src"), remote_path="/root/electrai/src", copy=True) + .add_local_dir( + str(ROOT / "scripts"), remote_path="/root/electrai/scripts", copy=True + ) + .add_local_dir(str(ROOT / "tests"), remote_path="/root/electrai/tests", copy=True) + .add_local_dir(str(ROOT / "data"), remote_path="/root/electrai/data", copy=True) + .add_local_file( + str(ROOT / "pyproject.toml"), + remote_path="/root/electrai/pyproject.toml", + copy=True, + ) + .run_commands("cd /root/electrai && pip install --no-deps -e .") +) + +app = modal.App("electrai-ci", image=image) + + +@app.function(gpu="L4", timeout=600, retries=0) +def run_e2e_test(epochs: int = 5, check: bool = True): + """Run e2e training test on GPU.""" + import json + import logging + import sys + + log = logging.getLogger(__name__) + + sys.path.insert(0, "/root/electrai/scripts") + from e2e_train import run_training + + results = run_training(epochs=epochs, gpu=True, verbose=True) + log.info("Platform: %s", results["platform"]) + log.info("Final val_loss: %.6f", results["final_val_loss"]) + if results["final_train_loss"] is not None: + log.info("Final train_loss: %.6f", results["final_train_loss"]) + log.info("Wallclock: %.1fs", results["wallclock_s"]) + + if check: + expected_file = Path("/root/electrai/tests/expected_values.json") + expected_values = json.loads(expected_file.read_text()) + platform = results["platform"] + if platform not in expected_values: + raise ValueError( + f"No expected values for platform {platform!r}, " + f"available: {list(expected_values.keys())}" + ) + expected = expected_values[platform] + if expected.get("final_val_loss") is None: + raise ValueError(f"Expected values for {platform!r} are null") + expected_val_loss = expected["final_val_loss"] + diff = abs(results["final_val_loss"] - expected_val_loss) + tolerance = 0.001 + if diff > tolerance: + raise AssertionError( + f"val_loss {results['final_val_loss']:.6f} differs from expected " + f"{expected_val_loss:.6f} by {diff:.6f} (tolerance: {tolerance})" + ) + log.info( + "PASS: val_loss matches expected within tolerance (%.6f <= %f)", + diff, + tolerance, + ) + + return results + + +@app.local_entrypoint() +def main(epochs: int = 5, check: bool = True): + import logging + + logging.basicConfig(level=logging.INFO) + results = run_e2e_test.remote(epochs=epochs, check=check) + logging.getLogger(__name__).info( + "Results: val_loss=%.6f in %.1fs", + results["final_val_loss"], + results["wallclock_s"], + ) diff --git a/modal/globus_load.py b/modal/globus_load.py new file mode 100644 index 00000000..2c495df7 --- /dev/null +++ b/modal/globus_load.py @@ -0,0 +1,103 @@ +"""Load della data onto the electrai-data Modal Volume via Globus. + +Runs Globus Connect Personal (GCP) inside a CPU-only Modal container with the +`electrai-data` Volume mounted at /data, registered as a personal endpoint using +a setup key generated in the Globus web UI. GCP makes only outbound connections, +so it works behind Modal's NAT. You then initiate the della -> this endpoint +transfer from the Globus file manager, writing into /data; the Volume is +committed periodically and on exit. + +Steps: + 1. In the Globus web UI (https://app.globus.org), create a Globus Connect + Personal collection and copy its setup key. + 2. Start the endpoint host (runs up to ~24h): + modal run modal/globus_load.py --setup-key "" + 3. In the Globus file manager (https://app.globus.org/file-manager) transfer: + source: ROSENGROUP share, .../mp/chg_datasets/functionals/{gga,gga+u} + dest: this endpoint, path /data/mp/chg_datasets/functionals/ + The data/label entries are symlinks into rho_gga{,+u}; confirm Globus + follows them (lands real .zarr stores). If it copies symlinks instead, + transfer rho_gga and rho_gga+u and adjust the config roots. + 4. When the Globus task shows complete, stop this run (ctrl-C / `modal app + stop electrai-globus-load`). +""" + +from __future__ import annotations + +import modal + +app = modal.App("electrai-globus-load") + +data_volume = modal.Volume.from_name("electrai-data", create_if_missing=True) + +VOLUME_ROOT = "/data" +GCP_URL = ( + "https://downloads.globus.org/globus-connect-personal/linux/stable/" + "globusconnectpersonal-latest.tgz" +) + +image = ( + modal.Image.debian_slim() + .apt_install("wget", "ca-certificates", "tar") + .run_commands( + "cd /opt && wget -q " + f"{GCP_URL}" + " -O gcp.tgz && tar xzf gcp.tgz && rm gcp.tgz " + "&& mv globusconnectpersonal-* globusconnectpersonal" + ) +) + + +@app.function( + image=image, + volumes={VOLUME_ROOT: data_volume}, + timeout=86400, # 24h max: one transfer window + cpu=4.0, +) +def host_endpoint(setup_key: str, run_hours: float = 23.5, commit_every_s: int = 300): + import logging + import subprocess + import time + from pathlib import Path + + log = logging.getLogger(__name__) + logging.basicConfig(level=logging.INFO) + + gcp = "/opt/globusconnectpersonal/globusconnectpersonal" + + # Non-interactive endpoint registration using the web-UI setup key. + log.info("Registering Globus Connect Personal endpoint...") + subprocess.run([gcp, "-setup", "--setup-key", setup_key], check=True) + + # Grant read/write to the mounted Volume (path,writable,shareable). + cfg_dir = Path.home() / ".globusonline" / "lta" + cfg_dir.mkdir(parents=True, exist_ok=True) + (cfg_dir / "config-paths").write_text(f"{VOLUME_ROOT},1,0\n") + log.info("config-paths -> %s,1,0", VOLUME_ROOT) + + # Start GCP in the background (outbound-only). + log.info("Starting Globus Connect Personal...") + proc = subprocess.Popen([gcp, "-start"]) + log.info( + "Endpoint up. Initiate the della -> %s transfer in the Globus web UI now.", + VOLUME_ROOT, + ) + + deadline = time.time() + run_hours * 3600 + try: + while time.time() < deadline: + time.sleep(commit_every_s) + data_volume.commit() # persist whatever has landed so far + status = subprocess.run( + [gcp, "-status"], capture_output=True, text=True, check=False + ) + log.info("globus status: %s", (status.stdout or status.stderr).strip()) + finally: + proc.terminate() + data_volume.commit() + log.info("Stopped GCP and committed Volume.") + + +@app.local_entrypoint() +def main(setup_key: str, run_hours: float = 23.5): + host_endpoint.remote(setup_key=setup_key, run_hours=run_hours) diff --git a/modal/populate_volume.py b/modal/populate_volume.py new file mode 100644 index 00000000..d3d5ce71 --- /dev/null +++ b/modal/populate_volume.py @@ -0,0 +1,218 @@ +"""Populate Modal Volume with training data from S3, packing zarr stores. + +Modal Volumes are capped at ~500K inodes per volume. Our zarr v3 layout has +~8 inodes per `.zarr/` (3 files + 5 directory entries) times ~226K stores +is ~1.8M inodes — well over the cap. + +To fit, each S3-side `/.../.zarr/{zarr.json, +charge_density_total/zarr.json, charge_density_total/c/0/0/0}` is packed into +a single `/.../.zarr.zip` on the Volume (zarr's `ZipStore` reads +it as a normal store). Non-zarr S3 keys (filelists, split files) pass through +unchanged. Net: ~226K + a few standalone files ≈ ~230K inodes — well under +the cap. + +The matching loader change (auto-detect `.zarr.zip` via ZipStore) lives in +`src/electrai/dataloader/utils.py:load_zarr`. +""" + +from __future__ import annotations + +import modal + +app = modal.App("electrai-populate") +volume = modal.Volume.from_name("electrai-data", create_if_missing=True) + + +WORKERS = 32 +BATCH = 2000 # commit every BATCH item completions (stores or standalones) + + +@app.function( + image=modal.Image.debian_slim(python_version="3.12").pip_install("boto3"), + volumes={"/data": volume}, + secrets=[modal.Secret.from_name("oa-electrai-read")], + timeout=86400, # 24h + retries=0, + cpu=4.0, + memory=4096, +) +def sync_s3( + bucket: str = "oa-electrai", + prefix: str = "mp/chg_datasets", + dest: str = "/data/mp/chg_datasets", + wipe_first: bool = False, +): + """Sync S3 -> Volume, packing each `.zarr/` into `.zarr.zip`. + + If `wipe_first=True`, recursively removes `dest` before populating to free + inodes — required when re-populating after a previous unpacked attempt. + """ + import logging + import shutil + import zipfile + from concurrent.futures import ThreadPoolExecutor, as_completed + from pathlib import Path + + import boto3 + from botocore.config import Config + + log = logging.getLogger(__name__) + logging.basicConfig(level=logging.INFO) + + s3 = boto3.client( + "s3", + config=Config(max_pool_connections=WORKERS + 8, retries={"max_attempts": 5}), + ) + + if wipe_first: + target = Path(dest) + if target.exists(): + log.info("Wiping %s (this can take a few minutes for many inodes)…", target) + shutil.rmtree(target) + volume.commit() + log.info("Wipe complete; committed.") + + # ---- list S3 --------------------------------------------------------- + paginator = s3.get_paginator("list_objects_v2") + objects = [ + obj + for page in paginator.paginate(Bucket=bucket, Prefix=prefix) + for obj in page.get("Contents", []) + ] + total_bytes = sum(o["Size"] for o in objects) + log.info("Listed %d S3 objects, %.1f GiB", len(objects), total_bytes / (1024**3)) + + # ---- group by .zarr/ store ------------------------------------------ + # Each store-key is the S3 prefix up to and including `.zarr`. Inner name + # within the zip is the remainder after `.zarr/`. + stores: dict[str, list[tuple[dict, str]]] = {} + standalones: list[dict] = [] + for obj in objects: + key = obj["Key"] + if ".zarr/" in key: + store_key, inner = key.split(".zarr/", 1) + store_key = store_key + ".zarr" + stores.setdefault(store_key, []).append((obj, inner)) + else: + standalones.append(obj) + log.info( + "Grouping: %d zarr stores to pack, %d standalone files", + len(stores), + len(standalones), + ) + + # ---- error log on volume -------------------------------------------- + err_log_path = Path(dest).parent / "_populate_errors.log" + err_log_path.parent.mkdir(parents=True, exist_ok=True) + err_log = err_log_path.open("a") + err_log.write(f"\n=== run start {bucket}/{prefix} (packed) ===\n") + err_log.flush() + + # ---- workers -------------------------------------------------------- + def pack_store(store_key: str, parts: list[tuple[dict, str]]): + rel = store_key[len(prefix) :].lstrip("/") + zip_path = Path(dest) / (rel + ".zip") + if zip_path.exists() and zip_path.stat().st_size > 0: + return ("skip", None) + try: + zip_path.parent.mkdir(parents=True, exist_ok=True) + tmp = zip_path.with_name(zip_path.name + ".tmp") + with zipfile.ZipFile(tmp, "w", zipfile.ZIP_STORED) as z: + for obj, inner in parts: + body = s3.get_object(Bucket=bucket, Key=obj["Key"])["Body"].read() + z.writestr(inner, body) + tmp.rename(zip_path) + return ("dl", None) + except Exception as e: + return ("err", f"{store_key}: {e!r}") + + def copy_standalone(obj: dict): + key = obj["Key"] + rel = key[len(prefix) :].lstrip("/") + local_path = Path(dest) / rel + if local_path.exists() and local_path.stat().st_size == obj["Size"]: + return ("skip", None) + try: + local_path.parent.mkdir(parents=True, exist_ok=True) + tmp = local_path.with_name(local_path.name + ".tmp") + with tmp.open("wb") as f: + s3.download_fileobj(bucket, key, f) + tmp.rename(local_path) + return ("dl", None) + except Exception as e: + return ("err", f"{key}: {e!r}") + + # ---- process: stores first (the bulk), then standalones -------------- + work: list[tuple[str, object]] = [("store", item) for item in stores.items()] + [ + ("standalone", obj) for obj in standalones + ] + total = len(work) + log.info("Total work items: %d", total) + + packed = skipped = errors = seen = 0 + try: + with ThreadPoolExecutor(max_workers=WORKERS) as pool: + for start in range(0, total, BATCH): + batch = work[start : start + BATCH] + futures = [] + for kind, payload in batch: + if kind == "store": + store_key, parts = payload + futures.append(pool.submit(pack_store, store_key, parts)) + else: + futures.append(pool.submit(copy_standalone, payload)) + for future in as_completed(futures): + res, msg = future.result() + seen += 1 + if res == "dl": + # pack_store returns "dl" too; we count both as "packed" + # for stores and "copied" for standalones; cheaper to + # just bucket together. + packed += 1 + elif res == "skip": + skipped += 1 + else: + errors += 1 + err_log.write(msg + "\n") + if errors <= 50: + log.warning("error: %s", msg) + err_log.flush() + volume.commit() + log.info( + "Progress: %d/%d (done %d, skipped %d, errors %d) — committed", + seen, + total, + packed, + skipped, + errors, + ) + finally: + err_log.close() + + log.info("Done: %d packed/copied, %d skipped, %d errors", packed, skipped, errors) + volume.commit() + + +@app.local_entrypoint() +def main( + bucket: str = "oa-electrai", + prefix: str = "mp/chg_datasets", + dest: str = "/data/mp/chg_datasets", + wipe_first: bool = False, +): + """Fire-and-forget: spawn sync_s3 and return so the remote run survives any + local CLI disconnect. Pair with `modal run --detach`; monitor via the Modal + web UI or `modal app logs `. + + Pass `--wipe-first` on the first packed run to clear any unpacked partial + state from previous attempts (frees inodes). + """ + import logging + + logging.basicConfig(level=logging.INFO) + fc = sync_s3.spawn(bucket=bucket, prefix=prefix, dest=dest, wipe_first=wipe_first) + log = logging.getLogger(__name__) + log.info( + "Spawned sync_s3 FunctionCall id=%s (wipe_first=%s)", fc.object_id, wipe_first + ) + log.info("Monitor at https://modal.com/apps (look for electrai-populate)") diff --git a/modal/prep_volume.py b/modal/prep_volume.py new file mode 100644 index 00000000..7888ecd3 --- /dev/null +++ b/modal/prep_volume.py @@ -0,0 +1,105 @@ +"""Prepare the electrai-data Volume after a Globus transfer. + +The Globus transfer lands the real data under .../rho_gga and .../rho_gga+u plus +the small per-functional metadata (mp_filelist.txt, split.json). This helper: + + 1. (Re)creates the data/label symlinks the loader expects under + functionals/{gga,gga+u}/ pointing at the matching rho_* dirs. + 2. Builds subset smoke filelists (mp_filelist_smoke.txt) used by + config_gga_gga+u_f32_smoke.yaml. + 3. Sanity-checks that the first id of each filelist resolves to a real .zarr. + 4. Commits the Volume explicitly (no reliance on shell auto-commit). + +Idempotent: if Globus already landed real data/label dirs (i.e. it followed the +symlinks), those are left untouched. + +Usage: + modal run modal/prep_volume.py # 200-sample smoke filelists + modal run modal/prep_volume.py --smoke-n 50 +""" + +from __future__ import annotations + +import modal + +app = modal.App("electrai-prep") + +data_volume = modal.Volume.from_name("electrai-data", create_if_missing=True) + +VOLUME_ROOT = "/data" +# functional dir name -> real data dir name (both under mp/chg_datasets/) +FUNCTIONALS = {"gga": "rho_gga", "gga+u": "rho_gga+u"} + +image = modal.Image.debian_slim() + + +@app.function(image=image, volumes={VOLUME_ROOT: data_volume}, timeout=900) +def prep(smoke_n: int = 200): + import logging + from pathlib import Path + + log = logging.getLogger(__name__) + logging.basicConfig(level=logging.INFO) + + base = Path(VOLUME_ROOT) / "mp" / "chg_datasets" + + for func, rho in FUNCTIONALS.items(): + fdir = base / "functionals" / func + filelist = fdir / "mp_filelist.txt" + if not filelist.exists(): + raise FileNotFoundError( + f"{filelist} missing — transfer functionals/{func}/mp_filelist.txt " + "(and split.json) before running prep." + ) + + # 1. data/label symlinks -> ../../rho_*/{data,label} + for sub in ("data", "label"): + link = fdir / sub + target = Path("../..") / rho / sub # relative to fdir + real = base / rho / sub + if not real.exists(): + raise FileNotFoundError( + f"{real} missing — transfer the {rho} data dir before running prep." + ) + if link.is_symlink(): + link.unlink() + link.symlink_to(target) + log.info("relinked %s -> %s", link, target) + elif link.is_dir(): + log.info( + "%s is a real dir (Globus followed symlinks); leaving as-is", link + ) + else: + link.symlink_to(target) + log.info("linked %s -> %s", link, target) + + # 2. smoke filelist + ids = filelist.read_text().splitlines() + smoke = fdir / "mp_filelist_smoke.txt" + smoke.write_text("\n".join(ids[:smoke_n]) + "\n") + log.info("wrote %s (%d ids)", smoke, min(smoke_n, len(ids))) + + # 3. sanity check: first id resolves to a `.zarr.zip` (packed) or + # `.zarr/` (unpacked) store under data/. + first = ids[0] + store_dir = fdir / "data" / f"{first}.zarr" + store_zip = fdir / "data" / f"{first}.zarr.zip" + if not (store_zip.exists() or store_dir.exists()): + raise FileNotFoundError( + f"Neither {store_zip} nor {store_dir} found — " + "data/ symlink or transfer is incomplete." + ) + log.info( + "OK: %s resolves (%d total ids)", + store_zip if store_zip.exists() else store_dir, + len(ids), + ) + + # 4. persist + data_volume.commit() + log.info("Volume committed.") + + +@app.local_entrypoint() +def main(smoke_n: int = 200): + prep.remote(smoke_n=smoke_n) diff --git a/modal/train.py b/modal/train.py new file mode 100644 index 00000000..68e88800 --- /dev/null +++ b/modal/train.py @@ -0,0 +1,206 @@ +"""Modal training entrypoint for electrai. + +Run real training experiments on Modal GPUs with data from the +electrai-data Volume and checkpoints persisted to electrai-checkpoints. + +Handles both single-dataset and multi-dataset (`datasets:` list) configs. +Della absolute paths in the config are remapped onto the Volume mount, and +checkpoints are namespaced by `run_name` so the entrypoint auto-resumes from +`/last.ckpt` across successive (24h-capped) invocations. + +Usage: + # Multi-dataset experiment config on a single A100 (80GB) + modal run modal/train.py --config src/electrai/configs/MP/config_gga_gga+u_f32.yaml + + # Short subset smoke run + modal run modal/train.py --config src/electrai/configs/MP/config_gga_gga+u_f32_smoke.yaml + + # Pick a different GPU (e.g. multi-GPU DDP once throughput is known) + modal run modal/train.py --config --gpu "A100-80GB:8" +""" + +from __future__ import annotations + +from pathlib import Path + +import modal + +ROOT = Path(__file__).parent.parent + +# Persistent volumes +data_volume = modal.Volume.from_name("electrai-data", create_if_missing=True) +ckpt_volume = modal.Volume.from_name("electrai-checkpoints", create_if_missing=True) + +# Dependencies read from pyproject.toml (shared with ci.py, populate_volume.py) +image = ( + modal.Image.debian_slim(python_version="3.12") + .apt_install("git") + .pip_install_from_pyproject( + str(ROOT / "pyproject.toml"), optional_dependencies=["dev"] + ) + .add_local_dir(str(ROOT / "src"), remote_path="/root/electrai/src", copy=True) + .add_local_dir( + str(ROOT / "examples"), remote_path="/root/electrai/examples", copy=True + ) + .add_local_file( + str(ROOT / "pyproject.toml"), + remote_path="/root/electrai/pyproject.toml", + copy=True, + ) + .run_commands("cd /root/electrai && pip install --no-deps -e .") +) + +app = modal.App("electrai-train", image=image) + +# Della ROSENGROUP share root and where it is mounted on Modal. Config paths are +# written as della absolute paths; they are rewritten onto the Volume here. +DELLA_SHARE_PREFIX = "/scratch/gpfs/ROSENGROUP/common/globus_share_OA" +VOLUME_ROOT = "/data" +CKPT_ROOT = "/checkpoints" +DEFAULT_GPU = "A100-80GB" + + +def _remap_path(path: str | None) -> str | None: + """Rewrite a della share path onto the Volume mount. Idempotent.""" + if not path: + return path + if path.startswith(VOLUME_ROOT): + return path + if path.startswith(DELLA_SHARE_PREFIX): + rel = path[len(DELLA_SHARE_PREFIX) :].lstrip("/") + return str(Path(VOLUME_ROOT) / rel) + return path + + +def _remap_data_paths(cfg: dict) -> dict: + """Remap every dataset root/split_file in a config onto the Volume.""" + data = cfg.get("data", {}) + for ds in data.get("datasets") or []: + ds["root"] = _remap_path(ds.get("root")) + ds["split_file"] = _remap_path(ds.get("split_file")) + if data.get("root"): + data["root"] = _remap_path(data["root"]) + if data.get("split_file"): + data["split_file"] = _remap_path(data["split_file"]) + return cfg + + +def _dataset_roots(cfg: dict) -> list[str]: + data = cfg.get("data", {}) + if data.get("datasets"): + return [ds["root"] for ds in data["datasets"]] + if data.get("root"): + return [data["root"]] + return [] + + +@app.function( + gpu=DEFAULT_GPU, + volumes={VOLUME_ROOT: data_volume, CKPT_ROOT: ckpt_volume}, + secrets=[modal.Secret.from_name("wandb-credentials")], + timeout=86400, # 24h (Modal max); resume across runs via /last.ckpt + retries=0, +) +def train(config_json: str, gpu_type: str = DEFAULT_GPU): + """Run training with the given config (as JSON, converted to YAML remotely).""" + import json + import logging + import subprocess + import sys + + import yaml + + log = logging.getLogger(__name__) + logging.basicConfig(level=logging.INFO) + + cfg = json.loads(config_json) + + config_path = Path("/tmp/config.yaml") + with config_path.open("w") as f: + yaml.dump(cfg, f, default_flow_style=False) + + log.info("GPU: %s", gpu_type) + log.info("Checkpoint dir: %s", cfg.get("ckpt_path")) + + # Verify each dataset filelist exists on the Volume and log sample counts. + total = 0 + for root in _dataset_roots(cfg): + fp = Path(root) + if not fp.exists(): + raise FileNotFoundError(f"Filelist not found on Volume: {fp}") + n = len(fp.read_text().strip().splitlines()) + total += n + log.info("Dataset %s: %d samples", root, n) + log.info("Total samples: %d", total) + + # Auto-resume: the entrypoint loads /last.ckpt if present. + ckpt_dir = Path(cfg.get("ckpt_path", CKPT_ROOT)) + ckpt_dir.mkdir(parents=True, exist_ok=True) + last_ckpt = ckpt_dir / "last.ckpt" + if last_ckpt.exists(): + log.info("Resuming from %s", last_ckpt) + else: + log.info("No checkpoint found, starting from scratch") + + result = subprocess.run( + [ + sys.executable, + "-m", + "electrai.entrypoints.main", + "train", + "--config", + str(config_path), + ], + cwd="/root/electrai", + check=False, + ) + + # Persist checkpoints + ckpt_volume.commit() + + if result.returncode != 0: + raise RuntimeError(f"Training failed with exit code {result.returncode}") + + log.info("Training complete. Checkpoints saved to electrai-checkpoints volume.") + + +@app.local_entrypoint() +def main(config: str, gpu: str = DEFAULT_GPU): + import json + import logging + import subprocess + + logging.basicConfig(level=logging.INFO) + log = logging.getLogger(__name__) + + # Load the YAML locally without importing electrai (yaml may not be in the + # modal CLI env). Pass the path as argv to avoid shell/string interpolation. + loaded = subprocess.run( + [ + "python3", + "-c", + "import sys, yaml, json; " + "print(json.dumps(yaml.safe_load(open(sys.argv[1]))))", + config, + ], + capture_output=True, + text=True, + check=True, + ) + cfg = json.loads(loaded.stdout) + + # Remap della paths onto the Volume and pin checkpoints to the ckpt Volume, + # namespaced by run so multiple experiments don't collide on last.ckpt. + cfg = _remap_data_paths(cfg) + run_name = cfg.get("run_name", "run") + cfg["ckpt_path"] = f"{CKPT_ROOT}/{run_name}" + + config_json = json.dumps(cfg, indent=2) + log.info("Config:\n%s", config_json) + + # Fire-and-forget so multi-day training survives any local CLI disconnect. + # Pair with `modal run --detach`; monitor via `modal app logs ` or + # the Modal web UI. See [[modal-long-running-detach-spawn]]. + fc = train.with_options(gpu=gpu).spawn(config_json=config_json, gpu_type=gpu) + log.info("Spawned train FunctionCall id=%s on %s", fc.object_id, gpu) + log.info("Monitor at https://modal.com/apps (look for electrai-train)") diff --git a/pyproject.toml b/pyproject.toml index ca0865c6..aeddaa2c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -62,6 +62,17 @@ repository = "https://github.com/Quantum-Accelerators/electrai" documentation = "https://quantum-accelerators.github.io/electrai/" changelog = "https://github.com/Quantum-Accelerators/electrai/blob/main/CHANGELOG.md" +# PyPI's linux-aarch64 torch wheels are CPU-only; Grace/GB200 nodes need the +# SBSA CUDA builds (+cu128) from the PyTorch index. +[tool.uv.sources] +torch = [{ index = "pytorch-cu128", marker = "sys_platform == 'linux' and platform_machine == 'aarch64'" }] +torchvision = [{ index = "pytorch-cu128", marker = "sys_platform == 'linux' and platform_machine == 'aarch64'" }] + +[[tool.uv.index]] +name = "pytorch-cu128" +url = "https://download.pytorch.org/whl/cu128" +explicit = true + [tool.setuptools.package-data] electrai = ["py.typed"] @@ -183,6 +194,9 @@ skip-magic-trailing-comma = true "scripts/benchmark_*.py" = [ "T201", # print used (CLI script logging) ] +"scripts/coreweave/*.py" = [ + "T201", # print used (CLI script logging) +] "inference_benchmark/charge3net/**" = [ # Reference copies of files that live in the charge3net fork; not # subject to electrai's lint policy. diff --git a/scripts/coreweave/count_voxel_bands.py b/scripts/coreweave/count_voxel_bands.py new file mode 100644 index 00000000..2ebc44a5 --- /dev/null +++ b/scripts/coreweave/count_voxel_bands.py @@ -0,0 +1,58 @@ +"""Count capped-dataset structures above the per-width cuDNN kernel cliffs. + +cuDNN's fast conv engines fall back to reference kernels (~400x slower) when +the U-Net decoder-concat tensor exceeds 2^31 elements. For concat channels +2C, the per-structure voxel ceiling is 2^31 / (2 * n_channels). This script +reads zarr shape metadata for every id in the capped filelists (data must be +staged locally; run on a warm node) and reports how many structures exceed +each width's ceiling — i.e. the data cost of a width-specific tighter cap. +""" + +from __future__ import annotations + +import statistics +from multiprocessing import Pool +from pathlib import Path + +import zarr + +STAGE_ROOT = Path("/uv/cache/electrai/mp/chg_datasets/functionals") +CEILINGS = { + "W192 (2*192 ch concat)": 2**31 // 384, # 5,592,405 voxels + "W256 (2*256 ch concat)": 2**31 // 512, # 4,194,304 voxels +} +CURRENT_CAP = 5_832_000 + + +def voxels(args): + root, mpid = args + try: + g = zarr.open_group(str(root / "data" / f"{mpid}.zarr"), mode="r") + s = g["charge_density_total"].shape + return s[0] * s[1] * s[2] + except Exception: + return -1 + + +def main(): + tasks = [] + for func in ("gga", "gga+u"): + root = STAGE_ROOT / func + ids = (root / "mp_filelist_capped.txt").read_text().split() + tasks += [(root, i) for i in ids] + + with Pool(32) as p: + sizes = p.map(voxels, tasks, chunksize=256) + + ok = [s for s in sizes if s > 0] + print(f"total {len(sizes)}, unreadable {len(sizes) - len(ok)}") + print(f"max voxels: {max(ok)} (current cap {CURRENT_CAP})") + print(f"mean voxels: {sum(ok) / len(ok):.0f}, median: {statistics.median(ok):.0f}") + for label, ceiling in CEILINGS.items(): + over = sum(1 for s in ok if s > ceiling) + print(f"over {label} ceiling {ceiling}: {over} ({100 * over / len(ok):.3f}%)") + print("COUNT DONE") + + +if __name__ == "__main__": + main() diff --git a/scripts/coreweave/dry_run.py b/scripts/coreweave/dry_run.py new file mode 100644 index 00000000..9231ea7c --- /dev/null +++ b/scripts/coreweave/dry_run.py @@ -0,0 +1,124 @@ +"""Dry run for the CoreWeave W96 campaign. + +Validates, on one GB200, everything the real run depends on except multi-GPU: + 1. the staged dataset (filelists, symlink layout, zarr reads via RhoRead), + 2. dataloader batch structure and read throughput, + 3. the W96 memory envelope: forward+backward+Adam step at the grid-size cap + (180^3, the largest structure the capped filelists allow) in bf16, + with activation checkpointing off. + +Run after scripts/coreweave/stage_data.sh: + uv run --no-sync python scripts/coreweave/dry_run.py \ + --config src/electrai/configs/MP/config_gga_gga+u_w96.yaml +""" + +from __future__ import annotations + +import argparse +import time +from pathlib import Path + +import torch +import yaml +from hydra.utils import instantiate + + +def describe(obj): + if isinstance(obj, torch.Tensor): + return f"Tensor{tuple(obj.shape)} {obj.dtype}" + if isinstance(obj, dict): + return {k: describe(v) for k, v in obj.items()} + if isinstance(obj, (list, tuple)): + return [describe(v) for v in obj] + return repr(obj)[:60] + + +def check_data(cfg, n_batches): + dm = instantiate(cfg["data"]) + dm.setup("fit") + train = dm.train_dataloader() + val = dm.val_dataloader() + print(f"train batches: {len(train)}, val batches: {len(val)}") + + start = time.monotonic() + for i, batch in enumerate(train): + if i == 0: + print("batch structure:", describe(batch)) + if i + 1 >= n_batches: + break + dt = time.monotonic() - start + print(f"read {n_batches} train batches in {dt:.1f}s ({dt / n_batches:.2f}s/batch)") + + +def check_memory(cfg, grids=(128, 180), iters=1): + model = instantiate(cfg["model"]).cuda() + n_params = sum(p.numel() for p in model.parameters()) + print(f"model params: {n_params / 1e6:.1f}M") + opt = torch.optim.Adam(model.parameters(), lr=1e-3) + + # 180^3 = 5,832,000 voxels: the exact cap from cap_filelists.py, i.e. the + # worst case any capped structure can present. With iters > 1 the first + # iteration absorbs cuDNN algorithm search (benchmark mode) and the last + # one is the steady state. + total = torch.cuda.get_device_properties(0).total_memory / 2**30 + for n in grids: + shape = (n, n, n) + torch.cuda.reset_peak_memory_stats() + x = torch.randn(1, 1, *shape, device="cuda") + times = [] + for _ in range(iters): + start = time.monotonic() + with torch.autocast("cuda", dtype=torch.bfloat16): + loss = model(x).float().square().mean() + loss.backward() + opt.step() + opt.zero_grad(set_to_none=True) + torch.cuda.synchronize() + times.append(time.monotonic() - start) + peak = torch.cuda.max_memory_allocated() / 2**30 + print( + f"grid {shape}: fwd+bwd+step {' / '.join(f'{t:.2f}s' for t in times)}, " + f"peak {peak:.1f} GiB / {total:.1f} GiB", + flush=True, + ) + del x, loss + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--config", required=True) + parser.add_argument("--n-batches", type=int, default=10) + parser.add_argument("--grids", default="128,180", help="comma-separated cube edges") + parser.add_argument("--iters", type=int, default=1, help="timed steps per grid") + parser.add_argument("--cudnn-benchmark", action="store_true") + parser.add_argument( + "--skip-data", action="store_true", help="model/memory probe only" + ) + parser.add_argument( + "--n-channels", type=int, help="override model.n_channels (width probes)" + ) + args = parser.parse_args() + + with Path(args.config).open() as f: + cfg = yaml.safe_load(f) + if args.n_channels: + cfg["model"]["n_channels"] = args.n_channels + print(f"n_channels override: {args.n_channels}") + + print("== gpu ==") + print( + torch.cuda.get_device_name(0), "capability", torch.cuda.get_device_capability(0) + ) + if args.cudnn_benchmark: + torch.backends.cudnn.benchmark = True + print("cudnn.benchmark = True") + if not args.skip_data: + print("== data ==") + check_data(cfg, args.n_batches) + print("== memory envelope ==") + check_memory(cfg, grids=[int(g) for g in args.grids.split(",")], iters=args.iters) + print("DRY RUN PASSED") + + +if __name__ == "__main__": + main() diff --git a/scripts/coreweave/gen_lrwd_sweep.py b/scripts/coreweave/gen_lrwd_sweep.py new file mode 100755 index 00000000..2ea1c717 --- /dev/null +++ b/scripts/coreweave/gen_lrwd_sweep.py @@ -0,0 +1,138 @@ +#!/usr/bin/env python3 +"""Generate per-trial configs for the LR/weight-decay width sweep. + +Derives each trial from the CoreWeave production config +(config_gga_gga+u_w96.yaml) with only these deltas: width, lr, weight_decay, +optimizer, the compressed 8-epoch schedule, the 12K sweep split files, and +per-trial run/ckpt names. Design rationale: docs/lr_wd_width_sweep.md. + +Stage A — LR grid at wd=0: + + uv run python scripts/coreweave/gen_lrwd_sweep.py --stage a + +Stage B — WD grid at the per-width best LR from stage A, plus one +(lr*/2, wd=1e-3) cross term and a verbatim rerun of the stage-A anchor to +measure run-to-run noise: + + uv run python scripts/coreweave/gen_lrwd_sweep.py --stage b \ + --best-lr 32=0.002 64=0.001 96=0.0005 + +Configs land in src/electrai/configs/MP/sweep_lrwd/; matching submit.sh +commands are printed but NOT executed. +""" + +from __future__ import annotations + +import argparse +import copy +from pathlib import Path + +import yaml + +REPO = Path(__file__).resolve().parents[2] +BASE_CONFIG = REPO / "src/electrai/configs/MP/config_gga_gga+u_w96.yaml" +OUT_DIR = REPO / "src/electrai/configs/MP/sweep_lrwd" + +WIDTHS = [32, 64, 96] +STAGE_A_LRS = [2.5e-4, 5e-4, 1e-3, 2e-3, 4e-3] +STAGE_B_WDS = [1e-4, 1e-3, 1e-2] + +SWEEP_EPOCHS = 8 +SWEEP_WARMUP = 1 +WB_PROJECT = "mp-gga-ggau-lrwd" +# Repo-relative: bundled in the submit.sh zip and read from the job workdir, +# so warm nodes whose .staged.ok marker skips the bucket sync still see them. +SWEEP_SPLITS = { + "gga+u": "data/MP/sweep_splits/gga+u_split_sweep12k.json", + "gga": "data/MP/sweep_splits/gga_split_sweep12k.json", +} + + +def fmt(x: float) -> str: + return f"{x:g}" + + +def make_trial(base: dict, width: int, lr: float, wd: float, suffix: str = "") -> str: + cfg = copy.deepcopy(base) + cfg["model"]["n_channels"] = width + cfg["lr"] = lr + cfg["weight_decay"] = wd + cfg["optimizer"] = "adamw" if wd > 0 else "adam" + cfg["epochs"] = SWEEP_EPOCHS + cfg["warmup_length"] = SWEEP_WARMUP + cfg["wb_pname"] = WB_PROJECT + for ds in cfg["data"]["datasets"]: + functional = next(f for f in SWEEP_SPLITS if f"/{f}/" in ds["root"]) + ds["split_file"] = SWEEP_SPLITS[functional] + + stem = f"sweep_lrwd_w{width}_lr{fmt(lr)}_wd{fmt(wd)}{suffix}" + cfg["run_name"] = stem + # Own stem => own ckpt namespace; a reused stem would resume from the + # earlier trial's last.ckpt in the bucket. + cfg["ckpt_path"] = f"/uv/cache/electrai/checkpoints/{stem}" + + OUT_DIR.mkdir(parents=True, exist_ok=True) + path = OUT_DIR / f"{stem}.yaml" + with path.open("w") as fp: + fp.write( + f"# Generated by scripts/coreweave/gen_lrwd_sweep.py from {BASE_CONFIG.name}\n" + ) + fp.write( + "# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits.\n" + ) + yaml.safe_dump(cfg, fp, sort_keys=False) + return stem + + +def parse_best_lr(pairs: list[str]) -> dict[int, float]: + best = {} + for pair in pairs: + w, _, lr = pair.partition("=") + best[int(w)] = float(lr) + missing = set(WIDTHS) - set(best) + if missing: + raise SystemExit(f"--best-lr missing widths: {sorted(missing)}") + return best + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--stage", choices=["a", "b"], required=True) + parser.add_argument( + "--best-lr", + nargs="*", + default=[], + metavar="WIDTH=LR", + help="stage b: per-width best LR from stage a, e.g. 32=0.002", + ) + args = parser.parse_args() + + with BASE_CONFIG.open() as fp: + base = yaml.safe_load(fp) + + stems = [] + if args.stage == "a": + stems = [ + make_trial(base, width, lr, 0.0) for width in WIDTHS for lr in STAGE_A_LRS + ] + else: + best = parse_best_lr(args.best_lr) + for width in WIDTHS: + lr = best[width] + stems.extend(make_trial(base, width, lr, wd) for wd in STAGE_B_WDS) + stems.append(make_trial(base, width, lr / 2, 1e-3)) + stems.append(make_trial(base, width, lr, 0.0, suffix="_rep2")) + + print(f"Wrote {len(stems)} configs to {OUT_DIR}\n") + print("Submit (from a clean worktree, one job per trial):") + for stem in stems: + job = "el-" + stem.removeprefix("sweep_lrwd_").replace(".", "p").replace( + "_", "-" + ) + print( + f" scripts/coreweave/submit.sh src/electrai/configs/MP/sweep_lrwd/{stem}.yaml {job}" + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/coreweave/make_sweep_split.py b/scripts/coreweave/make_sweep_split.py new file mode 100755 index 00000000..23e705e9 --- /dev/null +++ b/scripts/coreweave/make_sweep_split.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +"""Subsample the train indices of a split JSON for cheap sweep trials. + +Reads a split file as consumed by electrai.dataloader.split.split_data +({"train": [indices...], "validation": [...], ...}), draws a deterministic +random fraction of the *train* indices, and writes a new split file with +every other key (validation, test) copied through untouched — so val_loss +from sweep trials stays directly comparable to full-dataset runs. + +Run once per functional against a local copy of split_capped.json, e.g.: + + rclone copy cw:mp/chg_datasets/functionals/gga/split_capped.json /tmp/gga/ + uv run python scripts/coreweave/make_sweep_split.py \ + /tmp/gga/split_capped.json data/MP/sweep_splits/gga_split_sweep12k.json \ + --frac 0.1112 + +Using the same --frac for gga and gga+u keeps the 76/24 functional mix of +the full capped train set. +""" + +from __future__ import annotations + +import argparse +import json +import random +from pathlib import Path + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("infile", type=Path, help="source split JSON") + parser.add_argument("outfile", type=Path, help="destination split JSON") + parser.add_argument( + "--frac", + type=float, + required=True, + help="fraction of train indices to keep (0, 1]", + ) + parser.add_argument("--seed", type=int, default=42) + args = parser.parse_args() + + if not 0 < args.frac <= 1: + parser.error(f"--frac must be in (0, 1], got {args.frac}") + + with args.infile.open() as fp: + splits = json.load(fp) + + train = splits["train"] + n_keep = round(len(train) * args.frac) + rng = random.Random(args.seed) + # Sorted so epoch iteration order is index order, matching full-set splits + kept = sorted(rng.sample(train, n_keep)) + + out = {**splits, "train": kept} + args.outfile.parent.mkdir(parents=True, exist_ok=True) + with args.outfile.open("w") as fp: + json.dump(out, fp) + + other = {k: len(v) for k, v in splits.items() if k != "train"} + print(f"{args.infile} -> {args.outfile}") + print(f" train: {len(train)} -> {n_keep} (frac={args.frac}, seed={args.seed})") + print(f" passed through unchanged: {other}") + + +if __name__ == "__main__": + main() diff --git a/scripts/coreweave/run_training.sh b/scripts/coreweave/run_training.sh new file mode 100755 index 00000000..dd544f29 --- /dev/null +++ b/scripts/coreweave/run_training.sh @@ -0,0 +1,143 @@ +#!/bin/bash +# Training wrapper for Iris jobs on the CoreWeave GB200 cluster. +# +# Sequence: install rclone -> stage dataset (idempotent) -> restore last.ckpt +# from CAIOS if the node has none -> start a background checkpoint sync loop +# -> torchrun. Lightning resumes from $CKPT_DIR/last.ckpt automatically when +# present, so preempted jobs continue from at most CKPT_SYNC_S + the 20-min +# checkpoint interval behind. +# +# Env (all optional): +# CONFIG training config [config_gga_gga+u_w96.yaml] +# NPROC GPUs on this node [4] +# CKPT_DIR local checkpoint dir (must match ckpt_path in CONFIG) +# CKPT_S3 CAIOS checkpoint prefix [s3://rhoarnet-us-east-08a/checkpoints/gga_gga+u_w96] +# CKPT_SYNC_S sync interval seconds [600] +# STAGE_* see stage_data.sh +# +# AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY must hold CAIOS credentials; +# WANDB_API_KEY is required for wandb_mode: online. +set -euo pipefail + +REPO_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd) +cd "$REPO_ROOT" + +CONFIG=${CONFIG:-src/electrai/configs/MP/config_gga_gga+u_w96.yaml} +NPROC=${NPROC:-4} +# Default checkpoint locations derive from the config filename stem so that a +# different config (e.g. w128) can never accidentally resume another run's +# incompatible last.ckpt. +RUN_STEM=$(basename "$CONFIG" .yaml) +RUN_STEM=${RUN_STEM#config_} +CKPT_DIR=${CKPT_DIR:-/uv/cache/electrai/checkpoints/$RUN_STEM} +CKPT_S3=${CKPT_S3:-s3://rhoarnet-us-east-08a/checkpoints/$RUN_STEM} +CKPT_SYNC_S=${CKPT_SYNC_S:-600} +STAGE_ENDPOINT=${STAGE_ENDPOINT:-http://cwlota.com} +CKPT_REMOTE="cw:${CKPT_S3#s3://}" + +if ! command -v rclone >/dev/null 2>&1; then + RCDIR=$(mktemp -d) + case "$(uname -m)" in + x86_64) RCARCH=amd64 ;; + aarch64) RCARCH=arm64 ;; + *) + echo "run_training: unsupported arch: $(uname -m)" >&2 + exit 1 + ;; + esac + python3 - "$RCARCH" "$RCDIR" <<'PYEOF' +import io +import sys +import urllib.request +import zipfile + +url = f"https://downloads.rclone.org/rclone-current-linux-{sys.argv[1]}.zip" +zipfile.ZipFile(io.BytesIO(urllib.request.urlopen(url).read())).extractall(sys.argv[2]) +PYEOF + RCBIN=$(echo "$RCDIR"/rclone-*-linux-"$RCARCH") + chmod +x "$RCBIN/rclone" + export PATH="$RCBIN:$PATH" +fi + +# CAIOS remote, configured via env: virtual-host addressing is mandatory +export RCLONE_CONFIG_CW_TYPE=s3 +export RCLONE_CONFIG_CW_PROVIDER=Other +export RCLONE_CONFIG_CW_ENV_AUTH=true +export RCLONE_CONFIG_CW_ENDPOINT="$STAGE_ENDPOINT" +export RCLONE_CONFIG_CW_REGION=default +export RCLONE_CONFIG_CW_FORCE_PATH_STYLE=false + +bash scripts/coreweave/stage_data.sh + +# Restore for resume. Lightning versions the save_last file (last-v1.ckpt, +# last-v2.ckpt, ...) whenever an earlier run's last.ckpt already exists, but +# resume always reads last.ckpt — so fetch all last*.ckpt and promote the +# newest before launch. ALWAYS reconcile against the bucket with --update +# (copy only when the bucket file is newer): after several preemption +# bounces a retry can land on a node whose local last.ckpt is DAYS stale, +# and trusting it rolled a run back six epochs (2026-08-01, W160). +mkdir -p "$CKPT_DIR" +rclone copy "$CKPT_REMOTE" "$CKPT_DIR" --include 'last*.ckpt' --update --transfers 4 || true +NEWEST=$(ls -t "$CKPT_DIR"/last*.ckpt 2>/dev/null | head -1 || true) +if [[ -n "$NEWEST" && "$NEWEST" != "$CKPT_DIR/last.ckpt" ]]; then + cp -f "$NEWEST" "$CKPT_DIR/.last_promote_tmp" + mv -f "$CKPT_DIR/.last_promote_tmp" "$CKPT_DIR/last.ckpt" + echo "run_training: promoted $(basename "$NEWEST") -> last.ckpt" +fi + +# Guard: never resume from a checkpoint meaningfully older than the bucket's. +# A transient fetch failure above (swallowed by || true) can otherwise let a +# stale node copy win promotion — that silently rolled W128 back ~20 epochs +# on 2026-08-05. Failing hard makes the Iris retry refetch instead. +REMOTE_TS=$(rclone lsl "$CKPT_REMOTE" --include 'last.ckpt' 2>/dev/null | awk '{print $2" "$3}' | head -1) +if [[ -n "$REMOTE_TS" && -f "$CKPT_DIR/last.ckpt" ]]; then + remote_s=$(date -d "$REMOTE_TS" +%s 2>/dev/null || echo 0) + local_s=$(stat -c %Y "$CKPT_DIR/last.ckpt" 2>/dev/null || echo 0) + if ((remote_s > 0 && local_s > 0 && remote_s - local_s > 600)); then + echo "run_training: local last.ckpt is $((remote_s - local_s))s staler than bucket — refusing stale resume" >&2 + exit 1 + fi +fi +if [[ -f "$CKPT_DIR/last.ckpt" ]]; then + echo "run_training: resuming from last.ckpt" +else + echo "run_training: fresh start" +fi + +ckpt_sync() { + rclone copy "$CKPT_DIR" "$CKPT_REMOTE" --transfers 4 || true +} +# wandb runs offline (online init hits the viewer flags=null TypeError and a +# rank-0 crash deadlocks DDP); `wandb sync` uses a different API path and works +wandb_sync() { + for d in "$REPO_ROOT"/wandb/offline-run-*; do + [ -d "$d" ] && uv run --no-sync wandb sync "$d" >/dev/null 2>&1 + done + return 0 +} +(while true; do + sleep "$CKPT_SYNC_S" + ckpt_sync + wandb_sync +done) & +SYNC_PID=$! + +export PYTORCH_ALLOC_CONF=expandable_segments:True +export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True +export PYTHONPATH="$REPO_ROOT" + +uv run --no-sync torchrun --standalone --nproc_per_node="$NPROC" \ + src/electrai/entrypoints/main.py train --config "$CONFIG" & +TRAIN_PID=$! +trap 'kill -TERM "$TRAIN_PID" 2>/dev/null || true' TERM INT + +set +e +wait "$TRAIN_PID" +RC=$? +set -e + +kill "$SYNC_PID" 2>/dev/null || true +ckpt_sync +wandb_sync +echo "run_training: exited rc=$RC (final checkpoint + wandb sync done)" +exit "$RC" diff --git a/scripts/coreweave/stage_data.sh b/scripts/coreweave/stage_data.sh new file mode 100755 index 00000000..5006f46c --- /dev/null +++ b/scripts/coreweave/stage_data.sh @@ -0,0 +1,89 @@ +#!/bin/bash +# Stage the charge-density dataset from CAIOS object storage to node-local NVMe. +# +# Runs as the first step of an Iris task on the CoreWeave GB200 cluster. +# Idempotent: a marker file written after a successful sync lets preemption +# retries that land on a warm node (the cache dir is a hostPath that survives +# pod restarts) skip the download entirely. The dataset is immutable, so the +# marker is trusted unless STAGE_FORCE=1. +# +# Uses rclone: CAIOS requires virtual-host addressing for list operations, +# which s5cmd cannot emit (it is path-style only against custom endpoints). +# +# STAGE_ROOT lives under /uv/cache because that is the ONLY host-persistent +# mount Iris task pods get (hostPath /mnt/local/iris-cache/uv-cache). Writing +# anywhere else lands on the container overlay and counts against the pod's +# ephemeral-storage limit, which kills the pod mid-stage (exit 137). +# +# Env (all optional): +# STAGE_BUCKET source bucket [rhoarnet-us-east-08a] +# STAGE_PREFIX bucket prefix to mirror [mp/chg_datasets] +# STAGE_ROOT local destination root [/uv/cache/electrai] +# STAGE_ENDPOINT S3 endpoint [http://cwlota.com] +# STAGE_FORCE 1 = re-sync even if marker present +# +# AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY must hold CAIOS credentials +# (passed to the Iris job with -e; they override the cluster-injected ones). +set -euo pipefail + +STAGE_BUCKET=${STAGE_BUCKET:-rhoarnet-us-east-08a} +STAGE_PREFIX=${STAGE_PREFIX:-mp/chg_datasets} +STAGE_ROOT=${STAGE_ROOT:-/uv/cache/electrai} +STAGE_ENDPOINT=${STAGE_ENDPOINT:-http://cwlota.com} +DEST="$STAGE_ROOT/$STAGE_PREFIX" +MARKER="$DEST/.staged.ok" + +if [[ -f "$MARKER" && "${STAGE_FORCE:-0}" != "1" ]]; then + echo "stage_data: marker present ($(cat "$MARKER")), skipping sync" + exit 0 +fi + +if ! command -v rclone >/dev/null 2>&1; then + RCDIR=$(mktemp -d) + case "$(uname -m)" in + x86_64) RCARCH=amd64 ;; + aarch64) RCARCH=arm64 ;; + *) + echo "stage_data: unsupported arch: $(uname -m)" >&2 + exit 1 + ;; + esac + python3 - "$RCARCH" "$RCDIR" <<'PYEOF' +import io +import sys +import urllib.request +import zipfile + +url = f"https://downloads.rclone.org/rclone-current-linux-{sys.argv[1]}.zip" +zipfile.ZipFile(io.BytesIO(urllib.request.urlopen(url).read())).extractall(sys.argv[2]) +PYEOF + RCBIN=$(echo "$RCDIR"/rclone-*-linux-"$RCARCH") + chmod +x "$RCBIN/rclone" + export PATH="$RCBIN:$PATH" +fi + +# CAIOS remote, configured via env: virtual-host addressing is mandatory +export RCLONE_CONFIG_CW_TYPE=s3 +export RCLONE_CONFIG_CW_PROVIDER=Other +export RCLONE_CONFIG_CW_ENV_AUTH=true +export RCLONE_CONFIG_CW_ENDPOINT="$STAGE_ENDPOINT" +export RCLONE_CONFIG_CW_REGION=default +export RCLONE_CONFIG_CW_FORCE_PATH_STYLE=false + +mkdir -p "$DEST" +echo "stage_data: syncing s3://$STAGE_BUCKET/$STAGE_PREFIX -> $DEST" +start=$(date +%s) +rclone copy "cw:$STAGE_BUCKET/$STAGE_PREFIX" "$DEST" \ + --transfers 96 --checkers 128 --size-only --fast-list \ + --stats 60s --stats-one-line --log-level NOTICE + +# RhoRead resolves data/ and label/ as siblings of the filelist, so the +# functionals dirs need the same symlink shim prep_data.sh created on Lambda. +ln -sfn ../../rho_gga/data "$DEST/functionals/gga/data" +ln -sfn ../../rho_gga/label "$DEST/functionals/gga/label" +ln -sfn "../../rho_gga+u/data" "$DEST/functionals/gga+u/data" +ln -sfn "../../rho_gga+u/label" "$DEST/functionals/gga+u/label" + +n_files=$(find "$DEST" -type f | wc -l) +echo "stage_data: staged $n_files files in $(($(date +%s) - start))s" +date -u +"%Y-%m-%dT%H:%M:%SZ $n_files files" >"$MARKER" diff --git a/scripts/coreweave/submit.sh b/scripts/coreweave/submit.sh new file mode 100755 index 00000000..8ba6f417 --- /dev/null +++ b/scripts/coreweave/submit.sh @@ -0,0 +1,47 @@ +#!/bin/bash +# Submit a CoreWeave training job through Iris. Run from the LAPTOP, with the +# repo checkout you want bundled as the current directory (use a clean +# worktree: the bundle is a zip of cwd, and stray large files bloat it). +# +# Usage: +# scripts/coreweave/submit.sh [extra iris args...] +# Example: +# scripts/coreweave/submit.sh src/electrai/configs/MP/config_gga_gga+u_w128.yaml \ +# electrai-w128-gga-ggau +# +# Env overrides: +# MARIN_REPO marin checkout providing the iris client [~/code/marin] +# GPUS iris --gpu spec [GB200x4] +# CPUS/MEMORY/DISK [64 / 200GB / 60GB] +# +# Credentials are read locally (CAIOS keys from the `coreweave` AWS profile, +# W&B key from ~/.netrc) and passed as job env vars; nothing is printed. +set -euo pipefail + +CONFIG=${1:?usage: submit.sh [extra iris args...]} +JOB_NAME=${2:?usage: submit.sh [extra iris args...]} +shift 2 + +MARIN_REPO=${MARIN_REPO:-$HOME/code/marin} +GPUS=${GPUS:-GB200x4} +CPUS=${CPUS:-64} +MEMORY=${MEMORY:-200GB} +DISK=${DISK:-60GB} +MAX_RETRIES=${MAX_RETRIES:-25} + +[[ -f $CONFIG ]] || { + echo "submit: config not found in cwd bundle: $CONFIG" >&2 + exit 1 +} + +KUBECONFIG=${KUBECONFIG:-$HOME/.kube/config-coreweave} \ + uv run --project "$MARIN_REPO" --package marin-iris \ + iris --cluster=cw-us-east-08a job run \ + --enable-extra-resources --gpu "$GPUS" --cpu "$CPUS" --memory "$MEMORY" --disk "$DISK" \ + --priority batch --max-retries "$MAX_RETRIES" --job-name "$JOB_NAME" --no-wait \ + -e AWS_ACCESS_KEY_ID "$(aws configure get aws_access_key_id --profile coreweave)" \ + -e AWS_SECRET_ACCESS_KEY "$(aws configure get aws_secret_access_key --profile coreweave)" \ + -e WANDB_API_KEY "$(awk '/machine api.wandb.ai/{f=1} f && /password/{print $2; exit}' ~/.netrc)" \ + -e CONFIG "$CONFIG" \ + "$@" \ + -- bash -lc 'cd "${IRIS_WORKDIR:-/app}" && bash scripts/coreweave/run_training.sh' diff --git a/scripts/lambda/MONITOR.md b/scripts/lambda/MONITOR.md new file mode 100644 index 00000000..105d975c --- /dev/null +++ b/scripts/lambda/MONITOR.md @@ -0,0 +1,129 @@ +# Monitoring an in-flight Lambda training run + +A 100-epoch run is ~26 days. Babysitting that with a human-driven status check +is impractical, so this doc describes the lightweight monitor we've been using: +an hourly LLM-driven check using a fixed prompt + a one-shot status script. + +The companion script `monitor_status.sh` produces a single snapshot you can +read directly; the LLM prompt fires the same snapshot hourly and decides +whether anything needs attention. + +## What "healthy" looks like (the 3-part liveness rule) + +Training is healthy iff **all three** are true: + +| Signal | Threshold | What it confirms | +|---|---|---| +| (a) `train.log` mtime | within 60s of now | log file actively being written | +| (b) All GPUs (4 on lambda2, 8 on multi-node) | ≥ 80% util | trainer actively computing | +| (c) `last.ckpt` mtime | within 8h of now | per-epoch checkpoint advance | + +**Only flag a stall if at least 2 of the 3 fail.** Single-signal failures are +almost always false alarms. + +## Why each non-obvious signal exists + +- **Don't trust the visible step counter in `tail train.log`.** The log is + multi-GB and grows by ~1 MB/min. Each progress update is a `\r`-overwritten + line, but each overwrite also appends a fresh log entry. The "last 200 lines" + from a `tail` is therefore a snapshot of ~1 minute of training and looks + identical between hourly checks even when training is fine. The mtime of the + file itself is the trustworthy freshness signal. +- **GPU util is the most direct signal training is doing real work.** All four + H100s should be 90-100% busy. A dataloader stall drops them; a dead trainer + drops them; a NCCL hang drops them. +- **`last.ckpt` mtime is the per-epoch advance marker.** Lightning writes it + at the end of each validation, so it cleanly tells you "we finished epoch + N". With 6-7h epochs, 8h is the right threshold. + +## False alarms we've already learned + +- `tail` on a fast-writing log returns a snapshot that looks stale between + consecutive hourly checks. Use mtime + GPU util instead. +- `stat ` can return ENOENT on NFS/virtiofs while `ls ` of + the same path works. Don't conclude "file missing" from one `stat`. +- "Backup log entry stuck" can just mean the 10-min cadence hasn't fired yet. + Check against the loop's cadence, not against arbitrary timestamps. + +## Setup: the hourly cron monitor + +Inside a Claude Code session, the recurring task is scheduled via `CronCreate` +with this prompt (verbatim, hourly at minute `:27` to avoid herd timing): + +```text +check the Lambda H100:4 full training run for config_gga_gga+u_f32 on lambda2. + +Run on lambda2 via `ssh lambda2`: +1. `tmux ls 2>&1 | head -3` — verify `electrai-train` session alive + (expect 3 windows: train, backup, wandb-sync) +2. `date -Iseconds; stat -c '%y %n' \ + /lambda/nfs//checkpoints/train.log \ + /lambda/nfs//checkpoints//last.ckpt 2>&1` + — liveness via mtimes (NOT step counter) +3. `nvidia-smi --query-gpu=utilization.gpu --format=csv,noheader` + — all should be 90-100% +4. `tmux capture-pane -t electrai-train:train -p -S -100 2>&1 \ + | tr '\r' '\n' | grep -oE 'Epoch [0-9]+: *[0-9]+%.*it/s.*' | tail -2` + — current epoch/step/it/s from live tmux pane +5. `ls -la /lambda/nfs//checkpoints// 2>&1 | tail -6` + — checkpoint files +6. `tail -3 /lambda/nfs//checkpoints/backup.log` + — S3 ckpt backup loop (10-min cadence) +7. `tail -8 /lambda/nfs//checkpoints/wandb-sync.log` + — wandb-sync loop; expect periodic "Syncing: ... done." lines + +Liveness rule (avoid false alarms): training is healthy iff +(a) train.log mtime within 60s of now AND +(b) all GPUs >= 80% util AND +(c) last.ckpt mtime within 8h of now. +The step counter in the tail of train.log is NOT a reliable freshness +signal. Only flag a stall if at least 2 of those three signals show +>threshold staleness. + +Report concisely: +- Current epoch + step + it/s (from tmux pane) +- Latest val_loss_epoch (and delta vs previous) +- New checkpoint files since last check +- wandb-sync state: time of last "Syncing: ... done." vs now +- Any errors / restarts (training exits != 0; wandb sync failures) +- ETA estimate: epochs remaining × current epoch wall-clock + +If the strict liveness rule (a&b&c) fails OR a checkpoint hasn't +advanced for >8h OR wandb-sync has logged failures for >2 consecutive +cycles, surface the failure and consider intervention. +``` + +For multi-node (see [PORT_PLAN.md](PORT_PLAN.md) Phase 7), add a second SSH +target for the worker node and check its GPUs + tmux separately. + +## Ad-hoc usage + +```sh +bash scripts/lambda/monitor_status.sh # default: lambda2, full_gga_gga+u_f32 +bash scripts/lambda/monitor_status.sh lambda3-a # different host +bash scripts/lambda/monitor_status.sh lambda3-a smoke_gga_gga+u_f32_smoke +``` + +## Intervention thresholds + +| Symptom | Action | +|---|---| +| Liveness rule (a+b+c) fails for 1 cycle | wait a cycle — likely transient | +| Liveness rule fails for 2+ cycles | ssh in, check `ps aux | grep python`, GPU XID errors in `dmesg` | +| `last.ckpt` >8h old | ssh in, check if validation is hung; consider killing + resuming from S3 mirror | +| wandb-sync fails for 2+ cycles | check `wandb-sync.log`; if backend is down, sync will catch up later — don't intervene unless training itself is also failing | +| Worker node unreachable (multi-node only) | DDP will time out and trainer will auto-resume; verify worker GPU/tmux when reachable again | +| `nvidia-smi` reports XID errors | hardware issue; preserve state via S3, file Lambda support ticket, consider switching instances | + +## What this monitor doesn't do + +- It doesn't auto-restart anything. The `while true` resume loop in + `run_training.sh` handles trainer crashes; the wandb-sync and backup + windows have their own loops too. The monitor's job is to surface + failures the operator should investigate. +- It doesn't watch loss curves for divergence — that's wandb's job. The + monitor checks that wandb is *receiving* data, not that the data is + good. +- It doesn't track Lambda billing. Cost is roughly $11.96/hr (4×) or + $47.84/hr (16×); back-of-envelope from "epochs left × wall-clock per + epoch × $/hr". diff --git a/scripts/lambda/MONITOR_EC2.md b/scripts/lambda/MONITOR_EC2.md new file mode 100644 index 00000000..e52a46d4 --- /dev/null +++ b/scripts/lambda/MONITOR_EC2.md @@ -0,0 +1,177 @@ +# EC2 resident monitor — active operator for the Lambda run + +This is the **active** counterpart to [MONITOR.md](MONITOR.md). MONITOR.md +describes an *observe-only* hourly check (it explicitly "doesn't auto-restart +anything"). This doc describes a dedicated always-on EC2 box running a resident +Claude agent that uses the same liveness rule and snapshot script, but **also +attempts a bounded set of safe fixes on its own and escalates to Slack only when +a fix fails or the situation is outside its allowed actions.** + +Use one or the other, not both pointed at the same run. + +## Why a separate box +The MONITOR.md cron lives *inside* a Claude Code session — it needs a live +session running somewhere (a laptop). A 100-epoch run is ~26 days, so the +watcher must be durable and independent of both your laptop and the training +box. A tiny EC2 instance gives you: 24/7 uptime, co-location with +`s3://oa-electrai` (us-east-1), an IAM instance role instead of static keys, and +a blast radius separate from the GPUs it babysits. + +## Components (all in `scripts/lambda/`) +| File | Role | +|---|---| +| `monitor_status.sh` | one-shot read-only snapshot of one node (existing) | +| `monitor_status_all.sh` | wraps the above over every node in `$HOSTS` (head + worker) | +| `monitor_agent_prompt.md` | the operator prompt fed to Claude each tick | +| `monitor_loop.sh` | resident supervisor loop (run by systemd) | +| `monitor_settings.json` | Claude permission allowlist (fail-closed) | +| `monitor.env.example` | EnvironmentFile template (targets + secrets) | +| `monitor_watchdog.sh` | independent "who watches the watcher" check | +| `systemd/electrai-monitor.service` | runs the loop, `Restart=always` | +| `systemd/electrai-monitor-watchdog.{service,timer}` | runs the watchdog every 30 min | + +### How a tick works +1. The loop prepends a live **CONTEXT** header (HOSTS, RUN, journal path, + maintenance flag, snapshot command) to `monitor_agent_prompt.md`. +2. Claude reads the tail of the **journal** (its memory across ticks), runs + `monitor_status_all.sh`, judges health by MONITOR.md's 3-part liveness rule, + and either does nothing, remediates, or escalates. +3. Claude's reply is appended to the journal. Continuity is the journal, **not** + the conversation buffer — that's what makes this survivable over 26 days and + across restarts/token refreshes. +4. A successful tick touches the **heartbeat** file; the watchdog keys off that, + so a silently failing Claude (expired token, quota, network) still gets + caught even though the journal header updates every tick. + +## REMEDIATION RUNBOOK (the agent reads this section) + +The training script already self-heals transient crashes (`run_training.sh` runs +the trainer in a `while true; … sleep 30` loop and Lightning auto-resumes from +`last.ckpt`). So the agent's autonomous scope is deliberately narrow: the cases +the self-heal loop **can't** cover. Confirm a real stall with the liveness rule +(>= 2 of 3 failing) before acting. + +### Allowed autonomous actions (safe, idempotent, reversible) +| Situation | Action | +|---|---| +| `electrai-train` tmux session absent entirely | relaunch: `ssh 'cd ~/electrai && bash scripts/lambda/run_training.sh full'` (auto-resumes) | +| A loop window died (backup or wandb-sync window gone, training fine) | restart just that window in the existing session | +| Train alive but wedged >= 2 cycles (GPUs idle + `train.log` stale → NCCL hang) | kill + relaunch the train window once (`NCCL_ASYNC_ERROR_HANDLING=1` should make resume clean) | +| (multi-node) worker tmux died but host reachable | restart the worker training command | +| Single-cycle / single-signal failure | wait one tick — likely transient (per MONITOR.md) | + +### Circuit breaker +At most **2** autonomous restart actions for the same issue within a rolling +**60-minute** window — count them from your journal recall. On the 3rd +occurrence, stop remediating and **escalate** instead of thrashing. + +### Escalate-only → Slack (never autonomous) +| Situation | Why escalate | +|---|---| +| `nvidia-smi` / `dmesg` XID hardware errors | hardware; preserve state, human picks whether to switch instances (cost) | +| Worker host *unreachable* / instance down (not just tmux) | provisioning is a cost decision for a human | +| `last.ckpt` still > 8h old after one allowed restart | validation genuinely hung; needs a look | +| Anything risking checkpoint/data loss; disk full | irreversible | +| Persistent failure past the circuit breaker | stop and page | +| Its own auth failures (Claude token / wandb / AWS / SSH) | the agent can't fix itself — surfaced by the watchdog | +| wandb-sync failing while training is healthy | **log only, do not escalate** (per MONITOR.md) | + +When unsure of the root cause, do not act on the cluster — escalate. + +### Maintenance mode +During the H100:4 → H100:16 cutover (see [PORT_PLAN.md](PORT_PLAN.md)) there is a +lot of expected churn. Pause autonomous remediation: +```sh +touch ~/.config/electrai-monitor/MAINTENANCE # observe + journal only +rm ~/.config/electrai-monitor/MAINTENANCE # resume normal operation +``` + +## Provisioning the box + +### 1. Instance +- `t4g.medium` (2 vCPU / 4 GB, ARM) — no GPU; Claude runs on Anthropic's + servers, the box only SSHes/parses/loops. ~$24/mo. +- Ubuntu 24.04 LTS, **us-east-1** (co-located with `s3://oa-electrai`). +- 30 GB gp3 EBS. Allocate an **Elastic IP** (stable address for your SSH in and + for allowlisting on the Lambda boxes). +- Security group: inbound SSH (22) **from your IP only**; outbound all. + +### 2. IAM instance role (no static keys) +Attach a role with **read-only** access to the checkpoint prefix — restores +happen on the Lambda box with its own creds, so the monitor never needs write: +``` +s3:GetObject, s3:ListBucket on arn:aws:s3:::oa-electrai (prefix checkpoints/lambda/*) +``` + +### 3. Software +```sh +sudo apt-get update && sudo apt-get install -y git tmux awscli jq curl +curl -LsSf https://astral.sh/uv/install.sh | sh # optional, parity with Lambda +# Node LTS + Claude Code (per current install docs), then: +claude setup-token # interactive once — binds your Max plan (long-lived token in ~/.claude) +git clone -b betsy/gga-gga+u-f32 git@github.com:Quantum-Accelerators/electrai.git ~/electrai +``` + +### 4. SSH to the Lambda boxes (dedicated, revocable key) +Do **not** copy your personal `id_ed25519_lambda`. Mint a key just for the +monitor and authorize it on each Lambda node: +```sh +ssh-keygen -t ed25519 -f ~/.ssh/id_ed25519_monitor -N '' +# add ~/.ssh/id_ed25519_monitor.pub to ~/.ssh/authorized_keys on lambda2 (and lambda3-a/-b) +``` +Replicate the host aliases in `~/.ssh/config` (`lambda2`, `lambda3-a`, +`lambda3-b`) pointing at `id_ed25519_monitor`, then verify: +```sh +ssh lambda2 nvidia-smi +bash ~/electrai/scripts/lambda/monitor_status.sh lambda2 +``` +Consider locking the Lambda nodes' SSH inbound to the EC2 Elastic IP + your laptop. + +### 5. Config + secrets +```sh +mkdir -p ~/.config/electrai-monitor ~/.local/state/electrai-monitor +cp ~/electrai/scripts/lambda/monitor.env.example ~/.config/electrai-monitor/monitor.env +chmod 600 ~/.config/electrai-monitor/monitor.env +# edit: HOSTS, RUN, SLACK_WEBHOOK_URL, WANDB_API_KEY +``` + +### 6. Install the services +```sh +sudo cp ~/electrai/scripts/lambda/systemd/electrai-monitor*.service /etc/systemd/system/ +sudo cp ~/electrai/scripts/lambda/systemd/electrai-monitor-watchdog.timer /etc/systemd/system/ +sudo systemctl daemon-reload +sudo systemctl enable --now electrai-monitor.service +sudo systemctl enable --now electrai-monitor-watchdog.timer +journalctl -u electrai-monitor -f # or: tail -f ~/.local/state/electrai-monitor/journal.md +``` + +### 7. Smoke test (before trusting it) +1. Confirm a healthy tick lands a `STATUS: HEALTHY` block in the journal. +2. At a safe moment, `ssh lambda2 'tmux kill-session -t electrai-train'` and + confirm the agent detects the stall, relaunches the run, and logs `STATUS: + ACTED`. Watch that the trainer resumes from `last.ckpt`. +3. Force an escalate-only case (e.g. set `HOSTS` to an unreachable host) and + confirm a Slack message arrives. +4. Stop `electrai-monitor.service` and confirm the watchdog Slacks within its + interval. + +## Cost +~$24/mo instance + ~$3/mo EBS + negligible egress. Claude runs on your Max +subscription token, so no per-token API bill — but a continuous loop on Opus +consumes Max usage; tune `MONITOR_INTERVAL` (default 15 min) and `MONITOR_MODEL` (default +Opus) if you bump Max limits. Trivial next to the ~$11.96/hr (4×) / ~$47.84/hr +(16×) training spend. + +## Multi-node cutover checklist +1. `touch ~/.config/electrai-monitor/MAINTENANCE` before the cutover. +2. Edit `monitor.env`: `HOSTS="lambda3-a lambda3-b"`, `RUN="full_gga_gga+u_f32_16x"`. +3. Add `lambda3-a` / `lambda3-b` to `~/.ssh/config` and authorize the monitor key. +4. (Already in `monitor_settings.json`: ssh allow entries for both.) +5. `sudo systemctl restart electrai-monitor` and confirm a healthy multi-node tick. +6. `rm ~/.config/electrai-monitor/MAINTENANCE` to re-enable remediation. + +## Open items before go-live +- **Sign-off on the allowed-action list above** — this is the one place the + monitor touches a live, expensive run. +- Confirm the Max-plan token survives unattended (the watchdog catches expiry, + but plan for periodic `claude setup-token` refresh). diff --git a/scripts/lambda/MULTINODE.md b/scripts/lambda/MULTINODE.md new file mode 100644 index 00000000..60152b13 --- /dev/null +++ b/scripts/lambda/MULTINODE.md @@ -0,0 +1,302 @@ +# Multi-node Lambda training runbook (`scripts/lambda/MULTINODE.md`) + +Drop-in 2-node companion to the single-node runbook (`README.md`). Same data +layout, same configs, same tmux session structure -- just two Lambda +instances coordinating over Ethernet via `torchrun` + NCCL. + +The training entrypoint (`src/electrai/entrypoints/train.py`) already reads +`LOCAL_WORLD_SIZE` / `WORLD_SIZE` from the environment and derives +`num_nodes` for the Lightning Trainer. We don't touch model code; this is +purely a launcher + ops doc. + +> **Heads-up.** Each Lambda instance has its own `/lambda/nfs/` +> filesystem -- they are NOT cross-mounted. Data and checkpoints live on +> each node's local NVMe / NFS independently. Only the S3 checkpoint backup +> on the head node is shared state. + +## Topology + +``` + 29500/tcp + NCCL data + Lambda instance A <-----------------------> Lambda instance B + (head, NODE_RANK=0) 100 Gbps eth (worker, NODE_RANK=1) + 8 x H100 80GB SXM 8 x H100 80GB SXM + /lambda/nfs/ /lambda/nfs/ + /home/ubuntu (NVMe) /home/ubuntu (NVMe) +``` + +- **Rendezvous:** torchrun static (`--master_addr` + `--master_port=29500`). + Simpler than `c10d` for two known-IP nodes and matches the runbook's + "head node IP" mental model. Swap for `--rdzv_backend=c10d` if/when we + want elastic worker restarts (not needed today). +- **Global batch size goes from 4 -> 16.** The Lightning `Trainer` sees + `devices="auto"` (= 8) and `num_nodes=2`. See "LR scaling" below. + +--- + +## Pre-flight checklist (in order) + +### 1. Provision a second Lambda instance + +Pick the same `gpu_8x_h100` SKU. After it boots, SSH in as `ubuntu@`. + +### 2. One-time env setup on the new node + +Same as single-node: + +```bash +curl -fsSL https://raw.githubusercontent.com/Quantum-Accelerators/electrai/betsy/gga-gga+u-f32/scripts/lambda/setup.sh | bash +``` + +Then, on the new node: + +```bash +aws configure # use the betsy-della creds (read + ckpt write) +echo "export WANDB_API_KEY='...'" >> ~/.bashrc +``` + +### 3. Sync data to the new node's NVMe (~1-3 h, parallel to step 4 below) + +```bash +bash ~/electrai/scripts/lambda/data_sync.sh +bash ~/electrai/scripts/lambda/prep_data.sh +``` + +Each node needs its own full copy of the data on its own NVMe. There is +no cross-node sharing. + +### 4. (Head node, in parallel) stage a known-good checkpoint on the head + +If you're switching mid-run from the existing 4 GPU run, first cleanly stop +training so `last.ckpt` is consistent: + +```bash +# inside the existing electrai-train tmux session on the head node: +tmux send-keys -t electrai-train:train C-c # graceful stop +# wait for Lightning to checkpoint + exit; verify last.ckpt timestamp is fresh +ls -lh $CKPT_ROOT/last.ckpt +``` + +Then sync the checkpoint to the worker so any rank can resume: + +```bash +# head node -> worker node (run on head): +rsync -avz $CKPT_ROOT/last.ckpt ubuntu@:$CKPT_ROOT/ +``` + +(Lightning only restores from `last.ckpt` on the global-rank-0 process, so +strictly the worker doesn't need it -- but having both copies makes +disaster recovery cheaper.) + +### 5. Find each node's PRIVATE / internal IP + +We need the IP that the OTHER node actually routes packets to -- usually a +private 10.x or 172.x address on the 100 Gbps NIC, NOT the public NAT'd +address Lambda lists in their console. + +On each node: + +```bash +# show all interfaces and their addresses +ip -brief addr + +# the IP you want is on the high-bandwidth NIC (look for the 100Gb one, +# usually enp* or ens* with a 10.x or 172.x address) +``` + +Pick the head node's private IP -- call it `HEAD_IP` for the rest of this +doc. Both nodes will use it as `MASTER_ADDR`. + +### 6. Network connectivity check + +From the **worker** node: + +```bash +ping -c 3 $HEAD_IP # latency sanity (should be < 1 ms) +nc -zv $HEAD_IP 29500 # MASTER_PORT reachable? +``` + +If `nc` fails, check Lambda's per-instance firewall (cloud-init may have +left iptables open by default, but verify with `sudo iptables -L -n`). + +### 7. Find the right `NCCL_SOCKET_IFNAME` + +The launcher auto-detects this via `ip route get $MASTER_ADDR`. To +double-check manually on the worker: + +```bash +ip -o route get $HEAD_IP | awk '{for(i=1;i<=NF;i++)if($i=="dev")print $(i+1)}' +``` + +That prints the interface name (e.g. `enp25s0`, `ens9`). If it's `lo`, +`docker0`, or a vlan, something's wrong -- pass `NCCL_SOCKET_IFNAME=...` +explicitly to the launcher. + +### 8. NCCL allreduce smoke test (strongly recommended) + +Validates that NCCL can actually talk across the two nodes -- way faster +to debug here than inside Lightning. + +```bash +# On HEAD: +NODE_RANK=0 MASTER_ADDR=$HEAD_IP NCCL_IB_DISABLE=1 NCCL_DEBUG=INFO \ + uv run torchrun --nnodes=2 --node_rank=0 --nproc_per_node=8 \ + --master_addr=$HEAD_IP --master_port=29500 \ + scripts/lambda/nccl_test.py + +# On WORKER (within ~60s of starting the head): +NODE_RANK=1 MASTER_ADDR=$HEAD_IP NCCL_IB_DISABLE=1 NCCL_DEBUG=INFO \ + uv run torchrun --nnodes=2 --node_rank=1 --nproc_per_node=8 \ + --master_addr=$HEAD_IP --master_port=29500 \ + scripts/lambda/nccl_test.py +``` + +Expect 16 lines of the form `[rank N/16 local=M] allreduce got=120.0 +expected=120.0 ok=True` (sum of `0..15` is 120). + +If it hangs at "initializing process group", see "Diagnostics" below. + +--- + +## Smoke test (200 samples, 2 epochs) + +Same config as single-node smoke, but with `WANDB_MODE_OVERRIDE=disabled` +to keep multi-node tests out of wandb. + +On both nodes, start within ~60 s of each other: + +```bash +# HEAD: +NODE_RANK=0 MASTER_ADDR=$HEAD_IP WANDB_MODE_OVERRIDE=disabled \ + bash ~/electrai/scripts/lambda/run_training_multinode.sh smoke + +# WORKER: +NODE_RANK=1 MASTER_ADDR=$HEAD_IP WANDB_MODE_OVERRIDE=disabled \ + bash ~/electrai/scripts/lambda/run_training_multinode.sh smoke +``` + +Validate: + +- `tmux attach -t electrai-train` on each node shows training advancing. +- Head log shows `world_size=16, num_nodes=2` in Lightning's startup banner. +- Loss curves match the 4 GPU smoke (within stochasticity); if they don't, + something is wrong with DDP gradient sync. +- ~half the steps per epoch vs single-node 8 GPU (since each rank sees the + same per-rank batch but global throughput is 2x). + +--- + +## Full run + +Once smoke passes, on both nodes (within ~60 s of each other): + +```bash +# HEAD: +NODE_RANK=0 MASTER_ADDR=$HEAD_IP WANDB_MODE_OVERRIDE=offline \ + bash ~/electrai/scripts/lambda/run_training_multinode.sh full + +# WORKER: +NODE_RANK=1 MASTER_ADDR=$HEAD_IP WANDB_MODE_OVERRIDE=offline \ + bash ~/electrai/scripts/lambda/run_training_multinode.sh full +``` + +The head node also runs the `backup` and `wandb-sync` tmux windows +(workers skip them; only global rank 0 writes checkpoints, and that +process lives on the head). + +--- + +## LR scaling (open question, flag for review) + +Going from 4 -> 16 GPUs means global batch size goes from 4 -> 16 +(`batch_size=1` per rank, see `config_gga_gga+u_f32.yaml`). + +The current config has `lr: 0.001`. Linear-scaling rule says `lr * 4 = +0.004`; square-root rule says `lr * 2 = 0.002`. The conservative starting +point is somewhere between. + +**Recommendation: start with `lr = 0.0025` (2.5x), run 1 epoch, compare +val loss against the 4 GPU baseline at the same epoch.** If loss is +clearly higher than the 4 GPU run's epoch 1, drop back to `lr * 2`. If +loss matches or improves, hold there for a few epochs, then optionally +push to `lr * 3` for the rest of the campaign. + +To override the LR without editing the committed config: + +```bash +# easiest: copy the config and edit lr in the copy +cp src/electrai/configs/MP/config_gga_gga+u_f32.yaml \ + src/electrai/configs/MP/config_gga_gga+u_f32_2node.yaml +sed -i 's/^lr: .*/lr: 0.0025/' \ + src/electrai/configs/MP/config_gga_gga+u_f32_2node.yaml +# then point SRC_CFG at the new file (or add a "full2node" MODE). +``` + +This is the one piece of config that genuinely needs a 1-epoch +verification before committing to the full 100-epoch campaign. Don't +skip it. + +--- + +## Switching from the in-flight 4 GPU run + +1. Verify what's running: + `tmux ls -t electrai-train` on `lambda2`. +2. Stop training gracefully so `last.ckpt` is consistent: + `tmux send-keys -t electrai-train:train C-c`, then wait for + `last.ckpt` mtime to update and the training process to exit. +3. (Optional, recommended) Bump the S3 backup once so the worker can + pull from there if the head dies: + `aws s3 sync $CKPT_ROOT s3://oa-electrai/checkpoints/lambda/` +4. Provision the second instance, follow steps 2-7 above. +5. Run the smoke test (steps "Smoke test" above) to validate cross-node + DDP works on this hardware pairing. +6. Launch the full multi-node run. + +The same `last.ckpt` resumes -- Lightning's `ckpt_path=` argument in +`train.py` is symmetric across world sizes (DDP just shards the loaded +state across whatever ranks are present). + +--- + +## Diagnostics (common multi-node failures) + +| Symptom | Likely cause | Fix | +|---|---|---| +| Hangs at "initializing process group", no NCCL output yet | torchrun rendezvous failed (TCP store). | Check `MASTER_ADDR` is reachable; `nc -zv $MASTER_ADDR 29500` from worker. | +| `NCCL WARN ... connection closed by remote peer` followed by hang | NCCL picked the wrong NIC (e.g. docker0 or a vlan). | Set `NCCL_SOCKET_IFNAME=` explicitly. Use the iface from `ip route get $MASTER_ADDR`. | +| `NCCL timeout` after some training steps | Stragglers / one rank stuck (e.g. dataloader). | Inspect each rank's stack with `py-spy dump --pid `. Often a single bad zarr file. | +| `world_size` doesn't match `nnodes * nproc_per_node` | One side launched with the wrong `--nnodes`. | Check that both launchers used `NUM_NODES=2`. | +| `RuntimeError: Address already in use` on head | A previous run left a torchrun TCPStore on 29500. | `pkill -f torchrun; sleep 3` then relaunch. Or pick a different `MASTER_PORT`. | +| Workers exit instantly with rank 0 unreachable | Wrong `MASTER_ADDR` (used the public IP instead of private). | Re-run step 5 above to pick the private IP. | +| Loss curves diverge between single-node and multi-node | Effective batch size changed; LR not scaled. | See "LR scaling" above. | +| `NCCL: failed to bind to interface` | Firewall blocking 29500 or NCCL's chosen port range. | `sudo iptables -L -n`; on Lambda's default cloud-init image this should be open, but verify. | +| Only one node prints "initializing process group" | Static rendezvous timed out (>~5 min) waiting for the late joiner. | Restart both within ~60 s of each other. | + +--- + +## Tunables (env vars) + +Inherited from `run_training.sh`: + +| var | default | what | +|---|---|---| +| `REPO_DIR` | `~/electrai` | repo location | +| `DATA_ROOT` | `$NFS_ROOT/data` | local data root | +| `CKPT_ROOT` | `$NFS_ROOT/checkpoints` | local checkpoint dir | +| `S3_CKPT_BUCKET` | `oa-electrai` | S3 bucket for ckpt backups | +| `S3_CKPT_PREFIX` | `checkpoints/lambda-multinode` | ckpt backup prefix | +| `CKPT_BACKUP_S` | `600` | seconds between backups | +| `TMUX_SESSION` | `electrai-train` | tmux session name | +| `WANDB_MODE_OVERRIDE` | (unset) | override `wandb_mode` (`offline` / `disabled`) | + +Multi-node specific: + +| var | default | what | +|---|---|---| +| `NODE_RANK` | (required) | 0 for head, 1+ for workers | +| `MASTER_ADDR` | (required) | head node's private IP, reachable from all workers | +| `NUM_NODES` | `2` | total node count | +| `NPROC_PER_NODE` | `8` | GPUs per node | +| `MASTER_PORT` | `29500` | torchrun rendezvous port | +| `NCCL_SOCKET_IFNAME` | (auto) | NIC for NCCL; autodetect via `ip route get $MASTER_ADDR` | diff --git a/scripts/lambda/PORT_PLAN.md b/scripts/lambda/PORT_PLAN.md new file mode 100644 index 00000000..3be04d91 --- /dev/null +++ b/scripts/lambda/PORT_PLAN.md @@ -0,0 +1,265 @@ +# Plan: port the in-flight H100:4 run to H100:16 (2 nodes × 8 GPUs) + +Companion to [MULTINODE.md](MULTINODE.md) — that file is the static operator +runbook; this file is the migration plan for taking a *running* 4× campaign +and porting it onto a 2-node 16-GPU cluster with the lowest risk and the +earliest abort signal. + +Status at the time of writing: H100:4 on `lambda2`, ~48h in, ~epoch 7, +`ckpt_epoch=06_val_loss=0.011800.ckpt` saved. Multi-node scaffolding lives +in `scripts/lambda/{run_training_multinode.sh, MULTINODE.md, nccl_test.py}`. + +## Why this is risky enough to need a plan + +Going from 1 node to 2 changes several things at once: + +- **DDP world size 4 → 16**, so the effective batch size goes 4 → 16 (bs=1 + per rank). Need to scale `lr`. +- **All-reduce hops over Ethernet**, not NVLink. Cheap for our 12M-param + model in theory, but NCCL config is hardware-specific. +- **Doubled failure surface** — two instances means ~2× the probability of + hardware/network failure stopping the run. +- **Capacity uncertain** — Lambda has been showing 8× H100 SXM unavailable + on demand. Provisioning two simultaneously may take time. + +Goal: don't commit to 16× until we have measured proof it's worth it. + +## Decision gates + +> **Smoke gate:** the 16× smoke must hit **≥3.5× the 4× run's +> samples/sec** (≥15.4 samples/sec ≈ ≥3.85 it/s at world-size 16). Below +> 2.5× → abort. +> +> **LR gate:** one 16× epoch resumed from `last.ckpt` with a candidate `lr` +> must give a `val_loss` no worse than 1.5× the 4× run's value at the same +> epoch index. If it spikes or plateaus, try a different `lr` (cheap; one +> epoch each). + +Hitting both gates before cutover bounds the downside of a bad migration. + +## Phase 1 — Provision (~30 min) + +1. Spin up two Lambda 8× H100 SXM instances. Call them `lambda3-a` and + `lambda3-b`. Configure `~/.ssh/config` aliases (Identity, ProxyJump if + needed, ForwardAgent yes). +2. On each: clone the repo, check out `betsy/gga-gga+u-f32`, run + `scripts/lambda/setup.sh`. +3. On each: `aws configure` (region `us-east-1`) and append + `WANDB_API_KEY` to `~/.bashrc`. +4. Verify each has a `/lambda/nfs/...` mount and ≥1.4 TB free NVMe. + +## Phase 2 — Data on both nodes (~3–4h, parallel) + +Each node needs its own NVMe copy — NFS is too slow (we measured 4× drop in +throughput on the current run). In two parallel terminals: + +```sh +ssh -A lambda3-a 'cd ~/electrai && tmux new-session -d -s data-sync \ + "bash scripts/lambda/data_sync.sh 2>&1 | tee -a ~/data_sync.log"' +ssh -A lambda3-b 'cd ~/electrai && tmux new-session -d -s data-sync \ + "bash scripts/lambda/data_sync.sh 2>&1 | tee -a ~/data_sync.log"' +``` + +When each completes (`OK.` line in `~/data_sync.log`), run on **both**: + +```sh +DATA_ROOT=~/data bash ~/electrai/scripts/lambda/prep_data.sh +``` + +Do the Phase 3 network prep in parallel while the data syncs. + +## Phase 3 — Network & NCCL validation (~30 min) + +1. **Find the head node's reachable IP** (private/internal, not the public + NAT'd address): + + ```sh + ssh lambda3-a 'ip -4 addr show | grep -A1 -E "(ens|enp|eth)" | grep inet' + ``` + + Pick the 10.x or 172.x address. Set `MASTER_ADDR`. + +2. **Confirm worker → head reach on port 29500**: + + ```sh + ssh lambda3-a 'nc -l 29500' & + ssh lambda3-b "nc -zv $MASTER_ADDR 29500" + ``` + + Should succeed. If blocked, open the port or pick another and set + `MASTER_PORT`. + +3. **Find the right NCCL socket interface** (from the worker): + + ```sh + ssh lambda3-b "ip route get $MASTER_ADDR" \ + | awk '{for (i=1;i<=NF;i++) if ($i=="dev") print $(i+1)}' + ``` + + Set `NCCL_SOCKET_IFNAME` to that interface name. + +4. **Run the NCCL smoke** — `scripts/lambda/nccl_test.py` does a + cross-node all-reduce. Expect <30s success. + + If it hangs or NCCL errors out, **fix here, not in real training** — + debugging NCCL while losing real training time is much harder. Common + fixes: wrong `NCCL_SOCKET_IFNAME`, wrong IP family, firewall on the + NCCL ephemeral port range. + +## Phase 4 — Multi-node smoke (~20 min) + +Goal: validate the loader + DDP topology + throughput **before** committing. + +```sh +# on head, separate ssh session per node +ssh -A lambda3-a 'cd ~/electrai && DATA_ROOT=~/data NUM_NODES=2 NODE_RANK=0 \ + MASTER_ADDR=10.x.x.x WANDB_MODE_OVERRIDE=disabled \ + bash scripts/lambda/run_training_multinode.sh smoke' + +# on worker, same time +ssh -A lambda3-b 'cd ~/electrai && DATA_ROOT=~/data NUM_NODES=2 NODE_RANK=1 \ + MASTER_ADDR=10.x.x.x WANDB_MODE_OVERRIDE=disabled \ + bash scripts/lambda/run_training_multinode.sh smoke' +``` + +What to measure: + +- World size = 16 in the Lightning startup log (`Initializing distributed: + GLOBAL_RANK: 0, MEMBER: 1/16`) +- Per-step `it/s` from the live tmux pane on the head node +- `val_loss` after 2 epochs (expect ~0.10, comparable to the 4× smoke) + +**Decision point #1 — smoke gate.** If samples/sec ≥ 15.4 (i.e. +it/s × 16 ≥ 15.4), continue. If not, abort and stay on 4×. + +## Phase 5 — LR-scaling 1-epoch validation (~3h, recommended) + +Going 4 → 16 ranks quadruples the effective batch. Standard scaling is +either √4 = 2× `lr` or linear 4× `lr`. The config is at `lr: 0.001`. + +1. **Copy `last.ckpt` from the 4× run to the head node's NFS:** + + ```sh + ssh lambda3-a 'mkdir -p /lambda/nfs//checkpoints/full_gga_gga+u_f32 \ + && aws s3 cp s3://oa-electrai/checkpoints/lambda/full_gga_gga+u_f32/last.ckpt \ + /lambda/nfs//checkpoints/full_gga_gga+u_f32/last.ckpt' + ``` + +2. **Make a candidate config** at `configs/MP/config_gga_gga+u_f32_16x.yaml` + based on the full config, with: + - `lr: 0.0025` (2.5×, the geometric midpoint of √ and linear) + - `run_name: gga_gga+u_f32_16x` (fresh wandb run) +3. **Run 1 epoch** from that ckpt using the new config across both nodes. +4. **Compare** the resulting `val_loss` to the 4× trajectory at the same + epoch index. + +**Decision point #2 — LR gate.** If `val_loss` is monotonically descending +and within 50% of the 4× value at the same epoch, commit. If it spikes or +plateaus higher, try `lr: 0.004` (4×, linear) or `lr: 0.0015` (1.5×) — each +is a cheap 1-epoch experiment. + +## Phase 6 — Cutover (~10 min) + +1. **Stop the 4× run cleanly.** Two options: + - Wait for current epoch to end. Worst-case ~6h cost; lossless. + Recommended. + - Hard-stop now (loses progress within the current epoch; Lightning + resumes from the last completed-epoch ckpt anyway). + + ```sh + ssh lambda2 'tmux kill-session -t electrai-train' + ``` + +2. Wait for the lambda2 wandb-sync window to flush the final state, then + `tmux kill-session -t electrai-train` on lambda2 if not already. +3. **Launch real 16× training** with the chosen `lr`: + + ```sh + # head + ssh -A lambda3-a 'cd ~/electrai && DATA_ROOT=~/data NUM_NODES=2 NODE_RANK=0 \ + MASTER_ADDR=10.x.x.x WANDB_MODE_OVERRIDE=offline \ + bash scripts/lambda/run_training_multinode.sh full' + # worker + ssh -A lambda3-b 'cd ~/electrai && DATA_ROOT=~/data NUM_NODES=2 NODE_RANK=1 \ + MASTER_ADDR=10.x.x.x WANDB_MODE_OVERRIDE=offline \ + bash scripts/lambda/run_training_multinode.sh full' + ``` + +4. Confirm the head node has 3 tmux windows: train, backup, wandb-sync. + Worker has 1: train. +5. Tear down lambda2 — but leave it up for 30 min as a fallback in case + the 16× run shows an early problem. + +## Phase 7 — Monitoring the 16× run + +Most of the existing hourly cron monitor transfers with minor edits: + +| Aspect | H100:4 (lambda2) | H100:16 (lambda3-a head + lambda3-b worker) | +|---|---|---| +| ssh target | `lambda2` | `lambda3-a` (head; canonical state lives here) | +| GPU check | 4 GPUs ≥ 80% | 8 GPUs on each node ≥ 80% | +| Per-step check | `tmux capture-pane -t electrai-train:train` | same, on head | +| Checkpoints | local NFS + S3 | head's NFS + S3 (worker writes nothing) | +| wandb-sync | on lambda2 | on head (lambda3-a) | +| **Extra check** | n/a | also `ssh lambda3-b` to confirm worker tmux + GPUs alive | + +### New failure modes + +- **Worker rank disappears** → DDP times out → head ranks exit → auto-resume + starts back from `last.ckpt`. Cron should ssh the worker every cycle and + confirm its training tmux + nvidia-smi are alive. Wasted time if the + worker is dead a long time. +- **NCCL silent hang** — training stops moving but no error. Catch via the + existing liveness rule (train.log mtime stale + GPUs idle). With + `NCCL_ASYNC_ERROR_HANDLING=1` already set in the script, this should + surface as a clean exception → auto-resume. +- **wandb run id changes** from `8yzlii32` to whatever the 16× run's + `offline-run-*` directory becomes. Update the wandb-sync log path + cron + monitor to the new offline dir on the head. + +## Concerns / risks (ranked by likelihood × impact) + +1. **HIGH — Multi-node networking misconfig.** + `NCCL_SOCKET_IFNAME` and the routable IP require real-hardware + verification. **Do not skip the Phase 3 NCCL smoke.** Mitigation: + `NCCL_DEBUG=INFO` on first run, ready to revisit. +2. **HIGH — Throughput doesn't scale enough.** + <2.5× of current and we just paid setup time + a few hours of double + compute for nothing. The decision gates above catch this in <8h + instead of mid-campaign. +3. **MED — LR scaling lands wrong on first try.** + Multi-epoch validation is too expensive; one-epoch is what we can + afford. If the model diverges, worst case is one bad epoch (~$300) + before we revert. +4. **MED — Doubled instance failure surface.** + Two H100:8 instances means ~2× the probability of one dying mid-run. + S3 checkpoint backup already mitigates total data loss; auto-resume + handles single failures. We lose at most the in-progress epoch. +5. **MED — Lambda capacity for the 2nd instance.** + Single-node H100:8 has been intermittently unavailable. + File a Lambda support ticket in parallel with this plan to gauge + timing. +6. **LOW — wandb dashboard discontinuity.** + New run id; the loss curve will look like a fresh run starting at the + ckpt-6 loss value. Paper over with wandb run groups, or just accept it. +7. **LOW — NFS visible only on head.** + Worker doesn't need NFS (only rank-0 writes). Already handled in the + multi-node script. +8. **LOW — Cost overrun in setup phase.** + Even with everything going wrong, 24h of double-instance dead time + ≈ $575. Acceptable. + +## When to actually do this + +Don't migrate until at least one of these is true: + +- Lambda confirms 2× H100:8 instances are available simultaneously +- val_loss curve on H100:4 plateaus enough that we expect to *need* the + full 100 epochs (right now it's monotonic, dropping ~5% epoch-over-epoch + — could plateau at any epoch) +- A specific deadline (e.g. April 2026 funding milestone) forces our hand + +If we hit any of these, this plan is ~12h end-to-end from "instances +provisioned" to "real training resumed at 16×". The pre-drafted scripts +handle most of the mechanics; the human work is provisioning, NCCL config +tuning, and the LR validation read. diff --git a/scripts/lambda/README.md b/scripts/lambda/README.md new file mode 100644 index 00000000..03c50e12 --- /dev/null +++ b/scripts/lambda/README.md @@ -0,0 +1,124 @@ +# Lambda training runbook (`scripts/lambda/`) + +End-to-end runbook for training `config_gga_gga+u_f32` on a Lambda Cloud GPU +instance (4× or 8× H100). Parallel to the Modal pipeline under `modal/` but +adapted for a reserved VM: data lives on local NVMe, training runs in tmux, +checkpoints back up to S3 hourly. + +The four scripts are idempotent and can be re-run safely. + +## Prerequisites + +- Lambda H100 (or A100) instance, SSH'able as `ubuntu@…`. +- AWS credentials with **read** access to `s3://oa-electrai/mp/chg_datasets/*` + (the `electrai-modal-reader` IAM user works; or a Lambda-specific read-only + user in the same AWS account). +- Optional but recommended: AWS credentials with **write** to a checkpoint + prefix like `s3://oa-electrai/checkpoints/lambda/` so training checkpoints + back up off-instance. +- A `WANDB_API_KEY` for `PrinceOA` (any teammate's key works). + +## Step 0 — Boot the instance and SSH in + +```bash +ssh ubuntu@ # or `ssh lambda` if you set up an SSH config +``` + +## Step 1 — One-time env setup + +```bash +# inside the Lambda instance +curl -fsSL https://raw.githubusercontent.com/Quantum-Accelerators/electrai/betsy/gga-gga+u-f32/scripts/lambda/setup.sh | bash +# or, if you've already cloned the repo: +# cd ~/electrai && bash scripts/lambda/setup.sh +``` + +The script installs `uv`, `aws`, and `tmux` (if missing), clones the repo at +the right branch, runs `uv sync`, and validates credentials. If AWS or wandb +isn't configured, it prints the exact next command. + +After setup: +- `aws configure` (or export `AWS_ACCESS_KEY_ID` / `AWS_SECRET_ACCESS_KEY`). +- `export WANDB_API_KEY='...'` in `~/.bashrc`. + +## Step 2 — Sync data from S3 (~1–3 h) + +```bash +bash ~/electrai/scripts/lambda/data_sync.sh +``` + +Pulls the packed `.zarr.zip` tree (~1.2 TiB, ~226K objects) from +`s3://oa-electrai/mp/chg_datasets/` to `~/data/mp/chg_datasets/`. Idempotent; +safe to re-run if interrupted. + +## Step 3 — Wire data/label symlinks + smoke filelists + +```bash +bash ~/electrai/scripts/lambda/prep_data.sh +``` + +Recreates the `functionals/{gga,gga+u}/{data,label}` symlinks into the +`rho_*` real dirs, writes `mp_filelist_smoke.txt` (first 200 ids) per +functional, and sanity-checks that the first id resolves to a real +`.zarr.zip`. + +## Step 4 — Smoke training + +```bash +bash ~/electrai/scripts/lambda/run_training.sh smoke +``` + +Launches under tmux session `electrai-train`. Validates the end-to-end +pipeline (DDP launches, ZipStore loads, wandb logs, checkpoints write) on +400 samples × 2 epochs in ~10 min. Throughput on Lambda's local NVMe is +the actual number we use to predict the full-run cost. + +Watch: +```bash +tmux attach -t electrai-train +# or +tail -f ~/checkpoints/train.log +``` + +## Step 5 — Full training run + +```bash +bash ~/electrai/scripts/lambda/run_training.sh full +``` + +Same tmux pattern, but the *full* config (113K samples × 100 epochs). The +training loop auto-resumes from `last.ckpt` if a previous run crashed, and a +sibling tmux window backs the checkpoint dir up to S3 every 10 min so even +a catastrophic instance loss only sets us back at most one window. + +ETA on 4× H100, given the Modal smoke (1.28 it/s at H100:8 → ~5 samples/sec +on H100:4), is ~22 days for 100 epochs — over the 2-week ceiling. Plan for +either fewer epochs, more GPUs, or accepting a longer wall-clock. The +Lambda smoke result decides. + +## Tunables (env vars) + +| var | default | what | +|---|---|---| +| `REPO_DIR` | `~/electrai` | repo location | +| `DATA_ROOT` | `~/data` | local data root | +| `CKPT_ROOT` | `~/checkpoints` | local checkpoint dir | +| `BRANCH` | `betsy/gga-gga+u-f32` | branch to check out | +| `S3_BUCKET` | `oa-electrai` | S3 source bucket | +| `S3_PREFIX` | `mp/chg_datasets` | prefix on S3 | +| `S3_CKPT_BUCKET` | `oa-electrai` | S3 bucket for ckpt backups | +| `S3_CKPT_PREFIX` | `checkpoints/lambda` | ckpt backup prefix | +| `CKPT_BACKUP_S` | `600` | seconds between backups | +| `SMOKE_N` | `200` | ids per functional in smoke filelists | +| `TMUX_SESSION` | `electrai-train` | tmux session name | + +## Recovery + +If the instance dies mid-run: +1. Boot a fresh Lambda H100, run `setup.sh`. +2. **Skip data_sync.sh and prep_data.sh** if the data is already on this + instance's NVMe (re-runs are idempotent but cost time). If on a new + instance, run them. +3. Pull the latest checkpoint from S3: + `aws s3 sync s3://oa-electrai/checkpoints/lambda/ ~/checkpoints/` +4. Re-run `run_training.sh full` — Lightning resumes from `last.ckpt`. diff --git a/scripts/lambda/cap_filelists.py b/scripts/lambda/cap_filelists.py new file mode 100755 index 00000000..0f20b0a3 --- /dev/null +++ b/scripts/lambda/cap_filelists.py @@ -0,0 +1,96 @@ +#!/usr/bin/env python3 +"""Generate a grid-size-capped filelist + remapped split for a functional dir. + +Why: the W64 fp16 full-MP run crash-loops because the largest structures +(charge-density grids up to ~540^3 vs a ~109^3 median) make a single DDP step +exceed the 30-min NCCL watchdog timeout (and approach the 80 GB OOM ceiling). +Capping the grid size drops that tail (~1.5% of structures at ~180^3). + +Reads /mp_filelist.txt and /split.json. split.json stores +POSITIONAL indices into the filelist, so filtering the filelist requires +remapping the split indices — done here. + +Writes /mp_filelist_.txt and /split_.json. + +Usage: + python cap_filelists.py [--cap-voxels N] [--suffix capped] [--workers 32] +""" + +from __future__ import annotations + +import argparse +import json +from multiprocessing import Pool +from pathlib import Path + +import zarr + + +def _voxels(args): + root, i, idx = args + d = Path(root) / "data" + zp, dp = d / f"{idx}.zarr.zip", d / f"{idx}.zarr" + try: + if zp.exists(): + s = zarr.storage.ZipStore(str(zp), mode="r") + try: + shp = zarr.open_group(s, mode="r")["charge_density_total"].shape + finally: + s.close() + elif dp.exists(): + shp = zarr.open_group(str(dp), mode="r")["charge_density_total"].shape + else: + return (i, None) + nx, ny, nz = (int(x) for x in shp) + return (i, nx * ny * nz) + except Exception: + return (i, None) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("root") + ap.add_argument("--cap-voxels", type=int, default=5_832_000) # ~180^3 + ap.add_argument("--suffix", default="capped") + ap.add_argument("--workers", type=int, default=32) + a = ap.parse_args() + + root = Path(a.root) + ids = [ + line.strip() + for line in (root / "mp_filelist.txt").read_text().splitlines() + if line.strip() + ] + n = len(ids) + with Pool(a.workers) as p: + sizes = dict( + p.map( + _voxels, + [(str(root), i, idx) for i, idx in enumerate(ids)], + chunksize=64, + ) + ) + + keep = [ + i for i in range(n) if sizes.get(i) is not None and sizes[i] <= a.cap_voxels + ] + missing = sum(1 for i in range(n) if sizes.get(i) is None) + remap = {old: new for new, old in enumerate(keep)} + + (root / f"mp_filelist_{a.suffix}.txt").write_text( + "\n".join(ids[i] for i in keep) + "\n" + ) + split = json.loads((root / "split.json").read_text()) + new_split = {k: [remap[o] for o in lst if o in remap] for k, lst in split.items()} + (root / f"split_{a.suffix}.json").write_text(json.dumps(new_split)) + + print( # noqa: T201 + f"{root.name}: total={n} keep={len(keep)} drop={n - len(keep)} " + f"({100 * (n - len(keep)) / n:.2f}%) missing={missing} " + f"cap={a.cap_voxels} (~{a.cap_voxels ** (1 / 3):.0f}^3) " + f"splits={ {k: len(v) for k, v in new_split.items()} }" + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/lambda/data_sync.sh b/scripts/lambda/data_sync.sh new file mode 100755 index 00000000..5db01af9 --- /dev/null +++ b/scripts/lambda/data_sync.sh @@ -0,0 +1,48 @@ +#!/usr/bin/env bash +# Sync packed zarr training data from S3 to local NVMe. +# Idempotent (aws s3 sync skips files that already exist with matching size). + +set -euo pipefail + +S3_BUCKET="${S3_BUCKET:-oa-electrai}" +S3_PREFIX="${S3_PREFIX:-mp/chg_datasets}" +NFS_ROOT="${NFS_ROOT:-$(ls -d /lambda/nfs/* 2>/dev/null | head -1)}" +[ -z "$NFS_ROOT" ] && { echo "ERROR: no /lambda/nfs/* mount found; pass NFS_ROOT=... explicitly"; exit 1; } +DATA_ROOT="${DATA_ROOT:-$NFS_ROOT/data}" +DEST="$DATA_ROOT/$S3_PREFIX" + +mkdir -p "$DEST" + +echo "=== syncing s3://$S3_BUCKET/$S3_PREFIX/ -> $DEST ===" +echo "(this is ~1.2 TiB / ~226K .zarr.zip files; expect 1-3h on Lambda NVMe)" +echo + +# --no-progress keeps the log quiet; --only-show-errors keeps it informative +# without thousands of "upload: ... -> ..." lines. +aws s3 sync \ + "s3://$S3_BUCKET/$S3_PREFIX/" "$DEST/" \ + --only-show-errors --no-progress + +echo +echo "=== verifying ===" +# Count files in each major subtree +for sub in rho_gga rho_gga+u functionals; do + count=$(find "$DEST/$sub" -type f 2>/dev/null | wc -l) + size=$(du -sh "$DEST/$sub" 2>/dev/null | awk '{print $1}') + printf " %-15s %8s files %8s\n" "$sub" "$count" "$size" +done + +total=$(find "$DEST" -type f | wc -l) +total_sz=$(du -sh "$DEST" 2>/dev/null | awk '{print $1}') +echo " --------------------------------------" +printf " %-15s %8s files %8s\n" "TOTAL" "$total" "$total_sz" + +# Sanity: S3 holds the unpacked zarr layout (~678K files: ~226K stores × +# ~3 inner files + standalone files). The .zarr.zip packed form is +# Modal-Volume-specific; not present on S3. +if [ "$total" -lt 670000 ]; then + echo "WARN: file count looks low; sync may not be complete." + exit 1 +fi + +echo "OK." diff --git a/scripts/lambda/monitor.env.example b/scripts/lambda/monitor.env.example new file mode 100644 index 00000000..4328c5d1 --- /dev/null +++ b/scripts/lambda/monitor.env.example @@ -0,0 +1,29 @@ +# EnvironmentFile for the resident monitor. Copy to +# ~/.config/electrai-monitor/monitor.env (chmod 600 — it holds secrets) +# and edit. NOT committed with real values. + +# --- targets --- +# Single node (H100:4 phase): +HOSTS="lambda2" +RUN="full_gga_gga+u_f32" +# Multi-node (after PORT_PLAN.md cutover) — head first, then worker: +# HOSTS="lambda3-a lambda3-b" +# RUN="full_gga_gga+u_f32_16x" + +# --- cadence / model --- +MONITOR_INTERVAL="900" # seconds between ticks (15 min) +MONITOR_MODEL="claude-opus-4-8" # Opus: strong reasoning for remediation calls on the live run + +# --- paths (defaults are fine) --- +# MONITOR_JOURNAL="$HOME/.local/state/electrai-monitor/journal.md" +# MONITOR_HEARTBEAT="$HOME/.local/state/electrai-monitor/heartbeat" +# MONITOR_MAINT_FLAG="$HOME/.config/electrai-monitor/MAINTENANCE" + +# --- secrets --- +# Slack incoming-webhook used for escalation only: +SLACK_WEBHOOK_URL="https://hooks.slack.com/services/REPLACE/ME" +# wandb read key (lets the agent cross-check run heartbeat/metrics if it wants): +WANDB_API_KEY="REPLACE_ME" + +# --- watchdog --- +WATCHDOG_STALE_S="3600" # heartbeat must be newer than this (1h) diff --git a/scripts/lambda/monitor_agent_prompt.md b/scripts/lambda/monitor_agent_prompt.md new file mode 100644 index 00000000..9cdbc9fd --- /dev/null +++ b/scripts/lambda/monitor_agent_prompt.md @@ -0,0 +1,68 @@ +You are the resident operations monitor for an in-flight ElectrAI training run on +Lambda Cloud GPUs. You are invoked once per loop tick. Your job: confirm the run +is healthy, autonomously fix a SMALL, well-defined set of failures, and escalate +everything else to a human via Slack. The CONTEXT block prepended above this +prompt gives you the live targets (HOSTS, RUN), the journal path, the maintenance +flag, and the exact snapshot command for this tick. + +Authoritative references — read them, they are the source of truth: +- `scripts/lambda/MONITOR.md` → the 3-part liveness rule, the false-alarm + catalog, and the intervention table. +- `scripts/lambda/MONITOR_EC2.md` → the REMEDIATION RUNBOOK (allowed vs + escalate-only actions) and the circuit breaker. + +Do this, in order: + +1. RECALL. Read the last ~40 lines of the journal: `tail -n 40 ` + (path is in the CONTEXT block). This is your memory of prior ticks — recent + observations, actions taken, and restart counts. The circuit breaker is + enforced by counting your own past actions here, so read it first. + +2. PROBE. Run the snapshot command from the CONTEXT block. For multi-node, the + worker node legitimately shows NFS/checkpoint errors (only the head holds that + state) — judge the worker on its tmux + GPU lines only. + +3. JUDGE using MONITOR.md's liveness rule: healthy iff (a) `train.log` mtime + < 60s, (b) all GPUs >= 80% util, (c) `last.ckpt` mtime < 8h. Flag a stall ONLY + if >= 2 of the 3 fail — single-signal failures are almost always false alarms; + wait for the next tick. Honor the documented false alarms (don't trust the + `tail` step counter; don't conclude "file missing" from one failed `stat`). + +4. ACT: + - HEALTHY → emit one STATUS block (format below) and stop. Take NO action. + - MAINTENANCE = yes (see CONTEXT) → observe and report only. Do NOT remediate + and do NOT escalate routine churn; the operator is mid-cutover. Say so. + - PROBLEM → consult the REMEDIATION RUNBOOK in MONITOR_EC2.md: + * ALLOWED autonomous action AND circuit breaker permits (per your recall) + → perform the minimal idempotent fix over ssh, then state exactly what + you did and why. Restarts are safe: the trainer auto-resumes from + last.ckpt. + * ESCALATE-ONLY, or an allowed fix already failed this cycle, or the + circuit breaker is tripped, or you are not confident of the root cause + → ESCALATE to Slack (below) and stop. Do NOT touch the cluster. + - When in doubt, do NOT act on the cluster — escalate. A false restart on a + live run is worse than a false page. + +ESCALATION (only when the runbook says to): check the CONTEXT line +"Slack escalation channel". + - If it says "configured", push to Slack with a plain one-line curl: + curl -fsS -X POST -H 'Content-type: application/json' --data '{"text":""}' "$SLACK_WEBHOOK_URL" + Use the $SLACK_WEBHOOK_URL variable as-is; NEVER print/echo/interpolate its value. + - If it says "NOT configured", do NOT curl — the operator is reading the journal + directly. Make the escalation impossible to miss: begin your reply with + *** ESCALATION (no Slack configured — operator is reading the journal) *** + Either way your full reply is journaled, so the diagnosis is never lost. + Message (one line): host(s), which of (a/b/c) failed, what you tried, current + epoch / it/s / val_loss if known, and the specific ask of the human. + +NEVER do autonomously (escalate instead): + - delete checkpoints or data, or run `rm`; provision / terminate / switch + instances; edit configs or hyperparameters; restore/pull over a run that is + actually healthy. + - exceed the circuit breaker — stop and page rather than thrash restarts. + +RESPONSE FORMAT — your entire reply is appended verbatim to the journal, so keep +it tight (a few lines): + STATUS: HEALTHY | STALL | ACTED | ESCALATED | MAINTENANCE + metrics: + action: diff --git a/scripts/lambda/monitor_loop.sh b/scripts/lambda/monitor_loop.sh new file mode 100644 index 00000000..58ef31a2 --- /dev/null +++ b/scripts/lambda/monitor_loop.sh @@ -0,0 +1,77 @@ +#!/usr/bin/env bash +# Resident monitor loop for the EC2 watcher box. Each tick injects a small live +# CONTEXT header in front of monitor_agent_prompt.md and runs Claude headless; +# Claude's reply is appended to the journal (its durable memory across ticks). +# +# Continuity is by JOURNAL, not by conversation buffer: a 26-day run would blow +# any context window, and the journal survives restarts and token refreshes. +# +# Run under systemd (see systemd/electrai-monitor.service). Config + secrets come +# from the EnvironmentFile (see monitor.env.example). +# +# Heartbeat: only a *successful* tick touches $HEARTBEAT. The watchdog checks that +# file (not the journal) so a silently failing Claude (expired token, quota, +# network) is detected even though the journal header still updates each tick. + +set -uo pipefail + +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +REPO_DIR="${REPO_DIR:-$HOME/electrai}" + +# Optional env file (systemd already loads it; this covers manual runs). +ENV_FILE="${MONITOR_ENV:-$HOME/.config/electrai-monitor/monitor.env}" +if [ -f "$ENV_FILE" ]; then set -a; . "$ENV_FILE"; set +a; fi + +HOSTS="${HOSTS:-lambda2}" +RUN="${RUN:-full_gga_gga+u_f32}" +INTERVAL="${MONITOR_INTERVAL:-900}" # seconds between ticks (15 min) +JOURNAL="${MONITOR_JOURNAL:-$HOME/.local/state/electrai-monitor/journal.md}" +HEARTBEAT="${MONITOR_HEARTBEAT:-$HOME/.local/state/electrai-monitor/heartbeat}" +MAINT_FLAG="${MONITOR_MAINT_FLAG:-$HOME/.config/electrai-monitor/MAINTENANCE}" +PROMPT_FILE="${MONITOR_PROMPT:-$HERE/monitor_agent_prompt.md}" +SETTINGS="${MONITOR_SETTINGS:-$HERE/monitor_settings.json}" +MODEL="${MONITOR_MODEL:-claude-opus-4-8}" +CLAUDE_BIN="${CLAUDE_BIN:-claude}" + +mkdir -p "$(dirname "$JOURNAL")" "$(dirname "$HEARTBEAT")" +cd "$REPO_DIR" || { echo "FATAL: REPO_DIR=$REPO_DIR missing" >&2; exit 1; } + +echo "$(date -Iseconds) monitor_loop start: HOSTS='$HOSTS' RUN='$RUN' interval=${INTERVAL}s model=$MODEL" >> "$JOURNAL" + +while true; do + ts="$(date -Iseconds)" + maint="no"; [ -f "$MAINT_FLAG" ] && maint="yes" + slack="configured" + if [ -z "${SLACK_WEBHOOK_URL:-}" ] || printf '%s' "${SLACK_WEBHOOK_URL:-}" | grep -qi replace; then + slack="NOT configured (journal-only escalation)" + fi + + ctx="CONTEXT (injected $ts): +- Nodes to check (HOSTS): $HOSTS +- Run name (RUN): $RUN +- Journal path (tail it for prior state): $JOURNAL +- Maintenance mode: $maint (if yes: observe + journal only, do NOT remediate) +- Snapshot command for this tick (HOSTS and RUN are already exported in your env): + bash scripts/lambda/monitor_status_all.sh \"$RUN\" +- Slack escalation channel: $slack (webhook, if any, is in env as \$SLACK_WEBHOOK_URL — never print it) +" + prompt="${ctx} +$(cat "$PROMPT_FILE")" + + echo "" >> "$JOURNAL" + echo "=== $ts tick (maint=$maint) ===" >> "$JOURNAL" + + if "$CLAUDE_BIN" -p "$prompt" \ + --model "$MODEL" \ + --settings "$SETTINGS" \ + --permission-mode default \ + --add-dir "$REPO_DIR" \ + >> "$JOURNAL" 2>&1; then + date -Iseconds > "$HEARTBEAT" + else + rc=$? + echo "$(date -Iseconds) ERROR: claude tick failed rc=$rc (auth? quota? network?) — heartbeat NOT updated" >> "$JOURNAL" + fi + + sleep "$INTERVAL" +done diff --git a/scripts/lambda/monitor_settings.json b/scripts/lambda/monitor_settings.json new file mode 100644 index 00000000..de5c8187 --- /dev/null +++ b/scripts/lambda/monitor_settings.json @@ -0,0 +1,27 @@ +{ + "_comment": "Permission allowlist for the resident monitor (loaded via `claude --settings`). In headless `-p` mode, allowlisted Bash prefixes auto-approve and everything else fails closed (no interactive prompt is possible). Containment of remote ssh command CONTENT is by the operator prompt's hard rules + the dedicated revocable SSH key + read-only AWS, NOT by these patterns. Add lambda3-a / lambda3-b ssh entries at multi-node cutover.", + "permissions": { + "allow": [ + "Read", + "Grep", + "Glob", + "Bash(tail:*)", + "Bash(date:*)", + "Bash(bash scripts/lambda/monitor_status_all.sh:*)", + "Bash(bash scripts/lambda/monitor_status.sh:*)", + "Bash(aws s3 ls:*)", + "Bash(ssh lambda:*)", + "Bash(ssh lambda2:*)", + "Bash(ssh lambda3-a:*)", + "Bash(ssh lambda3-b:*)", + "Bash(curl:*)" + ], + "deny": [ + "Bash(rm:*)", + "Bash(aws s3 rm:*)", + "Bash(aws s3 cp:*)", + "Bash(aws s3 sync:*)", + "Bash(aws ec2:*)" + ] + } +} diff --git a/scripts/lambda/monitor_status.sh b/scripts/lambda/monitor_status.sh new file mode 100755 index 00000000..19d5ec70 --- /dev/null +++ b/scripts/lambda/monitor_status.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# One-shot status snapshot for an in-flight Lambda training run. +# +# Usage: +# bash monitor_status.sh [host=lambda2] [run_name=full_gga_gga+u_f32] +# +# Designed to pair with the hourly LLM-driven monitor (see MONITOR.md). The +# output is intentionally machine-greppable plus human-readable. + +set -uo pipefail + +HOST="${1:-lambda2}" +RUN="${2:-full_gga_gga+u_f32}" +SESSION="${SESSION:-electrai-train}" + +ssh "$HOST" " + NFS_ROOT=\"\${NFS_ROOT:-\$(ls -d /lambda/nfs/* 2>/dev/null | head -1)}\" + CKPT_DIR=\"\$NFS_ROOT/checkpoints\" + RUN_DIR=\"\$CKPT_DIR/$RUN\" + + echo '=== tmux ===' + tmux ls 2>&1 | head -5 + + echo '=== now ===' + date -Iseconds + + echo '=== mtimes (train.log within 60s of now, last.ckpt within 8h = live) ===' + stat -c '%y %n' \"\$CKPT_DIR/train.log\" \"\$RUN_DIR/last.ckpt\" 2>&1 + + echo '=== GPU util (all >= 80% = active) ===' + nvidia-smi --query-gpu=utilization.gpu --format=csv,noheader + + echo '=== current step (live tmux pane) ===' + tmux capture-pane -t $SESSION:train -p -S -100 2>&1 \\ + | tr '\\r' '\\n' \\ + | grep -oE 'Epoch [0-9]+: *[0-9]+%.*it/s.*' | tail -2 + + echo '=== checkpoints ===' + ls -la \"\$RUN_DIR/\" 2>&1 | tail -8 + + echo '=== backup log (S3 ckpt mirror, 10-min cadence) ===' + tail -3 \"\$CKPT_DIR/backup.log\" 2>&1 + + echo '=== wandb-sync log (10-min cadence; expect Syncing: ... done.) ===' + tail -8 \"\$CKPT_DIR/wandb-sync.log\" 2>&1 +" diff --git a/scripts/lambda/monitor_status_all.sh b/scripts/lambda/monitor_status_all.sh new file mode 100644 index 00000000..0413f323 --- /dev/null +++ b/scripts/lambda/monitor_status_all.sh @@ -0,0 +1,24 @@ +#!/usr/bin/env bash +# Multi-host wrapper around monitor_status.sh. Prints a labeled snapshot for +# every node in HOSTS so the EC2 monitor agent can evaluate head + worker in a +# single read. +# +# Usage: +# bash monitor_status_all.sh # HOSTS=lambda2, default run +# HOSTS="lambda3-a lambda3-b" bash monitor_status_all.sh full_gga_gga+u_f32_16x +# +# Note on multi-node: only the HEAD node holds NFS/checkpoint/wandb-sync state. +# The worker's snapshot will legitimately show NFS/checkpoint errors — judge the +# worker on its `=== tmux ===` and `=== GPU util ===` sections only. + +set -uo pipefail + +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +RUN="${1:-full_gga_gga+u_f32}" +HOSTS="${HOSTS:-lambda2}" + +for h in $HOSTS; do + echo "################## NODE: $h ##################" + bash "$HERE/monitor_status.sh" "$h" "$RUN" || echo "(monitor_status.sh exited rc=$? for $h)" + echo +done diff --git a/scripts/lambda/monitor_watchdog.sh b/scripts/lambda/monitor_watchdog.sh new file mode 100644 index 00000000..b98ac368 --- /dev/null +++ b/scripts/lambda/monitor_watchdog.sh @@ -0,0 +1,41 @@ +#!/usr/bin/env bash +# Independent watchdog for the resident monitor — NO LLM, no cluster access. +# "Who watches the watcher": confirms the monitor service is up and that a tick +# actually SUCCEEDED recently (via the heartbeat file, which only a clean Claude +# run touches). Pings Slack if not. Run from a systemd timer or cron (~30 min). +# +# Catches the failure the monitor itself cannot report: expired Claude token, +# quota exhaustion, a wedged loop, or the EC2 box itself degrading. + +set -uo pipefail + +ENV_FILE="${MONITOR_ENV:-$HOME/.config/electrai-monitor/monitor.env}" +if [ -f "$ENV_FILE" ]; then set -a; . "$ENV_FILE"; set +a; fi + +HEARTBEAT="${MONITOR_HEARTBEAT:-$HOME/.local/state/electrai-monitor/heartbeat}" +SERVICE="${MONITOR_SERVICE:-electrai-monitor.service}" +STALE_S="${WATCHDOG_STALE_S:-3600}" + +problems=() + +if command -v systemctl >/dev/null 2>&1; then + systemctl is-active --quiet "$SERVICE" || problems+=("service $SERVICE not active") +fi + +if [ -f "$HEARTBEAT" ]; then + age=$(( $(date +%s) - $(stat -c %Y "$HEARTBEAT") )) + [ "$age" -gt "$STALE_S" ] && problems+=("last successful tick ${age}s ago (> ${STALE_S}s) — Claude may be de-authed / out of quota / wedged") +else + problems+=("heartbeat $HEARTBEAT missing — monitor has never completed a clean tick") +fi + +[ ${#problems[@]} -eq 0 ] && exit 0 + +msg=":rotating_light: electrai-monitor watchdog on $(hostname): $(printf '%s; ' "${problems[@]}")" +if [ -n "${SLACK_WEBHOOK_URL:-}" ]; then + curl -fsS -X POST -H 'Content-type: application/json' \ + --data "$(printf '{"text":%s}' "\"${msg//\"/\\\"}\"")" \ + "$SLACK_WEBHOOK_URL" >/dev/null 2>&1 || true +fi +echo "$(date -Iseconds) WATCHDOG: $msg" >&2 +exit 1 diff --git a/scripts/lambda/nccl_test.py b/scripts/lambda/nccl_test.py new file mode 100644 index 00000000..757bc4be --- /dev/null +++ b/scripts/lambda/nccl_test.py @@ -0,0 +1,58 @@ +"""Minimal cross-node NCCL allreduce smoke test. + +Run with torchrun on BOTH nodes (same args as run_training_multinode.sh): + + # head: + NODE_RANK=0 MASTER_ADDR= NCCL_DEBUG=INFO \\ + NCCL_SOCKET_IFNAME= NCCL_IB_DISABLE=1 \\ + uv run torchrun --nnodes=2 --node_rank=0 --nproc_per_node=8 \\ + --master_addr= --master_port=29500 \\ + scripts/lambda/nccl_test.py + + # worker: + NODE_RANK=1 MASTER_ADDR= NCCL_DEBUG=INFO \\ + NCCL_SOCKET_IFNAME= NCCL_IB_DISABLE=1 \\ + uv run torchrun --nnodes=2 --node_rank=1 --nproc_per_node=8 \\ + --master_addr= --master_port=29500 \\ + scripts/lambda/nccl_test.py + +Verifies process group init + an all_reduce across world_size ranks. If this +hangs at init, fix the NCCL_SOCKET_IFNAME / firewall before touching the +real training command. +""" + +from __future__ import annotations + +import os + +import torch +import torch.distributed as dist + + +def main() -> None: + local_rank = int(os.environ["LOCAL_RANK"]) + rank = int(os.environ["RANK"]) + world_size = int(os.environ["WORLD_SIZE"]) + + torch.cuda.set_device(local_rank) + dist.init_process_group(backend="nccl") + + tensor = torch.full((1,), float(rank), device=f"cuda:{local_rank}") + dist.all_reduce(tensor, op=dist.ReduceOp.SUM) + + # Expected sum: 0+1+...+(world_size-1) == world_size*(world_size-1)/2 + expected = world_size * (world_size - 1) / 2 + got = tensor.item() + ok = abs(got - expected) < 1e-6 + print( # noqa: T201 + f"[rank {rank}/{world_size} local={local_rank}] " + f"allreduce got={got} expected={expected} ok={ok}", + flush=True, + ) + + dist.barrier() + dist.destroy_process_group() + + +if __name__ == "__main__": + main() diff --git a/scripts/lambda/prep_data.sh b/scripts/lambda/prep_data.sh new file mode 100755 index 00000000..fed02dc9 --- /dev/null +++ b/scripts/lambda/prep_data.sh @@ -0,0 +1,74 @@ +#!/usr/bin/env bash +# Recreate the data/label symlinks and build smoke filelists in the local data +# tree. Lambda equivalent of modal/prep_volume.py. + +set -euo pipefail + +NFS_ROOT="${NFS_ROOT:-$(ls -d /lambda/nfs/* 2>/dev/null | head -1)}" +[ -z "$NFS_ROOT" ] && { echo "ERROR: no /lambda/nfs/* mount found; pass NFS_ROOT=... explicitly"; exit 1; } +DATA_ROOT="${DATA_ROOT:-$NFS_ROOT/data}" +BASE="$DATA_ROOT/mp/chg_datasets" +SMOKE_N="${SMOKE_N:-200}" + +echo "=== relinking functionals/{gga,gga+u}/{data,label} -> rho_* ===" +for entry in "gga:rho_gga" "gga+u:rho_gga+u"; do + func="${entry%:*}" + rho="${entry#*:}" + fdir="$BASE/functionals/$func" + mkdir -p "$fdir" + for sub in data label; do + link="$fdir/$sub" + target="../../$rho/$sub" + real="$BASE/$rho/$sub" + if [ ! -d "$real" ]; then + echo " ERROR: $real missing -- data_sync.sh hasn't completed" + exit 1 + fi + if [ -L "$link" ]; then rm "$link"; fi + if [ -d "$link" ] && [ ! -L "$link" ]; then + echo " $link is a real dir; leaving as-is" + continue + fi + ln -s "$target" "$link" + echo " linked $link -> $target" + done +done + +echo +echo "=== writing smoke filelists (first $SMOKE_N ids) ===" +for func in gga 'gga+u'; do + fdir="$BASE/functionals/$func" + fl="$fdir/mp_filelist.txt" + smoke="$fdir/mp_filelist_smoke.txt" + if [ ! -f "$fl" ]; then + echo " ERROR: $fl missing" + exit 1 + fi + head -n "$SMOKE_N" "$fl" > "$smoke" + n=$(wc -l < "$smoke") + echo " wrote $smoke ($n ids)" +done + +echo +echo "=== sanity check (first id of each filelist resolves) ===" +# Loader auto-detects either packed (.zarr.zip) or unpacked (.zarr/ dir). +# S3 holds unpacked; Modal Volume held packed; Lambda gets whichever S3 has. +for func in gga 'gga+u'; do + fdir="$BASE/functionals/$func" + first=$(head -n 1 "$fdir/mp_filelist.txt") + zip="$fdir/data/$first.zarr.zip" + store="$fdir/data/$first.zarr" + if [ -f "$zip" ]; then + fmt="packed (.zarr.zip)"; resolved="$zip" + elif [ -d "$store" ]; then + fmt="unpacked (.zarr/)"; resolved="$store" + else + echo " ERROR: neither $zip nor $store resolves -- transfer incomplete" + exit 1 + fi + n=$(wc -l < "$fdir/mp_filelist.txt") + echo " OK ($func, $fmt): $resolved ($n total ids)" +done + +echo +echo "prep_data complete. DATA_ROOT=$DATA_ROOT" diff --git a/scripts/lambda/run_training.sh b/scripts/lambda/run_training.sh new file mode 100755 index 00000000..4370b002 --- /dev/null +++ b/scripts/lambda/run_training.sh @@ -0,0 +1,127 @@ +#!/usr/bin/env bash +# Launch training in a detached tmux session. Auto-resume from last.ckpt; a +# parallel rclone-style aws sync backs checkpoints up to S3 every CKPT_BACKUP_S. +# +# Usage: +# bash scripts/lambda/run_training.sh smoke # subset + 2 epochs +# bash scripts/lambda/run_training.sh full # full 113K + 100 epochs + +set -euo pipefail + +MODE="${1:-smoke}" # smoke | full +REPO_DIR="${REPO_DIR:-$HOME/electrai}" +NFS_ROOT="${NFS_ROOT:-$(ls -d /lambda/nfs/* 2>/dev/null | head -1)}" +[ -z "$NFS_ROOT" ] && { echo "ERROR: no /lambda/nfs/* mount found; pass NFS_ROOT=... explicitly"; exit 1; } +DATA_ROOT="${DATA_ROOT:-$NFS_ROOT/data}" +CKPT_ROOT="${CKPT_ROOT:-$NFS_ROOT/checkpoints}" +S3_CKPT_BUCKET="${S3_CKPT_BUCKET:-oa-electrai}" +S3_CKPT_PREFIX="${S3_CKPT_PREFIX:-checkpoints/lambda}" +CKPT_BACKUP_S="${CKPT_BACKUP_S:-600}" # 10 min +TMUX_SESSION="${TMUX_SESSION:-electrai-train}" +UV_BIN="${UV_BIN:-$HOME/.local/bin/uv}" + +case "$MODE" in + smoke) SRC_CFG="src/electrai/configs/MP/config_gga_gga+u_f32_smoke.yaml" ;; + full) SRC_CFG="src/electrai/configs/MP/config_gga_gga+u_f32.yaml" ;; + w64) SRC_CFG="src/electrai/configs/MP/config_gga_gga+u_w64.yaml" ;; + *) echo "MODE must be 'smoke', 'full', or 'w64'"; exit 1 ;; +esac + +cd "$REPO_DIR" + +# Pull WANDB_API_KEY out of ~/.bashrc. Ubuntu's default ~/.bashrc starts with +# `case $- in *i*) ;; *) return;; esac` which early-returns for non-interactive +# shells, so plain `source ~/.bashrc` doesn't help. Grep the export line +# directly and eval it. +if [ -z "${WANDB_API_KEY:-}" ] && [ -f "$HOME/.bashrc" ]; then + eval "$(grep -E '^[[:space:]]*export WANDB_API_KEY=' "$HOME/.bashrc" | tail -1)" +fi +if [ -z "${WANDB_API_KEY:-}" ]; then + echo "ERROR: WANDB_API_KEY env var not set." + echo " Add to ~/.bashrc: export WANDB_API_KEY='...' then re-run." + exit 1 +fi + +mkdir -p "$CKPT_ROOT" + +# Rewrite the config so dataset paths and ckpt_path point at our local roots. +# Two source path conventions exist in committed configs: +# - della: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/... +# - modal: /data/... (smoke config uses this since it was authored for Modal) +# Both get rewritten to live under $DATA_ROOT. +RUNTIME_CFG="$CKPT_ROOT/runtime-config.yaml" +WANDB_SED=() +if [ -n "${WANDB_MODE_OVERRIDE:-}" ]; then + WANDB_SED=(-e 's|^wandb_mode: .*|wandb_mode: '"$WANDB_MODE_OVERRIDE"'|') +fi +sed \ + -e 's| /scratch/gpfs/ROSENGROUP/common/globus_share_OA/| '"$DATA_ROOT"'/|g' \ + -e 's| /data/mp/chg_datasets/| '"$DATA_ROOT"'/mp/chg_datasets/|g' \ + -e 's|^ckpt_path: .*|ckpt_path: '"$CKPT_ROOT"'/${MODE}_${RUN}|' \ + "${WANDB_SED[@]}" \ + "$SRC_CFG" \ + | python3 -c " +import sys, re +text = sys.stdin.read() +# substitute MODE/RUN tokens from env +text = text.replace('\${MODE}', '$MODE') +# pull run_name from the config itself for the ckpt subdir +m = re.search(r'^run_name:\s*(.+)$', text, re.M) +run = (m.group(1).strip() if m else 'run') +text = text.replace('\${RUN}', run) +print(text) +" > "$RUNTIME_CFG" + +# Sanity-print the rewritten paths +echo "=== runtime config datasets/ckpt ===" +grep -E '^\s*(root|split_file|ckpt_path):' "$RUNTIME_CFG" + +# Pull entity/project from the config so the wandb URL printed below is correct +# for whichever mode/config was selected. +WB_ENTITY="$(grep -E '^entity:' "$RUNTIME_CFG" | awk '{print $2}')" +WB_PROJECT="$(grep -E '^wb_pname:' "$RUNTIME_CFG" | awk '{print $2}')" + +# Kill any existing session of the same name +if tmux has-session -t "$TMUX_SESSION" 2>/dev/null; then + echo "killing existing tmux session $TMUX_SESSION" + tmux kill-session -t "$TMUX_SESSION" +fi + +# Launch training + backup loop in one tmux session with two windows. +PYTHONPATH="$REPO_DIR/src${PYTHONPATH:+:$PYTHONPATH}" +export PYTHONPATH + +# Window 1: training. Re-exec on crash so a transient failure doesn't end +# the session; Lightning auto-resumes from last.ckpt on the next start. +# - set -o pipefail captures uv's exit code through the `| tee` +# - uv is invoked by absolute path so we don't depend on a sourced PATH +TRAIN_CMD="set -o pipefail; cd '$REPO_DIR' && export WANDB_API_KEY='$WANDB_API_KEY' && \ + while true; do \ + echo \"\$(date -Iseconds) starting training\" | tee -a $CKPT_ROOT/train.log; \ + '$UV_BIN' run python -m electrai.entrypoints.main train --config '$RUNTIME_CFG' 2>&1 | tee -a $CKPT_ROOT/train.log; \ + rc=\$?; \ + echo \"\$(date -Iseconds) training exited rc=\$rc\" | tee -a $CKPT_ROOT/train.log; \ + [ \$rc -eq 0 ] && break; \ + sleep 30; \ + done" + +# Window 2: checkpoint backup loop, every CKPT_BACKUP_S seconds. +BACKUP_CMD="while true; do \ + echo \"\$(date -Iseconds) backing up checkpoints\" | tee -a $CKPT_ROOT/backup.log; \ + aws s3 sync '$CKPT_ROOT' 's3://$S3_CKPT_BUCKET/$S3_CKPT_PREFIX/' --only-show-errors 2>&1 | tee -a $CKPT_ROOT/backup.log; \ + sleep $CKPT_BACKUP_S; \ +done" + +tmux new-session -d -s "$TMUX_SESSION" -n train "$TRAIN_CMD" +tmux new-window -t "$TMUX_SESSION" -n backup "$BACKUP_CMD" +tmux ls + +cat < \ +# bash scripts/lambda/run_training_multinode.sh full +# +# # worker (rank 1): +# NODE_RANK=1 MASTER_ADDR= \ +# bash scripts/lambda/run_training_multinode.sh full +# +# Required env: +# NODE_RANK 0 for head, 1..NUM_NODES-1 for workers +# MASTER_ADDR IP of the head node, reachable from all workers (use the +# PRIVATE / internal interface IP, not the public NAT'd one) +# +# Optional env (with defaults): +# NUM_NODES 2 +# NPROC_PER_NODE 8 (1 rank per H100) +# MASTER_PORT 29500 +# NCCL_SOCKET_IFNAME (autodetect via `ip route get $MASTER_ADDR`) +# WANDB_MODE_OVERRIDE (inherit pattern from run_training.sh; "offline" +# for online wandb, "disabled" for smokes) +# See MULTINODE.md for the full operator runbook. + +set -uo pipefail + +MODE="${1:-smoke}" # smoke | full + +# -- required inputs ----------------------------------------------------------- +: "${NODE_RANK:?NODE_RANK is required (0 = head, 1+ = worker)}" +: "${MASTER_ADDR:?MASTER_ADDR is required (head node private IP)}" + +# -- defaults ------------------------------------------------------------------ +NUM_NODES="${NUM_NODES:-2}" +NPROC_PER_NODE="${NPROC_PER_NODE:-8}" +MASTER_PORT="${MASTER_PORT:-29500}" + +REPO_DIR="${REPO_DIR:-$HOME/electrai}" +NFS_ROOT="${NFS_ROOT:-$(ls -d /lambda/nfs/* 2>/dev/null | head -1)}" +[ -z "$NFS_ROOT" ] && { echo "ERROR: no /lambda/nfs/* mount found; pass NFS_ROOT=... explicitly"; exit 1; } +DATA_ROOT="${DATA_ROOT:-$NFS_ROOT/data}" +CKPT_ROOT="${CKPT_ROOT:-$NFS_ROOT/checkpoints}" +S3_CKPT_BUCKET="${S3_CKPT_BUCKET:-oa-electrai}" +S3_CKPT_PREFIX="${S3_CKPT_PREFIX:-checkpoints/lambda-multinode}" +CKPT_BACKUP_S="${CKPT_BACKUP_S:-600}" +TMUX_SESSION="${TMUX_SESSION:-electrai-train}" +UV_BIN="${UV_BIN:-$HOME/.local/bin/uv}" + +# -- pick the right config ----------------------------------------------------- +case "$MODE" in + smoke) SRC_CFG="src/electrai/configs/MP/config_gga_gga+u_f32_smoke.yaml" ;; + full) SRC_CFG="src/electrai/configs/MP/config_gga_gga+u_f32.yaml" ;; + *) echo "MODE must be 'smoke' or 'full'"; exit 1 ;; +esac + +cd "$REPO_DIR" + +# -- WANDB key (same trick as run_training.sh; Ubuntu's ~/.bashrc returns early +# for non-interactive shells, so we grep the export line directly) ------------- +if [ -z "${WANDB_API_KEY:-}" ] && [ -f "$HOME/.bashrc" ]; then + eval "$(grep -E '^[[:space:]]*export WANDB_API_KEY=' "$HOME/.bashrc" | tail -1)" +fi +if [ -z "${WANDB_API_KEY:-}" ]; then + echo "ERROR: WANDB_API_KEY env var not set." + echo " Add to ~/.bashrc: export WANDB_API_KEY='...' then re-run." + exit 1 +fi + +mkdir -p "$CKPT_ROOT" + +# -- auto-detect the NIC that reaches MASTER_ADDR ------------------------------ +# `ip route get ` prints the interface used to reach that IP. On Lambda +# this is usually the second NIC (the 100 Gbps internal one) -- e.g. `enp...` +# or `ens...`. If autodetection picks the wrong NIC (e.g. the management +# interface), set NCCL_SOCKET_IFNAME explicitly. +if [ -z "${NCCL_SOCKET_IFNAME:-}" ]; then + DETECTED_IF=$(ip -o route get "$MASTER_ADDR" 2>/dev/null | awk '{ + for (i=1;i<=NF;i++) if ($i=="dev") { print $(i+1); exit } + }') + if [ -n "$DETECTED_IF" ]; then + NCCL_SOCKET_IFNAME="$DETECTED_IF" + echo "NCCL_SOCKET_IFNAME auto-detected -> $NCCL_SOCKET_IFNAME (route to $MASTER_ADDR)" + else + echo "WARN: could not auto-detect NIC for $MASTER_ADDR; falling back to NCCL default." + echo " Set NCCL_SOCKET_IFNAME= (see MULTINODE.md)." + fi +fi + +# -- rewrite the config so dataset paths and ckpt_path point at our local roots +# Same sed pipeline as run_training.sh; both nodes regenerate this independently +# (idempotent, same input -> same output). ------------------------------------ +RUNTIME_CFG="$CKPT_ROOT/runtime-config.yaml" +WANDB_SED=() +if [ -n "${WANDB_MODE_OVERRIDE:-}" ]; then + WANDB_SED=(-e 's|^wandb_mode: .*|wandb_mode: '"$WANDB_MODE_OVERRIDE"'|') +fi +sed \ + -e 's| /scratch/gpfs/ROSENGROUP/common/globus_share_OA/| '"$DATA_ROOT"'/|g' \ + -e 's| /data/mp/chg_datasets/| '"$DATA_ROOT"'/mp/chg_datasets/|g' \ + -e 's|^ckpt_path: .*|ckpt_path: '"$CKPT_ROOT"'/${MODE}_${RUN}|' \ + "${WANDB_SED[@]}" \ + "$SRC_CFG" \ + | python3 -c " +import sys, re +text = sys.stdin.read() +text = text.replace('\${MODE}', '$MODE') +m = re.search(r'^run_name:\s*(.+)$', text, re.M) +run = (m.group(1).strip() if m else 'run') +text = text.replace('\${RUN}', run) +print(text) +" > "$RUNTIME_CFG" + +echo "=== runtime config datasets/ckpt ===" +grep -E '^\s*(root|split_file|ckpt_path):' "$RUNTIME_CFG" + +# -- tmux ---------------------------------------------------------------------- +if tmux has-session -t "$TMUX_SESSION" 2>/dev/null; then + echo "killing existing tmux session $TMUX_SESSION" + tmux kill-session -t "$TMUX_SESSION" +fi + +PYTHONPATH="$REPO_DIR/src${PYTHONPATH:+:$PYTHONPATH}" +export PYTHONPATH + +# -- NCCL env ----------------------------------------------------------------- +# Why each var (all of these get passed through tmux into the training shell): +# NCCL_IB_DISABLE=1 Lambda Ethernet-only fabric; force the socket +# transport so NCCL doesn't waste time probing IB. +# NCCL_SOCKET_IFNAME=... Pin NCCL to the NIC that actually reaches +# MASTER_ADDR (auto-detected above). Without this, +# NCCL may pick docker0 / a vlan and hang. +# NCCL_DEBUG=INFO Keep this loud for the first few multi-node runs; +# drop to WARN once stable to cut log noise. +# NCCL_ASYNC_ERROR_HANDLING=1 +# Surface NCCL errors as Python exceptions instead +# of silent hangs. Strongly recommended on Ethernet. +# NCCL_P2P_LEVEL=NVL Intra-node: use NVLink only (H100 SXM has NVL). +# Inter-node always falls back to socket regardless. +NCCL_ENV=( + "NCCL_IB_DISABLE=1" + "NCCL_DEBUG=INFO" + "NCCL_ASYNC_ERROR_HANDLING=1" + "NCCL_P2P_LEVEL=NVL" +) +if [ -n "${NCCL_SOCKET_IFNAME:-}" ]; then + NCCL_ENV+=("NCCL_SOCKET_IFNAME=$NCCL_SOCKET_IFNAME") +fi +NCCL_EXPORTS=$(printf 'export %s; ' "${NCCL_ENV[@]}") + +# -- training command ---------------------------------------------------------- +# Rendezvous backend choice: +# We use torchrun's classic --master_addr / --master_port (a.k.a. "static" +# rendezvous). For exactly 2 known-IP nodes this is simpler and friendlier +# to firewall config than c10d, and matches the rest of the runbook's +# "head node IP" mental model. Switch to `--rdzv_backend=c10d +# --rdzv_endpoint=$MASTER_ADDR:$MASTER_PORT --rdzv_id=electrai` if/when we +# want elastic restarts (e.g. workers can drop/rejoin) -- not needed today. +# +# torchrun sets LOCAL_RANK, RANK, WORLD_SIZE, LOCAL_WORLD_SIZE in each worker +# process. train.py already reads LOCAL_WORLD_SIZE / WORLD_SIZE to derive +# num_nodes for the Lightning Trainer. +# +# We wrap in a while-loop for the same crash-resume behaviour as the +# single-node script. Lightning auto-resumes from last.ckpt on restart. +TORCHRUN_CMD=( + "$UV_BIN" run torchrun + --nnodes="$NUM_NODES" + --node_rank="$NODE_RANK" + --nproc_per_node="$NPROC_PER_NODE" + --master_addr="$MASTER_ADDR" + --master_port="$MASTER_PORT" + -m electrai.entrypoints.main train --config "$RUNTIME_CFG" +) +# expand to a single shell string for tmux's -d send-keys form +TORCHRUN_STR="${TORCHRUN_CMD[*]}" + +TRAIN_CMD="set -o pipefail; cd '$REPO_DIR' && \ + export WANDB_API_KEY='$WANDB_API_KEY' && \ + $NCCL_EXPORTS \ + while true; do \ + echo \"\$(date -Iseconds) [node $NODE_RANK/$NUM_NODES] starting training\" | tee -a $CKPT_ROOT/train.log; \ + $TORCHRUN_STR 2>&1 | tee -a $CKPT_ROOT/train.log; \ + rc=\$?; \ + echo \"\$(date -Iseconds) [node $NODE_RANK/$NUM_NODES] training exited rc=\$rc\" | tee -a $CKPT_ROOT/train.log; \ + [ \$rc -eq 0 ] && break; \ + sleep 30; \ + done" + +tmux new-session -d -s "$TMUX_SESSION" -n train "$TRAIN_CMD" + +# -- head-node-only: ckpt backup + wandb sync windows -------------------------- +# In Lightning DDP only global rank 0 writes checkpoints, and global rank 0 +# is always on the head node (NODE_RANK=0). So workers don't need backup or +# wandb-sync windows. +if [ "$NODE_RANK" = "0" ]; then + BACKUP_CMD="while true; do \ + echo \"\$(date -Iseconds) backing up checkpoints\" | tee -a $CKPT_ROOT/backup.log; \ + aws s3 sync '$CKPT_ROOT' 's3://$S3_CKPT_BUCKET/$S3_CKPT_PREFIX/' --only-show-errors 2>&1 | tee -a $CKPT_ROOT/backup.log; \ + sleep $CKPT_BACKUP_S; \ + done" + tmux new-window -t "$TMUX_SESSION" -n backup "$BACKUP_CMD" + + # wandb-sync window: only useful when WANDB_MODE_OVERRIDE=offline (the + # working pattern for PrinceOA right now -- online login crashes). For + # smoke runs with WANDB_MODE_OVERRIDE=disabled there's nothing to sync, + # but starting the window is cheap and idempotent. + WANDB_SYNC_CMD="cd '$REPO_DIR' && export WANDB_API_KEY='$WANDB_API_KEY' && \ + while true; do \ + echo \"\$(date -Iseconds) wandb sync sweep\" | tee -a $CKPT_ROOT/wandb-sync.log; \ + find . -type d -name 'offline-run-*' -print 2>/dev/null \ + | xargs -r -n1 '$UV_BIN' run wandb sync 2>&1 \ + | tee -a $CKPT_ROOT/wandb-sync.log; \ + sleep 300; \ + done" + tmux new-window -t "$TMUX_SESSION" -n wandb-sync "$WANDB_SYNC_CMD" +fi + +tmux ls + +cat </dev/null 2>&1; then + curl -LsSf https://astral.sh/uv/install.sh | sh + export PATH="$HOME/.local/bin:$PATH" +fi +uv --version + +echo "=== aws cli ===" +if ! command -v aws >/dev/null 2>&1; then + sudo apt-get update -qq + sudo apt-get install -y -qq awscli +fi +aws --version + +echo "=== tmux ===" +if ! command -v tmux >/dev/null 2>&1; then + sudo apt-get install -y -qq tmux +fi +tmux -V + +echo "=== repo ===" +if [ ! -d "$REPO_DIR/.git" ]; then + git clone "$REPO_URL" "$REPO_DIR" +fi +cd "$REPO_DIR" +git fetch origin +git checkout "$BRANCH" +git pull --ff-only origin "$BRANCH" + +echo "=== uv sync ===" +uv sync + +echo "=== checking credentials ===" +if ! aws sts get-caller-identity >/dev/null 2>&1; then + echo "AWS creds not configured. Run: aws configure" + echo " (need read access to s3://oa-electrai/mp/chg_datasets/*)" + exit 1 +fi +echo "AWS identity:" +aws sts get-caller-identity --output text --query 'Arn' + +if [ -z "${WANDB_API_KEY:-}" ]; then + echo "WARNING: WANDB_API_KEY env var not set." + echo " Add to ~/.bashrc: export WANDB_API_KEY='...'" + echo " (Or pass inline before run_training.sh.)" +fi + +echo +echo "=== setup OK ===" +echo "Repo at : $REPO_DIR" +echo "Branch : $(git rev-parse --abbrev-ref HEAD)" +echo "Commit : $(git rev-parse --short HEAD)" +echo +echo "Next:" +echo " bash scripts/lambda/data_sync.sh # ~1-3h, 1.2 TiB from S3 to local NVMe" +echo " bash scripts/lambda/prep_data.sh # relink data/label dirs, build smoke filelists" +echo " bash scripts/lambda/run_training.sh smoke # validate end-to-end" +echo " bash scripts/lambda/run_training.sh full # full 100-epoch campaign" diff --git a/scripts/lambda/systemd/electrai-monitor-watchdog.service b/scripts/lambda/systemd/electrai-monitor-watchdog.service new file mode 100644 index 00000000..7449afeb --- /dev/null +++ b/scripts/lambda/systemd/electrai-monitor-watchdog.service @@ -0,0 +1,10 @@ +[Unit] +Description=ElectrAI monitor watchdog (checks the monitor is alive; pings Slack) + +[Service] +Type=oneshot +User=ubuntu +Environment=HOME=/home/ubuntu +Environment=PATH=/usr/local/bin:/usr/bin:/bin +EnvironmentFile=-/home/ubuntu/.config/electrai-monitor/monitor.env +ExecStart=/usr/bin/env bash /home/ubuntu/electrai/scripts/lambda/monitor_watchdog.sh diff --git a/scripts/lambda/systemd/electrai-monitor-watchdog.timer b/scripts/lambda/systemd/electrai-monitor-watchdog.timer new file mode 100644 index 00000000..1baf162a --- /dev/null +++ b/scripts/lambda/systemd/electrai-monitor-watchdog.timer @@ -0,0 +1,10 @@ +[Unit] +Description=Run the ElectrAI monitor watchdog every 30 minutes + +[Timer] +OnBootSec=10min +OnUnitActiveSec=30min +AccuracySec=1min + +[Install] +WantedBy=timers.target diff --git a/scripts/lambda/systemd/electrai-monitor.service b/scripts/lambda/systemd/electrai-monitor.service new file mode 100644 index 00000000..84de003c --- /dev/null +++ b/scripts/lambda/systemd/electrai-monitor.service @@ -0,0 +1,19 @@ +[Unit] +Description=ElectrAI Lambda training monitor (resident Claude agent) +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User=ubuntu +Environment=HOME=/home/ubuntu +# claude (native installer / npm global), uv, and aws can land in any of these: +Environment=PATH=/home/ubuntu/.local/bin:/home/ubuntu/.npm-global/bin:/usr/local/bin:/usr/bin:/bin +WorkingDirectory=/home/ubuntu/electrai +EnvironmentFile=-/home/ubuntu/.config/electrai-monitor/monitor.env +ExecStart=/usr/bin/env bash /home/ubuntu/electrai/scripts/lambda/monitor_loop.sh +Restart=always +RestartSec=30 + +[Install] +WantedBy=multi-user.target diff --git a/scripts/lambda/train_watchdog.sh b/scripts/lambda/train_watchdog.sh new file mode 100755 index 00000000..90fa9cc9 --- /dev/null +++ b/scripts/lambda/train_watchdog.sh @@ -0,0 +1,100 @@ +#!/usr/bin/env bash +# One-shot training watchdog for the Lambda W64 run. Run from cron (~10 min). +# +# Catches the failure that silently burned ~2 days on 2026-07-13: training went +# to NaN but kept "running" (GPUs busy, session up), and save_top_k masked it so +# the best-checkpoint list looked frozen rather than alarming. +# +# Detects, on each tick: +# NaN - sustained nan train_loss (2 strikes, ~20 min) -> ALERT + stop run +# STALL - train.log not updated in STALL_S while session up -> ALERT +# DOWN - tmux session gone -> ALERT +# +# Emits: ALERT.txt in the checkpoint dir (mirrored to S3 by the backup loop), +# a line in ~/train_watchdog.log, and a Slack post if SLACK_WEBHOOK_URL is set. +# De-dups Slack so an unchanged condition doesn't ping every tick. +# +# Install: (echo '*/10 * * * * bash $HOME/electrai/scripts/lambda/train_watchdog.sh') | crontab - +set -uo pipefail + +ENV_FILE="${WATCHDOG_ENV:-$HOME/.config/electrai-monitor/monitor.env}" +[ -f "$ENV_FILE" ] && { set -a; . "$ENV_FILE"; set +a; } + +SESSION="${WATCHDOG_SESSION:-electrai-train}" +NFS_ROOT="${NFS_ROOT:-$(ls -d /lambda/nfs/* 2>/dev/null | head -1)}" +CKPT_ROOT="${CKPT_ROOT:-$NFS_ROOT/checkpoints}" +LOG="${WATCHDOG_LOG:-$CKPT_ROOT/train.log}" +STALL_S="${WATCHDOG_STALL_S:-2700}" # 45 min without a log write +NAN_STRIKES_MAX="${WATCHDOG_NAN_STRIKES:-2}" # consecutive nan ticks before acting +STOP_ON_NAN="${WATCHDOG_STOP_ON_NAN:-1}" +ALERT_FILE="$CKPT_ROOT/ALERT.txt" +WLOG="$HOME/train_watchdog.log" +STATE_DIR="${WATCHDOG_STATE:-$HOME/.local/state/electrai-watchdog}" +mkdir -p "$STATE_DIR" +STRIKE_FILE="$STATE_DIR/nan_strikes" +LASTMSG_FILE="$STATE_DIR/last_alert" + +ts() { date -Iseconds; } + +alert() { + local kind="$1"; shift + local msg="[$kind] $*" + printf '%s %s\n' "$(ts)" "$msg" | tee -a "$WLOG" > "$ALERT_FILE" + # Slack, de-duplicated on (kind+msg) + if [ -n "${SLACK_WEBHOOK_URL:-}" ] && [ "$msg" != "$(cat "$LASTMSG_FILE" 2>/dev/null)" ]; then + local safe; safe=$(printf 'electrai W64 watchdog (%s): %s' "$(hostname)" "$msg" | tr -d '"\n\r' | cut -c1-400) + curl -fsS -X POST -H 'Content-type: application/json' \ + --data "{\"text\":\":rotating_light: $safe\"}" "$SLACK_WEBHOOK_URL" >/dev/null 2>&1 || true + fi + printf '%s' "$msg" > "$LASTMSG_FILE" +} + +ok() { printf '%s OK %s\n' "$(ts)" "$*" >> "$WLOG"; rm -f "$ALERT_FILE"; : > "$LASTMSG_FILE"; } + +# --- DOWN: session gone --- +if ! tmux has-session -t "$SESSION" 2>/dev/null; then + alert DOWN "tmux session '$SESSION' is gone -- training not running" + exit 2 +fi + +# --- STALL: log not written recently --- +if [ -f "$LOG" ]; then + age=$(( $(date +%s) - $(stat -c %Y "$LOG") )) + if [ "$age" -gt "$STALL_S" ]; then + alert STALL "train.log not updated in ${age}s (> ${STALL_S}s) -- stalled/hung" + exit 3 + fi +else + alert DOWN "train.log missing at $LOG" + exit 2 +fi + +# --- NaN: sustained nan in the newest run's recent train_loss values --- +start=$(grep -n "starting training" "$LOG" 2>/dev/null | tail -1 | cut -d: -f1); start="${start:-1}" +recent=$(tail -n +"$start" "$LOG" 2>/dev/null | tr '\r' '\n' | grep -oE 'train_loss_step=[0-9na.]+' | tail -30) +last=$(printf '%s' "$recent" | tail -1) +nan_n=$(printf '%s\n' "$recent" | grep -c 'nan') + +# Real divergence: latest value nan AND >=25/30 recent are nan (startup shows +# nan only briefly before numeric losses appear, so it won't hold across ticks). +if [ -n "$last" ] && printf '%s' "$last" | grep -q 'nan' && [ "$nan_n" -ge 25 ]; then + strikes=$(( $(cat "$STRIKE_FILE" 2>/dev/null || echo 0) + 1 )) + echo "$strikes" > "$STRIKE_FILE" + printf '%s NaN-strike %s/%s (nan %s/30, last=%s)\n' "$(ts)" "$strikes" "$NAN_STRIKES_MAX" "$nan_n" "$last" >> "$WLOG" + if [ "$strikes" -ge "$NAN_STRIKES_MAX" ]; then + if [ "$STOP_ON_NAN" = "1" ]; then + tmux kill-session -t "$SESSION" 2>/dev/null + pkill -9 -f "[v]env/bin/python3 -m electrai" 2>/dev/null + alert NAN-STOPPED "Sustained NaN over ${strikes} ticks -- STOPPED the run to stop compute waste. Recover: relaunch from newest good ckpt_epoch=* (NOT last.ckpt)." + else + alert NAN "Sustained NaN over ${strikes} ticks -- training is producing NaN." + fi + exit 4 + fi + exit 0 +fi + +# healthy +echo 0 > "$STRIKE_FILE" +ok "last=${last:-} nan=${nan_n}/30 logage=${age}s" +exit 0 diff --git a/scripts/review_lrwd_sweep.py b/scripts/review_lrwd_sweep.py new file mode 100755 index 00000000..b74e2ffb --- /dev/null +++ b/scripts/review_lrwd_sweep.py @@ -0,0 +1,95 @@ +#!/usr/bin/env python3 +"""Aggregate the LR/WD width-sweep W&B runs into per-(width, lr, wd) results. + +Preemption restarts create a fresh W&B run per segment (no explicit id=), so +runs are grouped by (n_channels, lr, weight_decay) from run config and merged +per epoch before picking best/final val_loss. val_loss is NormMAE; x100 = %. + + uv run python scripts/review_lrwd_sweep.py [project] +""" +# ruff: noqa: T201 + +from __future__ import annotations + +import sys + +import wandb +import wandb.sdk.lib.server as _wandb_server + +# Server returns "flags": null, crashing the viewer query inside Api() login +# (same wandb bug that forces offline mode + sidecar sync on the cluster). +_orig_query = _wandb_server.Server.query_with_timeout + + +def _query_tolerant(self, *args, **kwargs): + try: + _orig_query(self, *args, **kwargs) + except TypeError: + self._flags = {} + + +_wandb_server.Server.query_with_timeout = _query_tolerant + +ENTITY = "PrinceOA" +PROJECT = sys.argv[1] if len(sys.argv) > 1 else "mp-gga-ggau-lrwd" + + +def main() -> None: + api = wandb.Api() + runs = api.runs(f"{ENTITY}/{PROJECT}") + + trials: dict[tuple[int, float, float, str], dict[int, float]] = {} + for run in runs: + cfg = run.config + model = cfg.get("model") or {} + width = model.get("n_channels") + lr = cfg.get("lr") + wd = cfg.get("weight_decay", 0.0) + if width is None or lr is None: + print(f" (skipping run {run.name}: no width/lr in config)") + continue + # The _rep2 noise-bar reruns share (width, lr, wd) with their stage-A + # twins; the config run_name (the config stem) keeps them apart while + # still merging preemption-restart segments, which share it. + rep = "rep2" if str(cfg.get("run_name", "")).endswith("_rep2") else "" + per_epoch = trials.setdefault((width, float(lr), float(wd), rep), {}) + for h in run.scan_history(keys=["epoch", "val_loss_epoch"]): + ep, val = h.get("epoch"), h.get("val_loss_epoch") + if ep is None or val is None: + continue + # Restart segments can re-log an epoch; keep the better value + per_epoch[int(ep)] = min(val, per_epoch.get(int(ep), float("inf"))) + + print( + f"{'width':>5} {'lr':>8} {'wd':>7} {'rep':>4} {'epochs':>6} {'best val':>9} " + f"{'best%':>6} {'final val':>9} {'@ep':>3}" + ) + best_by_width: dict[int, tuple[float, float]] = {} + for (width, lr, wd, rep), per_epoch in sorted(trials.items()): + if not per_epoch: + continue + best_ep = min(per_epoch, key=per_epoch.get) + last_ep = max(per_epoch) + best = per_epoch[best_ep] + print( + f"{width:>5} {lr:>8g} {wd:>7g} {rep:>4} {len(per_epoch):>6} {best:>9.6f} " + f"{best * 100:>6.3f} {per_epoch[last_ep]:>9.6f} {best_ep:>3}" + ) + if ( + wd == 0.0 + and not rep + and best < best_by_width.get(width, (float("inf"),))[0] + ): + best_by_width[width] = (best, lr) + + if best_by_width: + args = " ".join(f"{w}={lr:g}" for w, (_, lr) in sorted(best_by_width.items())) + print(f"\nBest LR per width (wd=0 trials): {args}") + print( + f"Stage B: uv run python scripts/coreweave/gen_lrwd_sweep.py " + f"--stage b --best-lr {args}" + ) + + +if __name__ == "__main__": + main() diff --git a/src/electrai/configs/MP/config_gga_gga+u_f32.yaml b/src/electrai/configs/MP/config_gga_gga+u_f32.yaml new file mode 100644 index 00000000..dde48c02 --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_f32.yaml @@ -0,0 +1,48 @@ +# Dataset / loader parameters +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga/mp_filelist.txt + split_file: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga/split.json + dataset_id: 1 + - root: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga+u/mp_filelist.txt + split_file: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga+u/split.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 + +# Model +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 + +# Training parameters +precision: 32 +epochs: 100 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 + +# Weights and biases +wandb_mode: online +entity: PrinceOA +wb_pname: mp-large-scale +run_name: gga_gga+u_f32 + +# checkpoints +ckpt_path: /scratch/gpfs/ROSENGROUP/ho0950/electrai/examples/MP/experiments/chgcar/functionals/gga_gga+u_f32/checkpoints + +# set HF as well diff --git a/src/electrai/configs/MP/config_gga_gga+u_f32_smoke.yaml b/src/electrai/configs/MP/config_gga_gga+u_f32_smoke.yaml new file mode 100644 index 00000000..3c0427b5 --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_f32_smoke.yaml @@ -0,0 +1,56 @@ +# Smoke test for config_gga_gga+u_f32 on Modal: subset filelists + few epochs. +# Validates data wiring, DDP/checkpoint path, and wandb logging before the full run. +# +# Create the subset filelists on the Volume first (after the Globus transfer): +# modal run modal/prep_volume.py # also (re)links data/label, commits the Volume +# +# Paths are already in Volume (/data) form, so train.py's della->Volume remap is a no-op. + +# Dataset / loader parameters +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /data/mp/chg_datasets/functionals/gga/mp_filelist_smoke.txt + split_file: null + val_frac: 0.1 + dataset_id: 1 + - root: /data/mp/chg_datasets/functionals/gga+u/mp_filelist_smoke.txt + split_file: null + val_frac: 0.1 + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 + +# Model +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 + +# Training parameters +precision: 32 +epochs: 2 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 + +# Weights and biases +wandb_mode: online +entity: PrinceOA +wb_pname: mp-large-scale +run_name: gga_gga+u_f32_smoke + +# checkpoints (overridden by modal/train.py to /checkpoints/) +ckpt_path: /checkpoints/gga_gga+u_f32_smoke diff --git a/src/electrai/configs/MP/config_gga_gga+u_w128.yaml b/src/electrai/configs/MP/config_gga_gga+u_w128.yaml new file mode 100644 index 00000000..0b265397 --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w128.yaml @@ -0,0 +1,70 @@ +# Width-128 point of the width ablation. IDENTICAL to config_gga_gga+u_w96.yaml +# (same capped GGA + GGA+U dataset/splits, bf16-mixed, lr 0.001, warmup 4) with +# a SINGLE delta so the only variable vs W96 is width: +# - model.n_channels: 96 -> 128 +# Compare against W96 on this exact dataset (W&B mp-gga-ggau-width run +# `revived-energy-7`/z21di7sl lineage: val_loss 0.007929 at epoch 12) and W64 +# (best 0.008398 at epoch 16, run zz3oecp7). +# +# Memory: ~193M params; activations ~4/3 of W96's measured 28.2 GiB peak at the +# 180^3 cap -> ~40-45 GiB expected, no activation checkpointing needed on +# GB200 (185 GiB). Verify with scripts/coreweave/dry_run.py before launch. +# Compute: conv FLOPs ~(128/96)^2 = 1.78x W96 per step; wall-clock increase is +# smaller because larger channel counts improve GPU occupancy (W96 MFU ~10%). +# Reuses the same *_capped filelists/splits as W64/W96; do NOT re-cap. +# +# Paths target the CoreWeave marin-us-east-08a cluster: dataset staged from the +# rhoarnet-us-east-08a CAIOS bucket to node-local NVMe by +# scripts/coreweave/stage_data.sh (run before training; run_training.sh does). +# Tracked in W&B project `mp-gga-ggau-width`. + +# Dataset / loader parameters +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 + +# Model +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 128 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 + +# Training parameters +precision: bf16-mixed +epochs: 100 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 + +# Weights and biases — OFFLINE + sidecar sync: online wandb.init crashes with +# the viewer flags=null TypeError and a rank-0 death silently deadlocks DDP. +# run_training.sh re-syncs the offline run dirs every CKPT_SYNC_S seconds. +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: gga_gga+u_w128 + +# checkpoints: node-local NVMe (survives same-node pod restarts); synced to and +# restored from the CAIOS bucket by run_training.sh (path derived from this +# config's filename stem — keep them consistent) +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w128 diff --git a/src/electrai/configs/MP/config_gga_gga+u_w128_lr2e-3.yaml b/src/electrai/configs/MP/config_gga_gga+u_w128_lr2e-3.yaml new file mode 100644 index 00000000..ead2777a --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w128_lr2e-3.yaml @@ -0,0 +1,41 @@ +# W128 at the sweep-tuned lr 2e-3 (vs the live 1e-3 run, W&B mp-gga-ggau-width). +# Logs to its OWN W&B project so the two W128 runs stay separate. Same +# 100-epoch cosine and capped dataset as the incumbent -> matched-epoch +# comparison against gga_gga+u_w128 tells us what 1e-3 left on the table. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 128 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 100 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-w128-lr2e3 +run_name: config_gga_gga+u_w128_lr2e-3 +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w128_lr2e-3 diff --git a/src/electrai/configs/MP/config_gga_gga+u_w128_test_della.yaml b/src/electrai/configs/MP/config_gga_gga+u_w128_test_della.yaml new file mode 100644 index 00000000..eb651c3f --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w128_test_della.yaml @@ -0,0 +1,55 @@ +# Test-set eval of the W128 lr2e-3 checkpoint (epoch 55, val 0.005785) on Della. +# Withheld test set = the `test` key of split_capped.json (GGA 1700 + GGA+U 524), +# staged from s3://oa-electrai into bb9080 scratch and remapped onto the UNCAPPED +# COMMON filelists (byte-identical to the bucket's) so the shared functionals +# dirs are used read-only. gga+u on Della is named gga+u_sad (filelist verified +# identical to the bucket's gga+u). +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga/mp_filelist.txt + split_file: /scratch/gpfs/ROSENGROUP/bb9080/w128_test_eval/gga/split_capped_on_uncapped.json + dataset_id: 1 + # gga+u_pads, NOT gga+u_sad: the trained model's gga+u inputs are PADS. + # Probe job 12111541 on 3 test samples: PADS 0.29-1.15% NMAE vs SAD + # 8.6-14.5% (worse than raw SAD input error) — the staged CoreWeave "gga+u" + # data was PADS content. Labels/filelists are identical between variants. + - root: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga+u_pads/mp_filelist.txt + split_file: /scratch/gpfs/ROSENGROUP/bb9080/w128_test_eval/gga+u/split_capped_on_uncapped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 0 + val_workers: 4 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 128 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +# fp32, matching the proven W128 throughput benchmark on this checkpoint/GPU +# (job 12046200, peak 45.5 GB). bf16-mixed autocast OOM'd on A100: a single +# ~75 GiB allocation inside a decoder conv on a 108^3 sample (job 12049722) — +# pathological cuDNN algo/workspace for bf16 5^3 convs on A100 (training ran +# bf16 fine on GB200, a different GPU/cuDNN). +precision: 32 +epochs: 100 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 +wandb_mode: disabled +entity: PrinceOA +wb_pname: mp-gga-ggau-w128-lr2e3 +run_name: w128_epoch55_test_eval +ckpt_path: /scratch/gpfs/ROSENGROUP/bb9080/w128_test_eval/ckpt +save_pred: false +log_dir: /scratch/gpfs/ROSENGROUP/bb9080/w128_test_eval/results +out_dir: /scratch/gpfs/ROSENGROUP/bb9080/w128_test_eval/results/preds diff --git a/src/electrai/configs/MP/config_gga_gga+u_w160.yaml b/src/electrai/configs/MP/config_gga_gga+u_w160.yaml new file mode 100644 index 00000000..8feb75be --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w160.yaml @@ -0,0 +1,61 @@ +# Width-160 point of the width ablation. Same recipe as W96/W128 (capped +# GGA + GGA+U dataset/splits, bf16-mixed, lr 0.001, warmup 4) with a SINGLE +# delta: model.n_channels -> 160 (~302M params). This is the WIDEST width +# that clears the cuDNN large-tensor kernel cliff on the UNMODIFIED capped +# dataset (measured 2026-07-22: W160@180^3 = 2.36s/step, 48.3 GiB peak; +# W176@180^3 = 590s/step — cliff between 1.87e9 and 2.05e9 concat elements). +# Full comparability with W64/W96/W128: same filelists, do NOT re-cap. +# Est ~8.5-9h/epoch on 4x GB200 (~7 days to a W96-style plateau). +# lr 0.001 / warmup 4 kept for single-variable comparability; note the +# optimizer regime is increasingly untuned at this scale. +# Tracked in W&B project `mp-gga-ggau-width`. +# Dataset / loader parameters +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 + +# Model +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 160 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 + +# Training parameters +precision: bf16-mixed +epochs: 100 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 + +# Weights and biases — OFFLINE + sidecar sync: online wandb.init crashes with +# the viewer flags=null TypeError and a rank-0 death silently deadlocks DDP. +# run_training.sh re-syncs the offline run dirs every CKPT_SYNC_S seconds. +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: gga_gga+u_w160 + +# checkpoints: node-local NVMe (survives same-node pod restarts); synced to and +# restored from the CAIOS bucket by run_training.sh (path derived from this +# config's filename stem — keep them consistent) +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w160 diff --git a/src/electrai/configs/MP/config_gga_gga+u_w192.yaml b/src/electrai/configs/MP/config_gga_gga+u_w192.yaml new file mode 100644 index 00000000..21807344 --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w192.yaml @@ -0,0 +1,58 @@ +# Width-192 point of the width ablation. Same recipe as W96/W128 (capped +# GGA + GGA+U, bf16-mixed, lr 0.001, warmup 4) with n_channels: 192 (~435M +# params). Motivated by the W256 kernel cliff: cuDNN heuristics select +# pathological conv kernels for 256-channel convs on grids >~150^3 (1495s/step +# at 180^3), and benchmark-mode search is unusable with this dataset's +# variable shapes — W192 is the next rung IF it stays on the fast side of the +# cliff. Validate with scripts/coreweave/dry_run.py --grids 128,160,180 +# --iters 2 in heuristic mode before any launch. Do NOT re-cap the filelists. +# Dataset / loader parameters +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 + +# Model +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 192 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 + +# Training parameters +precision: bf16-mixed +epochs: 100 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 + +# Weights and biases — OFFLINE + sidecar sync: online wandb.init crashes with +# the viewer flags=null TypeError and a rank-0 death silently deadlocks DDP. +# run_training.sh re-syncs the offline run dirs every CKPT_SYNC_S seconds. +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: gga_gga+u_w192 + +# checkpoints: node-local NVMe (survives same-node pod restarts); synced to and +# restored from the CAIOS bucket by run_training.sh (path derived from this +# config's filename stem — keep them consistent) +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w192 diff --git a/src/electrai/configs/MP/config_gga_gga+u_w256.yaml b/src/electrai/configs/MP/config_gga_gga+u_w256.yaml new file mode 100644 index 00000000..0b878f3a --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w256.yaml @@ -0,0 +1,58 @@ +# Width-256 EXPLORATORY point of the width ablation. Same recipe as W96/W128 +# (capped GGA + GGA+U, bf16-mixed, lr 0.001, warmup 4) with n_channels: 256. +# ~774M params. Projected from measured W96/W128 scaling: peak ~85 GiB at the +# 180^3 cap (fits GB200 185 GiB without activation checkpointing), conv FLOPs +# 4x W128 -> est 12-16h/epoch on 4x GB200. CAVEAT: at this scale lr 0.001 / +# warmup 4 may be the wrong optimizer regime — a launch is a deliberate +# decision, not implied by this config existing. Validate with +# scripts/coreweave/dry_run.py first. Do NOT re-cap the filelists. +# Dataset / loader parameters +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 + +# Model +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 256 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 + +# Training parameters +precision: bf16-mixed +epochs: 100 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 + +# Weights and biases — OFFLINE + sidecar sync: online wandb.init crashes with +# the viewer flags=null TypeError and a rank-0 death silently deadlocks DDP. +# run_training.sh re-syncs the offline run dirs every CKPT_SYNC_S seconds. +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: gga_gga+u_w256 + +# checkpoints: node-local NVMe (survives same-node pod restarts); synced to and +# restored from the CAIOS bucket by run_training.sh (path derived from this +# config's filename stem — keep them consistent) +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w256 diff --git a/src/electrai/configs/MP/config_gga_gga+u_w32_lr2e-3.yaml b/src/electrai/configs/MP/config_gga_gga+u_w32_lr2e-3.yaml new file mode 100644 index 00000000..c55de657 --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w32_lr2e-3.yaml @@ -0,0 +1,43 @@ +# Stage C of the LR/WD width sweep (docs/lr_wd_width_sweep.md): full-scale +# confirmation at the tuned recipe. Identical to the incumbent production +# config INCLUDING the 100-epoch cosine -- only lr differs -- so val curves +# compare against the incumbent at matched epochs. Budget: halt manually +# around epoch 10-12 (do not set epochs lower: that would compress the +# cosine and flatter this run vs the incumbent). +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 100 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: config_gga_gga+u_w32_lr2e-3 +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w32_lr2e-3 diff --git a/src/electrai/configs/MP/config_gga_gga+u_w64.yaml b/src/electrai/configs/MP/config_gga_gga+u_w64.yaml new file mode 100644 index 00000000..560ff2a2 --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w64.yaml @@ -0,0 +1,60 @@ +# Width-64 variant of the mp-large-scale run (config_gga_gga+u_f32.yaml). +# Same GGA + GGA+U hyperparameters, with four deltas: +# - model.n_channels: 32 -> 64 (width 64) +# - precision: 32 -> bf16-mixed (fp32 OOMs at W64; fp16 overflowed to NaN at +# epoch 3 on 2026-07-13, so use bf16 for its fp32-range exponent) +# - warmup_length: 1 -> 4 +# - grid-size cap: uses *_capped filelists/splits (structures with grids +# >5,832,000 voxels ~180^3 dropped, ~1.5%). Without the cap, the largest +# structures (up to ~540^3) make a single DDP step exceed the 30-min NCCL +# watchdog and crash-loop. Regenerate with scripts/lambda/cap_filelists.py. +# Tracked in W&B project `mp-gga-ggau-width`. + +# Dataset / loader parameters +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /scratch/gpfs/ROSENGROUP/common/globus_share_OA/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 + +# Model +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 + +# Training parameters +precision: bf16-mixed +epochs: 100 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 + +# Weights and biases +wandb_mode: online +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: gga_gga+u_w64 + +# checkpoints (overridden by scripts/lambda/run_training.sh to $CKPT_ROOT/_) +ckpt_path: ./checkpoints/gga_gga+u_w64 + +# set HF as well diff --git a/src/electrai/configs/MP/config_gga_gga+u_w64_lr4e-3.yaml b/src/electrai/configs/MP/config_gga_gga+u_w64_lr4e-3.yaml new file mode 100644 index 00000000..ae0cd94b --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w64_lr4e-3.yaml @@ -0,0 +1,43 @@ +# Stage C of the LR/WD width sweep (docs/lr_wd_width_sweep.md): full-scale +# confirmation at the tuned recipe. Identical to the incumbent production +# config INCLUDING the 100-epoch cosine -- only lr differs -- so val curves +# compare against the incumbent at matched epochs. Budget: halt manually +# around epoch 10-12 (do not set epochs lower: that would compress the +# cosine and flatter this run vs the incumbent). +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 100 +lr: 0.004 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: config_gga_gga+u_w64_lr4e-3 +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w64_lr4e-3 diff --git a/src/electrai/configs/MP/config_gga_gga+u_w96.yaml b/src/electrai/configs/MP/config_gga_gga+u_w96.yaml new file mode 100644 index 00000000..50142b9e --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w96.yaml @@ -0,0 +1,72 @@ +# Width-96 point of the width ablation. IDENTICAL to config_gga_gga+u_w64.yaml +# (same capped GGA + GGA+U dataset/splits, bf16-mixed, lr 0.001, warmup 4) with +# a SINGLE delta so the only variable vs W64 is width: +# - model.n_channels: 64 -> 96 +# Compare against the W64 result on this exact dataset: val NMAE 0.84% +# (ckpt_epoch=16_val_loss=0.008398), W&B mp-gga-ggau-width run zz3oecp7. +# +# Memory: W96 activations ~1.5x W64 (~115 GB), which OOMs an 80 GB A100/H100. +# This run targets a GB200 (185 GiB visible/GPU), so it fits WITHOUT activation +# checkpointing -> `use_checkpoint` stays off, matching W64 exactly. If ever run +# on 80 GB GPUs, set model.use_checkpoint: True (costs ~30% compute). +# Reuses the same *_capped filelists/splits as W64 (validated 111,257 ids); +# do NOT re-cap, or the W64 comparison breaks. +# +# Paths target the CoreWeave marin-us-east-08a cluster: the dataset is staged +# from the rhoarnet-us-east-08a CAIOS bucket to node-local NVMe by +# scripts/coreweave/stage_data.sh, which also creates the functionals// +# {data,label} symlinks RhoRead needs. Run stage_data.sh before training. +# Tracked in W&B project `mp-gga-ggau-width`. + +# Dataset / loader parameters +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 + +# Model +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 + +# Training parameters +precision: bf16-mixed +epochs: 100 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 + +# Weights and biases — OFFLINE + sidecar sync: online wandb.init crashes with +# the viewer flags=null TypeError (same wandb bug as lambda2; confirmed on +# GB200 2026-07-18) and the rank-0 death silently deadlocks DDP. +# run_training.sh re-syncs the offline run dirs every CKPT_SYNC_S seconds. +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: gga_gga+u_w96 + +# checkpoints: node-local NVMe (survives same-node pod restarts); synced to and +# restored from the CAIOS bucket by the CoreWeave run wrapper +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w96 + +# set HF as well diff --git a/src/electrai/configs/MP/config_gga_gga+u_w96_lr2e-3.yaml b/src/electrai/configs/MP/config_gga_gga+u_w96_lr2e-3.yaml new file mode 100644 index 00000000..30ce4748 --- /dev/null +++ b/src/electrai/configs/MP/config_gga_gga+u_w96_lr2e-3.yaml @@ -0,0 +1,43 @@ +# Stage C of the LR/WD width sweep (docs/lr_wd_width_sweep.md): full-scale +# confirmation at the tuned recipe. Identical to the incumbent production +# config INCLUDING the 100-epoch cosine -- only lr differs -- so val curves +# compare against the incumbent at matched epochs. Budget: halt manually +# around epoch 10-12 (do not set epochs lower: that would compress the +# cosine and flatter this run vs the incumbent). +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga/split_capped.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/split_capped.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 100 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 4 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-width +run_name: config_gga_gga+u_w96_lr2e-3 +ckpt_path: /uv/cache/electrai/checkpoints/gga_gga+u_w96_lr2e-3 diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.00025_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.00025_wd0.yaml new file mode 100644 index 00000000..dbbcd22b --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.00025_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.00025 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.00025_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.00025_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.0005_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.0005_wd0.yaml new file mode 100644 index 00000000..dd39944b --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.0005_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.0005 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.0005_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.0005_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.001_wd0.001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.001_wd0.001.yaml new file mode 100644 index 00000000..75d7185c --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.001_wd0.001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.001 +weight_decay: 0.001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.001_wd0.001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.001_wd0.001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.001_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.001_wd0.yaml new file mode 100644 index 00000000..d1cf47ba --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.001_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.001_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.001_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.0001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.0001.yaml new file mode 100644 index 00000000..c942a9e1 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.0001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.0001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.002_wd0.0001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.002_wd0.0001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.001.yaml new file mode 100644 index 00000000..8dc20598 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.002_wd0.001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.002_wd0.001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.01.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.01.yaml new file mode 100644 index 00000000..4702236e --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.01.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.01 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.002_wd0.01 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.002_wd0.01 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.yaml new file mode 100644 index 00000000..c7189368 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.002_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.002_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0_rep2.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0_rep2.yaml new file mode 100644 index 00000000..6ef2e287 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.002_wd0_rep2.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.002_wd0_rep2 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.002_wd0_rep2 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.004_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.004_wd0.yaml new file mode 100644 index 00000000..1c52ab16 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w32_lr0.004_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 32 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.004 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w32_lr0.004_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w32_lr0.004_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.00025_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.00025_wd0.yaml new file mode 100644 index 00000000..769add74 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.00025_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.00025 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.00025_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.00025_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.0005_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.0005_wd0.yaml new file mode 100644 index 00000000..f6221379 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.0005_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.0005 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.0005_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.0005_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.001_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.001_wd0.yaml new file mode 100644 index 00000000..44eb4fa3 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.001_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.001_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.001_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.002_wd0.001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.002_wd0.001.yaml new file mode 100644 index 00000000..7f7d3eef --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.002_wd0.001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.002_wd0.001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.002_wd0.001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.002_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.002_wd0.yaml new file mode 100644 index 00000000..87e1d6b1 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.002_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.002_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.002_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.0001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.0001.yaml new file mode 100644 index 00000000..c020d6a6 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.0001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.004 +weight_decay: 0.0001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.004_wd0.0001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.004_wd0.0001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.001.yaml new file mode 100644 index 00000000..57eec956 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.004 +weight_decay: 0.001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.004_wd0.001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.004_wd0.001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.01.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.01.yaml new file mode 100644 index 00000000..76d5dbfe --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.01.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.004 +weight_decay: 0.01 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.004_wd0.01 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.004_wd0.01 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.yaml new file mode 100644 index 00000000..422b3e37 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.004 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.004_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.004_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0_rep2.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0_rep2.yaml new file mode 100644 index 00000000..df8a6ea2 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.004_wd0_rep2.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.004 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.004_wd0_rep2 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.004_wd0_rep2 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.008_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.008_wd0.yaml new file mode 100644 index 00000000..7ae605ab --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w64_lr0.008_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 64 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.008 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w64_lr0.008_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w64_lr0.008_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.00025_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.00025_wd0.yaml new file mode 100644 index 00000000..9ca19f75 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.00025_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.00025 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.00025_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.00025_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.0005_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.0005_wd0.yaml new file mode 100644 index 00000000..de10e352 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.0005_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.0005 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.0005_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.0005_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.001_wd0.001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.001_wd0.001.yaml new file mode 100644 index 00000000..67522f52 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.001_wd0.001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.001 +weight_decay: 0.001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.001_wd0.001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.001_wd0.001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.001_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.001_wd0.yaml new file mode 100644 index 00000000..df3ea370 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.001_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.001 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.001_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.001_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.0001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.0001.yaml new file mode 100644 index 00000000..e710713a --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.0001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.0001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.002_wd0.0001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.002_wd0.0001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.001.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.001.yaml new file mode 100644 index 00000000..ea0b4330 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.001.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.001 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.002_wd0.001 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.002_wd0.001 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.01.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.01.yaml new file mode 100644 index 00000000..269578ad --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.01.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.01 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.002_wd0.01 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.002_wd0.01 +optimizer: adamw diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.yaml new file mode 100644 index 00000000..1223cc94 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.002_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.002_wd0 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0_rep2.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0_rep2.yaml new file mode 100644 index 00000000..08ece595 --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.002_wd0_rep2.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.002 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.002_wd0_rep2 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.002_wd0_rep2 +optimizer: adam diff --git a/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.004_wd0.yaml b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.004_wd0.yaml new file mode 100644 index 00000000..375133ad --- /dev/null +++ b/src/electrai/configs/MP/sweep_lrwd/sweep_lrwd_w96_lr0.004_wd0.yaml @@ -0,0 +1,40 @@ +# Generated by scripts/coreweave/gen_lrwd_sweep.py from config_gga_gga+u_w96.yaml +# Deltas: width/lr/wd/optimizer, 8-epoch schedule, 12K sweep splits. +data: + _target_: electrai.dataloader.dataset.RhoRead + datasets: + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga_split_sweep12k.json + dataset_id: 1 + - root: /uv/cache/electrai/mp/chg_datasets/functionals/gga+u/mp_filelist_capped.txt + split_file: data/MP/sweep_splits/gga+u_split_sweep12k.json + dataset_id: 2 + precision: f32 + batch_size: 1 + train_workers: 8 + val_workers: 2 + pin_memory: false + drop_last: false + augmentation: false + random_seed: 42 +model: + _target_: electrai.model.resunet.ResUNet3D + in_channels: 1 + out_channels: 1 + n_channels: 96 + n_residual_blocks: 1 + kernel_size: 5 + depth: 2 +precision: bf16-mixed +epochs: 8 +lr: 0.004 +weight_decay: 0.0 +warmup_length: 1 +beta1: 0.9 +beta2: 0.99 +wandb_mode: offline +entity: PrinceOA +wb_pname: mp-gga-ggau-lrwd +run_name: sweep_lrwd_w96_lr0.004_wd0 +ckpt_path: /uv/cache/electrai/checkpoints/sweep_lrwd_w96_lr0.004_wd0 +optimizer: adam diff --git a/src/electrai/dataloader/dataset.py b/src/electrai/dataloader/dataset.py index a4845f51..27d1ba7f 100644 --- a/src/electrai/dataloader/dataset.py +++ b/src/electrai/dataloader/dataset.py @@ -200,15 +200,19 @@ def __init__(self, datapath: str, precision: str, augmentation: bool, **kwargs): if not member_list: raise ValueError(f"Filelist at {datapath} is empty.") self.member_list = member_list - # Detect zarr vs CHGCAR by checking which extension the first entry has + # Detect zarr vs CHGCAR; zarr may be unpacked (`.zarr/`) or packed + # (`.zarr.zip`) — load_zarr handles both. first = member_list[0] - if (self.root / "data" / f"{first}.zarr").exists(): + data_dir = self.root / "data" + if (data_dir / f"{first}.zarr.zip").exists() or ( + data_dir / f"{first}.zarr" + ).exists(): self.fmt = "zarr" - elif (self.root / "data" / f"{first}.CHGCAR").exists(): + elif (data_dir / f"{first}.CHGCAR").exists(): self.fmt = "chgcar" else: raise ValueError( - f"No .zarr or .CHGCAR file found for '{first}' in {self.root / 'data'}" + f"No .zarr(.zip) or .CHGCAR file found for '{first}' in {data_dir}" ) def __len__(self): diff --git a/src/electrai/dataloader/utils.py b/src/electrai/dataloader/utils.py index 9c346883..c99debf6 100644 --- a/src/electrai/dataloader/utils.py +++ b/src/electrai/dataloader/utils.py @@ -40,18 +40,39 @@ def load_numpy_rho( def load_zarr(root: str | bytes | os.PathLike, index: str): + """Read a zarr store as either a directory tree (`.zarr/`) or a single + packed zip (`.zarr.zip`). Volumes capped by inode count (e.g. Modal) + benefit greatly from the packed form — one inode per store instead of ~8. + """ import zarr - def _read(path): - z = zarr.open_group(str(path), mode="r") - if "structure" not in z.attrs: - raise KeyError(f"'structure' attribute missing from zarr store at {path}") - arr = np.array(z["charge_density_total"]) - volume = json.loads(z.attrs["structure"])["lattice"]["volume"] - return arr / volume + def _read(zarr_dir: Path, zarr_zip: Path): + store = None + if zarr_zip.exists(): + store = zarr.storage.ZipStore(str(zarr_zip), mode="r") + z = zarr.open_group(store, mode="r") + elif zarr_dir.exists(): + z = zarr.open_group(str(zarr_dir), mode="r") + else: + raise FileNotFoundError(f"No zarr store at {zarr_zip} or {zarr_dir}") + try: + if "structure" not in z.attrs: + raise KeyError( + f"'structure' attribute missing from zarr store at " + f"{zarr_zip if zarr_zip.exists() else zarr_dir}" + ) + arr = np.array(z["charge_density_total"]) + volume = json.loads(z.attrs["structure"])["lattice"]["volume"] + return arr / volume + finally: + if store is not None: + store.close() - data = _read(Path(root) / "data" / f"{index}.zarr") - label = _read(Path(root) / "label" / f"{index}.zarr") + root = Path(root) + data = _read(root / "data" / f"{index}.zarr", root / "data" / f"{index}.zarr.zip") + label = _read( + root / "label" / f"{index}.zarr", root / "label" / f"{index}.zarr.zip" + ) return data, label diff --git a/src/electrai/entrypoints/train.py b/src/electrai/entrypoints/train.py index d841519e..649b6fa7 100644 --- a/src/electrai/entrypoints/train.py +++ b/src/electrai/entrypoints/train.py @@ -1,6 +1,7 @@ from __future__ import annotations import os +from datetime import UTC, datetime, timedelta from pathlib import Path from types import SimpleNamespace @@ -9,6 +10,7 @@ from hydra.utils import instantiate from lightning.pytorch import Trainer from lightning.pytorch.callbacks import LearningRateMonitor, ModelCheckpoint +from lightning.pytorch.strategies import DDPStrategy from electrai.lightning import LightningGenerator @@ -40,8 +42,15 @@ def train(args): if wandb_mode != "disabled": from lightning.pytorch.loggers import WandbLogger + # Short width-first run name (w128_0729-1912): auto-generated names + # made restart segments of different-width runs indistinguishable, + # and full config-derived names were too long for the project view. + model_cfg = getattr(cfg, "model", None) or {} + n_ch = model_cfg.get("n_channels") if isinstance(model_cfg, dict) else None + stamp = datetime.now(UTC).strftime("%m%d-%H%M") + run_name = f"w{n_ch}_{stamp}" if n_ch else getattr(cfg, "run_name", None) wandb_logger = WandbLogger( - project=cfg.wb_pname, entity=cfg.entity, config=vars(cfg) + project=cfg.wb_pname, entity=cfg.entity, name=run_name, config=vars(cfg) ) else: wandb_logger = None @@ -53,12 +62,25 @@ def train(args): save_top_k=2, mode="min", filename="ckpt_{epoch:02d}_{val_loss:.6f}", - save_last=True, + ) + + # Frequent resume checkpoint. An epoch here is tens of thousands of steps + # and can crash mid-way (e.g. a large sample tripping the NCCL watchdog), so + # save `last.ckpt` periodically; a restart then resumes instead of redoing + # the epoch from scratch. `val_loss` only exists at epoch end, so this is a + # separate monitor-less callback (save_top_k=0 -> last.ckpt only). + # Step-based, NOT train_time_interval: the wall-clock trigger needs a DDP + # broadcast to align ranks and their clocks can disagree about when the + # interval fired, which deadlocked a 4-rank run mid-epoch (ranks in + # mismatched collectives; NCCL watchdog never fires). Step counts are + # identical on every rank, so no alignment collective is needed. + last_ckpt_cb = ModelCheckpoint( + dirpath=ckpt_path, every_n_train_steps=2000, save_top_k=0, save_last=True ) lr_monitor = LearningRateMonitor(logging_interval="epoch") - callbacks = [checkpoint_cb, lr_monitor] + callbacks = [checkpoint_cb, last_ckpt_cb, lr_monitor] hf_cfg = getattr(cfg, "hf", None) if hf_cfg and hf_cfg.get("repo_id"): @@ -82,7 +104,11 @@ def train(args): precision=cfg.precision, devices="auto", num_nodes=num_nodes, - strategy="ddp", + # Raise the NCCL collective timeout from the 30-min default: a single + # large sample's forward/backward can stall the all-reduce on the other + # ranks past 30 min and abort the whole group. The grid-size cap keeps + # steps short; this is a safety net for the occasional slow one. + strategy=DDPStrategy(timeout=timedelta(hours=2)), log_every_n_steps=1, gradient_clip_val=getattr(cfg, "gradient_clip_value", 1.0), ) diff --git a/src/electrai/lightning.py b/src/electrai/lightning.py index c1123880..fc5b6916 100644 --- a/src/electrai/lightning.py +++ b/src/electrai/lightning.py @@ -59,7 +59,14 @@ def _loss_calculation(self, batch): return loss def configure_optimizers(self): - optimizer = torch.optim.Adam( + # "adam" applies weight_decay as coupled L2 (rescaled by the adaptive + # denominator); "adamw" decays weights directly. At weight_decay=0 the + # two are identical, so existing configs are unaffected. + opt_name = (getattr(self.cfg, "optimizer", None) or "adam").lower() + if opt_name not in ("adam", "adamw"): + raise ValueError(f"Unknown optimizer '{opt_name}': use 'adam' or 'adamw'") + opt_cls = torch.optim.AdamW if opt_name == "adamw" else torch.optim.Adam + optimizer = opt_cls( self.model.parameters(), lr=float(self.cfg.lr), weight_decay=float(self.cfg.weight_decay), diff --git a/uv.lock b/uv.lock index abf58e23..e203a0b4 100644 --- a/uv.lock +++ b/uv.lock @@ -4,9 +4,11 @@ requires-python = ">=3.11, <3.14" resolution-markers = [ "python_full_version >= '3.12' and sys_platform == 'win32'", "python_full_version < '3.12' and sys_platform == 'win32'", - "python_full_version >= '3.12' and sys_platform == 'linux'", + "python_full_version >= '3.12' and platform_machine == 'aarch64' and sys_platform == 'linux'", + "python_full_version >= '3.12' and platform_machine != 'aarch64' and sys_platform == 'linux'", "python_full_version >= '3.12' and sys_platform != 'linux' and sys_platform != 'win32'", - "python_full_version < '3.12' and sys_platform == 'linux'", + "python_full_version < '3.12' and platform_machine == 'aarch64' and sys_platform == 'linux'", + "python_full_version < '3.12' and platform_machine != 'aarch64' and sys_platform == 'linux'", "python_full_version < '3.12' and sys_platform != 'linux' and sys_platform != 'win32'", ] @@ -574,12 +576,16 @@ name = "cuda-bindings" version = "12.9.4" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "cuda-pathfinder", marker = "sys_platform == 'linux'" }, + { name = "cuda-pathfinder" }, ] wheels = [ + { url = "https://files.pythonhosted.org/packages/a9/2b/ebcbb60aa6dba830474cd360c42e10282f7a343c0a1f58d24fbd3b7c2d77/cuda_bindings-12.9.4-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:a6a429dc6c13148ff1e27c44f40a3dd23203823e637b87fd0854205195988306", size = 11840604, upload-time = "2025-10-21T14:51:34.565Z" }, { url = "https://files.pythonhosted.org/packages/45/e7/b47792cc2d01c7e1d37c32402182524774dadd2d26339bd224e0e913832e/cuda_bindings-12.9.4-cp311-cp311-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:c912a3d9e6b6651853eed8eed96d6800d69c08e94052c292fec3f282c5a817c9", size = 12210593, upload-time = "2025-10-21T14:51:36.574Z" }, + { url = "https://files.pythonhosted.org/packages/0c/c2/65bfd79292b8ff18be4dd7f7442cea37bcbc1a228c1886f1dea515c45b67/cuda_bindings-12.9.4-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:694ba35023846625ef471257e6b5a4bc8af690f961d197d77d34b1d1db393f56", size = 11760260, upload-time = "2025-10-21T14:51:40.79Z" }, { url = "https://files.pythonhosted.org/packages/a9/c1/dabe88f52c3e3760d861401bb994df08f672ec893b8f7592dc91626adcf3/cuda_bindings-12.9.4-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fda147a344e8eaeca0c6ff113d2851ffca8f7dfc0a6c932374ee5c47caa649c8", size = 12151019, upload-time = "2025-10-21T14:51:43.167Z" }, + { url = "https://files.pythonhosted.org/packages/05/8b/b4b2d1c7775fa403b64333e720cfcfccef8dcb9cdeb99947061ca5a77628/cuda_bindings-12.9.4-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:cf8bfaedc238f3b115d957d1fd6562b7e8435ba57f6d0e2f87d0e7149ccb2da5", size = 11570071, upload-time = "2025-10-21T14:51:47.472Z" }, { url = "https://files.pythonhosted.org/packages/63/56/e465c31dc9111be3441a9ba7df1941fe98f4aa6e71e8788a3fb4534ce24d/cuda_bindings-12.9.4-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:32bdc5a76906be4c61eb98f546a6786c5773a881f3b166486449b5d141e4a39f", size = 11906628, upload-time = "2025-10-21T14:51:49.905Z" }, + { url = "https://files.pythonhosted.org/packages/ec/07/6aff13bc1e977e35aaa6b22f52b172e2890c608c6db22438cf7ed2bf43a6/cuda_bindings-12.9.4-cp313-cp313t-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3adf4958dcf68ae7801a59b73fb00a8b37f8d0595060d66ceae111b1002de38d", size = 11566797, upload-time = "2025-10-21T14:51:54.581Z" }, { url = "https://files.pythonhosted.org/packages/a3/84/1e6be415e37478070aeeee5884c2022713c1ecc735e6d82d744de0252eee/cuda_bindings-12.9.4-cp313-cp313t-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:56e0043c457a99ac473ddc926fe0dc4046694d99caef633e92601ab52cbe17eb", size = 11925991, upload-time = "2025-10-21T14:51:56.535Z" }, ] @@ -632,8 +638,10 @@ dependencies = [ { name = "pymatgen" }, { name = "pyyaml" }, { name = "scikit-learn" }, - { name = "torch" }, - { name = "torchvision" }, + { name = "torch", version = "2.10.0", source = { registry = "https://pypi.org/simple" }, marker = "platform_machine != 'aarch64' or sys_platform != 'linux'" }, + { name = "torch", version = "2.10.0+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" }, marker = "platform_machine == 'aarch64' and sys_platform == 'linux'" }, + { name = "torchvision", version = "0.25.0", source = { registry = "https://pypi.org/simple" }, marker = "platform_machine != 'aarch64' or sys_platform != 'linux'" }, + { name = "torchvision", version = "0.25.0+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" }, marker = "platform_machine == 'aarch64' and sys_platform == 'linux'" }, { name = "wandb" }, { name = "zarr" }, ] @@ -692,8 +700,10 @@ requires-dist = [ { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.0.285" }, { name = "s3fs", marker = "extra == 'zarr-conversion'", specifier = ">=2024.5.0" }, { name = "scikit-learn", specifier = ">=1.7.2" }, - { name = "torch", specifier = "~=2.10.0" }, - { name = "torchvision", specifier = ">=0.24.0" }, + { name = "torch", marker = "platform_machine != 'aarch64' or sys_platform != 'linux'", specifier = "~=2.10.0" }, + { name = "torch", marker = "platform_machine == 'aarch64' and sys_platform == 'linux'", specifier = "~=2.10.0", index = "https://download.pytorch.org/whl/cu128" }, + { name = "torchvision", marker = "platform_machine != 'aarch64' or sys_platform != 'linux'", specifier = ">=0.24.0" }, + { name = "torchvision", marker = "platform_machine == 'aarch64' and sys_platform == 'linux'", specifier = ">=0.24.0", index = "https://download.pytorch.org/whl/cu128" }, { name = "wandb", specifier = ">=0.12.10" }, { name = "zarr", specifier = ">=3.1.3" }, { name = "zarr", marker = "extra == 'zarr-conversion'", specifier = ">=3.1.3" }, @@ -1113,7 +1123,8 @@ dependencies = [ { name = "packaging" }, { name = "pytorch-lightning" }, { name = "pyyaml" }, - { name = "torch" }, + { name = "torch", version = "2.10.0", source = { registry = "https://pypi.org/simple" }, marker = "platform_machine != 'aarch64' or sys_platform != 'linux'" }, + { name = "torch", version = "2.10.0+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" }, marker = "platform_machine == 'aarch64' and sys_platform == 'linux'" }, { name = "torchmetrics" }, { name = "tqdm" }, { name = "typing-extensions" }, @@ -1638,6 +1649,7 @@ name = "nvidia-cublas-cu12" version = "12.8.4.1" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/29/99/db44d685f0e257ff0e213ade1964fc459b4a690a73293220e98feb3307cf/nvidia_cublas_cu12-12.8.4.1-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:b86f6dd8935884615a0683b663891d43781b819ac4f2ba2b0c9604676af346d0", size = 590537124, upload-time = "2025-03-07T01:43:53.556Z" }, { url = "https://files.pythonhosted.org/packages/dc/61/e24b560ab2e2eaeb3c839129175fb330dfcfc29e5203196e5541a4c44682/nvidia_cublas_cu12-12.8.4.1-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:8ac4e771d5a348c551b2a426eda6193c19aa630236b418086020df5ba9667142", size = 594346921, upload-time = "2025-03-07T01:44:31.254Z" }, ] @@ -1646,6 +1658,7 @@ name = "nvidia-cuda-cupti-cu12" version = "12.8.90" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/d5/1f/b3bd73445e5cb342727fd24fe1f7b748f690b460acadc27ea22f904502c8/nvidia_cuda_cupti_cu12-12.8.90-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:4412396548808ddfed3f17a467b104ba7751e6b58678a4b840675c56d21cf7ed", size = 9533318, upload-time = "2025-03-07T01:40:10.421Z" }, { url = "https://files.pythonhosted.org/packages/f8/02/2adcaa145158bf1a8295d83591d22e4103dbfd821bcaf6f3f53151ca4ffa/nvidia_cuda_cupti_cu12-12.8.90-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:ea0cb07ebda26bb9b29ba82cda34849e73c166c18162d3913575b0c9db9a6182", size = 10248621, upload-time = "2025-03-07T01:40:21.213Z" }, ] @@ -1655,6 +1668,7 @@ version = "12.8.93" source = { registry = "https://pypi.org/simple" } wheels = [ { url = "https://files.pythonhosted.org/packages/05/6b/32f747947df2da6994e999492ab306a903659555dddc0fbdeb9d71f75e52/nvidia_cuda_nvrtc_cu12-12.8.93-py3-none-manylinux2010_x86_64.manylinux_2_12_x86_64.whl", hash = "sha256:a7756528852ef889772a84c6cd89d41dfa74667e24cca16bb31f8f061e3e9994", size = 88040029, upload-time = "2025-03-07T01:42:13.562Z" }, + { url = "https://files.pythonhosted.org/packages/eb/d1/e50d0acaab360482034b84b6e27ee83c6738f7d32182b987f9c7a4e32962/nvidia_cuda_nvrtc_cu12-12.8.93-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:fc1fec1e1637854b4c0a65fb9a8346b51dd9ee69e61ebaccc82058441f15bce8", size = 43106076, upload-time = "2025-03-07T01:41:59.817Z" }, ] [[package]] @@ -1662,6 +1676,7 @@ name = "nvidia-cuda-runtime-cu12" version = "12.8.90" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/7c/75/f865a3b236e4647605ea34cc450900854ba123834a5f1598e160b9530c3a/nvidia_cuda_runtime_cu12-12.8.90-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:52bf7bbee900262ffefe5e9d5a2a69a30d97e2bc5bb6cc866688caa976966e3d", size = 965265, upload-time = "2025-03-07T01:39:43.533Z" }, { url = "https://files.pythonhosted.org/packages/0d/9b/a997b638fcd068ad6e4d53b8551a7d30fe8b404d6f1804abf1df69838932/nvidia_cuda_runtime_cu12-12.8.90-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:adade8dcbd0edf427b7204d480d6066d33902cab2a4707dcfc48a2d0fd44ab90", size = 954765, upload-time = "2025-03-07T01:40:01.615Z" }, ] @@ -1670,9 +1685,10 @@ name = "nvidia-cudnn-cu12" version = "9.10.2.21" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cublas-cu12", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cublas-cu12" }, ] wheels = [ + { url = "https://files.pythonhosted.org/packages/fa/41/e79269ce215c857c935fd86bcfe91a451a584dfc27f1e068f568b9ad1ab7/nvidia_cudnn_cu12-9.10.2.21-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:c9132cc3f8958447b4910a1720036d9eff5928cc3179b0a51fb6d167c6cc87d8", size = 705026878, upload-time = "2025-06-06T21:52:51.348Z" }, { url = "https://files.pythonhosted.org/packages/ba/51/e123d997aa098c61d029f76663dedbfb9bc8dcf8c60cbd6adbe42f76d049/nvidia_cudnn_cu12-9.10.2.21-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:949452be657fa16687d0930933f032835951ef0892b37d2d53824d1a84dc97a8", size = 706758467, upload-time = "2025-06-06T21:54:08.597Z" }, ] @@ -1681,9 +1697,10 @@ name = "nvidia-cufft-cu12" version = "11.3.3.83" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-nvjitlink-cu12", marker = "sys_platform == 'linux'" }, + { name = "nvidia-nvjitlink-cu12" }, ] wheels = [ + { url = "https://files.pythonhosted.org/packages/60/bc/7771846d3a0272026c416fbb7e5f4c1f146d6d80704534d0b187dd6f4800/nvidia_cufft_cu12-11.3.3.83-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:848ef7224d6305cdb2a4df928759dca7b1201874787083b6e7550dd6765ce69a", size = 193109211, upload-time = "2025-03-07T01:44:56.873Z" }, { url = "https://files.pythonhosted.org/packages/1f/13/ee4e00f30e676b66ae65b4f08cb5bcbb8392c03f54f2d5413ea99a5d1c80/nvidia_cufft_cu12-11.3.3.83-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:4d2dd21ec0b88cf61b62e6b43564355e5222e4a3fb394cac0db101f2dd0d4f74", size = 193118695, upload-time = "2025-03-07T01:45:27.821Z" }, ] @@ -1693,6 +1710,7 @@ version = "1.13.1.3" source = { registry = "https://pypi.org/simple" } wheels = [ { url = "https://files.pythonhosted.org/packages/bb/fe/1bcba1dfbfb8d01be8d93f07bfc502c93fa23afa6fd5ab3fc7c1df71038a/nvidia_cufile_cu12-1.13.1.3-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:1d069003be650e131b21c932ec3d8969c1715379251f8d23a1860554b1cb24fc", size = 1197834, upload-time = "2025-03-07T01:45:50.723Z" }, + { url = "https://files.pythonhosted.org/packages/1e/f5/5607710447a6fe9fd9b3283956fceeee8a06cda1d2f56ce31371f595db2a/nvidia_cufile_cu12-1.13.1.3-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:4beb6d4cce47c1a0f1013d72e02b0994730359e17801d395bdcbf20cfb3bb00a", size = 1120705, upload-time = "2025-03-07T01:45:41.434Z" }, ] [[package]] @@ -1700,6 +1718,7 @@ name = "nvidia-curand-cu12" version = "10.3.9.90" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/45/5e/92aa15eca622a388b80fbf8375d4760738df6285b1e92c43d37390a33a9a/nvidia_curand_cu12-10.3.9.90-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:dfab99248034673b779bc6decafdc3404a8a6f502462201f2f31f11354204acd", size = 63625754, upload-time = "2025-03-07T01:46:10.735Z" }, { url = "https://files.pythonhosted.org/packages/fb/aa/6584b56dc84ebe9cf93226a5cde4d99080c8e90ab40f0c27bda7a0f29aa1/nvidia_curand_cu12-10.3.9.90-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:b32331d4f4df5d6eefa0554c565b626c7216f87a06a4f56fab27c3b68a830ec9", size = 63619976, upload-time = "2025-03-07T01:46:23.323Z" }, ] @@ -1708,11 +1727,12 @@ name = "nvidia-cusolver-cu12" version = "11.7.3.90" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cublas-cu12", marker = "sys_platform == 'linux'" }, - { name = "nvidia-cusparse-cu12", marker = "sys_platform == 'linux'" }, - { name = "nvidia-nvjitlink-cu12", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cublas-cu12" }, + { name = "nvidia-cusparse-cu12" }, + { name = "nvidia-nvjitlink-cu12" }, ] wheels = [ + { url = "https://files.pythonhosted.org/packages/c8/32/f7cd6ce8a7690544d084ea21c26e910a97e077c9b7f07bf5de623ee19981/nvidia_cusolver_cu12-11.7.3.90-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:db9ed69dbef9715071232caa9b69c52ac7de3a95773c2db65bdba85916e4e5c0", size = 267229841, upload-time = "2025-03-07T01:46:54.356Z" }, { url = "https://files.pythonhosted.org/packages/85/48/9a13d2975803e8cf2777d5ed57b87a0b6ca2cc795f9a4f59796a910bfb80/nvidia_cusolver_cu12-11.7.3.90-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:4376c11ad263152bd50ea295c05370360776f8c3427b30991df774f9fb26c450", size = 267506905, upload-time = "2025-03-07T01:47:16.273Z" }, ] @@ -1721,9 +1741,10 @@ name = "nvidia-cusparse-cu12" version = "12.5.8.93" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-nvjitlink-cu12", marker = "sys_platform == 'linux'" }, + { name = "nvidia-nvjitlink-cu12" }, ] wheels = [ + { url = "https://files.pythonhosted.org/packages/bc/f7/cd777c4109681367721b00a106f491e0d0d15cfa1fd59672ce580ce42a97/nvidia_cusparse_cu12-12.5.8.93-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:9b6c161cb130be1a07a27ea6923df8141f3c295852f4b260c65f18f3e0a091dc", size = 288117129, upload-time = "2025-03-07T01:47:40.407Z" }, { url = "https://files.pythonhosted.org/packages/c2/f5/e1854cb2f2bcd4280c44736c93550cc300ff4b8c95ebe370d0aa7d2b473d/nvidia_cusparse_cu12-12.5.8.93-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:1ec05d76bbbd8b61b06a80e1eaf8cf4959c3d4ce8e711b65ebd0443bb0ebb13b", size = 288216466, upload-time = "2025-03-07T01:48:13.779Z" }, ] @@ -1732,6 +1753,7 @@ name = "nvidia-cusparselt-cu12" version = "0.7.1" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/73/b9/598f6ff36faaece4b3c50d26f50e38661499ff34346f00e057760b35cc9d/nvidia_cusparselt_cu12-0.7.1-py3-none-manylinux2014_aarch64.whl", hash = "sha256:8878dce784d0fac90131b6817b607e803c36e629ba34dc5b433471382196b6a5", size = 283835557, upload-time = "2025-02-26T00:16:54.265Z" }, { url = "https://files.pythonhosted.org/packages/56/79/12978b96bd44274fe38b5dde5cfb660b1d114f70a65ef962bcbbed99b549/nvidia_cusparselt_cu12-0.7.1-py3-none-manylinux2014_x86_64.whl", hash = "sha256:f1bb701d6b930d5a7cea44c19ceb973311500847f81b634d802b7b539dc55623", size = 287193691, upload-time = "2025-02-26T00:15:44.104Z" }, ] @@ -1740,6 +1762,7 @@ name = "nvidia-nccl-cu12" version = "2.27.5" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/bb/1c/857979db0ef194ca5e21478a0612bcdbbe59458d7694361882279947b349/nvidia_nccl_cu12-2.27.5-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:31432ad4d1fb1004eb0c56203dc9bc2178a1ba69d1d9e02d64a6938ab5e40e7a", size = 322400625, upload-time = "2025-06-26T04:11:04.496Z" }, { url = "https://files.pythonhosted.org/packages/6e/89/f7a07dc961b60645dbbf42e80f2bc85ade7feb9a491b11a1e973aa00071f/nvidia_nccl_cu12-2.27.5-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:ad730cf15cb5d25fe849c6e6ca9eb5b76db16a80f13f425ac68d8e2e55624457", size = 322348229, upload-time = "2025-06-26T04:11:28.385Z" }, ] @@ -1749,6 +1772,7 @@ version = "12.8.93" source = { registry = "https://pypi.org/simple" } wheels = [ { url = "https://files.pythonhosted.org/packages/f6/74/86a07f1d0f42998ca31312f998bd3b9a7eff7f52378f4f270c8679c77fb9/nvidia_nvjitlink_cu12-12.8.93-py3-none-manylinux2010_x86_64.manylinux_2_12_x86_64.whl", hash = "sha256:81ff63371a7ebd6e6451970684f916be2eab07321b73c9d244dc2b4da7f73b88", size = 39254836, upload-time = "2025-03-07T01:49:55.661Z" }, + { url = "https://files.pythonhosted.org/packages/2a/a2/8cee5da30d13430e87bf99bb33455d2724d0a4a9cb5d7926d80ccb96d008/nvidia_nvjitlink_cu12-12.8.93-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:adccd7161ace7261e01bb91e44e88da350895c270d23f744f0820c818b7229e7", size = 38386204, upload-time = "2025-03-07T01:49:43.612Z" }, ] [[package]] @@ -1756,6 +1780,7 @@ name = "nvidia-nvshmem-cu12" version = "3.4.5" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/1d/6a/03aa43cc9bd3ad91553a88b5f6fb25ed6a3752ae86ce2180221962bc2aa5/nvidia_nvshmem_cu12-3.4.5-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:0b48363fc6964dede448029434c6abed6c5e37f823cb43c3bcde7ecfc0457e15", size = 138936938, upload-time = "2025-09-06T00:32:05.589Z" }, { url = "https://files.pythonhosted.org/packages/b5/09/6ea3ea725f82e1e76684f0708bbedd871fc96da89945adeba65c3835a64c/nvidia_nvshmem_cu12-3.4.5-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:042f2500f24c021db8a06c5eec2539027d57460e1c1a762055a6554f72c369bd", size = 139103095, upload-time = "2025-09-06T00:32:31.266Z" }, ] @@ -1764,6 +1789,7 @@ name = "nvidia-nvtx-cu12" version = "12.8.90" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/10/c0/1b303feea90d296f6176f32a2a70b5ef230f9bdeb3a72bddb0dc922dc137/nvidia_nvtx_cu12-12.8.90-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:d7ad891da111ebafbf7e015d34879f7112832fc239ff0d7d776b6cb685274615", size = 91161, upload-time = "2025-03-07T01:42:23.922Z" }, { url = "https://files.pythonhosted.org/packages/a2/eb/86626c1bbc2edb86323022371c39aa48df6fd8b0a1647bc274577f72e90b/nvidia_nvtx_cu12-12.8.90-py3-none-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:5b17e2001cc0d751a5bc2c6ec6d26ad95913324a4adb86788c944f8ce9ba441f", size = 89954, upload-time = "2025-03-07T01:42:44.131Z" }, ] @@ -2308,7 +2334,8 @@ dependencies = [ { name = "lightning-utilities" }, { name = "packaging" }, { name = "pyyaml" }, - { name = "torch" }, + { name = "torch", version = "2.10.0", source = { registry = "https://pypi.org/simple" }, marker = "platform_machine != 'aarch64' or sys_platform != 'linux'" }, + { name = "torch", version = "2.10.0+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" }, marker = "platform_machine == 'aarch64' and sys_platform == 'linux'" }, { name = "torchmetrics" }, { name = "tqdm" }, { name = "typing-extensions" }, @@ -2747,6 +2774,14 @@ wheels = [ name = "torch" version = "2.10.0" source = { registry = "https://pypi.org/simple" } +resolution-markers = [ + "python_full_version >= '3.12' and sys_platform == 'win32'", + "python_full_version < '3.12' and sys_platform == 'win32'", + "python_full_version >= '3.12' and platform_machine != 'aarch64' and sys_platform == 'linux'", + "python_full_version >= '3.12' and sys_platform != 'linux' and sys_platform != 'win32'", + "python_full_version < '3.12' and platform_machine != 'aarch64' and sys_platform == 'linux'", + "python_full_version < '3.12' and sys_platform != 'linux' and sys_platform != 'win32'", +] dependencies = [ { name = "cuda-bindings", marker = "platform_machine == 'x86_64' and sys_platform == 'linux'" }, { name = "filelock" }, @@ -2799,6 +2834,47 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/0e/13/e76b4d9c160e89fff48bf16b449ea324bda84745d2ab30294c37c2434c0d/torch-2.10.0-cp313-none-macosx_11_0_arm64.whl", hash = "sha256:cdf2a523d699b70d613243211ecaac14fe9c5df8a0b0a9c02add60fb2a413e0f", size = 79498248, upload-time = "2026-01-21T16:23:09.315Z" }, ] +[[package]] +name = "torch" +version = "2.10.0+cu128" +source = { registry = "https://download.pytorch.org/whl/cu128" } +resolution-markers = [ + "python_full_version >= '3.12' and platform_machine == 'aarch64' and sys_platform == 'linux'", + "python_full_version < '3.12' and platform_machine == 'aarch64' and sys_platform == 'linux'", +] +dependencies = [ + { name = "cuda-bindings" }, + { name = "filelock" }, + { name = "fsspec" }, + { name = "jinja2" }, + { name = "networkx" }, + { name = "nvidia-cublas-cu12" }, + { name = "nvidia-cuda-cupti-cu12" }, + { name = "nvidia-cuda-nvrtc-cu12" }, + { name = "nvidia-cuda-runtime-cu12" }, + { name = "nvidia-cudnn-cu12" }, + { name = "nvidia-cufft-cu12" }, + { name = "nvidia-cufile-cu12" }, + { name = "nvidia-curand-cu12" }, + { name = "nvidia-cusolver-cu12" }, + { name = "nvidia-cusparse-cu12" }, + { name = "nvidia-cusparselt-cu12" }, + { name = "nvidia-nccl-cu12" }, + { name = "nvidia-nvjitlink-cu12" }, + { name = "nvidia-nvshmem-cu12" }, + { name = "nvidia-nvtx-cu12" }, + { name = "setuptools", marker = "python_full_version >= '3.12'" }, + { name = "sympy" }, + { name = "triton" }, + { name = "typing-extensions" }, +] +wheels = [ + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.10.0%2Bcu128-cp311-cp311-manylinux_2_28_aarch64.whl", hash = "sha256:85ed7944655ea6fd69377692e9cbfd7bba28d99696ceae79985e7caa99cf0a95", upload-time = "2026-01-21T15:21:36Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.10.0%2Bcu128-cp312-cp312-manylinux_2_28_aarch64.whl", hash = "sha256:6f09cdf2415516be028ae82e6b985bcfc3eac37bc52ab401142689f6224516ca", upload-time = "2026-01-21T15:22:03Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.10.0%2Bcu128-cp313-cp313-manylinux_2_28_aarch64.whl", hash = "sha256:bdbcc703382f948e951c063448c9406bf38ce66c41dd698d9e2733fcf96c037a", upload-time = "2026-01-21T15:22:29Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torch-2.10.0%2Bcu128-cp313-cp313t-manylinux_2_28_aarch64.whl", hash = "sha256:f1f8b840c64b645a4bc61a393db48effb9c92b2dc26c8373873911f0750d1ea7", upload-time = "2026-01-21T15:23:28Z" }, +] + [[package]] name = "torchmetrics" version = "1.8.2" @@ -2807,7 +2883,8 @@ dependencies = [ { name = "lightning-utilities" }, { name = "numpy" }, { name = "packaging" }, - { name = "torch" }, + { name = "torch", version = "2.10.0", source = { registry = "https://pypi.org/simple" }, marker = "platform_machine != 'aarch64' or sys_platform != 'linux'" }, + { name = "torch", version = "2.10.0+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" }, marker = "platform_machine == 'aarch64' and sys_platform == 'linux'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/85/2e/48a887a59ecc4a10ce9e8b35b3e3c5cef29d902c4eac143378526e7485cb/torchmetrics-1.8.2.tar.gz", hash = "sha256:cf64a901036bf107f17a524009eea7781c9c5315d130713aeca5747a686fe7a5", size = 580679, upload-time = "2025-09-03T14:00:54.077Z" } wheels = [ @@ -2818,10 +2895,18 @@ wheels = [ name = "torchvision" version = "0.25.0" source = { registry = "https://pypi.org/simple" } +resolution-markers = [ + "python_full_version >= '3.12' and sys_platform == 'win32'", + "python_full_version < '3.12' and sys_platform == 'win32'", + "python_full_version >= '3.12' and platform_machine != 'aarch64' and sys_platform == 'linux'", + "python_full_version >= '3.12' and sys_platform != 'linux' and sys_platform != 'win32'", + "python_full_version < '3.12' and platform_machine != 'aarch64' and sys_platform == 'linux'", + "python_full_version < '3.12' and sys_platform != 'linux' and sys_platform != 'win32'", +] dependencies = [ { name = "numpy" }, { name = "pillow" }, - { name = "torch" }, + { name = "torch", version = "2.10.0", source = { registry = "https://pypi.org/simple" } }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/3e/be/c704bceaf11c4f6b19d64337a34a877fcdfe3bd68160a8c9ae9bea4a35a3/torchvision-0.25.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:db74a551946b75d19f9996c419a799ffdf6a223ecf17c656f90da011f1d75b20", size = 1874923, upload-time = "2026-01-21T16:27:46.574Z" }, @@ -2842,6 +2927,26 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/63/cc/0ea68b5802e5e3c31f44b307e74947bad5a38cc655231d845534ed50ddb8/torchvision-0.25.0-cp313-cp313t-win_amd64.whl", hash = "sha256:5e6b449e9fa7d642142c0e27c41e5a43b508d57ed8e79b7c0a0c28652da8678c", size = 4344260, upload-time = "2026-01-21T16:27:17.018Z" }, ] +[[package]] +name = "torchvision" +version = "0.25.0+cu128" +source = { registry = "https://download.pytorch.org/whl/cu128" } +resolution-markers = [ + "python_full_version >= '3.12' and platform_machine == 'aarch64' and sys_platform == 'linux'", + "python_full_version < '3.12' and platform_machine == 'aarch64' and sys_platform == 'linux'", +] +dependencies = [ + { name = "numpy" }, + { name = "pillow" }, + { name = "torch", version = "2.10.0+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" } }, +] +wheels = [ + { url = "https://download-r2.pytorch.org/whl/cu128/torchvision-0.25.0%2Bcu128-cp311-cp311-manylinux_2_28_aarch64.whl", hash = "sha256:5d576c65d40198627e0fad03bddeb0ef536371312f2bdfcc804c22fd28fa6018", upload-time = "2026-01-21T22:32:21Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torchvision-0.25.0%2Bcu128-cp312-cp312-manylinux_2_28_aarch64.whl", hash = "sha256:8623e534ef6a815bd6407d4b52dd70c7154e2eda626ad4b9cb895d36c5a3305b", upload-time = "2026-01-21T22:32:23Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torchvision-0.25.0%2Bcu128-cp313-cp313-manylinux_2_28_aarch64.whl", hash = "sha256:12c253520a26483fe3c614f63ff16eca6d9b0b4ebe510699b7d15d88e6c0cd35", upload-time = "2026-01-21T22:32:26Z" }, + { url = "https://download-r2.pytorch.org/whl/cu128/torchvision-0.25.0%2Bcu128-cp313-cp313t-manylinux_2_28_aarch64.whl", hash = "sha256:58b2971b55c761f1d2491bd80fcc4618ea97d363d387a9dd3aff23220cbee264", upload-time = "2026-01-21T22:32:28Z" }, +] + [[package]] name = "tqdm" version = "4.67.1" @@ -2859,9 +2964,13 @@ name = "triton" version = "3.6.0" source = { registry = "https://pypi.org/simple" } wheels = [ + { url = "https://files.pythonhosted.org/packages/0f/2c/96f92f3c60387e14cc45aed49487f3486f89ea27106c1b1376913c62abe4/triton-3.6.0-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:49df5ef37379c0c2b5c0012286f80174fcf0e073e5ade1ca9a86c36814553651", size = 176081190, upload-time = "2026-01-20T16:16:00.523Z" }, { url = "https://files.pythonhosted.org/packages/e0/12/b05ba554d2c623bffa59922b94b0775673de251f468a9609bc9e45de95e9/triton-3.6.0-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e8e323d608e3a9bfcc2d9efcc90ceefb764a82b99dea12a86d643c72539ad5d3", size = 188214640, upload-time = "2026-01-20T16:00:35.869Z" }, + { url = "https://files.pythonhosted.org/packages/17/5d/08201db32823bdf77a0e2b9039540080b2e5c23a20706ddba942924ebcd6/triton-3.6.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:374f52c11a711fd062b4bfbb201fd9ac0a5febd28a96fb41b4a0f51dde3157f4", size = 176128243, upload-time = "2026-01-20T16:16:07.857Z" }, { url = "https://files.pythonhosted.org/packages/ab/a8/cdf8b3e4c98132f965f88c2313a4b493266832ad47fb52f23d14d4f86bb5/triton-3.6.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:74caf5e34b66d9f3a429af689c1c7128daba1d8208df60e81106b115c00d6fca", size = 188266850, upload-time = "2026-01-20T16:00:43.041Z" }, + { url = "https://files.pythonhosted.org/packages/3c/12/34d71b350e89a204c2c7777a9bba0dcf2f19a5bfdd70b57c4dbc5ffd7154/triton-3.6.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:448e02fe6dc898e9e5aa89cf0ee5c371e99df5aa5e8ad976a80b93334f3494fd", size = 176133521, upload-time = "2026-01-20T16:16:13.321Z" }, { url = "https://files.pythonhosted.org/packages/f9/0b/37d991d8c130ce81a8728ae3c25b6e60935838e9be1b58791f5997b24a54/triton-3.6.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:10c7f76c6e72d2ef08df639e3d0d30729112f47a56b0c81672edc05ee5116ac9", size = 188289450, upload-time = "2026-01-20T16:00:49.136Z" }, + { url = "https://files.pythonhosted.org/packages/ce/4e/41b0c8033b503fd3cfcd12392cdd256945026a91ff02452bef40ec34bee7/triton-3.6.0-cp313-cp313t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1722e172d34e32abc3eb7711d0025bb69d7959ebea84e3b7f7a341cd7ed694d6", size = 176276087, upload-time = "2026-01-20T16:16:18.989Z" }, { url = "https://files.pythonhosted.org/packages/35/f8/9c66bfc55361ec6d0e4040a0337fb5924ceb23de4648b8a81ae9d33b2b38/triton-3.6.0-cp313-cp313t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d002e07d7180fd65e622134fbd980c9a3d4211fb85224b56a0a0efbd422ab72f", size = 188400296, upload-time = "2026-01-20T16:00:56.042Z" }, ]