Transformer

This commit is contained in:
Efim Beshmenev
2026-07-12 21:48:44 +03:00
parent ffd46f89e3
commit 9e3dd7ce8b
11614 changed files with 16818 additions and 4458 deletions
@@ -0,0 +1,58 @@
format szilassi-global-search-v5
run_id 90000b96-af3c-4d90-8233-0ba9c155673b
node_id LOOKICH
seed 1160878469
topology_from 0
topology_to 58
backend cuda-fp32
objective_version 5
cuda_math standard-fp32
cuda_chains_requested 0
cuda_chains_effective 61440
cuda_iterations_per_batch 64
cuda_depth_batches 6
cuda_session_cache 59
algorithm hybrid-quality-diversity-neural-v3-transformer-ranker-v1
control_baseline_min_fraction 0.25
strategy_weights adaptive-total:16,floors:baseline4/replica1/adaptive1/pbt1/injected3,max6
fresh_fractions depth:1/4,breadth:7/8,injected-protected
replica_exchange group:8,temperature-ratio:16
pbt rotating-pairs,depth-chance:0.08,breadth-chance:0.04
verification_quotas overall:128,per-strategy:max(8,overall/5),per-injected-seed:1
fp32_population_sample per-producing-strategy:64,finite-chain-bests-before-host-selection,deterministic-circular-offset,exact-numerator-denominator-recorded
accounted_fp32_steps iterations-plus-initialization-fresh-stagnation-injection-pbt-and-retries
map_elites_bins determinant:8,edge:8,turn:8,extent:8,crossing-face-mask:12,intersection-face-mask:12
map_elites_max_cells_per_topology 4096
cem diagonal-weighted,covariance-adaptation:0.18,new-injected-only
intersection_loss segment-depth-times-boundary-depth,weight:0.01,cap:4
repair injected-quarter,offending-face-biased,neural-policy-with-exploration
neural_enabled 1
neural_schema 2
neural_model_format 2
neural per-topology-ensemble:5,residual-mlp:45-128-128-128,value-policy,online-adamw,replay:4096,recent:25%
neural_safety baseline-floor:25%,policy-exploration:25%,exact-cuda-plus-cpu-dd-authoritative
transformer_enabled 1
transformer_model_id 568ec2d600768cc5d2189f2c45ef6218689e8537937b51693eeb461b298d9f45-s1511499089-p56c66b83
transformer global-ensemble:3,set-transformer:12-faces-plus-cls,width:64,heads:4,layers:3,ff:256,host-fp32-ranker
transformer_scope guided-seeds-only,ranked:12/64,score-independent-controls:4/64,map-cem-random:48/64
transformer_failure_mode missing-invalid-nonfinite:fallback-to-online-mlp
training_archive_schema 3
training_archive_format 3
training_archive_layout run-uuid/immutable-64MiB-crc-shards-plus-durable-wal
training_archive_records seed-proposals,stratified-fp32-chain-bests,cpu-verified,anchor-injected-result,spsa-steps,legacy-replay
training_cache_limit_bytes 200000000000
training_cache_accounted_bytes_at_start 7175636440
training_neural_model_reserve_bytes 265316868
training_collection_enabled_at_start 1
training_recovered_wals 0
training_wals_deferred_at_limit 0
training_recovered_records 0
training_recovered_torn_bytes_discarded 0
training_limit_behavior freeze-collection-and-online-learning;search-continues;rewrite-warning
spsa adam:6,cpu-double,all-plus-minus-and-updated-double-states,c-plus-i-smooth-loss,fp32-roundtrip,final-canonical-dd-gate
topology_scheduler ucb-plus-reward-plus-staleness,full-refresh-every-5
topology_scheduler_mode quality-first
worst_priority_coefficient 0.65
cpu_iterations_per_trial 50000
degeneracy_formula worst-plus-0.05-rest
degeneracy_weight 0.01
1 format szilassi-global-search-v5
2 run_id 90000b96-af3c-4d90-8233-0ba9c155673b
3 node_id LOOKICH
4 seed 1160878469
5 topology_from 0
6 topology_to 58
7 backend cuda-fp32
8 objective_version 5
9 cuda_math standard-fp32
10 cuda_chains_requested 0
11 cuda_chains_effective 61440
12 cuda_iterations_per_batch 64
13 cuda_depth_batches 6
14 cuda_session_cache 59
15 algorithm hybrid-quality-diversity-neural-v3-transformer-ranker-v1
16 control_baseline_min_fraction 0.25
17 strategy_weights adaptive-total:16,floors:baseline4/replica1/adaptive1/pbt1/injected3,max6
18 fresh_fractions depth:1/4,breadth:7/8,injected-protected
19 replica_exchange group:8,temperature-ratio:16
20 pbt rotating-pairs,depth-chance:0.08,breadth-chance:0.04
21 verification_quotas overall:128,per-strategy:max(8,overall/5),per-injected-seed:1
22 fp32_population_sample per-producing-strategy:64,finite-chain-bests-before-host-selection,deterministic-circular-offset,exact-numerator-denominator-recorded
23 accounted_fp32_steps iterations-plus-initialization-fresh-stagnation-injection-pbt-and-retries
24 map_elites_bins determinant:8,edge:8,turn:8,extent:8,crossing-face-mask:12,intersection-face-mask:12
25 map_elites_max_cells_per_topology 4096
26 cem diagonal-weighted,covariance-adaptation:0.18,new-injected-only
27 intersection_loss segment-depth-times-boundary-depth,weight:0.01,cap:4
28 repair injected-quarter,offending-face-biased,neural-policy-with-exploration
29 neural_enabled 1
30 neural_schema 2
31 neural_model_format 2
32 neural per-topology-ensemble:5,residual-mlp:45-128-128-128,value-policy,online-adamw,replay:4096,recent:25%
33 neural_safety baseline-floor:25%,policy-exploration:25%,exact-cuda-plus-cpu-dd-authoritative
34 transformer_enabled 1
35 transformer_model_id 568ec2d600768cc5d2189f2c45ef6218689e8537937b51693eeb461b298d9f45-s1511499089-p56c66b83
36 transformer global-ensemble:3,set-transformer:12-faces-plus-cls,width:64,heads:4,layers:3,ff:256,host-fp32-ranker
37 transformer_scope guided-seeds-only,ranked:12/64,score-independent-controls:4/64,map-cem-random:48/64
38 transformer_failure_mode missing-invalid-nonfinite:fallback-to-online-mlp
39 training_archive_schema 3
40 training_archive_format 3
41 training_archive_layout run-uuid/immutable-64MiB-crc-shards-plus-durable-wal
42 training_archive_records seed-proposals,stratified-fp32-chain-bests,cpu-verified,anchor-injected-result,spsa-steps,legacy-replay
43 training_cache_limit_bytes 200000000000
44 training_cache_accounted_bytes_at_start 7175636440
45 training_neural_model_reserve_bytes 265316868
46 training_collection_enabled_at_start 1
47 training_recovered_wals 0
48 training_wals_deferred_at_limit 0
49 training_recovered_records 0
50 training_recovered_torn_bytes_discarded 0
51 training_limit_behavior freeze-collection-and-online-learning;search-continues;rewrite-warning
52 spsa adam:6,cpu-double,all-plus-minus-and-updated-double-states,c-plus-i-smooth-loss,fp32-roundtrip,final-canonical-dd-gate
53 topology_scheduler ucb-plus-reward-plus-staleness,full-refresh-every-5
54 topology_scheduler_mode quality-first
55 worst_priority_coefficient 0.65
56 cpu_iterations_per_trial 50000
57 degeneracy_formula worst-plus-0.05-rest
58 degeneracy_weight 0.01