{ "artifact_provenance": { "corpus_manifest_sha256": "d065f2065712e9c83d73d3aa58e7453205e4ccbea8dcc232b71a57d54dd17d97", "corpus_offsets_sha256": "c0fef973b028cb4927e64182d07328d320aa9984357704051e2a1c8705843e0d", "corpus_source_sha256": "25bd27eec0d21e0ec3e33a0467c91f9f3d1d6f78adc25369720cc0289e191bf3", "corpus_tokens_sha256": "e5d4fc1f685dbec12ce458e257e4d9bfb39fc084cbbcedc5963229800ee8ac52", "corpus_words": 8900000, "corpus_words_sha256": "4036ddbe4921b35fb732d9cffb46f968c58ed8463def22bb846a35f5250790c3", "document_order_sha256": null, "factor_manifest_sha256": null, "factorization_manifest_sha256": "52f31760faac4620008f19dbb6533b4b1856831f7e2b94fdb178b3b06803605a", "influence_preflight_sha256": null, "repositories": { "encoded_corpus": { "repo": "miguelcsx/babylm-bbpe16k-512-encoded", "repo_type": "dataset", "revision": "a62a6f274f3d86c05c8dd967a0c7200573ae9ef6" }, "encoded_factorized_real": { "repo": "miguelcsx/factorized-strict-small-real-corpus", "repo_type": "dataset", "revision": "b86f852224ea8f3c150f395209e447f5223c7aa0" }, "factorization_priors": { "repo": "miguelcsx/factorized-strict-small-priors", "repo_type": "dataset", "revision": "3ea2f0d08ce87f20ce8986a4a08374ca8cc4f23f" }, "tokenizer": { "repo": "miguelcsx/causal-focus-bbpe16k-tokenizer", "repo_type": "model", "revision": "5c95ae8fe9cc0f745f8d643192775fbceff5fedb" }, "tokenizer_factorized": { "repo": "miguelcsx/factorized-strict-small-bpe16k-tokenizer", "repo_type": "model", "revision": "a1f87b2b439ea988ba2b94b62898ed44780fb5e5" } }, "structured_priors_manifest_sha256": "ee75fe93d2dffd4bc7d86d1a1fd9f1aa1d76a956012dc6357274a252f40d8bbf", "tokenizer_sha256": "2b8d1b3f51c0d8f276a64ea6a69efa50b4d9899780f1b282b4d0dcb7be5bbb0f", "tokenizer_word_start_sha256": "42d9a905c49df4c54dbe3bc93e567c9bbe491ec6fbee5fcbed70e8d1ad533d49" }, "compliance": { "auxiliary_linguistic_training_words": 2000000, "brown_training_words": 1000000, "competition_status": "strict_small_conservative_accounting", "conservative_corpus_accounting_words": 10000000, "conservative_counted_words": 100000000, "corpus_induced_priors": true, "corpus_limit_words": 10000000, "evaluation_holdout_words": 1000000, "exposure_limit_words": 100000000, "external_teacher": false, "generated_words": 0, "induction_corpus_disjoint_from_lm_and_holdout": true, "leaderboard_checkpoint_words": 88000000, "model_exposure_words": 100000000, "model_training_corpus_words": 8900000, "ppmi_induction_words": 1000000, "relational_corpus_fraction": 0.025, "same_run_checkpoint_distribution": false, "signals_derived_from_current_model_input": false, "strict_small_status": "corpus_10m_model_views_100m", "syntax_training_words": 0, "teacher_queries": 0, "tokenizer_training_words": 8900000, "total_exposure_words": 100000000, "track": "strict-small", "unique_corpus_words": 10000000, "within_100m_conservative_budget": true, "within_track_corpus_budget": true, "within_track_exposure_budget": true }, "factor_priors": { "angular_margin": 0.1, "dual": { "enabled": false, "learning_rate": 0.001, "max_weight": 1.0, "targets": [ 1.0, 4.0, 0.1 ] }, "enabled": false, "lambda_conceptual": 0.0, "lambda_lexical": 0.0, "lambda_syntax": 0.0, "max_positions": 256, "radial_margin": 0.02, "radial_weight": 1.0, "randomize": false, "randomize_seed": 42, "syntax_depth_weight": 0.1, "syntax_distance_weight": 0.1, "syntax_window": 32, "temperature": 0.07, "warmup_words": 1000000 }, "factorization": { "context_window": 4, "covariance_weight": 0.005, "cross_covariance_weight": 0.002, "enabled": true, "gradient_budget": 0.1, "gradient_ema_momentum": 0.95, "graph_mode": "real", "loss_enabled": true, "max_gradient_scale": 1000.0, "null_seed": 271828, "pairs_per_microbatch": 128, "ramp_words": 5000000, "regularizer_words": 512, "signal_mode": "induced_priors", "temperature": 0.07, "variance_weight": 0.02 }, "geometry": { "curvature": 1.0, "distance_margin": 0.1, "enabled": false, "lambda_radial": 0.02, "lambda_related": 0.02, "max_tokens": 256, "radial_margin": 0.05, "warmup_words": 5000000 }, "model": { "absolute_positions": false, "attention_dropout": 0.1, "bos_token_id": 1, "cognitive_readout_layer": 0, "cognitive_readout_weight": 0.0, "direct_sum_dims": [], "direct_sum_heads": [], "direct_sum_intermediate_sizes": [], "dropout": 0.1, "eos_token_id": 2, "expert_intermediate_size": null, "experts_per_token": 1, "factor_readout_dims": [ 128, 128, 128 ], "factor_readout_mode": "heads", "factor_readout_reflectors": 384, "factor_readout_seed": 314159, "future_offsets": [], "geometry_curvature": 1.0, "geometry_lexical_dim": 0, "hidden_size": 384, "initializer_range": 0.03227486121839514, "intermediate_size": 1280, "lexical_residual_buckets": 0, "lexical_residual_dim": 0, "lexical_residual_scale": 1.0, "mask_token_id": 4, "max_seq_len": 512, "num_attention_heads": 6, "num_experts": 1, "num_hidden_layers": 12, "pad_token_id": 3, "position_buckets": 32, "recurrent_steps": 1, "residual_mixing": true, "rtd_auxiliary": true, "state_mixer_kernel": 0, "structured_projection_dim": 0, "use_alibi": false, "use_rope": false, "value_gating": true, "vocab_size": 16384 }, "release": "TOLM", "repository": { "commit": "748fa8a55539a43dd27bcf08b42abfa009012223", "tracked_dirty": false }, "structured_priors": { "decay_end_words": 7000000, "enabled": false, "hold_until_words": 2000000, "init_scale": 0.0, "lambda_lexical": 0.0, "lambda_orth": 0.0, "lambda_syntax": 0.0, "lexical_dim": 128, "max_positions": 256, "prior_mode": "contrastive", "randomize": false, "representation": "slices", "require_aligned_artifacts": false, "syntax_dim": 128, "temperature": 0.07, "warmup_words": 100000 }, "training": { "adaptive_masking": { "enabled": false, "max_mask_prob": 4.0, "min_mask_prob": 0.25, "momentum": 0.99 }, "batch_size": 16, "beta1": 0.9, "beta2": 0.98, "causal_fraction": 0.0, "causal_noise_kind": "random", "causal_noise_probability": 0.0, "causal_unit": "segment", "cooldown_fraction": 0.016, "data2vec_layers": 4, "data2vec_weight": 0.0, "device": "xpu:4", "document_curriculum": { "enabled": false }, "ema_decay": 0.9998, "epsilon": 1e-08, "exact_word_checkpoints": true, "exposure_words": 100000000, "final_lr_ratio": 0.1, "frequency_aware_masking": { "enabled": false, "max_mask_prob": 4.0, "min_mask_prob": 0.25, "temperature": 1.0 }, "future_loss_weight": 0.0, "gradient_clip": 2.0, "label_smoothing": 0.0, "learning_progress": { "buckets_per_axis": 4, "enabled": false, "fast_momentum": 0.9, "forgetting_weight": 1.0, "slow_momentum": 0.99, "uniform_floor": 0.3, "window_documents": 4000 }, "learning_rate": 0.0035, "learning_rate_schedule": "cosine", "log_interval": 50, "mask_probability_end": 0.5, "mask_probability_schedule": "uniform", "mask_probability_start": 0.15, "mask_replace_probability": 0.8, "mask_schedule": "complementary", "masked_fraction": 1.0, "masking_unit": "word", "microbatch_tokens": 8192, "mixed_precision": "bf16", "num_workers": 0, "objective_period": 16, "objective_rng_reset_words": [], "objective_sanity_window_steps": 512, "objective_schedule": "coverage", "optimizer": "lamb", "packing_strategy": "dense", "pin_memory": false, "random_replace_probability": 0.1, "recombine_probability": 0.0, "recombine_strategy": "random", "recovery_interval_words": 1000000, "resource_memory": { "enabled": false }, "router_aux_weight": 0.0, "save_steps_words": [ 1000000, 2000000, 3000000, 4000000, 5000000, 6000000, 7000000, 8000000, 9000000, 10000000, 20000000, 30000000, 40000000, 50000000, 60000000, 70000000, 80000000, 88000000, 90000000, 100000000 ], "schedule_total_words": 100000000, "span_max_length": 3, "stable_fraction": 0.9, "telemetry": { "enabled": true, "gradient_checkpoints_words": [ 1000000, 10000000, 40000000, 70000000, 88000000 ], "sampler_trace": true }, "threads_per_process": 12, "tokens_per_update": 16384, "warmup_fraction": 0.016, "weight_decay": 0.1, "z_loss_weight": 0.0001 }, "variant": "factorized_heads", "words_seen": 100000000 }