Varshith19's picture
Update logbook: Reproduction: How much can language models memorize?
8673629 verified
Raw History Blame Contribute Delete
4.39 kB
{
"schema_version": 2,
"title": "Reproduction: How much can language models memorize?",
"emoji": "🎯",
"space_id": "Varshith19/repro-how-much-can-language-models-memorize",
"paper": null,
"tags": [
"icml2026-repro",
"paper-bA6BgSbaUi"
],
"updated_at": "2026-07-22T12:49:11+00:00",
"root": {
"slug": "index",
"title": "Reproduction: How much can language models memorize?",
"file": "pages/index.md",
"children": [
{
"slug": "executive-summary",
"title": "Executive summary",
"file": "pages/executive-summary/page.md",
"children": []
},
{
"slug": "claim-1-gpt-style-transformers-trained-on-uniform-random-data-show-an-empirical-memorization-capacity-plateau-of-about-3-6-bits-per-parameter-figure-1",
"title": "Claim 1: GPT-style transformers trained on uniform random data show an empirical memorization-capacity plateau of about 3.6 bits per parameter (Figure 1)",
"file": "pages/claim-1-gpt-style-transformers-trained-on-uniform-random-data-show-an-empirical-memorization-capacity-plateau-of-about-3-6-bits-per-parameter-figure-1/page.md",
"children": []
},
{
"slug": "claim-2-capacity-estimates-across-model-widths-and-depths-support-a-roughly-linear-bits-per-parameter-scaling-law-with-bfloat16-to-float32-increasing-capacity-only-modestly-table-1",
"title": "Claim 2: Capacity estimates across model widths and depths support a roughly linear bits-per-parameter scaling law, with bfloat16 to float32 increasing capacity only modestly (Table 1)",
"file": "pages/claim-2-capacity-estimates-across-model-widths-and-depths-support-a-roughly-linear-bits-per-parameter-scaling-law-with-bfloat16-to-float32-increasing-capacity-only-modestly-table-1/page.md",
"children": []
},
{
"slug": "claim-3-on-text-data-unintended-memorization-rises-with-model-size-but-decreases-once-models-begin-generalizing-relative-to-an-oracle-reference-model-figure-2",
"title": "Claim 3: On text data, unintended memorization rises with model size but decreases once models begin generalizing relative to an oracle reference model (Figure 2)",
"file": "pages/claim-3-on-text-data-unintended-memorization-rises-with-model-size-but-decreases-once-models-begin-generalizing-relative-to-an-oracle-reference-model-figure-2/page.md",
"children": []
},
{
"slug": "claim-4-double-descent-begins-when-dataset-information-content-exceeds-estimated-model-capacity-in-both-synthetic-bitstrings-and-text-experiments-figures-3-and-4",
"title": "Claim 4: Double descent begins when dataset information content exceeds estimated model capacity in both synthetic bitstrings and text experiments (Figures 3 and 4)",
"file": "pages/claim-4-double-descent-begins-when-dataset-information-content-exceeds-estimated-model-capacity-in-both-synthetic-bitstrings-and-text-experiments-figures-3-and-4/page.md",
"children": []
},
{
"slug": "claim-5-the-paper-derives-and-evaluates-scaling-law-predictions-for-membership-inference-as-a-function-of-model-capacity-and-dataset-size-figure-7",
"title": "Claim 5: The paper derives and evaluates scaling-law predictions for membership inference as a function of model capacity and dataset size (Figure 7)",
"file": "pages/claim-5-the-paper-derives-and-evaluates-scaling-law-predictions-for-membership-inference-as-a-function-of-model-capacity-and-dataset-size-figure-7/page.md",
"children": []
},
{
"slug": "conclusion",
"title": "Conclusion",
"file": "pages/conclusion/page.md",
"children": []
}
]
},
"traces": [],
"workspace": {
"file": "workspace.json",
"file_count": 0,
"total_size": 0,
"bucket_id": null
},
"agent_view_tokens": 5591,
"trace_view_tokens": 49,
"workspace_view_tokens": 8,
"revision": "bec1d8cc4213c1dd707a",
"traces_ref": {
"repo_id": "Varshith19/repro-how-much-can-language-models-memorize-traces",
"repo_type": "dataset",
"repo_url": "https://huggingface.co/datasets/Varshith19/repro-how-much-can-language-models-memorize-traces",
"private": true
},
"trace_dataset": "https://huggingface.co/datasets/Varshith19/repro-how-much-can-language-models-memorize-traces"
}