{ "seed": 85, "query_selection": "first100 sorted by SHA256 UTF8(85:query_id), independent of labels and scores", "query_ids": [ "881", "1125", "1309", "201", "1193", "47", "652", "456", "154", "537", "1224", "308", "1145", "9", "761", "530", "963", "1310", "325", "1306", "925", "557", "665", "30", "448", "774", "90", "813", "1109", "1120", "973", "755", "828", "1302", "182", "703", "1249", "1053", "371", "572", "1005", "1027", "280", "68", "858", "368", "481", "1138", "74", "258", "1050", "1166", "361", "900", "815", "365", "1406", "1263", "776", "780", "309", "622", "1034", "203", "104", "764", "283", "542", "304", "1276", "1361", "1341", "242", "155", "32", "658", "1307", "616", "469", "591", "1165", "95", "1136", "160", "717", "638", "64", "990", "139", "817", "1401", "978", "156", "779", "835", "687", "667", "671", "461", "454" ], "split": "BEIR train development subset; original SciFact train", "corpus_documents": 5183, "top_k": 50, "batch_size": 16, "max_length": 512, "threads": 4, "device": "cpu", "dtype": "float32", "activation": "Identity raw logits", "text": "(title + single space + text).strip()", "prompt": "none", "decode": "not applicable, scalar scoring", "primary_metric": "ndcg_cut_10", "tie_break": "score DESC, document ID string DESC", "confirmation": "300 BEIR test queries not evaluated", "resource_rule": "smoke first5 selected queries top10; full100x50 if extrapolated inference<900s and peakRSS<4GiB; otherwise preserve incomplete, no score-driven shrink" }