E01	100	knowledge/status/drbench_judge_four_negatives.log	published DRBench tasks scored
E02	254	knowledge/status/drbench_judge_four_negatives.log	lexical containment, mean insight-recall permille at matched precision
E03	306	knowledge/status/drbench_judge_four_negatives.log	sparse PPMI late-interaction, mean insight-recall permille at matched precision
E04	284	knowledge/status/drbench_judge_four_negatives.log	domain-augmented PPMI model, mean insight-recall permille
E05	307	knowledge/status/drbench_judge_four_negatives.log	IDF-weighted semantic, mean insight-recall permille
E06	133	knowledge/status/drbench_judge_four_negatives.log	trained dense embed_v1, mean insight-recall permille
E07	145	knowledge/status/drbench_judge_four_negatives.log	dense after common-component removal, mean insight-recall permille
E08	316	knowledge/status/drbench_judge_four_negatives.log	learned linear fusion, held-out insight-recall permille
E09	318	knowledge/status/drbench_judge_four_negatives.log	best single signal held-out (semantic-IDF), insight-recall permille
E10	650	knowledge/status/drbench_judge_four_negatives.log	nx_dr_fuse train_pairwise_acc permille, fitted directly on training data
E11	2003	knowledge/status/drbench_judge_four_negatives.log	labelled candidate answers across the task set
E12	833	knowledge/status/drbench_judge_four_negatives.log	insight-term vocabulary coverage permille, general model
E13	1000	knowledge/status/drbench_judge_four_negatives.log	insight-term vocabulary coverage permille, domain model
E14	15	knowledge/status/drbench_judge_four_negatives.log	isolated probe score, general model
E15	235	knowledge/status/drbench_judge_four_negatives.log	isolated probe score, domain model, same insight
E16	48	knowledge/status/drbench_judge_four_negatives.log	tasks where sparse beat trained dense
E17	12	knowledge/status/drbench_judge_four_negatives.log	tasks where trained dense beat sparse
E18	10	knowledge/status/drbench_judge_four_negatives.log	tasks where common-component-removed dense beat sparse
E19	46	knowledge/status/drbench_judge_four_negatives.log	tasks where sparse beat common-component-removed dense
E20	908	knowledge/status/drbench_judge_four_negatives.log	domain corpus size in kilobytes
E21	21	knowledge/status/drbench_judge_four_negatives.log	tasks where IDF weighting was better
E22	58	knowledge/status/drbench_judge_four_negatives.log	tasks where IDF weighting tied
E23	32	knowledge/status/drbench_scale_judge_measurement.log	tasks where semantic beat lexical
E24	17	knowledge/status/drbench_scale_judge_measurement.log	tasks where lexical beat semantic
E25	51	knowledge/status/drbench_scale_judge_measurement.log	tasks where lexical and semantic tied
E26	24	knowledge/status/drbench_judge_four_negatives.log	embed_v1 embedding dimensionality
E27	102318	knowledge/status/drbench_judge_four_negatives.log	PPMI model vocabulary size in words
E28	70	knowledge/status/drbench_judge_four_negatives.log	tasks in the fusion training split
E29	30	knowledge/status/drbench_judge_four_negatives.log	tasks in the disjoint held-out split
E30	634	knowledge/status/drbench_judge_four_negatives.log	dense probe score for the on-topic term
E31	549	knowledge/status/drbench_judge_four_negatives.log	dense probe score for the off-topic term
E32	123	knowledge/status/drbench_judge_four_negatives.log	sparse probe score for the on-topic term
E33	25	knowledge/status/drbench_judge_four_negatives.log	sparse probe score for the off-topic term
E34	547	knowledge/status/drbench_judge_four_negatives.log	centered dense probe score, on-topic term
E35	400	knowledge/status/drbench_judge_four_negatives.log	centered dense probe score, off-topic term
E36	5	knowledge/status/drbench_judge_four_negatives.log	number of hypotheses tested and rejected
