{"n_units_with_improvement": 307388, "n_sampled": 80000, "groups": [{"name": "Extend the evidence \u2014 more datasets, domains, models, scale", "share": 0.2215, "exemplar": "Expand evaluation to additional datasets or tasks to demonstrate robustness and scalability.", "top_objects": [["empirical_scope", 0.349], ["method_design", 0.216], ["stats_metrics", 0.096]]}, {"name": "Compare against stronger baselines", "share": 0.0859, "exemplar": "Compare against recent state-of-the-art methods.", "top_objects": [["baselines_ablations", 0.426], ["empirical_scope", 0.168], ["method_design", 0.094]]}, {"name": "Fix the surface \u2014 typos, notation, figures", "share": 0.0848, "exemplar": "Correct typos, add missing notation, and clarify figure captions.", "top_objects": [["clarity", 0.473], ["theory", 0.202], ["stats_metrics", 0.072]]}, {"name": "Measure the true cost \u2014 runtime, memory, GPU-hours", "share": 0.0704, "exemplar": "Report inference-time savings and runtime measurements to demonstrate actual speed improvements.", "top_objects": [["compute_cost", 0.543], ["empirical_scope", 0.131], ["method_design", 0.093]]}, {"name": "Strengthen the theory \u2014 proofs, bounds, guarantees", "share": 0.0686, "exemplar": "Provide theoretical results on learnability or optimizability.", "top_objects": [["theory", 0.5], ["empirical_scope", 0.145], ["method_design", 0.099]]}, {"name": "Justify assumptions & scope the claims", "share": 0.0662, "exemplar": "Justify the assumption or acknowledge its limitations.", "top_objects": [["theory", 0.29], ["method_design", 0.166], ["empirical_scope", 0.154]]}, {"name": "Articulate the contribution", "share": 0.0655, "exemplar": "Articulate a clearer principle behind the method's design.", "top_objects": [["method_design", 0.225], ["theory", 0.161], ["empirical_scope", 0.156]]}, {"name": "Release & document \u2014 code, data, repro details", "share": 0.0505, "exemplar": "Clarify reproducibility details, code release plans, and LLM usage.", "top_objects": [["empirical_scope", 0.229], ["reproducibility", 0.193], ["method_design", 0.149]]}, {"name": "Situate in the literature", "share": 0.0444, "exemplar": "Deepen engagement with specific prior works and clarify differentiation.", "top_objects": [["clarity", 0.253], ["related_work", 0.249], ["empirical_scope", 0.088]]}, {"name": "Validate against humans & the real world", "share": 0.0423, "exemplar": "Conduct a validation study comparing model outputs to human judgments.", "top_objects": [["stats_metrics", 0.333], ["empirical_scope", 0.257], ["method_design", 0.146]]}, {"name": "Show it on real visual data", "share": 0.0406, "exemplar": "Investigate more tasks with real-world images to provide more convincing evidence.", "top_objects": [["empirical_scope", 0.443], ["clarity", 0.133], ["method_design", 0.109]]}, {"name": "Report error bars & significance", "share": 0.0386, "exemplar": "Report mean \u00b1 std or confidence intervals across multiple runs.", "top_objects": [["stats_metrics", 0.392], ["empirical_scope", 0.215], ["robustness_sensitivity", 0.08]]}, {"name": "Ablate the components", "share": 0.0378, "exemplar": "Isolate components through ablation studies to demonstrate their individual contributions.", "top_objects": [["baselines_ablations", 0.304], ["empirical_scope", 0.173], ["method_design", 0.169]]}, {"name": "Stress-test under attack & shift", "share": 0.033, "exemplar": "Demonstrate robustness across different model scales and under adversarial attack.", "top_objects": [["empirical_scope", 0.384], ["robustness_sensitivity", 0.163], ["theory", 0.131]]}, {"name": "Analyze the failures", "share": 0.03, "exemplar": "Identify and report specific failure modes or conditions where the approach does not work well.", "top_objects": [["empirical_scope", 0.189], ["stats_metrics", 0.16], ["method_design", 0.153]]}, {"name": "Probe hyperparameter sensitivity", "share": 0.0198, "exemplar": "Conduct a sensitivity analysis to demonstrate robustness to hyperparameter variations.", "top_objects": [["robustness_sensitivity", 0.535], ["empirical_scope", 0.134], ["theory", 0.118]]}]}
