{"target": "paper decision (accept vs reject), tribunal-level", "decision_rule": "1 if 'accept' in decision.lower() else 0; decision NOT NULL, withdrawn=0, desk_rejected=0", "years_with_decisions": [2018, 2020, 2021, 2022, 2023, 2024, 2025, 2026], "workshop_invite_quirk_n": 90, "n_papers": 39484, "base_rate": 0.3945, "seed": 7, "n_folds": 5, "objects": ["baselines_ablations", "clarity", "compute_cost", "empirical_scope", "method_design", "novelty", "problem_framing", "related_work", "reproducibility", "robustness_sensitivity", "stats_metrics", "theory"], "aucs": {"tally": 0.6453, "tally_nrev": 0.6518, "text": 0.7403, "nrev_only": 0.5064, "rating_ref": 0.8955}, "auc_folds": {"tally": [0.6331, 0.6532, 0.6479, 0.6467, 0.6455], "tally_nrev": [0.6405, 0.6601, 0.6545, 0.6509, 0.6529], "text": [0.7371, 0.7484, 0.7392, 0.7407, 0.7361], "nrev_only": [0.5094, 0.5059, 0.508, 0.4962, 0.5124], "rating_ref": [0.896, 0.8931, 0.8974, 0.8951, 0.8957]}, "per_year_auc": {"tally": {"2018": 0.7453, "2020": 0.727, "2021": 0.7012, "2022": 0.7046, "2023": 0.6748, "2024": 0.6551, "2025": 0.6233, "2026": 0.6347}, "tally_nrev": {"2018": 0.7439, "2020": 0.7267, "2021": 0.7011, "2022": 0.7038, "2023": 0.6802, "2024": 0.659, "2025": 0.6256, "2026": 0.6396}, "text": {"2018": 0.7261, "2020": 0.7665, "2021": 0.7503, "2022": 0.7404, "2023": 0.7465, "2024": 0.7361, "2025": 0.7179, "2026": 0.7051}, "rating_ref": {"2018": 0.9612, "2020": 0.9579, "2021": 0.9544, "2022": 0.955, "2023": 0.9629, "2024": 0.9414, "2025": 0.944, "2026": 0.8904}}, "n_rating_ref": 39484, "n_text_zero_vector": 48, "tally_coef": {"objects_sorted_ascending": ["novelty", "problem_framing", "baselines_ablations", "related_work", "stats_metrics", "clarity", "theory", "empirical_scope", "method_design", "reproducibility", "compute_cost", "robustness_sensitivity"], "coef": {"baselines_ablations": -0.173, "clarity": -0.077, "compute_cost": 0.0058, "empirical_scope": -0.0453, "method_design": -0.0379, "novelty": -0.3624, "problem_framing": -0.2245, "related_work": -0.1447, "reproducibility": -0.0353, "robustness_sensitivity": 0.0366, "stats_metrics": -0.0912, "theory": -0.0613}, "coef_fold_std": {"baselines_ablations": 0.0055, "clarity": 0.0043, "compute_cost": 0.005, "empirical_scope": 0.0055, "method_design": 0.0065, "novelty": 0.0106, "problem_framing": 0.0068, "related_work": 0.0064, "reproducibility": 0.0144, "robustness_sensitivity": 0.0054, "stats_metrics": 0.0065, "theory": 0.0008}}, "notes": "Criticism units = official_reviewer, temporal_position=initial_review only (pre-rebuttal, pre-meta-review). rating_ref pools raw scores across 4 scoring-form eras (2018-2026) -- see per_year_auc.rating_ref for the honest per-era version. 2019 has no decision data in this corpus and is excluded entirely, not by choice.", "wall_clock_seconds": 101.8}
