{"kpi": {"units": 410586, "reviews": 74380, "papers": 19474}, "taxonomy": {"version": "v1", "created_from": "clusters-inspected_object.json / clusters-reasoning.json (12k sample, seed 7)", "notes": ["Categories were induced by human reading of HDBSCAN cluster exemplars.", "Excluded from centroids: inspected_object cluster 13 (heterogeneous), reasoning cluster 10 (meta/mixed norm phrasing).", "Assignment = nearest category centroid (mean of member-cluster embeddings), cosine similarity stored per unit."], "inspected_object": [{"key": "empirical_scope", "label_ja": "実験の範囲・一般化可能性", "label_en": "Empirical scope & generalizability", "clusters": [20, 12, 23, 33, 32, 27, 19, 1], "definition": "Breadth and representativeness of empirical evaluation: dataset scale, benchmark coverage, generalization to other models/settings, robustness scope, attack/threat coverage."}, {"key": "baselines_ablations", "label_ja": "ベースライン比較・アブレーション", "label_en": "Baselines & ablations", "clusters": [24, 5, 0], "definition": "Choice, fairness, and completeness of baseline comparisons and ablation studies."}, {"key": "theory", "label_ja": "理論・仮定・形式的保証", "label_en": "Theory, assumptions & guarantees", "clusters": [22, 21, 4], "definition": "Theoretical assumptions, formal guarantees, proofs, and the loss/objective's formal properties."}, {"key": "method_design", "label_ja": "手法設計の正当性", "label_en": "Method design rationale", "clusters": [2, 15, 11, 30, 14, 18], "definition": "Justification of design choices: architecture, reward/objective design, training procedure, mechanism claims."}, {"key": "compute_cost", "label_ja": "計算コスト・スケーラビリティ", "label_en": "Compute cost & scalability", "clusters": [17, 7], "definition": "Training/inference computational cost, latency, scalability analysis."}, {"key": "clarity", "label_ja": "明瞭性・記述品質", "label_en": "Clarity & presentation", "clusters": [35, 31, 25, 34], "definition": "Writing quality, figures, notation, exposition and structure."}, {"key": "novelty", "label_ja": "新規性・貢献の位置づけ", "label_en": "Novelty & contribution", "clusters": [29], "definition": "Novelty of the contribution and its articulation relative to existing methods."}, {"key": "related_work", "label_ja": "関連研究・引用", "label_en": "Related work & citations", "clusters": [16], "definition": "Coverage and accuracy of citations and positioning within prior literature."}, {"key": "stats_metrics", "label_ja": "統計的厳密性・評価指標", "label_en": "Statistical rigor & metrics", "clusters": [28, 9, 3], "definition": "Statistical reporting, significance, choice and validity of evaluation metrics including LLM-as-judge setups."}, {"key": "robustness_sensitivity", "label_ja": "頑健性・感度", "label_en": "Robustness & sensitivity", "clusters": [6, 8], "definition": "Hyperparameter sensitivity, noise robustness, failure-mode analysis."}, {"key": "reproducibility", "label_ja": "再現性・コード公開", "label_en": "Reproducibility & code", "clusters": [10], "definition": "Code availability and sufficiency of detail for independent replication."}, {"key": "problem_framing", "label_ja": "問題設定・動機", "label_en": "Problem framing & motivation", "clusters": [26], "definition": "Framing, motivation, and significance of the problem being addressed."}], "reasoning": [{"key": "novelty_standard", "label_ja": "新規性規範", "label_en": "Novelty standard", "clusters": [13], "definition": "Contribution must show conceptual/epistemic difference from prior art, not mere recombination."}, {"key": "claim_evidence_match", "label_ja": "主張-証拠スコープ照合", "label_en": "Claim-evidence scope matching", "clusters": [2], "definition": "Generality of claims must be matched by breadth of empirical demonstration."}, {"key": "fair_comparison", "label_ja": "公平比較規範", "label_en": "Fair comparison norm", "clusters": [3], "definition": "Baselines must be comparable in tuning, compute, and recency for the comparison to be meaningful."}, {"key": "statistical_identifiability", "label_ja": "統計的識別可能性", "label_en": "Statistical identifiability", "clusters": [14], "definition": "Without variance/significance reporting, signal cannot be distinguished from noise."}, {"key": "confound_hypothesis", "label_ja": "交絡・別因仮説の構成", "label_en": "Confound / alternative explanation", "clusters": [6, 0], "definition": "Results could be explained by leakage, memorization, weak protocols, or other artifacts."}, {"key": "design_justification", "label_ja": "設計正当化要求", "label_en": "Design justification demand", "clusters": [1, 7], "definition": "Design choices need mechanistic grounding or comparison against plausible alternatives."}, {"key": "robustness_norm", "label_ja": "頑健性規範", "label_en": "Robustness norm", "clusters": [5], "definition": "A practical method should be stable across hyperparameters and settings."}, {"key": "cost_benefit", "label_ja": "実用性のコスト便益", "label_en": "Practical cost-benefit", "clusters": [8, 9], "definition": "Gains must be weighed against computational/engineering cost for practical utility."}, {"key": "construct_validity", "label_ja": "測定の構成概念妥当性", "label_en": "Measurement construct validity", "clusters": [4], "definition": "Metrics must actually measure what they claim to measure."}, {"key": "reproducibility_norm", "label_ja": "再現可能性規範", "label_en": "Reproducibility norm", "clusters": [16], "definition": "Enough detail must be disclosed for independent re-implementation and verification."}, {"key": "presentation_trust", "label_ja": "提示品質=信頼シグナル", "label_en": "Presentation as trust signal", "clusters": [15], "definition": "Polish signals care; presentation flaws undermine trust in the work's rigor."}, {"key": "merit_recognition", "label_ja": "長所の承認", "label_en": "Merit recognition", "clusters": [12, 11], "definition": "Positive appraisal: breadth/soundness of evidence or contribution acknowledged as convincing."}]}, "triples": [{"o": "baselines_ablations", "r": "claim_evidence_match", "v": "conditional", "n": 112}, {"o": "baselines_ablations", "r": "claim_evidence_match", "v": "mixed", "n": 11}, {"o": "baselines_ablations", "r": "claim_evidence_match", "v": "negative", "n": 1352}, {"o": "baselines_ablations", "r": "claim_evidence_match", "v": "positive", "n": 74}, {"o": "baselines_ablations", "r": "claim_evidence_match", "v": "uncertain", "n": 33}, {"o": "baselines_ablations", "r": "confound_hypothesis", "v": "conditional", "n": 69}, {"o": "baselines_ablations", "r": "confound_hypothesis", "v": "negative", "n": 627}, {"o": "baselines_ablations", "r": "confound_hypothesis", "v": "positive", "n": 35}, {"o": "baselines_ablations", "r": "confound_hypothesis", "v": "uncertain", "n": 33}, {"o": "baselines_ablations", "r": "construct_validity", "v": "conditional", "n": 28}, {"o": "baselines_ablations", "r": "construct_validity", "v": "mixed", "n": 1}, {"o": "baselines_ablations", "r": "construct_validity", "v": "negative", "n": 378}, {"o": "baselines_ablations", "r": "construct_validity", "v": "positive", "n": 24}, {"o": "baselines_ablations", "r": "construct_validity", "v": "uncertain", "n": 12}, {"o": "baselines_ablations", "r": "cost_benefit", "v": "conditional", "n": 56}, {"o": "baselines_ablations", "r": "cost_benefit", "v": "mixed", "n": 1}, {"o": "baselines_ablations", "r": "cost_benefit", "v": "negative", "n": 548}, {"o": "baselines_ablations", "r": "cost_benefit", "v": "positive", "n": 30}, {"o": "baselines_ablations", "r": "cost_benefit", "v": "uncertain", "n": 20}, {"o": "baselines_ablations", "r": "design_justification", "v": "conditional", "n": 374}, {"o": "baselines_ablations", "r": "design_justification", "v": "mixed", "n": 12}, {"o": "baselines_ablations", "r": "design_justification", "v": "negative", "n": 5334}, {"o": "baselines_ablations", "r": "design_justification", "v": "positive", "n": 154}, {"o": "baselines_ablations", "r": "design_justification", "v": "uncertain", "n": 124}, {"o": "baselines_ablations", "r": "fair_comparison", "v": "conditional", "n": 576}, {"o": "baselines_ablations", "r": "fair_comparison", "v": "mixed", "n": 30}, {"o": "baselines_ablations", "r": "fair_comparison", "v": "negative", "n": 9199}, {"o": "baselines_ablations", "r": "fair_comparison", "v": "positive", "n": 192}, {"o": "baselines_ablations", "r": "fair_comparison", "v": "uncertain", "n": 191}, {"o": "baselines_ablations", "r": "merit_recognition", "v": "conditional", "n": 175}, {"o": "baselines_ablations", "r": "merit_recognition", "v": "mixed", "n": 28}, {"o": "baselines_ablations", "r": "merit_recognition", "v": "negative", "n": 1326}, {"o": "baselines_ablations", "r": "merit_recognition", "v": "positive", "n": 2872}, {"o": "baselines_ablations", "r": "merit_recognition", "v": "uncertain", "n": 47}, {"o": "baselines_ablations", "r": "novelty_standard", "v": "conditional", "n": 218}, {"o": "baselines_ablations", "r": "novelty_standard", "v": "mixed", "n": 34}, {"o": "baselines_ablations", "r": "novelty_standard", "v": "negative", "n": 3280}, {"o": "baselines_ablations", "r": "novelty_standard", "v": "positive", "n": 154}, {"o": "baselines_ablations", "r": "novelty_standard", "v": "uncertain", "n": 71}, {"o": "baselines_ablations", "r": "presentation_trust", "v": "conditional", "n": 39}, {"o": "baselines_ablations", "r": "presentation_trust", "v": "mixed", "n": 9}, {"o": "baselines_ablations", "r": "presentation_trust", "v": "negative", "n": 535}, {"o": "baselines_ablations", "r": "presentation_trust", "v": "positive", "n": 28}, {"o": "baselines_ablations", "r": "presentation_trust", "v": "uncertain", "n": 15}, {"o": "baselines_ablations", "r": "reproducibility_norm", "v": "conditional", "n": 17}, {"o": "baselines_ablations", "r": "reproducibility_norm", "v": "negative", "n": 359}, {"o": "baselines_ablations", "r": "reproducibility_norm", "v": "positive", "n": 16}, {"o": "baselines_ablations", "r": "reproducibility_norm", "v": "uncertain", "n": 10}, {"o": "baselines_ablations", "r": "robustness_norm", "v": "conditional", "n": 201}, {"o": "baselines_ablations", "r": "robustness_norm", "v": "mixed", "n": 12}, {"o": "baselines_ablations", "r": "robustness_norm", "v": "negative", "n": 1431}, {"o": "baselines_ablations", "r": "robustness_norm", "v": "positive", "n": 148}, {"o": "baselines_ablations", "r": "robustness_norm", "v": "uncertain", "n": 90}, {"o": "baselines_ablations", "r": "statistical_identifiability", "v": "conditional", "n": 95}, {"o": "baselines_ablations", "r": "statistical_identifiability", "v": "mixed", "n": 6}, {"o": "baselines_ablations", "r": "statistical_identifiability", "v": "negative", "n": 1838}, {"o": "baselines_ablations", "r": "statistical_identifiability", "v": "positive", "n": 27}, {"o": "baselines_ablations", "r": "statistical_identifiability", "v": "uncertain", "n": 26}, {"o": "clarity", "r": "claim_evidence_match", "v": "conditional", "n": 61}, {"o": "clarity", "r": "claim_evidence_match", "v": "mixed", "n": 5}, {"o": "clarity", "r": "claim_evidence_match", "v": "negative", "n": 487}, {"o": "clarity", "r": "claim_evidence_match", "v": "positive", "n": 27}, {"o": "clarity", "r": "claim_evidence_match", "v": "uncertain", "n": 30}, {"o": "clarity", "r": "confound_hypothesis", "v": "conditional", "n": 33}, {"o": "clarity", "r": "confound_hypothesis", "v": "mixed", "n": 3}, {"o": "clarity", "r": "confound_hypothesis", "v": "negative", "n": 288}, {"o": "clarity", "r": "confound_hypothesis", "v": "positive", "n": 24}, {"o": "clarity", "r": "confound_hypothesis", "v": "uncertain", "n": 35}, {"o": "clarity", "r": "construct_validity", "v": "conditional", "n": 41}, {"o": "clarity", "r": "construct_validity", "v": "mixed", "n": 1}, {"o": "clarity", "r": "construct_validity", "v": "negative", "n": 632}, {"o": "clarity", "r": "construct_validity", "v": "positive", "n": 21}, {"o": "clarity", "r": "construct_validity", "v": "uncertain", "n": 15}, {"o": "clarity", "r": "cost_benefit", "v": "conditional", "n": 17}, {"o": "clarity", "r": "cost_benefit", "v": "mixed", "n": 1}, {"o": "clarity", "r": "cost_benefit", "v": "negative", "n": 157}, {"o": "clarity", "r": "cost_benefit", "v": "positive", "n": 24}, {"o": "clarity", "r": "cost_benefit", "v": "uncertain", "n": 14}, {"o": "clarity", "r": "design_justification", "v": "conditional", "n": 209}, {"o": "clarity", "r": "design_justification", "v": "mixed", "n": 13}, {"o": "clarity", "r": "design_justification", "v": "negative", "n": 2730}, {"o": "clarity", "r": "design_justification", "v": "positive", "n": 206}, {"o": "clarity", "r": "design_justification", "v": "uncertain", "n": 83}, {"o": "clarity", "r": "fair_comparison", "v": "conditional", "n": 54}, {"o": "clarity", "r": "fair_comparison", "v": "mixed", "n": 11}, {"o": "clarity", "r": "fair_comparison", "v": "negative", "n": 641}, {"o": "clarity", "r": "fair_comparison", "v": "positive", "n": 36}, {"o": "clarity", "r": "fair_comparison", "v": "uncertain", "n": 18}, {"o": "clarity", "r": "merit_recognition", "v": "conditional", "n": 137}, {"o": "clarity", "r": "merit_recognition", "v": "mixed", "n": 38}, {"o": "clarity", "r": "merit_recognition", "v": "negative", "n": 1641}, {"o": "clarity", "r": "merit_recognition", "v": "positive", "n": 3384}, {"o": "clarity", "r": "merit_recognition", "v": "uncertain", "n": 46}, {"o": "clarity", "r": "novelty_standard", "v": "conditional", "n": 204}, {"o": "clarity", "r": "novelty_standard", "v": "mixed", "n": 58}, {"o": "clarity", "r": "novelty_standard", "v": "negative", "n": 3092}, {"o": "clarity", "r": "novelty_standard", "v": "positive", "n": 214}, {"o": "clarity", "r": "novelty_standard", "v": "uncertain", "n": 61}, {"o": "clarity", "r": "presentation_trust", "v": "conditional", "n": 222}, {"o": "clarity", "r": "presentation_trust", "v": "mixed", "n": 125}, {"o": "clarity", "r": "presentation_trust", "v": "negative", "n": 14854}, {"o": "clarity", "r": "presentation_trust", "v": "positive", "n": 373}, {"o": "clarity", "r": "presentation_trust", "v": "uncertain", "n": 114}, {"o": "clarity", "r": "reproducibility_norm", "v": "conditional", "n": 31}, {"o": "clarity", "r": "reproducibility_norm", "v": "mixed", "n": 4}, {"o": "clarity", "r": "reproducibility_norm", "v": "negative", "n": 983}, {"o": "clarity", "r": "reproducibility_norm", "v": "positive", "n": 39}, {"o": "clarity", "r": "reproducibility_norm", "v": "uncertain", "n": 13}, {"o": "clarity", "r": "robustness_norm", "v": "conditional", "n": 77}, {"o": "clarity", "r": "robustness_norm", "v": "mixed", "n": 8}, {"o": "clarity", "r": "robustness_norm", "v": "negative", "n": 453}, {"o": "clarity", "r": "robustness_norm", "v": "positive", "n": 42}, {"o": "clarity", "r": "robustness_norm", "v": "uncertain", "n": 43}, {"o": "clarity", "r": "statistical_identifiability", "v": "conditional", "n": 31}, {"o": "clarity", "r": "statistical_identifiability", "v": "mixed", "n": 2}, {"o": "clarity", "r": "statistical_identifiability", "v": "negative", "n": 756}, {"o": "clarity", "r": "statistical_identifiability", "v": "positive", "n": 18}, {"o": "clarity", "r": "statistical_identifiability", "v": "uncertain", "n": 16}, {"o": "compute_cost", "r": "claim_evidence_match", "v": "conditional", "n": 172}, {"o": "compute_cost", "r": "claim_evidence_match", "v": "mixed", "n": 12}, {"o": "compute_cost", "r": "claim_evidence_match", "v": "negative", "n": 1018}, {"o": "compute_cost", "r": "claim_evidence_match", "v": "positive", "n": 70}, {"o": "compute_cost", "r": "claim_evidence_match", "v": "uncertain", "n": 125}, {"o": "compute_cost", "r": "confound_hypothesis", "v": "conditional", "n": 116}, {"o": "compute_cost", "r": "confound_hypothesis", "v": "mixed", "n": 4}, {"o": "compute_cost", "r": "confound_hypothesis", "v": "negative", "n": 612}, {"o": "compute_cost", "r": "confound_hypothesis", "v": "positive", "n": 51}, {"o": "compute_cost", "r": "confound_hypothesis", "v": "uncertain", "n": 63}, {"o": "compute_cost", "r": "construct_validity", "v": "conditional", "n": 33}, {"o": "compute_cost", "r": "construct_validity", "v": "mixed", "n": 5}, {"o": "compute_cost", "r": "construct_validity", "v": "negative", "n": 367}, {"o": "compute_cost", "r": "construct_validity", "v": "positive", "n": 38}, {"o": "compute_cost", "r": "construct_validity", "v": "uncertain", "n": 23}, {"o": "compute_cost", "r": "cost_benefit", "v": "conditional", "n": 1260}, {"o": "compute_cost", "r": "cost_benefit", "v": "mixed", "n": 35}, {"o": "compute_cost", "r": "cost_benefit", "v": "negative", "n": 8387}, {"o": "compute_cost", "r": "cost_benefit", "v": "positive", "n": 567}, {"o": "compute_cost", "r": "cost_benefit", "v": "uncertain", "n": 634}, {"o": "compute_cost", "r": "design_justification", "v": "conditional", "n": 633}, {"o": "compute_cost", "r": "design_justification", "v": "mixed", "n": 19}, {"o": "compute_cost", "r": "design_justification", "v": "negative", "n": 4541}, {"o": "compute_cost", "r": "design_justification", "v": "positive", "n": 316}, {"o": "compute_cost", "r": "design_justification", "v": "uncertain", "n": 315}, {"o": "compute_cost", "r": "fair_comparison", "v": "conditional", "n": 334}, {"o": "compute_cost", "r": "fair_comparison", "v": "mixed", "n": 27}, {"o": "compute_cost", "r": "fair_comparison", "v": "negative", "n": 2614}, {"o": "compute_cost", "r": "fair_comparison", "v": "positive", "n": 141}, {"o": "compute_cost", "r": "fair_comparison", "v": "uncertain", "n": 132}, {"o": "compute_cost", "r": "merit_recognition", "v": "conditional", "n": 256}, {"o": "compute_cost", "r": "merit_recognition", "v": "mixed", "n": 20}, {"o": "compute_cost", "r": "merit_recognition", "v": "negative", "n": 863}, {"o": "compute_cost", "r": "merit_recognition", "v": "positive", "n": 2684}, {"o": "compute_cost", "r": "merit_recognition", "v": "uncertain", "n": 89}, {"o": "compute_cost", "r": "novelty_standard", "v": "conditional", "n": 254}, {"o": "compute_cost", "r": "novelty_standard", "v": "mixed", "n": 60}, {"o": "compute_cost", "r": "novelty_standard", "v": "negative", "n": 1956}, {"o": "compute_cost", "r": "novelty_standard", "v": "positive", "n": 263}, {"o": "compute_cost", "r": "novelty_standard", "v": "uncertain", "n": 67}, {"o": "compute_cost", "r": "presentation_trust", "v": "conditional", "n": 26}, {"o": "compute_cost", "r": "presentation_trust", "v": "mixed", "n": 6}, {"o": "compute_cost", "r": "presentation_trust", "v": "negative", "n": 401}, {"o": "compute_cost", "r": "presentation_trust", "v": "positive", "n": 12}, {"o": "compute_cost", "r": "presentation_trust", "v": "uncertain", "n": 15}, {"o": "compute_cost", "r": "reproducibility_norm", "v": "conditional", "n": 32}, {"o": "compute_cost", "r": "reproducibility_norm", "v": "mixed", "n": 2}, {"o": "compute_cost", "r": "reproducibility_norm", "v": "negative", "n": 453}, {"o": "compute_cost", "r": "reproducibility_norm", "v": "positive", "n": 20}, {"o": "compute_cost", "r": "reproducibility_norm", "v": "uncertain", "n": 14}, {"o": "compute_cost", "r": "robustness_norm", "v": "conditional", "n": 352}, {"o": "compute_cost", "r": "robustness_norm", "v": "mixed", "n": 14}, {"o": "compute_cost", "r": "robustness_norm", "v": "negative", "n": 1256}, {"o": "compute_cost", "r": "robustness_norm", "v": "positive", "n": 177}, {"o": "compute_cost", "r": "robustness_norm", "v": "uncertain", "n": 252}, {"o": "compute_cost", "r": "statistical_identifiability", "v": "conditional", "n": 89}, {"o": "compute_cost", "r": "statistical_identifiability", "v": "mixed", "n": 8}, {"o": "compute_cost", "r": "statistical_identifiability", "v": "negative", "n": 1164}, {"o": "compute_cost", "r": "statistical_identifiability", "v": "positive", "n": 15}, {"o": "compute_cost", "r": "statistical_identifiability", "v": "uncertain", "n": 43}, {"o": "empirical_scope", "r": "claim_evidence_match", "v": "conditional", "n": 1478}, {"o": "empirical_scope", "r": "claim_evidence_match", "v": "mixed", "n": 59}, {"o": "empirical_scope", "r": "claim_evidence_match", "v": "negative", "n": 7726}, {"o": "empirical_scope", "r": "claim_evidence_match", "v": "positive", "n": 552}, {"o": "empirical_scope", "r": "claim_evidence_match", "v": "uncertain", "n": 670}, {"o": "empirical_scope", "r": "confound_hypothesis", "v": "conditional", "n": 720}, {"o": "empirical_scope", "r": "confound_hypothesis", "v": "mixed", "n": 12}, {"o": "empirical_scope", "r": "confound_hypothesis", "v": "negative", "n": 4774}, {"o": "empirical_scope", "r": "confound_hypothesis", "v": "positive", "n": 268}, {"o": "empirical_scope", "r": "confound_hypothesis", "v": "uncertain", "n": 408}, {"o": "empirical_scope", "r": "construct_validity", "v": "conditional", "n": 331}, {"o": "empirical_scope", "r": "construct_validity", "v": "mixed", "n": 30}, {"o": "empirical_scope", "r": "construct_validity", "v": "negative", "n": 2565}, {"o": "empirical_scope", "r": "construct_validity", "v": "positive", "n": 146}, {"o": "empirical_scope", "r": "construct_validity", "v": "uncertain", "n": 148}, {"o": "empirical_scope", "r": "cost_benefit", "v": "conditional", "n": 437}, {"o": "empirical_scope", "r": "cost_benefit", "v": "mixed", "n": 26}, {"o": "empirical_scope", "r": "cost_benefit", "v": "negative", "n": 2067}, {"o": "empirical_scope", "r": "cost_benefit", "v": "positive", "n": 374}, {"o": "empirical_scope", "r": "cost_benefit", "v": "uncertain", "n": 245}, {"o": "empirical_scope", "r": "design_justification", "v": "conditional", "n": 2360}, {"o": "empirical_scope", "r": "design_justification", "v": "mixed", "n": 66}, {"o": "empirical_scope", "r": "design_justification", "v": "negative", "n": 15079}, {"o": "empirical_scope", "r": "design_justification", "v": "positive", "n": 1120}, {"o": "empirical_scope", "r": "design_justification", "v": "uncertain", "n": 1104}, {"o": "empirical_scope", "r": "fair_comparison", "v": "conditional", "n": 802}, {"o": "empirical_scope", "r": "fair_comparison", "v": "mixed", "n": 64}, {"o": "empirical_scope", "r": "fair_comparison", "v": "negative", "n": 5912}, {"o": "empirical_scope", "r": "fair_comparison", "v": "positive", "n": 323}, {"o": "empirical_scope", "r": "fair_comparison", "v": "uncertain", "n": 295}, {"o": "empirical_scope", "r": "merit_recognition", "v": "conditional", "n": 827}, {"o": "empirical_scope", "r": "merit_recognition", "v": "mixed", "n": 67}, {"o": "empirical_scope", "r": "merit_recognition", "v": "negative", "n": 2281}, {"o": "empirical_scope", "r": "merit_recognition", "v": "positive", "n": 7936}, {"o": "empirical_scope", "r": "merit_recognition", "v": "uncertain", "n": 265}, {"o": "empirical_scope", "r": "novelty_standard", "v": "conditional", "n": 874}, {"o": "empirical_scope", "r": "novelty_standard", "v": "mixed", "n": 156}, {"o": "empirical_scope", "r": "novelty_standard", "v": "negative", "n": 8005}, {"o": "empirical_scope", "r": "novelty_standard", "v": "positive", "n": 984}, {"o": "empirical_scope", "r": "novelty_standard", "v": "uncertain", "n": 264}, {"o": "empirical_scope", "r": "presentation_trust", "v": "conditional", "n": 176}, {"o": "empirical_scope", "r": "presentation_trust", "v": "mixed", "n": 12}, {"o": "empirical_scope", "r": "presentation_trust", "v": "negative", "n": 1883}, {"o": "empirical_scope", "r": "presentation_trust", "v": "positive", "n": 49}, {"o": "empirical_scope", "r": "presentation_trust", "v": "uncertain", "n": 97}, {"o": "empirical_scope", "r": "reproducibility_norm", "v": "conditional", "n": 159}, {"o": "empirical_scope", "r": "reproducibility_norm", "v": "mixed", "n": 12}, {"o": "empirical_scope", "r": "reproducibility_norm", "v": "negative", "n": 2027}, {"o": "empirical_scope", "r": "reproducibility_norm", "v": "positive", "n": 76}, {"o": "empirical_scope", "r": "reproducibility_norm", "v": "uncertain", "n": 50}, {"o": "empirical_scope", "r": "robustness_norm", "v": "conditional", "n": 1666}, {"o": "empirical_scope", "r": "robustness_norm", "v": "mixed", "n": 39}, {"o": "empirical_scope", "r": "robustness_norm", "v": "negative", "n": 6025}, {"o": "empirical_scope", "r": "robustness_norm", "v": "positive", "n": 714}, {"o": "empirical_scope", "r": "robustness_norm", "v": "uncertain", "n": 967}, {"o": "empirical_scope", "r": "statistical_identifiability", "v": "conditional", "n": 265}, {"o": "empirical_scope", "r": "statistical_identifiability", "v": "mixed", "n": 18}, {"o": "empirical_scope", "r": "statistical_identifiability", "v": "negative", "n": 3219}, {"o": "empirical_scope", "r": "statistical_identifiability", "v": "positive", "n": 50}, {"o": "empirical_scope", "r": "statistical_identifiability", "v": "uncertain", "n": 138}, {"o": "method_design", "r": "claim_evidence_match", "v": "conditional", "n": 661}, {"o": "method_design", "r": "claim_evidence_match", "v": "mixed", "n": 22}, {"o": "method_design", "r": "claim_evidence_match", "v": "negative", "n": 2718}, {"o": "method_design", "r": "claim_evidence_match", "v": "positive", "n": 215}, {"o": "method_design", "r": "claim_evidence_match", "v": "uncertain", "n": 394}, {"o": "method_design", "r": "confound_hypothesis", "v": "conditional", "n": 539}, {"o": "method_design", "r": "confound_hypothesis", "v": "mixed", "n": 15}, {"o": "method_design", "r": "confound_hypothesis", "v": "negative", "n": 3137}, {"o": "method_design", "r": "confound_hypothesis", "v": "positive", "n": 200}, {"o": "method_design", "r": "confound_hypothesis", "v": "uncertain", "n": 288}, {"o": "method_design", "r": "construct_validity", "v": "conditional", "n": 143}, {"o": "method_design", "r": "construct_validity", "v": "mixed", "n": 8}, {"o": "method_design", "r": "construct_validity", "v": "negative", "n": 1204}, {"o": "method_design", "r": "construct_validity", "v": "positive", "n": 43}, {"o": "method_design", "r": "construct_validity", "v": "uncertain", "n": 52}, {"o": "method_design", "r": "cost_benefit", "v": "conditional", "n": 445}, {"o": "method_design", "r": "cost_benefit", "v": "mixed", "n": 16}, {"o": "method_design", "r": "cost_benefit", "v": "negative", "n": 1961}, {"o": "method_design", "r": "cost_benefit", "v": "positive", "n": 421}, {"o": "method_design", "r": "cost_benefit", "v": "uncertain", "n": 199}, {"o": "method_design", "r": "design_justification", "v": "conditional", "n": 1682}, {"o": "method_design", "r": "design_justification", "v": "mixed", "n": 61}, {"o": "method_design", "r": "design_justification", "v": "negative", "n": 10589}, {"o": "method_design", "r": "design_justification", "v": "positive", "n": 1010}, {"o": "method_design", "r": "design_justification", "v": "uncertain", "n": 793}, {"o": "method_design", "r": "fair_comparison", "v": "conditional", "n": 547}, {"o": "method_design", "r": "fair_comparison", "v": "mixed", "n": 22}, {"o": "method_design", "r": "fair_comparison", "v": "negative", "n": 3142}, {"o": "method_design", "r": "fair_comparison", "v": "positive", "n": 201}, {"o": "method_design", "r": "fair_comparison", "v": "uncertain", "n": 266}, {"o": "method_design", "r": "merit_recognition", "v": "conditional", "n": 670}, {"o": "method_design", "r": "merit_recognition", "v": "mixed", "n": 66}, {"o": "method_design", "r": "merit_recognition", "v": "negative", "n": 1741}, {"o": "method_design", "r": "merit_recognition", "v": "positive", "n": 6332}, {"o": "method_design", "r": "merit_recognition", "v": "uncertain", "n": 261}, {"o": "method_design", "r": "novelty_standard", "v": "conditional", "n": 880}, {"o": "method_design", "r": "novelty_standard", "v": "mixed", "n": 133}, {"o": "method_design", "r": "novelty_standard", "v": "negative", "n": 7795}, {"o": "method_design", "r": "novelty_standard", "v": "positive", "n": 1029}, {"o": "method_design", "r": "novelty_standard", "v": "uncertain", "n": 276}, {"o": "method_design", "r": "presentation_trust", "v": "conditional", "n": 101}, {"o": "method_design", "r": "presentation_trust", "v": "mixed", "n": 9}, {"o": "method_design", "r": "presentation_trust", "v": "negative", "n": 1393}, {"o": "method_design", "r": "presentation_trust", "v": "positive", "n": 46}, {"o": "method_design", "r": "presentation_trust", "v": "uncertain", "n": 52}, {"o": "method_design", "r": "reproducibility_norm", "v": "conditional", "n": 83}, {"o": "method_design", "r": "reproducibility_norm", "v": "mixed", "n": 8}, {"o": "method_design", "r": "reproducibility_norm", "v": "negative", "n": 1087}, {"o": "method_design", "r": "reproducibility_norm", "v": "positive", "n": 34}, {"o": "method_design", "r": "reproducibility_norm", "v": "uncertain", "n": 39}, {"o": "method_design", "r": "robustness_norm", "v": "conditional", "n": 889}, {"o": "method_design", "r": "robustness_norm", "v": "mixed", "n": 23}, {"o": "method_design", "r": "robustness_norm", "v": "negative", "n": 3072}, {"o": "method_design", "r": "robustness_norm", "v": "positive", "n": 357}, {"o": "method_design", "r": "robustness_norm", "v": "uncertain", "n": 526}, {"o": "method_design", "r": "statistical_identifiability", "v": "conditional", "n": 145}, {"o": "method_design", "r": "statistical_identifiability", "v": "mixed", "n": 14}, {"o": "method_design", "r": "statistical_identifiability", "v": "negative", "n": 1954}, {"o": "method_design", "r": "statistical_identifiability", "v": "positive", "n": 24}, {"o": "method_design", "r": "statistical_identifiability", "v": "uncertain", "n": 94}, {"o": "novelty", "r": "claim_evidence_match", "v": "conditional", "n": 11}, {"o": "novelty", "r": "claim_evidence_match", "v": "mixed", "n": 6}, {"o": "novelty", "r": "claim_evidence_match", "v": "negative", "n": 165}, {"o": "novelty", "r": "claim_evidence_match", "v": "positive", "n": 16}, {"o": "novelty", "r": "claim_evidence_match", "v": "uncertain", "n": 4}, {"o": "novelty", "r": "confound_hypothesis", "v": "conditional", "n": 7}, {"o": "novelty", "r": "confound_hypothesis", "v": "mixed", "n": 3}, {"o": "novelty", "r": "confound_hypothesis", "v": "negative", "n": 141}, {"o": "novelty", "r": "confound_hypothesis", "v": "positive", "n": 11}, {"o": "novelty", "r": "confound_hypothesis", "v": "uncertain", "n": 2}, {"o": "novelty", "r": "construct_validity", "v": "conditional", "n": 9}, {"o": "novelty", "r": "construct_validity", "v": "mixed", "n": 1}, {"o": "novelty", "r": "construct_validity", "v": "negative", "n": 93}, {"o": "novelty", "r": "construct_validity", "v": "positive", "n": 2}, {"o": "novelty", "r": "construct_validity", "v": "uncertain", "n": 5}, {"o": "novelty", "r": "cost_benefit", "v": "conditional", "n": 6}, {"o": "novelty", "r": "cost_benefit", "v": "mixed", "n": 3}, {"o": "novelty", "r": "cost_benefit", "v": "negative", "n": 55}, {"o": "novelty", "r": "cost_benefit", "v": "positive", "n": 17}, {"o": "novelty", "r": "cost_benefit", "v": "uncertain", "n": 2}, {"o": "novelty", "r": "design_justification", "v": "conditional", "n": 71}, {"o": "novelty", "r": "design_justification", "v": "mixed", "n": 19}, {"o": "novelty", "r": "design_justification", "v": "negative", "n": 1368}, {"o": "novelty", "r": "design_justification", "v": "positive", "n": 78}, {"o": "novelty", "r": "design_justification", "v": "uncertain", "n": 26}, {"o": "novelty", "r": "fair_comparison", "v": "conditional", "n": 17}, {"o": "novelty", "r": "fair_comparison", "v": "mixed", "n": 6}, {"o": "novelty", "r": "fair_comparison", "v": "negative", "n": 302}, {"o": "novelty", "r": "fair_comparison", "v": "positive", "n": 16}, {"o": "novelty", "r": "fair_comparison", "v": "uncertain", "n": 8}, {"o": "novelty", "r": "merit_recognition", "v": "conditional", "n": 31}, {"o": "novelty", "r": "merit_recognition", "v": "mixed", "n": 13}, {"o": "novelty", "r": "merit_recognition", "v": "negative", "n": 187}, {"o": "novelty", "r": "merit_recognition", "v": "positive", "n": 563}, {"o": "novelty", "r": "merit_recognition", "v": "uncertain", "n": 9}, {"o": "novelty", "r": "novelty_standard", "v": "conditional", "n": 392}, {"o": "novelty", "r": "novelty_standard", "v": "mixed", "n": 160}, {"o": "novelty", "r": "novelty_standard", "v": "negative", "n": 6682}, {"o": "novelty", "r": "novelty_standard", "v": "positive", "n": 306}, {"o": "novelty", "r": "novelty_standard", "v": "uncertain", "n": 116}, {"o": "novelty", "r": "presentation_trust", "v": "conditional", "n": 8}, {"o": "novelty", "r": "presentation_trust", "v": "mixed", "n": 3}, {"o": "novelty", "r": "presentation_trust", "v": "negative", "n": 123}, {"o": "novelty", "r": "presentation_trust", "v": "positive", "n": 9}, {"o": "novelty", "r": "presentation_trust", "v": "uncertain", "n": 6}, {"o": "novelty", "r": "reproducibility_norm", "v": "conditional", "n": 3}, {"o": "novelty", "r": "reproducibility_norm", "v": "negative", "n": 31}, {"o": "novelty", "r": "reproducibility_norm", "v": "positive", "n": 2}, {"o": "novelty", "r": "robustness_norm", "v": "conditional", "n": 19}, {"o": "novelty", "r": "robustness_norm", "v": "mixed", "n": 2}, {"o": "novelty", "r": "robustness_norm", "v": "negative", "n": 96}, {"o": "novelty", "r": "robustness_norm", "v": "positive", "n": 12}, {"o": "novelty", "r": "robustness_norm", "v": "uncertain", "n": 11}, {"o": "novelty", "r": "statistical_identifiability", "v": "conditional", "n": 12}, {"o": "novelty", "r": "statistical_identifiability", "v": "mixed", "n": 1}, {"o": "novelty", "r": "statistical_identifiability", "v": "negative", "n": 166}, {"o": "novelty", "r": "statistical_identifiability", "v": "positive", "n": 1}, {"o": "novelty", "r": "statistical_identifiability", "v": "uncertain", "n": 6}, {"o": "problem_framing", "r": "claim_evidence_match", "v": "conditional", "n": 59}, {"o": "problem_framing", "r": "claim_evidence_match", "v": "mixed", "n": 5}, {"o": "problem_framing", "r": "claim_evidence_match", "v": "negative", "n": 380}, {"o": "problem_framing", "r": "claim_evidence_match", "v": "positive", "n": 44}, {"o": "problem_framing", "r": "claim_evidence_match", "v": "uncertain", "n": 18}, {"o": "problem_framing", "r": "confound_hypothesis", "v": "conditional", "n": 60}, {"o": "problem_framing", "r": "confound_hypothesis", "v": "mixed", "n": 3}, {"o": "problem_framing", "r": "confound_hypothesis", "v": "negative", "n": 357}, {"o": "problem_framing", "r": "confound_hypothesis", "v": "positive", "n": 34}, {"o": "problem_framing", "r": "confound_hypothesis", "v": "uncertain", "n": 10}, {"o": "problem_framing", "r": "construct_validity", "v": "conditional", "n": 25}, {"o": "problem_framing", "r": "construct_validity", "v": "mixed", "n": 1}, {"o": "problem_framing", "r": "construct_validity", "v": "negative", "n": 236}, {"o": "problem_framing", "r": "construct_validity", "v": "positive", "n": 14}, {"o": "problem_framing", "r": "construct_validity", "v": "uncertain", "n": 4}, {"o": "problem_framing", "r": "cost_benefit", "v": "conditional", "n": 17}, {"o": "problem_framing", "r": "cost_benefit", "v": "mixed", "n": 1}, {"o": "problem_framing", "r": "cost_benefit", "v": "negative", "n": 118}, {"o": "problem_framing", "r": "cost_benefit", "v": "positive", "n": 63}, {"o": "problem_framing", "r": "cost_benefit", "v": "uncertain", "n": 5}, {"o": "problem_framing", "r": "design_justification", "v": "conditional", "n": 150}, {"o": "problem_framing", "r": "design_justification", "v": "mixed", "n": 11}, {"o": "problem_framing", "r": "design_justification", "v": "negative", "n": 1572}, {"o": "problem_framing", "r": "design_justification", "v": "positive", "n": 182}, {"o": "problem_framing", "r": "design_justification", "v": "uncertain", "n": 54}, {"o": "problem_framing", "r": "fair_comparison", "v": "conditional", "n": 46}, {"o": "problem_framing", "r": "fair_comparison", "v": "mixed", "n": 6}, {"o": "problem_framing", "r": "fair_comparison", "v": "negative", "n": 348}, {"o": "problem_framing", "r": "fair_comparison", "v": "positive", "n": 49}, {"o": "problem_framing", "r": "fair_comparison", "v": "uncertain", "n": 20}, {"o": "problem_framing", "r": "merit_recognition", "v": "conditional", "n": 92}, {"o": "problem_framing", "r": "merit_recognition", "v": "mixed", "n": 16}, {"o": "problem_framing", "r": "merit_recognition", "v": "negative", "n": 474}, {"o": "problem_framing", "r": "merit_recognition", "v": "positive", "n": 2876}, {"o": "problem_framing", "r": "merit_recognition", "v": "uncertain", "n": 22}, {"o": "problem_framing", "r": "novelty_standard", "v": "conditional", "n": 179}, {"o": "problem_framing", "r": "novelty_standard", "v": "mixed", "n": 60}, {"o": "problem_framing", "r": "novelty_standard", "v": "negative", "n": 2478}, {"o": "problem_framing", "r": "novelty_standard", "v": "positive", "n": 420}, {"o": "problem_framing", "r": "novelty_standard", "v": "uncertain", "n": 49}, {"o": "problem_framing", "r": "presentation_trust", "v": "conditional", "n": 25}, {"o": "problem_framing", "r": "presentation_trust", "v": "mixed", "n": 6}, {"o": "problem_framing", "r": "presentation_trust", "v": "negative", "n": 689}, {"o": "problem_framing", "r": "presentation_trust", "v": "positive", "n": 37}, {"o": "problem_framing", "r": "presentation_trust", "v": "uncertain", "n": 12}, {"o": "problem_framing", "r": "reproducibility_norm", "v": "conditional", "n": 9}, {"o": "problem_framing", "r": "reproducibility_norm", "v": "negative", "n": 92}, {"o": "problem_framing", "r": "reproducibility_norm", "v": "positive", "n": 10}, {"o": "problem_framing", "r": "reproducibility_norm", "v": "uncertain", "n": 2}, {"o": "problem_framing", "r": "robustness_norm", "v": "conditional", "n": 42}, {"o": "problem_framing", "r": "robustness_norm", "v": "mixed", "n": 2}, {"o": "problem_framing", "r": "robustness_norm", "v": "negative", "n": 190}, {"o": "problem_framing", "r": "robustness_norm", "v": "positive", "n": 81}, {"o": "problem_framing", "r": "robustness_norm", "v": "uncertain", "n": 22}, {"o": "problem_framing", "r": "statistical_identifiability", "v": "conditional", "n": 23}, {"o": "problem_framing", "r": "statistical_identifiability", "v": "mixed", "n": 2}, {"o": "problem_framing", "r": "statistical_identifiability", "v": "negative", "n": 338}, {"o": "problem_framing", "r": "statistical_identifiability", "v": "positive", "n": 9}, {"o": "problem_framing", "r": "statistical_identifiability", "v": "uncertain", "n": 5}, {"o": "related_work", "r": "claim_evidence_match", "v": "conditional", "n": 19}, {"o": "related_work", "r": "claim_evidence_match", "v": "mixed", "n": 2}, {"o": "related_work", "r": "claim_evidence_match", "v": "negative", "n": 163}, {"o": "related_work", "r": "claim_evidence_match", "v": "positive", "n": 8}, {"o": "related_work", "r": "claim_evidence_match", "v": "uncertain", "n": 3}, {"o": "related_work", "r": "confound_hypothesis", "v": "conditional", "n": 11}, {"o": "related_work", "r": "confound_hypothesis", "v": "negative", "n": 106}, {"o": "related_work", "r": "confound_hypothesis", "v": "positive", "n": 10}, {"o": "related_work", "r": "confound_hypothesis", "v": "uncertain", "n": 5}, {"o": "related_work", "r": "construct_validity", "v": "conditional", "n": 7}, {"o": "related_work", "r": "construct_validity", "v": "mixed", "n": 1}, {"o": "related_work", "r": "construct_validity", "v": "negative", "n": 83}, {"o": "related_work", "r": "construct_validity", "v": "positive", "n": 6}, {"o": "related_work", "r": "construct_validity", "v": "uncertain", "n": 3}, {"o": "related_work", "r": "cost_benefit", "v": "conditional", "n": 9}, {"o": "related_work", "r": "cost_benefit", "v": "negative", "n": 31}, {"o": "related_work", "r": "cost_benefit", "v": "positive", "n": 9}, {"o": "related_work", "r": "cost_benefit", "v": "uncertain", "n": 1}, {"o": "related_work", "r": "design_justification", "v": "conditional", "n": 46}, {"o": "related_work", "r": "design_justification", "v": "mixed", "n": 3}, {"o": "related_work", "r": "design_justification", "v": "negative", "n": 912}, {"o": "related_work", "r": "design_justification", "v": "positive", "n": 38}, {"o": "related_work", "r": "design_justification", "v": "uncertain", "n": 11}, {"o": "related_work", "r": "fair_comparison", "v": "conditional", "n": 40}, {"o": "related_work", "r": "fair_comparison", "v": "mixed", "n": 1}, {"o": "related_work", "r": "fair_comparison", "v": "negative", "n": 364}, {"o": "related_work", "r": "fair_comparison", "v": "positive", "n": 10}, {"o": "related_work", "r": "fair_comparison", "v": "uncertain", "n": 8}, {"o": "related_work", "r": "merit_recognition", "v": "conditional", "n": 62}, {"o": "related_work", "r": "merit_recognition", "v": "mixed", "n": 6}, {"o": "related_work", "r": "merit_recognition", "v": "negative", "n": 551}, {"o": "related_work", "r": "merit_recognition", "v": "positive", "n": 378}, {"o": "related_work", "r": "merit_recognition", "v": "uncertain", "n": 11}, {"o": "related_work", "r": "novelty_standard", "v": "conditional", "n": 212}, {"o": "related_work", "r": "novelty_standard", "v": "mixed", "n": 37}, {"o": "related_work", "r": "novelty_standard", "v": "negative", "n": 4736}, {"o": "related_work", "r": "novelty_standard", "v": "positive", "n": 108}, {"o": "related_work", "r": "novelty_standard", "v": "uncertain", "n": 64}, {"o": "related_work", "r": "presentation_trust", "v": "conditional", "n": 29}, {"o": "related_work", "r": "presentation_trust", "v": "mixed", "n": 4}, {"o": "related_work", "r": "presentation_trust", "v": "negative", "n": 1749}, {"o": "related_work", "r": "presentation_trust", "v": "positive", "n": 22}, {"o": "related_work", "r": "presentation_trust", "v": "uncertain", "n": 19}, {"o": "related_work", "r": "reproducibility_norm", "v": "conditional", "n": 15}, {"o": "related_work", "r": "reproducibility_norm", "v": "negative", "n": 162}, {"o": "related_work", "r": "reproducibility_norm", "v": "positive", "n": 5}, {"o": "related_work", "r": "reproducibility_norm", "v": "uncertain", "n": 5}, {"o": "related_work", "r": "robustness_norm", "v": "conditional", "n": 18}, {"o": "related_work", "r": "robustness_norm", "v": "mixed", "n": 4}, {"o": "related_work", "r": "robustness_norm", "v": "negative", "n": 51}, {"o": "related_work", "r": "robustness_norm", "v": "positive", "n": 4}, {"o": "related_work", "r": "robustness_norm", "v": "uncertain", "n": 7}, {"o": "related_work", "r": "statistical_identifiability", "v": "conditional", "n": 8}, {"o": "related_work", "r": "statistical_identifiability", "v": "negative", "n": 125}, {"o": "related_work", "r": "statistical_identifiability", "v": "positive", "n": 2}, {"o": "related_work", "r": "statistical_identifiability", "v": "uncertain", "n": 2}, {"o": "reproducibility", "r": "claim_evidence_match", "v": "conditional", "n": 54}, {"o": "reproducibility", "r": "claim_evidence_match", "v": "mixed", "n": 1}, {"o": "reproducibility", "r": "claim_evidence_match", "v": "negative", "n": 253}, {"o": "reproducibility", "r": "claim_evidence_match", "v": "positive", "n": 14}, {"o": "reproducibility", "r": "claim_evidence_match", "v": "uncertain", "n": 24}, {"o": "reproducibility", "r": "confound_hypothesis", "v": "conditional", "n": 37}, {"o": "reproducibility", "r": "confound_hypothesis", "v": "negative", "n": 313}, {"o": "reproducibility", "r": "confound_hypothesis", "v": "positive", "n": 17}, {"o": "reproducibility", "r": "confound_hypothesis", "v": "uncertain", "n": 20}, {"o": "reproducibility", "r": "construct_validity", "v": "conditional", "n": 15}, {"o": "reproducibility", "r": "construct_validity", "v": "mixed", "n": 1}, {"o": "reproducibility", "r": "construct_validity", "v": "negative", "n": 173}, {"o": "reproducibility", "r": "construct_validity", "v": "positive", "n": 4}, {"o": "reproducibility", "r": "construct_validity", "v": "uncertain", "n": 6}, {"o": "reproducibility", "r": "cost_benefit", "v": "conditional", "n": 37}, {"o": "reproducibility", "r": "cost_benefit", "v": "mixed", "n": 2}, {"o": "reproducibility", "r": "cost_benefit", "v": "negative", "n": 266}, {"o": "reproducibility", "r": "cost_benefit", "v": "positive", "n": 55}, {"o": "reproducibility", "r": "cost_benefit", "v": "uncertain", "n": 15}, {"o": "reproducibility", "r": "design_justification", "v": "conditional", "n": 79}, {"o": "reproducibility", "r": "design_justification", "v": "mixed", "n": 6}, {"o": "reproducibility", "r": "design_justification", "v": "negative", "n": 675}, {"o": "reproducibility", "r": "design_justification", "v": "positive", "n": 38}, {"o": "reproducibility", "r": "design_justification", "v": "uncertain", "n": 39}, {"o": "reproducibility", "r": "fair_comparison", "v": "conditional", "n": 37}, {"o": "reproducibility", "r": "fair_comparison", "v": "mixed", "n": 2}, {"o": "reproducibility", "r": "fair_comparison", "v": "negative", "n": 270}, {"o": "reproducibility", "r": "fair_comparison", "v": "positive", "n": 17}, {"o": "reproducibility", "r": "fair_comparison", "v": "uncertain", "n": 19}, {"o": "reproducibility", "r": "merit_recognition", "v": "conditional", "n": 90}, {"o": "reproducibility", "r": "merit_recognition", "v": "mixed", "n": 5}, {"o": "reproducibility", "r": "merit_recognition", "v": "negative", "n": 281}, {"o": "reproducibility", "r": "merit_recognition", "v": "positive", "n": 774}, {"o": "reproducibility", "r": "merit_recognition", "v": "uncertain", "n": 19}, {"o": "reproducibility", "r": "novelty_standard", "v": "conditional", "n": 54}, {"o": "reproducibility", "r": "novelty_standard", "v": "mixed", "n": 11}, {"o": "reproducibility", "r": "novelty_standard", "v": "negative", "n": 555}, {"o": "reproducibility", "r": "novelty_standard", "v": "positive", "n": 58}, {"o": "reproducibility", "r": "novelty_standard", "v": "uncertain", "n": 17}, {"o": "reproducibility", "r": "presentation_trust", "v": "conditional", "n": 23}, {"o": "reproducibility", "r": "presentation_trust", "v": "mixed", "n": 2}, {"o": "reproducibility", "r": "presentation_trust", "v": "negative", "n": 509}, {"o": "reproducibility", "r": "presentation_trust", "v": "positive", "n": 14}, {"o": "reproducibility", "r": "presentation_trust", "v": "uncertain", "n": 11}, {"o": "reproducibility", "r": "reproducibility_norm", "v": "conditional", "n": 120}, {"o": "reproducibility", "r": "reproducibility_norm", "v": "mixed", "n": 6}, {"o": "reproducibility", "r": "reproducibility_norm", "v": "negative", "n": 2377}, {"o": "reproducibility", "r": "reproducibility_norm", "v": "positive", "n": 132}, {"o": "reproducibility", "r": "reproducibility_norm", "v": "uncertain", "n": 30}, {"o": "reproducibility", "r": "robustness_norm", "v": "conditional", "n": 44}, {"o": "reproducibility", "r": "robustness_norm", "v": "mixed", "n": 2}, {"o": "reproducibility", "r": "robustness_norm", "v": "negative", "n": 269}, {"o": "reproducibility", "r": "robustness_norm", "v": "positive", "n": 22}, {"o": "reproducibility", "r": "robustness_norm", "v": "uncertain", "n": 37}, {"o": "reproducibility", "r": "statistical_identifiability", "v": "conditional", "n": 29}, {"o": "reproducibility", "r": "statistical_identifiability", "v": "negative", "n": 609}, {"o": "reproducibility", "r": "statistical_identifiability", "v": "positive", "n": 1}, {"o": "reproducibility", "r": "statistical_identifiability", "v": "uncertain", "n": 6}, {"o": "robustness_sensitivity", "r": "claim_evidence_match", "v": "conditional", "n": 118}, {"o": "robustness_sensitivity", "r": "claim_evidence_match", "v": "mixed", "n": 2}, {"o": "robustness_sensitivity", "r": "claim_evidence_match", "v": "negative", "n": 595}, {"o": "robustness_sensitivity", "r": "claim_evidence_match", "v": "positive", "n": 33}, {"o": "robustness_sensitivity", "r": "claim_evidence_match", "v": "uncertain", "n": 92}, {"o": "robustness_sensitivity", "r": "confound_hypothesis", "v": "conditional", "n": 130}, {"o": "robustness_sensitivity", "r": "confound_hypothesis", "v": "mixed", "n": 4}, {"o": "robustness_sensitivity", "r": "confound_hypothesis", "v": "negative", "n": 706}, {"o": "robustness_sensitivity", "r": "confound_hypothesis", "v": "positive", "n": 32}, {"o": "robustness_sensitivity", "r": "confound_hypothesis", "v": "uncertain", "n": 72}, {"o": "robustness_sensitivity", "r": "construct_validity", "v": "conditional", "n": 49}, {"o": "robustness_sensitivity", "r": "construct_validity", "v": "mixed", "n": 2}, {"o": "robustness_sensitivity", "r": "construct_validity", "v": "negative", "n": 385}, {"o": "robustness_sensitivity", "r": "construct_validity", "v": "positive", "n": 15}, {"o": "robustness_sensitivity", "r": "construct_validity", "v": "uncertain", "n": 24}, {"o": "robustness_sensitivity", "r": "cost_benefit", "v": "conditional", "n": 177}, {"o": "robustness_sensitivity", "r": "cost_benefit", "v": "mixed", "n": 7}, {"o": "robustness_sensitivity", "r": "cost_benefit", "v": "negative", "n": 840}, {"o": "robustness_sensitivity", "r": "cost_benefit", "v": "positive", "n": 86}, {"o": "robustness_sensitivity", "r": "cost_benefit", "v": "uncertain", "n": 81}, {"o": "robustness_sensitivity", "r": "design_justification", "v": "conditional", "n": 437}, {"o": "robustness_sensitivity", "r": "design_justification", "v": "mixed", "n": 17}, {"o": "robustness_sensitivity", "r": "design_justification", "v": "negative", "n": 2511}, {"o": "robustness_sensitivity", "r": "design_justification", "v": "positive", "n": 126}, {"o": "robustness_sensitivity", "r": "design_justification", "v": "uncertain", "n": 201}, {"o": "robustness_sensitivity", "r": "fair_comparison", "v": "conditional", "n": 205}, {"o": "robustness_sensitivity", "r": "fair_comparison", "v": "mixed", "n": 9}, {"o": "robustness_sensitivity", "r": "fair_comparison", "v": "negative", "n": 1335}, {"o": "robustness_sensitivity", "r": "fair_comparison", "v": "positive", "n": 27}, {"o": "robustness_sensitivity", "r": "fair_comparison", "v": "uncertain", "n": 90}, {"o": "robustness_sensitivity", "r": "merit_recognition", "v": "conditional", "n": 181}, {"o": "robustness_sensitivity", "r": "merit_recognition", "v": "mixed", "n": 10}, {"o": "robustness_sensitivity", "r": "merit_recognition", "v": "negative", "n": 640}, {"o": "robustness_sensitivity", "r": "merit_recognition", "v": "positive", "n": 899}, {"o": "robustness_sensitivity", "r": "merit_recognition", "v": "uncertain", "n": 81}, {"o": "robustness_sensitivity", "r": "novelty_standard", "v": "conditional", "n": 160}, {"o": "robustness_sensitivity", "r": "novelty_standard", "v": "mixed", "n": 16}, {"o": "robustness_sensitivity", "r": "novelty_standard", "v": "negative", "n": 1221}, {"o": "robustness_sensitivity", "r": "novelty_standard", "v": "positive", "n": 131}, {"o": "robustness_sensitivity", "r": "novelty_standard", "v": "uncertain", "n": 66}, {"o": "robustness_sensitivity", "r": "presentation_trust", "v": "conditional", "n": 24}, {"o": "robustness_sensitivity", "r": "presentation_trust", "v": "mixed", "n": 3}, {"o": "robustness_sensitivity", "r": "presentation_trust", "v": "negative", "n": 393}, {"o": "robustness_sensitivity", "r": "presentation_trust", "v": "positive", "n": 4}, {"o": "robustness_sensitivity", "r": "presentation_trust", "v": "uncertain", "n": 7}, {"o": "robustness_sensitivity", "r": "reproducibility_norm", "v": "conditional", "n": 34}, {"o": "robustness_sensitivity", "r": "reproducibility_norm", "v": "mixed", "n": 3}, {"o": "robustness_sensitivity", "r": "reproducibility_norm", "v": "negative", "n": 612}, {"o": "robustness_sensitivity", "r": "reproducibility_norm", "v": "positive", "n": 6}, {"o": "robustness_sensitivity", "r": "reproducibility_norm", "v": "uncertain", "n": 13}, {"o": "robustness_sensitivity", "r": "robustness_norm", "v": "conditional", "n": 1271}, {"o": "robustness_sensitivity", "r": "robustness_norm", "v": "mixed", "n": 10}, {"o": "robustness_sensitivity", "r": "robustness_norm", "v": "negative", "n": 4783}, {"o": "robustness_sensitivity", "r": "robustness_norm", "v": "positive", "n": 138}, {"o": "robustness_sensitivity", "r": "robustness_norm", "v": "uncertain", "n": 720}, {"o": "robustness_sensitivity", "r": "statistical_identifiability", "v": "conditional", "n": 68}, {"o": "robustness_sensitivity", "r": "statistical_identifiability", "v": "mixed", "n": 2}, {"o": "robustness_sensitivity", "r": "statistical_identifiability", "v": "negative", "n": 1024}, {"o": "robustness_sensitivity", "r": "statistical_identifiability", "v": "positive", "n": 7}, {"o": "robustness_sensitivity", "r": "statistical_identifiability", "v": "uncertain", "n": 35}, {"o": "stats_metrics", "r": "claim_evidence_match", "v": "conditional", "n": 276}, {"o": "stats_metrics", "r": "claim_evidence_match", "v": "mixed", "n": 26}, {"o": "stats_metrics", "r": "claim_evidence_match", "v": "negative", "n": 2403}, {"o": "stats_metrics", "r": "claim_evidence_match", "v": "positive", "n": 112}, {"o": "stats_metrics", "r": "claim_evidence_match", "v": "uncertain", "n": 76}, {"o": "stats_metrics", "r": "confound_hypothesis", "v": "conditional", "n": 160}, {"o": "stats_metrics", "r": "confound_hypothesis", "v": "mixed", "n": 8}, {"o": "stats_metrics", "r": "confound_hypothesis", "v": "negative", "n": 1405}, {"o": "stats_metrics", "r": "confound_hypothesis", "v": "positive", "n": 53}, {"o": "stats_metrics", "r": "confound_hypothesis", "v": "uncertain", "n": 79}, {"o": "stats_metrics", "r": "construct_validity", "v": "conditional", "n": 426}, {"o": "stats_metrics", "r": "construct_validity", "v": "mixed", "n": 36}, {"o": "stats_metrics", "r": "construct_validity", "v": "negative", "n": 4249}, {"o": "stats_metrics", "r": "construct_validity", "v": "positive", "n": 195}, {"o": "stats_metrics", "r": "construct_validity", "v": "uncertain", "n": 165}, {"o": "stats_metrics", "r": "cost_benefit", "v": "conditional", "n": 122}, {"o": "stats_metrics", "r": "cost_benefit", "v": "mixed", "n": 15}, {"o": "stats_metrics", "r": "cost_benefit", "v": "negative", "n": 946}, {"o": "stats_metrics", "r": "cost_benefit", "v": "positive", "n": 100}, {"o": "stats_metrics", "r": "cost_benefit", "v": "uncertain", "n": 44}, {"o": "stats_metrics", "r": "design_justification", "v": "conditional", "n": 472}, {"o": "stats_metrics", "r": "design_justification", "v": "mixed", "n": 31}, {"o": "stats_metrics", "r": "design_justification", "v": "negative", "n": 4838}, {"o": "stats_metrics", "r": "design_justification", "v": "positive", "n": 171}, {"o": "stats_metrics", "r": "design_justification", "v": "uncertain", "n": 189}, {"o": "stats_metrics", "r": "fair_comparison", "v": "conditional", "n": 409}, {"o": "stats_metrics", "r": "fair_comparison", "v": "mixed", "n": 29}, {"o": "stats_metrics", "r": "fair_comparison", "v": "negative", "n": 3870}, {"o": "stats_metrics", "r": "fair_comparison", "v": "positive", "n": 185}, {"o": "stats_metrics", "r": "fair_comparison", "v": "uncertain", "n": 169}, {"o": "stats_metrics", "r": "merit_recognition", "v": "conditional", "n": 336}, {"o": "stats_metrics", "r": "merit_recognition", "v": "mixed", "n": 47}, {"o": "stats_metrics", "r": "merit_recognition", "v": "negative", "n": 1497}, {"o": "stats_metrics", "r": "merit_recognition", "v": "positive", "n": 3239}, {"o": "stats_metrics", "r": "merit_recognition", "v": "uncertain", "n": 112}, {"o": "stats_metrics", "r": "novelty_standard", "v": "conditional", "n": 398}, {"o": "stats_metrics", "r": "novelty_standard", "v": "mixed", "n": 99}, {"o": "stats_metrics", "r": "novelty_standard", "v": "negative", "n": 2943}, {"o": "stats_metrics", "r": "novelty_standard", "v": "positive", "n": 256}, {"o": "stats_metrics", "r": "novelty_standard", "v": "uncertain", "n": 136}, {"o": "stats_metrics", "r": "presentation_trust", "v": "conditional", "n": 148}, {"o": "stats_metrics", "r": "presentation_trust", "v": "mixed", "n": 16}, {"o": "stats_metrics", "r": "presentation_trust", "v": "negative", "n": 1936}, {"o": "stats_metrics", "r": "presentation_trust", "v": "positive", "n": 47}, {"o": "stats_metrics", "r": "presentation_trust", "v": "uncertain", "n": 84}, {"o": "stats_metrics", "r": "reproducibility_norm", "v": "conditional", "n": 56}, {"o": "stats_metrics", "r": "reproducibility_norm", "v": "mixed", "n": 4}, {"o": "stats_metrics", "r": "reproducibility_norm", "v": "negative", "n": 980}, {"o": "stats_metrics", "r": "reproducibility_norm", "v": "positive", "n": 22}, {"o": "stats_metrics", "r": "reproducibility_norm", "v": "uncertain", "n": 19}, {"o": "stats_metrics", "r": "robustness_norm", "v": "conditional", "n": 503}, {"o": "stats_metrics", "r": "robustness_norm", "v": "mixed", "n": 20}, {"o": "stats_metrics", "r": "robustness_norm", "v": "negative", "n": 2942}, {"o": "stats_metrics", "r": "robustness_norm", "v": "positive", "n": 217}, {"o": "stats_metrics", "r": "robustness_norm", "v": "uncertain", "n": 232}, {"o": "stats_metrics", "r": "statistical_identifiability", "v": "conditional", "n": 181}, {"o": "stats_metrics", "r": "statistical_identifiability", "v": "mixed", "n": 17}, {"o": "stats_metrics", "r": "statistical_identifiability", "v": "negative", "n": 3242}, {"o": "stats_metrics", "r": "statistical_identifiability", "v": "positive", "n": 42}, {"o": "stats_metrics", "r": "statistical_identifiability", "v": "uncertain", "n": 85}, {"o": "theory", "r": "claim_evidence_match", "v": "conditional", "n": 799}, {"o": "theory", "r": "claim_evidence_match", "v": "mixed", "n": 37}, {"o": "theory", "r": "claim_evidence_match", "v": "negative", "n": 3023}, {"o": "theory", "r": "claim_evidence_match", "v": "positive", "n": 174}, {"o": "theory", "r": "claim_evidence_match", "v": "uncertain", "n": 427}, {"o": "theory", "r": "confound_hypothesis", "v": "conditional", "n": 213}, {"o": "theory", "r": "confound_hypothesis", "v": "mixed", "n": 5}, {"o": "theory", "r": "confound_hypothesis", "v": "negative", "n": 1347}, {"o": "theory", "r": "confound_hypothesis", "v": "positive", "n": 77}, {"o": "theory", "r": "confound_hypothesis", "v": "uncertain", "n": 103}, {"o": "theory", "r": "construct_validity", "v": "conditional", "n": 99}, {"o": "theory", "r": "construct_validity", "v": "mixed", "n": 11}, {"o": "theory", "r": "construct_validity", "v": "negative", "n": 859}, {"o": "theory", "r": "construct_validity", "v": "positive", "n": 59}, {"o": "theory", "r": "construct_validity", "v": "uncertain", "n": 64}, {"o": "theory", "r": "cost_benefit", "v": "conditional", "n": 282}, {"o": "theory", "r": "cost_benefit", "v": "mixed", "n": 20}, {"o": "theory", "r": "cost_benefit", "v": "negative", "n": 1191}, {"o": "theory", "r": "cost_benefit", "v": "positive", "n": 211}, {"o": "theory", "r": "cost_benefit", "v": "uncertain", "n": 145}, {"o": "theory", "r": "design_justification", "v": "conditional", "n": 1537}, {"o": "theory", "r": "design_justification", "v": "mixed", "n": 63}, {"o": "theory", "r": "design_justification", "v": "negative", "n": 10334}, {"o": "theory", "r": "design_justification", "v": "positive", "n": 863}, {"o": "theory", "r": "design_justification", "v": "uncertain", "n": 721}, {"o": "theory", "r": "fair_comparison", "v": "conditional", "n": 310}, {"o": "theory", "r": "fair_comparison", "v": "mixed", "n": 20}, {"o": "theory", "r": "fair_comparison", "v": "negative", "n": 1825}, {"o": "theory", "r": "fair_comparison", "v": "positive", "n": 88}, {"o": "theory", "r": "fair_comparison", "v": "uncertain", "n": 147}, {"o": "theory", "r": "merit_recognition", "v": "conditional", "n": 689}, {"o": "theory", "r": "merit_recognition", "v": "mixed", "n": 44}, {"o": "theory", "r": "merit_recognition", "v": "negative", "n": 1835}, {"o": "theory", "r": "merit_recognition", "v": "positive", "n": 5745}, {"o": "theory", "r": "merit_recognition", "v": "uncertain", "n": 249}, {"o": "theory", "r": "novelty_standard", "v": "conditional", "n": 865}, {"o": "theory", "r": "novelty_standard", "v": "mixed", "n": 195}, {"o": "theory", "r": "novelty_standard", "v": "negative", "n": 7530}, {"o": "theory", "r": "novelty_standard", "v": "positive", "n": 855}, {"o": "theory", "r": "novelty_standard", "v": "uncertain", "n": 334}, {"o": "theory", "r": "presentation_trust", "v": "conditional", "n": 188}, {"o": "theory", "r": "presentation_trust", "v": "mixed", "n": 23}, {"o": "theory", "r": "presentation_trust", "v": "negative", "n": 2855}, {"o": "theory", "r": "presentation_trust", "v": "positive", "n": 100}, {"o": "theory", "r": "presentation_trust", "v": "uncertain", "n": 70}, {"o": "theory", "r": "reproducibility_norm", "v": "conditional", "n": 63}, {"o": "theory", "r": "reproducibility_norm", "v": "mixed", "n": 6}, {"o": "theory", "r": "reproducibility_norm", "v": "negative", "n": 1198}, {"o": "theory", "r": "reproducibility_norm", "v": "positive", "n": 20}, {"o": "theory", "r": "reproducibility_norm", "v": "uncertain", "n": 33}, {"o": "theory", "r": "robustness_norm", "v": "conditional", "n": 1252}, {"o": "theory", "r": "robustness_norm", "v": "mixed", "n": 32}, {"o": "theory", "r": "robustness_norm", "v": "negative", "n": 4363}, {"o": "theory", "r": "robustness_norm", "v": "positive", "n": 324}, {"o": "theory", "r": "robustness_norm", "v": "uncertain", "n": 733}, {"o": "theory", "r": "statistical_identifiability", "v": "conditional", "n": 209}, {"o": "theory", "r": "statistical_identifiability", "v": "mixed", "n": 15}, {"o": "theory", "r": "statistical_identifiability", "v": "negative", "n": 2142}, {"o": "theory", "r": "statistical_identifiability", "v": "positive", "n": 44}, {"o": "theory", "r": "statistical_identifiability", "v": "uncertain", "n": 91}], "obj_extras": {"baselines_ablations": {"n": 26207, "with_improvement": 24074, "reviewer_explicit": 5091}, "clarity": {"n": 26714, "with_improvement": 23644, "reviewer_explicit": 5869}, "compute_cost": {"n": 23632, "with_improvement": 20485, "reviewer_explicit": 4577}, "empirical_scope": {"n": 61563, "with_improvement": 52155, "reviewer_explicit": 11161}, "method_design": {"n": 39793, "with_improvement": 32818, "reviewer_explicit": 6789}, "novelty": {"n": 9409, "with_improvement": 5977, "reviewer_explicit": 1554}, "problem_framing": {"n": 7272, "with_improvement": 5996, "reviewer_explicit": 1274}, "related_work": {"n": 9033, "with_improvement": 7915, "reviewer_explicit": 1944}, "reproducibility": {"n": 6550, "with_improvement": 5889, "reviewer_explicit": 1396}, "robustness_sensitivity": {"n": 15045, "with_improvement": 13535, "reviewer_explicit": 2878}, "stats_metrics": {"n": 31251, "with_improvement": 27041, "reviewer_explicit": 6020}, "theory": {"n": 38502, "with_improvement": 32463, "reviewer_explicit": 7115}}, "confidence": {"object_share_ge_075": 0.9447, "reasoning_share_ge_075": 0.8787, "object_mean_sim": 0.821, "reasoning_mean_sim": 0.7957}, "sample_reviews": [{"review_id": "004ApSUcMS", "paper_id": "suZoTnu0qb", "paper_title": "Enhancing the Medical Context-Awareness Ability of LLMs via Multifaceted Self-Refinement Learning", "decision": "Reject", "summary": "The reviewer conducts a decomposition audit, focusing on whether each design choice (facets, attributes, KD) is empirically justified and isolable. The logic posits that unexplained design parameters and confounded components constitute an unsupported contribution, regardless of reported performance metrics.", "units": [{"unit_index": 0, "inspected_object": "The paper's notation system and formatting consistency.", "observation": "The reviewer finds the notation 'difficult to follow' due to inconsistencies in subscript/superscript usage and a lack of a unified notation table.", "reasoning": "Clear notation is a prerequisite for verifying the method's correctness and reproducibility; ambiguity here creates a barrier to evaluating the technical soundness of the proposed pipeline.", "judgment": "The presentation quality is insufficiently rigorous, constituting a formal weakness that hinders scrutiny.", "valence": "negative", "suggested_improvement": "Provide a comprehensive notation table and unify subscript/superscript conventions throughout the manuscript.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8827066421508789, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8017251491546631}, {"unit_index": 1, "inspected_object": "The selection of three specific facets (decision-making, communication, safety) for the query generation framework.", "observation": "The reviewer notes that the choice of these three facets is presented without empirical justification or derivation from domain expertise.", "reasoning": "Design choices regarding categorical inputs should be grounded in data (e.g., error taxonomies) or clinical input rather than appearing arbitrary; without this grounding, it is unclear if the selected facets are independently necessary or merely redundant partitions.", "judgment": "The theoretical basis for the core component design is weak and lacks sufficient evidentiary support.", "valence": "negative", "suggested_improvement": "Conduct ablations to measure the marginal gain of each facet individually and in combination, or provide external evidence (e.g., clinician interviews, baseline error analysis) justifying the selection.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8102061152458191, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7737475633621216}, {"unit_index": 2, "inspected_object": "The specific count of seven attributes used in the attribute-conditioned query generation.", "observation": "The reviewer identifies the number 'seven' as an unmotivated hyperparameter/design choice, questioning its origin.", "reasoning": "Specific numerical parameters in method design require narrative or empirical justification to distinguish them from arbitrary conveniences; their absence suggests the design may not be fully accounted for.", "judgment": "A specific design parameter appears arbitrary and unsupported by explanation.", "valence": "negative", "suggested_improvement": "Explain the rationale behind selecting exactly seven attributes, citing either domain analysis, prior work, or ablation results.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7998681664466858, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8289949297904968}, {"unit_index": 3, "inspected_object": "The knowledge distillation (KD) component within the MuSeR pipeline.", "observation": "The reviewer observes that the pipeline combines MuSeR with KD but does not isolate their respective contributions.", "reasoning": "If the performance gains are driven primarily by KD rather than the novel MuSeR components, the contribution of the proposed method is confounded; decomposition is necessary to verify independent efficacy.", "judgment": "The isolation of the novel method's contribution is unclear due to the compound treatment with standard distillation techniques.", "valence": "negative", "suggested_improvement": "Perform an ablation separating the MuSeR contribution from the KD contribution, including reporting on compute and cost implications.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8319166898727417, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7746797800064087}]}, {"review_id": "00bMH9Bj9i", "paper_id": "PhEHuo7oMm", "paper_title": "ReST-KV: Robust KV Cache Eviction with Layer-wise Output Reconstruction and Spatial-Temporal Smoothing", "decision": "Accept (Poster)", "summary": "The reviewer affirms the principled reframing of eviction and broad experimental scope but critically questions the theoretical justification of the greedy heuristic, the parsimony of the spatial-temporal smoothing mechanism, and the clarity of the method's scope. The review emphasizes argumentation and clarity over additional empirical results.", "units": [{"unit_index": 0, "inspected_object": "Reframing eviction as output reconstruction", "observation": "The reviewer explicitly praises this approach as more principled and robust than key-query similarity.", "reasoning": "The reviewer holds a normative view that defensible eviction methods should be grounded in models of downstream effects rather than proxy heuristics, viewing this framing as genuine merit.", "judgment": "Positive assessment of the core conceptual framing.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8159016370773315, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7810586094856262}, {"unit_index": 1, "inspected_object": "Experimental scope and ablation study", "observation": "The method is tested on multiple models, benchmarks, and cache budgets; the ablation shows effectiveness.", "reasoning": "The reviewer views breadth of coverage across diverse settings as a proxy for robustness of evidence and acknowledges the ablation as informative, though without commenting on statistical rigor or individual experiment quality.", "judgment": "Minimal endorsement based on empirical breadth.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8259792327880859, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7882845401763916}, {"unit_index": 2, "inspected_object": "Greedy removal justification via local linearity assumptions", "observation": "The paper justifies greedy one-token-at-a-time removal using local linearity assumptions despite softmax being highly non-linear.", "reasoning": "The reviewer identifies a tension between the paper's claim to model attention redistribution and its heuristic implementation; removing one token can cause drastic, non-local redistribution, making the local linearity assumption strong and potentially faith-based rather than proven.", "judgment": "Critical concern about theoretical grounding matching practical behavior.", "valence": "negative", "suggested_improvement": "A brief discussion of why this approximation holds in practice would be beneficial.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8620104193687439, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8011177778244019}, {"unit_index": 3, "inspected_object": "Adaptive Window-Based Spatial Smoothing mechanism", "observation": "The mechanism computes average indices, derives shifts, and defines new parameters, appearing complex with interacting hyperparameters.", "reasoning": "The reviewer judges the mechanism as overly-engineered because it introduces geometric intuitions and complexity without a clear underlying principle or simple intuitive explanation, raising parsimony concerns.", "judgment": "Negative aesthetic and parsimony judgment regarding clarity and naturalness.", "valence": "negative", "suggested_improvement": "Provide a simple and intuitive explanation of the mechanism.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7805021405220032, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7197872996330261}, {"unit_index": 4, "inspected_object": "Sensitivity to hyperparameter β", "observation": "The reviewer questions whether performance is brittle to the setting of β.", "reasoning": "This probes whether the 'overly-engineered' spatial-temporal smoothing mechanism is actually earning its complexity; high sensitivity would confirm fragility, while robustness would partially rehabilitate it.", "judgment": "Uncertain/Conditional evaluation of mechanism robustness.", "valence": "conditional", "suggested_improvement": "Analyze sensitivity to hyperparameter β.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8948906660079956, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8056490421295166}, {"unit_index": 5, "inspected_object": "Scope of application: Prefill vs. Decoding", "observation": "It is unclear whether ReST-KV evicts only from the prompt during prefill or also from generated tokens during decoding.", "reasoning": "If the method only operates at prefill, its applicability to long-generation scenarios where memory constraints are highest (decoding phase) is limited, contradicting claims about long sequences and decoding latency.", "judgment": "Factual clarification needed to assess scope of applicability.", "valence": "uncertain", "suggested_improvement": "Clarify whether eviction occurs during decoding.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7512937784194946, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7594908475875854}, {"unit_index": 6, "inspected_object": "Comparison with LaCache baseline", "observation": "The paper lacks comparison with LaCache, an ICML'25 paper.", "reasoning": "State-of-the-art claims are contingent on evaluation against the most recent competitors; missing this baseline is a gap in empirical completeness that could be exploited by readers.", "judgment": "Mild criticism regarding empirical completeness and literature currency.", "valence": "negative", "suggested_improvement": "Compare results with LaCache.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8119480013847351, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8153015971183777}]}, {"review_id": "00fI9xBvg5", "paper_id": "gLPxpqYRqH", "paper_title": "MoDA: Modulation Adapter for Fine-Grained Visual Understanding in Instructional MLLMs", "decision": "Reject", "summary": "The reviewer systematically challenges the paper's robustness, novelty, and verifiability through counterfactual testing and specific empirical criticisms, concluding that fundamental flaws outweigh reported gains.", "units": [{"unit_index": 0, "inspected_object": "MMBench-Cn performance with SigLIP-S2", "observation": "Performance drops from 68.0 to 63.6.", "reasoning": "The reviewer treats this magnitude of regression as decisive evidence that the method is brittle and fails to generalize cross-attention modulation knowledge beyond its training linguistic space, contradicting the 'general-purpose' framing.", "judgment": "The method lacks robustness and generalization capability on non-English benchmarks.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7281623482704163, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7969558238983154}, {"unit_index": 1, "inspected_object": "Cross-attention mechanism novelty", "observation": "The mechanism resembles instruction-conditioned attention found in InstructBLIP and Q-Former.", "reasoning": "Applying an established mechanism in a new context constitutes incremental refinement rather than a novel contribution, making the paper's positioning an overclaim relative to its architectural delta.", "judgment": "The architectural contribution is not novel.", "valence": "negative", "suggested_improvement": "Clearly articulate what is genuinely novel about MoDA relative to InstructBLIP and Q-Former.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.861416220664978, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8346332907676697}, {"unit_index": 2, "inspected_object": "Soft modulation vs. hard sparsity distinction", "observation": "The method uses soft re-weighting for channel encoding.", "reasoning": "Soft re-weighting cannot fully disentangle mixed signals (e.g., relevant object features vs. irrelevant background textures); only hard, instruction-conditioned feature selectors enforcing channel sparsity would suffice for true disentanglement.", "judgment": "The mechanism is insufficient for the claimed level of disentanglement.", "valence": "negative", "suggested_improvement": "Implement a hard-sparsity mechanism or provide a convincing argument for why soft modulation suffices.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8003091812133789, "reasoning_key": "design_justification", "reasoning_sim": 0.8123684525489807}, {"unit_index": 3, "inspected_object": "Choice of LLM layers for token extraction", "observation": "The paper uses tokens from initial layers of the LLM.", "reasoning": "This choice appears driven by convenience rather than optimality, lacking justification against using later, more semantically rich layers, which undermines confidence in the design decision.", "judgment": "The design choice is suspicious and unjustified.", "valence": "negative", "suggested_improvement": "Perform ablation studies on layer choice to validate the selection of initial layers.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8076709508895874, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7788752913475037}, {"unit_index": 4, "inspected_object": "Cross-attention depth ablation", "observation": "Depth comparison was conducted on the Linear MLP variant, not the superior Cross-Attention variant.", "reasoning": "Conclusions about optimal depth for one architecture cannot be validly transferred to another structurally different architecture, creating a logical gap in the ablation strategy.", "judgment": "The ablation strategy is methodologically flawed and incomplete.", "valence": "negative", "suggested_improvement": "Conduct cross-attention depth ablations on the Cross-Attention variant.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7800155282020569, "reasoning_key": "design_justification", "reasoning_sim": 0.7723008990287781}, {"unit_index": 5, "inspected_object": "Related work coverage", "observation": "Missing citations for Instruction-Guided Fusion, MoReS/LLaVA Steering, and EAGLE.", "reasoning": "These works are highly relevant precedents; their absence weakens novelty claims and indicates a deficiency in scholarly positioning and related-work section.", "judgment": "The literature review is deficient and novelty claims are weakened.", "valence": "negative", "suggested_improvement": "Engage with the cited precedents and explain how MoDA differs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.7854881882667542, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7475450038909912}, {"unit_index": 6, "inspected_object": "Availability of code and weights", "observation": "Code and weights are not provided.", "reasoning": "Specific claims regarding computational efficiency (<1% FLOPs) and architectural gains render key findings unverifiable without independent replication, shifting the burden of proof unmet.", "judgment": "The findings lack verifiability and trustworthiness.", "valence": "negative", "suggested_improvement": "Release code and weights.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8277276754379272, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7883766293525696}]}, {"review_id": "00v5D69eTB", "paper_id": "iQwMr0tuJC", "paper_title": "GOAT: A Training Framework for Goal-Oriented Agent with Tools", "decision": null, "summary": "The reviewer performs a novelty and recency audit, judging the paper's contributions as insufficient because the pipeline resembles prior graph-based methods and the experimental baselines are temporally outdated. The review emphasizes comparative positioning over internal validity, concluding that without cross-framework comparisons and modern baselines, the evidence does not support strong claims of superiority.", "units": [{"unit_index": 0, "inspected_object": "The 'call-first' generation strategy (generating a query from an executable API path)", "observation": "The reviewer identifies this as a strength, noting it is more constrained and reliable than the reverse approach.", "reasoning": "Process-oriented evaluation: the design's internal logic is sound and reduces ambiguity in tool selection.", "judgment": "Positive assessment of the specific design choice as a robust engineering decision.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7750605344772339, "reasoning_key": "design_justification", "reasoning_sim": 0.741534411907196}, {"unit_index": 1, "inspected_object": "Novelty of the core pipeline relative to prior art (ToolFlow, Magnet, ToolDial)", "observation": "The paper's own Table 1 lists contemporaneous works using graph-based synthesis; the reviewer notes these are highly similar.", "reasoning": "If prior art uses the same graph-based approach for agent data synthesis, the claimed contribution is incremental rather than fundamental. The reviewer dismisses the filtering steps as pragmatic engineering, not novel research.", "judgment": "Negative assessment of novelty; the method is not new enough to constitute a primary contribution.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8320836424827576, "reasoning_key": "design_justification", "reasoning_sim": 0.8625689744949341}, {"unit_index": 2, "inspected_object": "Experimental baselines (text-davinci-003, Llama2-13B/Vicuna-13B)", "observation": "Baselines include models from 2022–2023, which are considered outdated for a late 2025 submission.", "reasoning": "Comparisons must reflect the state of the art at submission time. Using older models invalidates the experimental claims of superiority because they do not represent current competitive standards.", "judgment": "Negative assessment of evidence quality; the experimental validation is weak due to temporal obsolescence of baselines.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8691039681434631, "reasoning_key": "fair_comparison", "reasoning_sim": 0.824959933757782}, {"unit_index": 3, "inspected_object": "Cross-framework comparison (training on ToolFlow-generated data vs. GOATBench)", "observation": "The review lacks a direct comparison where other frameworks generate data evaluated on the same benchmark.", "reasoning": "A cross-framework test is required to isolate data quality superiority. Without this, it is unclear if GOAT's performance gains stem from better data or other factors.", "judgment": "Negative assessment of completeness; the absence of this experiment is a decisive gap in proving data generation superiority.", "valence": "negative", "suggested_improvement": "Perform a cross-framework comparison by training on ToolFlow-generated data and evaluating on GOATBench.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8195506930351257, "reasoning_key": "design_justification", "reasoning_sim": 0.820514440536499}, {"unit_index": 4, "inspected_object": "Open-source release of the GOATBench benchmark", "observation": "The benchmark code/data has not been open-sourced.", "reasoning": "Community norm dictates that benchmarks should be public to ensure reproducibility and utility. Absence limits the artifact's contribution value.", "judgment": "Negative assessment of contribution utility; lack of release diminishes the benchmark's impact.", "valence": "negative", "suggested_improvement": "Open-source the GOATBench benchmark code and data.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8377282023429871, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8114879131317139}]}, {"review_id": "00vC2EXTiB", "paper_id": "OMjcX1Z6Uu", "paper_title": "Markovian Transformers for Informative Language Modeling", "decision": "Accept (Poster)", "summary": "The reviewer provides qualified endorsement based on conceptual elegance and technical soundness but raises mechanistic skepticism due to insufficient causal attribution of gains, missing counterfactual baselines, lack of comparison to established methods, and incomplete scholarly positioning.", "units": [{"unit_index": 0, "inspected_object": "The causal attribution of empirical gains to the Markovian bottleneck versus other method components.", "observation": "The reviewer finds that the paper bundles multiple interventions (Markovian bottleneck, actor–reward coupling, within-batch normalization, reward design) without decomposing their individual contributions.", "reasoning": "A novel method must be shown to contribute beyond its constituent parts; conflating these interventions prevents determining if the headline result is driven by the novel constraint or mundane training details.", "judgment": "Insufficient evidence to establish that the Markovian bottleneck is the active ingredient for the observed performance gains.", "valence": "negative", "suggested_improvement": "Perform ablation studies to isolate the contribution of each component.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8322706818580627, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8073307871818542}, {"unit_index": 1, "inspected_object": "The absence of a counterfactual baseline where the model predicts from (Question + CoT) under the same RL training.", "observation": "The reviewer notes the missing baseline that retains the CoT and RL training but removes the Markovian bottleneck.", "reasoning": "To verify the Markovian constraint itself drives improvements, one must compare against the closest possible alternative lacking only that constraint; if (Question + CoT) performs equally well, the bottleneck is not the distinctive intervention.", "judgment": "Critical experimental control is missing, leaving uncertainty about the specific causal role of the Markovian property.", "valence": "negative", "suggested_improvement": "Add a baseline using (Question + CoT) with identical RL training to isolate the effect of the bottleneck.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7532690167427063, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8225232362747192}, {"unit_index": 2, "inspected_object": "The generalization of the 'informativeness' objective to knowledge-grounded tasks.", "observation": "The reviewer questions whether the informativeness criterion, effective for deductive/mathematical tasks like GSM8K, is sufficient for factual QA or MMLU where world knowledge may not be fully expressible in bounded CoTs.", "reasoning": "Deductive tasks have a closed reasoning space, while knowledge tasks have an open one; the implicit assumption is that the learning signal becomes ill-defined when answers depend on external facts rather than logical derivation.", "judgment": "Potential domain boundary limitation regarding the applicability of the proposed objective.", "valence": "conditional", "suggested_improvement": "Discuss or test the method's performance on knowledge-grounded tasks to define scope boundaries.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7932456135749817, "reasoning_key": "design_justification", "reasoning_sim": 0.7812148332595825}, {"unit_index": 3, "inspected_object": "Comparison against established process-supervised or fine-tuned CoT methods.", "observation": "The reviewer identifies a lack of comparison against baselines such as STaR.", "reasoning": "Novel methods should demonstrate a real advantage over existing techniques in the same subfield; without such positioning, it is unclear if the contribution is incremental or substantive compared to simpler methods.", "judgment": "Contribution cannot be assessed as more than incremental without comparative benchmarking.", "valence": "negative", "suggested_improvement": "Compare results against STaR and other established reasoning-enhancement techniques.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.831987202167511, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8382165431976318}, {"unit_index": 4, "inspected_object": "Scholarly positioning relative to prior work on CoT faithfulness.", "observation": "The reviewer notes two missing citations: Paul et al., 2024 (faithfulness measurement) and Ferreira et al., 2025 (causal attribution for reward hacking).", "reasoning": "The paper's claims about load-bearing CoTs overlap with existing literature on CoT faithfulness; acknowledging this work is necessary to differentiate the current contribution.", "judgment": "Incomplete scholarly engagement with relevant prior attempts to address CoT faithfulness.", "valence": "negative", "suggested_improvement": "Cite and discuss Paul et al. (2024) and Ferreira et al. (2025) to position the work within the faithfulness literature.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8275965452194214, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7768799662590027}, {"unit_index": 5, "inspected_object": "The conceptual novelty and technical soundness of the Markovian bottleneck and GRPO-style algorithm.", "observation": "The reviewer acknowledges the idea as novel, conceptually elegant, and technically sound, noting the formalization of the Markovian LM and the close reading of the method section.", "reasoning": "The core mechanism is intelligible and the implementation appears rigorous, providing a foundation for potential value despite current evidentiary gaps.", "judgment": "Conceptual merit and technical execution are high, supporting conditional endorsement.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8682869672775269, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7782760262489319}, {"unit_index": 6, "inspected_object": "The probing analyses including perturbation sensitivity and cross-model transfer.", "observation": "The reviewer treats the perturbation sensitivity and cross-model transfer results as thoughtful probing analyses rather than mere accuracy reporting.", "reasoning": "These analyses demonstrate an attempt to show causal reliance on the CoT, which adds value beyond simple performance metrics.", "judgment": "Empirical evidence is promising and demonstrates good experimental design in probing causal reliance.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8290086984634399, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7086417078971863}]}, {"review_id": "00wKhWEVu5", "paper_id": "rF8iG1QW7Y", "paper_title": "WatermarkLab: A Comprehensive Framework for Robust Image Watermarks Benchmarking and Development", "decision": "Reject", "summary": "The reviewer critically evaluates the framework's breadth and metric design positively while demanding deeper explanatory depth for empirical findings, broader domain generalization, inclusion of computational costs, and more comprehensive attack coverage to meet practitioner and theoretical standards.", "units": [{"unit_index": 0, "inspected_object": "Integration of diverse watermarking paradigms and extensive attack taxonomy", "observation": "The framework integrates diverse watermarking paradigms (zero-bit, multi-bit, IGW, PGW, robust reversible) and an extensive attack taxonomy spanning compression, color, geometric, noise, diffusion-based, and filtering attacks.", "reasoning": "The reviewer recognizes the logistical effort involved in unifying such heterogeneous methods under one interface, viewing this breadth as a genuine contribution to benchmarking infrastructure.", "judgment": "Positive evaluation of the framework's comprehensive scope and engineering output.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7458674907684326, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7917759418487549}, {"unit_index": 1, "inspected_object": "Proposed metrics (TPR@x%FPR and RQ-AUC)", "observation": "The metrics TPR@x%FPR and RQ-AUC are identified as principled methodological tools.", "reasoning": "The reviewer evaluates the paper's value through its transferable intellectual artifacts, noting that these metrics transcend the specific framework and serve as generalizable tools for the field.", "judgment": "Positive evaluation of the methodological contribution and its utility beyond the immediate system.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8570533990859985, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8095167875289917}, {"unit_index": 2, "inspected_object": "IGW/PGW performance gap and explanatory depth", "observation": "GaussianShading shows superior cumulative RQ-AUC compared to StegaStamp; authors attribute this to stronger implicit correlation with image content, but no formal analysis or information-theoretic treatment is provided.", "reasoning": "The reviewer holds that a benchmarking paper should not merely report differences but provide frameworks for understanding why those differences exist; without this, the benchmark becomes a black-box leaderboard rather than a scientific instrument.", "judgment": "Negative evaluation regarding the lack of explanatory depth for empirical findings.", "valence": "negative", "suggested_improvement": "Provide formal analysis or information-theoretic treatment to explain performance differences.", "support_status": "mixed", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8453277945518494, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7398982644081116}, {"unit_index": 3, "inspected_object": "VINE crop/cropout asymmetry", "observation": "VINE exhibits a surprising asymmetry: high vulnerability to boundary crops (TPR 0.01 at 10% boundary crop) but high robustness to cropouts (90% cropout tolerated).", "reasoning": "The reviewer treats surprising empirical results as diagnostic opportunities to reveal insights about watermark embedding strategies and inform better method design, rather than accepting them as mere anomalies.", "judgment": "Questioning the interpretive value of the empirical pattern; seeking mechanistic explanation.", "valence": "conditional", "suggested_improvement": "Analyze what the asymmetry reveals about watermark embedding strategies to inform future method design.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7268260717391968, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7756218910217285}, {"unit_index": 4, "inspected_object": "Missing computational cost data", "observation": "The paper lacks data on training time, inference latency, memory usage, and model sizes.", "reasoning": "The reviewer assumes that a benchmarking framework's purpose includes facilitating method selection for real-world deployment; without computational cost data, practitioners cannot evaluate the three-way tradeoff including efficiency.", "judgment": "Critical negative evaluation regarding the utility of the benchmark for practical deployment decisions.", "valence": "negative", "suggested_improvement": "Include training time, inference latency, memory, and model size data.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.817707359790802, "reasoning_key": "cost_benefit", "reasoning_sim": 0.789945662021637}, {"unit_index": 5, "inspected_object": "Domain generalization of experiments", "observation": "All experiments use MS-COCO, with no testing on medical, satellite, or artistic imagery.", "reasoning": "The reviewer assumes real-world watermarking deployment will encounter domain shift; evaluating methods only on a single domain fails to capture potential robustness issues under distributional change.", "judgment": "Negative evaluation regarding the limited scope of experimental validation.", "valence": "negative", "suggested_improvement": "Test methods on diverse domains such as medical, satellite, or artistic imagery to assess generalization.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8524463176727295, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7925242185592651}, {"unit_index": 6, "inspected_object": "Completeness of attack coverage", "observation": "The framework covers 34 attacks but omits adversarial perturbations targeting watermark removal, overwriting attacks, inpainting beyond diffusion, and physical-world attacks beyond print-capture.", "reasoning": "The reviewer holds an external standard for a complete watermarking threat model; the omission suggests the current coverage does not fully satisfy the claim of comprehensiveness against adaptive adversaries.", "judgment": "Negative evaluation regarding the completeness of the threat model.", "valence": "negative", "suggested_improvement": "Expand attack categories to include adversarial perturbations, overwriting, advanced inpainting, and physical-world attacks.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8000324964523315, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7381040453910828}, {"unit_index": 7, "inspected_object": "Evaluation against adaptive adversaries", "observation": "The review questions whether predefined transformation attacks are sufficient for assessing true robustness.", "reasoning": "The reviewer considers white-box adversaries optimizing watermark removal to be the gold standard for robustness evaluation; static attacks may underestimate vulnerabilities.", "judgment": "Negative evaluation regarding the sufficiency of the robustness evaluation methodology.", "valence": "negative", "suggested_improvement": "Evaluate robustness against adaptive white-box adversaries who optimize for watermark removal.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8334038257598877, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7572839260101318}, {"unit_index": 8, "inspected_object": "Potential for ensemble methods", "observation": "The reviewer notes the framework's infrastructure could enable research into combining methods.", "reasoning": "The reviewer sees potential for the framework to support new research directions, specifically exploring if combining methods yields complementary robustness properties.", "judgment": "Positive forward-looking assessment of the framework's research potential.", "valence": "positive", "suggested_improvement": "Investigate ensemble approaches to leverage complementary robustness from different methods.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8040319681167603, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7983928322792053}]}, {"review_id": "010PY9B9Xh", "paper_id": "QASlwMxdeP", "paper_title": "DeblurSDI: Blind Image Deblurring Using Self-diffusion", "decision": null, "summary": "The reviewer confirms the method's technical execution but raises three practical concerns regarding computational efficiency, hyperparameter robustness, and real-world generalization based on synthetic benchmarks.", "units": [{"unit_index": 0, "inspected_object": "Computational cost of the diffusion process (S=200 inner optimization loops)", "observation": "The method uses S=200 inner optimization loops per diffusion step.", "reasoning": "This parameter implies high computational cost, which may lead to long processing times and hinder real-time applicability compared to single-forward-pass methods.", "judgment": "The method's practical deployment is constrained by its inference speed.", "valence": "negative", "suggested_improvement": "Benchmark inference time against single-forward-pass methods and consider acceleration strategies.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8315638303756714, "reasoning_key": "cost_benefit", "reasoning_sim": 0.821556568145752}, {"unit_index": 1, "inspected_object": "Hyperparameter sensitivity and selection methodology", "observation": "Performance depends on T, S, learning rate, and L1 weight, with a reported sensitivity analysis.", "reasoning": "Finding optimal combinations may still be a challenge in practical applications, treating hyperparameter tuning as a practical burden rather than a demonstrated stability feature.", "judgment": "The method lacks demonstrated robustness to suboptimal settings, posing a usability risk.", "valence": "negative", "suggested_improvement": "Explain the selection methodology and quantify degradation with suboptimal settings.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.90628582239151, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8441179394721985}, {"unit_index": 2, "inspected_object": "Real-world generalization capability", "observation": "Experiments rely exclusively on synthetic blur datasets.", "reasoning": "Real-world blurs are more complex, non-linear, and spatially varying; performance on synthetic data remains to be further verified for real-world scenarios.", "judgment": "The validity of the method for real-world image deblurring is unverified.", "valence": "negative", "suggested_improvement": "Test on real-world images with such degradations.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8575319051742554, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7109146118164062}]}, {"review_id": "010tKaXCZw", "paper_id": "4vXNp4wujY", "paper_title": "AMS-Quant: Adaptive Mantissa Sharing for Floating-Point Quantization", "decision": "Reject", "summary": "The reviewer identifies novelty in mantissa sharing as a key strength but critiques deployment portability and missing computational cost reporting. They pose conditional questions on scalability, composability, and generality to map the method's boundaries.", "units": [{"unit_index": 0, "inspected_object": "Novelty of mantissa sharing compared to prior exponent/scaling factor sharing", "observation": "Existing works explored sharing exponent bits or scaling factors, but mantissa sharing is surprisingly unexplored.", "reasoning": "The paper fills a genuine gap in the bit-sharing design space, violating the expectation that such strategies would be exhausted.", "judgment": "The idea is novel and contributes a valuable conceptual insight.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8179042339324951, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7419861555099487}, {"unit_index": 1, "inspected_object": "Clarity of writing and adaptive search mechanism effectiveness", "observation": "The paper presents clear writing and an effective adaptive search mechanism.", "reasoning": "These elements support the paper's overall quality and utility.", "judgment": "The presentation and core mechanism are strengths.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8232616782188416, "reasoning_key": "merit_recognition", "reasoning_sim": 0.759558379650116}, {"unit_index": 2, "inspected_object": "Reporting of computational cost for adaptive mantissa sharing", "observation": "Authors did not report search time for the adaptive mantissa sharing algorithm.", "reasoning": "Even if offline, ascertaining computational cost is essential for practical feasibility assessment.", "judgment": "The lack of reported search time is a weakness limiting the evaluation of practical deployment.", "valence": "negative", "suggested_improvement": "Report the search time for the adaptive mantissa sharing algorithm.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8246408700942993, "reasoning_key": "cost_benefit", "reasoning_sim": 0.828037440776825}, {"unit_index": 3, "inspected_object": "Deployment portability regarding non-integer bit-widths", "observation": "Non-integer bit-widths based on packing/unpacking cannot integrate with inference frameworks like TensorRT; TPU/NPU deployment may need extra effort.", "reasoning": "Quantization methods should ideally integrate with standard inference stacks; the current approach limits ecosystem compatibility.", "judgment": "The method's portability and integration potential are weak due to hardware/framework constraints.", "valence": "negative", "suggested_improvement": "Address integration with standard inference frameworks like TensorRT or clarify deployment efforts for other hardware.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7891870737075806, "reasoning_key": "design_justification", "reasoning_sim": 0.7848262190818787}, {"unit_index": 4, "inspected_object": "Scalability of search time for large models", "observation": "Search time behavior for large models is unknown.", "reasoning": "Understanding scalability is necessary to determine if offline costs remain tractable as model size grows.", "judgment": "Uncertainty remains about the method's scalability to larger models.", "valence": "conditional", "suggested_improvement": "Investigate and report search time scaling behavior for large models.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8358707427978516, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8134592771530151}, {"unit_index": 5, "inspected_object": "Composability with GPTQ or AWQ", "observation": "It is unclear if the method can be layered on top of existing post-training quantization techniques like GPTQ or AWQ.", "reasoning": "A good quantization method should be a building block that works with the broader ecosystem.", "judgment": "Uncertainty exists regarding the method's composability with prior approaches.", "valence": "conditional", "suggested_improvement": "Evaluate or discuss the method's compatibility with GPTQ or AWQ.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7504162788391113, "reasoning_key": "design_justification", "reasoning_sim": 0.7543557286262512}, {"unit_index": 6, "inspected_object": "Extension to activation quantization", "observation": "The paper focuses on weight quantization; application to activations is not addressed.", "reasoning": "Applying mantissa sharing to activations would broaden the contribution's scope and generality.", "judgment": "The generalizability of the core idea beyond weights is unproven.", "valence": "conditional", "suggested_improvement": "Explore or discuss the extension of mantissa sharing to activation quantization.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "theory", "object_sim": 0.8212677240371704, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7302141189575195}]}, {"review_id": "015Dp5FNEo", "paper_id": "xo4MHUjiDL", "paper_title": "MolReasoner: Toward Effective and Interpretable Reasoning for Molecular LLMs", "decision": null, "summary": "The reviewer systematically challenges the paper's novelty, empirical rigor, and theoretical depth. They argue that the framework is a familiar application of existing methods (R1), the reward design is flawed, baselines are weak, ablations are insufficient, and theoretical support is absent.", "units": [{"unit_index": 0, "inspected_object": "The architectural provenance of the two-stage SFT-then-RL design (MolReasoner framework lineage)", "observation": "The reviewer infers that MolReasoner appears to be an application of the R1 framework in molecular reasoning, noting the familiarity of the SFT-as-distillation and RL-as-reward-fine-tuning pipeline.", "reasoning": "The reviewer applies a norm that novelty is structural rather than domain-specific; applying a known recipe (R1 paradigm) to a new domain constitutes only incremental contribution unless the adaptation itself is non-obvious or technically challenging.", "judgment": "The paper's conceptual contribution is judged as modest/incremental due to this recognized structural lineage.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.770066499710083, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8443642258644104}, {"unit_index": 1, "inspected_object": "The multi-level molecular reward definitions (language-similarity and structural-similarity rewards)", "observation": "The reviewer claims the rewards 'actually reward any molecules' and predicts they may result in poor generation performance.", "reasoning": "The reviewer reasons from the reward structure (rewarding molecule-like strings) to a counterfactual prediction of failure: if rewards do not discriminate between chemically valid/meaningful outputs and meaningless ones, the RL stage will optimize for proxy metrics without improving actual molecular quality.", "judgment": "The reward design is potentially harmful because it may lead to reward-gaming and poor empirical generalization.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8196389079093933, "reasoning_key": "design_justification", "reasoning_sim": 0.8046571612358093}, {"unit_index": 2, "inspected_object": "The baseline selection for comparative evaluation", "observation": "The review identifies missing comparators, specifically naming QWQ and the original version of DeepSeek-R1.", "reasoning": "The reviewer holds an implicit norm that a paper claiming to advance reasoning must benchmark against current state-of-the-art reasoning LLMs, even if not designed for molecules; the absence of these strong baselines makes the empirical evidence insufficient to establish superiority.", "judgment": "The empirical justification for the method's effectiveness is weakened by the lack of fair and hard comparisons.", "valence": "negative", "suggested_improvement": "Include comparisons with QWQ and the original version of DeepSeek-R1.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8957958221435547, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7960034608840942}, {"unit_index": 3, "inspected_object": "The ablation study isolating the contribution of the RL stage", "observation": "The reviewer finds the ablation 'not convincing' and notes a lack of discussion regarding the specific benefits of the RL stage over Mol-SFT alone.", "reasoning": "A two-stage framework's value requires causal attribution; without demonstrating that the RL stage independently drives gains beyond the SFT stage/data quality, the claim that RL transitions the model toward chemical reasoning remains unsubstantiated.", "judgment": "The causal contribution of the RL component is uncertain and insufficiently demonstrated.", "valence": "negative", "suggested_improvement": "Provide more discussion about the benefits of the RL stage and its marginal contribution over Mol-SFT.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.784920334815979, "reasoning_key": "design_justification", "reasoning_sim": 0.7804585099220276}, {"unit_index": 4, "inspected_object": "The theoretical grounding of the proposed framework", "observation": "The reviewer states the entire framework lacks theoretical support.", "reasoning": "The reviewer holds a norm that framework papers should offer principled justification (e.g., convergence guarantees, sample-complexity bounds, or reward-design principles) for why the design works, rather than relying solely on empirical demonstration.", "judgment": "The paper lacks sufficient explanatory depth and principled justification for its design choices.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "low", "object_key": "problem_framing", "object_sim": 0.8038946390151978, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8217357397079468}, {"unit_index": 5, "inspected_object": "The choice of SELFIES as the molecular representation", "observation": "The reviewer asks for justification of using SELFIES in the given context.", "reasoning": "The reviewer probes whether the choice of SELFIES was principled or merely convenient, noting its known properties (guaranteed validity vs. awkwardness for certain structures) relative to alternatives like SMILES or graph-based representations.", "judgment": "The design decision regarding molecular representation requires explicit justification to confirm it was not arbitrary.", "valence": "conditional", "suggested_improvement": "Justify the use of SELFIES in the given context.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7698535323143005, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7433180212974548}]}, {"review_id": "016vLVtNEE", "paper_id": "lycAfkgTQI", "paper_title": "Shift-Invariant Attribute Scoring for Kolmogorov-Arnold Networks via Shapley Value", "decision": "Reject", "summary": "The reviewer systematically challenges the paper's validity by highlighting insufficient experimental scope, contradictions between text and figures, inconsistencies between code and paper descriptions, and a misguided motivation regarding shift-invariance that ignores simpler alternatives.", "units": [{"unit_index": 0, "inspected_object": "Experimental scope and depth (synthetic and real-world)", "observation": "Synthetic experiments use only one hidden layer with five nodes; real-world results are described as inconclusive and limited to two short paragraphs.", "reasoning": "The proposed method involves complex mechanisms (Shapley values, multiple selection strategies). A proportionality norm applies: the experimental burden must match the method's complexity to robustly demonstrate claims. The current setup is too thin to support the paper's assertions.", "judgment": "Insufficient evidence for the core claims due to limited experimental scope.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8593165874481201, "reasoning_key": "design_justification", "reasoning_sim": 0.8580362796783447}, {"unit_index": 1, "inspected_object": "Figure 5 visual evidence vs. textual claim", "observation": "The paper claims 'ShapKAN consistently outperforms Vanilla KAN in generalization,' but Figure 5 shows overlapping, almost identical lines with no clear winner.", "reasoning": "A strict norm of textual fidelity to evidence requires that prose accurately represent figures. A direct contradiction between claim and visual data is a serious substantive flaw, not just a presentation issue.", "judgment": "The textual claim is incorrect and unsupported by the provided evidence.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.853146493434906, "reasoning_key": "presentation_trust", "reasoning_sim": 0.819127082824707}, {"unit_index": 2, "inspected_object": "Code implementation vs. Paper description (KAN version)", "observation": "The code utilizes KAN 2.0, while the paper describes importance scores based on L1 norms from KAN 1.0.", "reasoning": "This creates a methodological inconsistency where the experiments do not test the formulation described in the paper. Furthermore, KAN 2.0 uses variance-based importance, which may already possess the shift-invariance property the authors claim as their contribution, undermining novelty.", "judgment": "Internal inconsistency invalidates the baseline comparison and challenges the novelty of the contribution.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.796880304813385, "reasoning_key": "design_justification", "reasoning_sim": 0.8595895767211914}, {"unit_index": 3, "inspected_object": "PyKAN caching behavior and importance score stability", "observation": "The paper treats PyKAN's overwriting of cached data (causing inconsistent scores across domains) as a problem/disadvantage.", "reasoning": "From a local explanation perspective, importance scores should be data-dependent. Different subsets of data yielding different scores is desired behavior, not a bug. The paper attacks a non-problem or destroys useful information by pursuing shift-invariance.", "judgment": "The motivation to fix this behavior is misguided; the observed behavior is functionally desirable.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8248737454414368, "reasoning_key": "design_justification", "reasoning_sim": 0.8108967542648315}, {"unit_index": 4, "inspected_object": "Motivation for shift-invariance and alternatives", "observation": "The relevance of shift-invariance for importance scores is unclear, and simpler alternatives like batch normalization or dataset normalization are not discussed.", "reasoning": "There is a burden of justification for complex methods. If simpler fixes can achieve similar goals, the complex method's necessity is unproven. The reviewer questions why stable scores are desired when local variation is informative.", "judgment": "The problem formulation is weak because the necessity of shift-invariance is not justified against simpler alternatives.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8486701846122742, "reasoning_key": "design_justification", "reasoning_sim": 0.8672255277633667}]}, {"review_id": "01CHqXTmHn", "paper_id": "qXlWhgkpBp", "paper_title": "Is my action policy safe? PolIC3 to the rescue", "decision": "Reject", "summary": "The reviewer conducts a technical soundness audit focusing on the theoretical gaps in the paper's decoupled verification method. They identify critical weaknesses in the necessity-only surrogate test, the unspecified under-approximation construction, and the scalability of successor selection, while also questioning the semantic fidelity of ASNet mappings. The review demands quantitative characterization of approximation errors and implementation details to validate the method's practical trustworthiness.", "units": [{"unit_index": 0, "inspected_object": "The split test (Theorem 2 and Corollary 1) replacing the exact policy frame test with two necessary-condition checks.", "observation": "The theorems establish that failures of the relaxed test imply failures of the exact test, but do not establish the converse; the condition is necessary but not sufficient.", "reasoning": "The reviewer applies a standard of completeness for verification surrogates: an approximation should characterize its deviation from the exact method to ensure it does not produce excessive false positives or weak pruning in practice.", "judgment": "The asymmetry creates a risk of spurious flags causing unnecessary refinement, making the practical utility of the method uncertain without further characterization.", "valence": "negative", "suggested_improvement": "Provide a short discussion quantifying the gap between the relaxed and exact tests.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.7193836569786072, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7938858270645142}, {"unit_index": 1, "inspected_object": "The construction of the under-approximation set A used in the policy-side check.", "observation": "The algorithmic details for computing A via SMT are unspecified, raising concerns about its behavior under large guards, disjunctions, or derived predicates.", "reasoning": "Errors or looseness in constructing A directly affect the soundness of the rejection step and the strength of pruning; implementation details must be specified when they impact theoretical guarantees.", "judgment": "The unspecified construction is a potential source of unsoundness or degraded performance if not handled carefully.", "valence": "negative", "suggested_improvement": "Specify how A is computed, particularly regarding complex guards and disjunctions, and provide sensitivity analysis of pruning to |A|.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.7851738333702087, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7362789511680603}, {"unit_index": 2, "inspected_object": "The choice to enumerate commands rather than solve the approximate policy frame test for successor selection.", "observation": "This approach is simple but potentially expensive when branching is high.", "reasoning": "Scalability claims require evidence that specific implementation choices do not become bottlenecks on hard instances; efficiency should be backed by analysis of potential cost centers.", "judgment": "The scalability of the method is uncertain due to the potential computational expense of command enumeration in high-branching scenarios.", "valence": "negative", "suggested_improvement": "Provide a cost bound or empirical trace demonstrating that this choice does not undermine performance on hard instances.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8103345036506653, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7952404022216797}, {"unit_index": 3, "inspected_object": "The interface functions F and G mapping ASNet policies to the JANI model semantics.", "observation": "It is unclear what guarantees exist that these functions preserve the original ASNet's decisions over the JANI model semantics.", "reasoning": "Claims of supporting complex policies require evidence of semantic fidelity; reductions between formalisms must preserve behavior or bound deviations.", "judgment": "The validity of the ASNet support claim is unverified without proof of semantic preservation or bounded deviation.", "valence": "negative", "suggested_improvement": "Provide a proof of semantic preservation, a bounded deviation analysis, or a mechanism to flag ambiguous states.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8092864751815796, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7762642502784729}, {"unit_index": 4, "inspected_object": "The IC3 refinement process, specifically clause sizes and variable orders in hard FFNN cases.", "observation": "The reviewer suspects the refinement might produce unnecessarily large clauses, which would hurt scalability, and notes a lack of data on clause structures.", "reasoning": "Experience with IC3 implementations suggests that clause learning heuristics dramatically affect performance; large clauses indicate inefficient reason generation.", "judgment": "The scalability of the refinement step is potentially compromised by inefficient clause growth, requiring heuristic improvements.", "valence": "negative", "suggested_improvement": "Show a heuristic that consistently shrinks reasons and provide data on clause sizes and variable orders.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8428344130516052, "reasoning_key": "design_justification", "reasoning_sim": 0.7667819857597351}]}, {"review_id": "01I3ec7Q01", "paper_id": "AhAIcQzeF1", "paper_title": "EmbedMol: An Open Billion-scale Molecular Embedding Dataset for Molecular Discovery", "decision": null, "summary": "The reviewer performs a feasibility audit focusing on operational claims, demanding decomposition of efficiency metrics, demonstration of model flexibility, and benchmarking against standard retrieval methods to validate the novelty and utility of the dataset.", "units": [{"unit_index": 0, "inspected_object": "The paper's aggregate speedup claim (37x versus fingerprints, 1.5x versus re-running encoder)", "observation": "The reviewer accepts the headline numbers but finds them opaque without component-level decomposition.", "reasoning": "A speedup metric is only interpretable if the cost of each pipeline step (e.g., embedding generation vs. preprocessing) is accounted for; otherwise, it is unclear whether the gain comes from avoiding trivial tasks like SMILES parsing or from genuine efficiency improvements in the core method.", "judgment": "The efficiency claim lacks sufficient evidentiary support to be considered a robust contribution without granular timing breakdowns.", "valence": "negative", "suggested_improvement": "Provide a step-by-step timing breakdown of the pipeline, specifically detailing the time taken for embedding and preprocessing steps.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7957426905632019, "reasoning_key": "design_justification", "reasoning_sim": 0.8056251406669617}, {"unit_index": 1, "inspected_object": "The lock-in effect of precomputed embeddings on future model flexibility", "observation": "The dataset locks users into a specific encoder architecture, preventing easy switching to other models or fine-tuning.", "reasoning": "Precomputed embeddings act as a commitment device that may hinder adaptability; for a dataset to be a valuable general-purpose resource, it should demonstrate transferability or allow for fine-tuning convergence rather than being a frozen representation.", "judgment": "The lack of evidence regarding fine-tuning and cross-task evaluation raises concerns about the dataset's long-term utility and flexibility.", "valence": "negative", "suggested_improvement": "Demonstrate the embeddings' transferability through fine-tuning convergence results and cross-task evaluations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8325909972190857, "reasoning_key": "design_justification", "reasoning_sim": 0.8263289928436279}, {"unit_index": 2, "inspected_object": "The novelty and attribution of the retrieval speedup relative to existing approximate nearest neighbor (ANN) algorithms", "observation": "The paper does not adequately compare its approach against standard ANN libraries (e.g., HNSW, FAISS) for high-dimensional vector search.", "reasoning": "If standard ANN tools already provide sub-linear search over high-dimensional vectors, the claimed speedups may be attributable to the search infrastructure rather than the dataset itself, reducing the contribution to data engineering rather than methodological innovation.", "judgment": "The review questions whether the speedup is novel or expected given existing retrieval systems literature, suggesting insufficient benchmarking against adjacent fields.", "valence": "negative", "suggested_improvement": "Benchmark the approach against existing approximate nearest neighbor searching algorithms to disentangle the contribution of the vectors from the search infrastructure.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8231579661369324, "reasoning_key": "design_justification", "reasoning_sim": 0.762651801109314}]}, {"review_id": "01IGXEMM81", "paper_id": "INAfPtuwtx", "paper_title": "Scalable Multi-Agent Autonomous Learning in Complex Unpredictable Environments", "decision": "Reject", "summary": "The reviewer appreciates the engineering effort and real-world experiments but critiques the lack of methodological novelty, outdated literature, unclear formal grounding, and confusing figures. The central judgment hinges on the absence of benchmark evidence, which is deemed necessary to prove generalizability and superiority over existing MARL methods.", "units": [{"unit_index": 0, "inspected_object": "Literature Review Recency and Coverage", "observation": "The Related Work section cites works primarily from four or five years ago, failing to engage with recent developments in High-Level Reinforcement Learning (HRL) and Task Partitioning & Role Assignment.", "reasoning": "The absence of specific recent references (2021–2025) suggests the authors have not positioned their work against the current research frontier, raising doubts about whether they are aware of the competitive landscape and the novelty of their contribution relative to existing methods.", "judgment": "Weakness: The literature review is outdated and insufficient for establishing methodological novelty.", "valence": "negative", "suggested_improvement": "Incorporate the nine cited recent references across HRL and task partitioning to demonstrate engagement with the current state of the art.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "related_work", "object_sim": 0.8308873772621155, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7895323634147644}, {"unit_index": 1, "inspected_object": "Formal Notation Grounding", "observation": "Section 3.1 presents clear and rigorous formal notation but lacks concrete grounding in the forest firefighting application domain.", "reasoning": "Without anchoring symbols to the real-world problem, readers cannot verify the motivation and reasonableness behind each assumption; formal rigor alone is insufficient if it does not match the empirical setting.", "judgment": "Weakness: The formalism needs better integration with the application context to ensure clarity and interpretability.", "valence": "negative", "suggested_improvement": "Illustrate each symbol using the forest firefighting example to ground the abstract framework in the concrete application.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.794448971748352, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8355814218521118}, {"unit_index": 2, "inspected_object": "Figure 1 Clarity", "observation": "Figure 1, intended as the core visual anchor, is confusing and conveys less clarity than its caption, failing to effectively differentiate the Refocus and Refine processes.", "reasoning": "For a paper contributing a framework rather than a specific algorithm, the figure is the primary vehicle for communicating the conceptual architecture; if it fails, the reader's ability to grasp the contribution is impaired.", "judgment": "Weakness: The core figure is ineffective at communicating the framework's structure.", "valence": "negative", "suggested_improvement": "Redesign Figure 1 to visually differentiate the Refocus and Refine processes.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.833378791809082, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7955531477928162}, {"unit_index": 3, "inspected_object": "Methodological Novelty vs. Engineering Contribution", "observation": "The paper presents an intuitive framework where experimental components are based on or similar to existing methods, and analysis is descriptive rather than demonstrative of superiority.", "reasoning": "ICLR norms prioritize algorithmic or theoretical innovation over engineering deployment; the current work reads more like a technical report because it lacks novel algorithmic contributions that distinguish it from existing methods.", "judgment": "Weakness: The paper lacks the methodological novelty required for the venue, despite its engineering merit.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.9001117944717407, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7929084300994873}, {"unit_index": 4, "inspected_object": "Formal Model Coherence and Scalability", "observation": "The reviewer raises questions about whether the agent set varies over time, if groups can overlap, if simultaneous partitioning creates inefficiencies, and if constraints are satisfiable at scale.", "reasoning": "These probes stress-test whether the formal model's assumptions are realistic and scalable under the conditions claimed by the paper (large-scale, dynamic, unpredictable); if the formalism breaks down or implies unfeasible action conflicts, the framework's validity is compromised.", "judgment": "Uncertainty/Weakness: The realism and scalability of the formal model are questionable without further clarification.", "valence": "negative", "suggested_improvement": "Clarify the dynamic nature of $A_g$, group overlap policies, coordination overhead implications, and constraint satisfiability at scale.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8191388845443726, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7881453633308411}, {"unit_index": 5, "inspected_object": "Autonomy Claim vs. Expert Heuristics", "observation": "It is unclear whether reward functions and knowledge parameters are prior knowledge or expert-defined heuristics.", "reasoning": "If these quantities require expert specification, the 'self-learning' claim is weakened, suggesting the system relies on human-crafted structure rather than being fully autonomous or novel.", "judgment": "Uncertainty: The autonomy of the framework is uncertain depending on the source of key parameters.", "valence": "conditional", "suggested_improvement": "Clarify whether $r(w^t_j)$ and $\rho(a_i)$ are learned automatically or defined by experts.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8156399726867676, "reasoning_key": "robustness_norm", "reasoning_sim": 0.772899329662323}, {"unit_index": 6, "inspected_object": "Experimental Setup Consistency", "observation": "There is an apparent inconsistency between real and simulated drones coexisting and a large discrepancy in scale between simulated drones (3,000) and MARL comparison agents (25).", "reasoning": "This raises factual concerns about the coherence of the experimental design and whether claims of scalability are supported by comparable experiments.", "judgment": "Weakness: The experimental setup has inconsistencies that undermine confidence in the scalability claims.", "valence": "negative", "suggested_improvement": "Resolve inconsistencies in the experimental setup regarding drone types and scale comparisons.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8207007050514221, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7784925103187561}, {"unit_index": 7, "inspected_object": "Evaluation Metric Breadth", "observation": "Drone capability may be evaluated solely based on fire-extinguishing capacity.", "reasoning": "Evaluating performance on a single dimension may not capture the full complexity of the task or the framework's true value.", "judgment": "Uncertainty: The evaluation metrics might be too narrow to fully assess framework performance.", "valence": "conditional", "suggested_improvement": "Consider broader evaluation metrics beyond fire-extinguishing capacity.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8511344194412231, "reasoning_key": "fair_comparison", "reasoning_sim": 0.819974422454834}, {"unit_index": 8, "inspected_object": "Benchmark Evidence for Generalizability", "observation": "The paper lacks results on standard benchmarks such as SMAC, relying instead on real-world experiments.", "reasoning": "Practical success alone does not establish superiority or generalizability; benchmark evidence is crucial to assess the framework's potential impact on future MARL research and demonstrate comparative advantage over existing algorithms.", "judgment": "Critical Weakness: The absence of benchmark evidence prevents assessment of the framework's general value and methodological contribution.", "valence": "negative", "suggested_improvement": "Provide results on standard benchmarks such as SMAC to demonstrate competitive or superior performance.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8756713271141052, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7934737801551819}]}, {"review_id": "01L41rf1kv", "paper_id": "IZV9k5BGxi", "paper_title": "Global and Local Topology-Aware Graph Generation via Dual Conditioning Diffusion", "decision": "Accept (Poster)", "summary": "The reviewer accepts the technical core and empirical effectiveness but identifies three specific gaps in methodological specification: lack of criteria for feature extraction choice, incomplete ablation of the bidirectional conditioning mechanism, and undefined procedures for pre-determined structural parameters. The review frames these as deficiencies in self-sufficiency rather than fatal flaws, suggesting the paper needs clarification to be fully convincing.", "units": [{"unit_index": 0, "inspected_object": "Selection of global feature extraction methods across different tasks", "observation": "The paper uses different global feature extraction methods for different tasks without articulating the selection criterion.", "reasoning": "A method paper should provide a principled rule for choosing among available tools rather than making ad hoc per-dataset decisions; the absence of a rationale jeopardizes generalization to new tasks and lacks epistemic transparency for practitioners.", "judgment": "The methodology is incomplete because it lacks a decision procedure for tool selection.", "valence": "negative", "suggested_improvement": "Provide a decision procedure or heuristic for selecting feature extraction methods based on dataset characteristics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.808858335018158, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8086887001991272}, {"unit_index": 1, "inspected_object": "Ablation study design for bidirectional conditioning mechanism", "observation": "The ablation study does not isolate the performance of a single-direction conditioning variant (local-to-global only).", "reasoning": "A claimed contribution of joint modeling must be validated by showing that both directions matter; without isolating one direction, it is unclear whether the full bidirectional design is necessary or over-engineered.", "judgment": "The current ablation design is incomplete regarding the necessity of the dual mechanism.", "valence": "negative", "suggested_improvement": "Conduct an ablation with only local-to-global condition to validate the necessity of the global-to-local path.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7887685894966125, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7559065222740173}, {"unit_index": 2, "inspected_object": "Pre-determined node and cluster counts in the sampling process", "observation": "The number of nodes and clusters must be fixed in advance at the start of generation, but the method for deciding these numbers is unspecified.", "reasoning": "A generative model should ideally produce outputs not constrained by fixed structural templates from the training distribution; fixing these parameters limits the model's utility in open-ended generation tasks and raises concerns about its true generative scope versus mere interpolation.", "judgment": "The method has a boundary condition regarding generalization to novel graph sizes that is not addressed.", "valence": "negative", "suggested_improvement": "Clarify how node and cluster counts are decided (e.g., dataset statistics or learned prior) and discuss implications for generalization ability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.821315586566925, "reasoning_key": "design_justification", "reasoning_sim": 0.8505422472953796}, {"unit_index": 3, "inspected_object": "Writing quality and clarity", "observation": "The paper is well-written and easy to follow.", "reasoning": "Clear communication makes the paper accessible enough for fair evaluation, allowing the reviewer to assess the reasonable design and solid results effectively.", "judgment": "The presentation is good and facilitates understanding.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8289667367935181, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8210018873214722}, {"unit_index": 4, "inspected_object": "Methodological design", "observation": "The design is described as 'reasonable'.", "reasoning": "The design is accepted as sound but not noted as novel or elegant; it forms the basis for the empirical validation.", "judgment": "The design is acceptable and functional.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7950607538223267, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7649447321891785}, {"unit_index": 5, "inspected_object": "Empirical results", "observation": "The empirical results show effectiveness.", "reasoning": "The evidence supports the claims of the model's performance, though the reviewer expresses no particular enthusiasm beyond acceptance.", "judgment": "The results are effective and support the paper's claims.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8528801202774048, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7437890768051147}]}, {"review_id": "01Q5en9F8v", "paper_id": "Wraipti24J", "paper_title": "Machine Unlearning in Low-dimensional Feature Subspace", "decision": "Reject", "summary": "The reviewer conducts a foundational critique, arguing that the paper's eigenvalue-based motivation is mathematically invalid and contradicted by model behavior, that the lack of theoretical guarantees undermines the method's structural claims, and that the black-box evaluation protocol fails to test against a plausible white-box adversary.", "units": [{"unit_index": 0, "inspected_object": "Eigenvalue-based separability analysis and the claim that similar eigenvalues imply indistinguishable subspaces", "observation": "The reviewer identifies a logical error in inferring subspace indistinguishability from eigenvalue similarity alone.", "reasoning": "Similar eigenvalues do not determine principal directions; two matrices can have identical eigenvalues but span different subspaces. Furthermore, the model achieves high classification accuracy on penultimate hidden states via a linear transform, which contradicts the premise that the features are truly indistinguishable.", "judgment": "The paper's motivational claim regarding subspace geometry is logically flawed and internally inconsistent with the model's demonstrated capability.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8391520380973816, "reasoning_key": "design_justification", "reasoning_sim": 0.7808582186698914}, {"unit_index": 1, "inspected_object": "Theoretical framework and guarantees for the unlearning method", "observation": "The method lacks formal theoretical grounding or guarantees, relying instead on empirical observations and numeric results.", "reasoning": "Methods making structural claims about information preservation or suppression in feature spaces require formal justification. The absence of theory, combined with the flawed empirical motivation (eigenvalue analysis), leaves the method's foundation weakened.", "judgment": "The methodological approach is insufficiently grounded for the claims made.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8410934805870056, "reasoning_key": "design_justification", "reasoning_sim": 0.8741047382354736}, {"unit_index": 2, "inspected_object": "Membership Inference Attack (MIA) evaluation protocol", "observation": "The evaluation uses a black-box MIA to demonstrate unlearning success.", "reasoning": "A white-box attacker would have access to internal representations. Since only a projection is added to unchanged parameters, a white-box attack could likely recover original features or bypass the projection, leading to high performance. Thus, the reported success may be an artifact of the limited attack model.", "judgment": "The evidence for unlearning efficacy is insufficient under a stronger adversary threat model.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7859821319580078, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7934715151786804}]}, {"review_id": "01egveVpXm", "paper_id": "mGqG9Q93KH", "paper_title": "DiT360: High-Fidelity Panoramic Image Generation via Hybrid Training", "decision": null, "summary": "The reviewer positively evaluates the execution quality and ablation structure of the hybrid training framework, viewing strong results as proof of principled design. Simultaneously, they identify gaps in scope validation, specifically requesting broader domain testing, loss weight sensitivity analysis, and quantitative failure characterization for high-frequency details to fully map the method's applicability boundaries.", "units": [{"unit_index": 0, "inspected_object": "Hybrid training strategy and overall framework execution", "observation": "The reviewer finds the hybrid training strategy well-motivated and notes that while not inherently novel, it is executed well.", "reasoning": "The reviewer applies a standard where high-quality execution of existing ideas constitutes a valid contribution, supported by the clarity of ablations which demonstrate thoughtful design rather than ad hoc patching.", "judgment": "The design and execution of the framework is considered 'well done' and convincing.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8496873378753662, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8348855972290039}, {"unit_index": 1, "inspected_object": "Qualitative results on boundary consistency and polar artifacts", "observation": "The reviewer identifies specific visual improvements in boundary consistency and reduction of polar artifacts as concrete outcomes.", "reasoning": "These visible improvements are treated as direct evidence that the method addresses the core problem of panoramic image generation, aligning with the paper's stated goals.", "judgment": "The qualitative results are viewed as particularly persuasive and convincing.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8452286124229431, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7512350082397461}, {"unit_index": 2, "inspected_object": "Scope of validation (domain generalization)", "observation": "The method is only tested on indoor panoramas (Matterport3D), leaving outdoor or dynamic scenes untested.", "reasoning": "The reviewer assumes AR/VR applications require robustness across diverse domains; the absence of testing outside the primary domain creates uncertainty about generalization.", "judgment": "The narrow domain coverage is identified as a weakness regarding scope.", "valence": "negative", "suggested_improvement": "Test the method on outdoor or dynamic scenes to establish generalization beyond static indoor environments.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.817376971244812, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8420960903167725}, {"unit_index": 3, "inspected_object": "Loss weight sensitivity (lambda_1 and lambda_2)", "observation": "The paper lacks ablations analyzing the sensitivity of performance to the hybrid loss weights.", "reasoning": "Systematic component analysis via ablations is viewed as the gold standard for validating design choices; missing this analysis leaves uncertainty about whether success depends on fragile hyperparameter tuning.", "judgment": "The absence of sensitivity analysis is flagged as a weakness in robustness.", "valence": "negative", "suggested_improvement": "Provide an ablation or sensitivity analysis on the choice of loss weights.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7880958914756775, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7983977198600769}, {"unit_index": 4, "inspected_object": "High-frequency details and human faces", "observation": "The paper does not address or quantify performance on high-frequency details and human faces, which are primary use cases for AR/VR.", "reasoning": "The reviewer expects panoramic images for AR/VR to render faces and fine details convincingly; the lack of quantitative failure characterization for these critical content types represents a gap in validation.", "judgment": "The omission is framed as a use-case gap rather than a technical flaw, but still a notable absence.", "valence": "negative", "suggested_improvement": "Provide more quantitative analysis of failure cases, specifically for high-frequency details and human faces.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7987880110740662, "reasoning_key": "construct_validity", "reasoning_sim": 0.7544216513633728}, {"unit_index": 5, "inspected_object": "Evaluation apparatus (user study and metrics)", "observation": "The reviewer notes the presence of a user study and a broad set of evaluation metrics.", "reasoning": "The reviewer accepts these components as sufficient indicators of methodological thoroughness without questioning their design.", "judgment": "The evaluation approach is appreciated as adequate.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8227483034133911, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8460818529129028}]}, {"review_id": "01kxKF0mEj", "paper_id": "AFJMB9SkHT", "paper_title": "FideDiff: Efficient Diffusion Model for High-Fidelity Image Motion Deblurring", "decision": "Accept (Poster)", "summary": "The reviewer performs a verification-oriented evaluation, expressing qualified enthusiasm for the paper's sound theoretical foundation, empirical robustness despite synthetic training, and competent engineering. However, they identify multiple gaps in exposition—including unclear latent generation, unspecified concatenation mechanisms, ambiguous data synthesis, and unidentified figure elements—that hinder full reproducibility and verification, leading to clarification requests rather than fundamental objections.", "units": [{"unit_index": 0, "inspected_object": "Time-consistency training mechanism and single-step distillation theory", "observation": "The reviewer identifies time-consistency training as the key innovation, accepting the mathematical justification that sharing a sharp ground truth across steps makes single-step generation possible despite infeasible Markov prediction calculations.", "reasoning": "The reviewer follows the formal argument and finds it sound, distinguishing this method from other one-step methods by noting it preserves diffusion inductive bias rather than discarding it during distillation.", "judgment": "The theoretical foundation is sound and the conceptual contribution is meaningful for addressing sensitivity to severe blurs.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8244216442108154, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7730273604393005}, {"unit_index": 1, "inspected_object": "Experimental performance on datasets with synthetic-only training", "observation": "The model shows better performance on all datasets compared to other diffusion-based methods and achieves high-fidelity results surpassing conventional methods, achieved despite using only synthetic training data.", "reasoning": "The reviewer treats the success under synthetic-only training as evidence of robustness, expecting such a limitation to hurt generalization more than observed; this empirical strength establishes practical value.", "judgment": "The engineering is competent and the results are empirically strong, supporting the paper's practical value.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8694037795066833, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8276911377906799}, {"unit_index": 2, "inspected_object": "ControlNet adaptation for blur kernel injection", "observation": "The reviewer notes the specific adaptation of ControlNet for injecting blur kernels into the architecture.", "reasoning": "The reviewer has examined the design choices around conditional information and found them appropriate, indicating the authors understand the problem domain well enough to adapt existing tools effectively.", "judgment": "The architectural modifications are appropriate and demonstrate domain competence.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7796315550804138, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8269009590148926}, {"unit_index": 3, "inspected_object": "Generation of initial latent z_0 from low-quality input Z_LQ", "observation": "The reviewer cannot reconstruct how the initial latent is generated from the low-quality input based on the paper's description.", "reasoning": "This gap prevents full verification of the pipeline's entry point and raises reproducibility concerns; the reviewer needs to understand the process to confirm the method works as described.", "judgment": "The exposition regarding the initial latent generation is unclear and hinders verification.", "valence": "negative", "suggested_improvement": "Clarify the procedure for generating z_0 from Z_LQ.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.821739673614502, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8127344250679016}, {"unit_index": 4, "inspected_object": "Training data compatibility and scope limitations", "observation": "The reviewer observes that the architecture's training data requirements constrain generalization, noting limited compatibility without elaborating on specific incompatibilities.", "reasoning": "The reviewer treats this as a scope limitation where the training data constraints may affect the model's ability to generalize beyond the training distribution, though they do not detail the exact failure modes.", "judgment": "There is a potential scope limitation due to training data compatibility constraints.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "low", "object_key": "empirical_scope", "object_sim": 0.852862536907196, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8058820366859436}, {"unit_index": 5, "inspected_object": "Concatenation mechanism for latent feature z and blur kernel", "observation": "The specific technical implementation of concatenating the latent feature z with the blur kernel is unspecified in the text.", "reasoning": "This missing detail prevents the reviewer from fully understanding the architecture's internal mechanics and verifying the integration of conditional information.", "judgment": "The technical implementation details for feature-kernel concatenation are insufficiently described.", "valence": "negative", "suggested_improvement": "Specify the concatenation mechanism for latent feature z and the blur kernel.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8117953538894653, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.743201732635498}, {"unit_index": 6, "inspected_object": "Three-stage training procedure versus end-to-end training", "observation": "The reviewer questions why the model cannot be trained end-to-end instead of using the described three-stage procedure.", "reasoning": "The reviewer recognizes three-stage training as a deliberate design choice with potential downsides (complexity, error accumulation) and wants to understand if this is a fundamental constraint or an engineering compromise affecting elegance and stability.", "judgment": "The necessity of the staged approach is unclear, creating uncertainty about the method's optimal formulation.", "valence": "conditional", "suggested_improvement": "Explain why end-to-end training is not possible and whether the three-stage approach is a fundamental constraint.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7952497005462646, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7479425072669983}, {"unit_index": 7, "inspected_object": "Adaptive timestep prediction component accuracy", "observation": "The reviewer expresses skepticism about the accuracy of the time prediction component and notes a lack of evaluation evidence.", "reasoning": "Without diagnostic analysis or ablation, the reviewer cannot determine if the timestep prediction component is load-bearing or decorative, nor can they assess the system's robustness to prediction errors.", "judgment": "The utility and accuracy of the adaptive timestep prediction component are unverified.", "valence": "uncertain", "suggested_improvement": "Provide an evaluation or ablation study demonstrating the accuracy and impact of the time prediction component.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8055111169815063, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7997863292694092}, {"unit_index": 8, "inspected_object": "Data generation process in supplementary material", "observation": "The reviewer finds the description of the data synthesis process ambiguous, specifically regarding the relationship between 13-frame and 11-frame average samples and the probabilistic augmentation scheme.", "reasoning": "Ambiguity in the data generation process prevents the reviewer from reconstructing the exact training data distribution, which is necessary to assess the credibility of generalization claims.", "judgment": "The data generation process is unclear, hindering assessment of generalization validity.", "valence": "negative", "suggested_improvement": "Clarify the data generation process, including the frame averaging and augmentation schemes.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8061396479606628, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8003416061401367}, {"unit_index": 9, "inspected_object": "Figure 4 blue block identification", "observation": "The reviewer cannot identify the 'blue block' shown in Figure 4.", "reasoning": "This figure comprehension issue prevents mapping the visual presentation to the textual description, representing a barrier to understanding the architecture's components.", "judgment": "The visual presentation lacks clarity regarding specific architectural components.", "valence": "negative", "suggested_improvement": "Label or clarify the blue block in Figure 4.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.763346254825592, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7686968445777893}]}, {"review_id": "01rSfGvZwL", "paper_id": "fHKbqw5ZND", "paper_title": "ReinforceGen: Hybrid Skill Policies with Automated Data Generation and Reinforcement Learning", "decision": "Reject", "summary": "The reviewer conducts a causal decomposition audit, accepting the system works but demanding evidence that each component is necessary and individually effective. They value robustness mechanisms but criticize the lack of standalone validation and granular attribution of performance gains.", "units": [{"unit_index": 0, "inspected_object": "Termination classifier mechanism", "observation": "The reviewer identifies the termination classifier as a strength because it minimizes the gap between training and deployment, addressing distributional mismatch.", "reasoning": "The reviewer values mechanisms that make the system robust to its own imperfections and address failure modes rather than simply improving average performance.", "judgment": "Positive evaluation of the design's potential for reliability under distribution shift.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7937983274459839, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7836536169052124}, {"unit_index": 1, "inspected_object": "Initiation pose predictor mechanism", "observation": "The reviewer identifies the initiation pose predictor as a strength because it can trigger replanning when necessary, introducing closed-loop corrective behavior.", "reasoning": "The reviewer values components that introduce closed-loop correction into pipelines that might otherwise be open-loop brittle, consistent with a focus on reliability.", "judgment": "Positive evaluation of the design's capacity for corrective action.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7974336743354797, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7922549247741699}, {"unit_index": 2, "inspected_object": "Necessity of Imitation Learning (IL) component", "observation": "The reviewer questions whether IL is necessary or merely convenient, suspecting it might be a computational crutch rather than a performance enabler.", "reasoning": "The reviewer demands counterfactual evidence (pure RL pipeline under same budget) to validate causal attribution, assuming component-level attribution is expected.", "judgment": "Uncertainty regarding the necessity of the IL component; skepticism about its unique contribution beyond efficiency.", "valence": "negative", "suggested_improvement": "Provide a counterfactual comparison with a purely reinforcement-learning pipeline under the same training budget.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8249053955078125, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7432637810707092}, {"unit_index": 3, "inspected_object": "Performance relative to privileged-state/larger-demo baselines", "observation": "The method underperforms settings using privileged state information and/or substantially larger human-demo corpora.", "reasoning": "The reviewer seeks to distinguish whether the performance gap is structural (inherent to low-data/vision-only regime) or methodological (fixable with better algorithms).", "judgment": "Concern about the scope and nature of the contribution (regime vs. method).", "valence": "conditional", "suggested_improvement": "Clarify if the gap is structural or methodological to contextualize the contribution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8555898666381836, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8187952637672424}, {"unit_index": 4, "inspected_object": "Standalone accuracy of the termination classifier", "observation": "The standalone quality/accuracy of the termination classifier is unreported.", "reasoning": "A component's mechanism can be elegant while its effectiveness remains unvalidated; the reviewer requires standalone validation for auxiliary learned components.", "judgment": "Negative judgment on the reporting completeness; unwillingness to credit the mechanism without operational performance evidence.", "valence": "negative", "suggested_improvement": "Report standalone accuracy metrics for the termination classifier.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8234022855758667, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8005533814430237}, {"unit_index": 5, "inspected_object": "Real-world validation", "observation": "The paper lacks real-world setting validation.", "reasoning": "The reviewer holds a norm that simulation results require extension to physical settings for ecological validity.", "judgment": "Negative judgment on ecological validity, though noted as potentially underdeveloped or mismatched with the paper's simulation-focused scope.", "valence": "negative", "suggested_improvement": "Include real-world validation or justify its absence relative to the simulation scope.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8425210118293762, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7794448733329773}, {"unit_index": 6, "inspected_object": "Quantification of RL's contribution", "observation": "The paper reports an aggregate 89% increase but does not decompose the contributions of IL, data augmentation, and RL fine-tuning.", "reasoning": "The reviewer assumes additive decomposition of gains and finds the aggregate figure insufficiently granular to attribute specific performance improvements to specific components.", "judgment": "Dissatisfaction with the granularity of ablation reporting; uncertainty about individual component efficacy.", "valence": "negative", "suggested_improvement": "Provide a quantitative decomposition of the performance gains attributable to RL alone, IL alone, and their interaction.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8426442742347717, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7689650058746338}]}, {"review_id": "01xPWViqnk", "paper_id": "A8ez8ThZWq", "paper_title": "REMA: A Unified Reasoning Manifold Framework for Interpreting Large Language Model", "decision": "Reject", "summary": "The reviewer critically evaluates the paper's theoretical framing and evaluation methodology, identifying a significant mismatch between the use of 'manifold' terminology and the lack of formal theory, as well as substantial flaws in the correctness evaluation protocol. Secondary concerns focus on estimator reliability, hyperparameter justification, causal mechanisms, cross-dataset consistency, and practical utility.", "units": [{"unit_index": 0, "inspected_object": "The paper's use of the term 'manifold' in its title and abstract without providing formal mathematical definitions or theorems.", "observation": "The reviewer notes the absence of manifold-theoretic concepts such as connectivity, curvature, geodesics, tangent spaces, or a formal definition of a 'reasoning manifold', despite the paper claiming to offer theoretical insights.", "reasoning": "The reviewer applies an implicit standard that papers using specific mathematical terminology in their framing must either engage with the established apparatus of that theory or explicitly disclaim theoretical ambitions; the absence creates a category error between the rhetorical promise and the actual content.", "judgment": "The reviewer judges the framing as 'clearly misleading' and identifies this mismatch as a fundamental flaw requiring either modification of the central claim or incorporation of more theoretical foundations.", "valence": "negative", "suggested_improvement": "Modify the central claim to remove the theoretical implication or incorporate more theoretical foundations including formal definitions.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7029446959495544, "reasoning_key": "novelty_standard", "reasoning_sim": 0.797691285610199}, {"unit_index": 1, "inspected_object": "The evaluation protocol for determining answer correctness, specifically the use of strict string matching.", "observation": "The reviewer identifies that strict string matching misclassifies semantically equivalent answers (e.g., `(x+1)(x-1)` vs `x^2-1`) as incorrect.", "reasoning": "The reviewer reasons that if the framework depends on distinguishing correct from erroneous reasoning, errors in labeling propagate through the analysis, potentially corrupting the geometric comparison which is the paper's core contribution; this represents a construct validity threat.", "judgment": "The reviewer judges this measurement validity issue as a 'substantial' weakness comparable to the theoretical framing issues.", "valence": "negative", "suggested_improvement": "Use community resources like EleutherAI's lm-evaluation-harness to handle semantic equivalence.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8125961422920227, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8381210565567017}, {"unit_index": 2, "inspected_object": "The reliability of TwoNN and KSG estimators used for entropy dimension (ID) and mutual information (MI) estimation.", "observation": "The reviewer questions whether these estimators are accurate enough given known instability in sparse or high-dimensional settings.", "reasoning": "The reviewer infers that if the estimators are unreliable, the finding of a 'low-dimensional' structure could be an artifact rather than a genuine property, undermining the empirical basis of the paper.", "judgment": "The reviewer expresses uncertainty about the robustness of the empirical claims due to potential estimator instability.", "valence": "conditional", "suggested_improvement": "Provide evidence of estimator stability or justification for their accuracy in the specific settings used.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8367732763290405, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8148894906044006}, {"unit_index": 3, "inspected_object": "The choice of hyperparameter α=2 for layer-wise divergence localization.", "observation": "The reviewer notes a lack of justification for this specific parameter value.", "reasoning": "The reviewer reasons that without sensitivity analysis or theoretical grounding, the main failure-localization results may not be robust or generalizable to other settings.", "judgment": "The reviewer views the lack of justification as a gap in methodological rigor.", "valence": "negative", "suggested_improvement": "Perform sensitivity analysis or provide theoretical grounding for the choice of α=2.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8049859404563904, "reasoning_key": "robustness_norm", "reasoning_sim": 0.827999472618103}, {"unit_index": 4, "inspected_object": "The causal explanation for the formation of the low-dimensional 'manifold'.", "observation": "The reviewer observes that the paper describes the phenomenon but does not explain its causes.", "reasoning": "The reviewer applies a standard that moves beyond description to mechanism, expecting the paper to explain whether the phenomenon arises from architecture, training, or other factors.", "judgment": "The reviewer finds the current level of explanation insufficient for understanding the underlying mechanics.", "valence": "negative", "suggested_improvement": "Investigate and report what contributed to the formation of the low-dimensional structure.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.790289580821991, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7839203476905823}, {"unit_index": 5, "inspected_object": "The consistency of 'reasoning manifold' properties across different datasets.", "observation": "The reviewer notes varying ID values across datasets in Table 1.", "reasoning": "The reviewer reasons that if dimensionality varies wildly by dataset, it is unclear if the same kind of object is being observed across settings, questioning the universality of the 'manifold' concept.", "judgment": "The reviewer expresses doubt about the cross-dataset consistency of the proposed framework.", "valence": "uncertain", "suggested_improvement": "Analyze and discuss the consistency of manifold properties across datasets.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8192176222801208, "reasoning_key": "construct_validity", "reasoning_sim": 0.7719568610191345}, {"unit_index": 6, "inspected_object": "The actionability of the findings for modifying LLM behavior or enhancing reliability.", "observation": "The reviewer asks whether the findings can be used to modify behavior and requests small-scale preliminary experiments.", "reasoning": "The reviewer applies a standard of demonstrated utility over speculative potential, requiring evidence that the theoretical/empirical insights translate to practical improvements.", "judgment": "The reviewer finds the current demonstration of utility insufficient.", "valence": "negative", "suggested_improvement": "Conduct small-scale preliminary experiments to demonstrate the utility of the findings.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.772777795791626, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8246732950210571}, {"unit_index": 7, "inspected_object": "The formatting of Table 3.", "observation": "The reviewer notes that Table 3 appears within the references section.", "reasoning": "The reviewer considers this a concrete presentation barrier that affects readability and professionalism.", "judgment": "The reviewer judges this as a minor but necessary correction.", "valence": "negative", "suggested_improvement": "Move Table 3 out of the references section to its appropriate location in the text.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8516870141029358, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8237996101379395}]}, {"review_id": "020d6CP2YN", "paper_id": "vfiXH5yi9l", "paper_title": "Efficient Zero-Shot Coordination via Offline Policy Diversity and Online Belief Reasoning", "decision": null, "summary": "The reviewer evaluates the paper primarily through a literature-mapping lens, dismissing the contribution as trivial due to prior work (CooT) and criticizing the empirical evidence for lacking modern baselines, broad environmental validation, and robust human trials. While acknowledging the motivation, the reviewer concludes the paper is insufficiently established compared to field standards.", "units": [{"unit_index": 0, "inspected_object": "The paper's core methodological claim regarding the use of offline RL for zero-shot coordination (ZSC).", "observation": "The reviewer identifies prior work, specifically CooT (arXiv:2506.23549), that has already explored offline methods for ZSC problems.", "reasoning": "The reviewer applies an implicit norm that a contribution must either introduce a new problem formulation or demonstrate a clear empirical leap over existing methods; since prior work exists, the mere application is deemed insufficient novelty.", "judgment": "The contribution is assessed as 'somewhat trivial' and lacking in originality.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.76521897315979, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8699650764465332}, {"unit_index": 1, "inspected_object": "The experimental results showing performance improvement.", "observation": "The reviewer finds the performance improvement to be 'not substantial' and notes that offline training often struggles to outperform online ZSC approaches.", "reasoning": "Using a counterfactual comparison, the reviewer questions whether the proposed method would have been better off using online methods directly, given that the offline approach does not yield a decisive win, thus failing its own promise of significant advantage.", "judgment": "The method fails to demonstrate a convincing empirical advantage.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8135536313056946, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7834913730621338}, {"unit_index": 2, "inspected_object": "The set of baselines used in the evaluation.", "observation": "The reviewer identifies specific missing baselines: MEP, TrajeDi, and COLE.", "reasoning": "The reviewer holds a norm that empirical evidence must be comparative against the current state of the art; the absence of these modern baselines suggests the evaluation is limited and outdated, undermining the credibility of the claims.", "judgment": "The evaluation is not convincing due to limited and somewhat outdated baselines.", "valence": "negative", "suggested_improvement": "Include comparisons with MEP, TrajeDi, and COLE.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8203874826431274, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8027774095535278}, {"unit_index": 3, "inspected_object": "The generalizability of the method across environments.", "observation": "The evaluation is conducted in a relatively narrow environment (Hanabi).", "reasoning": "The reviewer applies a norm that ZSC methods should be validated across multiple coordination environments with different structural properties; testing only in Hanabi leaves the method's applicability to other settings unproven.", "judgment": "Generalizability is unproven.", "valence": "negative", "suggested_improvement": "Test the method in additional environments such as Overcooked.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8488059639930725, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7250718474388123}, {"unit_index": 4, "inspected_object": "The human-AI coordination experiments.", "observation": "The number of participants is relatively small, and the evaluation metrics are limited.", "reasoning": "The reviewer invokes a statistical power argument, suggesting that with few participants and limited metrics, the observed human-AI performance could be noise rather than a reliable signal.", "judgment": "Reliability concerns exist regarding the human-AI results.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8343353867530823, "reasoning_key": "construct_validity", "reasoning_sim": 0.7549924254417419}, {"unit_index": 5, "inspected_object": "The construction and cost of the offline dataset.", "observation": "The reviewer asks whether the training cost of generating the offline dataset (using prior methods like OBL) has been accounted for.", "reasoning": "The reviewer applies a lifecycle accounting norm: if the offline dataset required expensive online training of prior methods, then the paper's 'reduced online interaction' claim may be misleading or unfair.", "judgment": "Uncertainty remains about the fairness and transparency of the sample-efficiency claims.", "valence": "conditional", "suggested_improvement": "Clarify the total training cost pipeline, including dataset generation costs.", "support_status": "mixed", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8258609771728516, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7919606566429138}, {"unit_index": 6, "inspected_object": "The visual presentation of Figure 3 and Table 1.", "observation": "Figure 3 has a misaligned blue background, and Table 1 has unclear highlighting.", "reasoning": "The reviewer interprets visual polish as a proxy for care in reporting; sloppy figures undermine confidence in the underlying results and suggest a lack of rigor.", "judgment": "Presentation quality issues negatively impact perceived credibility.", "valence": "negative", "suggested_improvement": "Correct the misalignment in Figure 3 and clarify highlighting in Table 1.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8757473826408386, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8239759802818298}]}, {"review_id": "0078hjmItT", "paper_id": "Omf1As3tZt", "paper_title": "Wasserstein Distributionally Robust Minimax Regret Optimization for Multimodal Machine Learning", "decision": "Reject", "summary": "The reviewer audits the manuscript against professional norms of self-contained notation, canonical theoretical structure, and comprehensive empirical validation, finding deficiencies in all three areas despite acknowledging the theoretical foundation.", "units": [{"unit_index": 0, "inspected_object": "Notation and definitional hygiene (specifically undefined symbols like phi(f) and conflated sets U/B)", "observation": "The reviewer identified eight specific instances of inconsistent notation, undefined symbols, or labeling errors across the manuscript.", "reasoning": "The reviewer applies a standard that consistent notation is diagnostic of mathematical rigor; poor presentational discipline implies potential undisciplined mathematics.", "judgment": "The paper fails to meet the required standard of care for publication due to these presentational deficits.", "valence": "negative", "suggested_improvement": "Define every symbol at first use and ensure consistent notation throughout the text.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.757171094417572, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8091498017311096}, {"unit_index": 1, "inspected_object": "Empirical breadth and baseline comparisons", "observation": "The empirical evaluation relies on a single dataset (HANCOCK) and only two baselines (ERM-LR, WDRO-LR), lacking comparisons to group-DRO or modern multimodal architectures.", "reasoning": "The narrow scope of evidence is insufficient to support claims of practical utility and generalizability, raising suspicion that novelty might be incremental rather than substantive.", "judgment": "The empirical work is too thin to validate the theory and establish advantage over existing approaches.", "valence": "negative", "suggested_improvement": "Include comparisons to group-DRO and modern multimodal deep architectures, and provide ablations over key parameters.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8984742164611816, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.788679838180542}, {"unit_index": 2, "inspected_object": "Theoretical presentation structure (Proposition 2.1, Lemmas 3.1–3.4)", "observation": "The main text includes proofs for standard results (strong duality) and supporting lemmas that could be moved to the appendix.", "reasoning": "Disciplinary conventions dictate that standard results should be cited rather than re-derived, and the main text should foreground primary results.", "judgment": "The packaging of theoretical results does not follow canonical exposition norms.", "valence": "negative", "suggested_improvement": "Cite canonical references for standard results and move supporting lemmas to the appendix.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8769662380218506, "reasoning_key": "presentation_trust", "reasoning_sim": 0.750703752040863}, {"unit_index": 3, "inspected_object": "Algorithm applicability in non-convex settings", "observation": "The reviewer questions whether Algorithm 1's updates are well-defined when extended to non-convex deep learning scenarios.", "reasoning": "Theoretical guarantees rely on convexity and duality; it is uncertain if the algorithm remains valid or effective in non-convex regimes without clarification.", "judgment": "Uncertainty regarding the method's definition and validity in non-convex deep fusion settings.", "valence": "uncertain", "suggested_improvement": "Clarify how Learner/Oracle updates function in the non-convex case via stochastic gradient steps.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8606294989585876, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8176576495170593}]}, {"review_id": "00RLD51ncN", "paper_id": "2NBS9ilNqM", "paper_title": "The Alignment Waltz: Jointly Training Agents to Collaborate for Safety", "decision": "Accept (Poster)", "summary": "The reviewer recognizes the novelty of the positive-sum framework and the clarity of the paper but raises substantive reservations about the lack of mechanistic motivation, insufficient experimental probing of the DIR mechanism (including missing ablations and single-round limitations), and inadequate benchmarking against other multi-agent RL methods.", "units": [{"unit_index": 0, "inspected_object": "The conceptual motivation for collaborative training enhancing helpfulness and safety.", "observation": "The reviewer notes the paper asserts that collaborative training enhances helpfulness and reduces harm but lacks a mechanistic explanation for why this connection exists.", "reasoning": "The reviewer applies an implicit standard that central claims must be motivated by a clear causal story rather than just empirical demonstration; without understanding *why* collaboration yields safety benefits, the claim remains unsupported theoretically.", "judgment": "The motivation is insufficiently explained, creating a gap between the claimed benefits and the provided rationale.", "valence": "negative", "suggested_improvement": "Describe the specific role of the feedback agent during training and inference, particularly its function in providing reflection on over-refusal for revision.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7635658383369446, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8129941821098328}, {"unit_index": 1, "inspected_object": "The experimental design regarding the number of collaborative rounds (set to 1).", "observation": "The reviewer observes that the collaborative round number is fixed at 1, despite the Dynamic Improvement Reward (DIR) being described as dynamic and evolving over time.", "reasoning": "If the system only runs one round, the 'dynamic' aspect of the reward is barely exercised, raising questions about whether a single round is sufficient to harness or demonstrate the power of the proposed mechanism.", "judgment": "The single-round setting may be insufficient to fully validate the dynamic nature of the DIR mechanism.", "valence": "negative", "suggested_improvement": "Conduct experiments with multiple collaborative rounds to test if the framework's benefits scale and if the feedback remains coherent across iterations.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8106446862220764, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7535879611968994}, {"unit_index": 2, "inspected_object": "The ablation study structure in Section 3.3.", "observation": "The reviewer identifies that all settings (A), (B), and (C) in the ablation table include the Dynamic Improvement Reward (DIR), meaning there is no 'without DIR' condition.", "reasoning": "Since DIR is the distinctive component of the approach, its absence from the ablation means the paper cannot empirically isolate whether DIR is actually responsible for the observed improvements versus the general positive-sum framework.", "judgment": "The lack of a DIR-free ablation is a significant omission that weakens the claim that DIR drives the performance gains.", "valence": "negative", "suggested_improvement": "Include an ablation setting that removes the DIR component to isolate its specific contribution to the system's performance.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.792883038520813, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7773014307022095}, {"unit_index": 3, "inspected_object": "The stopping criteria for adaptive engagement of the feedback agent.", "observation": "The reviewer notes that the stopping criteria for when the feedback agent engages are not verified in the experiments, despite the abstract claiming adaptive engagement.", "reasoning": "Without verification, it is unclear if the system correctly decides when feedback is needed; the empirical results do not confirm that the adaptive mechanism functions as intended.", "judgment": "The adaptive engagement mechanism lacks empirical validation regarding its decision-making process.", "valence": "negative", "suggested_improvement": "Verify and report on the correctness of the stopping criteria decisions made by the feedback agent.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8338770866394043, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.826206624507904}, {"unit_index": 4, "inspected_object": "The comparison baselines against other multi-agent RL frameworks.", "observation": "The reviewer identifies a missing baseline: other multi-agent RL frameworks, specifically citing Zheng et al. 2024.", "reasoning": "To demonstrate that the specific design choices (positive-sum game, DIR) are advantageous, the paper must show superiority over existing multi-agent methods, not just single-agent or non-RL baselines.", "judgment": "The comparative evaluation is incomplete because it fails to benchmark against relevant multi-agent RL competitors.", "valence": "negative", "suggested_improvement": "Benchmark the proposed method against other multi-agent RL frameworks, such as Zheng et al. 2024, to demonstrate distinct advantages.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8174951076507568, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7605841159820557}, {"unit_index": 5, "inspected_object": "The novelty of the positive-sum game formulation compared to zero-sum games.", "observation": "The reviewer acknowledges that the paper proposes a positive-sum game between agents, contrasting with previous multi-agent frameworks that use zero-sum games.", "reasoning": "This distinction places the work in a unique position in the design space, offering a different paradigm for agent collaboration.", "judgment": "The shift from zero-sum to positive-sum collaboration is a novel and valuable design choice.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.8190650343894958, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7250314950942993}, {"unit_index": 6, "inspected_object": "The clarity of exposition and overall presentation.", "observation": "The reviewer finds the paper 'overall easy to follow' and notes the accurate summary of technical content.", "reasoning": "Clear exposition facilitates understanding of the complex multi-agent interactions and the DIR mechanism, reducing barriers to entry for readers.", "judgment": "The presentation quality is high, making the technical contributions accessible.", "valence": "positive", "suggested_improvement": "Address minor grammatical errors noted in the text.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8404306173324585, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7093766927719116}, {"unit_index": 7, "inspected_object": "The visualization of the Dynamic Improvement Reward (DIR).", "observation": "The reviewer requests visualization of the DIR behavior.", "reasoning": "Visualizing how the reward evolves would help verify if it tracks the conversation agent's improvements as designed, providing intuitive insight into the mechanism's dynamics.", "judgment": "The current lack of DIR visualization limits the interpretability of the reward's behavior.", "valence": "conditional", "suggested_improvement": "Add visualizations showing how the DIR converges or tracks improvements over time.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8167733550071716, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7140341997146606}]}, {"review_id": "00s7AkDp7R", "paper_id": "Y3L3WotMPm", "paper_title": "MemOrb: A Plug-and-Play Verbal-Reinforcement Memory Layer for E-Commerce Customer Service", "decision": null, "summary": "The reviewer performs a pre-evaluative gatekeeping assessment, concluding the paper is not ready for detailed review due to insufficient technical detail, poor presentation, limited baselines, questionable venue fit, and weak scholarly hygiene. A single positive acknowledgment of problem identification is isolated and does not mitigate the structural deficiencies.", "units": [{"unit_index": 0, "inspected_object": "The paper's framing of the problem and motivation", "observation": "The reviewer acknowledges that the paper correctly identifies some areas where existing LLM-based agents fail.", "reasoning": "This is accepted as a valid observation of the problem space, but it is isolated from the solution evaluation; the reviewer does not specify which areas are correct or why, indicating only a surface-level scan of the introduction.", "judgment": "The problem identification is plausible and correct, but this positive assessment does not extend to the proposed solution or its presentation.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "low", "object_key": "problem_framing", "object_sim": 0.8679443597793579, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7177505493164062}, {"unit_index": 1, "inspected_object": "Figure 1 and its caption", "observation": "The reviewer cites Figure 1 as evidence of major writing problems without describing specific defects like illegibility or missing labels.", "reasoning": "The figure serves as a synecdoche for a general impression of poor presentation; the lack of specific critique suggests the judgment is based on an overall aesthetic or clarity failure rather than a detailed technical audit.", "judgment": "The presentation quality is insufficient, contributing to the perception that the paper is not ready for detailed review.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "clarity", "object_sim": 0.8274838924407959, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7587432265281677}, {"unit_index": 2, "inspected_object": "The technical description of the approach", "observation": "The reviewer notes almost no technical detail beyond very limited pseudocode.", "reasoning": "Technical detail is treated as a prerequisite for evaluation; without sufficient methodological description, the reviewer cannot assess technical relevance, methodological contribution, or novelty.", "judgment": "The paper is in principle unassessable in its current form due to lack of technical substance.", "valence": "negative", "suggested_improvement": "Clarify the main technical contribution and provide sufficient detail to allow assessment of methodological relevance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7937892079353333, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7826899290084839}, {"unit_index": 3, "inspected_object": "Venue fit relative to ICLR", "observation": "The reviewer states the topic is largely not relevant to ICLR.", "reasoning": "The judgment rests on an implicit assumption that customer service applications with LLM agents fall outside ICLR's scope, which is focused on representation learning and deep learning methods.", "judgment": "The paper lacks topical relevance for the target venue.", "valence": "negative", "suggested_improvement": "Clarify the relevance of the work to ICLR's scope.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "clarity", "object_sim": 0.6849837899208069, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.6964212656021118}, {"unit_index": 4, "inspected_object": "Baselines and evaluation breadth", "observation": "The reviewer flags that baselines are far too limited for proper evaluation.", "reasoning": "Limited baselines undermine the evidentiary value of experimental claims, preventing a robust comparison of performance.", "judgment": "The evaluation is insufficiently rigorous to support the paper's claims.", "valence": "negative", "suggested_improvement": "Include more comprehensive baselines to validate the claimed improvements.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.877365231513977, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8473826050758362}, {"unit_index": 5, "inspected_object": "Bibliography and citation format", "observation": "The bibliography consists nearly entirely of arXiv links, which the reviewer assumes are automatically generated.", "reasoning": "The citation format is interpreted as evidence of carelessness or a lack of engagement with peer-reviewed literature, reflecting poorly on scholarly rigor.", "judgment": "The paper demonstrates low scholarly hygiene and potential neglect of established literature norms.", "valence": "negative", "suggested_improvement": "Replace arXiv links with references to published venues to demonstrate proper scholarly engagement.", "support_status": "memo_inferred", "confidence": "low", "object_key": "related_work", "object_sim": 0.8229591250419617, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7915357947349548}]}, {"review_id": "01VgqIxjJ4", "paper_id": "EbIL1Po0Vf", "paper_title": "Deep Low Rank Projector for KV Cache Compression", "decision": null, "summary": "The reviewer critically evaluates the paper's theoretical tightness, conceptual novelty, and experimental fairness, challenging the principled basis of the regularizer, the distinctiveness of the DLN motivation against existing LoRA work, and the asymmetry in capacity between the proposed method and baselines.", "units": [{"unit_index": 0, "inspected_object": "Theorem 3.1 derivation (Equations 20 and 21)", "observation": "The reviewer identifies that the denominator in the bound should be $N^N$ rather than $N$, making the bound exponentially loose, though the inequality direction remains correct.", "reasoning": "Because the bound is arbitrarily loose, the specific choice of the regularizer form ($Σ||D_n||_F$) becomes unprincipled; a tight bound is required to justify the design choice theoretically.", "judgment": "The theoretical justification for the regularizer's form is weak because it relies on a loose bound that does not constrain the design principles.", "valence": "negative", "suggested_improvement": "Provide a normative argument justifying why this specific loose bound is the right one to use for the regularizer's design.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8330206274986267, "reasoning_key": "design_justification", "reasoning_sim": 0.8220054507255554}, {"unit_index": 1, "inspected_object": "Motivation for Deep Linear Networks (DLNs)", "observation": "The reviewer notes that for N=2, the method essentially reduces to finding the lowest ranked update and fine-tuning, and that determining task-dependent ranks is not conceptually novel compared to existing LoRA methods.", "reasoning": "If the mechanism collapses to standard low-rank fine-tuning and the outcome (rank selection) is already known, the claimed novelty of the DLN architecture and its implicit bias is overstated or incidental.", "judgment": "The conceptual contribution regarding the necessity and novelty of the DLN component is questionable due to reduction to mundane procedures.", "valence": "negative", "suggested_improvement": "Discuss and compare with existing LoRA methods that optimize singular values (e.g., arXiv:2405.19597) to clarify the distinct novelty.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.811217188835144, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8052432537078857}, {"unit_index": 2, "inspected_object": "Experimental fairness (LoRA baseline rank vs. proposed method rank selection)", "observation": "The proposed method selects ranks based on an energy threshold (approx 10%, 25%, 50% compression), allowing it to use more task-dependent parameters, while the LoRA baseline appears fixed at rank 32.", "reasoning": "This asymmetry gives the proposed method more architectural freedom (variable parameter count) than the baseline, meaning performance differences could be attributed to capacity asymmetry rather than intrinsic merit.", "judgment": "The experimental comparison is potentially unfair due to unequal capacity constraints between the proposed method and the baseline.", "valence": "negative", "suggested_improvement": "Clarify if all models were finetuned with rank 32 irrespective of compression, or adjust the baseline to match the variable capacity of the proposed method.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.8571511507034302, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8111765384674072}, {"unit_index": 3, "inspected_object": "Necessity of the regularization parameter alpha", "observation": "The reviewer poses a counterfactual: if alpha=0 works, the regularization may be unnecessary; if higher alpha reduces selected rank, it has a causal effect.", "reasoning": "Testing whether the regularizer is load-bearing or decorative helps determine if the proposed component adds value beyond the base model.", "judgment": "Uncertainty remains regarding the causal effect and necessity of the regularization component.", "valence": "uncertain", "suggested_improvement": "Conduct experiments varying alpha to demonstrate the regularizer's impact on rank selection and performance.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.827123761177063, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7267502546310425}]}, {"review_id": "01myf25tck", "paper_id": "Y74tGjpsjq", "paper_title": "PAML: MoE-Based Partitioning and Merging Framework for Solving Large-scale Multi-task VRPs", "decision": null, "summary": "The reviewer conducts a novelty-and-rigor audit, attributing system components to prior work to judge limited innovation, demanding technical specificity for the MoE mechanism, questioning the necessity of the merging step via counterfactual comparison, noting missing comparisons with recent SOTA, and correcting factual errors about prior work, while acknowledging the merging strategy's strength.", "units": [{"unit_index": 0, "inspected_object": "The paper's architectural components (partitioner, solver, merging strategy)", "observation": "The reviewer identifies the partitioner as borrowed from TAM and the solver from MVMoE, isolating the merging integration as the only novel element.", "reasoning": "Novelty is treated as a combinatorial property; if constituent components are known derivatives, the innovation must reside in their combination. The reviewer judges this specific combination (merging) as insufficiently transformative for the venue.", "judgment": "Limited innovation due to derivative components.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8256099224090576, "reasoning_key": "design_justification", "reasoning_sim": 0.8125534653663635}, {"unit_index": 1, "inspected_object": "The MoE-based partitioner claim and mechanism details", "observation": "The paper titles itself with 'MoE' but provides no details on expert routing or gating mechanisms in the main text or appendix.", "reasoning": "A transparency norm requires that named mechanisms be operationally explained. The absence of such detail constitutes a soundness problem rather than a mere presentation issue.", "judgment": "Insufficient technical specificity undermining soundness.", "valence": "negative", "suggested_improvement": "Provide details on how experts are routed and handle variant constraints.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.831527829170227, "reasoning_key": "design_justification", "reasoning_sim": 0.8321419358253479}, {"unit_index": 2, "inspected_object": "Experimental comparisons against direct end-to-end generation", "observation": "The paper lacks an ablation-style comparison between 'partitioning + merging' and direct end-to-end large subproblem generation.", "reasoning": "A parsimony/minimalism norm dictates that non-obvious design choices (like adding a merging step) must be justified by demonstrating necessity against simpler alternatives. Without this comparison, the added complexity is unjustified.", "judgment": "Unjustified architectural complexity due to missing justification.", "valence": "negative", "suggested_improvement": "Compare 'partitioning + merging' against direct end-to-end large subproblem generation to demonstrate the necessity of the merging step.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8459932804107666, "reasoning_key": "design_justification", "reasoning_sim": 0.8574138879776001}, {"unit_index": 3, "inspected_object": "Comparison with recent state-of-the-art methods (CaDA)", "observation": "The paper does not compare its results with CaDA (ICML 2025).", "reasoning": "A temporal relevance norm requires benchmarking against the most recent frontier methods, not just the ones the paper builds upon. Absence suggests lack of awareness or failure to improve upon the current state.", "judgment": "Weakness in literature positioning and currency.", "valence": "negative", "suggested_improvement": "Include comparisons with recent state-of-the-art methods like CaDA.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8119493126869202, "reasoning_key": "design_justification", "reasoning_sim": 0.8646674752235413}, {"unit_index": 4, "inspected_object": "Description of prior work (MVMoE training procedure)", "observation": "The paper incorrectly characterizes MVMoE's training as using multi-task loss and claims it is built on Berto et al.'s platform, whereas it trains one task per batch with backpropagation.", "reasoning": "Scholarly integrity and credibility checks require accurate characterization of prior work. Mischaracterization suggests the authors have not deeply engaged with the methods they claim to extend, undermining confidence in the paper's validity.", "judgment": "Factual errors undermine scholarly credibility and trust in the method's foundation.", "valence": "negative", "suggested_improvement": "Correct the description of MVMoE's training procedure and platform basis.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7587291598320007, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8531515598297119}, {"unit_index": 5, "inspected_object": "Equation numbering in Section 4.1", "observation": "There is an equation numbering error in Section 4.1.", "reasoning": "Presentation polish issues compound substantive concerns and signal careful reading, though they are minor relative to novelty and soundness.", "judgment": "Minor presentation flaw.", "valence": "negative", "suggested_improvement": "Fix equation numbering error in Section 4.1.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7412635087966919, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8133528828620911}, {"unit_index": 6, "inspected_object": "The merging strategy's preservation of global information", "observation": "The merging strategy preserves global information better than TAM while balancing solving complexity.", "reasoning": "Acknowledgement of genuine engineering value and effectiveness despite overall novelty concerns.", "judgment": "Genuine strength in the merging strategy.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7761948704719543, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8025708794593811}]}, {"review_id": "01rpC7AaQV", "paper_id": "ZY9nFE36OY", "paper_title": "BRIDGEBENCH: AN OFFLINE CONSTRAINED MULTI- AGENT REINFORCEMENT LEARNING BENCHMARK FOR INFRASTRUCTURE MANAGEMENT", "decision": null, "summary": "The reviewer rejects the benchmark's premise due to premature combination of complex subfields (offline, MARL, constrained RL) leading to unsound data requirements, concerns about unrealistic world model data, unclear baseline descriptions, and poor presentation quality.", "units": [{"unit_index": 0, "inspected_object": "The combination of offline RL, multi-agent RL (MARL), and constrained RL in a single benchmark.", "observation": "Offline RL, MARL, and constrained RL are individually challenging and not fully understood; combining them multiplies data requirements significantly.", "reasoning": "Offline RL requires strong data coverage to work; constraints must also be supported by data for every agent. Combining these elements creates a combinatorial complexity that makes applicability and usability highly suspicious given the current state of the field.", "judgment": "The fundamental premise of the benchmark is methodologically flawed and premature.", "valence": "negative", "suggested_improvement": "Isolate one challenging element at a time to enable careful study, as current benchmarks do, rather than combining multiple hard subfields prematurely.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8411011695861816, "reasoning_key": "design_justification", "reasoning_sim": 0.811790943145752}, {"unit_index": 1, "inspected_object": "Use of data-driven world models for offline policy evaluation.", "observation": "Data-driven world models are used, but these models can produce unrealistic data.", "reasoning": "If the benchmark relies on learned world models for evaluation, the results are only as trustworthy as the models' fidelity. Unrealistic simulated data undermines the reliability of evaluations intended for real-world deployment.", "judgment": "The world model component introduces a threat to validity independent of the combinatorial complexity argument.", "valence": "negative", "suggested_improvement": "Provide more details about the world model used for offline policy evaluation, as it is a crucial part of the benchmark currently not covered in the main text.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.7774848937988281, "reasoning_key": "design_justification", "reasoning_sim": 0.8644372224807739}, {"unit_index": 2, "inspected_object": "Description and categorization of hybrid baselines (specifically CQL-IQL).", "observation": "Hybrid baseline names like CQL-IQL are introduced without adequate explanation of how components are combined or why they constitute state-of-the-art offline MARL methods.", "reasoning": "The reviewer perceives CQL-IQL as a composition of two single-agent offline RL methods rather than a genuine multi-agent method. The paper's framing suggests a conflation of single-agent and multi-agent capabilities, which challenges the validity of the multi-agent claims.", "judgment": "The baseline descriptions are unclear and potentially misleading regarding the nature of the methods evaluated.", "valence": "negative", "suggested_improvement": "Explain how the components of hybrid baselines like CQL-IQL are combined into a multi-agent method, or acknowledge if they are not genuinely multi-agent methods.", "support_status": "mixed", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.8597325682640076, "reasoning_key": "design_justification", "reasoning_sim": 0.8077878952026367}, {"unit_index": 3, "inspected_object": "Figure legibility (Figures 3 and 4).", "observation": "The text within Figures 3 and 4 is too small to read.", "reasoning": "Illegible figures undermine the paper's ability to communicate its experimental results effectively.", "judgment": "Presentation quality is insufficient for effective communication.", "valence": "negative", "suggested_improvement": "Increase font size in figures to ensure readability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8500311374664307, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8404766321182251}, {"unit_index": 4, "inspected_object": "Scope and generality of the benchmark tasks.", "observation": "It is unclear whether the benchmark consists of only one task (bridge maintenance) with different settings or many different tasks.", "reasoning": "A benchmark with only one task is less useful than one with multiple diverse tasks. The reviewer probes this to determine if the contribution is a single environment or a general framework.", "judgment": "Uncertainty about the benchmark's scope limits its perceived utility and generalizability.", "valence": "conditional", "suggested_improvement": "Clarify whether the benchmark includes multiple diverse tasks or just one task with varying settings.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.8646618127822876, "reasoning_key": "design_justification", "reasoning_sim": 0.8131962418556213}]}, {"review_id": "023IWHtq43", "paper_id": "aooJUiadOm", "paper_title": "ReXMoE: Reusing Experts with Minimal Overhead in Mixture-of-Experts", "decision": "Reject", "summary": "The reviewer identifies three main weaknesses: insufficient analysis of prefill latency degradation, lack of controlled experiments to validate the decoupling claim, and absence of stress tests for expert collapse. The reviewer demands more rigorous empirical validation to support the paper's claims of practical viability and architectural novelty.", "units": [{"unit_index": 0, "inspected_object": "Prefill latency performance for short sequences with R8 configuration", "observation": "Figure 2(a) shows up to a 77% slowdown in prefill speed, which the reviewer interprets as a significant degradation in practical applicability.", "reasoning": "The implicit standard is that MoE architectures must improve efficiency; trading inference speed for modeling gains requires explicit justification and diagnostic breakdowns of time spent during prefill.", "judgment": "The paper's acknowledgment of increased I/O operations is too superficial to support claims of practical viability for latency-sensitive applications.", "valence": "negative", "suggested_improvement": "Provide a detailed breakdown of where time is spent during prefill and identify use cases where the prefill penalty is acceptable.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7916446924209595, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7550031542778015}, {"unit_index": 1, "inspected_object": "Claim that ReXMoE decouples expert dimensionality from per-layer budgets", "observation": "All experiments use fixed architectural configurations (Table 1), so the decoupling is asserted but never demonstrated through systematic variation.", "reasoning": "Counterfactual comparisons (64 larger experts with R4 vs. 256 smaller experts with R1) are needed to determine if the mechanism of reuse itself, or merely parameter allocation, drives performance gains.", "judgment": "The causal attribution of performance gains is unclear because the paper has not isolated the architectural degree of freedom from parameter-counting effects.", "valence": "negative", "suggested_improvement": "Conduct counterfactual comparisons varying expert size and reuse factor to isolate the contribution of the reuse mechanism versus parameter allocation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8392469882965088, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8093892335891724}, {"unit_index": 2, "inspected_object": "Comparison against external baselines in Table 3", "observation": "Baselines differ in training data and configurations, making it difficult to isolate the contribution of expert reuse.", "reasoning": "The standard for demonstrating a new method's contribution is minimal intervention on existing strong baselines rather than bespoke training runs with multiple differing dimensions.", "judgment": "The empirical validation lacks controlled comparisons necessary to attribute improvements specifically to the proposed method.", "valence": "negative", "suggested_improvement": "Apply ReXMoE to an existing MoE architecture to provide more controlled comparisons isolating the contribution of expert reuse.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8729994297027588, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7466889023780823}, {"unit_index": 3, "inspected_object": "Potential for expert collapse at higher reuse factors", "observation": "The reviewer suspects that cross-layer sharing may exacerbate expert collapse, a known failure mode in MoE where a few experts dominate.", "reasoning": "This concern arises from modeling the method's behavior under stress and exploring fundamental scaling limits, indicating a need to test robustness beyond reported results.", "judgment": "The method's robustness at higher reuse factors is uncertain and potentially fragile due to risk of expert collapse.", "valence": "negative", "suggested_improvement": "Explore fundamental scaling limits and investigate interactions with routing mechanisms, capacity factors, auxiliary losses, and expert granularities to stress-test the method.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8216334581375122, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8471978902816772}]}, {"review_id": "02Co0Vt4ra", "paper_id": "jVyUlri4Rw", "paper_title": "Judge's Verdict: A Comprehensive Analysis of LLM Judge Capability Through Human Agreement", "decision": "Reject", "summary": "The reviewer conducts a methodological audit focusing on construct validity, identifying significant concerns regarding scale isomorphism, threshold arbitrariness, and baseline generalization while acknowledging the paper's novelty and structural clarity.", "units": [{"unit_index": 0, "inspected_object": "Scale mapping between human raters (0/0.5/1.0) and LLM judges (0/2/4 normalized to [0,1])", "observation": "The reviewer identifies a discrepancy in the measurement scales used for human versus model judgments, noting that the LLM rubric is RAGAS-style and normalized.", "reasoning": "If the two scales are not isomorphic after normalization, agreement statistics like Cohen's kappa may be systematically biased rather than reflecting true alignment.", "judgment": "The validity of the reported agreement metrics is uncertain due to potential scale incongruence.", "valence": "negative", "suggested_improvement": "Provide the exact operational mapping between human labels and LLM scores at scoring time to verify if computations compare like with like.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8625952005386353, "reasoning_key": "construct_validity", "reasoning_sim": 0.7719530463218689}, {"unit_index": 1, "inspected_object": "Threshold choices for filtering (r ≥ 0.80) and classification (|z| < 1 vs z > 1)", "observation": "The reviewer notes that the thresholds are motivated but appear ad-hoc, lacking principled decision-theoretic analysis or preregistration.", "reasoning": "Methodological hygiene requires that classification cutoffs in benchmark papers should either be derived from first principles or pre-committed before data analysis to avoid post-hoc fitting.", "judgment": "The threshold decisions represent a rigor deficit against standards for pre-committed analytic choices.", "valence": "negative", "suggested_improvement": "Clarify whether thresholds were preregistered or provide a decision-theoretic justification for their selection.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8340147137641907, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7673096656799316}, {"unit_index": 2, "inspected_object": "Human baseline agreement (κ = 0.801)", "observation": "The reviewer observes that a single aggregate human-agreement figure is treated as near-universal across all datasets and labels.", "reasoning": "If human agreement varies by dataset, label type, or item difficulty, using one static baseline for all model comparisons could lead to misclassification of models on specific subsets.", "judgment": "The use of a universal baseline is an assumption that risks overgeneralization and inaccurate relative ranking.", "valence": "negative", "suggested_improvement": "Characterize human agreement per dataset or label type rather than collapsing it into a single number.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.840508759021759, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7789462804794312}, {"unit_index": 3, "inspected_object": "Data generation stack and prompt standardization", "observation": "The reviewer inspects the provenance of RAG outputs and notes that humans and models judge outputs produced by a small set of models/prompts, raising concerns about prompt leakage and distributional bias.", "reasoning": "Systematic artifacts from similar instructions or limited generation sources may inflate apparent judge competence or bias results toward specific answer styles, limiting generalizability.", "judgment": "The evaluation's generalizability is uncertain due to potential in-distribution bias and prompt leakage risks.", "valence": "negative", "suggested_improvement": "Assess whether prompts inadvertently encode expected answer formats and clarify if the evaluation set spans diverse generation conditions.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8351899981498718, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.796956479549408}, {"unit_index": 4, "inspected_object": "Novelty and tier clarity", "observation": "The reviewer acknowledges the paper moves beyond correlation to agreement-aware validation and finds the tier classifications clear and actionable.", "reasoning": "The contribution addresses a specific gap in methodological validation, and the output structure provides practical utility.", "judgment": "The paper demonstrates genuine novelty and presents a useful framework.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8742910623550415, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7896324992179871}]}, {"review_id": "02FUPOSAyS", "paper_id": "yPfBP3EFGG", "paper_title": "STNAdam: Stochastic Two-track Nesterov-Accelerated Adaptive Momentum Estimation", "decision": "Reject", "summary": "The reviewer conducts a rigorous consistency audit, systematically identifying gaps between claimed mechanisms (Adam-style, Nesterov) and actual implementations, and highlighting theoretical assumptions that do not cover experimental settings or practical constraints.", "units": [{"unit_index": 0, "inspected_object": "STNAdam's second moment update (scalar vs coordinate-wise)", "observation": "The algorithm uses a scalar second moment rather than the coordinate-wise scaling found in canonical Adam.", "reasoning": "Canonical Adam provides adaptive preconditioning via coordinate-wise scaling; a scalar version is insufficient for anisotropic gradients and does not fulfill the 'Adam-style' claim.", "judgment": "The 'Adam-style' label is misleading because the defining mechanism of adaptive preconditioning is absent.", "valence": "negative", "suggested_improvement": "Use coordinate-wise second moments to genuinely claim Adam-style adaptivity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7977195382118225, "reasoning_key": "design_justification", "reasoning_sim": 0.8482254147529602}, {"unit_index": 1, "inspected_object": "STNAdam's extrapolation track gradient evaluation", "observation": "The extrapolation track reuses gradients from the regular update point instead of evaluating them at the extrapolated point.", "reasoning": "Nesterov acceleration relies on curvature awareness gained by evaluating gradients at the extrapolated point; without this, the two-track design offers little real benefit compared to standard methods.", "judgment": "The Nesterov acceleration claim is undermined as the implementation is largely cosmetic.", "valence": "negative", "suggested_improvement": "Evaluate gradients at the extrapolated point to genuinely claim Nesterov acceleration.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8010852932929993, "reasoning_key": "design_justification", "reasoning_sim": 0.7235240936279297}, {"unit_index": 2, "inspected_object": "ℓ₁/₂ norm weak convexity assumption", "observation": "The paper claims the ℓ₁/₂ norm is weakly convex, but its curvature diverges near zero.", "reasoning": "Theoretical proofs rely on weak convexity; if the experimental penalty violates this mathematical property, the theory does not cover the experiments, creating a domain mismatch.", "judgment": "The theoretical guarantee is invalid for the experimental setup due to formal defect.", "valence": "negative", "suggested_improvement": "Use a penalty like MCP or SCAD that satisfies weak convexity, or prove results for ℓ₁/₂ directly.", "support_status": "mixed", "confidence": "high", "object_key": "theory", "object_sim": 0.8149868845939636, "reasoning_key": "design_justification", "reasoning_sim": 0.759202241897583}, {"unit_index": 3, "inspected_object": "KŁ preservation under expectation in Lemma 5", "observation": "The proof assumes the expected objective inherits the KŁ property, ignoring stochastic dependence and variance-reduced estimator structure.", "reasoning": "Deterministic KŁ arguments do not automatically extend to stochastic settings with dependent samples; the argument is purely formal and glosses over technical difficulties.", "judgment": "The KŁ preservation assumption is unjustified for the stochastic setting described.", "valence": "negative", "suggested_improvement": "Clarify under what conditions KŁ preservation holds for dependent estimators.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8104274868965149, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7348756194114685}, {"unit_index": 4, "inspected_object": "Convergence rate claims (abstract vs theorems)", "observation": "The abstract promises 'almost sure' convergence, while the theorems only deliver 'expected' convergence.", "reasoning": "There is a rhetorical overstatement where the abstract claims stronger convergence properties than the formal results support.", "judgment": "The abstract is misleading regarding the strength of the convergence guarantees.", "valence": "negative", "suggested_improvement": "State expected convergence in the abstract if almost sure convergence is not proven.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8400101661682129, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7677775621414185}, {"unit_index": 5, "inspected_object": "μ < 0.5 constraint in theory", "observation": "The theoretical results require μ < 0.5, which conflicts with practical settings where real-world Adam users operate.", "reasoning": "If the theory requires a parameter regime not used in practice, the theoretical results may not transfer to the intended application domain.", "judgment": "The theoretical assumptions are practically incompatible with typical usage.", "valence": "negative", "suggested_improvement": "Relax the μ < 0.5 constraint or justify why it is acceptable for the target applications.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8414278030395508, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7813820838928223}, {"unit_index": 6, "inspected_object": "Random parameter sampling specification", "observation": "The paper specifies random parameters γ, α, λ drawn from state-dependent intervals that can become invalid, without detailing distributions.", "reasoning": "Underspecified randomness prevents reproducibility and raises concerns about whether the intervals are well-defined during execution.", "judgment": "The algorithmic specification is genuinely underspecified for practical implementation.", "valence": "negative", "suggested_improvement": "Provide full details on how random parameters are sampled and their distributions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.859882652759552, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8275936245918274}, {"unit_index": 7, "inspected_object": "Experimental methodology (ablation and baselines)", "observation": "Missing ablation studies and unreported baseline hyperparameters.", "reasoning": "Without ablations, it is impossible to attribute gains to specific components; without baseline details, comparisons are not reproducible or fair.", "judgment": "The empirical claims lack necessary attribution and reproducibility details.", "valence": "negative", "suggested_improvement": "Provide ablation results to demonstrate which component drives gains and include full hyperparameter details.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.903099536895752, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8360925316810608}]}, {"review_id": "02aNvwp51W", "paper_id": "y5GYHWCmco", "paper_title": "TAME: A Task-Agnostic Framework for Robust Graph Neural Network Explanations via Structural Mixup", "decision": "Reject", "summary": "The reviewer acknowledges rigorous theoretical foundations but raises substantial concerns regarding novelty relative to prior art, internal consistency of claims versus mechanisms, and methodological transparency in experiments and baselines.", "units": [{"unit_index": 0, "inspected_object": "Theoretical derivation of the task-agnostic loss function via GIB reformulation through InfoNCE", "observation": "The reviewer explicitly praises the theoretical foundation, noting the derivation is rigorous.", "reasoning": "Deep engagement with the mathematical framing confirms internal validity and soundness of the core methodological contribution.", "judgment": "Positive assessment of the paper's theoretical soundness; the derivation is accepted as a strong point without qualification.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8563061952590942, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8008623719215393}, {"unit_index": 1, "inspected_object": "Novelty claim relative to MixupExplainer and RegExplainer", "observation": "The reviewer frames TAME's contribution as potentially an incremental improvement on MixupExplainer's structural sampling step rather than a distinct mechanism.", "reasoning": "A valid contribution must offer a distinct mechanism or insight beyond recombination of existing components; the current description fails to clearly differentiate TAME from prior art (MixupExplainer/RegExplainer).", "judgment": "Substantive concern regarding novelty; the claimed contribution is viewed as insufficiently distinct from nearest predecessors.", "valence": "negative", "suggested_improvement": "Provide detailed clarification of how TAME differs substantially from MixupExplainer and how the InfoNCE-based loss differs from RegExplainer's loss.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.8835334181785583, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8197890520095825}, {"unit_index": 2, "inspected_object": "Terminological consistency (interpretability vs. explainability)", "observation": "The paper uses 'interpretability' where 'explainability' is used in the cited taxonomic survey and community conventions.", "reasoning": "Terminological discipline serves as a proxy for conceptual rigor and alignment with the explainability literature's established distinctions.", "judgment": "Negative judgment on terminological precision; the paper should align with subfield conventions to demonstrate proper grasp of the field.", "valence": "negative", "suggested_improvement": "Consistently use 'explainability' rather than 'interpretability' to align with community standards.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7912030220031738, "reasoning_key": "novelty_standard", "reasoning_sim": 0.719989001750946}, {"unit_index": 3, "inspected_object": "Internal consistency between motivation and mechanism (soft-mask usage)", "observation": "The paper criticizes soft-mask methods for failing to drive edge weights toward binary values, yet TAME itself uses an MLP-generated soft mask.", "reasoning": "An apparent contradiction exists between the stated problem framing (motivation) and the chosen solution design (mechanism); this undermines the narrative that TAME solves the specific limitations it identifies.", "judgment": "Negative judgment on architectural consistency; the critique suggests the problem framing may be rhetorical if the method employs the same problematic mechanism.", "valence": "negative", "suggested_improvement": "Clarify the distinction between the soft-mask approach criticized in prior work and the soft-mask mechanism used in TAME.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8630166053771973, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7631439566612244}, {"unit_index": 4, "inspected_object": "Scope of the 'task-agnostic' claim", "observation": "The reviewer questions whether the framework's agnosticism is a deep property or an artifact of the specific tasks evaluated (node-level classification/regression).", "reasoning": "Testing the boundaries of generalizability claims requires demonstrating applicability beyond immediate experiments to assess true impact and generality.", "judgment": "Uncertain/Negative judgment on scope; the extent of the task-agnostic capability is unverified and potentially limited.", "valence": "conditional", "suggested_improvement": "Discuss or demonstrate the extension of TAME to other tasks such as node-level classification or edge-level link prediction.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7891743183135986, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8666438460350037}, {"unit_index": 5, "inspected_object": "Methodological transparency of the Sampling Neighbors procedure", "observation": "The reviewer expresses uncertainty about how negative neighbors are selected in the Sampling Neighbors procedure.", "reasoning": "Reproducibility and trust require complete specification of algorithmic details; insufficient description prevents verification of the method.", "judgment": "Negative judgment on methodological transparency; the lack of detail hinders understanding and replication.", "valence": "negative", "suggested_improvement": "Provide more detail on the sampling procedure, specifically how negative neighbors are selected.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8357334733009338, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.9017451405525208}, {"unit_index": 6, "inspected_object": "Fairness and validity of experimental comparisons with baselines", "observation": "The reviewer questions whether baseline architectures (MixupExplainer, RegExplainer) were reimplemented faithfully or unified under a shared structure.", "reasoning": "Performance gains are only attributable to the proposed method if implementation asymmetries are ruled out; faithful or unified baselines are required for valid comparison.", "judgment": "Negative judgment on experimental validity; potential threat to the central empirical claim due to unclear baseline implementation.", "valence": "negative", "suggested_improvement": "Clarify whether baselines were reimplemented faithfully or unified under a shared explainer structure.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.9022631049156189, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8427449464797974}]}, {"review_id": "02agbFQClz", "paper_id": "kT8ZJzf6Jb", "paper_title": "Judging with Confidence: Calibrating Autoraters to Preference Distributions", "decision": "Reject", "summary": "The reviewer critically evaluates the paper's empirical support for calibrated probabilistic judgments, identifying significant gaps in the synthetic target's representativeness, train-test overlap risks, missing parsability metrics, and lack of downstream validation.", "units": [{"unit_index": 0, "inspected_object": "Synthetic ground truth construction (single teacher, personas, weights, temperature)", "observation": "The target distribution is constructed using a single teacher model with hard-coded persona weights and high temperature.", "reasoning": "These specific design choices are ad-hoc and arbitrary; if the target is unprincipled, the benchmark for calibration is also unprincipled.", "judgment": "The target distribution lacks representativeness and generalizability.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8422195911407471, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7509788274765015}, {"unit_index": 1, "inspected_object": "Evaluation protocol (train/test data overlap)", "observation": "Headline MSE and ECE results are computed on a test set generated by the same pipeline used for training.", "reasoning": "This overlap prevents distinguishing between genuine calibration improvement and imitation of the teacher's quirks or overfitting to the data-generating process.", "judgment": "The headline results demonstrate imitation rather than robust generalization.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7958803176879883, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7827522158622742}, {"unit_index": 2, "inspected_object": "RL training parsability rates", "observation": "The paper claims the reward drives parsability toward 1 but never reports the empirical parsability rate.", "reasoning": "A method relying on a parser is only as reliable as its success rate; missing this statistic leaves the method's practical robustness unclear.", "judgment": "The reliability of the parsing mechanism is unsupported.", "valence": "negative", "suggested_improvement": "Report empirical parsability rates.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8335901498794556, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8185157179832458}, {"unit_index": 3, "inspected_object": "Output distribution shape", "observation": "The reviewer expects sparsed, discontinuous pseudo-probabilities rather than true calibrated probabilities based on prior experience.", "reasoning": "If outputs are discrete scores, the model is imitating numerical output behavior rather than learning to represent uncertainty mathematically.", "judgment": "The probabilistic nature of the outputs is suspect and requires inspection at the individual prediction level.", "valence": "uncertain", "suggested_improvement": "Show the actual output distribution of individual predictions.", "support_status": "mixed", "confidence": "medium", "object_key": "theory", "object_sim": 0.7786698341369629, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7588106989860535}, {"unit_index": 4, "inspected_object": "Downstream experiments", "observation": "No downstream tasks (e.g., risk management) are tested despite the paper's motivation.", "reasoning": "If calibrated probabilities provide tangible benefits, they should improve application pipelines; without this evidence, the practical utility is unverified.", "judgment": "The claimed practical benefit is not demonstrated.", "valence": "negative", "suggested_improvement": "Add downstream experiments demonstrating practical benefit.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.771335244178772, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7952350378036499}, {"unit_index": 5, "inspected_object": "Comparison against BT-style baselines", "observation": "The framework's independence assumption is not compared against Bradley-Terry style baselines.", "reasoning": "It is unclear if the framework's assumptions are a liability compared to standard pairwise comparison methods.", "judgment": "The comparative advantage regarding independence assumptions is unverified.", "valence": "negative", "suggested_improvement": "Compare against BT-style baselines.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8645258545875549, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8152604699134827}]}, {"review_id": "02eTihT6R6", "paper_id": "pXYwksqDyE", "paper_title": "HyperClick: Advancing Reliable GUI Grounding via Uncertainty Calibration", "decision": null, "summary": "The reviewer conducts a module-level audit focusing on the practical significance of the primary contribution and the robustness of key modeling assumptions. They accept the experimental rigor but judge the effect size of the core innovation as too small and the center-confidence assumption as insufficiently validated against realistic data distributions.", "units": [{"unit_index": 0, "inspected_object": "The magnitude of improvement provided by the confidence reward component, as reported in ablation studies.", "observation": "The ablation results show very small differences between configurations, specifically a marginal 0.5% improvement attributed to the confidence reward.", "reasoning": "The evaluative standard applied is proportionality: the central mechanism (confidence reward) should produce a non-trivial effect commensurate with the paper's stated importance and framing of the problem as 'critical'. A 0.5% gain is judged insufficient to justify the contribution level, reflecting a concern for practical rather than just statistical significance.", "judgment": "The core contribution is underwhelming relative to its claims; the effect size does not support a high contribution rating.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7527774572372437, "reasoning_key": "design_justification", "reasoning_sim": 0.8080549240112305}, {"unit_index": 1, "inspected_object": "The modeling assumption that peak confidence occurs at the geometric center of the bounding box.", "observation": "The truncated Gaussian confidence model relies on the assumption that labels are at the exact center, while human annotation practices may place them elsewhere.", "reasoning": "The standard applied is robustness to assumption violation. The reviewer reasons from a general prior about data generation (humans click approximate locations) to a specific threat to generalizability. If the assumption does not hold for plausible subsets of data, the model's validity is undermined.", "judgment": "The assumption is potentially fragile and undermines the claim of robustness/generalizability.", "valence": "negative", "suggested_improvement": "Conduct an experiment (sensitivity analysis) to demonstrate how performance degrades when the center assumption is violated.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8119551539421082, "reasoning_key": "design_justification", "reasoning_sim": 0.8650779724121094}]}, {"review_id": "02mwa2qshI", "paper_id": "lDA2C9nJg6", "paper_title": "Fire on Motion: Optimizing Video Pass-bands for Efficient Spiking Action Recognition", "decision": "Reject", "summary": "The reviewer audits the consistency between the paper's headline claims (minimal parameters, specific mechanism utility) and its experimental evidence, identifying ambiguities in parameter tuning, missing controls for mechanism necessity, and limited scope validation.", "units": [{"unit_index": 0, "inspected_object": "The paper's claim of having only two learnable parameters versus the experimental ablation of amplitude parameter A.", "observation": "Table 4 shows that amplitude parameter A materially changes accuracy, yet the paper emphasizes 'two learnable scalars'.", "reasoning": "If A requires per-dataset tuning or search to maintain performance, the method is not truly a 'two-parameter' solution in deployment, undermining claims of training efficiency and fair comparison.", "judgment": "The framing of the method as strictly 'two-scalar' is potentially misleading regarding its actual complexity and tuning requirements.", "valence": "negative", "suggested_improvement": "Clarify whether A (and φ) are fixed or tuned per dataset to disambiguate between learned parameters and hyperparameters.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7621275782585144, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7827149629592896}, {"unit_index": 1, "inspected_object": "The necessity of the time-varying pre-filter design compared to simpler alternatives.", "observation": "The paper contrasts PBO+LIF against LIF-only but does not test a time-invariant learnable band-pass filter.", "reasoning": "Without a control condition isolating the time-varying aspect, it is unclear if the proposed complexity earns its keep or if a simpler time-invariant design would achieve similar results.", "judgment": "The specific contribution of the time-varying mechanism remains unproven due to lack of minimal-pair comparison.", "valence": "negative", "suggested_improvement": "Include an ablation or comparison with a time-invariant learnable band-pass filter to isolate the contribution of the time-varying design.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8158843517303467, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8071532249450684}, {"unit_index": 2, "inspected_object": "Expository clarity regarding technical definitions.", "observation": "The term 'DC' (direct current) is used before being defined in the text.", "reasoning": "Definitions should precede usage to ensure self-contained clarity for the expert audience.", "judgment": "Minor presentation flaw indicating a lack of polish in expository ordering.", "valence": "negative", "suggested_improvement": "Define 'DC' before using the term in the manuscript.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8255600929260254, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7496266961097717}, {"unit_index": 3, "inspected_object": "Robustness of gains to standard hyperparameter choices.", "observation": "The review notes a gap in testing sensitivity to clip length, stride, and resolution.", "reasoning": "Understanding robustness to these settings is critical for assessing practical deployability and whether gains are fragile.", "judgment": "Uncertainty about the method's stability under varying standard hyperparameter configurations.", "valence": "conditional", "suggested_improvement": "Conduct sensitivity analysis on clip length, stride, and resolution.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8974974155426025, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8114690184593201}, {"unit_index": 4, "inspected_object": "Scalability and generalizability to larger, temporally complex datasets.", "observation": "The paper does not evaluate performance on Something-Somethingv2 (SSv2).", "reasoning": "SSv2 requires motion understanding rather than static appearance, providing a direct test of the pass-band mismatch claim for dynamic tasks.", "judgment": "Limited evidence of generalizability to temporal-related benchmarks.", "valence": "uncertain", "suggested_improvement": "Evaluate the method on larger datasets, specifically Something-Somethingv2, to test scalability and temporal generalization.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8505703210830688, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7268002033233643}]}, {"review_id": "030gf86bqM", "paper_id": "orwQupGIl0", "paper_title": "DFCA: Decentralized Federated Clustering Algorithm", "decision": null, "summary": "The reviewer provides qualified appreciation for the paper's technical competence and theoretical soundness but systematically challenges its novelty relative to IFCA, practical viability on low-power devices, privacy guarantees, and the adequacy of its empirical validation and presentation.", "units": [{"unit_index": 0, "inspected_object": "Novelty of the proposed approach relative to IFCA", "observation": "The reviewer identifies significant similarities between the proposed decentralized clustering method and the existing IFCA baseline.", "reasoning": "The reviewer applies an implicit standard that a clustering method inheriting IFCA's core mechanism must demonstrate that the decentralization itself constitutes a substantial, algorithmic contribution rather than just an architectural shift. The magnitude of the delta is deemed insufficient for novelty.", "judgment": "The contribution lacks sufficient incremental value and novelty.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8193241953849792, "reasoning_key": "design_justification", "reasoning_sim": 0.7917079329490662}, {"unit_index": 1, "inspected_object": "Memory footprint and deployment feasibility on low-powered devices", "observation": "The algorithm requires storing all cluster models on each client, which leads to exploding memory costs as the number of clusters grows.", "reasoning": "The reviewer contrasts this resource requirement with the practical constraints of low-powered IoT frameworks, where storage capacity is limited, finding the method incompatible with the claimed real-world decentralized network settings.", "judgment": "The method is impractical for deployment in the target low-power environments.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.796235978603363, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7546419501304626}, {"unit_index": 2, "inspected_object": "Privacy implications of degenerate clusters", "observation": "The algorithm allows for degenerate clusters (single-client clusters), causing the cluster model to coincide with the individual client's model.", "reasoning": "This architecture implies that a client stores another client's trained model, which acts as a proxy for their data distribution. This violates standard FL privacy assumptions that clients should not have access to other clients' model parameters.", "judgment": "The design presents a significant privacy risk due to potential exposure of other clients' information via their models.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7668788433074951, "reasoning_key": "design_justification", "reasoning_sim": 0.7393894791603088}, {"unit_index": 3, "inspected_object": "Hyperparameter sensitivity analysis for the number of clusters (k)", "observation": "The paper fixes the number of clusters a priori and only evaluates performance with k=4.", "reasoning": "Since the number of clusters is a user-specified input, robustness should be demonstrated across a range of cluster counts. Evaluating only one value raises concerns about cherry-picking or lack of robustness.", "judgment": "The experimental validation is under-parameterized and fails to demonstrate robustness to key hyperparameters.", "valence": "negative", "suggested_improvement": "Evaluate performance across a range of cluster counts to demonstrate robustness.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8796281814575195, "reasoning_key": "design_justification", "reasoning_sim": 0.8132428526878357}, {"unit_index": 4, "inspected_object": "Reporting precision regarding data heterogeneity", "observation": "The paper claims robustness to data heterogeneity but does not specify the Dirichlet alpha value used to construct the federation.", "reasoning": "Data heterogeneity is a spectrum, not a binary property. Without specifying the parameter, the claim of robustness is under-specified and cannot be properly evaluated or replicated.", "judgment": "The experimental reporting is imprecise and incomplete regarding critical settings.", "valence": "negative", "suggested_improvement": "Specify the exact Dirichlet alpha value and heterogeneity levels tested.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8418152332305908, "reasoning_key": "design_justification", "reasoning_sim": 0.8117009997367859}, {"unit_index": 5, "inspected_object": "Comparison baselines in the experimental evaluation", "observation": "The paper compares against generic decentralized FL algorithms but omits other clustering DFL algorithms or methods specifically designed to mitigate data heterogeneity.", "reasoning": "To establish the value proposition of the proposed method, it must show superiority against the strongest available competitors, particularly those also designed to handle heterogeneity. Generic comparisons are insufficient.", "judgment": "The benchmarking is inadequate and fails to clearly position the contribution's advantage.", "valence": "negative", "suggested_improvement": "Include comparisons to other clustering DFL algorithms and methods designed to mitigate data heterogeneity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.919800877571106, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8578949570655823}, {"unit_index": 6, "inspected_object": "Placement of the related work section", "observation": "The related work section is placed at the end of the experimental results.", "reasoning": "This placement is unconventional and compromises readability. It is interpreted as a structural error rather than a deliberate choice, violating standard norms for paper organization.", "judgment": "The paper's structure is confusing and negatively impacts readability.", "valence": "negative", "suggested_improvement": "Move the related work section to a more conventional location (e.g., after the introduction).", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8166959881782532, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8279412984848022}, {"unit_index": 7, "inspected_object": "Pseudocode clarity regarding local SGD iterations and communication rounds", "observation": "The pseudocode appears to equate the number of local SGD iterations with the number of communication rounds.", "reasoning": "In standard FL, these quantities are decoupled. This conflation suggests either a typographical error or a conceptual confusion in the algorithm's specification, undermining the clarity and precision of the presentation.", "judgment": "The algorithmic presentation is ambiguous and potentially misleading.", "valence": "negative", "suggested_improvement": "Clarify the distinction between local update steps and communication rounds in the pseudocode.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8293823599815369, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8101336359977722}, {"unit_index": 8, "inspected_object": "Theoretical extension using algebraic graph theory", "observation": "The reviewer acknowledges the theoretical extension of IFCA's proofs using algebraic graph theory.", "reasoning": "While the reviewer finds the technical execution sound and the paper well-motivated, they do not consider this theoretical competence a distinguishing strength sufficient to offset the novelty and practicality concerns.", "judgment": "The theoretical work is competent and sound but does not constitute a primary strength.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8247222304344177, "reasoning_key": "novelty_standard", "reasoning_sim": 0.838972270488739}]}, {"review_id": "032W49Rni8", "paper_id": "a8uipkMIZN", "paper_title": "A Theoretical Framework for Escaping Local Optima in MSE toward Global Convergence", "decision": "Reject", "summary": "The reviewer conducts a verificatory analysis focusing on the mismatch between the abstract's theoretical claims and the body's lack of formal proofs. They challenge the paper's theoretical foundation by citing conflicting prior literature, question novelty due to poor positioning against existing methods, and critique empirical sufficiency based on limited scope and lack of generalization gains.", "units": [{"unit_index": 0, "inspected_object": "Abstract claim of theoretical guarantees vs. body content", "observation": "The abstract asserts 'theoretical guarantees for avoiding spurious local traps,' but the body contains no formal theorems, assumptions, or proofs establishing convergence or non-vanishing gradients beyond heuristics.", "reasoning": "A paper using the phrase 'theoretical guarantees' must provide formal apparatus (theorems, assumptions, proofs). The absence of these constitutes a fundamental soundness failure, not merely a stylistic issue.", "judgment": "The paper fails to substantiate its central theoretical claim, resulting in a low soundness assessment.", "valence": "negative", "suggested_improvement": "Provide formal theorems and proofs establishing convergence or non-vanishing gradients.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8360135555267334, "reasoning_key": "design_justification", "reasoning_sim": 0.8104054927825928}, {"unit_index": 1, "inspected_object": "Paper's taxonomy of local optima (Cases 1-3) relative to prior literature", "observation": "Zhou et al. proved that under MSE loss in an unconstrained feature model, there are no spurious saddle points and every local minimum is global. The reviewer posits that this result may cause the paper's Cases 2 and 3 to collapse into Case 1, rendering the taxonomy vacuous.", "reasoning": "If a prior theorem establishes that all local minima are global in a relevant setting, the paper's identification of 'spurious local optima' must either be wrong, rely on different assumptions, or be a trivial restatement. The paper has not reconciled its claims with this existing theoretical result.", "judgment": "The paper's theoretical contribution is undermined by a lack of consistency with established prior results, creating uncertainty about the validity of its motivating problem.", "valence": "negative", "suggested_improvement": "Provide a formal basis, such as a counterexample, differing setting, additional constraints, and corresponding proof, to reconcile the paper's taxonomy with Zhou et al.'s findings.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8472608327865601, "reasoning_key": "design_justification", "reasoning_sim": 0.8283361196517944}, {"unit_index": 2, "inspected_object": "Method positioning against existing optimization techniques", "observation": "The proposed QMSE method resembles perturbed/annealed gradients, gradient noise injection, sharpness-aware, or entropy-based objectives, yet the paper fails to position itself within these families.", "reasoning": "A new method must explain what it shares with known families, what it adds, and why the differences matter. The failure to perform this comparative mapping raises concerns about the novelty of the contribution.", "judgment": "Novelty concerns arise due to the lack of clear differentiation from existing methods.", "valence": "negative", "suggested_improvement": "Position the method against related families (perturbed gradients, sharpness-aware objectives) and explicitly articulate distinctiveness.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.822736382484436, "reasoning_key": "design_justification", "reasoning_sim": 0.8066637516021729}, {"unit_index": 3, "inspected_object": "Experimental scope and baseline comparisons", "observation": "Experiments use simple CNNs on small datasets (MNIST, CIFAR-10, CIFAR-100) without modern architectures (e.g., ResNet-18) or key baselines (Adam, RMSProp, SGD+noise, Lookahead, SAM, label smoothing, CE).", "reasoning": "A method claiming improved optimization should be tested against a broad set of optimizers known to address similar issues (escaping sharp/degenerate regions). The limited scope and missing baselines weaken the empirical evidence.", "judgment": "Empirical evidence is insufficient to support general claims of optimization improvement.", "valence": "negative", "suggested_improvement": "Include broader baselines and test on modern architectures to demonstrate robustness and scalability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.9147666096687317, "reasoning_key": "design_justification", "reasoning_sim": 0.8663412928581238}, {"unit_index": 4, "inspected_object": "Training vs. validation accuracy gains", "observation": "Gains are primarily in training accuracy, while validation accuracy remains similar to MSE, and the model underfits badly on CIFAR-100.", "reasoning": "The purpose of escaping local optima is ultimately better generalization, not just better training fit. Limited evidence of improved validation accuracy suggests the method does not significantly aid useful convergence.", "judgment": "The practical utility of the method is questionable due to lack of generalization improvements.", "valence": "negative", "suggested_improvement": "Demonstrate improvements in validation accuracy or generalization metrics, not just training performance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8336033225059509, "reasoning_key": "design_justification", "reasoning_sim": 0.8713783621788025}, {"unit_index": 5, "inspected_object": "Writing quality and notation consistency", "observation": "The paper contains typographical errors ('mean squrared', 'modi-fied'), grammatical issues, and inconsistent notation formatting.", "reasoning": "Presentation errors and ambiguity impair readability and signal a lack of rigor, contributing to a negative assessment of the paper's overall quality.", "judgment": "Low presentation score due to significant writing and formatting issues.", "valence": "negative", "suggested_improvement": "Correct typos, fix grammar, and standardize notation formatting.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8542515635490417, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8622226715087891}]}, {"review_id": "037uFeCinx", "paper_id": "K6AYxvp3jl", "paper_title": "Polysemous Language Gaussian Splatting via Matching-based Mask Lifting", "decision": null, "summary": "The reviewer systematically challenges the paper's conceptual foundations by verifying claims against prior literature, assessing novelty through component provenance, questioning the linkage between motivation and design, and critiquing the interpretability of performance gains due to insufficient ablations.", "units": [{"unit_index": 0, "inspected_object": "Claim that contrastive methods lack one-to-one assignment capability and only encode relativeness (W1-1)", "observation": "Reviewer asserts that prior contrastive learning inherently encodes 'relativity' rather than strict one-to-one assignment, allowing classification of Gaussians as 'chair' if relatively closer to that embedding than others.", "reasoning": "The reviewer uses empirical evidence from ScanNet20/200 benchmarks where prior methods assign different categories to Gaussians within the same scene, refuting the paper's characterization of a fundamental limitation in existing work.", "judgment": "The paper's motivation based on this claimed limitation is undermined by inaccurate characterization of prior capabilities.", "valence": "negative", "suggested_improvement": "Correct the description of how contrastive methods handle assignment and relativity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8377853035926819, "reasoning_key": "design_justification", "reasoning_sim": 0.803877592086792}, {"unit_index": 1, "inspected_object": "Claim that Dr.Splat trains a per-scene feature compressor (W2-1, W2-2)", "observation": "Reviewer identifies that Dr.Splat uses Product Quantization with a codebook trained once on LVIS and reused, not per-scene training.", "reasoning": "The paper claims Dr.Splat 'trains a feature compressor for each scene' (line 148), which contradicts the reviewer's understanding of the method's global codebook approach. This mischaracterization supports the paper's motivation of being 'training-free', which the reviewer argues is false or misleading.", "judgment": "The paper's central motivation regarding training efficiency is built on an incorrect premise about prior work.", "valence": "negative", "suggested_improvement": "Accurately describe Dr.Splat's mechanism and its training scope.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7037743926048279, "reasoning_key": "design_justification", "reasoning_sim": 0.7885742783546448}, {"unit_index": 2, "inspected_object": "Novelty of Section 3.2 (Object-level grouping)", "observation": "Section 3.2 is described as identical to Gaussian Grouping [D] with only minor implementation details differing.", "reasoning": "The reviewer applies a standard that requires identifiable high-level conceptual differences for novelty. Since no such difference is found, the component is deemed not novel despite better performance.", "judgment": "Limited conceptual contribution due to lack of high-level distinction from Gaussian Grouping.", "valence": "negative", "suggested_improvement": "Articulate the high-level conceptual difference between the proposed method and Gaussian Grouping [D].", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8286501169204712, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8665581345558167}, {"unit_index": 3, "inspected_object": "Novelty of Section 3.3 (Instance feature extraction)", "observation": "Section 3.3 is highly similar to Mosaic3D's data generation pipeline [A] and applicable to point clouds, not specialized to 3D Gaussians.", "reasoning": "The reviewer evaluates novelty based on whether a technique is specialized to the domain or merely a general application of existing pipelines. The similarity to Mosaic3D and general applicability suggest limited specific contribution.", "judgment": "Limited novelty as the module appears to be a generic application of existing data generation techniques.", "valence": "negative", "suggested_improvement": "Clarify the specific innovations in Section 3.3 beyond what Mosaic3D provides.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8097680807113647, "reasoning_key": "design_justification", "reasoning_sim": 0.8386610150337219}, {"unit_index": 4, "inspected_object": "Linkage between stated motivations and proposed solution design", "observation": "The reviewer finds no clear link between the three identified flaws in prior work (per-scene retraining, monosemous design, cross-view inconsistencies) and the proposed Gaussian grouping + text assignment pipeline.", "reasoning": "The reviewer values conceptual coherence and justification of design choices over raw performance. The absence of a logical mapping between problems and solutions undermines the paper's theoretical grounding.", "judgment": "The proposed design lacks justification relative to the stated problems, indicating poor conceptual coherence.", "valence": "negative", "suggested_improvement": "Explicitly demonstrate how the proposed modules address the specific flaws identified in the motivation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8599680662155151, "reasoning_key": "design_justification", "reasoning_sim": 0.8337194323539734}, {"unit_index": 5, "inspected_object": "Attribution of performance gains via ablation studies", "observation": "The reviewer suspects Section 3.3 is the dominant contributor to performance gains due to known CLIP modality gaps and text-to-text search effectiveness, but the paper's ablations do not isolate this effect.", "reasoning": "Without ablations isolating the instance feature extraction module, it is impossible to determine if the gains come from the proposed Gaussian grouping or simply from using text embeddings effectively. The current ablations are insufficient for attribution.", "judgment": "Performance claims are uninterpretable because the specific contribution of key components is not isolated.", "valence": "negative", "suggested_improvement": "Conduct ablations that specifically isolate the effect of Section 3.3 to test the hypothesis regarding text embedding effectiveness.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.8421292901039124, "reasoning_key": "design_justification", "reasoning_sim": 0.8084163665771484}, {"unit_index": 6, "inspected_object": "Terminology: 'efficiency of neural techniques' (Q1-1)", "observation": "The reviewer challenges the use of 'neural techniques' to describe 3DGS, arguing rasterization and splatting are not neural.", "reasoning": "Precision in terminology matters for accurate framing. 3DGS's efficiency comes from explicit representation and rendering algorithms, not neural properties. Mislabeling suggests sloppy scientific communication.", "judgment": "The paper's framing of 3DGS's advantages is technically inaccurate due to imprecise terminology.", "valence": "negative", "suggested_improvement": "Replace 'neural techniques' with more accurate descriptions of 3DGS's explicit representation and rendering algorithm.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8194884657859802, "reasoning_key": "design_justification", "reasoning_sim": 0.8673930764198303}, {"unit_index": 7, "inspected_object": "Definition of 'training-free' claim (Q4)", "observation": "The reviewer notes Section 3.2 requires parameter optimization, conflicting with the abstract's claim of abandoning feature optimization entirely.", "reasoning": "There is a tension between the 'training-free' label and the presence of optimization steps in the grouping module. The definition of 'training time' in Table 1 is unclear.", "judgment": "Uncertainty remains regarding whether the method is truly training-free as claimed.", "valence": "uncertain", "suggested_improvement": "Clarify the definition of 'training time' and resolve the apparent contradiction between optimization requirements and the training-free claim.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7005438804626465, "reasoning_key": "design_justification", "reasoning_sim": 0.7407182455062866}, {"unit_index": 8, "inspected_object": "Evaluation breadth (LeRF and ScanNet datasets)", "observation": "The reviewer notes approvingly that the paper evaluates on LeRF and ScanNet datasets.", "reasoning": "Broad evaluation across multiple datasets is a positive aspect of experimental rigor, even if other aspects of the paper are flawed.", "judgment": "Positive assessment of the evaluation setup's breadth.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8689094185829163, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7973812818527222}]}, {"review_id": "037v07Ey1g", "paper_id": "cAoUgzEtZ0", "paper_title": "Representation Convergence: Mutual Distillation is Secretly a Form of Regularization", "decision": "Reject", "summary": "The reviewer evaluates the paper primarily through the lens of empirical validation strategy, accepting the formal proof but criticizing the limited scope (single benchmark), weak baselines, and lack of computational analysis. The core tension is between the paper's honest self-positioning as a proof-of-concept and the reviewer's expectation for broader validation to justify enthusiasm.", "units": [{"unit_index": 0, "inspected_object": "Empirical validation scope (single benchmark)", "observation": "The paper uses only the Procgen benchmark for evaluation.", "reasoning": "A method claiming to improve generalization should be tested across multiple distribution-shift benchmarks; using a single suite is necessary but not sufficient to establish robustness.", "judgment": "The narrow scope limits the confidence in the generalizability of the results, preventing an enthusiastic endorsement despite the proof-of-concept nature.", "valence": "negative", "suggested_improvement": "Test on a more diverse set of environments/benchmarks.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.835017204284668, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8705269694328308}, {"unit_index": 1, "inspected_object": "Baseline comparison quality", "observation": "The SPO baseline performance is described as not up to current standards, and there is no comparison against specific data-augmentation methods like AutoRL.", "reasoning": "If the baseline is weak, beating it is not informative; comparing against stronger baselines is required to situate the method's unique contribution within the existing toolkit.", "judgment": "The experimental setup lacks discriminative power due to weak baselines and missing comparisons, making the performance claims less compelling.", "valence": "negative", "suggested_improvement": "Compare against stronger baselines, specifically data-augmentation methods such as AutoRL.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.9078994393348694, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8243910074234009}, {"unit_index": 2, "inspected_object": "Computational overhead of mutual distillation", "observation": "Mutual distillation requires training multiple policies, raising questions about computational cost.", "reasoning": "Practical adoption depends on whether the regularization benefit justifies the additional training costs; this is a cost-benefit consideration often absent from theory-focused papers.", "judgment": "The practical viability and value proposition of the method are unclear without an analysis of its computational overhead.", "valence": "conditional", "suggested_improvement": "Provide an analysis of the computational overhead associated with training multiple policies.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.816143810749054, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7734420895576477}, {"unit_index": 3, "inspected_object": "Formal proof and lower bound", "observation": "The reviewer accepts the existence of the theorem and the robustness term at a high level without interrogating assumptions or derivation.", "reasoning": "The formal result is acknowledged as a legitimate contribution, though the engagement is superficial compared to the scrutiny applied to experiments.", "judgment": "The theoretical contribution is accepted as valid but not deeply scrutinized, serving as a strength that is contrasted with empirical limitations.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8010290861129761, "reasoning_key": "merit_recognition", "reasoning_sim": 0.766765832901001}, {"unit_index": 4, "inspected_object": "Ablation studies", "observation": "The ablations are described as 'legitimate' by the reviewer.", "reasoning": "The reviewer checked that the ablations isolate the claimed mechanism rather than merely showing the method works, indicating they passed a validity check.", "judgment": "The ablation studies successfully support the claimed mechanism.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7679039239883423, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7508243322372437}]}, {"review_id": "03R8UmUhAk", "paper_id": "bUnG3WcSYX", "paper_title": "Evolving and Detecting Multi-Turn Deception using Geometric Signatures", "decision": null, "summary": "The reviewer systematically interrogates the paper's central interpretive claim by demanding theoretical grounding, baseline comparisons, and visual verification of geometric separability, while accepting the empirical scaffolding of the data generation pipeline.", "units": [{"unit_index": 0, "inspected_object": "Geometric feature selection rationale", "observation": "The reviewer finds the intuition behind geometric features intriguing but notes a lack of theoretical or empirical support demonstrating their discriminative value beyond heuristic reasoning.", "reasoning": "The reviewer applies a standard requiring justification for why specific spatial statistics (angular coverage, distance ratio) are privileged over generic properties of multi-turn dialogue or alternative spatial statistics like manifold curvature.", "judgment": "The current claim that these features work is under-defended and relies on heuristic reasoning rather than demonstrated necessity.", "valence": "negative", "suggested_improvement": "Conduct feature ablation studies or comparisons against alternative spatial statistics to demonstrate discriminative value.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8249409794807434, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7496104836463928}, {"unit_index": 1, "inspected_object": "Dataset diversity and generalization", "observation": "Experiments focus narrowly on eliciting dangerous information about explosives without cross-domain validation.", "reasoning": "Without testing on other harmful intent categories (e.g., social manipulation, misinformation), it is unclear if the geometric footprint is a property of deception generally or specific to this task.", "judgment": "The central hypothesis that multi-turn deceptive intent leaves a stable geometric footprint is potentially overstated due to limited scope.", "valence": "negative", "suggested_improvement": "Perform cross-domain validation using different harmful intent categories such as social manipulation or misinformation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.842913806438446, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7694525718688965}, {"unit_index": 2, "inspected_object": "Interpretability and visualization of results", "observation": "The reviewer acknowledges strong numerical performance but notes the paper lacks visual evidence of separability in embedding space.", "reasoning": "The reviewer holds an epistemological stance that genuine geometric footprints should be visually apparent (via PCA/t-SNE or trajectory plots) rather than just statistically significant in classifier decision boundaries.", "judgment": "The interpretive layer is weak because the claimed separability is not visually confirmed, leaving open the possibility of statistical artifacts.", "valence": "negative", "suggested_improvement": "Visualize embedding trajectories of deceptive versus benign conversations using PCA or t-SNE plots.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8438205122947693, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7772991061210632}, {"unit_index": 3, "inspected_object": "Baseline comparisons for parsimony", "observation": "The paper does not compare the geometric approach against simpler statistical or linguistic baselines.", "reasoning": "To establish the necessity of complex geometric features, the reviewer requires evidence that cheaper, conventional signals (perplexity shifts, topic coherence) do not suffice.", "judgment": "The contribution's necessity is unproven because it is unclear if the geometric approach offers advantages over simpler baselines.", "valence": "negative", "suggested_improvement": "Compare against simpler statistical or linguistic baselines such as perplexity shifts, topic coherence measures, or dialogue entropy.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8594787120819092, "reasoning_key": "construct_validity", "reasoning_sim": 0.7898464798927307}, {"unit_index": 4, "inspected_object": "Data generation pipeline", "observation": "The reviewer explicitly praises the multi-objective genetic optimization, co-evolving mutation operators, and human validation.", "reasoning": "These components are viewed as thoughtfully designed and well-motivated, providing a solid empirical foundation for the study.", "judgment": "The data generation methodology is sound and well-executed.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.808194100856781, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7928412556648254}]}, {"review_id": "03Xl9ItGQE", "paper_id": "slCmiGEX1D", "paper_title": "One Size Does not Fit All: Cross-Architectural Layer-wise Representations in Diffusion Models", "decision": null, "summary": "The reviewer evaluates the paper primarily as an editing-method contribution, criticizing the lack of justification for metric selection and the use of non-competitive baselines, while accepting the core cross-architectural analysis as valuable but under-supported by rigorous evaluation standards.", "units": [{"unit_index": 0, "inspected_object": "Selection of four low-level similarity metrics for injection layer identification", "observation": "The paper uses four specific low-level metrics without providing justification for their selection or ruling out alternatives.", "reasoning": "The choice of metrics is load-bearing as it determines which layers are selected; methodological transparency requires justifying why these particular metrics are appropriate and considering high-level alternatives like CLIP or LPIPS.", "judgment": "The metric selection procedure lacks sufficient methodological justification.", "valence": "negative", "suggested_improvement": "Justify the selection of the four metrics, discuss whether all are necessary, and consider high-level metrics like CLIP and LPIPS as complementary options.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7918862700462341, "reasoning_key": "construct_validity", "reasoning_sim": 0.775790810585022}, {"unit_index": 1, "inspected_object": "Qualitative comparison baselines in Figure 4", "observation": "The qualitative evaluation lacks sufficient baselines, particularly for style transfer, and omits stronger DiT-specific methods such as Flux Kontext.", "reasoning": "Competitive benchmarking norms require comparing against the strongest available alternatives to demonstrate method value, especially given that DiT editing methods have advanced beyond the chosen baselines (P2P, PnP, FreePromptEdit).", "judgment": "The baseline selection is insufficiently competitive and does not reflect the current state of the field.", "valence": "negative", "suggested_improvement": "Include stronger DiT-specific baselines like Flux Kontext and increase the number of editing examples to ensure comparability across methods.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8644204139709473, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8312535881996155}]}, {"review_id": "03Z49enjft", "paper_id": "EZn2TmBBfF", "paper_title": "From Curiosity to Caution: Mitigating Reward Hacking for Best-of-$N$ with Pessimism", "decision": "Accept (Poster)", "summary": "The reviewer adopts a practitioner-oriented stance, praising the method's computational practicality, problem importance, and rigorous ablation studies while demanding stronger empirical grounding through comparative baselines, multi-modal evidence, and clarification of theoretical transferability and portability.", "units": [{"unit_index": 0, "inspected_object": "The use of reward model features for prediction error calculation versus random features.", "observation": "The reviewer found the ablation studies in Section 3.2 to be thorough and successfully validating the design choice.", "reasoning": "The reviewer identifies methodological rigor where empirical claims are closely tested, rewarding the validation of key design decisions over theoretical novelty alone.", "judgment": "Positive assessment of the experimental design's internal validity.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.828942060470581, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8363738656044006}, {"unit_index": 1, "inspected_object": "The theoretical contribution based on linear-Gaussian models.", "observation": "The reviewer questions whether insights from a linear-Gaussian model transfer to the complex, high-dimensional geometry of transformer feature spaces.", "reasoning": "The implicit standard is that theoretical results should directly illuminate the empirical setting or be explicitly positioned as illustrative rather than probative; without this connection, the theory risks being irrelevant to the actual operating setting.", "judgment": "Skeptical of the theory's practical relevance despite accepting it as motivation.", "valence": "negative", "suggested_improvement": "Explicitly position the theory as illustrative or demonstrate its direct illumination of the empirical setting.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8440008759498596, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7479316592216492}, {"unit_index": 2, "inspected_object": "Comparative standing against other inference-time uncertainty estimation techniques.", "observation": "The paper compares primarily against a single baseline (standard BoN) rather than a landscape of alternatives.", "reasoning": "New methods should be situated within a broader design space of alternatives, not just against one baseline, to establish true comparative standing.", "judgment": "The comparison is insufficient for establishing the method's relative utility.", "valence": "negative", "suggested_improvement": "Compare the method against other plausible inference-time uncertainty estimation techniques and broader reward hacking mitigation literature.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8226156234741211, "reasoning_key": "fair_comparison", "reasoning_sim": 0.75065678358078}, {"unit_index": 3, "inspected_object": "Evidence modalities beyond automated metrics.", "observation": "The review lacks qualitative examples and human evaluation.", "reasoning": "Automated metrics alone are insufficient to assess real-world usefulness; multi-modal evidence (qualitative/human) is needed to bridge the gap between metric success and practical deployment value.", "judgment": "Incomplete evidence base for claiming real-world utility.", "valence": "negative", "suggested_improvement": "Provide more qualitative examples and conduct human evaluation.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8521385788917542, "reasoning_key": "construct_validity", "reasoning_sim": 0.8153616189956665}, {"unit_index": 4, "inspected_object": "Efficiency and necessity of the separate predictor network.", "observation": "The method trains a separate predictor network instead of using simpler uncertainty estimates like Monte Carlo dropout or Mahalanobis distance.", "reasoning": "If simpler methods yield comparable performance, the added complexity of training a separate network is unnecessary; the reviewer seeks justification for this specific design choice relative to efficiency.", "judgment": "Uncertainty regarding the necessity of the current architectural complexity.", "valence": "conditional", "suggested_improvement": "Demonstrate why the separate predictor is necessary compared to simpler uncertainty estimation methods.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.823944628238678, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8089209198951721}, {"unit_index": 5, "inspected_object": "Portability of the trained predictor across different base models.", "observation": "The predictor is trained on responses from Llama-3.2-3B, raising concerns about generalization to other base models.", "reasoning": "The assumption that 'typical responses' are a property of the task rather than the generating policy is unverified; if portability requires retraining for each new model, the method's deployability is limited.", "judgment": "Concerns about the method's deployment envelope and generalizability.", "valence": "uncertain", "suggested_improvement": "Clarify how the predictor generalizes to different base models and whether retraining is required.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8255756497383118, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8147177696228027}, {"unit_index": 6, "inspected_object": "Semantic depth of the OOD signal detected by prediction error.", "observation": "Current examples focus on formatting errors rather than deeper semantic or stylistic deviations.", "reasoning": "If the method only detects style/surface-level deviations, its usefulness for catching subtle reward hacking exploits is limited; the reviewer needs evidence that the signal captures substance.", "judgment": "Uncertainty about the breadth and depth of the method's detection capabilities.", "valence": "uncertain", "suggested_improvement": "Provide semantic or stylistic examples of what the method catches to demonstrate depth of OOD detection.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7964009642601013, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.766861081123352}, {"unit_index": 7, "inspected_object": "Computational practicality of the method.", "observation": "The method is computationally practical.", "reasoning": "Practical deployability matters to the reviewer; computational feasibility supports the method's viability as a practitioner-oriented solution.", "judgment": "Positive assessment of computational efficiency.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8669520020484924, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8248769044876099}, {"unit_index": 8, "inspected_object": "Importance of the problem addressed.", "observation": "The problem of reward hacking/uncertainty is important and persistent.", "reasoning": "Addressing a significant, ongoing problem adds inherent value to the contribution, independent of the specific method used.", "judgment": "Positive assessment of the problem significance.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7337635159492493, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7605758309364319}, {"unit_index": 9, "inspected_object": "Scope of experimental evaluation.", "observation": "Experiments span multiple domains across three benchmarks.", "reasoning": "Comprehensive evaluation across diverse domains strengthens the claim of robustness and general applicability.", "judgment": "Positive assessment of experimental scope.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.804691731929779, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8342409729957581}]}, {"review_id": "00HiU6bLWX", "paper_id": "xg9Z37gn7R", "paper_title": "Provable Low-Frequency Bias of In-Context Learning of Representations", "decision": "Reject", "summary": "The reviewer performs a rigorous assumption-focused audit, identifying definitional circularity in Theorem 1, unrealistic attention weight assumptions in Definition 3, generality mismatches in layerwise analysis, insufficient justification for matrix decomposition, and a critical mathematical error in the proof of Theorem 2.", "units": [{"unit_index": 0, "inspected_object": "Theorem 1 (context-direction convergence result)", "observation": "The reviewer finds that the theorem's conclusion follows almost directly from Definitions 1 and 3, which already assume convergence of weight fractions and representations.", "reasoning": "A meaningful convergence theorem should derive convergence from more primitive or less assumption-laden conditions rather than encoding the conclusion in its premises; the current setup exhibits definitional circularity, making the result uninformative.", "judgment": "The convergence result is trivial and lacks explanatory value because it is a consequence of assumptions rather than a derived dynamic property.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8846374750137329, "reasoning_key": "design_justification", "reasoning_sim": 0.7902571558952332}, {"unit_index": 1, "inspected_object": "Definition 3 (attention weight assumption)", "observation": "The reviewer observes that Definition 3 assumes nonzero attention weights converge to token proportions, reducing attention to a counting operation.", "reasoning": "This assumption strips the attention mechanism of its adaptive or computational character, rendering the study of its 'convergence' pointless as it describes a trivial process rather than an interesting mechanism.", "judgment": "The assumption is unrealistic and fails to capture the functional complexity of actual attention behavior.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7517297863960266, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7321053743362427}, {"unit_index": 2, "inspected_object": "Layerwise analysis assumptions (general mapping and spectral gap)", "observation": "The reviewer notes that the 'great mapping' assumption constrains σ to linear bounds and cites only linear examples, despite claims covering non-linearities like FFNs.", "reasoning": "Claims of generality must match technical coverage; if the paper asserts it handles non-linearities, the assumptions should genuinely accommodate them rather than restricting to linear functions for tractability.", "judgment": "The layerwise assumptions are unnatural and represent opportunistic assumption-making that overreaches the paper's advertised scope.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8674190640449524, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.808612048625946}, {"unit_index": 3, "inspected_object": "Appendix G decomposition assumption (A,B,O,T matrices)", "observation": "The reviewer identifies that Appendix G justifies the matrix decomposition by checking per-token type membership, which is a weaker condition than the required global low-dimensional linear combination.", "reasoning": "The evidence provided (per-token checks) does not establish the strength of the assumption (global decomposability); a weaker check cannot license a stronger theoretical claim.", "judgment": "The justification for the decomposition assumption is insufficient and logically mismatched with the assumption's strength.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8378438949584961, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7419685125350952}, {"unit_index": 4, "inspected_object": "Proof of Theorem 2 (Equation 62 and line 326 condition)", "observation": "The reviewer claims a mathematical error in applying Corollary 11, stating constant terms should be inverted, and notes the resulting condition is vacuous or misnamed.", "reasoning": "As written, the condition can always be satisfied by choosing γ₁ = 0, making it trivial; if corrected, it requires an unexplained lower bound on λ_q/λ_{q+1} and uses incorrect terminology ('spectral gap').", "judgment": "The proof contains a critical error, leading to a vacuous condition that lacks motivation and proper naming.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7725393772125244, "reasoning_key": "design_justification", "reasoning_sim": 0.731019914150238}]}, {"review_id": "00rDa4Mwg8", "paper_id": "7nTKiJLkWS", "paper_title": "Efficient and Sharp Off-Policy Learning under Unobserved Confounding", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper primarily through a lens of field positioning and novelty, concluding that while the work is technically sound and clearly presented, it offers only incremental improvements over existing methods in a saturated problem space with modest practical gains.", "units": [{"unit_index": 0, "inspected_object": "Problem formulation and novelty relative to prior literature", "observation": "The reviewer identifies the problem of robust offline policy learning under sensitivity models as well-studied, citing seven prior works spanning 2013–2023.", "reasoning": "The enumeration of prior work establishes that the problem space is saturated, creating a high threshold for new contributions. The reviewer uses this boundary-setting move to frame the paper's value as needing clear differentiation from existing solutions rather than opening new territory.", "judgment": "The paper demonstrates limited novelty because it addresses a well-populated research landscape without introducing a fundamentally new problem or conceptual framework.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.8962079286575317, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8113245368003845}, {"unit_index": 1, "inspected_object": "Technical contribution relative to Kallus and Zhou (2020)", "observation": "The reviewer concedes that the paper makes a technical contribution regarding the semi-parametric efficiency of its estimator compared to Kallus and Zhou (2020).", "reasoning": "While acknowledging the theoretical soundness of the derivation from naive plug-in to efficient estimation, the reviewer frames this as a narrow improvement over one specific prior work. The contribution is viewed as incremental within the established lineage of sensitivity analysis methods.", "judgment": "The technical contribution is real but modest, representing an optimization of existing approaches rather than a transformative advance.", "valence": "mixed", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8090440034866333, "reasoning_key": "design_justification", "reasoning_sim": 0.8472420573234558}, {"unit_index": 2, "inspected_object": "Practical significance of the efficient estimator vs. naive plug-in", "observation": "The reviewer observes that the new estimator appears to provide only modest improvements over the naive plug-in estimator based on experimental results.", "reasoning": "The reviewer shifts the comparison axis from the authors' focus (minimax instability) to a practical baseline (naive plug-in). This suggests an implicit norm that theoretical efficiency gains should translate into meaningful practical advantages; if the gain is marginal, the added complexity may not be justified for practitioners.", "judgment": "The practical utility of the proposed method is questionable due to the small performance gap relative to simpler alternatives.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8408096432685852, "reasoning_key": "design_justification", "reasoning_sim": 0.8622420430183411}, {"unit_index": 3, "inspected_object": "Motivation regarding inverse propensity weight instability", "observation": "The reviewer acknowledges that instability of inverse propensity weights is a known problem.", "reasoning": "By validating the motivation as addressing a known issue, the reviewer simultaneously deflates the novelty of the solution. Addressing a standard, well-known problem in the field is expected behavior for researchers in this area, reducing the perceived impact of the paper's primary motivation.", "judgment": "The core motivation is valid but does not constitute a significant novelty driver because the problem itself is widely recognized.", "valence": "mixed", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8144665956497192, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8321493864059448}]}, {"review_id": "01BozT1OXU", "paper_id": "9Ug3gHa9nL", "paper_title": "Offline Policy Learning for Nonparametric Contextual Bandits under Relaxed Coverage", "decision": "Reject", "summary": "The reviewer evaluates the paper's theoretical novelty and rigor positively but critiques the absence of empirical validation, connections to OPE literature, and discussions on parameter adaptation boundaries. The logic is comparative and norm-driven, seeking to position the work within interdisciplinary contexts rather than challenging core technical soundness.", "units": [{"unit_index": 0, "inspected_object": "The relaxed coverage notion", "observation": "Identified as a key novelty that generalizes prior assumptions and reflects more realistic conditions found in practice.", "reasoning": "The contribution is valuable because it subsumes or extends what came before, bridging nonparametric statistics and offline reinforcement learning by connecting two research communities.", "judgment": "Positive evaluation of the theoretical contribution's novelty and disciplinary positioning.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8318952322006226, "reasoning_key": "design_justification", "reasoning_sim": 0.7431926727294922}, {"unit_index": 1, "inspected_object": "The minimax upper and lower bounds", "observation": "The bounds are tight under partial coverage, and the proof structure uses a careful decomposition following classical nonparametric learning methods.", "reasoning": "Matching bounds represent the gold standard for theoretical contributions, and methodological continuity with established techniques serves as a marker of soundness.", "judgment": "Positive evaluation of the theoretical rigor and soundness of the proofs.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8477869629859924, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7727593183517456}, {"unit_index": 2, "inspected_object": "Empirical validation of the algorithm", "observation": "Absence of experiments illustrating the algorithm's behavior under varying coverage levels.", "reasoning": "Even theory papers should demonstrate algorithmic behavior to bridge theory to application; synthetic experiments are needed to verify if theoretical adaptivity translates to observable practical properties.", "judgment": "Critique regarding the lack of empirical demonstration despite theoretical soundness.", "valence": "negative", "suggested_improvement": "Conduct synthetic experiments with varying coverage levels to illustrate adaptivity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8697319030761719, "reasoning_key": "merit_recognition", "reasoning_sim": 0.765182375907898}, {"unit_index": 3, "inspected_object": "Connection to Off-Policy Evaluation (OPE) literature", "observation": "Absence of connection to recent OPE results.", "reasoning": "Contributions should be explicitly located within a web of prior results; stronger conceptual bridges to existing RL theory are expected to validate the paper's positioning.", "judgment": "Critique regarding insufficient disciplinary positioning relative to adjacent literatures.", "valence": "negative", "suggested_improvement": "Establish a stronger conceptual bridge to recent OPE results.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.783139169216156, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7295573353767395}, {"unit_index": 4, "inspected_object": "Adaptivity to unknown smoothness", "observation": "Absence of discussion on adaptivity to unknown smoothness, despite authors claiming adaptation is impossible without extra assumptions.", "reasoning": "Nonparametric statistics norms expect results to be adaptive to nuisance parameters; context about mild regularity assumptions under which adaptation might be feasible would map the boundary of the impossibility result.", "judgment": "Critique regarding the lack of contextualization around parameter adaptation boundaries.", "valence": "negative", "suggested_improvement": "Discuss adaptivity to unknown smoothness and clarify the boundary of the impossibility result.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8707907199859619, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8021864891052246}, {"unit_index": 5, "inspected_object": "Estimation of the concentrability coefficient (C*)", "observation": "Question regarding the estimation of C*, which is not required for the algorithm but has diagnostic value.", "reasoning": "Theoretical contributions should have a pathway to empirical application; understanding how practitioners might deploy the theory to measure the central theoretical object reveals its empirical content.", "judgment": "Inquiry into the practical usability and empirical measurability of the theoretical framework.", "valence": "conditional", "suggested_improvement": "Address the estimation of C* to demonstrate the empirical content of the relaxed coverage coefficient.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8260869979858398, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7589421272277832}]}, {"review_id": "01XYuDrCwc", "paper_id": "Op8RDj5qX1", "paper_title": "Optimizing optimizers for fast gradient-based learning", "decision": "Reject", "summary": "The reviewer conducts a contribution audit, arguing that the paper's reverse-engineering approach is trivially representable and thus uninformative, while also flagging Proposition 7 as a potential methodological error due to validation data contamination. The review emphasizes the need for principled constraints and proper literature positioning.", "units": [{"unit_index": 0, "inspected_object": "The paper's core contribution: mapping algebraic optimizer rules to optimization problems via reverse-engineering.", "observation": "The reviewer identifies the paper's achievement as a one-to-one mapping between known optimizers and specific constraints, noting that any algebraic rule can be represented this way (e.g., via Fenchel duals).", "reasoning": "The reviewer applies a standard of generative or selective power: a contribution must generate new algorithms, explain non-obvious phenomena, or provide selection criteria. Since the paper only reproduces known optimizers through an arbitrary-looking construction, it fails to meet this standard.", "judgment": "The result is conceptually weak because it constitutes a trivial mapping rather than a discovery or derivation from first principles.", "valence": "negative", "suggested_improvement": "Justify the choice of the positive semidefinite operator Q to demonstrate it is principled rather than arbitrary.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8195635676383972, "reasoning_key": "design_justification", "reasoning_sim": 0.8655472993850708}, {"unit_index": 1, "inspected_object": "Proposition 7: optimizing optimizers against validation loss.", "observation": "The reviewer observes that the proposal suggests using validation loss to tune the optimizer, with only a brief comment from authors regarding potential controversy.", "reasoning": "The reviewer applies the machine-learning norm that validation data must be held out from training signals. Using validation loss as the optimization target effectively makes the validation set part of the training objective, violating methodological integrity and invalidating generalization estimates.", "judgment": "The proposal is potentially methodologically flawed ('essentially equivalent to using the validation set as training set'), raising serious concerns about correctness.", "valence": "negative", "suggested_improvement": "Expand the discussion to clarify whether validation data is being used as training data and justify the epistemological status of such an approach.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8939045071601868, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7989630699157715}, {"unit_index": 2, "inspected_object": "Literature positioning relative to the Performance Estimation Problem (PEP) framework.", "observation": "The reviewer notes the absence of discussion on PEP (Drori and Teboulle), which formulates finding best optimizers as a semidefinite program with worst-case guarantees.", "reasoning": "The reviewer expects scholarly grounding in adjacent literature that provides normative criteria (worst-case performance) for optimizer selection. The paper's lack of engagement with PEP suggests it does not distinguish its ad-hoc constraint approach from existing frameworks that offer rigorous guarantees.", "judgment": "The paper lacks sufficient scholarly context and fails to position its contribution against relevant prior work that addresses similar goals with stronger theoretical backing.", "valence": "negative", "suggested_improvement": "Discuss the PEP framework to position the paper's contribution relative to existing work on optimizing over optimizers.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8537722229957581, "reasoning_key": "design_justification", "reasoning_sim": 0.8415570259094238}]}, {"review_id": "02B3FfvoR3", "paper_id": "msLnKDvhBx", "paper_title": "HSIC Bottleneck for Cross-Generator and Domain-Incremental Synthetic Image Detection", "decision": "Accept (Poster)", "summary": "The reviewer conducts a comparative and attributional evaluation, acknowledging empirical success while critically testing the novelty of the method against prior work (RINE, DualHSIC) and demanding rigorous component-level justification through ablations and complete comparisons.", "units": [{"unit_index": 0, "inspected_object": "Introduction of 3D Gaussian Splatting (3DGS) rendered images as a new synthetic image family for benchmarking", "observation": "The reviewer identifies this as the only element explicitly framed as a significant strength supporting acceptance, noting it expands the evaluation landscape beyond GAN/diffusion dichotomies.", "reasoning": "The reviewer values the expansion of the evaluation landscape as a genuine community contribution rather than an incremental extension, treating the new benchmark as a meaningful challenge that justifies acceptance despite other concerns.", "judgment": "Positive assessment of the benchmark's value as a significant contribution.", "valence": "positive", "suggested_improvement": "Provide cross-generator performance on 3DGS compared with previous methods to contextualize the benchmark's difficulty and test whether the method's improvements transfer to this new domain.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7787657380104065, "reasoning_key": "merit_recognition", "reasoning_sim": 0.78437739610672}, {"unit_index": 1, "inspected_object": "Claimed novelty of the HSIC bottleneck relative to RINE and DualHSIC approaches", "observation": "The reviewer asserts the method can be seen as a combination of existing RINE and DualHSIC approaches and notes that RINE is missing from Table 1.", "reasoning": "The reviewer applies a norm of complete comparison against the most relevant prior work; the absence of RINE creates uncertainty about whether the reported state-of-the-art results are robust or if the gap narrows when including this competitor.", "judgment": "Critical assessment questioning the intellectual contribution due to compositional reliance on known techniques without sufficient differentiation.", "valence": "negative", "suggested_improvement": "Include RINE in Table 1 or explain its absence to demonstrate the robustness of the reported margins over baselines.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8699541091918945, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7841907143592834}, {"unit_index": 2, "inspected_object": "Contribution of the specific term (1 - N(r_i)) in Equation 10 within the HGR replay component", "observation": "The reviewer requests an ablation to justify the contribution of this specific term, distinguishing between the HSIC relevance term and the k-center coverage term.", "reasoning": "The reviewer operates under a norm of component attribution, requiring individual justification for design choices rather than accepting aggregate performance numbers; this tests whether the specific design choice is principled or merely works in practice.", "judgment": "Skeptical assessment regarding the necessity and isolation of the specific design choice driving performance.", "valence": "conditional", "suggested_improvement": "Perform an ablation study isolating the contribution of the (1 - N(r_i)) term to demonstrate its specific impact on performance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8191699385643005, "reasoning_key": "merit_recognition", "reasoning_sim": 0.784711480140686}, {"unit_index": 3, "inspected_object": "Specification of the classifier architecture g_theta_g", "observation": "The reviewer notes that the paper does not specify the architecture of the classifier g_theta_g.", "reasoning": "Without knowing the model's structure, the reviewer cannot fully evaluate whether reported results are attributable to the proposed loss function or to the classifier's capacity, highlighting a barrier to experimental transparency.", "judgment": "Uncertain assessment due to lack of transparency preventing full evaluation of the method's efficacy.", "valence": "uncertain", "suggested_improvement": "Specify the architecture of the classifier g_theta_g to allow for proper attribution of results.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8400930166244507, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.85196852684021}, {"unit_index": 4, "inspected_object": "Source and provenance of real samples used in the 3DGS datasets", "observation": "The reviewer questions where the real samples from the 3DGS datasets originate.", "reasoning": "The reviewer assumes benchmark construction requires transparent data sources to ensure validity; overlapping distributions between real and synthetic samples could artificially affect the detection task's difficulty.", "judgment": "Concern about potential confounds in benchmark design affecting validity.", "valence": "negative", "suggested_improvement": "Clarify the source and distribution of real samples to ensure they do not overlap unintentionally with synthetic training data.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8348129987716675, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8271135687828064}]}, {"review_id": "02xiEhJhKz", "paper_id": "aJRZzDiGe5", "paper_title": "Finding the Broad Gini in the Bottle: Optimizing Equity, Efficiency, and Resilience in Grid Restoration", "decision": "Reject", "summary": "The reviewer evaluates the paper primarily through the lens of contribution, finding neither algorithmic nor modeling novelty sufficient. This central judgment is compounded by multiple technical objections regarding formal specification clarity, domain consistency, and missing performance data, leading to a uniformly negative assessment of the paper's readiness for publication.", "units": [{"unit_index": 0, "inspected_object": "Algorithmic novelty of the proposed method", "observation": "The paper presents a standard two-stage MILP stochastic problem solved with a standard MILP solver.", "reasoning": "In operations research, solving a known formulation with existing solvers does not constitute algorithmic novelty or contribution.", "judgment": "Insufficient algorithmic contribution.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.874588131904602, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7549455165863037}, {"unit_index": 1, "inspected_object": "Modeling contribution and metric combination", "observation": "The paper fails to assert that the specific combination of metrics has never been used or yields insights unavailable from prior work.", "reasoning": "Contributions in this domain must be clearly demarcated from a substantial body of work on power system restoration; mere utility is insufficient without demonstrated uniqueness.", "judgment": "Insufficient modeling contribution.", "valence": "negative", "suggested_improvement": "Assert that these metrics have never been used and demonstrate that their combination yields new insights and solutions that could not be gleaned from prior work.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8294861316680908, "reasoning_key": "novelty_standard", "reasoning_sim": 0.717465877532959}, {"unit_index": 2, "inspected_object": "Consistency between transmission benchmarks and radiality constraints", "observation": "The paper uses transmission benchmarks (IEEE-30, IEEE-145) but includes radiality constraints, which are requirements for distribution system modeling only.", "reasoning": "Mixing elements from different network domains without explanation indicates a fundamental confusion about the problem domain and undermines model credibility.", "judgment": "Significant modeling inconsistency indicating domain confusion.", "valence": "negative", "suggested_improvement": "Clarify whether the model applies to transmission or distribution systems and remove or justify the inclusion of radiality constraints accordingly.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8224710822105408, "reasoning_key": "design_justification", "reasoning_sim": 0.8475936055183411}, {"unit_index": 3, "inspected_object": "Definition of restoration", "observation": "The paper does not describe what the restoration is, whereas the reviewer expects it to involve determining the sequence of lines to re-energize.", "reasoning": "Failure to align with or explicitly define the core concept of restoration in power systems is a substantive gap.", "judgment": "Substantive definitional gap.", "valence": "negative", "suggested_improvement": "Explicitly define what constitutes restoration in the context of the paper's model.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.5932298898696899, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7078690528869629}, {"unit_index": 4, "inspected_object": "Load forecasting literature positioning", "observation": "The paper uses HMM-based load forecasting but dismisses it as a contribution due to the wide body of existing work.", "reasoning": "Methodological elements must be positioned relative to relevant literature to establish novelty; failure to do so is a weakness.", "judgment": "Lack of novel contribution from load forecasting component.", "valence": "negative", "suggested_improvement": "Position the HMM-based load forecasting method relative to existing literature and demonstrate its novelty.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "related_work", "object_sim": 0.7993628978729248, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8554254770278931}, {"unit_index": 5, "inspected_object": "Clarity of objective functions", "observation": "There are two distinct objective functions (linear combination of equations 16-19 and equation 23), and it is unclear which is used.", "reasoning": "A mathematical model should have a single, clearly identified objective function or explicitly explain the relationship between multiple objectives.", "judgment": "Ambiguity in model specification.", "valence": "negative", "suggested_improvement": "Clarify which objective function is used or explicitly explain the relationship between the two.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8263757824897766, "reasoning_key": "design_justification", "reasoning_sim": 0.7359715104103088}, {"unit_index": 6, "inspected_object": "Two-stage stochastic programming structure", "observation": "The formulation does not formally describe the problem as a first and second stage stochastic program, making it difficult to understand.", "reasoning": "Stochastic programs must clearly delineate decisions made before vs. after uncertainty realization; unit commitment decisions should adapt to conditions during emergencies.", "judgment": "Unclear stochastic structure preventing soundness evaluation.", "valence": "negative", "suggested_improvement": "Formally describe the problem as a first and second stage stochastic program and justify which decisions are fixed per scenario.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8016402125358582, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7082973718643188}, {"unit_index": 7, "inspected_object": "Data description", "observation": "Most data is described as caseXX without providing a description of the data.", "reasoning": "Readers will not know what caseXX refers to; self-contained data documentation is expected even for standard benchmarks.", "judgment": "Insufficient data documentation.", "valence": "negative", "suggested_improvement": "Provide a description of the data used, including details on caseXX benchmarks.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7896351218223572, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.704939603805542}, {"unit_index": 8, "inspected_object": "Computational performance information", "observation": "Runtimes and scalability of the model are not provided.", "reasoning": "For operations research, computational tractability is a core criterion; a model without efficiency evidence has limited practical value.", "judgment": "Missing critical performance evidence.", "valence": "negative", "suggested_improvement": "Provide runtimes and scalability analysis of the model.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8585501909255981, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8085053563117981}, {"unit_index": 9, "inspected_object": "Integration of game theory", "observation": "Game theory is stated as a crucial component but is not described as part of the workflow or model, appearing only as a comparison point.", "reasoning": "All claimed components should be integrated into the model description; absence suggests lack of care or coherence.", "judgment": "Unintegrated component undermining architectural coherence.", "valence": "negative", "suggested_improvement": "Describe how game theory is integrated into the workflow or model, or remove it if not integral.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7588373422622681, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.6756975650787354}]}, {"review_id": "03ETTo7ue7", "paper_id": "hPUjTTj64O", "paper_title": "Path-Consistency with Prefix Enhancement for Efficient Inference in LLM", "decision": "Reject", "summary": "The reviewer evaluates the paper as empirically sound but incrementally novel, questioning the robustness of efficiency claims against stronger baselines, the necessity of heavy engineering, and the applicability of theoretical guarantees to hard cases.", "units": [{"unit_index": 0, "inspected_object": "Efficiency gains (20–48% latency reduction, 20–60% fewer tokens)", "observation": "The reviewer acknowledges significant efficiency gains but notes the comparison is limited to a 20-sample baseline.", "reasoning": "The reviewer suspects the gains may be an artifact of the weak baseline and questions if prefix reuse locks in errors when larger budgets are used.", "judgment": "Efficiency is real but context-dependent; generalizability to larger budgets is uncertain.", "valence": "mixed", "suggested_improvement": "Compare against SC with larger or adaptive sample sizes to verify robustness of speed-up.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.792972207069397, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8031077980995178}, {"unit_index": 1, "inspected_object": "Novelty of early-commitment mechanism", "observation": "The method's core idea resembles existing beam/pruning and adaptive-consistency techniques.", "reasoning": "The reviewer holds a norm that novelty requires a new mechanism, not just a re-packaging of known search strategies.", "judgment": "The contribution is incremental.", "valence": "negative", "suggested_improvement": "Clarify how the proposed mechanism differs fundamentally from prior work.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8014849424362183, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8318703174591064}, {"unit_index": 2, "inspected_object": "Hyper-parameter engineering", "observation": "The method has five tunable components with no automatic or adaptive schedule.", "reasoning": "Heavy engineering implies poor external validity; results may not transfer to new tasks without expert tuning.", "judgment": "The method is heavily engineered and lacks practical ease-of-use.", "valence": "negative", "suggested_improvement": "Provide sensitivity analysis or an adaptive mechanism for hyper-parameters.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8767407536506653, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8145802021026611}, {"unit_index": 3, "inspected_object": "Prefix-reuse assumption and accuracy impact", "observation": "Fig. 2 shows 25–50% tokens wasted on wrong branches.", "reasoning": "Noisy confidence signals suggest aggressive prefix lengthening might mis-guide later samples and hurt accuracy.", "judgment": "The safety of prefix commitment under noisy conditions is questionable.", "valence": "negative", "suggested_improvement": "Analyze failure modes where aggressive prefix lengthening hurts accuracy.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8222765326499939, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8080160021781921}, {"unit_index": 4, "inspected_object": "Theoretical justification (Pvote >= 0.5)", "observation": "The proof assumes Pvote >= 0.5.", "reasoning": "This assumption excludes hard datasets where SC struggles, limiting the method's value proposition in difficult regimes.", "judgment": "The theoretical guarantee is too narrow for practical utility.", "valence": "negative", "suggested_improvement": "Provide bounds for harder datasets or guidance on setting Cthreshold.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8226242661476135, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7694810032844543}]}, {"review_id": "03GyNTIJYH", "paper_id": "96DTvuYq4h", "paper_title": "Uncertainty-Aware 3D Reconstruction for Dynamic Underwater Scenes", "decision": "Accept (Poster)", "summary": "The reviewer acts as a technical auditor and presentation consultant, focusing on internal consistency of notation, clarity of motivation, experimental reporting hygiene, and robustness evidence. The stance is positive but conditional on clarifications and additional reporting rather than fundamental methodological changes.", "units": [{"unit_index": 0, "inspected_object": "Mathematical derivation of the medium model, specifically Eq. (6) and the parameterization of σ_med", "observation": "The paper uses a single scalar `σ_med` inside the transmittance term `T_med(s)` but later separates `σ_med` into `σ_att` (structure) and `σ_bs` (backscatter), creating ambiguity about whether the equations explicitly reflect this decomposition and how wavelength-dependent parameters are handled.", "reasoning": "The reviewer applies a standard that equations must be self-documenting and faithful to the paper's conceptual claims of clean separation between structure and medium; the current notation leaves a gap that prevents verification of this factorization.", "judgment": "The derivation is ambiguous and requires clarification to maintain internal consistency with the stated contributions.", "valence": "negative", "suggested_improvement": "Write out explicit separated transmittance and emission terms and clarify the wavelength-dependent parameterization for RGB channels.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8382015824317932, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7839690446853638}, {"unit_index": 1, "inspected_object": "Introduction section presentation", "observation": "The introduction lacks a visual teaser or before/after comparison on a challenging underwater scene, making the paper's advantages unclear to readers.", "reasoning": "The reviewer treats the introduction as a persuasion device; without early visual dramatization of the value proposition (handling dynamics + noise), reader engagement may suffer despite the strength of the underlying work.", "judgment": "The presentation entry point is weak and fails to effectively sell the contribution visually.", "valence": "negative", "suggested_improvement": "Add a teaser in the introduction with a before/after comparison on a challenging underwater scene.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8247584700584412, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7927214503288269}, {"unit_index": 2, "inspected_object": "User study methodology reporting", "observation": "The description of the 40-person, 1,500-image user study lacks details on randomization, rating scales, and inter-rater reliability.", "reasoning": "The reviewer applies a standard requiring experimental hygiene and protocol transparency for human evaluations; without these details, readers cannot assess the validity or reproducibility of the subjective results.", "judgment": "The user study is promising but its reporting is insufficient for validation.", "valence": "negative", "suggested_improvement": "Provide descriptions of the protocol including randomization, rating scale, and inter-rater reliability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.7650710344314575, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7945768237113953}, {"unit_index": 3, "inspected_object": "Robustness to calibration noise", "observation": "The paper compares VGGT and COLMAP pose estimators with comparable results but does not quantify robustness to noisy inputs typical of underwater capture.", "reasoning": "The reviewer infers that comparing two pipelines is insufficient to demonstrate robustness to observation noise; controlled noise injection experiments are required to stress-test the uncertainty-aware loss under degraded input conditions.", "judgment": "The claim of robustness is not sufficiently supported by the current evidence base.", "valence": "negative", "suggested_improvement": "Add experiments with noisy calibration inputs to quantify robustness to calibration inaccuracies.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8321040272712708, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8011556267738342}, {"unit_index": 4, "inspected_object": "Typographical errors in the text", "observation": "The paper contains multiple typographical errors, such as 'Japanese Gradens' and 'a uncertainty-aware'.", "reasoning": "The presence of typos signals a lack of polish and careful proofreading, which undermines the professional presentation of the work.", "judgment": "The presentation quality is diminished by preventable errors.", "valence": "negative", "suggested_improvement": "Correct the identified typographical errors.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7630798816680908, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8299024105072021}]}, {"review_id": "03Itrv6a0g", "paper_id": "sfe6KFGRlD", "paper_title": "Dual-Stage Frequency-based Denoising for Generative Recommendation", "decision": "Reject", "summary": "The reviewer acts as a credibility auditor, prioritizing verification of empirical claims over conceptual criticism. They challenge the plausibility of large effect sizes, question the practical trade-offs of computational complexity, and demand mechanistic interpretability for frequency-domain operations, while requesting additional statistical rigor and case studies to substantiate the findings.", "units": [{"unit_index": 0, "inspected_object": "Magnitude of reported performance improvement (+23.83% NDCG@10 on Software)", "observation": "The reported improvement is exceptionally high and far exceeds typical improvements observed in the field.", "reasoning": "Large deviations from expected effect sizes are more likely explained by methodological artifacts than by genuine superiority, triggering a need for rigorous scrutiny of experimental setup details such as data preprocessing, splits, baseline implementation, and hyperparameter tuning.", "judgment": "The empirical claims require verification to rule out systematic errors or biases before their validity can be accepted.", "valence": "negative", "suggested_improvement": "Provide identical experimental conditions across baselines including thoroughly tuned hyperparameters, and report mean/variance across random seeds.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8081997632980347, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8242993950843811}, {"unit_index": 1, "inspected_object": "Computational cost and model complexity of the framework", "observation": "The framework is described as 'heavy' with high computational cost and model complexity, evidenced by an approximate 25-30% increase in training time.", "reasoning": "A method's value depends not only on accuracy but also on ease of adoption; the current evidence does not fully establish that the performance gains justify the significant complexity burden compared to simpler baselines.", "judgment": "The practical viability and deployment readiness of the method are questionable due to the non-trivial overhead.", "valence": "negative", "suggested_improvement": "Analyze the computational and memory complexity of CRFA compared to standard self-attention and assess feasibility for long sequences.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8740710020065308, "reasoning_key": "cost_benefit", "reasoning_sim": 0.855328381061554}, {"unit_index": 2, "inspected_object": "Explanatory completeness regarding frequency-domain operations", "observation": "The paper fails to explain what the frequency-domain operations actually do to the data, lacking visualization or case analysis of which behaviors are identified as noise or what patterns are learned.", "reasoning": "A method paper should offer mechanistic transparency and intuition for why its operations succeed, rather than providing only empirical validation without semantic content explanation.", "judgment": "The paper lacks necessary interpretability to fully understand the mechanism behind the method's success.", "valence": "negative", "suggested_improvement": "Provide a concrete case study or before/after visualization of a real user sequence to demonstrate the method's behavior.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8494768738746643, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7835597395896912}, {"unit_index": 3, "inspected_object": "RPML initialization sensitivity and design choices", "observation": "The Gaussian initialization appears simple, raising concerns about potential instability, overfitting, or masking of training dynamics.", "reasoning": "Granular methodological questions about sensitivity to parameters like alpha suggest that seemingly arbitrary design choices might significantly impact robustness and results.", "judgment": "The robustness of the method's specific design choices remains unverified and potentially fragile.", "valence": "uncertain", "suggested_improvement": "Investigate alternatives to initialization, analyze sensitivity to alpha, and examine training dynamics.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.7939668297767639, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8567653894424438}]}, {"review_id": "03JmK9fhjO", "paper_id": "DF6udvxuvY", "paper_title": "From Pixels to Words -- Towards Native Vision-Language Primitives at Scale", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper through a lens of methodological purism, prioritizing isolated architectural contributions and rigorous scaling diagnostics over holistic performance metrics. Key criticisms focus on the lack of evidence for scale-transferability of design choices, ambiguous attribution of gains to confounding factors like base models, and overconfident framing against modular baselines.", "units": [{"unit_index": 0, "inspected_object": "Pre-Buffer scaling consistency across model sizes", "observation": "The NEO-9B model uses a pre-Buffer that is 50% smaller than the one in NEO-2.2B, justified by performance-efficiency trade-offs.", "reasoning": "Ablation results obtained at the smaller scale (NEO-2.2B) may not extrapolate correctly to the larger post-LLM context; design choices validated at small scale require independent validation at larger scale when mediating vision-language interactions.", "judgment": "The justification for the size discrepancy lacks evidentiary support and raises concerns about scale-transferability.", "valence": "negative", "suggested_improvement": "Provide independent validation of the pre-Buffer design choices at the larger scale used in NEO-9B.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8571933507919312, "reasoning_key": "design_justification", "reasoning_sim": 0.8456370830535889}, {"unit_index": 1, "inspected_object": "Attribution of performance gains to architectural components", "observation": "The paper does not fully isolate the contribution of its native architecture from confounding factors such as data quality/quantity, training stages, or the use of a modern base LLM (Qwen3).", "reasoning": "Architectural claims require isolation from confounds to determine if the native VLM primitives are the primary driver of improvement rather than infrastructure advantages like better backbones or data.", "judgment": "The attribution of success to the proposed architecture is ambiguous due to uncontrolled variables.", "valence": "negative", "suggested_improvement": "Conduct controlled ablation experiments to isolate the effect of the architecture from data and base model improvements, or explicitly acknowledge this limitation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8317464590072632, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7613478302955627}, {"unit_index": 2, "inspected_object": "Framing of comparison with Modular VLMs", "observation": "The paper exhibits a 16% gap on InfoVQA against Qwen2-VL but characterizes its position relative to modular VLMs in an overly positive manner.", "reasoning": "Claims about competitiveness should be calibrated to actual performance gaps; aspirational framing that suggests parity where significant gaps exist violates norms of calibrated rhetoric.", "judgment": "The presentation overstates the competitive standing of the native approach relative to modular alternatives.", "valence": "negative", "suggested_improvement": "Adjust the phrasing in the 'Comparison with Modular VLMs' section to accurately reflect the magnitude of the performance gap while acknowledging constraints.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8627099394798279, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7793729901313782}, {"unit_index": 3, "inspected_object": "Scaling trajectory between 2B and 9B models", "observation": "Scaling improvements between the 2B and 9B models are modest compared to those seen in modular VLMs.", "reasoning": "Modest scaling gains serve as a diagnostic indicator of potential architectural limitations; a well-designed native architecture should scale comparably to modular alternatives.", "judgment": "The scaling behavior suggests deeper architectural limitations despite overall positive performance.", "valence": "negative", "suggested_improvement": "Analyze the reasons for modest scaling gains and discuss whether they indicate fundamental limits of the current architecture.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8294736742973328, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7727701663970947}, {"unit_index": 4, "inspected_object": "Writing style and authenticity", "observation": "Parts of the paper exhibit overuse of synonyms, sometimes resulting in nonsensical phrases, suggesting LLM-assisted generation without sufficient human editing.", "reasoning": "Presentation clarity and intentional word choice are expected standards; artifacts from automated writing tools obscure meaning and reduce perceived rigor.", "judgment": "The presentation suffers from clarity issues due to stylistic artifacts.", "valence": "negative", "suggested_improvement": "Edit the manuscript to remove LLM-generated artifacts and ensure clear, intentional prose.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "clarity", "object_sim": 0.7446436882019043, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8241285085678101}]}, {"review_id": "03TLWZg9jl", "paper_id": "ddf7XdLtNO", "paper_title": "Sequence Length Matters in Data Scheduling for Accelerating Language Model Pretraining", "decision": "Reject", "summary": "The reviewer evaluates the paper as an engineering contribution with strong empirical intuition but identifies three distinct weaknesses: incomplete baseline comparisons affecting positioning, poor figure placement increasing cognitive load, and insufficient theoretical justification for Lemma 2.", "units": [{"unit_index": 0, "inspected_object": "Comparative performance against reference-based data selection methods", "observation": "The paper compares DBSP only with reference-free online data selection methods, omitting reference-based baselines.", "reasoning": "If reference-based approaches outperform DBSP, the practical contribution becomes unclear; a reference-free method must justify its constraints by achieving comparable performance without reference overhead (standard of practical necessity).", "judgment": "Positioning concern: the restriction to reference-free baselines leaves the method's broader utility uncertain.", "valence": "negative", "suggested_improvement": "Include reference-based baselines or discuss trade-offs in terms of ppl, training time, computational cost, and resource efficiency.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8422138690948486, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8356918692588806}, {"unit_index": 1, "inspected_object": "Presentation structure and figure placement", "observation": "Figures 2 and 3 are referenced multiple times before they appear in the text.", "reasoning": "This forces readers to scroll back and forth, increasing cognitive load and creating navigational friction that hinders following the argument.", "judgment": "Presentation barrier: the document is not optimally organized for reader comprehension.", "valence": "negative", "suggested_improvement": "Reorganize figures to appear closer to their first mention.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8608633875846863, "reasoning_key": "presentation_trust", "reasoning_sim": 0.681720495223999}, {"unit_index": 2, "inspected_object": "Lemma 2 theoretical justification", "observation": "The expectation inequality E_pi >= E_D is stated without explicit proofs or stronger correlation assumptions.", "reasoning": "A theoretical claim requires either proof from explicit assumptions (e.g., monotonicity or rank correlation bounds) or clear hedging as conjectural; the current formulation is under-justified.", "judgment": "Technical gap: the lemma is plausible but lacks rigorous formal support.", "valence": "negative", "suggested_improvement": "Provide explicit proofs such as monotonicity or rank correlation bounds to justify the expectation inequality.", "support_status": "mixed", "confidence": "medium", "object_key": "theory", "object_sim": 0.8044159412384033, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7798153162002563}]}, {"review_id": "03USgbBZf5", "paper_id": "LhwPy8NMoN", "paper_title": "PLP-RC:Point–Line–Plane Fusion for Discriminative Relation Classification with LLMs", "decision": "Reject", "summary": "The reviewer evaluates the paper primarily on theoretical rigor and claim-evidence alignment, praising the conceptual elegance while critiquing the lack of formal mathematical grounding, the implausibility of cross-sentence capabilities given the architecture, and terminological confusion regarding discriminative vs. generative paradigms.", "units": [{"unit_index": 0, "inspected_object": "Theoretical grounding of the PLP mechanism", "observation": "The paper claims to be conceptually grounded in geometric and information-theoretic principles but provides no formal connection between its geometric representations and those principles.", "reasoning": "A framework invoking specific theoretical principles requires a formal mathematical bridge (e.g., mapping line/plane representations to mutual information) to move beyond metaphor; without it, the contribution's significance is hard to assess.", "judgment": "The geometric perspective is largely metaphorical without formal mathematical formulation or theoretical guarantees.", "valence": "negative", "suggested_improvement": "Operationalize the metaphor into mathematics that can be inspected and potentially falsified.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.797042191028595, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7851948142051697}, {"unit_index": 1, "inspected_object": "Cross-sentence capability claim", "observation": "The paper claims the framework can capture long-range dependencies, yet all experiments are sentence-level.", "reasoning": "The paper's Introduction frames the method as addressing cross-sentence limitations, creating an expectation of evidence; additionally, the autoregressive decoder's [EOS] token causal attention architecture cannot model cross-sentence context effectively.", "judgment": "There is a scope mismatch between the stated motivation and experimental design, and the architectural setup makes the claimed capability implausible.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.8066377639770508, "reasoning_key": "design_justification", "reasoning_sim": 0.8318130373954773}, {"unit_index": 2, "inspected_object": "Discriminative/Generative distinction", "observation": "The paper claims to avoid hallucinations by using a 'discriminative framework,' but the underlying model (Qwen3) is a decoder-only LLM pretrained with generative next-token prediction.", "reasoning": "Calling a generative backbone-based system 'discriminative' is imprecise terminology that constitutes a category error, potentially obscuring what the method actually does and undermining the central selling point regarding hallucination avoidance.", "judgment": "The paper confuses discriminative vs. generative paradigms.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8306238055229187, "reasoning_key": "design_justification", "reasoning_sim": 0.7875121831893921}, {"unit_index": 3, "inspected_object": "Conceptual framing and clarity", "observation": "The reviewer finds the point-line-plane abstraction conceptually elegant and the approach clear and interpretable.", "reasoning": "The innovative geometric fusion paradigm and the use of LLM embeddings discriminatively represent a strong conceptual ambition and presentation virtue.", "judgment": "The paper demonstrates genuine ambition and clear conceptual framing, though this is distinct from empirical validation.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.9099059700965881, "reasoning_key": "design_justification", "reasoning_sim": 0.7874863147735596}]}, {"review_id": "03UigV2tv8", "paper_id": "HtMt9XNZv6", "paper_title": "Transfer Bound of Graph Convolutional Networks across Arbitrary Sparsity", "decision": "Reject", "summary": "The reviewer grants technical correctness but challenges the paper's novelty and positioning based on three specific literary and conceptual gaps: a misreading of asymptotic notation in prior work, missing citations for standard terminology, and a failure to distinguish the method from equivalent unbounded operator approaches.", "units": [{"unit_index": 0, "inspected_object": "Paper's characterization of prior work on sparsity regimes (claim that prior works analyze GCN only under fixed sparsity settings)", "observation": "Prior works state sparsity results in asymptotic notation (e.g., average degree in Θ(log n)).", "reasoning": "By the definition of Θ, these results hold for any graph sequence whose degrees are within constant factors of each other; therefore, prior works already cover multiple sparsity regimes, not just a 'fixed' setting.", "judgment": "The paper's motivating claim that prior work is limited to 'fixed sparsity settings' is incorrect.", "valence": "negative", "suggested_improvement": "Acknowledge that prior work covers constant-factor variations and position the contribution as removing that restriction rather than moving from fixed to variable sparsity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8600125312805176, "reasoning_key": "design_justification", "reasoning_sim": 0.7365608215332031}, {"unit_index": 1, "inspected_object": "Terminology usage: 'generalized graphon' vs established literature term", "observation": "The concept labeled 'generalized graphon' in the paper corresponds to the existing literature term 'graphex'.", "reasoning": "Using non-standard terminology without citing the originating literature (Borgs, Chayes, Dhara and Sen) misrepresents the novelty and grounding of the work.", "judgment": "This is a minor but crucial literature gap requiring citation.", "valence": "negative", "suggested_improvement": "Study and cite the graphex literature to align terminology and attribution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7532864212989807, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7335570454597473}, {"unit_index": 2, "inspected_object": "Novelty relative to unbounded operator approaches (Maskey, Levie and Kutyniok; Levie et al. 2021)", "observation": "The paper changes the domain of the graphon to be unbounded, while unbounded operator approaches change the range.", "reasoning": "These two approaches are likely equivalent via a change of measure in the vast majority of cases; thus, the unbounded operator approach can perform all tasks claimed by the paper.", "judgment": "The paper's novelty is undermined because its approach may not be distinct from existing methods.", "valence": "negative", "suggested_improvement": "Provide a comprehensive comparison with unbounded operator approaches to demonstrate where the proposed approach differs or adds value.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "novelty", "object_sim": 0.8836261034011841, "reasoning_key": "merit_recognition", "reasoning_sim": 0.6820083260536194}]}, {"review_id": "03UpgJlcN3", "paper_id": "dsIBpBJWhm", "paper_title": "Physics-Informed Graph Convolutional Network for Data-Free Learning of Channel Flow Fields", "decision": null, "summary": "The reviewer prioritizes methodological novelty as a gatekeeping criterion, rejecting the work based on prior art and perceived simplicity of test cases, while acknowledging presentation quality but dismissing technical engagement due to the novelty deficit.", "units": [{"unit_index": 0, "inspected_object": "Methodological novelty of combining graph neural networks and PINNs", "observation": "The reviewer identifies a 2022 journal paper (GNN Galerkin networks) as prior art, noting the combination was presented three years ago and is cited by the authors.", "reasoning": "The reviewer applies an ICLR standard requiring novel frameworks/methodologies, treating the existence of prior art as dispositive evidence that the current work lacks methodological novelty.", "judgment": "The work is not new in terms of methodology and fails to meet venue standards for novelty.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.775418758392334, "reasoning_key": "novelty_standard", "reasoning_sim": 0.829416036605835}, {"unit_index": 1, "inspected_object": "Internal consistency of the 'data-free' claim versus loss function components", "observation": "The reviewer notes Figure 2 and Equations 1-2 include a data component in the loss function, contrasting with the title's 'data-free' claim.", "reasoning": "The reviewer interprets this discrepancy as a contradiction or misrepresentation, using it to reinforce a negative impression of the work's rigor or honesty.", "judgment": "The paper exhibits internal inconsistency or sloppiness regarding its core claims.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8624384999275208, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8036925792694092}, {"unit_index": 2, "inspected_object": "Complexity and difficulty of the test cases (square/L-shaped channels)", "observation": "The reviewer asserts the test cases are simpler than those solved in the cited prior work and insufficiently challenging.", "reasoning": "The reviewer holds that challenging applications (e.g., 3D turbulent flow) would have justified the work, implying the current simple cases do not demonstrate sufficient contribution or robustness.", "judgment": "The empirical evaluation lacks sufficient difficulty to validate the framework's value.", "valence": "negative", "suggested_improvement": "Try to use the framework for turbulent flow in 3D.", "support_status": "memo_inferred", "confidence": "low", "object_key": "theory", "object_sim": 0.826290488243103, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7720866799354553}, {"unit_index": 3, "inspected_object": "Presentation quality (readability, figures, equations)", "observation": "The reviewer acknowledges readability, figure quality, equation presentation, and illustrative value of Figure 1.", "reasoning": "These elements are noted as positive but are deemed surface-level and insufficient to compensate for the lack of novelty.", "judgment": "The presentation is clear and well-formatted, though secondary to substantive contributions.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8790397047996521, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7647028565406799}]}, {"review_id": "03YecWUUmR", "paper_id": "MYyVPutxVB", "paper_title": "Injecting Sensitivity Constraint Into Continual Learning Significantly Enhances Surrogate-Aided Optimization", "decision": null, "summary": "The reviewer systematically identifies gaps in specification, baseline completeness, benchmark diversity, and component isolation, concluding that while the idea has potential, the current execution lacks the rigorous validation and situational awareness required to support its broad claims.", "units": [{"unit_index": 0, "inspected_object": "Specification of the Online Continual Learning (OCL) algorithm variant", "observation": "The paper uses the general term 'Online Continual Learning' without specifying the exact algorithm or variant implemented, raising ambiguity about whether it is a simple replay buffer.", "reasoning": "The reviewer assumes OCL is a family of techniques with meaningful differences; using an umbrella term elides implementation details critical for reproducibility and comparability, potentially undercutting the generality of claims if a minimal instantiation was used.", "judgment": "The specification is insufficiently precise to support the paper's claims of versatility.", "valence": "negative", "suggested_improvement": "Specify which specific OCL variant or algorithm was implemented.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8435038924217224, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7485562562942505}, {"unit_index": 1, "inspected_object": "Design rationale for the sensitivity constraint formulation", "observation": "The paper adopts one specific form of sensitivity information (e.g., input-output Jacobian norms) without discussing alternatives like local Lipschitz bounds.", "reasoning": "A core novelty claim requires demonstrating awareness of the design space and justifying why the chosen form is not arbitrary compared to other plausible options for injecting sensitivity information.", "judgment": "The core novelty mechanism is insufficiently motivated.", "valence": "negative", "suggested_improvement": "Provide justification for the specific formulation and discuss alternative sensitivity formulations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8496231436729431, "reasoning_key": "novelty_standard", "reasoning_sim": 0.823419988155365}, {"unit_index": 2, "inspected_object": "Experimental baselines in the multi-objective multi-fidelity Bayesian optimization (MO-MFBO) setting", "observation": "Key baselines such as MF-OSEMO and iMOCA are absent from the evaluation.", "reasoning": "Claims of effectiveness require comparison against the strongest relevant baselines; the reviewer believes these methods could be straightforwardly adapted, so their absence signals oversight or selective reporting that weakens experimental support.", "judgment": "The experimental support for claimed effectiveness is weakened by missing comparisons.", "valence": "negative", "suggested_improvement": "Include comparisons to MF-OSEMO and iMOCA.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8230942487716675, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8670458197593689}, {"unit_index": 3, "inspected_object": "Benchmark diversity and scope of evaluation", "observation": "Evaluation is mainly limited to the Branin–Currin synthetic benchmark, lacking real-world problems or other synthetic benchmarks like Park, Levy, or Rosenbrock.", "reasoning": "Broad claims of generality require a broader evidential base; reliance on a single synthetic benchmark is insufficient to demonstrate applicability across diverse problem types.", "judgment": "The evidence base is too narrow to support claims of general applicability.", "valence": "negative", "suggested_improvement": "Evaluate on additional synthetic benchmarks and real-world problems (e.g., Mechanical Plate Vibration Design).", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8732165098190308, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8910254240036011}, {"unit_index": 4, "inspected_object": "Comparisons to recent joint forward–inverse operator methods", "observation": "The paper lacks comparisons to methods like Latent Neural Operator (LNO) and does not evaluate the FUSE pipeline on PDE tasks beyond Darcy flow.", "reasoning": "Recent related methods provide necessary context for positioning; evaluating only on Darcy flow limits the demonstration of the method's utility in broader PDE solving contexts.", "judgment": "The comparative positioning is incomplete regarding recent advances in neural operators.", "valence": "negative", "suggested_improvement": "Compare against LNO and evaluate on additional PDE tasks such as airfoil or Navier-Stokes.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.836243748664856, "reasoning_key": "design_justification", "reasoning_sim": 0.7470611929893494}, {"unit_index": 5, "inspected_object": "Isolation of OCL contribution relative to simple retraining", "observation": "It is unclear whether OCL provides benefits beyond computational savings compared to simply retraining the surrogate model with concatenated data.", "reasoning": "If OCL is merely a computationally cheaper way to achieve what full retraining does, its substantive contribution is questionable; the synergy between OCL and the sensitivity constraint needs isolation to justify the combined narrative.", "judgment": "The necessity and distinct value of the OCL component are uncertain.", "valence": "conditional", "suggested_improvement": "Provide analysis isolating the benefit of OCL versus simple online retraining.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8536717295646667, "reasoning_key": "cost_benefit", "reasoning_sim": 0.787634015083313}, {"unit_index": 6, "inspected_object": "Sensitivity to the weighting parameter λ", "observation": "The paper does not analyze how performance varies with λ or provide guidance on selecting it per task.", "reasoning": "Practical usability depends on robustness to hyperparameter choice; lack of sensitivity analysis suggests the method may be fragile or require difficult per-task tuning, diminishing its 'plug-in' versatility.", "judgment": "The practical usability and robustness of the method are unverified.", "valence": "negative", "suggested_improvement": "Analyze sensitivity to λ and provide principled guidance for parameter selection.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8708701133728027, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8627806901931763}]}, {"review_id": "03c4RbrRsC", "paper_id": "WjEAMyLDoh", "paper_title": "Sharp asymptotic theory for Q-learning with \\texttt{LD2Z} learning rate and its generalization", "decision": "Accept (Poster)", "summary": "The reviewer challenges the paper's interpretive framing and empirical completeness, arguing that the step-size schedule is effectively offline rather than online, that convergence claims obscure known trade-offs, and that experimental comparisons lack necessary baselines and clarity.", "units": [{"unit_index": 0, "inspected_object": "The LD2Z step-size schedule structure and its dependence on total sample size n.", "observation": "The schedule η_{t,n} = η(1 - t/n) requires knowing the total sample size n in advance, making it undefined for data arriving beyond n.", "reasoning": "This dependence on a fixed horizon contradicts the normative expectation that Q-learning is a streaming, online algorithm; if new data arrive, the schedule fails, effectively turning the algorithm into an offline one.", "judgment": "The paper commits a category error by presenting a finite-horizon/batch method as an online algorithm, which undermines its applicability in typical reinforcement learning settings.", "valence": "negative", "suggested_improvement": "Discuss the algorithm's behavior when n is mis-specified or grows, and explicitly acknowledge that the schedule is not truly online.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8543199896812439, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.6905068755149841}, {"unit_index": 1, "inspected_object": "The convergence rate claims in Theorem 3.5 and the paper's abstract framing.", "observation": "The scaling n^ν/(2(ν+1)) is slower than √n unless ν → ∞, indicating a trade-off between last-iterate speed and asymptotic efficiency.", "reasoning": "Stochastic approximation theory posits that fast last-iterate convergence and asymptotic efficiency are mutually exclusive; the paper's own theorem confirms this trade-off but its abstract claims a 'best-of-both-worlds property' without acknowledging the limitation.", "judgment": "The presentation is mathematically misleading because it implies dominance over prior rules while internally admitting a performance trade-off.", "valence": "negative", "suggested_improvement": "Explicitly acknowledge the trade-off between last-iterate convergence and asymptotic efficiency, potentially comparing with ROOT-SGD-style results to position the work honestly within the literature.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8540178537368774, "reasoning_key": "design_justification", "reasoning_sim": 0.7624830007553101}, {"unit_index": 2, "inspected_object": "The experimental design and comparator set in Figure 1.", "observation": "The experiments omit comparisons with linearly decaying step sizes (η_t ∝ 1/t), lack averaged-iterate analysis, and do not report hyperparameters such as the exponent α.", "reasoning": "Linear decay has theoretical guarantees for Q-learning and averaged iterates are standard for polynomial decay; the absence of these baselines and details creates suspicion that the observed dominance of LZ2D may result from unfair parameter choices rather than intrinsic superiority.", "judgment": "The empirical evidence for LZ2D's superiority is incomplete and potentially biased due to missing standard comparators and reproducibility details.", "valence": "negative", "suggested_improvement": "Add experiments with linearly decaying step sizes, include averaged-iterate analysis, and provide full hyperparameter reporting to ensure fair comparison and reproducibility.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8379546999931335, "reasoning_key": "design_justification", "reasoning_sim": 0.8066128492355347}, {"unit_index": 3, "inspected_object": "The measurement definition at Line 455 versus the plot in Figure 3.", "observation": "There is a discrepancy between the text defining the measurement as max over k_n ≤ t ≤ n and Figure 3 appearing to plot max over 1 ≤ t ≤ n.", "reasoning": "If different measurement windows are used for the LD2D method and the Brownian benchmark, the comparison becomes unfair and affects the interpretation of approximation quality; the alignment might be due to slow convergence of remainder terms rather than method superiority.", "judgment": "The technical inconsistency raises concerns about methodological rigor and the validity of the visual comparison presented.", "valence": "negative", "suggested_improvement": "Clarify the correct measurement definition and explain any discrepancies between the text and figures, particularly regarding the role of the k_n term.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7517513632774353, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8128923773765564}]}, {"review_id": "03dGa8mvRz", "paper_id": "tkEmIJv1tB", "paper_title": "OmniEVA: Embodied Versatile Planner via Task-Adaptive 3D-Grounded and Embodiment-aware Reasoning", "decision": "Accept (Poster)", "summary": "The reviewer acknowledges the paper's motivation and novelty but raises significant concerns about epistemic transparency: unclear data provenance, unattributed performance gains, overreaching real-world claims, and ambiguous gating mechanism behavior.", "units": [{"unit_index": 0, "inspected_object": "Training pipeline data provenance and terminology", "observation": "The reviewer identifies three training stages (TAGR pretraining, SFT, RFT) but cannot determine the specific datasets used at each stage or disambiguate terms like 'general embodied reasoning' versus 'omnibodied reasoning'.", "reasoning": "Without knowing the data inputs and having clear terminology to map results to model configurations, the reviewer cannot evaluate what the model has learned or assess whether ablations isolate claimed contributions.", "judgment": "The training pipeline is not delivered clearly, creating a fundamental barrier to evaluating the model's learning and the validity of the ablation studies.", "valence": "negative", "suggested_improvement": "Provide explicit details on the datasets used at each training stage and clarify the definitions of key terminology to map results to specific model configurations.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8165764808654785, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8152972459793091}, {"unit_index": 1, "inspected_object": "Attribution of performance gains to architectural innovations", "observation": "The paper reports strong performance using extensive training data and three training stages, but does not decompose the contribution of each stage or control for base model capability and data scale.", "reasoning": "Comparisons may be unfair if performance is driven by scale rather than the proposed mechanisms; without decompositional evidence, the paper offers limited insight beyond demonstrating that the configuration works.", "judgment": "The causal attribution of performance gains to the proposed method is weak due to the absence of evidence isolating component contributions from data scale effects.", "valence": "negative", "suggested_improvement": "Provide decompositional evidence showing which components matter and by how much, and address potential confounds related to base model capability and data scale.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8020225763320923, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8608018159866333}, {"unit_index": 2, "inspected_object": "Claims about real-world robotics applicability", "observation": "The paper makes claims about 'real-world robotics tasks' despite containing no real-robot tasks or results.", "reasoning": "Claims about real-world applicability should be calibrated to the evidence actually presented; extrapolating to real-world utility without empirical support is inappropriate.", "judgment": "The claims regarding real-world applicability are not appropriate given the lack of corresponding experimental evidence.", "valence": "negative", "suggested_improvement": "Remove or temper claims about real-world robotics tasks unless supported by real-robot experiments, or provide evidence linking simulated results to real-world performance.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.7962691783905029, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7844606041908264}, {"unit_index": 3, "inspected_object": "Gate activation patterns in Figure 4", "observation": "Figure 4 shows gate activation differences across object categories involving only semantics (e.g., armchair, towel), raising questions about whether this reflects geometric necessity or data distribution bias.", "reasoning": "If the gated router activates based on semantic regularities rather than task-relevant geometric needs, the mechanism's claimed task-adaptivity and interpretability are suspect.", "judgment": "The interpretation of gate activation patterns is uncertain, creating doubt about whether the mechanism learns meaningful geometric features or merely reflects statistical data biases.", "valence": "conditional", "suggested_improvement": "Explain the differences in gate activation across object categories and investigate whether they indicate a bias stemming from data distribution.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7668015956878662, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8097769618034363}]}, {"review_id": "03rLZTg7Go", "paper_id": "Vem6FQvRvq", "paper_title": "PM-KVQ: Progressive Mixed-precision KV Cache Quantization for Long-CoT LLMs", "decision": "Accept (Poster)", "summary": "The reviewer accepts the mathematical elegance of the method but applies a strict deployment-readiness standard, criticizing the lack of bf16 results, real-kernel validation, and reproducible experimental details as insufficient evidence for practical applicability.", "units": [{"unit_index": 0, "inspected_object": "Mathematical formulations (Eqs. 9-12, Eq. 3)", "observation": "The reviewer acknowledges concrete formulations for positional interpolation and a simple shrinking rule that avoids round-trip dequantization.", "reasoning": "The reviewer evaluates these equations for implementational elegance and hardware-level implications, accepting the method's internal logic as sound.", "judgment": "Positive evaluation of the core technical contributions and design elegance.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.805311918258667, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8097270727157593}, {"unit_index": 1, "inspected_object": "Experimental reporting: Precision formats (FP16 vs bf16)", "observation": "The paper only reports FP16 results, omitting bf16.", "reasoning": "bf16 is the de facto standard for inference with wider dynamic range; its exclusion leaves uncertainty about performance and compatibility in realistic deployment settings.", "judgment": "Weakness: The practical claims are unsubstantiated due to lack of evidence in the standard precision format.", "valence": "negative", "suggested_improvement": "Report bf16 results to demonstrate compatibility with realistic deployment settings.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8594929575920105, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7362608909606934}, {"unit_index": 2, "inspected_object": "Experimental validity: Fake quantization vs real kernels", "observation": "Accuracy is reported without real 2-bit/4-bit kernels.", "reasoning": "Memory management and throughput claims cannot be fully validated in simulation because actual memory footprint and kernel efficiency are artifacts of real hardware implementation; fake quant makes throughput claims suspect.", "judgment": "Weakness: The claim of robustness in practice is weakened by the absence of real-kernel validation.", "valence": "negative", "suggested_improvement": "Validate PM-KVQ using real 2-bit/4-bit kernels to substantiate practical robustness and throughput claims.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8370025157928467, "reasoning_key": "design_justification", "reasoning_sim": 0.7932929396629333}, {"unit_index": 3, "inspected_object": "Experimental setup: QwQ-32B hardware feasibility", "observation": "The reviewer questions how QwQ-32B is used on one A100 GPU, finding the memory calculation suspect.", "reasoning": "The reviewer attempted to reconstruct experimental conditions and found them implausible given the model size and hardware constraints.", "judgment": "Uncertainty/Skepticism regarding the feasibility and reproducibility of the reported experiments.", "valence": "negative", "suggested_improvement": "Clarify hardware configuration and memory usage details to verify feasibility.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.759997546672821, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7705390453338623}, {"unit_index": 4, "inspected_object": "Implementation choice: CVXPY solver", "observation": "The reviewer asks why CVXPY was used for the integer program.", "reasoning": "This probes whether the authors made a principled tool selection or defaulted to a familiar one, impacting the perceived intellectual rigor and engineering pathway.", "judgment": "Questioning of methodological scrutiny and engineering rationale.", "valence": "conditional", "suggested_improvement": "Provide justification for the choice of CVXPY over simpler solvers or heuristics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7628721594810486, "reasoning_key": "novelty_standard", "reasoning_sim": 0.763843297958374}, {"unit_index": 5, "inspected_object": "Implementation choice: Progressive quantization GPU utilization", "observation": "The reviewer questions if progressive quantization hinders GPU utilization by blocking parallel execution.", "reasoning": "This probes for hidden systems costs that an accuracy-focused evaluation might miss, challenging the practical viability of the proposed strategy.", "judgment": "Uncertainty regarding the efficiency of the proposed quantization strategy on hardware.", "valence": "conditional", "suggested_improvement": "Analyze or report on GPU utilization and parallel execution impacts of progressive quantization.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8033653497695923, "reasoning_key": "cost_benefit", "reasoning_sim": 0.852263331413269}, {"unit_index": 6, "inspected_object": "Reproducibility: Experimental parameters", "observation": "Missing details on batch size, output token counts, pass@k parameters, voting sample counts, and number of trials.", "reasoning": "A paper should provide enough information for independent replication; the absence of these details is treated as a deficiency preventing verification of claims.", "judgment": "Weakness: Insufficient detail for reproducibility.", "valence": "negative", "suggested_improvement": "Provide complete experimental parameters including batch size, token counts, pass@k, voting samples, and trial counts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8877378702163696, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.801339864730835}]}, {"review_id": "03ziMvvwdr", "paper_id": "laN0mgvF2Y", "paper_title": "Revisiting Mixture Policies in Entropy-Regularized Actor-Critic", "decision": "Reject", "summary": "The reviewer provides a qualified endorsement centered on the novelty of Proposition 3.3 and the internal consistency of the theory, while accepting empirical claims via a sufficiency heuristic and expressing uncertainty about parameter sensitivity and generalization.", "units": [{"unit_index": 0, "inspected_object": "Proposition 3.3 (robustness argument regarding Gaussian base policies vs. Gaussian mixture policies)", "observation": "The reviewer identifies a novel robustness argument showing that Gaussian base policies may lose stationary points when the entropy coefficient α exceeds 3/2 r_max, whereas GM policies maintain valid solutions.", "reasoning": "The reviewer finds this result intellectually satisfying because it establishes a concrete, interpretable threshold phenomenon and connects entropy regularization to multimodal policy landscapes in a clean way, which serves as a primary source of intellectual value.", "judgment": "The proposition is evaluated as genuinely novel and theoretically significant.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8382983207702637, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8259015679359436}, {"unit_index": 1, "inspected_object": "Propositions 3.1–3.2 and 3.4", "observation": "The reviewer classifies these propositions as extending known properties of entropy-regularized optimization rather than introducing entirely new formulations.", "reasoning": "The reviewer applies a standard distinguishing genuine novelty from refinement, placing these results in a lower tier of contribution compared to Proposition 3.3, though acknowledging they meaningfully deepen understanding.", "judgment": "These contributions are viewed as thoughtful refinements/extensions rather than fundamentally new formulations.", "valence": "conditional", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.7945222854614258, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7942739129066467}, {"unit_index": 2, "inspected_object": "MRP estimator (Theorem 4.3 and Proposition 4.7)", "observation": "The reviewer acknowledges variance-reduction guarantees but does not engage with the mechanics, assumptions, or practical meaningfulness of the marginalized reparameterization.", "reasoning": "The reviewer treats the estimator's evaluation at the level of its stated guarantee without testing the claim that mixtures lack low-variance gradient updates or whether variance reduction translates to empirical improvements.", "judgment": "The estimator is accepted based on stated theoretical guarantees without scrutiny of implementation details or practical translation.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8403022885322571, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7751404047012329}, {"unit_index": 3, "inspected_object": "Experimental suite (continuous-control benchmarks)", "observation": "The reviewer notes experiments cover a wide range of benchmarks and concludes they convincingly demonstrate improved exploration and stability under high-entropy/multimodal settings.", "reasoning": "The reviewer operates on a sufficiency heuristic: broad benchmarks combined with results consistent with the theory are deemed sufficient evidence, without requiring rigorous statistical analysis or specific baseline scrutiny.", "judgment": "The empirical claims are accepted as sufficiently demonstrated.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8307498693466187, "reasoning_key": "merit_recognition", "reasoning_sim": 0.787997305393219}, {"unit_index": 4, "inspected_object": "Theoretical consistency and organization", "observation": "The reviewer states theoretical results are internally consistent and logically organized.", "reasoning": "The reviewer signals that they followed the logical chain from assumptions to conclusions and found no gaps or contradictions, viewing this internal coherence as a necessary condition for endorsement.", "judgment": "The paper's argumentative structure is sound and coherent.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8131136894226074, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7621108889579773}, {"unit_index": 5, "inspected_object": "Degree of contribution (extensions vs. new formulations)", "observation": "The reviewer identifies that some theoretical contributions are mainly extensions of established analysis.", "reasoning": "While identifying this as a limitation in the degree of contribution, the reviewer softens the judgment by noting the extensions clearly present and meaningfully deepen understanding, resolving the tension between incrementalism and value.", "judgment": "The work has a modest degree of novelty in parts but remains valuable due to clear presentation and deepened understanding.", "valence": "mixed", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.7995195388793945, "reasoning_key": "merit_recognition", "reasoning_sim": 0.758719801902771}, {"unit_index": 6, "inspected_object": "Parameter sensitivity (K and α)", "observation": "The reviewer asks how sensitive results are to the number of mixture components K and entropy coefficient α.", "reasoning": "The reviewer seeks to understand boundary conditions and whether observed benefits are robust across the hyperparameter space or emerge only in narrow regimes.", "judgment": "Uncertainty remains regarding the robustness of benefits across different parameter settings.", "valence": "uncertain", "suggested_improvement": "Clarify sensitivity to the number of mixture components K and entropy coefficient α.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8734652996063232, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7983406782150269}, {"unit_index": 7, "inspected_object": "Generalization to non-Gaussian/discrete mixtures", "observation": "The reviewer asks if similar robustness holds for non-Gaussian or discrete mixture families.", "reasoning": "The reviewer explores whether the theoretical results are specific to Gaussian mixtures or extendable to other mixture families, seeking to define the scope of the robustness argument.", "judgment": "Uncertainty exists regarding the generalizability of the robustness findings beyond Gaussian mixtures.", "valence": "uncertain", "suggested_improvement": "Investigate or discuss whether robustness holds for non-Gaussian or discrete mixture families.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8312642574310303, "reasoning_key": "robustness_norm", "reasoning_sim": 0.822178840637207}]}, {"review_id": "040XScoWwF", "paper_id": "iPCgCxmSR4", "paper_title": "Fast Block Attention Computation via Dynamic Algorithm", "decision": null, "summary": "The reviewer evaluates the paper primarily through the lens of presentation completeness and adherence to structural norms, identifying multiple specific defects in notation, assumption motivation, and organization that collectively lead to the judgment that the paper is unfinished and problematic.", "units": [{"unit_index": 0, "inspected_object": "Notation in Definition 1.2 (symbols `hat{A}_{I,j}`, `hat{A}_{[I,j]}`, `n` vs `[n]`)", "observation": "The proposed notations are confusing and potentially ambiguous.", "reasoning": "Notation serves as a contract with the reader; violations indicate carelessness and hinder comprehension.", "judgment": "The presentation is problematic due to unclear notation.", "valence": "negative", "suggested_improvement": "Clarify the notation to be unambiguous and self-contained.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.6956700682640076, "reasoning_key": "presentation_trust", "reasoning_sim": 0.793533205986023}, {"unit_index": 1, "inspected_object": "Undefined symbol `H` near Figure 1", "observation": "The symbol `H` appears on the third line after Figure 1 without definition.", "reasoning": "Presenting undefined symbols suggests sloppy presentation and breaks the expectation of a self-contained artifact.", "judgment": "The paper exhibits sloppy presentation practices.", "valence": "negative", "suggested_improvement": "Define all symbols used, including `H`, to ensure clarity.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.712618350982666, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8371082544326782}, {"unit_index": 2, "inspected_object": "Assumptions 1.3 and 1.4", "observation": "The assumptions are stated but lack justification and explanation of their implications.", "reasoning": "Assumptions must be motivated with narratives explaining why they are natural or necessary; otherwise, they appear arbitrary.", "judgment": "The theoretical foundation lacks sufficient motivation and clarity.", "valence": "negative", "suggested_improvement": "Provide justification and explain the implications for Assumptions 1.3 and 1.4.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8119525909423828, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7762079834938049}, {"unit_index": 3, "inspected_object": "Section placement (Section 3: Related Work)", "observation": "Related work is placed in the middle of problem development rather than at the beginning or end.", "reasoning": "Standard academic structure places related work at the start or end to avoid interrupting the logical flow of problem setup.", "judgment": "The organization is 'a bit strange' and violates conventional structural norms.", "valence": "negative", "suggested_improvement": "Move Section 3 (Related Work) to the beginning or end of the paper.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "related_work", "object_sim": 0.8272645473480225, "reasoning_key": "presentation_trust", "reasoning_sim": 0.6977432370185852}, {"unit_index": 4, "inspected_object": "Structure and conclusion of Section 4", "observation": "The section ends abruptly after lemmas without experiments or discussion.", "reasoning": "A theory paper without experimental results supporting complexity analysis is considered incomplete by the reviewer's standards.", "judgment": "The paper appears unfinished.", "valence": "negative", "suggested_improvement": "Include experimental results to support the complexity analysis.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7800798416137695, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7836130261421204}]}, {"review_id": "012rrgl64k", "paper_id": "HPeiH7da0Z", "paper_title": "Sculptor: Empowering LLMs with Cognitive Agency via Active Context Management", "decision": "Accept (Poster)", "summary": "The reviewer engages in methodological auditing, accepting the empirical results but questioning the epistemic status of the contributions regarding scalability, autonomy, and comparative validity. The evaluation focuses on gaps in analysis (RL stability), comparison (related work), and autonomy (prompt engineering), framed as areas requiring clarification rather than fatal flaws.", "units": [{"unit_index": 0, "inspected_object": "RL training formulation (GSPO) and algorithmic choice", "observation": "The reviewer identifies GSPO as the specific algorithm used and notes it is 'known to suffer from' credit assignment issues in long-horizon tasks and training stability problems, while questioning why this variant was chosen over others like GRPO/PPO/XPO.", "reasoning": "The reviewer invokes an implicit standard of RL robustness and principled method selection, assuming a body of literature on policy-gradient instability that challenges the reliability of the current approach without direct empirical evidence of failure in this specific paper.", "judgment": "Uncertainty regarding whether the authors understand the theoretical underpinnings of their method's success or if they are relying on arbitrary choices that may not generalize.", "valence": "negative", "suggested_improvement": "Articulate the design rationale for choosing GSPO over other variants (GRPO/PPO/XPO) and demonstrate understanding of its stability properties.", "support_status": "memo_inferred", "confidence": "low", "object_key": "method_design", "object_sim": 0.861668050289154, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8456583023071289}, {"unit_index": 1, "inspected_object": "Tool design principles and protocol adaptability", "observation": "The reviewer observes that the framework uses six tools with strong considerations for determinism and reversibility but questions whether this fixed, human-designed structure limits the system's ability to evolve or self-adapt.", "reasoning": "The reviewer applies a standard of scalability and self-sufficiency, probing for a vision where context management is fully learned rather than statically scaffolded by humans, suggesting a tension between active model agency and static tooling.", "judgment": "Concern that the hybrid human-machine system represents a limitation in autonomy rather than a complete solution, leaving the boundary between human scaffolding and learned behavior undefined.", "valence": "conditional", "suggested_improvement": "Clarify whether the authors envision the tools themselves being learned or evolved, addressing the gap between human-designed artifacts and autonomous model capabilities.", "support_status": "memo_inferred", "confidence": "low", "object_key": "reproducibility", "object_sim": 0.8098448514938354, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7647169232368469}, {"unit_index": 2, "inspected_object": "Comparisons with external memory systems (MemGPT, MemoryLLM)", "observation": "The reviewer notes the absence of comparisons with MemGPT and MemoryLLM, despite the paper positioning itself against external memory as the primary solution.", "reasoning": "The reviewer assumes that close alternatives in different design spaces should be compared to validate the paper's claims of superiority or complementarity, challenging the paper's framing that ACM and external memory are purely complementary without empirical demonstration.", "judgment": "Doubt about whether the approach is genuinely better than alternatives or simply different, indicating a lack of engagement with the competitive landscape.", "valence": "negative", "suggested_improvement": "Include comparisons with MemGPT and MemoryLLM to empirically demonstrate the relationship (complementary vs. competing) between the proposed method and external memory systems.", "support_status": "memo_inferred", "confidence": "low", "object_key": "compute_cost", "object_sim": 0.8103640675544739, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8273258805274963}, {"unit_index": 3, "inspected_object": "Manual prompt engineering and zero-shot results", "observation": "The reviewer infers that the impressive zero-shot results may depend on carefully crafted prompts, suspecting hidden human labor that undermines the claim of 'inherent' tool-calling capabilities.", "reasoning": "The reviewer applies a standard of cognitive autonomy, arguing that if performance relies on manual engineering, the system's claimed agency is overstated and the results do not reflect the model's inherent capabilities as strongly as presented.", "judgment": "Suspicion that the paper's cognitive-agency framing overstates the model's autonomy due to reliance on human-crafted prompts, creating a gap between the observed results and the claimed mechanism.", "valence": "negative", "suggested_improvement": "Demonstrate that the results are robust to less engineered prompts or clarify the extent of human involvement required to achieve the reported performance.", "support_status": "memo_inferred", "confidence": "low", "object_key": "method_design", "object_sim": 0.8457085490226746, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7489097714424133}]}, {"review_id": "021Iwg61Ox", "paper_id": "tmIRNo66Rg", "paper_title": "TriVLA: A Triple-System-Based Unified Vision-Language-Action Model with Episodic World Modeling for General Robot Control", "decision": "Reject", "summary": "The reviewer performs a gatekeeping function focused on scientific hygiene, interrogating the epistemic scaffolding of the paper by probing for missing problem framing, potential evaluation leakage, unclear novelty boundaries, and terminological ambiguity.", "units": [{"unit_index": 0, "inspected_object": "Problem formulation and evaluation protocol in Section 3", "observation": "The paper introduces only the VLA mode without specifying the learning paradigm (supervised vs. imitation learning) or defining 'Zero-shot long-horizon evaluation'.", "reasoning": "The omission prevents locating the work within standard ML evaluation paradigms, raising suspicion that 'zero-shot' is used loosely to conflate 'not fine-tuned on the exact task' with a stronger claim, potentially inflating perceived performance.", "judgment": "The problem setup is insufficiently specified to judge the evaluation against community norms.", "valence": "negative", "suggested_improvement": "Specify the learning paradigm and clearly define the train/test splits and the meaning of 'Zero-shot long-horizon evaluation'.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7957544326782227, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7953661680221558}, {"unit_index": 1, "inspected_object": "Data provenance for System 3 (video diffusion model)", "observation": "System 3 is fine-tuned on 'self-collected data' without description of its correlation to evaluation tasks.", "reasoning": "If the fine-tuning data shares distributional overlap with test tasks, results may reflect memorization (evaluation leakage) rather than generalizable mechanism; testing the pretrained SVD without fine-tuning would isolate this contribution.", "judgment": "The possibility of evaluation leakage remains unaddressed, undermining trust in the reported gains.", "valence": "negative", "suggested_improvement": "Provide an ablation using the pretrained SVD model without fine-tuning to rule out data leakage.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8178877234458923, "reasoning_key": "design_justification", "reasoning_sim": 0.8427661657333374}, {"unit_index": 2, "inspected_object": "Novelty boundary relative to prior work (single-step prediction and Seer)", "observation": "The reviewer questions whether multi-step future prediction is a trivial extension of single-step methods and if the difference from Seer is well explained.", "reasoning": "Incremental extensions are not contributions unless non-trivial; if Seer already uses inverse dynamics conditioned on forecasted states, the proposed pipeline may be a re-implementation with different branding lacking mechanistic distinction.", "judgment": "The specific mechanistic or architectural novelty relative to existing literature is not sufficiently articulated.", "valence": "negative", "suggested_improvement": "Articulate the specific mechanistic or architectural differences from Seer and analyze how the number of predicted steps affects performance.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.9030486345291138, "reasoning_key": "design_justification", "reasoning_sim": 0.7368082404136658}, {"unit_index": 3, "inspected_object": "Technical definition of 'action flow-matching' (Equation 5 vs Equation 1)", "observation": "The reviewer suspects Equation (5) is qualitatively similar to the diffusion loss in Equation (1).", "reasoning": "Terminology should not obscure technical content; if the new term represents a standard diffusion-style loss, it constitutes conceptual inflation rather than a novel objective.", "judgment": "The terminology appears to potentially obscure the lack of novelty in the loss formulation.", "valence": "negative", "suggested_improvement": "Provide a conceptual explanation of why the two formulations are different in kind, not just in variable names.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7567744851112366, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7462584376335144}]}, {"review_id": "02Ahn4CjrR", "paper_id": "XRYwcQg2C6", "paper_title": "Black-box Optimization of LLM Outputs by Asking for Directions", "decision": "Reject", "summary": "The reviewer validates the core empirical insight but raises significant concerns about the standalone attribution of the method's contribution versus known techniques, the robustness to simple refusal defenses, unexplained performance variability across models, and the clarity of query efficiency metrics.", "units": [{"unit_index": 0, "inspected_object": "The empirical validation of the core premise that LLMs are poorly calibrated for absolute confidence but well-calibrated for binary comparisons, specifically Figure 3.", "observation": "Figure 3 contrasts the failure of absolute scoring with the success of binary comparison against ground-truth logits, providing convincing evidence.", "reasoning": "The reviewer values the comparison to a known-good signal (ground-truth logits) as the appropriate benchmark to establish that the comparative prompt extracts calibrated information rather than noise.", "judgment": "The validation is accepted as convincing and the central insight is deemed novel and significant.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7054252028465271, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7829793691635132}, {"unit_index": 1, "inspected_object": "The attribution of performance gains in the 'Transfer+ours' hybrid method shown in Table 1.", "observation": "High ASRs (e.g., 94.7% on GPT-4o mini) are achieved with the hybrid method, where the marginal improvement from the optimization step is small (1.8%) compared to the baseline transfer attack (92.9%).", "reasoning": "The reviewer applies a decompositional standard: a paper's central contribution should be demonstrably responsible for a meaningful portion of the reported success. The arithmetic suggests the known transfer attack does most of the work.", "judgment": "There is a potential gap between the paper's framing and its evidence regarding the standalone contribution of the proposed method.", "valence": "negative", "suggested_improvement": "Clarify or demonstrate the standalone contribution of the novel component separate from the known transfer technique.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8359820246696472, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7888951301574707}, {"unit_index": 2, "inspected_object": "The variability of performance gains across different model families (GPT vs. Claude) in Table 1.", "observation": "The benefit from the 'Transfer+ours' method is highly variable: tiny gain for GPT models (1.8%) but massive gain for Claude models (35.1% to 59.6%), which is not analyzed by the authors.", "reasoning": "This unexplained pattern creates interpretive ambiguity: it could mean the transfer attack is already near-perfect for GPT, or that Claude models provide a better optimization signal. The discrepancy needs explanation to properly frame the contribution.", "judgment": "The lack of analysis for this significant discrepancy is a gap in the paper's evaluation.", "valence": "negative", "suggested_improvement": "Analyze the reasons for the significant discrepancy in gains between GPT and Claude models.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8244505524635315, "reasoning_key": "design_justification", "reasoning_sim": 0.8535244464874268}, {"unit_index": 3, "inspected_object": "The robustness of the attack to simple defenses, specifically model refusals based on safety guidelines.", "observation": "The attack depends on the model's willingness to answer the comparative meta-prompt, which can be suppressed by a defense such as 'I cannot compare prompts in a way that might lead to a harmful outcome'.", "reasoning": "A security contribution should be resilient to obvious countermeasures. The reviewer identifies this dependence as a critical vulnerability of the attack itself, treating it as a practical attack surface issue.", "judgment": "The attack's susceptibility to refusal-based defenses is a significant weakness not adequately addressed.", "valence": "negative", "suggested_improvement": "Experiment with iteratively re-prompting or reformulating the comparison prompt to overcome such refusals.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7551610469818115, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8344820141792297}, {"unit_index": 4, "inspected_object": "The broader theoretical implications of the finding that larger/more capable models are more vulnerable to this attack.", "observation": "The finding shows strong evidence across model families (e.g., Qwen-VL-72B > 7B, GPT-5 mini > GPT-4o mini) that advanced reasoning capabilities correlate with vulnerability.", "reasoning": "The reviewer probes whether this implies that alignment techniques relying on advanced reasoning (like self-critique or Constitutional AI) are fundamentally flawed because that same capability can be turned against the model.", "judgment": "The paper has potentially consequential implications for alignment research that warrant exploration.", "valence": "conditional", "suggested_improvement": "Engage with the broader significance regarding whether reasoning-based alignment may be self-defeating.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.6579459309577942, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7874566316604614}, {"unit_index": 5, "inspected_object": "The query efficiency of the vision-LLM attacks compared to jailbreak attacks.", "observation": "Table 3 provides query counts for jailbreaks (4.9-79.3 queries), described as very efficient, but the average number of queries for successful vision attacks is not clearly reported despite a budget of 1,000.", "reasoning": "Practicality is an evaluative dimension; a practical attack should not require near the maximum budget. The reviewer suspects vision attacks may regularly require hundreds of queries, which would temper their practical significance compared to jailbreaks.", "judgment": "The efficiency of the vision attacks is unclear and potentially lower than jailbreaks, creating uncertainty about their practical utility.", "valence": "uncertain", "suggested_improvement": "Report the average number of queries for successful vision attacks to allow comparison with jailbreak efficiency.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7931443452835083, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8167230486869812}]}, {"review_id": "02j2h5v0Wy", "paper_id": "USjSdem7WO", "paper_title": "Just a Simple Transformation is Enough for Data Protection in Split Learning", "decision": null, "summary": "The reviewer systematically dismantles the paper by identifying a foundational category error in problem framing, insufficient attack baselines, unverified theoretical claims, metric inconsistencies, limited experimental design, and questionable interpretation of visual results.", "units": [{"unit_index": 0, "inspected_object": "Problem framing distinction between Vertical Federated Learning (VFL) and Two-Party Split Learning (SL)", "observation": "The paper describes a scenario where the server holds no private data and only receives intermediate representations, which aligns with Two-Party Split Learning rather than VFL.", "reasoning": "The reviewer asserts that this is a category error because the distinction carries substantive consequences for what attacks are possible and what defenses mean; evaluating against the wrong lineage makes the contribution unintelligible within the claimed literature.", "judgment": "The paper's entire framing is built on a mis-specified problem setting.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7968840599060059, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7760817408561707}, {"unit_index": 1, "inspected_object": "Selection of attack baselines (UnSplit, FSHA)", "observation": "The paper evaluates only UnSplit and FSHA, characterized as outdated methods developed before 2022, omitting more recent and powerful attacks like PCAT, FORA, and SDAR.", "reasoning": "A defense paper must be tested against the strongest available adversaries; the reviewer invokes the counterfactual that newer attacks may not experience the same poor results on MLP models, suggesting the central empirical finding might be an artifact of weak attack selection.", "judgment": "The attack baselines are insufficient to validate the resistance claims.", "valence": "negative", "suggested_improvement": "Evaluate against recent and powerful attacks such as PCAT, FORA, and SDAR to rule out the possibility that newer attacks succeed where older ones fail.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8358703255653381, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8330650329589844}, {"unit_index": 2, "inspected_object": "Theoretical Remarks 1 and 3 regarding attacker reconstruction capabilities", "observation": "The paper claims an attacker cannot reconstruct initial data X without theoretical proof or experimental verification, and conflates client transformations with information available to the attacker.", "reasoning": "The reviewer argues that if activations (H) remain the same, the attacker's inference problem is unchanged regardless of client-side transformations; furthermore, many attacks work with auxiliary data from different distributions, undermining the premise that prior distribution knowledge is necessary.", "judgment": "The theoretical claims are unsupported and conceptually confused.", "valence": "negative", "suggested_improvement": "Provide theoretical proof or experimental verification for reconstruction claims and clarify the independence of attacker information from client-side model/input transformations.", "support_status": "mixed", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.809370756149292, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8444226980209351}, {"unit_index": 3, "inspected_object": "Evaluation metrics (FID vs MSE) and internal consistency", "observation": "Table 1 shows MLP achieving better MSE than CNN on F-MNIST, while CNN achieves better FID than MLP on CIFAR-10, contradicting the paper's general conclusions about MLP superiority.", "reasoning": "The reviewer treats these results as internally contradictory evidence that the claimed pattern of MLP superiority does not hold uniformly across metrics and datasets, questioning why standard image quality metrics like SSIM, PSNR, and LPIPS are absent.", "judgment": "The evaluation metrics reveal inconsistencies that undermine the paper's conclusions.", "valence": "negative", "suggested_improvement": "Reconcile metric inconsistencies by including standard image quality metrics (SSIM, PSNR, LPIPS) and addressing why MLP superiority is not consistent across all metrics and datasets.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8699102401733398, "reasoning_key": "construct_validity", "reasoning_sim": 0.8060295581817627}, {"unit_index": 4, "inspected_object": "Experimental design regarding cut layer configuration and architecture diversity", "observation": "Only layer 1 is evaluated, where UnSplit performs poorly, and the study lacks diverse MLP architectures or MLP layers in CNN models.", "reasoning": "The reviewer infers that the paper chose the most favorable condition (shallow cut layer) rather than testing robustness comprehensively; a comparative claim requires systematic variation to isolate whether resistance is due to architecture class or other factors.", "judgment": "The experimental design is insufficiently thorough and potentially biased toward favorable results.", "valence": "negative", "suggested_improvement": "Test deeper cut layers and use diverse MLP architectures, including adding MLP layers to CNN models, to systematically test robustness.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7958588004112244, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7885982394218445}, {"unit_index": 5, "inspected_object": "Interpretation of visual reconstruction results in Figure 5 and Figure 7", "observation": "Attack results on MLP models show complete mismatches (e.g., '0' reconstructed as '1') rather than blurry outputs.", "reasoning": "The reviewer argues that a genuinely ineffective attack would produce degraded but recognizable outputs (blur), whereas complete mismatch suggests the attack's optimization landscape is broken or hyperparameters are inappropriate, reinterpreting the paper's evidence of success as evidence of experimental failure.", "judgment": "The visual results indicate poor optimization or inappropriate hyperparameters rather than fundamental attack weakness.", "valence": "negative", "suggested_improvement": "Investigate and report on hyperparameter settings and optimization landscapes to distinguish between attack ineffectiveness and experimental artifacts.", "support_status": "memo_inferred", "confidence": "low", "object_key": "clarity", "object_sim": 0.8216638565063477, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.815944492816925}, {"unit_index": 6, "inspected_object": "Scholarly presentation norms (references and LLM disclosure)", "observation": "References lack conference or journal names, and there is no disclosure of LLM usage as required by author guidelines.", "reasoning": "These issues signal a failure to adhere to scholarly norms and venue-specific requirements, representing procedural deficiencies separate from scientific content.", "judgment": "The presentation fails to meet basic scholarly standards.", "valence": "negative", "suggested_improvement": "Complete reference citations with venue details and disclose LLM usage in accordance with author guidelines.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8240604996681213, "reasoning_key": "presentation_trust", "reasoning_sim": 0.78883296251297}]}, {"review_id": "02qgPsD239", "paper_id": "A9sAYUD3XB", "paper_title": "HiST: Spatial Transcriptomics Prediction via Multi-Level Hyperbolic Representation Learning", "decision": null, "summary": "The reviewer questions whether the paper's central claim of hierarchical representation learning is adequately demonstrated, citing insufficient task-specific evaluation, lack of ablation for alignment modules, narrow baselines, and questionable dataset/niche definitions.", "units": [{"unit_index": 0, "inspected_object": "Evaluation of hierarchical feature utility via downstream tasks", "observation": "The only downstream task evaluated is gene expression prediction, with minimal mention of MSI status classification.", "reasoning": "If the method's value proposition is learning hierarchical features, evaluation should demonstrate utility for tasks where hierarchy matters or isolate the hierarchy's identifiable work; using a general performance task does not verify the specific claim.", "judgment": "The central conceptual claim that learned features are truly hierarchical is underdetermined by the evidence.", "valence": "negative", "suggested_improvement": "Evaluate on tasks where hierarchy specifically matters (e.g., MSI status classification) or provide ablations isolating the hierarchy's contribution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8406172394752502, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8218297362327576}, {"unit_index": 1, "inspected_object": "Functional role of alignment objectives (HCA/HEA)", "observation": "It remains unclear whether the alignment modules function as pre-training objectives rather than contributing to inference-time hierarchical structure.", "reasoning": "If the modules act as pre-training, the end-to-end training scheme conflates distinct phases; without separating these, it is unknown if the hierarchical alignment is the operative mechanism or merely a replaceable regularizer.", "judgment": "Uncertainty regarding the mechanistic necessity and functional role of the proposed alignment modules.", "valence": "conditional", "suggested_improvement": "Conduct an experiment separating pre-training from fine-tuning to determine if the method performs as well without the specific hierarchical alignment at inference.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7971863150596619, "reasoning_key": "design_justification", "reasoning_sim": 0.7454100251197815}, {"unit_index": 2, "inspected_object": "Comparative baselines for multi-scale/hierarchical capture", "observation": "The comparison set may be too narrow, lacking methods like DINOV2 that also capture global and local features through different mechanisms.", "reasoning": "A method claiming to exploit hierarchy should be compared against other methods capturing multi-scale information to establish that hyperbolic hierarchical alignment is the 'right' way to do so.", "judgment": "Insufficient evidence that the proposed hyperbolic approach is superior to alternative multi-scale capture mechanisms.", "valence": "negative", "suggested_improvement": "Include comparisons against alternative encoders that capture multi-scale information (e.g., DINOV2).", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8414373993873596, "reasoning_key": "design_justification", "reasoning_sim": 0.8071914315223694}, {"unit_index": 3, "inspected_object": "Dataset selection for validating hierarchical design", "observation": "The paper does not use Visium HD, which provides multi-level gene expression data aligning with the paper's goal, nor does it justify the representativeness of the three chosen tissues.", "reasoning": "Testing on unambiguously hierarchical data (Visium HD) is necessary to show design choices are necessary rather than incidental; limited tissue types raise concerns about generalizability.", "judgment": "Weakness in experimental scope reduces confidence in the generalizability and necessity of the method's design.", "valence": "negative", "suggested_improvement": "Test the method on Visium HD data and expand the range of tissue types to support broader conclusions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8524017930030823, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7969863414764404}, {"unit_index": 4, "inspected_object": "K-nearest neighbors choice for niche definition", "observation": "Using KNN imposes a discrete structure on continuous pathology image data, potentially introducing bias and reducing construct validity.", "reasoning": "Pathology images are continuous; defining niches via KNN creates a computational convenience rather than reflecting discovered biological properties; expanding context would offer a smoother, more faithful notion of locality.", "judgment": "The operationalization of 'niche' may not match the conceptual notion of 'hierarchy', raising questions about construct validity.", "valence": "negative", "suggested_improvement": "Consider expanding context around the central spot instead of using KNN to define niches, to avoid imposing artificial discreteness.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7881494164466858, "reasoning_key": "design_justification", "reasoning_sim": 0.7452350854873657}]}, {"review_id": "02tXBeKVoz", "paper_id": "zZyxHmId3w", "paper_title": "MIRACLE: Model-free Imitation and Reinforcement Learning for Adaptive Cut-Selection", "decision": "Accept (Poster)", "summary": "The reviewer performs a rigorous feasibility audit, identifying significant gaps in motivation, related work, implementation detail, and experimental justification, while acknowledging the work's potential direction.", "units": [{"unit_index": 0, "inspected_object": "The paper's central claim of 98.1% memory reduction and its practical motivation.", "observation": "The reviewer finds the memory-reduction focus compelling but notes a lack of justification for why memory is a bottleneck in real-world scenarios.", "reasoning": "The implicit standard is that a central contribution must be grounded in concrete problem settings rather than just impressive numbers; without this, the claim floats free of practical significance.", "judgment": "The motivation is underdeveloped and lacks causal attribution to specific method components addressing the bottleneck.", "valence": "negative", "suggested_improvement": "Articulate real-world scenarios where memory is the bottleneck and specify which method components address it.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7429385185241699, "reasoning_key": "novelty_standard", "reasoning_sim": 0.758087158203125}, {"unit_index": 1, "inspected_object": "Related work on learning-based cut selection.", "observation": "Three specific prior works on learning-based cut selection are neither cited nor compared against.", "reasoning": "The evaluative move is that if a paper cannot position itself against the closest existing methods, its claimed novelty is suspect and the omission is unfair to prior contributions.", "judgment": "The related-work section is incomplete, raising suspicion about the novelty claim.", "valence": "negative", "suggested_improvement": "Cite and compare against the three identified prior works on learning-based cut selection.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.82428377866745, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8505014181137085}, {"unit_index": 2, "inspected_object": "The training pipeline, specifically the discriminator network and phase training order.", "observation": "The reviewer repeatedly asks about the discriminator's training data, procedure, integration with PPO, and whether phases are trained sequentially.", "reasoning": "The implicit standard is that a learning method must be described at an operational level sufficient for a competent researcher to reimplement it; mathematical formalism alone is insufficient for soundness and reproducibility.", "judgment": "The implementation details are opaque, casting doubt on the method's soundness and reproducibility.", "valence": "negative", "suggested_improvement": "Provide detailed descriptions of the discriminator's training data, procedure, and the sequential or parallel nature of the phase training.", "support_status": "mixed", "confidence": "high", "object_key": "method_design", "object_sim": 0.8447865843772888, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8060604333877563}, {"unit_index": 3, "inspected_object": "Experimental design choices including instance selection, time limits, and metrics.", "observation": "The reviewer questions the criteria for selecting 300 test instances, the rationale for the 600-second time limit and 12 GB memory cap, and the use of a custom 'success rate' metric.", "reasoning": "The concern is that experimental constraints may cherry-pick easy cases (excluding hard MIPLIB instances) and that bespoke metrics obscure comparative performance relative to standardized measures like primal gap.", "judgment": "The experimental setup may be biased and lacks transparent justification, potentially favoring the method.", "valence": "negative", "suggested_improvement": "Justify instance selection criteria, explain experimental constraints, and adopt widely adopted metrics like primal gap or final objective value.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8198021054267883, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8221156001091003}, {"unit_index": 4, "inspected_object": "The expert policy choice (SCIP's heuristic cut selection).", "observation": "The reviewer asks why SCIP's heuristic rules serve as the expert, noting that stronger solvers or human-designed oracles might provide better supervision.", "reasoning": "The implicit norm is that the quality of the learned policy is bounded by the quality of the expert; using a heuristic expert may limit the potential performance gains.", "judgment": "The choice of expert policy is questionable and may constrain the method's effectiveness.", "valence": "conditional", "suggested_improvement": "Discuss or experiment with stronger experts or human-designed oracles to bound the policy quality.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8006964325904846, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7678697109222412}, {"unit_index": 5, "inspected_object": "The curriculum learning scheme.", "observation": "The reviewer requests an ablation to quantify the contribution of the curriculum learning scheme.", "reasoning": "Mechanistic understanding requires isolating components to determine their individual impact on performance.", "judgment": "The contribution of the curriculum learning scheme is unquantified.", "valence": "uncertain", "suggested_improvement": "Perform an ablation study to quantify the contribution of the curriculum learning scheme.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7340763211250305, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7846480011940002}]}, {"review_id": "031eDTtSs7", "paper_id": "4kz4586euw", "paper_title": "Simulation-Free Structure Learning for Stochastic Dynamics", "decision": "Reject", "summary": "The reviewer critically assesses the epistemic warrant of the method, focusing on three main gaps: the lack of identifiability justification for using sparse neural networks to infer structure, the insufficient differentiation from prior work ([SF]2M), and the overstated claims regarding causal/mechanistic understanding given the black-box nature of the model.", "units": [{"unit_index": 0, "inspected_object": "Justification for learning network structure via sparse neural network", "observation": "The paper lacks a principled justification for why the first sparse weight matrix of a highly non-linear neural network encodes the right network structure.", "reasoning": "In multi-layer non-linear functions, sparsity in the first layer corresponds to the approximator's dependency structure, not necessarily the system's causal or functional dependencies. Unlike closed-form methods (e.g., SINDy) where sparsity has clear semantic meaning, neural networks require an identifiability argument to ensure optimization recovers the true structure rather than just a fitting one.", "judgment": "Methodological gap: The interpretive link between learned sparsity and true system structure is unsupported.", "valence": "negative", "suggested_improvement": "Articulate an identifiability argument explaining why the optimization procedure recovers the true structure.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8087108135223389, "reasoning_key": "design_justification", "reasoning_sim": 0.8363686203956604}, {"unit_index": 1, "inspected_object": "Novelty relative to prior method [SF]2M", "observation": "The contribution appears to be a slight modification in the learning objective relative to [SF]2M.", "reasoning": "The reviewer perceives substantial overlap in the Schrödinger Bridge formulation. Without explicit articulation of the delta, the contribution risks being incremental if the only difference is the sparsity regularizer.", "judgment": "Positioning concern: The novelty claim is unsubstantiated by comparison.", "valence": "negative", "suggested_improvement": "Explicitly articulate the differences from [SF]2M to demonstrate genuine novelty.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.9336432218551636, "reasoning_key": "design_justification", "reasoning_sim": 0.8179135322570801}, {"unit_index": 2, "inspected_object": "Claims of learning causal dependencies", "observation": "The paper claims to learn causal dependencies and provide mechanistic understanding, but uses a black-box neural SDE.", "reasoning": "For neural systems, analyzing qualitative properties like bifurcations or attractors is largely an open question and not straightforward. A structure learning method should enable downstream scientific inference, not just produce a sparse matrix. The epistemic payoff is questionable if the model cannot be analyzed qualitatively.", "judgment": "Overstated claim: The epistemic warrant for causal/mechanistic claims is weak given the black-box nature.", "valence": "negative", "suggested_improvement": "Address how the learned model enables downstream scientific inference or qualify the causal claims.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8414045572280884, "reasoning_key": "design_justification", "reasoning_sim": 0.8266069889068604}]}, {"review_id": "039k1PwHy9", "paper_id": "ajnBafpqmE", "paper_title": "Aligning Visual Foundation Encoders to Tokenizers for Diffusion Models", "decision": "Accept (Poster)", "summary": "The reviewer employs a quantitative, parameter-focused strategy, identifying weaknesses primarily through unanswered questions about computational cost, scaling principles, regularization robustness, semantic-reconstruction trade-offs, and CFG dependency.", "units": [{"unit_index": 0, "inspected_object": "Computational cost of the three-stage alignment pipeline", "observation": "The paper does not report overall computational costs for all stages or compare them to other methods.", "reasoning": "A three-stage pipeline is implicitly assumed to be more expensive than end-to-end VAE training; without quantified costs, the method's practical viability remains uncertain.", "judgment": "The absence of cost data creates uncertainty regarding the method's practical utility.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8123658895492554, "reasoning_key": "cost_benefit", "reasoning_sim": 0.735844612121582}, {"unit_index": 1, "inspected_object": "Latent dimensionality scaling (32-dimensional latent)", "observation": "The reviewer questions whether generalizing to higher resolutions (e.g., ImageNet 512) requires only doubling the latent dimension and notes uncertainty about the principled nature of the 32-d choice.", "reasoning": "The phrasing indicates a probe for design principles rather than a definitive flaw; the lack of guidance on scaling makes the architectural choice appear potentially arbitrary.", "judgment": "Uncertainty exists regarding whether the latent dimension choice is principled or ad hoc.", "valence": "uncertain", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8282333612442017, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7495833039283752}, {"unit_index": 2, "inspected_object": "L2 regularization robustness", "observation": "The reviewer notes that L2 regularization may not be robust, particularly if the latent space lacks proper normalization.", "reasoning": "This is a hypothesis about potential failure modes based on the assumption that meaningful L2 regularization requires specific latent space properties which may not be verified.", "judgment": "Potential technical fragility in the method's assumptions.", "valence": "conditional", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8219606876373291, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8267500996589661}, {"unit_index": 3, "inspected_object": "Linear probing accuracy during training", "observation": "The drop in linear probing accuracy during training raises concerns about the trade-off between reconstruction fidelity and semantic capacity.", "reasoning": "The method claims to preserve semantics, yet the observed drop suggests the L2 regularization might not be working as intended or the claim is overstated, revealing an internal tension in the results.", "judgment": "Concerns about the effectiveness of semantic preservation relative to reconstruction.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8477565050125122, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7868140339851379}, {"unit_index": 4, "inspected_object": "rFID decrease", "observation": "The rFID decrease suggests that constraints on the adaptor and the stage-wise framework impact reconstruction accuracy.", "reasoning": "The reviewer infers from this metric that the stage-wise approach may be trading reconstruction quality for semantic richness, a trade-off the paper may not explicitly acknowledge.", "judgment": "Questioning whether reconstruction quality is sacrificed for semantic gains.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.6347812414169312, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8312512040138245}, {"unit_index": 5, "inspected_object": "Performance without classifier-free guidance (CFG)", "observation": "The paper claims improvement with and without CFG, but the reviewer asks for the performance before/during non-CFG settings.", "reasoning": "The question functions as a verification request to confirm that the method's benefit is not an artifact of the CFG setting, given the abstract's explicit claim.", "judgment": "Uncertainty about the generality of the generation improvement independent of CFG.", "valence": "uncertain", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8499559164047241, "reasoning_key": "novelty_standard", "reasoning_sim": 0.733798623085022}]}, {"review_id": "03BPnUaXE9", "paper_id": "iVeqLVyWO8", "paper_title": "One Model to Critique Them All: Rewarding Agentic Tool-Use via Efficient Reasoning", "decision": null, "summary": "The reviewer performs a mixed verification and gap-analysis, praising the data pipeline's construction quality while critically probing for hidden failure modes such as artifact exploitation and brittle reasoning. The evaluation hinges on a normative standard requiring demonstration that models learn genuine competence rather than shortcuts, leading to a qualified endorsement contingent on deeper introspection into error analysis and artifact bias.", "units": [{"unit_index": 0, "inspected_object": "ToolPref-Pairwise-30K data pipeline", "observation": "The reviewer checked some data samples and found them of high quality, noting the pipeline was carefully devised.", "reasoning": "Direct empirical verification via sampling supports a positive assessment of the construction quality and internal logic of the data generation process.", "judgment": "The data pipeline is sound and well-constructed.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8012562990188599, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8228794932365417}, {"unit_index": 1, "inspected_object": "Data pipeline design choices (preference intensity bins, complexity scores)", "observation": "The reviewer questions whether these design choices create blind spots or subtle biases/artifacts that reward models could exploit.", "reasoning": "A counterfactual probe assumes that systematic patterns in the data distribution might allow models to learn shortcuts rather than genuine task competence; the paper has not provided negative evidence ruling this out.", "judgment": "There is an unresolved risk that models are exploiting data artifacts rather than learning tool-use competence.", "valence": "negative", "suggested_improvement": "Provide analysis showing that models fail on cases where rule-based scoring would be misleading to demonstrate they are not cheating via surface-level features.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8377501368522644, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7741305232048035}, {"unit_index": 2, "inspected_object": "TRBENCH_{BFCL} benchmark", "observation": "The benchmark is described as an OOD, error-rich environment for fair evaluation.", "reasoning": "The benchmark's fitness for purpose is evaluated based on its ability to provide a challenging test that prevents gaming of in-distribution statistics, though long-term utility depends on community adoption.", "judgment": "The benchmark has potential value but is hedged due to uncertainty about its robustness against simultaneous gaming of training data and benchmark.", "valence": "conditional", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.7632794380187988, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7597832679748535}, {"unit_index": 3, "inspected_object": "Reward model architecture and training details", "observation": "The reviewer does not inspect the model architecture, hyperparameters, or GRPO implementation.", "reasoning": "GRPO is considered well-established in the field, so the modeling side is treated as a known quantity requiring no scrutiny.", "judgment": "The architectural components are standard and not a primary focus of evaluation.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8623093366622925, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.6672044992446899}, {"unit_index": 4, "inspected_object": "Overall contribution novelty", "observation": "The central algorithmic innovations are mostly in data construction and evaluation, with no significant technical contributions on the reward model side shown.", "reasoning": "A normative standard is applied where a sufficient contribution requires either new method/architecture or a dataset/benchmark enabling something previously impossible; the paper falls short on demonstrating necessity of design choices beyond sufficiency.", "judgment": "The contribution is incremental and lacks transformative novelty.", "valence": "negative", "suggested_improvement": "Provide more evidence that the data pipeline's design choices are necessary rather than merely sufficient to achieve reported gains.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8936468362808228, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8269578814506531}, {"unit_index": 5, "inspected_object": "Error analysis and failure modes", "observation": "The review requests more error analysis regarding whether models prefer responses that look correct but are subtly wrong or exhibit brittle reasoning.", "reasoning": "Aggregate accuracy numbers are insufficient; qualitative failure characterization is needed to determine practical utility in downstream applications like Best-of-N sampling.", "judgment": "The current evaluation lacks depth in characterizing how models fail, raising concerns about robustness in real-world scenarios.", "valence": "negative", "suggested_improvement": "Conduct qualitative failure characterization to show how models fail, specifically looking for brittle reasoning or preference for subtly wrong responses.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8273411989212036, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7914820909500122}]}, {"review_id": "03PsCr2JJc", "paper_id": "TX3oGD99CJ", "paper_title": "HiMoE-VLA: Hierarchical Mixture-of-Experts for Generalist Vision–Language–Action Policies", "decision": "Reject", "summary": "The reviewer conducts an architectural audit, accepting empirical results but demanding stronger causal evidence for the MoE design's necessity and effectiveness through comparative ablations and mechanistic routing analysis. Key concerns include internal inconsistencies in performance attribution and insufficient explanation of architectural components.", "units": [{"unit_index": 0, "inspected_object": "The premise that data from different action spaces are largely non-transferable and the justification for using a hierarchical MoE architecture.", "observation": "The reviewer questions whether the paper provides a mechanism explaining why transfer fails, rather than just asserting it, and asks why MoE is chosen over simpler alternatives like separate heads or full parameter sharing.", "reasoning": "The reviewer applies a standard of causal identification to architectural claims, believing the current evidence conflates multiple sources of improvement (e.g., pre-training, hyperparameters) and cannot attribute gains specifically to the MoE design without comparative ablations to isolate the mechanism.", "judgment": "The paper's central technical narrative regarding the necessity and superiority of the MoE architecture is insufficiently supported by the provided evidence.", "valence": "negative", "suggested_improvement": "Perform comparative ablations isolating the MoE mechanism from confounds, such as testing separate heads per action representation, GR00T-style approaches with embodiment embeddings, or fully shared parameters.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.736552894115448, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7935968637466431}, {"unit_index": 1, "inspected_object": "Expert specialization and routing behavior within the AS-MoE and HB-MoE architectures.", "observation": "The reviewer finds limited insight into how experts specialize and route data, noting a lack of mechanistic evidence showing that the architecture actually performs the claimed heterogeneity handling via distinct routing.", "reasoning": "For an architectural claim about handling heterogeneity through routing to be sound, there must be evidence that this specific mechanism is occurring and meaningful, rather than relying solely on aggregate performance numbers.", "judgment": "The review lacks sufficient mechanistic evidence to validate the functional role of the expert routing component.", "valence": "negative", "suggested_improvement": "Provide expert routing analysis across datasets with different action spaces and observations to demonstrate that routing is actually occurring.", "support_status": "mixed", "confidence": "high", "object_key": "method_design", "object_sim": 0.8210993409156799, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7865702509880066}, {"unit_index": 2, "inspected_object": "The 'w/o MoE' variant in Table 6(b) compared against prior VLAs in Table 1.", "observation": "The reviewer observes that the non-MoE baseline already outperforms all prior VLAs listed in Table 1, creating an internal inconsistency with the headline claim that MoE provides consistent gains.", "reasoning": "If a non-MoE variant beats prior work, the causal story attributing gains to the MoE architecture is undermined by the possibility that improvements stem from other pipeline elements like data, training procedure, or flow-matching, representing a confound in the reported results.", "judgment": "The paper's narrative regarding the source of performance gains is internally inconsistent and potentially misleading due to unisolated confounds.", "valence": "negative", "suggested_improvement": "Break down which training pipeline elements contributed most to the performance of the 'w/o MoE' variant to clarify the attribution of gains.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7988932728767395, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8067008852958679}, {"unit_index": 3, "inspected_object": "Clarification of 'MoE re-initialization during fine-tuning' (L240).", "observation": "The reviewer identifies a lack of procedural transparency regarding what exactly is being re-initialized and the rationale behind it.", "reasoning": "Procedural clarity is necessary for reproducibility and understanding the experimental setup; ambiguity here creates uncertainty about the exact conditions under which results were generated.", "judgment": "The experimental methodology description contains ambiguities that hinder full understanding of the fine-tuning process.", "valence": "negative", "suggested_improvement": "Clarify exactly what components are re-initialized during fine-tuning and explain the reasoning for this choice.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7944312691688538, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.838972270488739}, {"unit_index": 4, "inspected_object": "Parameter matching in the comparison between 'Full' and 'w/o MoE' in Table 6(b).", "observation": "The reviewer questions whether active parameters or total parameters are matched in the baseline comparison.", "reasoning": "Methodological precision in parameter counting is critical for ensuring the fairness of the comparison; mismatched metrics could invalidate the conclusion that MoE adds value.", "judgment": "The fairness of the comparative evaluation is uncertain due to ambiguous parameter accounting.", "valence": "uncertain", "suggested_improvement": "Specify whether active or total parameters are matched in the 'Full' vs 'w/o MoE' comparison to ensure methodological rigor.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8009970784187317, "reasoning_key": "construct_validity", "reasoning_sim": 0.8329535126686096}, {"unit_index": 5, "inspected_object": "Code and dataset release status.", "observation": "The reviewer notes the absence of information regarding code and dataset availability.", "reasoning": "Reproducibility is a standard requirement for validating empirical claims; without access to code or data, independent verification is impossible.", "judgment": "The paper currently lacks the necessary resources for independent reproducibility verification.", "valence": "negative", "suggested_improvement": "Release code and datasets to facilitate reproducibility checks.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7930601239204407, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8814849853515625}, {"unit_index": 6, "inspected_object": "Shared expert component functionality.", "observation": "The reviewer seeks to understand the functional role of the shared expert component, noting its purpose is not fully explained.", "reasoning": "Components whose purpose is not clearly defined create opacity in the architectural design, making it difficult to assess their contribution to the overall system.", "judgment": "The functional role of specific architectural components remains unclear.", "valence": "negative", "suggested_improvement": "Elaborate on the functional role of the shared expert component within the architecture.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8174746036529541, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7605987191200256}, {"unit_index": 7, "inspected_object": "Minor presentation issues including broken citations and table aggregation discrepancies.", "observation": "The reviewer identified a broken citation at L249 and a discrepancy in CALVIN aggregation (sum vs avg) in table headers.", "reasoning": "These errors indicate a need for careful proofreading and consistency checking, signaling potential lapses in final quality control despite the substantive strength of the work.", "judgment": "The presentation requires minor corrections to ensure accuracy and consistency.", "valence": "positive", "suggested_improvement": "Fix the broken citation and correct the table header aggregation labels to match the methodology.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8326999545097351, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8174824714660645}]}, {"review_id": "03X0xSDxOG", "paper_id": "22hBwIf7OC", "paper_title": "Plug-and-Play Compositionality for Boosting Continual Learning with Foundation Models", "decision": "Accept (Oral)", "summary": "The reviewer identifies significant concerns regarding experimental validity (baseline adaptation), conceptual novelty (resemblance to distillation), architectural compatibility (logit conflicts), and presentation clarity (complexity). The logic moves from specific technical observations to broader evaluative judgments about fairness, positioning, and reproducibility, driven by norms of benchmark fidelity and modular compatibility.", "units": [{"unit_index": 0, "inspected_object": "Choice and application of Class-Incremental Learning baselines in the experimental comparison protocol", "observation": "The compared methods were not used in the benchmark paper, and it is unclear how these specific CL methods were applied in this setting.", "reasoning": "Fair experimental comparison requires that baselines are evaluated in settings consistent with their original design or that deviations are explicitly justified; without this, the validity of the comparison is uncertain.", "judgment": "Uncertainty regarding the fairness and validity of the experimental comparison due to unexplained baseline adaptation.", "valence": "negative", "suggested_improvement": "Explain how the baselines were adapted to the current setting to justify the comparison.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8427741527557373, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8444769382476807}, {"unit_index": 1, "inspected_object": "Conceptual framing of CompSLOT relative to knowledge distillation", "observation": "CompSLOT uses an external model as a teacher to reduce catastrophic forgetting, resembling standard teacher-student distillation.", "reasoning": "Novelty claims should be positioned against the closest existing paradigm; reframing the method as familiar distillation diminishes the perceived conceptual novelty if not distinguished from prior work.", "judgment": "Deflation of novelty claim because the method appears to be a re-packaging of known distillation techniques without sufficient differentiation.", "valence": "negative", "suggested_improvement": "Explicitly distinguish the approach from the closest existing paradigm (knowledge distillation) to clarify the unique contribution.", "support_status": "memo_inferred", "confidence": "low", "object_key": "problem_framing", "object_sim": 0.9058736562728882, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8631839156150818}, {"unit_index": 2, "inspected_object": "Architectural compatibility of the plug-in with logit-constraining methods like CPrompt", "observation": "The plugin generates a set of logits, which may conflict with methods like CPrompt that constrain logits at different stages during incremental learning.", "reasoning": "A 'plug-and-play' module must have clearly specified interaction semantics with host systems; introducing a second logit mechanism where one already exists creates potential interaction effects that neither method was designed for.", "judgment": "Concern about architectural incompatibility and undefined interaction effects between the proposed plug-in and existing logit-level constraints.", "valence": "negative", "suggested_improvement": "Clarify how the plug-in integrates with methods that have their own output-side mechanisms, explaining the coexistence of logit operations.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.801270604133606, "reasoning_key": "design_justification", "reasoning_sim": 0.7621210813522339}, {"unit_index": 3, "inspected_object": "Presentation clarity and complexity of the method description", "observation": "The method is described as too complex to be reproduced easily, aligning with a low presentation score.", "reasoning": "Complexity in a method must be matched by explanatory clarity to enable reproduction; insufficient exposition of intricate details acts as a barrier to understanding and verification.", "judgment": "Negative evaluation of presentation quality due to excessive complexity hindering reproducibility.", "valence": "negative", "suggested_improvement": "Improve exposition and explanation to make the complex method reproducible.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8379903435707092, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8277159333229065}]}, {"review_id": "03b1CBJCqU", "paper_id": "kWdofGOTdc", "paper_title": "AutoTSAugment: Model-Agnostic Automated Data Augmentation for Unsupervised Contrastive-based Time Series Representation Learning", "decision": null, "summary": "The reviewer evaluates the paper as a useful engineering contribution with strong empirical breadth but critiques its novelty claims as incremental, its efficiency metrics as internally comparative rather than externally benchmarked, and its adaptive search design as inconsistent with its search framing.", "units": [{"unit_index": 0, "inspected_object": "The claimed novel search objective and framework as a conceptual contribution", "observation": "The reviewer identifies the paper's claims of novelty in the search objective and framework, but maps them onto existing AutoML formalizations (Baratchi et al., 2024) and contrastive learning heuristics.", "reasoning": "The reviewer applies a standard that novel contributions should introduce new algorithmic mechanisms or formalizations rather than recombining existing concepts. Since the work is viewed as applying an existing formalization to an existing heuristic, it is judged as incremental engineering rather than intellectual novelty.", "judgment": "The contribution is assessed as having low epistemic status (incremental), despite being acknowledged as a valuable engineering artifact.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.814289927482605, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8453388214111328}, {"unit_index": 1, "inspected_object": "The empirical evaluation scale (three baselines, 164 datasets)", "observation": "The reviewer notes the large scale of the empirical evaluation across classification and forecasting tasks.", "reasoning": "This breadth provides strong evidence of the framework's robustness and utility, serving as the primary counterbalance to concerns about novelty.", "judgment": "This aspect receives unqualified praise as the paper's primary strength.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8683592677116394, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7656711339950562}, {"unit_index": 2, "inspected_object": "The composition of the search space (six augmentation methods)", "observation": "The reviewer observes that the search space is limited to six common transformation methods.", "reasoning": "The reviewer infers that this fixed set bounds the framework's potential ceiling, assuming a truly automated system should discover or incorporate a wider range of augmentations rather than relying on a hand-picked subset.", "judgment": "The scope limits the framework's future utility and perceived automation capability.", "valence": "negative", "suggested_improvement": "Clarify how the framework's modularity mitigates this limitation or specify what an adequate search space would entail.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8576143383979797, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7450631260871887}, {"unit_index": 3, "inspected_object": "The efficiency claim of 63.25% speed-up", "observation": "The reviewer identifies that the speed-up figure compares search strategies within the framework (random vs. Bayesian) rather than against optimized external baselines.", "reasoning": "Efficiency claims must be benchmarked against the strongest relevant alternative (e.g., monolithic baselines with native on-the-fly augmentation). Internal comparisons do not demonstrate that the replacement framework is competitive in total cost against the status quo.", "judgment": "The efficiency claim is overstated and misleading because the comparison class is narrower than implied.", "valence": "negative", "suggested_improvement": "Provide end-to-end runtime comparisons against TS2Vec and other optimized baselines.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8398304581642151, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8433765769004822}, {"unit_index": 4, "inspected_object": "The adaptive search space experiment using RandSampling(n=1)", "observation": "The reviewer notes that using `RandSampling(n=1)` reduces the adaptive component to picking one recommended option, effectively abandoning the search process.", "reasoning": "If the framework is about automated search, the adaptive variant should inform the search process (e.g., via weighted sampling) rather than replace it with a single recommendation. This configuration undermines the internal consistency of the 'search' framing.", "judgment": "The design choice appears to be a convenient simplification that contradicts the core premise of automated search.", "valence": "negative", "suggested_improvement": "Justify the principled reason for this design choice or modify the adaptive component to genuinely enhance the search process.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8390123248100281, "reasoning_key": "design_justification", "reasoning_sim": 0.828909158706665}, {"unit_index": 5, "inspected_object": "Typo in the manuscript text", "observation": "The reviewer identifies a specific typo in the manuscript.", "reasoning": "Close reading indicates attention to detail and presentation quality.", "judgment": "Minor presentation issue flagged.", "valence": "negative", "suggested_improvement": "Correct the identified typo.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7519839406013489, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7893751263618469}]}, {"review_id": "03gHi8UBib", "paper_id": "wOmjeBN6hP", "paper_title": "Parallel-R1: Towards Parallel Thinking via Reinforcement Learning", "decision": "Accept (Poster)", "summary": "The reviewer performs an affirmative evaluation based on perceived novelty and practical utility, while marking boundaries regarding generalizability and computational reporting.", "units": [{"unit_index": 0, "inspected_object": "The core methodological contribution (progressive curriculum combining SFT-then-RL for parallel thinking)", "observation": "The reviewer identifies the work as the first RL framework explicitly designed to train parallel thinking in LLMs, contrasting it with prior SFT-based imitation methods.", "reasoning": "The reviewer accepts the SFT-vs-RL distinction as a meaningful axis of novelty, viewing RL training as more aligned with exploration and generalization than imitation.", "judgment": "The methodology is evaluated as highly novel and well-motivated because it addresses the cold-start problem in RL through a new paradigm.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7968014478683472, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7520908713340759}, {"unit_index": 1, "inspected_object": "Practical applicability and architectural integration", "observation": "The reviewer notes that the framework achieves gains while requiring minimal modifications to the model architecture.", "reasoning": "The reviewer holds an evaluative norm that contributions should be usable by the community, not just theoretically interesting; thus, low implementation barrier is valued.", "judgment": "The approach is judged favorably for its practical accessibility and ease of adoption.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8059936761856079, "reasoning_key": "merit_recognition", "reasoning_sim": 0.819845974445343}, {"unit_index": 2, "inspected_object": "Generalizability of parallel thinking beyond mathematics", "observation": "The reviewer questions whether similar gains could be achieved in other complex reasoning domains like commonsense or scientific reasoning.", "reasoning": "The reviewer expects that a framework claiming to instill a general reasoning skill should demonstrate transferability beyond the specific benchmark domain (math).", "judgment": "The scope of the contribution is currently limited and unverified outside of mathematical tasks, creating an unanswered empirical question.", "valence": "negative", "suggested_improvement": "Comment on the expected transferability of parallel thinking beyond math tasks or provide preliminary evidence.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8066394925117493, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7963413596153259}, {"unit_index": 3, "inspected_object": "Computational cost and resource reporting", "observation": "The reviewer notes limited discussion of efficiency or resource requirements compared to sequential RL or SFT-only approaches.", "reasoning": "Papers proposing expensive training methods should provide resource comparisons so readers can judge whether the gains justify the expenditure.", "judgment": "The paper fails to adequately situate its cost-benefit profile, which is a gap in reporting norms rather than scientific validity.", "valence": "negative", "suggested_improvement": "Provide discussion of efficiency or resource requirements compared to baselines.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8869412541389465, "reasoning_key": "cost_benefit", "reasoning_sim": 0.766036868095398}]}, {"review_id": "03me9o7w3B", "paper_id": "4PZMeopXzP", "paper_title": "PRISM-Physics: Causal DAG-Based Process Evaluation for Physics Reasoning", "decision": "Accept (Poster)", "summary": "The reviewer conducts a formal audit of the scoring policy's permissiveness, arguing that ignoring skipped steps and asymmetric representations undermines its diagnostic value for process-level evaluation, while also requesting verification of experimental details and logical ground-truth alignment.", "units": [{"unit_index": 0, "inspected_object": "Ancestor Closure Scoring Policy's handling of skipped steps and derivation validity", "observation": "The scoring scheme credits students for downstream formulas even if intermediate steps are skipped, and does not verify the correctness of assumptions leading to those formulas.", "reasoning": "A rigorous process-level evaluator should penalize logical gaps and invalid inferences rather than merely rewarding the presence of target formulas; the current scheme is 'too forgiving' because it ignores the derivation path.", "judgment": "The method fails to provide diagnostic rigor for process-level evaluation as claimed.", "valence": "negative", "suggested_improvement": "Verify alignment with true logical evaluation by checking if missing ancestor nodes break logical soundness.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8271814584732056, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7893714308738708}, {"unit_index": 1, "inspected_object": "Asymmetry between reference DAG structure and student answer representation", "observation": "The reference answer is modeled as a DAG with explicit logical dependencies, while the student's answer is extracted as a bag of formulas without logical dependencies.", "reasoning": "This structural asymmetry makes the evaluation lenient because it does not hold the student to the same structural standard as the reference, undermining the potential of the representational framework.", "judgment": "The evaluation is insufficiently rigorous due to under-exploited representational choices.", "valence": "negative", "suggested_improvement": "Extract the student's answer as a DAG and compare it to the reference DAG.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8140454292297363, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7749747037887573}, {"unit_index": 2, "inspected_object": "Alignment of scoring scheme with human judgment and logical ground truth", "observation": "The paper reports alignment with human scores but does not clarify if human experts used the Ancestor Closure Scoring or compared alternative methods based on 'true logic'.", "reasoning": "Human expert scoring may be a proxy that needs validation against logical correctness; agreement between the scheme and humans might be an artifact of shared leniency rather than establishing logical rigor.", "judgment": "The claim of logical correctness is unsupported without comparison to a gold standard of logical correctness distinct from human scores.", "valence": "conditional", "suggested_improvement": "Clarify whether human experts scored using the Ancestor Closure Scoring and compare against methods based on true logic.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8441892862319946, "reasoning_key": "construct_validity", "reasoning_sim": 0.8203967809677124}, {"unit_index": 3, "inspected_object": "Experimental details regarding context length and temperature", "observation": "Uncertainty about whether Appendix E.1 used 8K context and zero temperature for evaluating reasoning LLMs.", "reasoning": "Specificity of these parameters suggests they are critical for reproducibility and avoiding confounds in LLM evaluation results.", "judgment": "Reproducibility concerns require verification of stated experimental configurations.", "valence": "uncertain", "suggested_improvement": "Confirm the use of 8K context and zero temperature in Appendix E.1.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8261094689369202, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8109033107757568}]}, {"review_id": "03oS6GI3je", "paper_id": "HQuboWvFA1", "paper_title": "Monitoring Decomposition Attacks with Lightweight Sequential Monitors", "decision": "Accept (Poster)", "summary": "The reviewer accepts the dataset and problem framing but critically questions the defense's mechanistic robustness, specifically its reliance on prompt engineering, its cost scalability under adversarial decomposition, and its positional stability. They propose concrete sensitivity analyses to validate these concerns.", "units": [{"unit_index": 0, "inspected_object": "Defense mechanism's reliance on prompt engineering", "observation": "The defense's effectiveness appears to stem from the specific wording of carefully engineered prompts rather than inherent algorithmic robustness.", "reasoning": "If success depends on fragile prompt configurations without demonstrated robustness to variation, the finding is an artifact rather than a stable property of the method.", "judgment": "The defense contribution is conceptually fragile and potentially accidental.", "valence": "negative", "suggested_improvement": "Conduct a sensitivity analysis replacing engineered prompts with simpler ones to quantify performance drop.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7947803139686584, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8642237186431885}, {"unit_index": 1, "inspected_object": "Cost profile under adversarial decomposition", "observation": "The sequential monitor processes cumulative context, creating a quadratic cost profile that scales with attacker-induced sub-task decomposition.", "reasoning": "An attacker can deliberately decompose harmful goals into many benign subtasks, forcing the monitor to re-process full history repeatedly, making the defense economically untenable.", "judgment": "The method lacks economic viability under realistic adversarial scaling scenarios.", "valence": "negative", "suggested_improvement": "Measure token consumption as a function of sub-prompt count to quantify the economic attack surface.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8074617981910706, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.6811984777450562}, {"unit_index": 2, "inspected_object": "Positional robustness of harmful step detection", "observation": "The review does not establish whether detection performance varies based on where harmful steps appear within the sequence.", "reasoning": "Without stratified reporting by position, it is unclear if the monitor's success is due to content understanding or positional confounds (e.g., early/late placement bias).", "judgment": "The evaluation lacks sufficient rigor to rule out positional artifacts.", "valence": "conditional", "suggested_improvement": "Report where harmful steps fall within the subtask sequence and stratify performance by that position.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7933458089828491, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7877250909805298}, {"unit_index": 3, "inspected_object": "Dataset construction and empirical visualization", "observation": "The dataset demonstrates scale, diversity, and well-defined splits with detailed quantitative analyses and visualizations.", "reasoning": "The reviewer engaged deeply with specific artifacts (Table 4, Figure 2, Tables 1-3), indicating a high-quality empirical foundation.", "judgment": "The problem framing and dataset contribution are compelling and well-executed.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8471751809120178, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7961388826370239}, {"unit_index": 4, "inspected_object": "Benign injection rate handling", "observation": "The monitor's behavior under varying levels of benign noise has not been characterized.", "reasoning": "Realistic agent workloads contain vastly more benign tasks than harmful ones; performance must degrade gracefully rather than collapse abruptly under high benign ratios.", "judgment": "The method's robustness to realistic noise levels is unverified.", "valence": "uncertain", "suggested_improvement": "Fix the fraction of benign subtasks at 25%, 50%, 75% and report F1, cost, and latency for each using existing tasks.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7455223798751831, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7292086482048035}]}, {"review_id": "042pBmFZXo", "paper_id": "bsokKPMJ5v", "paper_title": "STAR : Semantic-ID Token-Embedding Alignment For Generative Recommenders", "decision": null, "summary": "The reviewer systematically audits the paper's empirical claims, identifying gaps in comparative fairness, baseline completeness, and evidence-claim alignment. The core judgment is that while the conceptual idea is sound, the execution lacks the rigorous controls and evidence needed to substantiate broader claims about efficiency, causality, and linguistic grounding.", "units": [{"unit_index": 0, "inspected_object": "Comparison architecture between STAR and sequential recommenders (e.g., TIGER)", "observation": "Sequential models use smaller transformers and train faster, while the paper claims superiority without accounting for efficiency trade-offs.", "reasoning": "Fair comparison requires controlling for architectural differences and computational cost; comparing performance deltas without addressing efficiency regimes is insufficient to claim superiority.", "judgment": "The comparative evidence is weak because it does not resolve the efficiency-performance trade-off.", "valence": "negative", "suggested_improvement": "Provide stronger theoretical or empirical evidence that accounts for data efficiency and computational cost in comparisons with sequential models.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7825268507003784, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7960359454154968}, {"unit_index": 1, "inspected_object": "Diagnostic analyses regarding embedding initialization", "observation": "The paper's preliminaries and diagnostics omit random embedding initialization as a baseline.", "reasoning": "If the central claim is that mean-of-vocabulary initialization causes collapse, random initialization serves as a necessary control to rule out alternative explanations for the observed collapse; its absence makes the causal story incomplete.", "judgment": "The diagnostic analysis is incomplete and potentially cherry-picked.", "valence": "negative", "suggested_improvement": "Include random initialization as a baseline in the diagnostic analysis to validate the causal claim about mean-of-vocabulary collapse.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8479423522949219, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7637734413146973}, {"unit_index": 2, "inspected_object": "Figure 4 and the claim of linguistic grounding", "observation": "Figure 4 shows distinctions among newly introduced token embeddings but does not demonstrate semantic meaning or alignment with linguistic properties.", "reasoning": "Showing that embeddings are distinct (separated) is necessary but not sufficient for claiming they are linguistically grounded; the evidence must show correspondence to linguistic properties, not just separation.", "judgment": "The evidence provided does not support the claim of 'linguistically meaningful embeddings.'", "valence": "negative", "suggested_improvement": "Provide evidence that aligned embeddings correspond to specific linguistic properties, rather than only showing they are well-separated.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8304838538169861, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7828678488731384}, {"unit_index": 3, "inspected_object": "Experimental setup regarding LLM backbones", "observation": "It is unclear whether all LLM-based recommenders share the same backbone.", "reasoning": "Internal validity requires that observed differences be attributable to the method (initialization) rather than architectural variations; different backbones would confound the results.", "judgment": "The experimental fairness is uncertain due to potential confounding by architecture.", "valence": "uncertain", "suggested_improvement": "Clarify whether all baselines share the same backbone to ensure a fair comparison.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8090369701385498, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8146091103553772}]}, {"review_id": "043IFRhzLl", "paper_id": "VTyL1y4Lab", "paper_title": "RL Is a Hammer and LLMs Are Nails: A Simple Reinforcement Learning Recipe for Strong Prompt Injection", "decision": "Reject", "summary": "The reviewer grants the paper's empirical soundness but critiques its technical novelty, arguing that applying standard RL methods without deep analysis of implications or defense interactions constitutes a weak contribution. The primary positive finding is the attack diversity collapse, which the reviewer views as an interesting but under-explored puzzle.", "units": [{"unit_index": 0, "inspected_object": "The training method (GRPO)", "observation": "The reviewer observes the paper uses 'off-the-shelf GRPO from Huggingface' with only 'a few hyperparameter changes'.", "reasoning": "The reviewer applies a standard of technical novelty that values algorithmic construction over configuration; because the method is standard and changes are minor, the technical contribution is deemed thin.", "judgment": "Low technical contribution due to lack of methodological innovation.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8158136010169983, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8123804330825806}, {"unit_index": 1, "inspected_object": "Empirical scope and dataset diversity", "observation": "The reviewer notes the paper tests on 'quite a few different datasets and models', including open-source and production models.", "reasoning": "While breadth is acknowledged, the reviewer prioritizes depth and mechanistic understanding over empirical coverage, finding the breadth insufficient to compensate for the lack of novelty or deep analysis.", "judgment": "Empirical breadth is noted but does not elevate the contribution score.", "valence": "conditional", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8566732406616211, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8019471764564514}, {"unit_index": 2, "inspected_object": "Attack diversity collapse", "observation": "The reviewer identifies the phenomenon where attacks 'collapse into one specific mode per training run' as a question-worthy finding.", "reasoning": "This observation represents the paper's most substantive intellectual contribution, surfacing an unresolved puzzle in RL-based attack generation, though the reviewer feels the treatment is too abstract.", "judgment": "The diversity collapse is an interesting finding worthy of exploration, but currently under-analyzed.", "valence": "positive", "suggested_improvement": "Provide qualitative analysis of attack evolution and concrete examples of reward hacking types.", "support_status": "mixed", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7235804200172424, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8048006296157837}, {"unit_index": 3, "inspected_object": "Scope of adversarial training implications", "observation": "The reviewer questions whether adversarial training on generated attacks leads to stronger robustness and if models remain easy to break.", "reasoning": "The reviewer expects a paper demonstrating a capability to also explore its implications for the defense landscape; the absence of this analysis marks the paper as incomplete regarding the attack-defense arms race.", "judgment": "The paper is underdeveloped because it fails to explore the defensive implications of its attack capabilities.", "valence": "negative", "suggested_improvement": "Explore whether adversarial training on generated attacks leads to stronger robustness and if models remain vulnerable to learned attackers.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.814109206199646, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7312986850738525}, {"unit_index": 4, "inspected_object": "Generalization to black-box production models", "observation": "The reviewer asks if attacks generalize to black-box production models without direct query access.", "reasoning": "The reviewer applies a standard of practical threat modeling, requiring evidence that the attack works in realistic scenarios where query access is unavailable, which the current results do not fully address.", "judgment": "The generalization to black-box settings is an unverified assumption that limits the paper's practical utility.", "valence": "negative", "suggested_improvement": "Demonstrate if attacks generalize to black-box production models without direct query access.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8400300741195679, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7982779145240784}, {"unit_index": 5, "inspected_object": "Presentation quality", "observation": "The reviewer assigns a low presentation score (2) despite not articulating specific presentation weaknesses in the text.", "reasoning": "There is a disconnect between the substantive prose and the harsh numerical score, suggesting an implicit negative judgment on framing or organization that is not explicitly justified in the review text.", "judgment": "Low presentation score is unjustified by the textual critique, creating an unresolved tension.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "low", "object_key": "clarity", "object_sim": 0.8505752682685852, "reasoning_key": "presentation_trust", "reasoning_sim": 0.769997775554657}]}, {"review_id": "049zMOZTS2", "paper_id": "vGMg9mu4ug", "paper_title": "Shape2Gcode: Direct G-code Generation from 3D Shape Data for Automated Manufacturing", "decision": null, "summary": "The reviewer conducts a domain-grounded reality check, systematically critiquing the paper's simulation-based claims against physical machining constraints. Key judgments highlight that the toolpaths are unsafe due to full-width cutting, the RL optimizes the wrong objective (geometry vs. efficiency), the visibility model ignores tool geometry, and the overall approach lacks practical economic viability regarding setup times and process efficiency.", "units": [{"unit_index": 0, "inspected_object": "Initial contour toolpath strategy and full-width cutting behavior", "observation": "The reviewer identifies that the paper's initial contour strategy appears to perform a 'full width cut', engaging the tool's entire diameter with the material.", "reasoning": "In physical machining, full-width cuts must be run extremely slowly to avoid breaking the tool; most effective roughing strategies control engagement angles. The observed behavior contradicts established best practices for tool safety and efficiency.", "judgment": "The generated toolpaths are likely unsuitable for running on a real CNC mill due to the high risk of tool breakage.", "valence": "negative", "suggested_improvement": "Adopt toolpath strategies that control tool engagement angles rather than using full-width cuts.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.743218183517456, "reasoning_key": "design_justification", "reasoning_sim": 0.7303211092948914}, {"unit_index": 1, "inspected_object": "Absence of feeds, speeds, and cutting forces in the optimization objective", "observation": "The paper's reinforcement learning reward functions do not incorporate feeds, speeds, or cutting forces, focusing instead on geometric metrics like IoU and chamfer distance.", "reasoning": "Optimizing a toolpath in practice requires removing material in the fastest time while minimizing cutting forces and avoiding tool breakage. By optimizing for geometric reconstruction rather than manufacturing efficiency, the RL solves a different problem than claimed in the abstract.", "judgment": "The RL component is optimizing a proxy objective (geometric accuracy) rather than the actual manufacturing objective (efficient, safe material removal), undermining the claim of producing 'optimized' G-code.", "valence": "negative", "suggested_improvement": "Incorporate feeds, speeds, and cutting force constraints into the reward function to align optimization with manufacturing reality.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.835547685623169, "reasoning_key": "design_justification", "reasoning_sim": 0.7675521969795227}, {"unit_index": 2, "inspected_object": "Process inefficiency regarding leads, links, and retracts", "observation": "The approach avoids computing leads and links and retracts the tool before each layer.", "reasoning": "CAM software typically tries to avoid frequent retraction as it is inefficient for production workflows. The current behavior suggests a lack of consideration for economic viability and practical workflow efficiency.", "judgment": "The generated G-code is practically inefficient and does not reflect standard CAM software behaviors aimed at minimizing non-cutting time.", "valence": "negative", "suggested_improvement": "Implement logic to compute leads, links, and optimize retraction paths to reduce non-cutting time.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7754532098770142, "reasoning_key": "design_justification", "reasoning_sim": 0.7894840240478516}, {"unit_index": 3, "inspected_object": "Visibility check algorithm in orientation selection module", "observation": "The paper uses a ray-casting visibility check that does not account for the tool radius or the shape of the cutter.", "reasoning": "Physical reachability in machining depends on the tool's physical footprint (radius and shape), especially in overhangs, narrow channels, and convex corners. A geometric line-of-sight check without these dimensions is insufficient for determining valid tool orientations.", "judgment": "The visibility definition is incorrect because it fails to model the physical constraints of the cutting tool.", "valence": "negative", "suggested_improvement": "Modify the visibility check to account for tool radius and cutter shape to accurately determine reachable surface points.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.787399172782898, "reasoning_key": "design_justification", "reasoning_sim": 0.7244403958320618}, {"unit_index": 4, "inspected_object": "Credit assignment between RL policies and hand-crafted CAM algorithms", "observation": "The majority of good results for IoU and chamfer distance come from the hand-crafted CAM algorithm, while the RL action space is limited.", "reasoning": "If the conventional components generate most of the high-quality output, the marginal contribution of the learned RL policies is minimal. This challenges the paper's central claim regarding the value of the ML approach.", "judgment": "The RL contribution is overstated; the performance gains are largely attributable to traditional algorithmic components rather than the learned policies.", "valence": "negative", "suggested_improvement": "Provide an ablation study isolating the specific performance gain attributable solely to the RL component compared to the baseline CAM algorithm.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8045479655265808, "reasoning_key": "design_justification", "reasoning_sim": 0.8533077836036682}, {"unit_index": 5, "inspected_object": "Handling of multi-setup requirements in orientation selection", "observation": "The paper does not discuss multiple setups, assuming continuous operation despite the need for part re-fixturing in real-world scenarios.", "reasoning": "Real-world manufacturing involves significant operator time and cost for setting up parts, particularly on 3-axis machines where the tool axis is fixed. Ignoring setup changes presents an incomplete view of 'automated machining workflows'.", "judgment": "The proposed solution does not adequately address the economic and operational realities of multi-setup manufacturing workflows.", "valence": "negative", "suggested_improvement": "Discuss or incorporate strategies for handling multiple setups and re-fixturing costs in the orientation selection framework.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7784249186515808, "reasoning_key": "design_justification", "reasoning_sim": 0.7360040545463562}]}, {"review_id": "04Ius8XLtf", "paper_id": "N2lMNqJsBw", "paper_title": "RL Squeezes, SFT Expands: A Comparative Study of Reasoning LLMs", "decision": "Accept (Poster)", "summary": "The reviewer provides qualified endorsement while demanding robustness checks on methodological choices, explicit links between topology and performance, clean separation of correct/incorrect signals, and clearer visualization of practical utility.", "units": [{"unit_index": 0, "inspected_object": "Analytical pipeline design choices (sentence segmentation, embedding model, k=2000 sampling cap, Euclidean distance)", "observation": "The reviewer identifies four specific parameters in the step-level reasoning graph methodology as potential sources of instability.", "reasoning": "Each parameter could plausibly change results; without sensitivity analysis, conclusions might be artifacts of arbitrary choices rather than robust phenomena.", "judgment": "The current evidence is insufficient to rule out parameter sensitivity; the findings require validation against alternative specifications.", "valence": "negative", "suggested_improvement": "Run ablations swapping the embedding model, changing cosine vs Euclidean distance, and adjusting sampling caps to demonstrate robustness.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8431020975112915, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8377600312232971}, {"unit_index": 1, "inspected_object": "Mapping between local graphlet proportions and model performance (Pass@1/Pass@k)", "observation": "Local graphlet proportions can remain similar across models whose accuracies diverge significantly.", "reasoning": "Descriptive topological differences do not automatically translate into predictive or explanatory power regarding practical performance metrics; topology-for-topology's-sake is insufficient.", "judgment": "The connection between the observed topological features and the quantities that matter (accuracy) is not established.", "valence": "negative", "suggested_improvement": "Explicitly demonstrate which topological features track Pass@1/Pass@k or provide a counterfactual showing predictive capability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8500313758850098, "reasoning_key": "design_justification", "reasoning_sim": 0.763785183429718}, {"unit_index": 2, "inspected_object": "Aggregation strategy in graph construction (mixing correct and incorrect responses)", "observation": "Step-level graphs aggregate all sampled responses per problem without separating by correctness.", "reasoning": "If RL compresses incorrect trajectories but the graph mixes correct and incorrect paths, the topological signature may be diluted or misattributed, threatening construct validity.", "judgment": "The measurement of 'functionality' concentration is ambiguous due to contamination of the signal by mixed solution qualities.", "valence": "negative", "suggested_improvement": "Build graphs separately for correct and incorrect responses to isolate the compression/expansion findings.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8134111166000366, "reasoning_key": "design_justification", "reasoning_sim": 0.7690204977989197}, {"unit_index": 3, "inspected_object": "Figure interpretability and redundancy (Figure 4 versions, Figure 7 metrics)", "observation": "The reviewer questions what the four versions of the same plot in Figure 4 add and what each metric in Figure 7 contributes.", "reasoning": "Visual redundancy reduces communication efficiency; figures should communicate content efficiently to be usable by researchers and practitioners.", "judgment": "The presentation of results is information-poor or redundant, hindering clarity.", "valence": "negative", "suggested_improvement": "Clarify the distinct contribution of each figure version/metric or consolidate redundant visualizations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8496735095977783, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7881109714508057}, {"unit_index": 4, "inspected_object": "Practical utility and generalizability of insights", "observation": "The reviewer asks how insights can be leveraged for improvement and whether findings generalize to code/scientific reasoning.", "reasoning": "The paper claims practical implications; without demonstrating leverageability or generalization, the claimed value is under-supported.", "judgment": "The paper falls short of fully convincing on its stated ambitions for practical impact and broad applicability.", "valence": "conditional", "suggested_improvement": "Discuss how insights can be leveraged for data construction/learning and test generalization on other domains like code.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8323137760162354, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7697352766990662}]}, {"review_id": "04eQLtNYWm", "paper_id": "WlQxDKQpo6", "paper_title": "RAG-ENHANCED ASPECT-BASED SENTIMENT ANALYSIS FOR MOBILE APPLICATION REVIEWS: A MULTIAGENT FRAMEWORK FOR DEVELOPER-ORIENTED INSIGHT GENERATION", "decision": "Reject", "summary": "The reviewer systematically critiques the paper's lack of scientific rigor, focusing on missing standard benchmarks, inadequate dataset transparency, unverified code claims, absent ablations, and insufficient baseline comparisons, concluding that the engineering effort does not meet the venue's scientific contribution standards.", "units": [{"unit_index": 0, "inspected_object": "Evaluation methodology and absence of standard ABSA benchmarks (SemEval, MAMS)", "observation": "The paper lacks evaluation on standard ABSA benchmarks, relying instead on a self-curated dataset with reported 98.23% sentiment accuracy and 82% aspect F1.", "reasoning": "Standard benchmarks like SemEval and MAMS are the default reference points for ABSA research; without them, the reported metrics cannot be interpreted meaningfully relative to prior work or positioned within the field's canonical datasets.", "judgment": "Weak external validity due to missing standard validation against established community norms.", "valence": "negative", "suggested_improvement": "Evaluate on SemEval and MAMS to position claims against field standards.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8560627698898315, "reasoning_key": "construct_validity", "reasoning_sim": 0.7370851635932922}, {"unit_index": 1, "inspected_object": "Dataset construction, annotation quality, and release status", "observation": "The dataset is self-curated without disclosed annotation processes, inter-annotator agreement metrics, or public release.", "reasoning": "A self-curated dataset's credibility depends on its annotation process and availability; without agreement metrics, label reliability is unassessable, and without release, reproducibility is impossible.", "judgment": "Lack of trust in ground truth labels and inability to verify reproducibility.", "valence": "negative", "suggested_improvement": "Provide annotation details, inter-annotator agreement metrics, and release the dataset publicly.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8487484455108643, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8119757175445557}, {"unit_index": 2, "inspected_object": "Claim of developer-actionable outputs and code correctness verification", "observation": "The system claims to produce 'production-ready developer feedback' but lacks human study or code correctness verification.", "reasoning": "Claims of actionable code suggestions require demonstration that suggestions actually work (e.g., via compilation checks, unit tests, or human evaluation); absence of such verification creates a gap between claims and evidence.", "judgment": "Fundamental gap between the paper's claims and its supporting evidence regarding code utility.", "valence": "negative", "suggested_improvement": "Verify code suggestions through compilation checks, unit tests, or human developer studies.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8562089800834656, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7671282291412354}, {"unit_index": 3, "inspected_object": "Component-level contribution (LoRA, dual-DB retrieval, multi-agent orchestration)", "observation": "The review notes qualitative descriptions of components but no ablation quantification.", "reasoning": "In a complex system with multiple moving parts, quantitative ablations are necessary to distinguish necessary components from decorative ones; qualitative descriptions are insufficient substitutes.", "judgment": "Inability to determine if each component earns its place in the architecture.", "valence": "negative", "suggested_improvement": "Provide numerical ablation results for LoRA, dual DB, and agents to quantify individual contributions.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8179229497909546, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7381436824798584}, {"unit_index": 4, "inspected_object": "Comparison to strong baselines (GPT-4/Claude prompting)", "observation": "No comparison provided against simple prompting of strong commercial LLMs without multi-agent infrastructure.", "reasoning": "Without comparing to simpler alternatives, it is unclear if the elaborate multi-agent RAG infrastructure provides justified performance gains over cost-effective baselines.", "judgment": "Unproven value proposition of the complex system architecture.", "valence": "negative", "suggested_improvement": "Compare performance against prompting GPT-4/Claude without multi-agents to justify complexity.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8646169900894165, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8153684735298157}, {"unit_index": 5, "inspected_object": "Scientific contribution versus engineering focus", "observation": "The paper presents a heavy system-engineering focus with less scientific contribution.", "reasoning": "The venue expects scientific insights, generalizable findings, or validated methods rather than merely a working pipeline; integration of existing components without demonstrated superiority does not constitute sufficient novelty.", "judgment": "Insufficient scientific contribution for the venue's standards despite practical relevance.", "valence": "negative", "suggested_improvement": "Demonstrate scientific insight or generalizable findings beyond system integration.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.774445652961731, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7901837229728699}, {"unit_index": 6, "inspected_object": "Novelty assessment relative to existing ABSA + RAG + code-repair research", "observation": "The reviewer claims over-claimed novelty, noting that each component (ABSA, RAG, code repair) exists independently.", "reasoning": "Integration alone, without demonstrated superiority or new capability, is viewed as insufficient novelty for the venue.", "judgment": "Over-claimed novelty; integration does not equal significant scientific advance.", "valence": "negative", "suggested_improvement": "Clarify specific novel capabilities or superior performance resulting from the integration.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "novelty", "object_sim": 0.8290941119194031, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7213972806930542}, {"unit_index": 7, "inspected_object": "Potential data leakage from public reviews to LLM training", "observation": "Possible overlap between user reviews used for evaluation and data potentially included in LLM training.", "reasoning": "If the model was trained on public app reviews similar to those in the evaluation set, results may reflect memorization rather than genuine capability, violating machine-learning hygiene norms.", "judgment": "Methodological risk of inflated performance due to potential data contamination.", "valence": "negative", "suggested_improvement": "Assess and report on potential data leakage risks between training and evaluation sets.", "support_status": "memo_inferred", "confidence": "low", "object_key": "stats_metrics", "object_sim": 0.8120679259300232, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7620779871940613}]}, {"review_id": "00lQ6OKyDW", "paper_id": "Iir0Y5yPZm", "paper_title": "Uncertainty Quantification for Regression: A Unified Framework based on Kernel Scores", "decision": "Reject", "summary": "The reviewer primarily acts as an auditor of novelty and experimental validity, identifying substantial weaknesses in the paper's differentiation from prior work (Gruber & Buettner) and demanding more rigorous, interpretable, and consistent experimental designs across robustness, active learning, and OOD tasks.", "units": [{"unit_index": 0, "inspected_object": "Novelty and positioning relative to Gruber & Buettner (ICML 2024)", "observation": "The reviewer identifies a concrete structural correspondence between the paper's framework (aleatoric uncertainty as expectation of kernel entropy, epistemic uncertainty as expected pairwise MMD) and Gruber & Buettner's distributional variance/bias-variance decomposition.", "reasoning": "The reviewer reasons that because the core conceptual components are very similar despite different application domains, the paper risks re-deriving or re-packaging existing results, which diminishes its contribution if not explicitly differentiated.", "judgment": "The contribution is potentially insufficiently novel due to lack of differentiation from closely related work.", "valence": "negative", "suggested_improvement": "Provide a thorough discussion and comparison with Gruber & Buettner to clarify how the entropy/divergence decomposition relates to their bias-variance decomposition.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8968132138252258, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8388795852661133}, {"unit_index": 1, "inspected_object": "Robustness analysis experimental design (Section 6.2)", "observation": "The current experiment corrupts the training target of an ensemble member, conflating training dynamics with test-time robustness.", "reasoning": "The reviewer argues that robustness should be a property of the measure's behavior given a fixed, trained model. The current design introduces a confound where observed effects could stem from how the ensemble absorbs outliers during training rather than the intrinsic properties of the uncertainty measure.", "judgment": "The experimental evidence does not adequately isolate or support claims about the intrinsic robustness of the uncertainty measure at test time.", "valence": "negative", "suggested_improvement": "Redesign the experiment to introduce outlier predictions into a trained ensemble at test time (e.g., Gaussian with large variance) to hold the model fixed and perturb its output.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8774533867835999, "reasoning_key": "design_justification", "reasoning_sim": 0.8485997915267944}, {"unit_index": 2, "inspected_object": "Completeness of robustness analysis across uncertainty types", "observation": "The robustness analysis is limited to aleatoric uncertainty (AU).", "reasoning": "The reviewer posits that epistemic uncertainty (EU) is arguably more sensitive to model disagreement and thus outlier predictions, implying that testing only AU provides an incomplete picture of the framework's robustness.", "judgment": "The evaluation of robustness is incomplete because it fails to examine EU, which is central to the framework's distinction.", "valence": "negative", "suggested_improvement": "Extend the robustness analysis to include epistemic uncertainty (EU).", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8405854105949402, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7669767141342163}, {"unit_index": 3, "inspected_object": "Interpretation of kernel hyperparameter gamma in active learning (Section 6.3)", "observation": "Larger gamma values lead to better active learning performance, but the mechanistic reason for this benefit is not explained.", "reasoning": "The reviewer expects a principled explanation connecting the kernel's mathematical properties (less sensitivity to small distances) to the downstream task's demands, rather than accepting empirical observation alone for a unified framework.", "judgment": "The paper lacks sufficient interpretive depth regarding design parameters, failing to provide concrete design guidelines as implied by its scope.", "valence": "negative", "suggested_improvement": "Provide a mechanistic explanation for why larger gamma is beneficial and analyze whether the relationship is monotonic or saturating.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8333492875099182, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.77543705701828}, {"unit_index": 4, "inspected_object": "Qualitative OOD assessment presentation (Figure 2)", "observation": "Color scales are not standardized across measures, making fair visual comparison difficult.", "reasoning": "The reviewer holds evidentiary standards that require falsifiable, numeric metrics for superiority claims; visual inspection without standardization is deemed insufficient evidence for OOD detection capability.", "judgment": "The qualitative evidence for OOD superiority is methodologically weak and not reproducible.", "valence": "negative", "suggested_improvement": "Standardize color scales or compute a quantitative metric such as the correlation between EU and the land-sea mask.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8233275413513184, "reasoning_key": "construct_validity", "reasoning_sim": 0.8016612529754639}, {"unit_index": 5, "inspected_object": "Computational cost trade-off of pairwise estimator", "observation": "The pairwise estimator has O(M^2) computational cost.", "reasoning": "The reviewer requires justification for design choices that impact practical usability, specifically asking why this cost is acceptable or necessary compared to alternatives.", "judgment": "The paper fails to justify a significant computational trade-off, raising concerns about practical viability.", "valence": "negative", "suggested_improvement": "Provide a brief justification for why the O(M^2) cost of the pairwise estimator is acceptable or necessary.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8293830156326294, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7933560013771057}, {"unit_index": 6, "inspected_object": "Internal consistency regarding gamma parameter setting", "observation": "There is an apparent contradiction between the median heuristic being mentioned as working well in Section 6 introduction and gamma being presented as a key tunable parameter in Section 6.3.", "reasoning": "The reviewer views reproducibility as dependent on clear methodology; inconsistent reporting of hyperparameter settings prevents full assessment and replication of results.", "judgment": "The methodology description contains inconsistencies that hinder reproducibility.", "valence": "negative", "suggested_improvement": "Clarify recommended practice for gamma selection and specify whether the median heuristic was used in Sections 6.1 and 6.2.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8068057894706726, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8485008478164673}]}, {"review_id": "01BQxO3Yf4", "paper_id": "3Z5Ygk2g1U", "paper_title": "Auditable Early Stopping for Agentic Routing: Ledger-Verified Run-Wise Certificates under Local DP", "decision": "Reject", "summary": "The reviewer performs a threshold assessment based on formal rigor and clarity, rejecting the paper because the lack of formal definitions and opaque presentation prevents verification of claims, despite acknowledging the problem domain's importance.", "units": [{"unit_index": 0, "inspected_object": "Problem formalization and definition of the routing task, privacy constraint, stopping criterion, and audit requirement.", "observation": "The reviewer finds that the paper lacks a clear statement of what problem it solves and that most definitions are presented without mathematical proof or justification.", "reasoning": "The abstract's dense terminology (prefix-DAGs, leaf sets, exponential races) is presented without a motivating example or formal problem statement, which the reviewer interprets as a structural failure where claims are asserted rather than justified, making verification impossible.", "judgment": "The absence of formal scaffolding prevents trust in the paper's claims, creating a 'dangerous trend' of unverifiable assertions.", "valence": "negative", "suggested_improvement": "Provide a clear formal definition of the problem and include mathematical proofs or lemmas to justify definitions.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7870325446128845, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7725890874862671}, {"unit_index": 1, "inspected_object": "Presentation and exposition clarity.", "observation": "The reviewer states the paper is very difficult to decipher and that concepts are not introduced in a logical way.", "reasoning": "The reviewer's attention is focused on the abstract and framing, compressing technical machinery into vague phrases like 'verifiable, replayable ledger,' suggesting an inability to map parts to a coherent argument due to opaque internal structure.", "judgment": "The presentation is poor, hindering any meaningful engagement with the method's logic.", "valence": "negative", "suggested_improvement": "Introduce concepts in a logical order and improve overall decipherability.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8811571598052979, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7935136556625366}, {"unit_index": 2, "inspected_object": "'Run-wise certificate' concept.", "observation": "The reviewer finds the statement regarding the 'run-wise certificate' unintelligible and asks for expansion.", "reasoning": "The term appears undefined or used inconsistently, and the reviewer cannot parse its meaning, viewing it as evidence of the paper's broader opacity rather than a genuine technical inquiry.", "judgment": "The concept is currently incomprehensible as presented.", "valence": "negative", "suggested_improvement": "Expand upon and clarify the definition and mechanism of the 'run-wise certificate'.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7657250165939331, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7694403529167175}]}, {"review_id": "01MpjVI753", "paper_id": "dIiV9FYt3i", "paper_title": "Stop Guessing When to Stop Testing: Efficient Model Evaluation with Just Enough Data", "decision": null, "summary": "The reviewer evaluates the paper primarily on practical relevance, arguing that the empirical demonstration relies on easy cases with large effect sizes rather than the harder, more costly scenarios where the method would be most valuable. Secondary critiques address incomplete literature coverage regarding always-valid inference and a missing abstract indicating presentation carelessness.", "units": [{"unit_index": 0, "inspected_object": "Empirical demonstration of efficiency gains on the Open VLM Leaderboard top-50", "observation": "The benchmark contains many heterogeneous model pairs with large effect sizes, making early stopping frequent and easy.", "reasoning": "If the method only yields significant cost reductions when models are easily separable (large effect sizes), it fails to demonstrate utility in the more practically relevant scenario where models are close and effect sizes are small, which is where evaluation costs are most prohibitive.", "judgment": "The central claim about efficiency gains is not convincingly established for the contexts that matter most; the work appears limited to an 'easy case'.", "valence": "negative", "suggested_improvement": "Demonstrate the method's effectiveness on a subset of closer models, such as the top-10, where effect sizes are smaller.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8445941805839539, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8077319860458374}, {"unit_index": 1, "inspected_object": "Related work section", "observation": "Citations for 'always-valid inference' paradigms (e.g., e-values, anytime-valid confidence sequences) are missing.", "reasoning": "These statistical paradigms are closely related to sequential testing, and their absence suggests an incomplete positioning of the paper within the broader relevant literature.", "judgment": "The literature engagement is deficient/incomplete.", "valence": "negative", "suggested_improvement": "Include citations for always-valid inference works to better position the paper.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.817516028881073, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7832695841789246}, {"unit_index": 2, "inspected_object": "Paper presentation (Abstract)", "observation": "The paper lacks an abstract.", "reasoning": "The omission of a standard structural component like an abstract suggests either an oversight or a lack of polish/care in preparing the submission artifact.", "judgment": "The paper appears unpolished or careless in its presentation.", "valence": "negative", "suggested_improvement": "Add an abstract to complete the paper artifact.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8674764037132263, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8052040338516235}]}, {"review_id": "02TJlPUEGt", "paper_id": "sieYp1CpYk", "paper_title": "The Unreasonable Effectiveness of Randomized Representations in Online Continual Graph Learning", "decision": null, "summary": "The reviewer conducts a boundary-testing evaluation, accepting the core empirical performance but demanding evidence for temporal robustness, quantitative efficiency, and mechanistic justification for design choices to determine the limits and true contribution of the method.", "units": [{"unit_index": 0, "inspected_object": "Frozen-embedding design (embeddings generated once and never updated)", "observation": "The reviewer observes that embeddings are static while the graph evolves, noting the absence of experiments assessing this temporal effect.", "reasoning": "As new nodes arrive, the structural context for old nodes changes; without analysis, it is plausible that predictions for earlier nodes degrade, raising concerns about reliability.", "judgment": "The paper has not ruled out a plausible failure mode regarding temporal robustness.", "valence": "negative", "suggested_improvement": "Conduct experiments to assess whether stability comes at a hidden cost to early nodes' predictions as the graph evolves.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8151559233665466, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7701199054718018}, {"unit_index": 1, "inspected_object": "Efficiency claim (lightweight computation, no replay buffer)", "observation": "The reviewer accepts the principle of lightweight computation but notes the claim is unquantified with no runtime, memory, or complexity comparisons provided.", "reasoning": "Qualitative descriptors like 'lightweight' are insufficient empirical support; quantitative evidence is required to verify practical claims against actual baselines.", "judgment": "The efficiency claim is unsubstantiated in its current form.", "valence": "negative", "suggested_improvement": "Provide quantitative comparisons of runtime, memory, or complexity against baselines to defend the lightweight claim.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8898603320121765, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7939510345458984}, {"unit_index": 2, "inspected_object": "Choice of random feature extractors (UGCN and GRNF)", "observation": "The reviewer questions why these two specific encoders were chosen and whether results generalize to other randomized encoders.", "reasoning": "A surprising result (random features performing well) demands mechanistic understanding; without justification or ablation, the insight into *why* it works is underdeveloped, limiting theoretical contribution.", "judgment": "The rationale for the specific encoder choice is unexplained, reducing the perceived incremental value/insight.", "valence": "negative", "suggested_improvement": "Perform ablation studies or provide justification for the selection of UGCN/GRNF to establish generality across randomized encoders.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8134463429450989, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8375592231750488}, {"unit_index": 3, "inspected_object": "Scope of applicability (OCGL benchmarks vs. real dynamic graphs)", "observation": "The reviewer asks if the method applies to real dynamic graphs where node attributes or edges change frequently, outside the stated OCGL scope.", "reasoning": "This question probes the boundary conditions of the claims; it tests whether the authors intend broad generalization beyond the specific benchmark setting.", "judgment": "Uncertainty remains regarding the extent of the method's applicability to broader dynamic graph scenarios.", "valence": "uncertain", "suggested_improvement": "Clarify the scope of claims or discuss applicability to real-world dynamic graphs with frequent attribute/edge changes.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8395195603370667, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8404932618141174}]}, {"review_id": "02fQESGQpg", "paper_id": "3DZeEUTwhq", "paper_title": "Are You Getting What You Pay For? Auditing Model Substitution in LLM APIs", "decision": "Reject", "summary": "The reviewer praises the paper's rigorous execution and deployable TEE solution but raises significant concerns about external validity, specifically regarding hardware scope, real-world validation, and adversarial sophistication.", "units": [{"unit_index": 0, "inspected_object": "Formal definition and experimental methodology", "observation": "The reviewer finds the formal definition strong and the experimental methodology sound.", "reasoning": "The paper provides a comprehensive and systematic evaluation, which establishes rigor.", "judgment": "Positive assessment of execution quality.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.777522087097168, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7843899726867676}, {"unit_index": 1, "inspected_object": "Negative results regarding software-only methods", "observation": "Software-only methods fail due to production nondeterminism defeating log-probability verification.", "reasoning": "This finding changes what practitioners should believe about the space, serving as a crucial practical insight rather than a limitation.", "judgment": "Valuable contribution.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7905120849609375, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7020052075386047}, {"unit_index": 2, "inspected_object": "TEE solution performance overhead", "observation": "The TEE solution shows surprisingly modest performance overhead (2.88% throughput under 64 concurrent requests).", "reasoning": "Modest overhead indicates the solution is deployable in an industry-facing context.", "judgment": "Genuinely deployable solution.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8164480924606323, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7962483763694763}, {"unit_index": 3, "inspected_object": "TEE evaluation scope (hardware/model breadth)", "observation": "Evaluation was limited to one model and one GPU.", "reasoning": "Memory constraints and multi-GPU setups could significantly impact feasibility; credibility scales with demonstrated threat surface breadth.", "judgment": "Severely constrains confidence in general applicability.", "valence": "negative", "suggested_improvement": "Test on larger models and multi-GPU setups.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8258042335510254, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7930102348327637}, {"unit_index": 4, "inspected_object": "Real-world validation of detection methods", "observation": "Experiments used controlled settings with known substitutions, not actual commercial APIs.", "reasoning": "Adversarial substitution attempts in the real world may be more subtle or have unknown ground truth; practical applicability remains uncertain without real-world evidence.", "judgment": "Uncertain practical applicability.", "valence": "negative", "suggested_improvement": "Validate against actual commercial APIs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8305405378341675, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8587924242019653}, {"unit_index": 5, "inspected_object": "Attack model sophistication", "observation": "Paper evaluates simple, static substitution strategies.", "reasoning": "A rational, adaptive adversary with economic incentives would likely use more sophisticated techniques (adaptive substitution, gradual degradation, hybrid evasion).", "judgment": "Insufficient adversarial realism.", "valence": "negative", "suggested_improvement": "Evaluate adaptive substitution, gradual model degradation, or hybrid evasion techniques.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8123783469200134, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.740584671497345}, {"unit_index": 6, "inspected_object": "Binary framing of solutions (software vs. TEE)", "observation": "Paper frames approaches as mutually exclusive (software-only vs. TEE-only).", "reasoning": "Hybrid approaches combining TEE attestation with statistical monitoring might be complementary and potentially more effective.", "judgment": "Potentially artificial binary framing.", "valence": "conditional", "suggested_improvement": "Explore hybrid approaches combining TEE attestation with statistical monitoring.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "problem_framing", "object_sim": 0.8334263563156128, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7756350636482239}]}, {"review_id": "05Zc2ebhyw", "paper_id": "3MqOBXtCxx", "paper_title": "Cost-Optimal Active AI Model Evaluation", "decision": "Reject", "summary": "The reviewer accepts the theoretical novelty of cost-optimal policies but judges the practical implementation as incomplete due to a significant gap between active and oracle performance, attributing this to imperfect uncertainty estimation. The review demands evidence of robustness through burn-in sensitivity analysis and alternative estimator comparisons to validate the practical claims.", "units": [{"unit_index": 0, "inspected_object": "The theoretical framework defining policies π_random and π_active as cost-optimal strategies for minimizing error under fixed budgets.", "observation": "The reviewer identifies the explicit solution for best sampling strategy to minimize error given a fixed monetary or computational budget as the paper's primary novelty, contrasting it with prior work that only improved efficiency for a fixed number of annotations.", "reasoning": "The reviewer applies a standard of theoretical rigor and novelty, accepting the premise that addressing the critical bottleneck in the GenAI lifecycle via cost-optimal policies is a significant contribution, provided the theory is sound.", "judgment": "Qualified endorsement of the theoretical foundation; the theory is sound and addresses a critical problem.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8452957272529602, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7635951042175293}, {"unit_index": 1, "inspected_object": "The practical instantiation of the Active policy compared against the Oracle policy.", "observation": "The practical Active policy underperforms the Oracle policy, which uses perfect uncertainty estimates u(x).", "reasoning": "Using a counterfactual inference, the reviewer assumes that if the active policy's advantage depends on accurate u(x) estimates, and the oracle (perfect u(x)) substantially outperforms the practical method, then the gap is causally attributed to the imperfection of the current uncertainty estimation methods, rather than other factors.", "judgment": "The practical methods are incomplete and not yet delivering on their promise because uncertainty estimation is a primary bottleneck limiting practical gains.", "valence": "negative", "suggested_improvement": "Improve uncertainty estimation methods to close the performance gap between the active policy and the oracle.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.7992510199546814, "reasoning_key": "fair_comparison", "reasoning_sim": 0.798122763633728}, {"unit_index": 2, "inspected_object": "The sensitivity of the practical policy's advantage to the burn-in size n_b.", "observation": "The reviewer notes the use of a 'policy burn-in' of the first 200 samples and questions whether this initial cost is accounted for in the total evaluation budget.", "reasoning": "Applying a norm of total-cost accounting, the reviewer posits that savings must be measured net of setup costs. The reviewer infers a tradeoff where too small a burn-in yields poor parameter estimates and too large a burn-in erases cost savings, creating uncertainty about the existence and width of a viable operational sweet spot.", "judgment": "Uncertainty regarding the practical viability and robustness of the method under realistic budget constraints due to unverified burn-in sensitivity.", "valence": "conditional", "suggested_improvement": "Provide experiments demonstrating the policy's advantage across a range of burn-in sizes to verify the existence of a robust sweet spot.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "theory", "object_sim": 0.7621570229530334, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8214611411094666}, {"unit_index": 3, "inspected_object": "The choice of the heuristic u(x) = G(1-G) for binary tasks.", "observation": "The reviewer identifies that the chosen heuristic assumes the weak rater is well-calibrated.", "reasoning": "The reviewer applies a standard that heuristics should be interrogated and justified. The lack of justification or exploration of alternatives raises concerns that the observed performance gap might be an artifact of this specific heuristic rather than a fundamental limitation of the approach.", "judgment": "Questioning the robustness of the results; the current heuristic choice requires justification or validation against alternatives to confirm the bottleneck is fixable.", "valence": "uncertain", "suggested_improvement": "Experiment with alternative u(x) estimators or justify the assumption that the weak rater is well-calibrated.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "theory", "object_sim": 0.8370287418365479, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8393262624740601}]}, {"review_id": "05hKaPTSd2", "paper_id": "o31oAjIra3", "paper_title": "Segmentation Helps Understanding: Mask-Infused Vision-Language Pre-training for 3D Medical Images", "decision": null, "summary": "The reviewer critically evaluates the paper's technical coherence, identifying fundamental mismatches in loss application, insufficient scalability, and unfair experimental comparisons. Key judgments center on the contradiction between claimed synergies and observed conflicts, as well as the deviation from community evaluation standards.", "units": [{"unit_index": 0, "inspected_object": "Application of Tversky loss in a contrastive learning context (Eq. 3)", "observation": "The paper applies a shared MLP per token to 20x20x10 patches, resulting in identical predictions for every voxel within a 4000-voxel patch.", "reasoning": "Tversky measures spatial overlap while contrastive learning operates on embedding similarity; this constitutes a paradigm mismatch and a category error. The implementation is token-level rather than voxel-level, failing to align with actual methods like nnU-Net or UNETR that use voxel-level supervision.", "judgment": "Fundamental incompatibility between the chosen loss function and the learning objective.", "valence": "negative", "suggested_improvement": "Justify why Tversky loss is mathematically appropriate for contrastive learning between embeddings rather than using InfoNCE.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8367434144020081, "reasoning_key": "design_justification", "reasoning_sim": 0.8244138360023499}, {"unit_index": 1, "inspected_object": "Claim of fine-grained supervision relative to baseline fVLM", "observation": "The method uses only 6 of 197 available segmentation classes, focusing on coarse anatomies (lung, heart, trachea).", "reasoning": "Using 6 coarse categories is barely finer than fVLM's organ-level approach, challenging the novelty of the contribution. Furthermore, Appendix J admits that handling more classes requires dealing with noisy labels but does not attempt it, suggesting the method does not scale beyond clean, coarse annotations.", "judgment": "Insufficient differentiation from baseline and limited scalability.", "valence": "negative", "suggested_improvement": "Show results scaling to more segmentation classes (e.g., 20, 50, 100) with noisy labels.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8683128952980042, "reasoning_key": "design_justification", "reasoning_sim": 0.7792748212814331}, {"unit_index": 2, "inspected_object": "Two-stage training procedure and visual enhancement module design", "observation": "The model trains VLP-only for 10^6 steps before adding segmentation, with αseg initialized at zero and λ=0 initialization for the visual enhancement module.", "reasoning": "This schedule suggests objectives fundamentally conflict rather than unify, contradicting the 'unified framework' claim. The λ=0 initialization implies the model actively avoids segmentation features initially, undermining the narrative of synergy. The architecture and dimension matching for the visual enhancement module are also missing.", "judgment": "Internal inconsistency between claimed unified synergy and implemented conflicting objectives.", "valence": "negative", "suggested_improvement": "Analyze what happens with joint training from the start instead of two-stage training.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7850020527839661, "reasoning_key": "design_justification", "reasoning_sim": 0.8656888604164124}, {"unit_index": 3, "inspected_object": "Comparison fairness with fVLM", "observation": "fVLM uses LLM-parsed organ descriptions while the proposed method uses simple 'This is <mask name>' prompts.", "reasoning": "The methods differ in both the supervision signal and the text encoder input, creating confounded variables that make attribution of performance differences impossible.", "judgment": "Experimental comparison is unfair due to uncontrolled variables.", "valence": "negative", "suggested_improvement": "Ensure baselines differ from the proposed method in only the intended way.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8156322240829468, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7460923790931702}, {"unit_index": 4, "inspected_object": "Zero-shot performance claims", "observation": "Table 3 shows fVLM beats the proposed method on zero-shot evaluation (0.778 vs 0.767 AUC).", "reasoning": "This result directly challenges the authors' claim that voxel-level supervision provides better understanding. The reviewer notes a contradiction between the claimed advantage and the empirical evidence provided by the authors themselves.", "judgment": "Empirical evidence contradicts central thesis regarding understanding capabilities.", "valence": "negative", "suggested_improvement": "Explain why fVLM beats the proposed method on zero-shot despite the claimed fine-grained advantage.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8015518188476562, "reasoning_key": "design_justification", "reasoning_sim": 0.8713457584381104}, {"unit_index": 5, "inspected_object": "Evaluation metric choice (patch-level Dice)", "observation": "The paper uses patch-level Dice instead of voxel-level metrics.", "reasoning": "Patch-level Dice systematically inflates performance compared to voxel-level evaluation. Community standards, such as those used in Medical Segmentation Decathlon, require voxel-level metrics for fair assessment.", "judgment": "Methodological choice inflates results and deviates from established norms.", "valence": "negative", "suggested_improvement": "Adopt voxel-level Dice metrics consistent with community standards like Medical Segmentation Decathlon.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8648087978363037, "reasoning_key": "design_justification", "reasoning_sim": 0.8037806749343872}, {"unit_index": 6, "inspected_object": "Decoder capacity analysis", "observation": "Table 6 only compares 'Light MLP' vs 'Heavy CNN' without intermediate decoder sizes.", "reasoning": "Understanding the functional relationship between decoder capacity and performance is necessary to validate the design, rather than just demonstrating existence at two endpoints.", "judgment": "Insufficient analysis of architectural sensitivity.", "valence": "negative", "suggested_improvement": "Conduct a decoder capacity sweep with intermediate sizes.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8329987525939941, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7632734179496765}, {"unit_index": 7, "inspected_object": "Analysis of voxel-mask learning benefits", "observation": "The paper claims synergy between voxel-mask learning and image-text alignment but lacks theoretical or empirical analysis.", "reasoning": "It is unclear if the objectives compete for capacity or synergize; the reviewer posits an alternative hypothesis that they might compete.", "judgment": "Lack of mechanistic understanding for claimed synergies.", "valence": "negative", "suggested_improvement": "Provide theoretical or empirical analysis of why voxel-mask learning helps image-text alignment.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.806670069694519, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7898322939872742}, {"unit_index": 8, "inspected_object": "Computational cost reporting", "observation": "Computational overhead is completely unaddressed despite training two tasks with different data types.", "reasoning": "Practical concerns about reproducibility and efficiency are relevant for methods combining multiple modalities.", "judgment": "Incomplete practical evaluation.", "valence": "negative", "suggested_improvement": "Report computational overhead (training time, memory) versus CT-CLIP.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8623021245002747, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8034887313842773}, {"unit_index": 9, "inspected_object": "Hyperparameter sensitivity", "observation": "Large standard deviations (±0.014) are reported for hyperparameters α, β per class, αseg, and λ schedule.", "reasoning": "High variance indicates potential instability or cherry-picking of results, raising concerns about robustness.", "judgment": "Concerns about result stability and robustness.", "valence": "negative", "suggested_improvement": "Demonstrate robustness through hyperparameter sensitivity analysis.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.902199387550354, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8197758793830872}]}, {"review_id": "05iISwoJxm", "paper_id": "SaiDRQU7Ez", "paper_title": "StreamSplat: Towards Online Dynamic 3D Reconstruction from Uncalibrated Video Streams", "decision": "Accept (Poster)", "summary": "The reviewer identifies significant documentation gaps including coordinate system ambiguities, under-specified algorithms, and inconsistent terminology, while also requesting additional qualitative and quantitative evidence to substantiate claims about temporal coherence and real-time performance.", "units": [{"unit_index": 0, "inspected_object": "Coordinate system formulation in Line 157", "observation": "The paper adds pixel coordinates (u,v) to unit-space offsets o_i.", "reasoning": "Pixel coordinates and unit-space offsets likely inhabit different dimensional spaces; direct addition is dimensionally inconsistent without an explicit transformation, creating ambiguity about the geometric validity of the operation.", "judgment": "The formulation may be incorrect or unclear if the coordinate system is rectilinear.", "valence": "negative", "suggested_improvement": "Clarify the coordinate transformation between pixel space and unit space.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.819078266620636, "reasoning_key": "design_justification", "reasoning_sim": 0.7448359131813049}, {"unit_index": 1, "inspected_object": "Algorithm 2 UPDATE operation and surrounding notation", "observation": "The description of the aggregation/fusion step is under-specified, with undefined projection parameters pi and informal terms like 'cached'.", "reasoning": "A methods paper should provide self-contained algorithmic specifications; informal implementation details and missing definitions prevent precise understanding of the procedure.", "judgment": "The algorithmic description is incomplete and lacks formal rigor.", "valence": "negative", "suggested_improvement": "Provide formal definitions for all operations and replace informal terms with precise algorithmic descriptions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7615416049957275, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7883297801017761}, {"unit_index": 2, "inspected_object": "Terminology consistency ('rendered frames' vs 'dynamic scenes')", "observation": "The term 'rendered frames' conflicts conceptually with 'dynamic scenes'.", "reasoning": "The terminology implies different ontologies for the output representation, suggesting an unexamined assumption about whether the method outputs static snapshots or dynamic representations.", "judgment": "There is a conceptual tension in the paper's own language that needs reconciliation.", "valence": "negative", "suggested_improvement": "Reconcile the terminology to clarify the nature of the output representation.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "clarity", "object_sim": 0.7702339887619019, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.743718147277832}, {"unit_index": 3, "inspected_object": "Temporal coherence and point correspondence evidence", "observation": "The reviewer suspects per-pixel feed-forward networks may struggle with precise point correspondence.", "reasoning": "Architectural priors suggest limitations in local consistency for this approach; single-point tracking may not adequately demonstrate general temporal coherence capabilities.", "judgment": "The current evidence is insufficient to fully validate the claimed temporal coherence for dense correspondence.", "valence": "conditional", "suggested_improvement": "Include multi-point tracking visualization (~30 points in a local neighborhood) to demonstrate local consistency.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8211055994033813, "reasoning_key": "design_justification", "reasoning_sim": 0.8287175297737122}, {"unit_index": 4, "inspected_object": "Runtime performance breakdown", "observation": "The paper claims real-time/online performance but does not break down runtime components.", "reasoning": "Without a breakdown, it is unclear if the 'online' claim is misleading (e.g., if encoder dominates runtime); feasibility depends on the distribution of computational cost.", "judgment": "The online claim requires more detailed evidentiary support regarding runtime distribution.", "valence": "conditional", "suggested_improvement": "Provide a runtime breakdown to verify the feasibility of the online framing.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8027937412261963, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8348202705383301}]}, {"review_id": "05s584OqKo", "paper_id": "vlpAgjkw39", "paper_title": "DistPFN: Test-Time Posterior Adjustment for Tabular Foundation Models under Label Shift", "decision": null, "summary": "The reviewer evaluates the paper based on three critical dimensions: novelty/scoping, experimental realism, and mechanistic identifiability. They judge the work as competent but insufficiently validated, citing a lack of differentiation from prior art, reliance on synthetic shift simulations, and potential degeneracy in the temperature scaling mechanism.", "units": [{"unit_index": 0, "inspected_object": "The novelty and scope of the DistPFN method as a general technique for logit-producing models", "observation": "The reviewer observes that the processing method is not first proposed in the paper and asserts it is not limited to ICL models, noting the paper lacks corresponding results demonstrating this generality or distinguishing it from prior art.", "reasoning": "The reviewer applies a standard that a paper claiming a generalizable method must either prove that generality empirically (e.g., by applying it to non-ICL models) or clearly delineate what is new relative to existing literature; the absence of either is treated as a deficiency in scoping and contribution.", "judgment": "The work is viewed as lacking sufficient novelty or clear differentiation from existing label-shift correction literature, creating ambiguity about whether the contribution is a new method or a re-packaging.", "valence": "negative", "suggested_improvement": "Demonstrate the method's applicability to non-ICL models (e.g., standard neural networks or gradient-boosted trees with logits) or clearly differentiate the contribution from existing literature.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7912987470626831, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8382732272148132}, {"unit_index": 1, "inspected_object": "The experimental protocol for simulating label shift via sampling", "observation": "The reviewer notes that the paper uses simple sampling to implement label shift across 253 datasets and questions whether this simulates complex real-world scenarios.", "reasoning": "The reviewer holds an expectation that synthetic or procedurally-induced shifts are insufficient evidence for robustness claims, arguing that real-world label shifts are more complex than simple class-conditional resampling; thus, the headline result may be an artifact of the shift-generation procedure rather than true ecological validity.", "judgment": "The experimental validation is considered weak because it relies on a 'toy' setup that does not align with community standards for realistic benchmarking.", "valence": "negative", "suggested_improvement": "Re-run experiments on TabReD or similar benchmarks with naturally-occurring shifts to verify if the method survives realistic shift conditions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8203964233398438, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8117010593414307}, {"unit_index": 2, "inspected_object": "The temperature-scaling mechanism in DistPFN-T", "observation": "The reviewer constructs a hypothetical scenario where training distribution is 7:3 and test-time shifts move in opposite directions (e.g., to 5:5 or 9:1), questioning if the same temperature value can adapt to these different shifts.", "reasoning": "The reviewer probes the identifiability of the temperature mechanism, assuming that a good method should have a monotonic, invertible relationship between its control parameter and the shift it corrects; if two different shifts produce the same discrepancy metric, the method would apply the same adjustment incorrectly.", "judgment": "The temperature mechanism appears underdetermined or potentially degenerate, raising concerns about internal consistency and whether the scalar temperature can uniquely map to the true shift direction and magnitude.", "valence": "negative", "suggested_improvement": "Analyze the temperature's behavior under different shift directions and magnitudes, possibly showing that the discrepancy metric is monotonic in shift severity to confirm identifiability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.797935426235199, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7756551504135132}]}, {"review_id": "060LPZmTuj", "paper_id": "2dJt9YKeg9", "paper_title": "When Evolution Meets Momentum: Orchestrating Goal-oriented and Process-oriented reasoning for LLM Inference Scaling", "decision": "Reject", "summary": "The reviewer acknowledges the conceptual novelty and broad empirical validation but expresses conditional skepticism regarding the method's generalizability and efficiency. The evaluation hinges on whether gains stem from the evolutionary mechanism itself versus manual initialization or increased compute expenditure.", "units": [{"unit_index": 0, "inspected_object": "The strategy pool's provenance and domain specificity", "observation": "The strategy pool is described as domain-specific and manually designed rather than automatically discovered.", "reasoning": "If the initial pool performs the heavy lifting of providing useful search directions, the method functions more as careful initialization with a search wrapper than as an evolutionary contribution; generalizability to non-coding domains remains unverified.", "judgment": "The core intellectual contribution regarding evolution is weakened by reliance on hand-crafted domain knowledge.", "valence": "negative", "suggested_improvement": "Demonstrate that the strategy pool can be automatically generated or adapted to non-coding domains.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8222929239273071, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7703742980957031}, {"unit_index": 1, "inspected_object": "Computational overhead and normalization of gains", "observation": "The paper lacks analysis of the engineering overhead from repeated similarity computations and does not report gains under strictly token- or time-normalized budgets.", "reasoning": "Without cost-benefit accounting per baseline, it is unclear if Pass@K improvements result from better compute allocation or simply spending more compute (via longer prompts and extra checks); equal-sample comparisons are insufficient for inference-scaling claims.", "judgment": "The practical value and efficiency of the method are unverified due to missing overhead analysis.", "valence": "negative", "suggested_improvement": "Quantify actual runtime complexity and verify if gains hold under strict token- or time-normalized budgets relative to baselines.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8788378238677979, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8367488384246826}, {"unit_index": 2, "inspected_object": "Integration complexity vs. plug-and-play claim", "observation": "The reviewer notes a tension between the claim of seamless integration without structural modifications and the question regarding significant engineering overhead.", "reasoning": "Conceptual cleanliness does not guarantee lightweight practical deployment; if integration requires substantial code changes or prompt construction effort, the 'plug-and-play' utility is diminished.", "judgment": "The practical viability is questioned because the ease of integration is not quantified against the added complexity.", "valence": "conditional", "suggested_improvement": "Provide evidence that integration is lightweight and quantify any engineering effort required beyond conceptual integration.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8147022128105164, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7115222811698914}]}, {"review_id": "061e3hpGV7", "paper_id": "E7jZqo0A50", "paper_title": "MARTI: A Framework for Multi-Agent LLM Systems Reinforced Training and Inference", "decision": "Accept (Poster)", "summary": "The reviewer challenges the paper on four independent grounds: definitional scope (not truly multi-agent), generalizability (Qwen-only models), efficiency (asynchronous rollout underperformance), and causal validity (missing reward ablations). A single diagnostic question probes mechanistic understanding of training dynamics.", "units": [{"unit_index": 0, "inspected_object": "The classification of the paper's setting as 'multi-agent LLM systems'", "observation": "Math word problems are not natively multi-agent LLM settings; they lack distinct roles, opposing objectives, or emergent strategic interaction found in zero-sum games, social deduction, or human-AI coordination.", "reasoning": "The reviewer holds a disciplinary norm that 'multi-agent' implies game-theoretic structure or heterogeneous agent types. Because the paper uses homogeneous models solving static problems, it fails to meet this standard, reducing its contribution to narrow empirical claims rather than field-defining insights.", "judgment": "The paper's framing is misaligned with the field's definition of multi-agent systems, resulting in a minimal contribution assessment.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8150864243507385, "reasoning_key": "design_justification", "reasoning_sim": 0.7950161099433899}, {"unit_index": 1, "inspected_object": "Generalizability of framework claims across model families", "observation": "Only Qwen-based models were used for experiments.", "reasoning": "The reviewer asserts that Qwen models behave differently from other base models (like Llama) regarding RL performance. Without cross-family variation, it is impossible to determine if the findings are generalizable to the broader space of LLM families.", "judgment": "The evidence is insufficient to support a general claim about the framework's efficacy.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8225347995758057, "reasoning_key": "design_justification", "reasoning_sim": 0.8148555755615234}, {"unit_index": 2, "inspected_object": "Efficiency and performance trade-offs of asynchronous rollouts", "observation": "Asynchronous generation underperforms synchronous behavior at concurrency 32 and shows only unimpressive improvement at higher concurrency.", "reasoning": "A systems contribution must demonstrate a clear Pareto improvement (better or equal performance at lower cost). The observed underperformance at moderate concurrency compromises the claimed efficiency benefit.", "judgment": "The efficiency mechanism is not clearly beneficial and may be compromised by performance costs.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8461663126945496, "reasoning_key": "design_justification", "reasoning_sim": 0.8204008340835571}, {"unit_index": 3, "inspected_object": "Causal attribution of gains to multi-agent architecture vs. reward shaping", "observation": "No ablation over reward shaping terms or novel components of MARTI was provided.", "reasoning": "Without isolating the effect of reward design, it is unclear whether the framework's gains are driven by the multi-agent architecture or simply by reward engineering choices. This leaves causal claims unverified.", "judgment": "The paper has plausible but unverified causal claims, leading to a low soundness assessment.", "valence": "negative", "suggested_improvement": "Perform ablations over reward shaping terms to isolate the contribution of the multi-agent architecture.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.825819730758667, "reasoning_key": "design_justification", "reasoning_sim": 0.8299668431282043}, {"unit_index": 4, "inspected_object": "Response length dynamics during training (Figure 5)", "observation": "Response length significantly increases for 2 agents and decreases for 1 agent over the course of training.", "reasoning": "This behavioral anomaly suggests potential issues such as learning to argue at length without accuracy gains, inadvertently rewarding verbosity, or side effects of optimization. The reviewer demands mechanistic understanding to verify the authors comprehend their system's dynamics.", "judgment": "The lack of explanation for this anomaly raises questions about the depth of understanding and robustness of the framework.", "valence": "conditional", "suggested_improvement": "Explain the mechanistic reasons for the divergent response length trends between 1-agent and 2-agent conditions.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8604891300201416, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8399879932403564}]}, {"review_id": "06PL7zHw0n", "paper_id": "c1fr8fmiaW", "paper_title": "Black-box Attack Robustness with Model Diversity and Randomization", "decision": "Reject", "summary": "The reviewer conducts a novelty audit against PuriDefense, abstracting both methods to a shared formal skeleton to argue that Disco's distinctions are implementation details. The review prioritizes conceptual differentiation over empirical performance or theoretical rigor, concluding the work is incremental.", "units": [{"unit_index": 0, "inspected_object": "The defense mechanism's conceptual structure and its relationship to PuriDefense (Guo et al. 2024)", "observation": "Both Disco and PuriDefense are abstracted into a shared formal skeleton: maintaining K diverse functions, randomly sampling N for each query, and using their combined output. The reviewer identifies the distinction between input-space purification (Disco) and model-space selection (PuriDefense) as an implementation detail rather than a conceptual difference.", "reasoning": "From an attacker's perspective, both methods present a randomly varying function sampled from a set. This abstraction erases the semantic difference between purifying inputs versus selecting models, leading to the inference that Disco does not constitute a conceptually novel defense paradigm distinct from PuriDefense.", "judgment": "The distinction between input-space and model-space randomization is superficial; Disco is not a new paradigm but a variant of PuriDefense.", "valence": "negative", "suggested_improvement": "Rebut the re-description by demonstrating why the implementation detail constitutes a fundamental conceptual distinction, or concede the overlap.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7220127582550049, "reasoning_key": "design_justification", "reasoning_sim": 0.8072600960731506}, {"unit_index": 1, "inspected_object": "Theoretical propositions (Propositions 3.1 and 3.2) regarding query complexity and attack misleadingness", "observation": "The reviewer notes that the theoretical results rely on standard inequalities (Hoeffding, Markov/Jensen) and argues they follow directly from the randomization principle established in PuriDefense.", "reasoning": "If the theoretical principles (diversity increases query complexity, diversity misleads attacks) were already shown empirically or theoretically in PuriDefense, then Disco's formalization adds no new insight, even if it is rigorous.", "judgment": "The theoretical contributions are not generative of new insight because they are derivable from prior work.", "valence": "negative", "suggested_improvement": "Clarify which specific theoretical insights are novel relative to PuriDefense, or provide evidence that PuriDefense did not establish these principles.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.801303505897522, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7491624355316162}, {"unit_index": 2, "inspected_object": "Training method and diversity promotion techniques (SVGD+ sample loss, DivDis, DivReg, ADP)", "observation": "The reviewer identifies the engineering contributions as either reasonable improvements or techniques already explored in prior literature.", "reasoning": "Engineering improvements and the application of known diversity techniques do not count as major conceptual advances if the underlying paradigm is considered derivative.", "judgment": "The training and diversity methods are evolutionary rather than revolutionary.", "valence": "negative", "suggested_improvement": "Demonstrate how the specific combination of these techniques yields qualitatively different properties than those achieved by prior works.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8247025012969971, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7401604056358337}, {"unit_index": 3, "inspected_object": "Novelty threshold and contribution criteria", "observation": "The reviewer applies a strict norm where conceptual novelty relative to the closest prior work (PuriDefense) is a necessary condition for contribution, overriding empirical robustness and formal analysis.", "reasoning": "The reviewer operates under the counterfactual: 'If PuriDefense exists, what does this paper add?' Since the answer is perceived as only engineering details, the paper fails the novelty gate regardless of other merits.", "judgment": "The work is technically competent but does not advance the field due to insufficient conceptual novelty.", "valence": "negative", "suggested_improvement": "Provide results that are impossible to derive from PuriDefense's framework and offer actionable insights not available from it.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.907467782497406, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7968661189079285}]}, {"review_id": "06WfSexpym", "paper_id": "D4xSuGvLZA", "paper_title": "BriLLM: Brain-inspired Large Language Model", "decision": "Reject", "summary": "The reviewer systematically identifies multiple areas where the paper's bold claims exceed its evidentiary support, focusing on the lack of biological validation, missing quantitative benchmarks, and undefined scalability derivations, resulting in a judgment that the evidence is fundamentally insufficient despite the idea's potential merit.", "units": [{"unit_index": 0, "inspected_object": "SiFu framework's formal definitions (Definition 1: SiFu graph, Definition 2: signal tensor)", "observation": "The definitions are internally consistent but lack validation against empirical data such as EEG or cortical activation patterns.", "reasoning": "The reviewer applies a fidelity standard requiring that 'brain-inspired' frameworks demonstrate correspondence to actual brain measurements rather than merely invoking biological terminology; without this anchor, the formal objects remain ungrounded in reality.", "judgment": "The conceptual ambition is acknowledged but deemed insufficient due to the absence of empirical grounding.", "valence": "negative", "suggested_improvement": "Validate the SiFu graph and signal tensor against EEG data or cortical activation patterns.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7439365983009338, "reasoning_key": "construct_validity", "reasoning_sim": 0.7935944199562073}, {"unit_index": 1, "inspected_object": "Competitive activation formula", "observation": "The formula ignores known biological phenomena: noise and synaptic delays.", "reasoning": "If the model claims to replicate brain information processing, its core computational primitive should incorporate known biological constraints; the absence of these terms suggests the model is 'brain-inspired in name only.'", "judgment": "The omission of noise and delay is treated as a design flaw indicating a lack of biological realism.", "valence": "negative", "suggested_improvement": "Quantify the impact of noise and synaptic delays on the competitive activation mechanism.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7573692202568054, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7382726073265076}, {"unit_index": 2, "inspected_object": "Experimental claims (performance, multi-modal compatibility, interpretability, scalability)", "observation": "The paper claims GPT-1-level performance and scalability to 100–200B parameters but provides no supporting metrics (perplexity, BLEU) or demonstrations on real datasets.", "reasoning": "Claims float free of empirical anchors without quantitative baselines; additionally, modern LLM papers are expected to benchmark against contemporary state-of-the-art models, not just 2018-era references.", "judgment": "The experimental evidence is fundamentally insufficient to support the stated performance and scalability claims.", "valence": "negative", "suggested_improvement": "Provide perplexity/BLEU metrics and benchmark against modern LLMs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8775178790092468, "reasoning_key": "fair_comparison", "reasoning_sim": 0.6897206902503967}, {"unit_index": 3, "inspected_object": "Scalability argument (100–200B parameter estimates)", "observation": "The derivation of the 100–200B parameter estimate is not explained, nor is the infrastructure required to support it described.", "reasoning": "Scaling claims should be backed by either empirical scaling curves or transparent extrapolation methodology; treating the number as a hypothesis rather than a finding requires justification.", "judgment": "The scalability claim lacks derivation transparency and remains an unsupported assertion.", "valence": "negative", "suggested_improvement": "Provide the scaling law or extrapolation logic used to derive the 100–200B parameter estimate.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8906016945838928, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7933501601219177}, {"unit_index": 4, "inspected_object": "Conceptual alignment with neuroscience principles", "observation": "The paper attempts a biologically-grounded paradigm shift but is derivative of SNNs and neurocognitive models.", "reasoning": "Conceptual alignment is necessary but not sufficient for contribution; without validation, the effort remains merely conceptual and does not constitute a novel achievement over existing models.", "judgment": "The contribution is partial and qualified by the lack of validation and derivative nature.", "valence": "conditional", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8182647824287415, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.797683835029602}, {"unit_index": 5, "inspected_object": "Unaddressed trade-offs and variability", "observation": "The paper fails to address computational cost, energy efficiency, and cultural bias in semantic mapping.", "reasoning": "Deployment implications require consideration of practical costs and societal biases; ignoring these factors leaves significant gaps in the evaluation of the model's utility and safety.", "judgment": "The oversight of trade-offs and variability represents a significant weakness in the paper's completeness.", "valence": "negative", "suggested_improvement": "Include analysis of computational cost, energy efficiency, and cultural bias in semantic mapping.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8486495614051819, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7885419130325317}]}, {"review_id": "06WqHZqIVH", "paper_id": "b9GH5mMBfG", "paper_title": "RoboPARA: Dual-Arm Robot Planning with Parallel Allocation and Recomposition Across Tasks", "decision": "Accept (Poster)", "summary": "The reviewer performs a feasibility assessment focused on deployment viability, identifying critical gaps in generalization, optimization realism, and safety while questioning the completeness of comparisons with multimodal approaches.", "units": [{"unit_index": 0, "inspected_object": "RoboPARA's reliance on predefined skill libraries and scenario templates", "observation": "The system requires structured inputs via predefined libraries/templates.", "reasoning": "General-purpose planning systems are expected to handle novel task specifications without substantial re-engineering; dependence on fixed structures limits scalability for long-horizon, multi-stage, or out-of-distribution tasks.", "judgment": "Limited generalization capability and scalability for complex novel tasks.", "valence": "negative", "suggested_improvement": "Enable abstraction mechanisms to allow the framework to adapt to new environments with minimal manual reformulation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8312162756919861, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7506130337715149}, {"unit_index": 1, "inspected_object": "Planning stage optimization objective minimizing estimated action duration", "observation": "Optimization depends on accurate duration estimates which are hard to obtain in real-world settings.", "reasoning": "If execution latency estimates are unreliable, the optimized schedule may not be time-optimal on physical hardware, creating a sim-to-real gap in the optimization logic.", "judgment": "Practical validity of the scheduling optimization is compromised by unrealistic assumptions about duration estimation.", "valence": "negative", "suggested_improvement": "Incorporate uncertainty into scheduling optimization or learn duration models from real-world data.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8063471913337708, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7633088231086731}, {"unit_index": 2, "inspected_object": "Absence of collision-avoidance discussion in dual-arm parallel execution", "observation": "The paper does not address safety risks inherent in concurrent physical action, specifically noting heuristics like preferential left-arm assignment.", "reasoning": "Any system proposing concurrent physical action must address safety as a first-class concern; ignoring collision risks makes the approach unsafe for real deployment.", "judgment": "Critical omission of safety mechanisms renders the current design potentially unsafe for physical deployment.", "valence": "negative", "suggested_improvement": "Include collision-avoidance mechanisms and discuss safety protocols for dual-arm parallel execution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7859557271003723, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.6786993145942688}, {"unit_index": 3, "inspected_object": "Real-robot experiment implementation details (Graph construction, pre-abstraction, action timing)", "observation": "Unclear how Graph is constructed during real experiments, whether scene objects require pre-abstraction, and how action execution times are defined.", "reasoning": "Understanding these details is necessary to assess whether duration estimates are grounded in actual measurement (addressing Weakness 2) and the manual effort required for instantiation (addressing Weakness 1).", "judgment": "Uncertainty regarding the practical feasibility and manual overhead of deploying the framework.", "valence": "conditional", "suggested_improvement": "Clarify Graph construction, pre-abstraction requirements, and action timing definitions in real-robot experiments.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7923876643180847, "reasoning_key": "construct_validity", "reasoning_sim": 0.7397869229316711}, {"unit_index": 4, "inspected_object": "Comparison against SOTA VLM models for dual-arm parallel tasks", "observation": "The review does not mention comparison with Vision-Language Models (VLMs).", "reasoning": "VLMs can perceive scenes directly and might offer better visual grounding and potential safety via collision avoidance compared to text-only LLMs; excluding them suggests an incomplete comparative set.", "judgment": "Missed opportunity for improvement and incomplete state-of-the-art comparison.", "valence": "negative", "suggested_improvement": "Attempt to use or compare against SOTA VLM models for dual-arm parallel tasks to evaluate potential benefits in visual grounding and safety.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8032225966453552, "reasoning_key": "design_justification", "reasoning_sim": 0.7136124968528748}]}, {"review_id": "06XCHdrdBT", "paper_id": "6jzadKYz3W", "paper_title": "Breaking Independence: Learning Correlated Views for Variational Incomplete Multi-View Clustering", "decision": "Reject", "summary": "The reviewer evaluates the paper as a theoretically motivated advance in variational IMVC but demands stronger proof of identifiability, stability, and scalability, while noting gaps in motivation and reporting transparency.", "units": [{"unit_index": 0, "inspected_object": "Conceptual framing and problem identification in variational IMVC", "observation": "The paper addresses the conditional independence assumption among view posteriors by modeling cross-view correlation of estimation errors.", "reasoning": "This approach is evaluated against a norm of principled generalization, representing a meaningful conceptual advance over prior work (DVIMC, CoDE) rather than just being empirically useful.", "judgment": "Positive: The contribution is theoretically motivated and represents a natural, satisfying next step in the lineage of adaptive correlation methods.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8563624620437622, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7621506452560425}, {"unit_index": 1, "inspected_object": "Motivation for estimating inter-view dependence via estimation errors", "observation": "The paper relies on citations (Winkler, Mancisidor et al.) to justify the abstraction of estimation errors rather than providing a deeper conceptual argument.", "reasoning": "The reviewer applies a norm that requires earning an abstraction through argument rather than citation, viewing the current justification as somewhat indirect and lacking higher-level motivation.", "judgment": "Negative: The conceptual case for the abstraction is not compellingly made, limiting the perceived depth of the foundation.", "valence": "negative", "suggested_improvement": "Provide a higher-level justification for why 'estimation errors' are the right abstraction for inter-view dependence, moving beyond simple citation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8189172148704529, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8095060586929321}, {"unit_index": 2, "inspected_object": "Generalizability of the proposed principle", "observation": "The method is presented specifically within the context of Variational IMVC without explicit discussion of broader applicability.", "reasoning": "The reviewer values contributions that speak to a wider community (multi-modal or self-supervised learning) and views narrow niche solutions as less valuable.", "judgment": "Negative: The contribution appears narrowly scoped, potentially limiting its impact on broader communities.", "valence": "negative", "suggested_improvement": "Discuss whether the principle generalizes beyond IMVC to broader multi-modal or self-supervised learning contexts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8239098787307739, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7548291683197021}, {"unit_index": 3, "inspected_object": "Mathematical derivation and parameterization of R", "observation": "The paper uses a decomposition Σ = DRD with normalized Cholesky parameterization for R, but lacks clear intuition for how learning R avoids degeneracy under high incompleteness or uncorrelated views.", "reasoning": "The reviewer expects clearer exposition of the robustness of learning dynamics under extreme conditions, not just mathematical correctness.", "judgment": "Negative: The intuition for stability under specific edge cases is insufficient, creating uncertainty about the method's behavior.", "valence": "negative", "suggested_improvement": "Provide clearer intuition about how learning R avoids degeneracy under high incompleteness or uncorrelated views.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8384896516799927, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8246833682060242}, {"unit_index": 4, "inspected_object": "Theoretical guarantees for R (Frobenius norm bound)", "observation": "The paper provides only a Frobenius norm bound for R.", "reasoning": "The reviewer applies a standard that a parameterization should be provably well-behaved; a purely structural bound does not guarantee stability or identifiability.", "judgment": "Negative: The theoretical guarantee is insufficient because it fails to ensure stability or identifiability.", "valence": "negative", "suggested_improvement": "Strengthen theoretical guarantees to demonstrate stability and identifiability, rather than relying solely on a structural norm bound.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8398137092590332, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7698386311531067}, {"unit_index": 5, "inspected_object": "Identifiability of the parameterization (L and D scaling)", "observation": "It is unclear if different L or scaling of D could lead to equivalent Σ = DRD, potentially yielding degenerate solutions.", "reasoning": "If the model is overparameterized such that multiple parameters yield the same output, the interpretability of learned correlations is undermined, which is a critical theoretical flaw.", "judgment": "Uncertain/Negative: There is a significant risk of non-identifiability that challenges the validity and interpretability of the learned correlations.", "valence": "negative", "suggested_improvement": "Clarify how parameter identifiability is ensured to prevent degenerate solutions from different parameterizations of L and D.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8713502883911133, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7819828391075134}, {"unit_index": 6, "inspected_object": "Reporting standards in experimental validation", "observation": "The practice of averaging results over five runs is not stated in the main text.", "reasoning": "Transparency in empirical reporting is required to verify the precision and reliability of the claims.", "judgment": "Negative: The lack of transparency in the main text undermines confidence in the reported empirical precision.", "valence": "negative", "suggested_improvement": "Move the statement about averaging over five runs into the main text.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8417112231254578, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7846943140029907}, {"unit_index": 7, "inspected_object": "Scalability and complexity analysis", "observation": "The method has O(NDV³) complexity, scaling cubically with views, but lacks empirical runtime/memory analysis across varying V.", "reasoning": "Theoretical complexity analysis is necessary but not sufficient; empirical evidence is needed to confirm practical applicability on larger scales.", "judgment": "Negative: The gap between theoretical complexity and demonstrated performance on small datasets raises concerns about practical utility at scale.", "valence": "negative", "suggested_improvement": "Provide an empirical analysis of how runtime and memory change with the number of views (V).", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8685810565948486, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8159618973731995}, {"unit_index": 8, "inspected_object": "Numerical stability during optimization", "observation": "Per-sample matrix inversion of R could be unstable when R is ill-conditioned.", "reasoning": "Safety in deployment requires ensuring that numerical operations remain stable even when intermediate matrices are ill-conditioned.", "judgment": "Negative: Potential numerical instability poses a risk to the safe deployment of the method.", "valence": "negative", "suggested_improvement": "Analyze or address the potential numerical instability arising from per-sample matrix inversion of ill-conditioned R.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8384314179420471, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7594733238220215}, {"unit_index": 9, "inspected_object": "Dataset scale and validation breadth", "observation": "All benchmarks used are standard, small to medium-scale datasets with no high-dimensional or large-scale validation.", "reasoning": "Validation on small datasets does not sufficiently demonstrate robustness or applicability in real-world, high-dimensional scenarios.", "judgment": "Negative: The scope of empirical validation is limited, leaving generalization to large-scale settings unverified.", "valence": "negative", "suggested_improvement": "Include high-dimensional or large-scale validation to demonstrate practical applicability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8583048582077026, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8310340642929077}]}, {"review_id": "06pMTRD0o6", "paper_id": "Bfuj1qeoOU", "paper_title": "Detecting Multilevel Manipulation from Limit Order Book via Cascaded Contrastive Representation Learning", "decision": null, "summary": "The reviewer conducts a normative audit focusing on novelty, completeness of baselines, interpretability, and practical deployability, finding the method incremental and lacking necessary comparative and analytical components.", "units": [{"unit_index": 0, "inspected_object": "Proposed cascaded architecture and contrastive loss novelty", "observation": "The reviewer decomposes the 'cascaded' design into a concatenation of LOB embeddings and manual features, and identifies the contrastive component as a direct adoption of Khosla et al. (2020) without financial or LOB-specific modification.", "reasoning": "A novel framework should introduce a new architectural principle, adapt known components in a non-trivial way to the domain, or provide theoretical justification for the combination; the reviewer finds none of these, concluding the contribution is not sufficiently differentiated from prior work.", "judgment": "The claim of a 'novel LOB-based representation learning framework' is unsupported due to incremental nature relative to known architectures.", "valence": "negative", "suggested_improvement": "Provide a theoretical justification for why the combination is more than the sum of its parts or demonstrate non-trivial adaptation to the LOB domain.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8462579250335693, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8052287697792053}, {"unit_index": 1, "inspected_object": "Interpretability of learned representations", "observation": "The paper does not provide analysis on what features or inter-level interactions matter most for detection.", "reasoning": "Given the paper's claim to offer 'broader insights' into multilevel manipulation, there is an expectation that it should provide evidence about which levels or interactions drive detection, rather than just reporting performance improvements.", "judgment": "The lack of interpretability analysis is a mild criticism that limits the paper's ability to fulfill its stated goal of providing broader insights.", "valence": "conditional", "suggested_improvement": "Perform post-hoc explanation analysis to identify which features or inter-level interactions are most important.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.826846718788147, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7729381918907166}, {"unit_index": 2, "inspected_object": "Baseline set for comparison", "observation": "The paper omits comparisons against specific deep anomaly detection baselines: DeepSAD, Deep Isolation Forest, and FeaWAD.", "reasoning": "A paper proposing a new deep representation learning approach for anomaly detection should benchmark against the current state-of-the-art in deep anomaly detection to assess whether gains are specific to the LOB representation task or generalizable advantages over generic methods.", "judgment": "The baseline set is incomplete, preventing a full assessment of the proposed framework's advantages over contemporary deep anomaly detection methods.", "valence": "negative", "suggested_improvement": "Include comparisons with DeepSAD, Deep Isolation Forest, and FeaWAD.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8722004294395447, "reasoning_key": "design_justification", "reasoning_sim": 0.7414825558662415}, {"unit_index": 3, "inspected_object": "Latency and throughput profile", "observation": "The paper does not report runtime or throughput analysis.", "reasoning": "The application domain of LOB data implies high-frequency trading and near-real-time decision-making, where runtime is critical for practical deployability; a method that is accurate but too slow has limited practical value.", "judgment": "The absence of latency analysis is a pragmatic concern regarding the practical deployability of the method in its target domain.", "valence": "negative", "suggested_improvement": "Conduct runtime or throughput analysis to evaluate deployment feasibility.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.7942507266998291, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7952274680137634}]}, {"review_id": "072bdkURS2", "paper_id": "nnRB90w2kv", "paper_title": "Flow Marching for a Generative PDE Foundation Model", "decision": null, "summary": "The reviewer conducts a forensic audit of the paper's empirical evidence, arguing that Table 1 mixes incompatible tasks and setups, rendering results uninterpretable. They further critique preprocessing choices, missing baselines, and insufficient ablations, concluding that the demonstration does not support the method's claims despite conceptual merit.", "units": [{"unit_index": 0, "inspected_object": "Table 1 baseline setups and task definitions", "observation": "The table mixes three distinct experimental setups: Setup 1 (DPOT/Unet, context length 10, multi-step, downsampled), Setup 2 (Lower-table Unet/FNO/CNextUnet, context length 1, single-step, native resolution), and Setup 3 (P2VAE, reconstruction on single frame).", "reasoning": "Comparisons require constant task definition, input representation, and evaluation protocol. Mixing reconstruction with prediction, multi-step rollout with single-step, and downsampled inputs with native resolution constitutes a category error that prevents fair interpretation of relative performance.", "judgment": "The results in Table 1 cannot be fairly interpreted and may mislead readers about the true nature of the model's performance due to setup heterogeneity.", "valence": "negative", "suggested_improvement": "Run baselines yourself under controlled conditions with consistent data, preprocessing, and evaluation protocols to ensure comparability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8113025426864624, "reasoning_key": "design_justification", "reasoning_sim": 0.8611286878585815}, {"unit_index": 1, "inspected_object": "Missing FFNO baseline in Table 1", "observation": "The FFNO baseline from 'The Well' is missing from Table 1, while errors from DPOT, MPP, and The Well are copied.", "reasoning": "Selective citation of prior results without explanation raises questions about the completeness and transparency of the comparison, although it is considered minor relative to larger comparability issues.", "judgment": "Minor concern regarding selective reporting of baselines; requires justification for omission.", "valence": "negative", "suggested_improvement": "Explain the selection logic for including only three out of four models or include the missing FFNO baseline.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7546623945236206, "reasoning_key": "fair_comparison", "reasoning_sim": 0.824658989906311}, {"unit_index": 2, "inspected_object": "Reconstruction vs. Prediction conflation in Table 1", "observation": "Table 1 compares a reconstruction task (P2VAE) to prediction tasks (other baselines). Reconstruction is inherently easier than prediction, as evidenced by VAE vs FMT losses in Table 2.", "reasoning": "If the autoencoder performs poorly at the easier reconstruction task compared to predictive baselines, it is likely the full generative model will perform worse than those same baselines at the harder prediction task. This creates a pessimistic inference about the method's actual standing.", "judgment": "The comparison is misleading because the component (autoencoder) does not clearly win even at its easier task, suggesting the full model likely won't win at its harder task.", "valence": "negative", "suggested_improvement": "Clarify the distinction between tasks or provide evidence that the generative model overcomes the autoencoder's limitations.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8546463251113892, "reasoning_key": "design_justification", "reasoning_sim": 0.836030900478363}, {"unit_index": 3, "inspected_object": "Table 3 rollout comparison baselines", "observation": "Table 3 compares rollout performance to only one baseline.", "reasoning": "Rollout claims require comparison against strong, well-understood deterministic baselines like FNO/UNet to establish robustness. Comparing against a single, potentially weak or unrepresentative baseline fails to justify the method's superiority.", "judgment": "Insufficient evidence for rollout superiority due to lack of strong comparative baselines.", "valence": "negative", "suggested_improvement": "Add comparisons to FNO and UNet baselines for rollout performance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8170157670974731, "reasoning_key": "design_justification", "reasoning_sim": 0.8388592600822449}, {"unit_index": 4, "inspected_object": "Discarding physical variables in data preprocessing", "observation": "The authors discard extra physical variables, labeling them as 'unimportant'.", "reasoning": "In domains like compressible Navier-Stokes, accessing new variables like energy is the point of solving the equations. Discarding these variables changes the nature of the prediction problem and may remove physically significant content.", "judgment": "Preprocessing decision is domain-inappropriate and potentially harmful to the validity of the prediction task.", "valence": "negative", "suggested_improvement": "Retain physically relevant variables or justify their exclusion based on domain-specific necessity rather than generic importance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8253465294837952, "reasoning_key": "design_justification", "reasoning_sim": 0.7787016034126282}, {"unit_index": 5, "inspected_object": "Downsampling physical fields", "observation": "The preprocessing pipeline downsamples physical fields.", "reasoning": "Downsampling destroys finer-scale features that could be important for phenomena like mixing behavior in Rayleigh-Benard Convection. This scale-sensitivity argument suggests the preprocessing removes critical information needed for accurate modeling.", "judgment": "Preprocessing degrades the quality of the input data by removing scale-sensitive features essential for the target phenomena.", "valence": "negative", "suggested_improvement": "Avoid downsampling if it removes features critical to the physical phenomena being modeled.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8285737037658691, "reasoning_key": "design_justification", "reasoning_sim": 0.8125923275947571}, {"unit_index": 6, "inspected_object": "Computational cost of FMT", "observation": "FMT requires 100 denoising steps versus 1-step direct prediction for baselines, resulting in longer runtime.", "reasoning": "Higher computational cost must be justified by superior performance. If the comparative advantage is unclear due to flawed baselines, the cost-benefit ratio is unfavorable.", "judgment": "Cost-benefit concern: high computational cost is not justified by clear performance gains in the current evidence base.", "valence": "negative", "suggested_improvement": "Demonstrate that the performance gain justifies the increased computational cost.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8296750783920288, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8307067155838013}, {"unit_index": 7, "inspected_object": "Ablation against pure flow matching and deterministic case", "observation": "The paper lacks a comparison showing Flow Marching is better than deterministic prediction and pure flow matching.", "reasoning": "To motivate the specific contribution of 'flow marching', the method must be shown to outperform its deterministic ancestor and its flow-matching cousin. Without this minimal experiment, the core claim is unmotivated.", "judgment": "Core contribution is unmotivated due to missing ablation studies against key variants.", "valence": "negative", "suggested_improvement": "Show that Flow Marching is better than the deterministic case and purely flow matching case.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8401963114738464, "reasoning_key": "design_justification", "reasoning_sim": 0.8343636393547058}]}, {"review_id": "0764oGi4Re", "paper_id": "cP2nOl3t3W", "paper_title": "Causal Canonical Modeling for Confounding Robust Treatment Evaluation", "decision": null, "summary": "The reviewer accepts the theoretical novelty of CCMs but finds the empirical validation weak due to poor posterior accuracy, lack of benchmarks, and missing asymptotic guarantees, while also questioning the feasibility of the chosen inferential target.", "units": [{"unit_index": 0, "inspected_object": "Causal canonical models (CCMs) and their theoretical density property", "observation": "The reviewer acknowledges that CCMs provide a solid theoretical foundation for approximating causal effects and accepts the proof that CCMs are dense in the space of all structural causal models.", "reasoning": "The reviewer explicitly credits the comprehensive analysis of the causal approximation property and the density result as a strength, accepting it without demanding further proof details in the review context.", "judgment": "The theoretical framework is recognized as providing a solid foundation and represents a valuable theoretical contribution.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8263199925422668, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8122884631156921}, {"unit_index": 1, "inspected_object": "Simulation study methodology and exposition", "observation": "The paper does not provide a sufficiently clear exposition of the method used in the simulation study, lacking pseudo-code for the entire algorithm.", "reasoning": "Without pseudo-code or clear procedural transparency, the approach remains a theoretical construction rather than an implementable procedure, hindering reproducibility and understanding.", "judgment": "The presentation of the simulation methodology is insufficiently clear for practical implementation and verification.", "valence": "negative", "suggested_improvement": "Include pseudo-code for the entire algorithm to make the approach more understandable and reproducible.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7739862203598022, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.783473551273346}, {"unit_index": 2, "inspected_object": "Empirical performance shown in Figure 6a", "observation": "In Figure 6a, the posteriors are quite far from the ground truth.", "reasoning": "The distance between posterior and ground truth undermines confidence in the method's empirical performance; since theoretical guarantees are not rigorous enough to compensate, this poor performance on the paper's own terms suggests insufficient empirical evidence.", "judgment": "The simulation results alone do not sufficiently convince the reviewer of the method's effectiveness due to observed poor performance relative to ground truth.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8261946439743042, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8133400082588196}, {"unit_index": 3, "inspected_object": "Benchmarking in the simulation study", "observation": "The simulation study lacks a benchmark that would demonstrate the performance of the method against alternatives.", "reasoning": "Without a comparative frame against existing approaches, the reviewer cannot calibrate whether the observed performance is acceptable or concerning, nor can they assess if the method works better than or comparably to alternatives.", "judgment": "The absence of benchmarks prevents proper evaluation of the method's utility and relative standing.", "valence": "negative", "suggested_improvement": "Add benchmarks comparing the proposed method against competing methods in the literature.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8374160528182983, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8378231525421143}, {"unit_index": 4, "inspected_object": "Asymptotic analysis and uncertainty quantification", "observation": "The paper lacks asymptotic analysis or uncertainty quantification despite having a density result.", "reasoning": "The density result addresses representation capacity but not estimation consistency or posterior concentration; the absence of any formal statement about behavior as sample size grows creates discomfort regarding the method's reliability and interpretability.", "judgment": "The lack of asymptotic analysis or uncertainty quantification leaves the method uncharacterized in a way that prevents readers from assessing its reliability.", "valence": "negative", "suggested_improvement": "Provide simple asymptotic results under strong assumptions or include uncertainty quantification to help readers assess theoretical guarantees.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8543883562088013, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8116584420204163}, {"unit_index": 5, "inspected_object": "Target parameter scope (full treatment response function vs ATE)", "observation": "The reviewer questions targeting the entire treatment response function and suggests estimating simpler parameters like the Average Treatment Effect (ATE).", "reasoning": "If the full response function is hard to estimate well, a coarser functional like the ATE might be more tractable and easier to validate, potentially offering a more modest but achievable objective with better empirical performance.", "judgment": "The ambition of estimating the full response function may exceed what the method can reliably deliver, suggesting a need for validation on simpler targets.", "valence": "conditional", "suggested_improvement": "Demonstrate how the method could be used to estimate the average treatment effect (ATE) and provide specific simulation results for this target.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.7769594788551331, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8068769574165344}]}, {"review_id": "07Ez0Eu6WD", "paper_id": "j0czDrEnFc", "paper_title": "A Bayesian Nonparametric Framework for Private, Fair, and Balanced Tabular Data Synthesis", "decision": "Accept (Poster)", "summary": "The reviewer acknowledges the architectural novelty of combining privacy, fairness, and balance mechanisms but judges the paper insufficient due to gaps between claims and evidence. Specifically, the reviewer critiques the lack of justification for the fairness metric, unsupported scalability claims, failure to isolate component effects in experiments, and inadequate handling of dataset limitations. The reviewer offers concrete requests for additional analyses and experiments to bridge these gaps.", "units": [{"unit_index": 0, "inspected_object": "The combination of DP-based privacy, MI-based fairness, and conditional generation for balance.", "observation": "The reviewer acknowledges the specific mechanisms (Dirichlet process, copula-based localization, VAE+GAN) and notes that unifying these in a single BNPL framework is novel.", "reasoning": "The reviewer identifies the paper as offering a 'complete solution' to DP + fairness and recognizes the novelty of the specific mechanism combination, granting genuine architectural contribution.", "judgment": "The paper possesses genuine novelty and a non-trivial contribution in its architectural design.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7702894806861877, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7986394762992859}, {"unit_index": 1, "inspected_object": "The choice of statistical parity (via MI minimization) as the sole fairness target.", "observation": "The reviewer notes that equalized odds and equalized opportunity are not addressed, creating a gap between the general claim of 'fairness' and the specific operationalization used.", "reasoning": "A normative assumption holds that fairness-aware methods must either justify their metric choice or demonstrate robustness across multiple definitions; the absence of such engagement suggests incomplete justification.", "judgment": "The fairness component lacks sufficient justification relative to field norms for metric selection.", "valence": "negative", "suggested_improvement": "Provide head-to-head comparisons or rank-correlations with fairness baselines beyond DECAF/TabFairGAN/FairGAN, or explicitly discuss compatibility with other fairness definitions.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8474668264389038, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8035422563552856}, {"unit_index": 2, "inspected_object": "Scalability claims attached to DirPMINE.", "observation": "The paper proposes DirPMINE for scalable MI estimation but provides no variance analysis, sample complexity bounds, or failure-mode discussion relative to standard MINE.", "reasoning": "Scalability is treated as a technical claim requiring formal support (variance, sample complexity) rather than a design aspiration; the lack of uncertainty quantification (confidence intervals) on metrics indicates potential instability.", "judgment": "The scalability claims are unsupported by necessary theoretical or empirical stability evidence.", "valence": "negative", "suggested_improvement": "Provide runtime/memory analysis, hyperparameter stability studies, and confidence intervals on MI/MMD/utility metrics.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8519976139068604, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8118283152580261}, {"unit_index": 3, "inspected_object": "Empirical demonstration via toy experiments isolating component effects.", "observation": "The current figures hint at effects but do not isolate mechanisms; the multi-component design prevents attribution of effects to individual components (DP resampling, localized privacy, class balancing).", "reasoning": "Multi-component methods require ablation-style isolation to demonstrate independent contributions; without this, the causal link between specific mechanisms and outcomes remains speculative.", "judgment": "The experimental validation fails to isolate the independent contributions of the method's components.", "valence": "negative", "suggested_improvement": "Conduct three specific toy experiments: one isolating DP resampling effects, one isolating localized privacy trade-offs, and one isolating the independent contributions of class balancing vs. fairness regularization.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8335968255996704, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7865852117538452}, {"unit_index": 4, "inspected_object": "Dataset choices (Adult and COMPAS).", "observation": "The reviewer flags documented limitations of Adult and COMPAS regarding representativeness and fairness properties, citing specific arXiv critiques.", "reasoning": "In fairness research, dataset bias is a subject of study; authors are expected to engage with the broader literature's self-critique of benchmark datasets rather than treating them as neutral ground.", "judgment": "The empirical evaluation relies on datasets whose known limitations are not adequately addressed or contextualized.", "valence": "negative", "suggested_improvement": "Include more diverse datasets or provide a detailed discussion of the existing datasets' limitations in light of broader fairness-ML literature critiques.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.810197114944458, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7437182664871216}]}, {"review_id": "07QRccdbO5", "paper_id": "9y2IyqaWxs", "paper_title": "PIANO: Physics-Informed Autoregressive Networks", "decision": null, "summary": "The reviewer primarily judges the work as lacking novelty because prior autoregressive PINN methods exist, while also identifying multiple gaps in the discussion regarding horizons, efficiency, robustness, and architecture.", "units": [{"unit_index": 0, "inspected_object": "Novelty of the autoregressive framework for PINNs", "observation": "Prior works exist that perform auto-regressive prediction in PINNs.", "reasoning": "The existence of prior works using the same core mechanism (autoregression) in the same subfield establishes that the contribution is incremental unless a clear delta is shown, which the reviewer finds absent.", "judgment": "Lack of novelty; low contribution.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8012959957122803, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7672133445739746}, {"unit_index": 1, "inspected_object": "Training vs. testing horizon mismatch", "observation": "No discussion regarding the training horizon using BPTT versus the testing horizon.", "reasoning": "Autoregressive models trained with finite rollout horizons should discuss behavior when inference extends beyond that horizon to address potential failure modes like error accumulation or distribution shift.", "judgment": "Incomplete analysis of generalization across time scales.", "valence": "negative", "suggested_improvement": "Discuss what happens if the prediction horizon is much larger than the training one.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8148167729377747, "reasoning_key": "design_justification", "reasoning_sim": 0.7863914966583252}, {"unit_index": 2, "inspected_object": "Inference time cost", "observation": "No quantification of the increase in inference time due to the autoregressive nature compared to other methods.", "reasoning": "A paper introducing a new framework should quantify computational overhead and efficiency relative to baselines.", "judgment": "Missing practical evaluation of computational efficiency.", "valence": "negative", "suggested_improvement": "Quantify the increase in inference time due to autoregression compared to other methods.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8209877610206604, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8089373111724854}, {"unit_index": 3, "inspected_object": "Robustness to qualitative regime changes", "observation": "No discussion on whether the autoregressive paradigm introduces more errors if the dynamical system's behavior qualitatively changes (e.g., chaotic systems).", "reasoning": "The autoregressive assumption may break down for systems with bifurcations or chaos where small errors are amplified, requiring robustness analysis.", "judgment": "Unaddressed robustness concern regarding dynamical regimes.", "valence": "negative", "suggested_improvement": "Investigate if autoregressive assumptions hold for chaotic systems or those with bifurcations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8269575238227844, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8069855570793152}, {"unit_index": 4, "inspected_object": "Architecture ablation (SSM vs. Attention)", "observation": "No ablation study comparing the State Space Model (SSM) used in the network with softmax attention.", "reasoning": "To determine if the framework's success depends on the specific sequence model or generalizes across architectures, an ablation is required.", "judgment": "Missing comparative analysis of architectural components.", "valence": "negative", "suggested_improvement": "Perform an ablation study replacing the SSM with softmax attention.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8262209296226501, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7316768169403076}]}, {"review_id": "01unspMuYs", "paper_id": "pWfq8mc2xC", "paper_title": "LoRAMax: Adaptive Low-Rank Gated Module  for Extreme-Scale language Style Modeling", "decision": null, "summary": "The reviewer identifies multiple epistemic obstacles to trusting the paper's claims, including unverifiable evidence from a private dataset, unspecified hybrid evaluation methods, missing ablations, thin theoretical grounding, and contextual opacity.", "units": [{"unit_index": 0, "inspected_object": "Dataset provenance and privacy status", "observation": "The experiments depend solely on a private dataset with an unclear source.", "reasoning": "Without access to the data, empirical claims cannot be independently verified or trusted.", "judgment": "Reproducibility is limited and external validation is impossible.", "valence": "negative", "suggested_improvement": "Disclose the dataset source or provide public access for verification.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.768564760684967, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.803662896156311}, {"unit_index": 1, "inspected_object": "Hybrid evaluation methodology (LLM-based scoring)", "observation": "Style and relevance scores partially depend on LLM-based scoring (DeepSeek-R1), mixing human and LLM annotations without clear specification.", "reasoning": "Using DeepSeek-R1 introduces potential bias toward similar model families, and the hybrid approach is underspecified for reconstruction.", "judgment": "Evaluation validity is questionable due to lack of transparency and potential systematic inflation of scores.", "valence": "negative", "suggested_improvement": "Clarify exactly how the hybrid approach works and address potential evaluator bias.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8651719689369202, "reasoning_key": "design_justification", "reasoning_sim": 0.7977267503738403}, {"unit_index": 2, "inspected_object": "Absence of ablation studies", "observation": "There are no ablation studies that isolate the effects of the different parts of the framework.", "reasoning": "Causal claims require controlled variation; without isolating components like decomposition, pretraining, gating, and fusion, individual contributions cannot be established.", "judgment": "The contribution of specific components is unestablished.", "valence": "negative", "suggested_improvement": "Conduct ablation studies to isolate the effects of different framework parts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7408319115638733, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7952406406402588}, {"unit_index": 3, "inspected_object": "Justification of linguistic dimensions", "observation": "The twelve linguistic dimensions are motivated by only one non-peer-reviewed paper (Fan et al. 2025), ignoring longer history in NLP style research.", "reasoning": "The theoretical grounding is thin; justification should be against a broader literature to ensure dimensions are well-defined and appropriate.", "judgment": "The justification for the chosen dimensions is inadequate.", "valence": "negative", "suggested_improvement": "Justify the choice of dimensions against a broader body of peer-reviewed literature.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7926708459854126, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7409065961837769}, {"unit_index": 4, "inspected_object": "Application context and presentation clarity", "observation": "The 'official account platform' is undefined, and Chinese examples are not understandable to non-Chinese speakers.", "reasoning": "Opacity in application context prevents assessment of generalizability and applicability; language barriers compound this presentation issue.", "judgment": "The work's applicability and generalizability cannot be fully assessed due to contextual opacity.", "valence": "negative", "suggested_improvement": "Clarify the application context and provide translations or explanations for non-Chinese examples.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8297238349914551, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8156954646110535}]}, {"review_id": "04AW0r4kJ3", "paper_id": "P00k4DFaXF", "paper_title": "Process-Verified Reinforcement Learning for Theorem Proving via Lean", "decision": "Accept (Poster)", "summary": "The reviewer balances appreciation for the paper's clean design and rigorous ablations with skepticism about the theoretical grounding of the reward function's fixed values and the robustness of empirical gains under further scaling.", "units": [{"unit_index": 0, "inspected_object": "The paper's procedural clarity and ablation studies.", "observation": "The reviewer finds thorough definitions, descriptions of procedures, and ablation studies confirming the significance of design decisions.", "reasoning": "These elements demonstrate internal rigor and allow the reviewer to appreciate the 'clean idea' of leveraging Lean's fine-grained feedback as a natural extension of the proof assistant's role.", "judgment": "Positive assessment of the paper's conceptual elegance and empirical thoroughness.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7847629189491272, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8078123927116394}, {"unit_index": 1, "inspected_object": "The scope of the experimental evaluation (full proof generation vs. search/repair).", "observation": "The analysis is limited to full proof generation, excluding search and iterative repair techniques common in automatic theorem proving.", "reasoning": "This limitation may bias the evaluation towards a human perspective rather than reflecting the broader ATP community's practice of interactive search, although the reviewer acknowledges this expectation might not align with the paper's specific contribution goals.", "judgment": "Negative assessment regarding scope representativeness, flagged as a weakness but with self-aware hedging about the reviewer's own bias.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8447014093399048, "reasoning_key": "novelty_standard", "reasoning_sim": 0.799409806728363}, {"unit_index": 2, "inspected_object": "The fixed reward value assigned to valid steps preceding incorrect steps.", "observation": "The reward assigns a hard-coded constant to syntactically valid steps that do not necessarily contribute to global proof progress.", "reasoning": "While the paper's ablations show that ordering between syntactic validity and invalidity matters, the specific fixed value lacks principled grounding because it conflates local syntactic well-formedness with semantic proof progress, potentially undermining the claim of providing dense, verifier-grounded credit signals.", "judgment": "Negative assessment of the reward design's conceptual justification, labeled as a 'nit' but substantive to the method's core claim.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7351110577583313, "reasoning_key": "design_justification", "reasoning_sim": 0.8215965628623962}, {"unit_index": 3, "inspected_object": "The robustness of empirical gains at scale.", "observation": "The differences between conditions in the outcome reward chart appear fairly small, and RL training dynamics are known to be noisy with potential for plateaus or negative trends.", "reasoning": "Without evidence from further scaling, it is uncertain whether the observed process-reward advantage persists or washes out in training noise, raising concerns about the fragility of the reported results.", "judgment": "Uncertain assessment of the long-term viability and robustness of the method's performance.", "valence": "uncertain", "suggested_improvement": "Provide data points from further scaled training runs to confirm that patterns hold beyond the current budget.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.86212557554245, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8211447596549988}]}, {"review_id": "04ayHBUrPG", "paper_id": "HW2jd9vfCU", "paper_title": "Pathway to $O(\\sqrt{d})$ Complexity bound under Wasserstein metric of flow-based models", "decision": "Reject", "summary": "The reviewer conducts a diagnostic critique focusing on communicative adequacy rather than mathematical falsification. They identify significant gaps between the paper's claims and its explicit presentation, including opaque dimensional dependence in the main theorem, unmapped error decompositions, overstated necessity claims, and a lack of synthesis. While acknowledging the paper's merit in achieving lower complexity bounds under general conditions, the reviewer judges the work as unfit for publication due to these structural and explanatory deficiencies.", "units": [{"unit_index": 0, "inspected_object": "Theorem 3.15 (Main Theorem)", "observation": "The theorem statement does not explicitly display the dimension variable d, requiring the reader to reverse-engineer that dimensional dependence is hidden in the constant M_0.", "reasoning": "A main contribution should make its scaling properties transparent; forcing the reader to deduce implicit dependencies violates communicative norms and requires excessive reconstructive effort.", "judgment": "The presentation of the main result is opaque and fails to meet standards for clarity in a theoretical paper.", "valence": "negative", "suggested_improvement": "Make the dimensional scaling explicit in the theorem statement or provide a clearly annotated reference table.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.805403470993042, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7886275053024292}, {"unit_index": 1, "inspected_object": "Abstract's error decomposition claim", "observation": "The abstract claims error is controlled by two parts (Lipschitzness and discretization), but these are not explicitly mapped to terms in the body text.", "reasoning": "Abstract promises must be directly redeemable in the main text without requiring the reader to perform interpretive labor to connect claims to equations.", "judgment": "The connection between the abstract's summary and the detailed results is insufficiently explicit.", "valence": "negative", "suggested_improvement": "Add explicit textual mapping identifying which terms in which equations correspond to the claimed decomposition parts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8354383111000061, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7429348826408386}, {"unit_index": 2, "inspected_object": "Corollaries 3.16 and 3.20 ('Requirement' language)", "observation": "The paper uses the word 'requirement' for the order of N, implying necessity, but the derivation only establishes sufficiency via non-tight inequalities.", "reasoning": "The Wasserstein-to-L^2 inequality and the Föllmer flow are not optimal transport maps, meaning the bound is loose; therefore, stating N is a 'requirement' overstates the epistemic strength of the result compared to an upper bound guarantee.", "judgment": "The logical status of the claims regarding N is overstated relative to the mathematical derivation.", "valence": "negative", "suggested_improvement": "Revise language from 'requirement' to 'sufficient condition' or tighten the derivation if necessity is intended.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7077494263648987, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7221240997314453}, {"unit_index": 3, "inspected_object": "Paper structure and ending", "observation": "The paper ends abruptly without a remarks or discussion section synthesizing results, contextualizing them, or noting limitations.", "reasoning": "A theoretical paper of this scope requires standard expository apparatus to synthesize main claims and compare against prior work; its absence makes the format unfit for the venue regardless of mathematical merit.", "judgment": "The paper is incomplete in its exposition and not ready for publication in its current form.", "valence": "negative", "suggested_improvement": "Include a discussion section that synthesizes main results, contextualizes them against prior work, and notes limitations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8157572746276855, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7796016931533813}, {"unit_index": 4, "inspected_object": "Proof technique novelty", "observation": "The reviewer questions how the Gaussian tail assumption yields a tighter complexity bound despite being more general and presumably harder.", "reasoning": "There is a lack of clarity on whether the improvement stems from a genuinely new technique or repackaging of existing arguments with different constants.", "judgment": "The source of the improvement is unclear, creating uncertainty about the distinctiveness of the methodological contribution.", "valence": "uncertain", "suggested_improvement": "Clarify the specific proof technique that differs from previous works to allow the tighter bound.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8722130060195923, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8364028930664062}]}, {"review_id": "05G4iKghyy", "paper_id": "vv6pZQAc5S", "paper_title": "Polar probe linearly decodes semantic structures from LLMs", "decision": "Reject", "summary": "The reviewer evaluates the paper primarily through the lenses of novelty, scalability, and causal rigor, concluding that the work is an incremental, non-scalable validation of prior findings that fails to meet current community standards for causal interpretability. While acknowledging strengths in execution and downstream links, the reviewer judges these insufficient to offset the lack of novelty and methodological depth.", "units": [{"unit_index": 0, "inspected_object": "Novelty of the central hypothesis (relation existence maps to distance, relation type to direction)", "observation": "The reviewer identifies that this finding was previously demonstrated by Tehenan et al. with stronger, causal evidence using interventions.", "reasoning": "The existence of prior work demonstrating the same core claim repositions the paper from a novel discovery to an incremental validation; the reviewer treats any prior demonstration as sufficient to diminish contribution regardless of domain specificity or methodological novelty.", "judgment": "The paper's contribution is merely incremental and insufficiently novel.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8410518169403076, "reasoning_key": "novelty_standard", "reasoning_sim": 0.853401780128479}, {"unit_index": 1, "inspected_object": "Experimental scale and generalizability of the polar code principle", "observation": "The tasks are minimalist (4–6 entities) and Figure 4 shows severe performance degradation as entity count increases.", "reasoning": "A genuine representational principle should scale; failure to scale in larger contexts suggests the finding is a brittle artifact of toy simplicity rather than a robust mechanism.", "judgment": "The claimed simple geometrical principle is likely invalid as a general mechanism due to lack of scalability.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8442062735557556, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7998490333557129}, {"unit_index": 2, "inspected_object": "Correlational methodology vs. causal evidence standards", "observation": "The study relies on probing (correlational analysis) without interventional experiments such as activation steering.", "reasoning": "The interpretability community has largely moved beyond correlational analyses toward causal evidence; the absence of interventions is a significant deficiency in modern interpretability research.", "judgment": "The methodology is outdated and deficient because it lacks causal verification.", "valence": "negative", "suggested_improvement": "Perform interventional experiments (e.g., activation steering) to demonstrate causal utility.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7974615097045898, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7203695178031921}, {"unit_index": 3, "inspected_object": "Presentation quality and analytical depth", "observation": "The Results section is largely a textual description of figures with little substantive analysis, making the paper feel padded.", "reasoning": "Thin analysis suggests thin thinking; the length of the paper is not matched by analytical depth, which negatively impacts the overall presentation score.", "judgment": "The presentation is weak and contributes to a low overall quality assessment.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8717769384384155, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8001739382743835}, {"unit_index": 4, "inspected_object": "Methodological choices: PCA visualization interpretation", "observation": "The reviewer questions what PC1 and PC2 represent in the PCA visualization used in the paper.", "reasoning": "Without clear justification for the projection axes, the interpretability claims regarding the geometric structure are challenged; the reviewer prefers hypothesis-direct visualization.", "judgment": "The PCA interpretation requires justification to support the paper's claims about semantic structures.", "valence": "conditional", "suggested_improvement": "Justify the PCA visualization by explaining what PC1 and PC2 represent.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "clarity", "object_sim": 0.7966880202293396, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7931190133094788}, {"unit_index": 5, "inspected_object": "Statistical robustness of Spearman correlation", "observation": "The reviewer expresses uncertainty about whether Spearman correlation was appropriate in every case.", "reasoning": "To ensure statistical validity, raw data and alternative metrics should be examined to confirm the robustness of the reported correlations.", "judgment": "The statistical reporting requires additional scrutiny to confirm appropriateness across all cases.", "valence": "conditional", "suggested_improvement": "Provide raw data or alternative metrics to validate the use of Spearman correlation.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8214765191078186, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7842379212379456}, {"unit_index": 6, "inspected_object": "Absence of baseline probes", "observation": "The paper does not include comparisons to simpler baseline probes.", "reasoning": "Without baselines, the added value of the Polar Probe method remains unclear and cannot be properly evaluated against standard methods.", "judgment": "The lack of baseline comparisons obscures the specific contribution of the proposed method.", "valence": "negative", "suggested_improvement": "Include comparisons to simpler baseline probes to clarify the Polar Probe's added value.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.824634313583374, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.793508768081665}]}, {"review_id": "05Il9DDmTl", "paper_id": "9CpHEbtvA9", "paper_title": "reAR: Rethinking Visual Autoregressive Models via Token-wise Consistency Regularization", "decision": "Accept (Poster)", "summary": "The reviewer conducts a conditional validation of the paper, identifying three primary weaknesses: the potential lack of novelty in the perturbation mechanism, insufficient empirical breadth beyond ImageNet, and ambiguous scope regarding applicability to VAR. Additionally, the reviewer critiques the clarity of Figure 1 and terminology. The evaluation is structured dialectically, balancing these concerns against acknowledged strengths like embedding alignment, with each weakness linked to a potential rating improvement if resolved.", "units": [{"unit_index": 0, "inspected_object": "The novelty of the perturbed context exposure mechanism", "observation": "The reviewer identifies that perturbing inputs is a common practice for robustness and questions why the proposed perturbation is non-obvious compared to existing literature.", "reasoning": "Applying an incremental novelty standard, the reviewer argues that borrowing a standard technique requires justification for its specific application to visual token contexts to address generator-tokenizer inconsistency, rather than being a derivative use of generic robustness training.", "judgment": "The perturbation component is viewed as potentially derivative or lacking sufficient originality unless mechanistic justification is provided.", "valence": "negative", "suggested_improvement": "Provide a mechanistic justification explaining why perturbing visual token contexts specifically addresses the generator-tokenizer inconsistency in a way generic robustness training would not.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8263161182403564, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8400440812110901}, {"unit_index": 1, "inspected_object": "The breadth of empirical validation on ImageNet only", "observation": "Experiments are conducted exclusively on the ImageNet dataset.", "reasoning": "Applying a proportionality norm, the reviewer asserts that claims addressing a fundamental problem in autoregressive models require evidence across multiple domains or tasks to justify the scope of the contribution.", "judgment": "Single-dataset validation is insufficient to support the broad claims of addressing a core bottleneck, posing a soundness concern due to limited generalizability evidence.", "valence": "negative", "suggested_improvement": "Demonstrate effectiveness across multiple domains or tasks, or temper the empirical claims to match the single-benchmark evidence.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8551674485206604, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7628189325332642}, {"unit_index": 2, "inspected_object": "The scope and boundary definition of the paper's claims", "observation": "The paper does not clearly specify whether it applies only to raster-order autoregressive modeling or also to other models like VAR (Visual Autoregressive/next-scale prediction), despite the broad title.", "reasoning": "Applying a precise scoping norm, the reviewer argues that the paper must explicitly delimit its claims to avoid overclaiming relative to its actual scope, particularly regarding alternative approaches like VAR.", "judgment": "The scope is ambiguous and potentially overclaimed, leading to clarity issues regarding the boundaries of the contribution.", "valence": "negative", "suggested_improvement": "Explicitly state what the method does not claim to address (e.g., VAR) or demonstrate/clarify applicability to VAR if intended.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.784307599067688, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7562629580497742}, {"unit_index": 3, "inspected_object": "Figure 1 (the orange cat image) and its explanatory power", "observation": "The reviewer finds the image of the orange cat confusing, noting that top and bottom images appear similar despite different token indices, and questions what is wrong with the image.", "reasoning": "Applying a visual evidence standard, the reviewer expects motivating examples to be self-explanatory and clearly demonstrate the claimed tokenizer inconsistency; the confusion suggests the figure fails to effectively prove the mechanism or is poorly presented.", "judgment": "The motivating example lacks clarity and fails to convincingly demonstrate the claimed phenomenon without further explanation.", "valence": "negative", "suggested_improvement": "Improve the clarity of Figure 1 to make the discrepancy between token indices and visual output self-evident, or provide better textual explanation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8108300566673279, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7852447628974915}, {"unit_index": 4, "inspected_object": "Definition of the term 'context'", "observation": "The term 'context' is used without explicit definition, relying on implication.", "reasoning": "Applying a terminological clarity norm, the reviewer notes that key terms should be defined explicitly for a general audience to ensure expository precision.", "judgment": "The lack of explicit definition creates a minor but genuine clarity issue regarding expository precision.", "valence": "negative", "suggested_improvement": "Define 'context' explicitly in the text.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7007966637611389, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7626476883888245}]}, {"review_id": "05dD4x7rJK", "paper_id": "OQ37q5JEtQ", "paper_title": "Improving Causal Inference Robustness via Reinforcement-guided Diffusion Models", "decision": null, "summary": "The reviewer systematically delimits the paper's contribution by arguing that experimental designs favor the method through structural alignment, that robust optimization cannot solve identifiability issues, that flexibility precludes formal guarantees, that computational costs limit applicability, and that presentation errors undermine credibility.", "units": [{"unit_index": 0, "inspected_object": "Experimental design for measurement error, missing values, and unmeasured confounding", "observation": "Gaussian noise and missing value generation mechanisms are structurally similar to the adversarial training perturbations used by CARD.", "reasoning": "If test-time corruption resembles training-time augmentation, the method is evaluated under conditions favorable by construction; the observed benefit diminution in unmeasured confounding (a structural change) reveals that the robustness in other cases may be inflated due to this alignment.", "judgment": "The experimental evidence for robustness is potentially biased by the similarity between evaluation and training mechanisms.", "valence": "negative", "suggested_improvement": "Reformulate experimental design to include corruption mechanisms orthogonal to CARD's training procedure to test robustness more fairly.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8667624592781067, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8192278742790222}, {"unit_index": 1, "inspected_object": "Theoretical scope regarding identifiability vs. optimization", "observation": "The paper presents an optimization algorithm rather than an identification result, acknowledging that data augmentation cannot recover missing causal information.", "reasoning": "Robust optimization improves stability within a given causal model but cannot correct fundamental bias arising from misspecified causal structures (unmeasured confounding); thus, the contribution is limited to stability enhancement rather than bias correction.", "judgment": "The contribution is more modest than the framing suggests, as it does not address identifiability.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8588929176330566, "reasoning_key": "design_justification", "reasoning_sim": 0.8326262831687927}, {"unit_index": 2, "inspected_object": "Lack of formal theoretical guarantees for the learning-based agent", "observation": "The RL-guided diffusion agent explores a large, less structured space of perturbations without predefined uncertainty sets.", "reasoning": "Unlike robust optimization methods with known uncertainty sets (e.g., Wasserstein balls), the flexibility of CARD makes it impossible to state a priori conditions for performance improvement, meaning behavior is only empirically validated, not theoretically bounded.", "judgment": "The method fails to meet the threshold for theoretical contribution required by the reviewer.", "valence": "negative", "suggested_improvement": "Provide formal sensitivity analysis or guarantees specifying conditions under which CARD improves performance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8205095529556274, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7823207378387451}, {"unit_index": 3, "inspected_object": "Computational cost of diffusion model training", "observation": "Training complexity increases from O(E) to O(ET) due to diffusion steps.", "reasoning": "For large-scale or real-time applications, this added cost may be prohibitive, limiting the broad applicability claimed in the abstract.", "judgment": "Practical deployment is hindered by computational demands.", "valence": "negative", "suggested_improvement": "Address computational cost seriously, perhaps by proposing efficiency improvements or quantifying the trade-off more precisely.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8010086417198181, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7963573932647705}, {"unit_index": 4, "inspected_object": "Presentation consistency in Table 2", "observation": "Bolding convention marking lower PEHE as 'better' is violated in S-learner comparisons where '+CARD' variants have higher PEHE yet are bolded.", "reasoning": "Inconsistent formatting can mislead readers into interpreting worse results as better, suggesting a lack of careful proofreading that undermines trust in the rigor of the presentation.", "judgment": "Presentation errors undermine reader credibility and readability.", "valence": "negative", "suggested_improvement": "Fix the presentation errors in Table 2.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8399040699005127, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8873153328895569}]}, {"review_id": "05qnlU8TnW", "paper_id": "2yWis3aAvj", "paper_title": "Neural Operator-based Curriculum Learning for Physics-Informed Neural Networks", "decision": null, "summary": "The reviewer critically evaluates the residual-masking transfer mechanism, identifying fundamental gaps in theoretical justification, potential internal inconsistencies with empirical evidence, and scalability issues with the NTK metric, while also noting presentation deficiencies.", "units": [{"unit_index": 0, "inspected_object": "Residual-masking initialization strategy for PINN training", "observation": "The mechanism assumes structural consistency of solutions across optimization stages, but the paper provides no account of when this holds or how optimization dynamics might differ.", "reasoning": "If the new model's optimization dynamics differ from previous stages, points that were 'good' in prior stages may no longer be effective, making the transfer unreliable without theoretical grounding.", "judgment": "The mechanism is potentially fragile and lacks sufficient justification to be trusted as a robust transfer method.", "valence": "negative", "suggested_improvement": "Provide theoretical or statistical analysis characterizing the conditions under which cross-stage consistency is guaranteed or likely.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7807958722114563, "reasoning_key": "robustness_norm", "reasoning_sim": 0.811173677444458}, {"unit_index": 1, "inspected_object": "Neural operator (FNO) output used for residual computation", "observation": "The neural operator does not inherently enforce physical consistency, so a low residual could occur by chance rather than indicating physical correctness.", "reasoning": "Conflating empirical fit with physical validity undermines the core purpose of PINNs; the residual metric may be misleading if the underlying model ignores physical laws.", "judgment": "The reliance on residual filtering via a black-box neural operator is epistemically weak because it bypasses physical constraints.", "valence": "negative", "suggested_improvement": "Address the risk that residuals reflect chance fit rather than physical validity, perhaps by incorporating physical constraints into the operator or validating residuals against known physics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7972248792648315, "reasoning_key": "construct_validity", "reasoning_sim": 0.7629920244216919}, {"unit_index": 2, "inspected_object": "Figure 6 showing NTK spectral structures across parameters", "observation": "Figure 6 suggests that NTK spectral structures vary significantly across parameters, which appears to contradict the assumption of mask consistency required for the transfer mechanism.", "reasoning": "If the solution structure (reflected by NTK spectra) varies dramatically, the residual masking from one stage may not transfer usefully to another, undermining the internal logic of the curriculum ordering.", "judgment": "There is an internal inconsistency between the paper's evidence (NTK variation) and its central assumption (mask consistency).", "valence": "negative", "suggested_improvement": "Resolve the tension between observed NTK variability and the assumption of consistent masks, or explain why the variability does not invalidate the transfer.", "support_status": "mixed", "confidence": "medium", "object_key": "theory", "object_sim": 0.8124495148658752, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7511276006698608}, {"unit_index": 3, "inspected_object": "NTK-based difficulty metric for curriculum planning", "observation": "The metric is conceptually insightful and moves away from heuristic ordering, but NTK computation is extremely expensive for large-scale models or datasets.", "reasoning": "While the measure of difficulty is valid, its computational cost limits practical applicability in broader settings, creating a scalability barrier.", "judgment": "The component is theoretically sound but practically limited by efficiency concerns.", "valence": "mixed", "suggested_improvement": "Investigate more efficient ways to perform difficulty-aware curriculum planning to mitigate the high computational cost of NTK computation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7790203094482422, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8391141295433044}, {"unit_index": 4, "inspected_object": "Terminology and presentation clarity", "observation": "Terms like 'function-level transfer' and 'spectral bias avoidance' may obscure the true contribution, and there are typographical errors (e.g., 'curricumun') and cluttered figures.", "reasoning": "Complex terminology may over-claim novelty and hide simpler core ideas, while presentation flaws suggest a lack of polish that reinforces skepticism about rigor.", "judgment": "The presentation obscures the contribution and detracts from the paper's perceived professionalism and clarity.", "valence": "negative", "suggested_improvement": "Adopt a more concise presentation focused on core ideas and correct typographical errors and figure layout issues.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8699052333831787, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7993610501289368}]}, {"review_id": "05xxcZktqm", "paper_id": "fIPng6j4eM", "paper_title": "Scaling Sequence-to-Sequence Generative Neural Rendering", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper primarily through high-level lenses of novelty, claim coherence, and output quality, criticizing the thin novelty of transferring video priors, challenging the breadth of the unification claim via a 4D test, and noting qualitative inconsistencies in demos, while acknowledging strong empirical results and ablations.", "units": [{"unit_index": 0, "inspected_object": "Prior art landscape and core method novelty", "observation": "The reviewer identifies the core method as sequence-to-sequence generation technology borrowed from video generation, noting that using video pretrained models for novel view synthesis is a common solution in the field.", "reasoning": "The reviewer applies a standard where value is inversely proportional to borrowing from adjacent domains; because the transfer of video priors is widespread, the paper's contribution must lie elsewhere, but the described structural designs are minor, leading to a judgment of thin contribution.", "judgment": "The paper's novelty is limited because its core technique resembles existing practice without significant new machinery.", "valence": "negative", "suggested_improvement": "Demonstrate a genuinely novel mechanism for transferring video priors to 3D rather than relying on a common solution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8720691204071045, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8149522542953491}, {"unit_index": 1, "inspected_object": "Unification claim scope (3D vs Video)", "observation": "The reviewer questions how the framework performs on 4D scenes (dynamic 3D) given the paper's claim to have 'unified' 3D and video modelling.", "reasoning": "The reviewer uses a counterfactual probe: if the unification is genuine, it should handle the intersection of both domains (time-varying 3D) without special-casing; failure to address this suggests the unification may be conceptual or marketing-based rather than robust.", "judgment": "The unification claim is challenged as potentially overextended or lacking coherence across the full space of relevant tasks.", "valence": "negative", "suggested_improvement": "Provide evidence of 4D performance to substantiate the unification claim or carefully scope claims.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7697258591651917, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7898719310760498}, {"unit_index": 2, "inspected_object": "Demo results quality", "observation": "The reviewer observes obvious inconsistency in the generated novel views in the demo results.", "reasoning": "The reviewer weighs qualitative outputs against quantitative claims of state-of-the-art performance, finding a gap that undermines trust in the headline claims.", "judgment": "The visual artifacts betray the quantitative metrics, indicating a lack of perceptual quality despite reported scores.", "valence": "negative", "suggested_improvement": "Present demo results without visible inconsistencies.", "support_status": "mixed", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7970525026321411, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8078781366348267}, {"unit_index": 3, "inspected_object": "Empirical results and ablation studies", "observation": "The reviewer acknowledges the empirical results matching per-scene optimization methods in many-view settings and finds value in the ablation studies on positional encoding and architecture designs.", "reasoning": "Systematic empirical exploration and strong performance in specific settings are valued as contributions, even if they do not compensate for novelty concerns.", "judgment": "The paper has merit due to strong empirical validation and thorough ablations.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8466298580169678, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7802954912185669}]}, {"review_id": "068t7ZeR0s", "paper_id": "RSvfY6dRVN", "paper_title": "Learning and Reusing Abstract Latent Actions in a Hippocampal-Entorhinal-Inspired World Model", "decision": "Reject", "summary": "The reviewer positively assesses the paper's novel biological grounding, clear presentation, and theoretically aligned experiments, while noting a gap in demonstrating physical understanding through downstream applications and requesting clarification on a latent space discrepancy.", "units": [{"unit_index": 0, "inspected_object": "The connection between neuroscience (HPC-MEC circuit) and machine learning world models.", "observation": "The reviewer found the motivation well-justified and the connection not forced, respecting the source domain while applying it productively.", "reasoning": "Novelty is rewarded when disciplined by the source domain constraints; this generates interesting points for future work rather than being merely descriptive.", "judgment": "Strong novelty grounded in biological fidelity.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8079690337181091, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8187019228935242}, {"unit_index": 1, "inspected_object": "The clarity of Figures 1 and 2 and the step-by-step architectural descriptions.", "observation": "The visual communication was effective enough to establish understanding without deep textual immersion, and descriptions were detailed but accessible.", "reasoning": "Clear presentation enables epistemic access, allowing the reviewer to mentally simulate the model's operation and reason about its architecture.", "judgment": "Presentation facilitates understanding and evaluation.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8438336253166199, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8015873432159424}, {"unit_index": 2, "inspected_object": "The experimental evaluation spanning three categories across seven datasets, specifically rotation and OOD tasks.", "observation": "Experiments leverage HPC-MEC circuit properties, such as testing abstraction and reuse of latent actions in the rotation task.", "reasoning": "The alignment between theoretical design principles and specific experimental tests elevates the evaluation from comprehensive to substantively convincing.", "judgment": "Experimental comprehensiveness demonstrates theoretical-experimental alignment.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.830785870552063, "reasoning_key": "merit_recognition", "reasoning_sim": 0.824607789516449}, {"unit_index": 3, "inspected_object": "The evaluation metrics focusing on generated video output quality compared to ground truth.", "observation": "The paper only evaluates visual fidelity against ground truth, which may not sufficiently demonstrate the model's ability to capture world physics.", "reasoning": "Visual fidelity can result from sophisticated interpolation rather than genuine prediction of underlying dynamics; functional validation is needed to prove utility.", "judgment": "Insufficient demonstration of physical understanding due to lack of downstream application evidence.", "valence": "negative", "suggested_improvement": "Include downstream tasks, such as robotic task performance, to provide stronger evidence of the advantages offered by the HPC-MEC module.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8282400369644165, "reasoning_key": "design_justification", "reasoning_sim": 0.8200427889823914}, {"unit_index": 4, "inspected_object": "The discrepancy between generated representations (g^{gen}) and predicted representations (p^{gen}) for two colored apples in Figure 5-B.", "observation": "Generated representations show overlap while predicted representations successfully differentiate the objects.", "reasoning": "The reviewer expects consistency between spaces if the model has learned to differentiate features; the anomaly requires a mechanistic explanation to confirm author understanding.", "judgment": "Unclear mechanism behind the latent space discrepancy.", "valence": "conditional", "suggested_improvement": "Provide an explanation for the observed behavior in the latent space analysis.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8183106780052185, "reasoning_key": "construct_validity", "reasoning_sim": 0.7730294466018677}]}, {"review_id": "06TamW6uTo", "paper_id": "5mMa7mC8Pl", "paper_title": "Functional-level Uncertainty Quantification for Calibrated Fine-tuning on LLMs", "decision": "Reject", "summary": "The reviewer identifies substantial methodological skepticism regarding task scope generalizability, confounding factors in baseline comparisons, theoretical validity under practical constraints, and presentation quality, while acknowledging efficiency advantages.", "units": [{"unit_index": 0, "inspected_object": "Reliance on multiple-choice QA benchmarks (OBQA, ARC, BOOLQ, ClimateQA, MMLU)", "observation": "The paper exclusively uses classification-style multiple-choice tasks.", "reasoning": "Multiple-choice QA has a bounded answer space and single correct answer, making it inherently suitable for calibration in ways that open-ended generation is not. The method's success may be an artifact of this constrained setting rather than a general property of functional-level uncertainty.", "judgment": "Uncertainty about the method's applicability to open-ended settings where factual recall with variable surface forms requires different calibration behaviors.", "valence": "negative", "suggested_improvement": "Extend evaluation to open-ended QA datasets such as TriviaQA.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8419529795646667, "reasoning_key": "design_justification", "reasoning_sim": 0.8183709979057312}, {"unit_index": 1, "inspected_object": "Comparative baselines including untuned LLaMA3.1-8B performance", "observation": "The review requests performance data for the untuned LLaMA3.1-8B model on the five tasks.", "reasoning": "Without the untuned baseline, it is impossible to decompose reported ECE reductions into gains from task learning (fine-tuning) versus gains from the proposed calibration mechanism. The reviewer suspects the improvement might be a side effect of accuracy improvement due to fine-tuning rather than the specific calibration method.", "judgment": "Confounding factor identified: inability to attribute ECE reduction specifically to the calibration method without controlling for task-learning effects.", "valence": "negative", "suggested_improvement": "Report performance of the untuned LLaMA3.1-8B to isolate calibration contribution from task-learning contribution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8233842253684998, "reasoning_key": "design_justification", "reasoning_sim": 0.8611159920692444}, {"unit_index": 2, "inspected_object": "Proposition 3.2 theoretical guarantees regarding non-convex optimization", "observation": "The proposition assumes global optimum over true distribution, but MoE training is non-convex.", "reasoning": "Theoretical guarantees derived under idealized convex/global assumptions may not hold or may degrade significantly when applied to practical non-convex optimization landscapes encountered during MoE training.", "judgment": "Skepticism about the practical relevance of the theoretical guarantee given the gap between idealized assumptions and actual training dynamics.", "valence": "negative", "suggested_improvement": "Clarify how the theoretical guarantee holds or degrades under non-convex optimization conditions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8597937226295471, "reasoning_key": "design_justification", "reasoning_sim": 0.8452976942062378}, {"unit_index": 3, "inspected_object": "Finite sample effects on calibration guarantees", "observation": "The proposition assumes access to the true distribution, but the method trains on finite data.", "reasoning": "Calibration guarantees derived under infinite-data assumptions may collapse or degrade significantly when applied to finite-sample regimes typical in LLM training.", "judgment": "Uncertainty about whether the calibration guarantee degrades gracefully or fails entirely in finite-sample settings.", "valence": "negative", "suggested_improvement": "Provide analysis or empirical evidence on how the guarantee behaves with finite samples.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.857986569404602, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8230835795402527}, {"unit_index": 4, "inspected_object": "Loss conflicts in Eq. 14 (CE, regularization, calibration loss)", "observation": "Eq. 14 combines cross-entropy, regularization, and calibration loss terms.", "reasoning": "These loss components may pull in conflicting directions; for example, CE encourages confident correct predictions while calibration loss may penalize overconfidence even when predictions are correct, potentially undermining the intended behavior.", "judgment": "Concern that conflicting optimization objectives may prevent the method from achieving its stated calibration goals.", "valence": "negative", "suggested_improvement": "Empirically verify that learned FLU ≈ P(correct) via reliability diagrams or calibration curves.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8014974594116211, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8033007979393005}, {"unit_index": 5, "inspected_object": "Presentation consistency and readability", "observation": "Naming inconsistency (", "reasoning": "Surface-level inconsistencies like model name capitalization can signal sloppy production, while poor table formatting (ECE and ACC on separate rows) hinders usability and comparison.", "judgment": "Low presentation quality due to minor but noticeable errors that affect professionalism and ease of use.", "valence": "negative", "suggested_improvement": "Fix naming inconsistencies and reformat Table 1 to place ECE and ACC on the same row per dataset.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8754163980484009, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8257960081100464}]}, {"review_id": "06V1S4PPTV", "paper_id": "fFgzAQAUqs", "paper_title": "SLIM-Brain: A Data- and Training-Efficient Foundation Model for fMRI Data Analysis", "decision": null, "summary": "The reviewer conducts a systematic audit of the paper's claims against evidence, focusing on the lack of empirical support for the 'data-efficiency' claim, the transparency of architectural choices, statistical rigor, terminology precision, and presentation clarity. The review identifies multiple gaps where the evidence does not fully support the assertions made.", "units": [{"unit_index": 0, "inspected_object": "The paper's central claim of 'data-efficiency' and its empirical validation.", "observation": "Experiments use a fixed small dataset size (~1,000 subjects) while baselines scale to significantly larger sizes (32,000–65,000). The authors relegate scaling experiments to future work despite reporting high training efficiency (~1 hour vs. 150 hours for baselines).", "reasoning": "If the model is genuinely data-efficient, scaling experiments should be computationally cheap. The absence of such experiments creates an evidential gap: it is unclear whether performance saturates quickly (supporting the claim), improves with more data (contradicting the claim), or if the current results are cherry-picked. This mismatch between the headline claim and the experimental design undermines the core contribution.", "judgment": "The central claim of data-efficiency is currently unsupported by the presented evidence, casting doubt on the paper's fundamental framing.", "valence": "negative", "suggested_improvement": "Provide at least one additional scaling point (e.g., using 5k or 10k subjects) to demonstrate the shape of the scaling curve.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8093864321708679, "reasoning_key": "design_justification", "reasoning_sim": 0.8661601543426514}, {"unit_index": 1, "inspected_object": "Architectural choice of Hiera-JEPA over Hiera-MAE.", "observation": "Ablation data shows JEPA is superior in Fingerprint (98.5% vs 90.0%) but inferior or comparable in Sex classification (91.1% vs 91.3%) and Age prediction (50.2% vs 52.6%).", "reasoning": "JEPA's advantage is task-specific rather than general. Without a clear rationale beyond single-benchmark performance, the selection of JEPA as the default architecture lacks transparency. The implicit standard is that design decisions not uniformly supported by data require explicit justification.", "judgment": "The architectural justification is weak due to lack of transparency regarding the selection criterion for JEPA versus MAE.", "valence": "negative", "suggested_improvement": "Provide a clearer rationale for choosing JEPA over MAE, or discuss adopting a task-adaptive approach given the mixed ablation results.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7805393934249878, "reasoning_key": "design_justification", "reasoning_sim": 0.8370017409324646}, {"unit_index": 2, "inspected_object": "Statistical rigor of the top-k selection improvement.", "observation": "The improvement from random to top-k selection is 3.2 percentage points (84.5% to 87.7%), but the report lacks confidence intervals or multiple-seed runs.", "reasoning": "Without uncertainty estimates, it is impossible to determine if this improvement is statistically significant or merely noise. The reviewer applies a consistent methodological standard that all performance claims, especially key contributions, must be accompanied by uncertainty estimates.", "judgment": "The statistical significance of the top-k improvement is unverified and potentially weak.", "valence": "negative", "suggested_improvement": "Include confidence intervals or multiple-seed runs to validate the statistical significance of the top-k improvement.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.829003632068634, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.879855215549469}, {"unit_index": 3, "inspected_object": "Terminology precision: 'Multi-scale'.", "observation": "The term 'multi-scale' is used to describe combining global and local features, despite Hiera's hierarchical architecture already providing multi-scale features.", "reasoning": "Using 'multi-scale' conflates the inherent properties of the backbone with the specific contribution of the top-k selection mechanism. This leads to conceptual ambiguity regarding which component drives the performance gains.", "judgment": "The terminology is imprecise and potentially misleading, obscuring the actual contribution.", "valence": "negative", "suggested_improvement": "Adopt the term 'multi-granularity' to accurately distinguish the contribution of top-k selection from the backbone's inherent multi-scale features.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8296768665313721, "reasoning_key": "design_justification", "reasoning_sim": 0.8168874979019165}, {"unit_index": 4, "inspected_object": "Terminology precision: 'Atlas-free'.", "observation": "The method avoids functional ROI parcellation but relies on a pre-defined template-based brain mask to remove background.", "reasoning": "Describing the method as 'atlas-free' blurs a nuanced distinction: it is atlas-free for functional parcellation but template-based for anatomical masking. This lack of operational transparency raises questions about generalizability and preprocessing choices.", "judgment": "The claim of being 'atlas-free' is imprecise and requires clarification to avoid misleading readers about the method's dependencies.", "valence": "negative", "suggested_improvement": "Clarify the distinction between functional atlas-free processing and template-based anatomical masking, and provide details on the template and block determination process.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8069212436676025, "reasoning_key": "design_justification", "reasoning_sim": 0.7965648770332336}, {"unit_index": 5, "inspected_object": "Figure 1 clarity and usability.", "observation": "Figure 1 contains an illegible mix of lines and points, has an unclear task label, and requires head rotation to read.", "reasoning": "The figure fails its primary communicative purpose due to poor visual design, hindering the reader's ability to understand the presented data.", "judgment": "Figure 1 is poorly designed and hinders comprehension.", "valence": "negative", "suggested_improvement": "Redesign Figure 1 to improve legibility, clarify labels, and ensure readability without physical manipulation of the document.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8698629140853882, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8012451529502869}, {"unit_index": 6, "inspected_object": "Consistency between Figure 2-c, Figure 3, and text specifications.", "observation": "Figure 2-c schematic shows 'x1' and '1 key window', while the text states k=8 windows. There is also ambiguity about whether windows are processed sequentially or batched.", "reasoning": "Discrepancies between visual schematics and technical specifications create confusion about the method's operational details, such as feature pooling and processing order.", "judgment": "There is a lack of consistency between the visual narrative and the technical specification.", "valence": "negative", "suggested_improvement": "Resolve discrepancies between figures and text, and clarify the processing order (sequential vs. batched) and feature pooling mechanism.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8966041207313538, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7768820524215698}, {"unit_index": 7, "inspected_object": "Placement of core methodological details.", "observation": "Key details such as the loss equation, lambda value, and k=8 hyperparameter are relegated to the appendix.", "reasoning": "Core methodological choices central to understanding the method should be in the main text. Burying them in supplementary material impedes reproducibility and immediate comprehension.", "judgment": "Important methodological details are inadequately presented in the main text.", "valence": "negative", "suggested_improvement": "Move core methodological details (loss equation, lambda, k) from the appendix to the main text.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8405373096466064, "reasoning_key": "design_justification", "reasoning_sim": 0.7823694944381714}]}, {"review_id": "06bixUxLH6", "paper_id": "FDuN3F1K9X", "paper_title": "InsCal: Calibrated Multi-Source Fully Test-Time Prompt Tuning for Object Detection", "decision": null, "summary": "The reviewer conducts a stress-test of the paper's claims against its experimental design and implementation details, identifying critical flaws in experimental isolation, methodological framing, and presentation clarity.", "units": [{"unit_index": 0, "inspected_object": "Experimental comparison in Table 2 (InsCal vs. FTTA baselines)", "observation": "InsCal is evaluated as a multi-source method using four source models, while competing FTTA baselines are treated as single-source.", "reasoning": "This structural asymmetry makes it difficult to isolate the performance gains of the proposed calibration method from the benefits of simply using more source models and data, violating the norm that a novel component should be isolable from auxiliary resources.", "judgment": "The experimental design does not allow for attributing performance gains to the calibration mechanism rather than the multi-source setup.", "valence": "negative", "suggested_improvement": "Conduct an ablation study or controlled comparison that isolates the effect of the number of source models.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.886184573173523, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8268126845359802}, {"unit_index": 1, "inspected_object": "Internal ablation results (Fig. 9)", "observation": "Adding sources can sometimes harm performance.", "reasoning": "This internal evidence undermines the implicit justification for the asymmetric multi-source comparison by showing that the relationship between source count and performance is not monotonic.", "judgment": "The paper's own data contradicts the assumption that more sources inherently justify the advantage seen in the main comparison.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7688972353935242, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7995011806488037}, {"unit_index": 2, "inspected_object": "Method framing as 'Fully Test-Time Adaptation' (FTTA)", "observation": "The method requires fine-tuning base GDINO on multiple source datasets and uses a target domain text prompt via the TGIA module.", "reasoning": "Standard FTTA premises involve adapting a single off-the-shelf pre-trained model without additional training; requiring domain-specific text descriptions constitutes a form of supervision that may not align with the 'fully test-time' setting where target images arrive without a priori knowledge.", "judgment": "The method's positioning within the literature as 'fully test-time' adaptation is questionable and potentially misleading due to its reliance on auxiliary training and supervision.", "valence": "negative", "suggested_improvement": "Clarify how the method fits within the FTTA definition or adjust the framing to reflect the use of auxiliary information.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "problem_framing", "object_sim": 0.8522742986679077, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.728084921836853}, {"unit_index": 3, "inspected_object": "TGIA module optimization process (Eq. 4)", "observation": "The paper does not specify when the optimization for the augmentation network occurs during inference.", "reasoning": "If the optimization runs for every test image, it introduces significant computational overhead that is not analyzed in the complexity discussion, raising questions about practical viability and resource consumption.", "judgment": "The implementation details are underspecified, creating uncertainty regarding the method's computational cost and reproducibility.", "valence": "negative", "suggested_improvement": "Specify the timing of the TGIA optimization and analyze its computational complexity relative to the adaptation loop.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.795418918132782, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8369212746620178}, {"unit_index": 4, "inspected_object": "Literature review completeness", "observation": "The literature review appears incomplete, omitting six recent papers on retrieval-augmented, Bayesian, continual, and dynamic test-time adaptation.", "reasoning": "The omission suggests the paper's claimed novelty may be situated within a more crowded landscape than acknowledged, failing to position the work against relevant recent advances.", "judgment": "The incomplete literature review undermines confidence in the paper's understanding of the current state-of-the-art.", "valence": "negative", "suggested_improvement": "Include and discuss the cited recent works to properly situate the paper's contributions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8088808655738831, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8045439124107361}, {"unit_index": 5, "inspected_object": "Figure 1(a) baseline selection", "observation": "Zero-shot performance is compared to a fully supervised fine-tuned baseline.", "reasoning": "A supervised upper bound does not effectively motivate an unsupervised adaptation method, as it represents a different operational regime entirely.", "judgment": "The motivational framing provided by this comparison is conceptually weak.", "valence": "negative", "suggested_improvement": "Replace or supplement the supervised fine-tune baseline with a more relevant unsupervised or semi-supervised baseline.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.8897222280502319, "reasoning_key": "fair_comparison", "reasoning_sim": 0.750519335269928}, {"unit_index": 6, "inspected_object": "Figure 3 diagram", "observation": "'Source Model 3' is listed twice in the diagram.", "reasoning": "This is a factual error in visual communication that reduces clarity.", "judgment": "The presentation contains errors that undermine rigor.", "valence": "negative", "suggested_improvement": "Correct the duplication error in Figure 3.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7968183755874634, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8048692941665649}, {"unit_index": 7, "inspected_object": "Figure 1 caption placement", "observation": "The caption for 1(a) is placed awkwardly close to the caption for 1(b).", "reasoning": "Layout issues contribute to confusion and reduce the overall quality of presentation.", "judgment": "The presentation quality is sloppy.", "valence": "negative", "suggested_improvement": "Adjust the layout to separate captions clearly.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7705296874046326, "reasoning_key": "presentation_trust", "reasoning_sim": 0.839920699596405}]}, {"review_id": "06ucUB7V4G", "paper_id": "yFFvGTtw81", "paper_title": "Whom to Trust? Adaptive Collaboration in Personalized Federated Learning", "decision": "Reject", "summary": "The reviewer acts as a meticulous auditor, focusing on completeness and fairness. They identify critical gaps in algorithmic explanation, unfair experimental comparisons due to resource asymmetry, and insufficient reporting of experimental robustness, while noting minor presentation flaws.", "units": [{"unit_index": 0, "inspected_object": "Confidence score calculation methods mentioned in L205 but not explained in detail", "observation": "The paper mentions two different methods to calculate confidence scores but fails to provide detailed exposition of these mechanisms.", "reasoning": "An algorithm's core mechanism must be fully specified for the paper to be reproducible and trustworthy; without explanation, readers cannot understand or verify what the algorithm actually does.", "judgment": "The paper is incomplete regarding its internal components and not ready for publication until this gap is addressed.", "valence": "negative", "suggested_improvement": "Provide a clear explanation of the two confidence score calculation methods and how they differ.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7791388034820557, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8260461091995239}, {"unit_index": 1, "inspected_object": "Comparative fairness of experimental baselines regarding public unlabeled dataset usage", "observation": "FedMosaic utilizes a public unlabeled dataset, whereas the FL baselines do not utilize such data.", "reasoning": "An algorithm should be compared against methods operating under the same resource assumptions; the asymmetry conflates algorithmic advantage with resource advantage, making the comparison unfair to some extent.", "judgment": "The experimental evaluation suffers from comparative unfairness due to asymmetric resource availability between the proposed method and baselines.", "valence": "negative", "suggested_improvement": "Compare FedMosaic against baselines that also leverage public data or knowledge distillation, such as those by Li & Wang (2019), Huang et al. (2022), or Yu et al. (2022).", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8444374799728394, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8419400453567505}, {"unit_index": 2, "inspected_object": "Experimental reporting details including variance, run counts, client counts, and consistency of confidence mechanisms", "observation": "The review notes missing information on standard deviations/confidence intervals, number of runs, inconsistent client counts across tables (15 vs 5), and inconsistent application of confidence mechanisms.", "reasoning": "Robustness and generalizability require stable results across varied conditions; low client counts limit scalability assessment, and inconsistent reporting raises concerns about cherry-picking or incomplete presentation of the design space.", "judgment": "The robustness and generalizability of the reported results are uncertain due to insufficient experimental detail and potential selection bias in reporting.", "valence": "negative", "suggested_improvement": "Report variance metrics, specify number of runs, ensure consistent client counts or explain variations, and apply both confidence mechanisms consistently across all tables.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8659898638725281, "reasoning_key": "robustness_norm", "reasoning_sim": 0.879994809627533}, {"unit_index": 3, "inspected_object": "Presentation mechanics including notation, formatting, and narrative coherence", "observation": "The reviewer identifies undefined symbols ($L_t^i$, diag, $C$), a missing footnote, unnecessary absolute values, typos, LaTeX quotation mark conventions, and questions the rationale behind the name 'FedMosaic'.", "reasoning": "Precision in presentation and typesetting reflects careful scholarship; undefined notation hinders readability, and the naming query suggests a desire for the paper to be more self-explanatory regarding its narrative structure.", "judgment": "The presentation is competent but not polished, containing enough minor errors and ambiguities to prevent a higher quality rating.", "valence": "negative", "suggested_improvement": "Define all symbols, fix typos and formatting issues, add missing footnotes, and clarify the rationale behind the algorithm's name to improve narrative coherence.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8828601241111755, "reasoning_key": "presentation_trust", "reasoning_sim": 0.865379273891449}]}, {"review_id": "074NWJMyhY", "paper_id": "UkwK6lrHo2", "paper_title": "Simultaneously Perturbed Optimistic Gradient Methods for Payoff-Based Learning in Games", "decision": "Reject", "summary": "The reviewer critically examines the paper's theoretical boundaries, parameter dependencies, and empirical validation, highlighting gaps in literature contextualization and proof transparency while suggesting concrete technical extensions.", "units": [{"unit_index": 0, "inspected_object": "Literature placement of the $n^{-2/3}$ convergence rate relative to known lower bounds", "observation": "The reviewer notes that Fiegel et al. appear to show an $n^{-1/4}$ lower bound, creating a potential gap in how the paper positions its faster rate.", "reasoning": "The paper claims a rate faster than sharpest known rates; without explicit engagement with relevant lower bounds (like Fiegel et al.), the positioning is incomplete and potentially misleading regarding novelty or tightness.", "judgment": "The literature context is insufficiently established due to missing comparison with specific prior lower bounds.", "valence": "negative", "suggested_improvement": "Explicitly address the relationship between the claimed $n^{-2/3}$ rate and known lower bounds such as those by Fiegel et al. and Cai et al.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8553593158721924, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7520977854728699}, {"unit_index": 1, "inspected_object": "Extension of results to constrained action spaces", "observation": "The reviewer questions whether the theoretical results extend beyond the unconstrained setting emphasized in the abstract.", "reasoning": "If the unconstrained assumption is essential rather than merely convenient, the scope of the contribution is limited; clarifying this boundary condition tests the robustness of the theoretical claim.", "judgment": "The scope limitation regarding constrained settings remains unexamined.", "valence": "conditional", "suggested_improvement": "Discuss whether results extend to constrained action spaces or explain why the unconstrained assumption is necessary.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8336319923400879, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8272709846496582}, {"unit_index": 2, "inspected_object": "Parameter knowledge requirements for input gamma", "observation": "The input parameter gamma appears to require knowledge of game-dependent constants L, G, and possibly N.", "reasoning": "Algorithms requiring problem-specific constants have limited applicability; the lack of discussion on robustness or adaptive tuning suggests a practical barrier to use.", "judgment": "The algorithm's practical utility is hindered by unknown parameter dependencies.", "valence": "negative", "suggested_improvement": "Provide a discussion of robustness or adaptive tuning schemes to mitigate dependence on specific constants.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8460404872894287, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8085283041000366}, {"unit_index": 3, "inspected_object": "Applicability to the merely monotone case (mu=0)", "observation": "The reviewer identifies a technical gap where the proof relies on strong monotonicity (mu>0) but suspects the argument might hold for mu=0.", "reasoning": "If the strong monotonicity assumption is not strictly necessary, the result could be generalized via techniques like player-wise regularization and annealing, increasing the theoretical contribution's breadth.", "judgment": "The restriction to strongly monotone games may be unnecessarily conservative.", "valence": "conditional", "suggested_improvement": "Investigate if the proof extends to the merely monotone case, potentially using a regularizer-annealing strategy.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "theory", "object_sim": 0.84818035364151, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7974037528038025}, {"unit_index": 4, "inspected_object": "Experimental scope and comparative validation", "observation": "Experiments focus on simple cases and lack comparison with existing payoff-based methods like Fiegel et al. in two-player zero-sum bilinear games.", "reasoning": "Empirical claims are strengthened by benchmarking against alternatives; isolation from standard baselines reduces confidence in the practical advantage of the proposed method.", "judgment": "The experimental validation is insufficiently comprehensive compared to state-of-the-art benchmarks.", "valence": "negative", "suggested_improvement": "Include comparisons with other payoff-based methods, specifically in two-player zero-sum bilinear games.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8662130236625671, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8276534676551819}, {"unit_index": 5, "inspected_object": "Relationship to Bandit Convex Optimization (BCO) lower bounds", "observation": "The reviewer queries how the paper's rate reconciles with Shamir's (2013) Omega(n^-1/2) lower bound for BCO, noting the phrasing 'break' the barrier.", "reasoning": "If the paper's setting is a special case of BCO, the claimed rate would contradict known lower bounds; explicit reconciliation of modeling differences is required to validate the rate claim.", "judgment": "There is a potential tension between the rate claim and established lower bounds in related settings that requires clarification.", "valence": "uncertain", "suggested_improvement": "Clarify the modeling and algorithmic differences between the paper's setting and BCO to reconcile the rate with known lower bounds.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8154575228691101, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7502182126045227}, {"unit_index": 6, "inspected_object": "Proof transparency regarding the mu>0 assumption", "observation": "The reviewer notes that Proposition 4.4 superficially looks like it might go through with mu=0 and asks where exactly the positive monotonicity is used.", "reasoning": "Transparent proofs should clearly delineate where assumptions are critical; ambiguity here suggests either a loose proof or an unnecessary constraint.", "judgment": "The necessity of the strong monotonicity assumption is not sufficiently justified in the proof structure.", "valence": "conditional", "suggested_improvement": "Explicitly identify where in the proof the mu>0 assumption is utilized.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8406565189361572, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7799938321113586}, {"unit_index": 7, "inspected_object": "Presentation clarity and notation", "observation": "The reviewer identified undefined notation V at line 105 of the paper.", "reasoning": "Undefined notation indicates a lapse in presentation quality that hinders readability, consistent with a moderate presentation score.", "judgment": "The presentation contains minor errors that reduce clarity.", "valence": "negative", "suggested_improvement": "Define all notation, specifically the symbol V.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8926987648010254, "reasoning_key": "presentation_trust", "reasoning_sim": 0.829073965549469}]}, {"review_id": "07D6LlBRUI", "paper_id": "j1J253hU2k", "paper_title": "Pareto-Guided Regional Sampling for Adaptive Collocation in Physics-Informed Neural Networks", "decision": null, "summary": "The reviewer acts as a gatekeeper of novelty, judging the method as an unproven combination of known tools due to a lack of comparative positioning, ablation evidence, and interpretability, rather than critiquing its internal coherence or mathematical validity.", "units": [{"unit_index": 0, "inspected_object": "The method's technical design composition (K-means clustering, NSGA-II algorithm, and weight sampling technique)", "observation": "The reviewer identifies the proposed method as a recognizable combination of pre-existing algorithms rather than a new primitive.", "reasoning": "The operative standard is that a contribution requires either a new algorithmic primitive or emergent behavior from integration; because the components are identifiable as known tools, the novelty of the combination itself is not demonstrated.", "judgment": "It is hard to identify the novelty of this paper's technical design.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7810639142990112, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8315481543540955}, {"unit_index": 1, "inspected_object": "Literature positioning and comparison class", "observation": "The paper lacks comparisons against three specific relevant works: a causality-respecting training method (CMAME 2024), RoPINN (NeurIPS 2024), and an NTK-perspective analysis (JCP 2022).", "reasoning": "A paper claiming to address region optimization and loss reweighting must position itself against the most recent and relevant work in that neighborhood to demonstrate its distinct contribution and theoretical grounding.", "judgment": "The current positioning is incomplete and fails to situate the work within the state of the art.", "valence": "negative", "suggested_improvement": "Compare against the identified baselines: the causality-respecting method, RoPINN, and engage with the NTK theoretical framework.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8516996502876282, "reasoning_key": "novelty_standard", "reasoning_sim": 0.804104745388031}, {"unit_index": 2, "inspected_object": "Component necessity and efficiency of the combinatorial approach", "observation": "The reviewer suspects some components may be decorative rather than load-bearing and notes potential efficiency costs (K-means time, GPU memory).", "reasoning": "If the method is a combination of known parts, ablation studies are required to prove each part's necessity, and efficiency analysis is needed to justify the computational complexity against performance gains.", "judgment": "The necessity of the combined components and the cost-effectiveness of the method are unproven.", "valence": "negative", "suggested_improvement": "Perform ablation studies removing components like weighting-informed sampling to test necessity, and provide an efficiency analysis regarding K-means time cost and GPU memory.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8324831128120422, "reasoning_key": "cost_benefit", "reasoning_sim": 0.760611891746521}, {"unit_index": 3, "inspected_object": "Visual interpretability of the method's behavior", "observation": "The paper lacks visualizations of subregion splitting and adaptive sampling during training.", "reasoning": "Interpretability is valued as a component of scientific contribution; methods that cannot be visually inspected are harder to trust, debug, or extend, reducing their communicability.", "judgment": "The method's legibility and trustworthiness are diminished by the lack of behavioral visualization.", "valence": "negative", "suggested_improvement": "Add visualizations showing subregion splitting and adaptive sampling during training.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8302941918373108, "reasoning_key": "presentation_trust", "reasoning_sim": 0.810302734375}]}, {"review_id": "07TzSqJMQ1", "paper_id": "lM1lpQdjAf", "paper_title": "TimeLAVA: Learning-Agnostic Valuation for Time Series Data", "decision": "Reject", "summary": "The reviewer evaluates the paper as technically sound but critically lacking in empirical characterization. Key judgments center on the need for precise terminology, comprehensive parameter sensitivity analysis, and validation of acknowledged limitations like reference-set dependence.", "units": [{"unit_index": 0, "inspected_object": "Terminology of the proposed distance measure ($W_{SW}$)", "observation": "The paper refers to its central measure as a 'metric', but the reviewer identifies that the triangular inequality may fail.", "reasoning": "A mathematical object failing the triangle inequality is technically a dissimilarity measure, not a metric; using precise terminology is a disciplinary norm for credibility and correct interpretation.", "judgment": "The claim of being a 'metric' is an overclaim requiring correction, though the issue is considered marginal.", "valence": "negative", "suggested_improvement": "Correct the terminology to describe the measure as a dissimilarity rather than a metric.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8092389106750488, "reasoning_key": "construct_validity", "reasoning_sim": 0.7663154006004333}, {"unit_index": 1, "inspected_object": "Sensitivity analysis of core method parameters (regularizer $k$ and balance $c$)", "observation": "The study examines window length and stride but omits sensitivity analysis for the regularizer $k$ and balance $c$, which are integral to the method's definition.", "reasoning": "Method papers must characterize their own internal hyperparameters to demonstrate robustness; omitting these leaves the method's behavior under parameter variation unverified.", "judgment": "The lack of parameter sensitivity analysis represents a gap in methodological completeness and self-knowledge.", "valence": "negative", "suggested_improvement": "Conduct sensitivity analysis for regularizer $k$ and balance $c$ in both supervised and unsupervised settings.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8601710200309753, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8529141545295715}, {"unit_index": 2, "inspected_object": "Reference set sensitivity in supervised settings", "observation": "The paper acknowledges sensitivity to the reference set but does not quantify this sensitivity or subject it to nested cross-validation.", "reasoning": "When a limitation is self-acknowledged, empirical evidence (such as nested cross-validation) is required to demonstrate that the limitation does not undermine the results or reliability.", "judgment": "The failure to empirically address a stated limitation constitutes a weakness in empirical self-consistency.", "valence": "negative", "suggested_improvement": "Apply nested cross-validation to quantify how much valuation outcomes depend on the choice of reference sequences.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8444826006889343, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7676261067390442}, {"unit_index": 3, "inspected_object": "Scalability of DWT for large window lengths", "observation": "The reviewer questions whether the Discrete Wavelet Transform (DWT) approach faces a curse of dimensionality or computational degradation with large window lengths.", "reasoning": "Practical viability depends on the method's performance in realistic, large-scale applications; without scalability details, the method's deployability remains uncertain.", "judgment": "The method's practical viability is currently unverified due to unclear scalability characteristics.", "valence": "uncertain", "suggested_improvement": "Provide details on computational complexity and performance when scaling to larger time series/window lengths.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8119569420814514, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7976230382919312}, {"unit_index": 4, "inspected_object": "Clarity of Figure 3 regarding anomaly types", "observation": "Local contextual, global contextual, and point anomalies appear visually similar in Figure 3, making distinctions hard to perceive.", "reasoning": "Visual evidence must clearly demonstrate the method's discriminative capabilities; if the figure fails to show differences, reader trust in the quantitative results is compromised.", "judgment": "The presentation of visual evidence is insufficient to verify the claimed distinctions between anomaly types.", "valence": "negative", "suggested_improvement": "Explicitly explain the differences between local contextual, global contextual, and point anomalies to improve communicative clarity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7834027409553528, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8149028420448303}, {"unit_index": 5, "inspected_object": "Formatting and structural presentation", "observation": "The review notes colored citations/references and an incorrect section reference at line 197.", "reasoning": "Minor formatting errors and citation inaccuracies detract from the overall polish and professionalism of the manuscript.", "judgment": "The presentation contains minor mechanical flaws that reduce polish.", "valence": "negative", "suggested_improvement": "Fix colored citations/references and correct the wrong section reference at line 197.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8968898057937622, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8348675966262817}]}, {"review_id": "07aTLosbb2", "paper_id": "wbttgzp7MT", "paper_title": "EmotionThinker: Prosody-Aware Reinforcement Learning for Explainable Speech Emotion Reasoning", "decision": "Accept (Oral)", "summary": "The reviewer evaluates the paper primarily as a constructive methodologist, focusing on data provenance, reward model calibration, and expository clarity. They identify specific gaps in validation and specification that threaten construct validity and evaluability, while withholding judgment on empirical results until these foundational issues are addressed.", "units": [{"unit_index": 0, "inspected_object": "Data construction pipeline and modality grounding", "observation": "The reasoning trace data is constructed with GPT4o without the actual speech input.", "reasoning": "If traces are generated from text transcripts rather than audio, the dataset may not be grounded in acoustic features, creating a modality gap that undermines the core claim of prosody-aware reasoning and risks systematic misalignment between the data and the model's intended capability.", "judgment": "The dataset's validity for training prosody-aware reasoning is threatened by potential lack of acoustic grounding.", "valence": "negative", "suggested_improvement": "Compare GPT-4o-based and human-based scoring distributions to verify alignment and check for systematic bias.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8457955718040466, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.764811635017395}, {"unit_index": 1, "inspected_object": "Reward model calibration and reliability", "observation": "There is little discussion or quantitative validation of the reward model's calibration, and no direct comparison of GPT-annotated versus human-annotated reward score distributions.", "reasoning": "The RL framework relies on the reward signal as its optimization target; if this signal is miscalibrated or unvalidated against human preferences, the resulting policy may optimize for incorrect objectives, making the entire GRPO-PTR framework suspect.", "judgment": "The reliability of the RL optimization is uncertain due to unvalidated reward signals.", "valence": "negative", "suggested_improvement": "Provide annotation accuracy statistics across emotion categories and speaker groups to demonstrate distributional robustness.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8477450609207153, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7889901995658875}, {"unit_index": 2, "inspected_object": "Expository clarity of the RL mechanism", "observation": "The description of the RL pipeline lacks improved logit flow regarding progressive reward scheduling and trustworthiness weighting.", "reasoning": "A contribution cannot be fairly evaluated or credited if its mechanics are insufficiently specified; clear exposition is a prerequisite for assessing the novelty and correctness of the proposed method.", "judgment": "The paper's contribution to the LLM RL community is currently difficult to fully evaluate due to unclear mechanism specification.", "valence": "negative", "suggested_improvement": "Improve the explanation of logit flow to clarify how progressive reward scheduling and trustworthiness weighting operate in the training loop.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "theory", "object_sim": 0.8407830595970154, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8320177793502808}, {"unit_index": 3, "inspected_object": "Overall empirical engagement and experimental results", "observation": "The reviewer does not engage with the paper's experimental results, benchmark comparisons, or baseline selections, treating empirical claims as plausible but under-validated until foundations are checked.", "reasoning": "The reviewer adopts a conditional stance where experimental claims rest on the validity of the dataset and reward model; silence on results indicates withholding judgment pending resolution of these foundational concerns rather than an endorsement of the results themselves.", "judgment": "The work is considered solid but not exceptional, with a moderate endorsement contingent on addressing validation gaps.", "valence": "conditional", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8583238124847412, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7570153474807739}, {"unit_index": 4, "inspected_object": "Minor presentation errors", "observation": "Presence of minor typos in the manuscript.", "reasoning": "Typos indicate a lack of final polish but do not impact the scientific substance or validity of the contributions.", "judgment": "The paper is otherwise well-presented despite minor typographical issues.", "valence": "positive", "suggested_improvement": "Correct minor typos.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8064186573028564, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7605899572372437}]}, {"review_id": "07ecvRO7th", "paper_id": "fwTRpXMsxB", "paper_title": "GT-Space: Enhancing Heterogeneous Collaborative Perception with Ground Truth Feature Space", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper positively for its novel GT-Space concept, scalable inference architecture, and decoupled performance, while raising major concerns about training complexity contradicting scalability claims and insufficient visual evidence for alignment.", "units": [{"unit_index": 0, "inspected_object": "GT-Space conceptual novelty", "observation": "The reviewer finds the GT-Space idea well-motivated and novel.", "reasoning": "The concept provides a unified reference that serves as a core contribution to the field's conceptual toolkit.", "judgment": "Positive evaluation of the paper's conceptual contribution.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.824094295501709, "reasoning_key": "merit_recognition", "reasoning_sim": 0.6912497282028198}, {"unit_index": 1, "inspected_object": "System scalability and flexibility (inference-time)", "observation": "The system uses frozen encoders, a single lightweight projector per agent, and no pairwise interactions.", "reasoning": "These architectural properties enable lightweight integration of new agents without pairwise complexity.", "judgment": "Positive evaluation of the system's deployment properties.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8769407272338867, "reasoning_key": "design_justification", "reasoning_sim": 0.7570421695709229}, {"unit_index": 2, "inspected_object": "Decoupling of performance from ego-agent capability", "observation": "GT-Space allows strong references to compensate for weak agents, addressing prior bottlenecks.", "reasoning": "This design choice mitigates the failure mode where fused results are limited by the ego agent's perception quality.", "judgment": "Positive evaluation of the system's robustness and design efficacy.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8299164772033691, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7861840724945068}, {"unit_index": 3, "inspected_object": "Empirical SOTA results", "observation": "The paper achieves state-of-the-art results on three datasets.", "reasoning": "Performance metrics confirm the effectiveness of the proposed framework.", "judgment": "Acknowledgment of empirical success.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8295019865036011, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7672857046127319}, {"unit_index": 4, "inspected_object": "Training complexity of the fusion network", "observation": "The contrastive loss is applied across all pairs of modalities, resulting in O(M²) complexity.", "reasoning": "This quadratic scaling during training undermines the paper's central scalability claim, which focuses on inference-time integration.", "judgment": "Major concern regarding consistency with scalability narrative.", "valence": "negative", "suggested_improvement": "Address whether the scalability claim covers training or only inference, or provide a defense/alternative for the O(M²) cost.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8038086295127869, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7507258653640747}, {"unit_index": 5, "inspected_object": "Quality of Figure 5 (visual evidence for alignment)", "observation": "Figure 5 shows stronger activations in fused feature maps, which is a consequence rather than a demonstration of alignment.", "reasoning": "Current visual evidence does not directly demonstrate that features occupy a common, aligned space; it could be explained by amplification without true alignment.", "judgment": "Evidentiary standard not met for the central mechanism.", "valence": "negative", "suggested_improvement": "Visualize F_GT itself or compare projected camera and LiDAR features side-by-side to demonstrate alignment quality.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8574756979942322, "reasoning_key": "design_justification", "reasoning_sim": 0.7609333992004395}]}, {"review_id": "07makAgdsc", "paper_id": "USEpVtH8qV", "paper_title": "JIONE: an Approach for Merging Large Language Models via Teacher–Student Prediction Refinement", "decision": "Reject", "summary": "The reviewer performs a validity audit, primarily challenging the conceptual definition of 'merging' and the experimental rigor of the attribution and statistical claims. The logic cascades from the definitional mismatch to specific failures in baseline selection, metric appropriateness, and theoretical grounding.", "units": [{"unit_index": 0, "inspected_object": "Conceptual framing of the JIONE method as 'merging'", "observation": "The reviewer observes that JIONE operates at the output level via sequential teacher–student refinement and produces no new model artifact, requiring both models for inference.", "reasoning": "The operative standard is that 'merging' in machine learning entails producing a single, independently deployable model artifact. Since JIONE functions as function composition rather than true merging, the paper's core claim is a fundamental category error.", "judgment": "The method is mislabeled; the conceptual framing is invalid under established definitions of model merging.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8483994603157043, "reasoning_key": "design_justification", "reasoning_sim": 0.8164020776748657}, {"unit_index": 1, "inspected_object": "Attribution of performance improvements", "observation": "The review notes the absence of a control evaluating the teacher model in isolation to distinguish its contribution from cross-model fusion.", "reasoning": "Without isolating the teacher's effect, it is impossible to attribute reported improvements to the proposed mechanism rather than to the superior capability of the teacher model alone, rendering the results uninterpretable.", "judgment": "The experimental design fails to support causal claims about the method's efficacy.", "valence": "negative", "suggested_improvement": "Include an evaluation of the teacher model in isolation to control for its individual contribution.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8341713547706604, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8004961609840393}, {"unit_index": 2, "inspected_object": "Robustness and symmetry of the method", "observation": "The reviewer identifies a lack of testing regarding role reversal (swapping teacher/student) or architecture swapping.", "reasoning": "A robust merging method should demonstrate consistent behavior or predictable outcomes under role reversal, ensuring results are not artifacts of specific model pairings.", "judgment": "The method's robustness and generalizability are unverified.", "valence": "conditional", "suggested_improvement": "Conduct experiments reversing teacher/student roles or swapping architectures to verify consistency.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "theory", "object_sim": 0.877242386341095, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8078699111938477}, {"unit_index": 3, "inspected_object": "Conflict resolution mechanism", "observation": "There is no formal reconciliation rule specified for when the teacher and student models produce contradictory predictions.", "reasoning": "A method operating at the output level must explicitly define its decision procedure (e.g., confidence weighting, priority rules) to resolve conflicts between component models.", "judgment": "The method lacks a necessary theoretical specification for handling disagreement.", "valence": "negative", "suggested_improvement": "Explicitly specify the conflict resolution mechanism used by the method.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7456130385398865, "reasoning_key": "robustness_norm", "reasoning_sim": 0.709274172782898}, {"unit_index": 4, "inspected_object": "Mathematical foundation section", "observation": "The section contains only two equations and lacks formal analysis of convergence, bias correction, or uncertainty propagation.", "reasoning": "Methods contributions are expected to provide theoretical scaffolding beyond empirical demonstration to justify their novelty and stability.", "judgment": "The mathematical presentation is superficial and insufficient for a methods contribution.", "valence": "negative", "suggested_improvement": "Expand the mathematical foundation to include formal analysis of convergence, bias, and uncertainty.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7796271443367004, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8128345012664795}, {"unit_index": 5, "inspected_object": "Statistical rigor and reproducibility", "observation": "The review finds an absence of confidence intervals, variance analysis, and multiple runs, alongside small sample sizes (e.g., SST-2 with 1,000 samples).", "reasoning": "Empirical claims require uncertainty quantification to distinguish reported improvements from noise and ensure reproducibility.", "judgment": "The statistical evidence is weak and does not reliably support the claimed improvements.", "valence": "negative", "suggested_improvement": "Report confidence intervals, perform variance analysis, and run multiple trials for all benchmarks.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8630856275558472, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8302391171455383}, {"unit_index": 6, "inspected_object": "Scale of models and terminology", "observation": "The paper uses small models (Phi-3-mini, Qwen1.5-1.8B) but refers to them as 'LLMs', running on free-tier hardware.", "reasoning": "Claims about 'LLM merging' should be demonstrated on models recognized by the community as large-scale (>70B); using small models undermines the scope and relevance of the claims.", "judgment": "The experimental scale does not match the terminology or implied contribution.", "valence": "negative", "suggested_improvement": "Test on larger-scale models or adjust terminology to accurately reflect the model sizes used.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8031260967254639, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7882961630821228}, {"unit_index": 7, "inspected_object": "Metric appropriateness for code generation", "observation": "The review questions the use of ROUGE for MBPP, a code-generation task.", "reasoning": "Surface-level text overlap metrics like ROUGE are inadequate for assessing functional correctness in code generation; exact match or pass@k are more appropriate standards.", "judgment": "The evaluation metric is inappropriate for the task being measured.", "valence": "negative", "suggested_improvement": "Replace ROUGE with exact match or pass@k for code generation benchmarks.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8240736126899719, "reasoning_key": "construct_validity", "reasoning_sim": 0.7742437124252319}, {"unit_index": 8, "inspected_object": "Baseline selection", "observation": "The review flags SLERP and Task Arithmetic as trivial for heterogeneous merging and notes the omission of recent techniques like MergeMoE, Rebasin, and ModelSoups++.", "reasoning": "A paper claiming to address heterogeneous merging must compare against state-of-the-art methods designed for that setting, not just classical weight-space approaches.", "judgment": "The baseline selection is inadequate and fails to position the work correctly within the current literature.", "valence": "negative", "suggested_improvement": "Include comparisons with recent heterogeneous merging techniques such as MergeMoE, Rebasin, and ModelSoups++.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.884096622467041, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7478929162025452}, {"unit_index": 9, "inspected_object": "Computational cost claims", "observation": "Execution time comparisons exist but lack FLOPs, memory footprint, or per-query cost quantification.", "reasoning": "Claims about efficiency or cost must be supported by comprehensive computational metrics, not just wall-clock time.", "judgment": "The computational analysis is insufficient to support efficiency claims.", "valence": "negative", "suggested_improvement": "Quantify computational costs using FLOPs, memory footprint, and per-query latency.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8686148524284363, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7696358561515808}]}, {"review_id": "07vGh1w4Ub", "paper_id": "whu2doKCti", "paper_title": "Binary Neural Network for Hyperspectral pansharpening", "decision": "Reject", "summary": "The reviewer affirms the technical strength and novelty of the method but withholds acceptance due to significant gaps in presentation clarity, evidentiary completeness (ablations and comparisons), and reproducibility norms.", "units": [{"unit_index": 0, "inspected_object": "Figure 3 architectural clarity", "observation": "Module names are not arranged horizontally for readability; the text mentions an 'encoder' not clearly visible in the figure; the paper does not explicitly indicate which convolutions are standard versus binary.", "reasoning": "The reviewer requires legibility to map textual descriptions onto visual representations, specifically needing to see where binarization occurs to evaluate if the method is genuinely binary. This reflects a norm of architectural transparency.", "judgment": "The figure presents friction and fails to adequately guide the reader through the architecture's core mechanism.", "valence": "negative", "suggested_improvement": "Arrange module names horizontally, clarify the encoder's presence, and mark binary vs. standard convolutions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8412948250770569, "reasoning_key": "design_justification", "reasoning_sim": 0.8216443657875061}, {"unit_index": 1, "inspected_object": "Novelty framing and literature context", "observation": "The reviewer credits the first BNN application to hyperspectral fusion but notes the paper lacks a brief review of existing binary networks and their limitations.", "reasoning": "While novelty is accepted, the reviewer doubts whether the paper makes the contribution legible to readers unfamiliar with BNN literature, requiring more work to demonstrate novelty rather than just claiming it.", "judgment": "The novelty claim is real but not sufficiently communicated or contextualized within the broader field.", "valence": "conditional", "suggested_improvement": "Briefly review existing binary networks and highlight their limitations to clarify the novelty.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8565220236778259, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8338508605957031}, {"unit_index": 2, "inspected_object": "Ablation study on HS-BiNet components", "observation": "There is no ablation on HS-BiNet components (Edge Injector, Multi-scale Extractor, Decoder Enhancement).", "reasoning": "Without ablations, the reviewer cannot distinguish whether performance gains come from the core innovation (ATISTE) or incidental design choices, violating the norm of component-level attribution.", "judgment": "The evidence base is incomplete regarding the causal contribution of specific architectural components.", "valence": "negative", "suggested_improvement": "Add ablations for Edge Injector, Multi-scale Extractor, and Decoder Enhancement.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7459790110588074, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8153166770935059}, {"unit_index": 3, "inspected_object": "Efficiency comparisons against full-precision models", "observation": "Efficiency comparisons are only provided among binary models, lacking comparison against full-precision baselines.", "reasoning": "Claims about suitability for edge deployment require establishing the practical value proposition by showing the efficiency gap relative to the natural alternative (full-precision), adhering to the norm of fair comparison.", "judgment": "The practical value proposition is not adequately established due to missing comparative context.", "valence": "negative", "suggested_improvement": "Compare efficiency against full-precision models to establish the trade-off between efficiency and accuracy.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.89195317029953, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7826526165008545}, {"unit_index": 4, "inspected_object": "Code and model release", "observation": "No code or model release is provided.", "reasoning": "The reviewer assumes that research contributions include means for others to build on them, viewing reproducibility as part of community contribution.", "judgment": "The paper is not yet a complete contribution to the research ecosystem.", "valence": "negative", "suggested_improvement": "Release code and models to enable reproduction and further development.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "reproducibility", "object_sim": 0.7952313423156738, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7508442997932434}]}, {"review_id": "06w6KOnA6R", "paper_id": "LNJ05tNHLs", "paper_title": "From Lab to Line: Deployment-Aware NMR–Text Expert Routing for Real-Time Apple Moldy Core Disease Screening and Explanation", "decision": "Reject", "summary": "The reviewer critically evaluates the manuscript primarily through the lens of communicative clarity, statistical rigor, and comparative justification. Key concerns include undefined terminology, insufficient motivation for multi-modality given strong image-only baselines, lack of robust statistical testing, and questionable necessity of the new metric. While the application domain is valued, the execution and evidentiary support are deemed insufficient.", "units": [{"unit_index": 0, "inspected_object": "Abstract and framing structure", "observation": "The abstract lists results without summarizing or contextualizing the work.", "reasoning": "An abstract must perform a rhetorical function of summarization and context; failure to do so indicates poor communicative structure and impedes reader comprehension.", "judgment": "The presentation is opaque and fails to meet standards for reader accessibility.", "valence": "negative", "suggested_improvement": "Rewrite the abstract to summarize and contextualize findings rather than listing results.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8833519220352173, "reasoning_key": "presentation_trust", "reasoning_sim": 0.822618842124939}, {"unit_index": 1, "inspected_object": "Terminology and acronyms", "observation": "Undefined acronyms, specifically 'AppleNMR-MM', appear in the text.", "reasoning": "Undefined terminology creates barriers to comprehension, preventing evaluation of the work's technical content.", "judgment": "The writing lacks accountability to the reader, contributing to low soundness perception.", "valence": "negative", "suggested_improvement": "Define all acronyms upon first use.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.717702329158783, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8349118232727051}, {"unit_index": 2, "inspected_object": "Motivation for multi-modal approach", "observation": "Prior works leveraging images only achieve >96% accuracy, yet the paper proposes a complex multi-modal method without sufficient justification.", "reasoning": "If simpler methods already perform well, the authors must explicitly justify why the added complexity of multi-modality is necessary for this task.", "judgment": "The problem motivation and design choice are unconvincing due to lack of comparative evidence against strong baselines.", "valence": "negative", "suggested_improvement": "Provide a clear justification for the necessity of the multi-modal approach given existing high-performing image-only methods.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8417147397994995, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7857480645179749}, {"unit_index": 3, "inspected_object": "Experimental baselines", "observation": "High-performing prior works cited in the related works section are not included as experimental baselines.", "reasoning": "Scientific rigor requires positioning new methods against the strongest available comparators to validate claimed improvements.", "judgment": "The experimental evidence is incomplete and undermines confidence in the reported performance gains.", "valence": "negative", "suggested_improvement": "Include the cited high-performance image-only methods as baselines in the experiments.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.899196982383728, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8251634836196899}, {"unit_index": 4, "inspected_object": "Statistical robustness and seeding", "observation": "The reproducibility statement mentions seeding but does not clarify if variation across runs is due to CUDA non-determinism or random initialization.", "reasoning": "Results based on a single seed or ambiguous seeding protocols may reflect artifacts rather than genuine model behavior, failing norms of statistical rigor.", "judgment": "The statistical significance and robustness of the results are doubtful.", "valence": "negative", "suggested_improvement": "Report results across multiple random seeds and clarify the source of variance to ensure statistical significance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8620728254318237, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8340072631835938}, {"unit_index": 5, "inspected_object": "TAAPM metric contribution", "observation": "The proposed TAAPM metric unifies components that are already individually accessible.", "reasoning": "A new metric must demonstrate incremental value or practical advantage over existing individual components to justify its introduction.", "judgment": "The contribution of the new metric is skeptical due to apparent redundancy.", "valence": "negative", "suggested_improvement": "Articulate the specific intellectual or practical advantage of the unified metric over existing individual components.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8194886445999146, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7624986171722412}, {"unit_index": 6, "inspected_object": "Dataset size and overfitting risk", "observation": "The dataset is small (n=237), raising concerns about overfitting in the context of modern machine learning research.", "reasoning": "Small datasets pose known risks of overfitting; without mitigation strategies or validation, the reliability of experimental claims is compromised.", "judgment": "The validity of the reported results is fundamentally challenged by the small sample size.", "valence": "negative", "suggested_improvement": "Address potential overfitting through cross-validation, data augmentation, or discussion of generalization limits.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8072631359100342, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7981724143028259}, {"unit_index": 7, "inspected_object": "Human expert baseline comparison", "observation": "The paper does not compare model performance against human experts.", "reasoning": "Contextualizing model performance against human expertise is necessary to establish the practical utility and relevance bar for the system.", "judgment": "The evaluation lacks a critical reference point for real-world applicability.", "valence": "negative", "suggested_improvement": "Include a comparison with human expert performance to contextualize the model's capabilities.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8370668292045593, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.782565176486969}, {"unit_index": 8, "inspected_object": "Application domain novelty", "observation": "The application area is identified as novel and very important.", "reasoning": "While the problem space is valuable, this strength is distinct from the quality of the proposed solution or its evaluation.", "judgment": "The problem is significant, but this does not compensate for weaknesses in method justification and presentation.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8733468651771545, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7947725653648376}]}, {"review_id": "07w6CClQvi", "paper_id": "FbssShlI4N", "paper_title": "FALCON: Few-step Accurate Likelihoods for Continuous Flows", "decision": "Accept (Oral)", "summary": "The reviewer praises the technical soundness and experimental execution but blocks acceptance due to unresolved novelty concerns relative to prior work, while also requesting clarifications on figures, baselines, and data interpretation.", "units": [{"unit_index": 0, "inspected_object": "Core Technical Proposal and Experimental Execution", "observation": "The reviewer identifies the method as an extension of CNFs using a regularization term for approximate invertibility to enable density estimation, supported by extensive experiments on peptide systems.", "reasoning": "The reviewer finds the technical execution sound, the background sufficient, and the experimental evidence convincing, leading to a positive assessment of the work's quality independent of its novelty.", "judgment": "The paper is well-executed, well-presented, and experimentally convincing.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8259729146957397, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8659622669219971}, {"unit_index": 1, "inspected_object": "Novelty relative to Rehmann et al., 2025", "observation": "A closely related prior work exists with a similar objective (combining CNFs and discrete-time NFs via approximate invertibility) and a similar, slightly more extensive set of experiments.", "reasoning": "The relationship between the two works is unclear without explicit engagement; this uncertainty prevents the reviewer from confirming that the contribution is sufficiently novel to warrant acceptance.", "judgment": "The paper's novelty is uncertain due to lack of comparison with prior art, acting as a blocking concern for acceptance.", "valence": "negative", "suggested_improvement": "Clarify similarities and provide a comparison across methods to position the contribution relative to prior work.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "novelty", "object_sim": 0.9081052541732788, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7798058390617371}, {"unit_index": 2, "inspected_object": "Figure 1 Visual Accuracy", "observation": "The debiased target in Figure 1 does not completely overlap with the true density as theoretically expected for debiasing.", "reasoning": "Figures should be visually accurate representations of theoretical claims; the current depiction may mislead regarding the recovery of the true distribution.", "judgment": "Minor visual inaccuracy requiring correction.", "valence": "negative", "suggested_improvement": "Correct the figure so the debiased target completely overlaps with the true density.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8379711508750916, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7975724339485168}, {"unit_index": 3, "inspected_object": "Figure 2 Baseline Inclusion", "observation": "Figure 2 lacks a discrete-time normalizing flow baseline.", "reasoning": "To properly contextualize the Flow Map approach within the broader family of normalizing flows, it should be compared against the discrete-time alternative it aims to improve upon.", "judgment": "Incomplete comparative context in visualization.", "valence": "negative", "suggested_improvement": "Include a discrete-time NF in Figure 2.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8241122364997864, "reasoning_key": "design_justification", "reasoning_sim": 0.7241460084915161}, {"unit_index": 4, "inspected_object": "Table 3 Scaling Pattern", "observation": "For large systems like Hexa-Alanine, the scalability advantage of FALCON over discrete-time NFs shrinks significantly, with other methods outperforming it on Wasserstein distance.", "reasoning": "The claim of superior scalability is undermined by data showing diminished advantages or reversals on larger systems; this pattern requires explanation rather than being ignored.", "judgment": "Scalability narrative is challenged by empirical results on large systems.", "valence": "negative", "suggested_improvement": "Discuss the scaling pattern and explain the performance drop on large systems.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7704296708106995, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8066807389259338}, {"unit_index": 5, "inspected_object": "Table 4 Efficiency Trend", "observation": "Efficiency results show mixed trends: ECNF varies by system size, SBG is slower except on the largest system, and no clear trend emerges.", "reasoning": "Experimental results should tell a coherent story; unexplained mixed results suggest a need for interpretation rather than just reporting raw numbers.", "judgment": "Lack of clear efficiency trend reduces interpretability of results.", "valence": "negative", "suggested_improvement": "Interpret and discuss the mixed efficiency results.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7501291632652283, "reasoning_key": "fair_comparison", "reasoning_sim": 0.79960697889328}]}, {"review_id": "08Gu7V4slg", "paper_id": "QzIQgloYgX", "paper_title": "No, of Course I Can! Deeper Fine-Tuning Attacks That Bypass Token-Level Safety Mechanisms", "decision": "Accept (Poster)", "summary": "The reviewer accepts the empirical efficacy of the attack but critiques the paper for lacking formal conceptual delineation of its novelty, sufficient methodological validation for harm metrics, and mechanistic explanations for generalization.", "units": [{"unit_index": 0, "inspected_object": "The empirical results of the attack on production-grade language models.", "observation": "The reviewer identifies strong empirical results, specifically noting high attack success rates validated by external parties such as OpenAI's bug bounty.", "reasoning": "High success rates on current, production-grade models constitute strong evidence of real-world impact and credibility for the core empirical claim.", "judgment": "The paper demonstrates a sound and effective attack with significant practical relevance.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8138370513916016, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7203333377838135}, {"unit_index": 1, "inspected_object": "The paper's framing of novelty as a 'new attack paradigm' relative to pre-filling attacks.", "observation": "The paper cites successful pre-filling attacks as inspiration while claiming a new paradigm, creating a tension in the genealogy of the contribution.", "reasoning": "Citing prior work creates a burden to explain the delta; without precise boundary conditions or formal definitions, claims of 'new paradigms' are underspecified and lack rigorous delineation from existing categories.", "judgment": "The conceptual contribution is unclear because the distinction between this attack and prior art is not adequately articulated or formalized.", "valence": "negative", "suggested_improvement": "Provide a formal, falsifiable definition of a 'deep' attack to clarify whether the contribution is a new category or a more effective instance of an existing one.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8593099117279053, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8017555475234985}, {"unit_index": 2, "inspected_object": "The evaluation methodology, specifically the reliance on LLM-as-a-judge and opaque human validation.", "observation": "The paper relies on LLM-as-a-judge for harmfulness assessment with opaque human validation, lacking standard metrics like inter-annotator agreement or sample size details.", "reasoning": "Without quantified human validation and reliability metrics, it is uncertain whether reported attack success rates accurately reflect true harmfulness rather than artifacts of the judge model.", "judgment": "The measurement validity is insufficient to fully support the claimed severity of the attack.", "valence": "negative", "suggested_improvement": "Report inter-annotator agreement, sample sizes, and disagreement protocols to establish measurement reliability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7427447438240051, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8027637004852295}, {"unit_index": 3, "inspected_object": "The mechanistic explanation for why the attack generalizes from benign to harmful prompts.", "observation": "The paper successfully demonstrates that the attack works but offers little insight into the underlying mechanism of its generalization.", "reasoning": "Demonstrating that an attack works is distinct from explaining why it works; understanding the generalization mechanism is necessary for a complete scientific contribution and for informing better defenses.", "judgment": "The explanatory depth is incomplete, limiting the paper's utility beyond mere demonstration.", "valence": "negative", "suggested_improvement": "Provide a hypothesis about the generalization mechanism to transform the paper from a demonstration into an explanation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7440295219421387, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7645866870880127}, {"unit_index": 4, "inspected_object": "The general utility of the model after the attack is applied.", "observation": "The review notes the absence of benchmark results measuring whether the attack degrades the model's normal capabilities.", "reasoning": "An attack's threat level depends on collateral damage; if the attack preserves general utility, it represents a surgical manipulation, whereas if it breaks the model, it may be less concerning as a targeted threat.", "judgment": "The practical severity and operational cost of the attack are unknown due to missing utility metrics.", "valence": "conditional", "suggested_improvement": "Include benchmark results to reveal the attack's operational cost and whether it acts as a surgical manipulation or a blunt instrument.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7498661279678345, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7544082999229431}]}, {"review_id": "08lMHzVzba", "paper_id": "W8ZFwYKbXo", "paper_title": "FATE: Feature-Wise Graph Attention with Multi-Period Temporal Encoding for Stock Return Forecasting", "decision": "Reject", "summary": "The reviewer conducts a completeness audit, endorsing the paper's motivation and ablation-based validation while identifying gaps in demonstrated interpretability, scalability discussion, and computational reporting. The logic separates technical competence from novelty, resulting in a qualified endorsement contingent on providing missing evidentiary details.", "units": [{"unit_index": 0, "inspected_object": "SuperAttention module and sparse graph construction interpretability claims", "observation": "The paper asserts that the feature-wise graph attention and sparse graphs provide interpretability, but provides no concrete examples, case studies, or visualizations of attention weights or learned graphs on actual stock subgraphs.", "reasoning": "Interpretability is treated as a communicative achievement requiring demonstration; without visual evidence of what the model attends to, the claim remains abstract and leaves readers with limited insight into how the model captures market dynamics.", "judgment": "The interpretability claim is unsupported due to lack of demonstrative evidence.", "valence": "negative", "suggested_improvement": "Provide concrete examples, case studies, or visualizations of attention weights and learned graphs on actual stock subgraphs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.810944139957428, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7682072520256042}, {"unit_index": 1, "inspected_object": "Experimental scope limited to China stock markets (CSI 500/1000)", "observation": "Experiments are confined to Chinese markets with no discussion of scalability to larger markets like the S&P 500 or global indices.", "reasoning": "The reviewer infers a potential generalizability problem regarding whether mechanisms like dynamic graphs and multi-period encoding would transfer to different market structures (regulatory environments, trading behaviors) and scales (number of stocks, data volume).", "judgment": "The paper lacks a discussion on feasibility and robustness beyond the tested setting.", "valence": "conditional", "suggested_improvement": "Include a discussion on the feasibility of scaling to larger markets and different market structures.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7613904476165771, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7982177138328552}, {"unit_index": 2, "inspected_object": "Computational overhead and training time details", "observation": "The paper does not provide quantitative details on training time or scalability to larger datasets.", "reasoning": "Evaluating against a norm of practical deployability, the reviewer considers computational cost part of the contribution; the absence of engineering dimensions prevents assessment of whether the model is practical for real-world deployment (e.g., daily rebalancing).", "judgment": "The paper is incomplete regarding its practical deployability and efficiency.", "valence": "uncertain", "suggested_improvement": "Provide more details on training time and scalability to larger datasets.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8818615674972534, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8449686169624329}, {"unit_index": 3, "inspected_object": "Ablation studies validating architectural modules", "observation": "The ablation studies demonstrate that each module contributes to the performance.", "reasoning": "The reviewer treats component-wise removal as the gold standard for architectural justification, viewing the ablations as primary empirical warrant for the architecture's validity and internal consistency.", "judgment": "The architectural justification is strong and validated by empirical evidence.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8269249200820923, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7998329401016235}, {"unit_index": 4, "inspected_object": "Overall novelty and contribution relative to technical execution", "observation": "The reviewer assigns a low contribution score (2) despite high soundness (3) and presentation (3) scores.", "reasoning": "The reviewer distinguishes between technical execution/novelty, viewing the work as a combination of existing ideas (multi-view graphs, multi-period encoding, attention) rather than a fundamentally new paradigm, applying a stricter novelty bar than the competent execution suggests.", "judgment": "The work is well-executed and sound but represents an incremental contribution rather than a transformative one.", "valence": "mixed", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.9095507860183716, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8390429019927979}]}, {"review_id": "09SG8E0AX0", "paper_id": "nGihWDdQFI", "paper_title": "CAED-Agent: an Agentic Framework to Automate Simulation-Based Experimental Design", "decision": "Reject", "summary": "The reviewer acts as a conceptual auditor, finding the problem framing misleading, novelty claims overstated relative to adjacent fields, and the method lacking theoretical grounding and robustness evidence. The review emphasizes the need for formalization, proper positioning against Bayesian decision theory, and testing in complex regimes.", "units": [{"unit_index": 0, "inspected_object": "Problem framing and terminology (simulation-based experimental design vs. simulator settings)", "observation": "The reviewer identifies a mismatch between the paper's claimed domain and its actual optimization targets, noting that the paper optimizes simulator settings (grid size, time step) rather than experimental variables or physical parameters.", "reasoning": "The reviewer applies a standard where 'experimental design' involves varying quantities of scientific interest; optimizing numerical discretization parameters does not meet this definition, making the title and abstract misleading regarding the problem scope.", "judgment": "The problem framing is misleading and the contribution does not generalize beyond the specific setup.", "valence": "negative", "suggested_improvement": "Clarify whether the method varies experimental variables or physical parameters to align with the claimed domain.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8666836619377136, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8067569136619568}, {"unit_index": 1, "inspected_object": "Novelty claim relative to cost-aware simulation optimization literature", "observation": "The reviewer notes prior work in Bayesian decision theory and amortized decision-making (Bharti et al., Gorecki et al.) that addresses similar cost-aware optimization goals.", "reasoning": "If the paper's approach is viewed as an LLM-driven heuristic version of existing amortized decision-making frameworks, the novelty claim is overstated because the core mechanism is not new relative to adjacent fields.", "judgment": "The novelty claim is overstated and the contribution is better characterized as a heuristic adaptation of existing ideas.", "valence": "negative", "suggested_improvement": "Engage with cost-aware methods in SBI and decision theory to properly scope the novelty claim.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.8727251887321472, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8219476342201233}, {"unit_index": 2, "inspected_object": "Formal status and theoretical grounding of the LLM agent", "observation": "The reviewer finds no clear notion of an optimization objective or theoretical grounding for the surrogate-LLM loop, describing it as heuristic without stability or convergence analysis.", "reasoning": "Optimization methods are expected to provide uncertainty quantification or formal guarantees (as BO does); the absence of such calibration in the LLM approach raises trustworthiness concerns about how users can verify the agent's suggestions.", "judgment": "The method lacks necessary theoretical grounding and trustworthiness guarantees compared to established optimization baselines.", "valence": "negative", "suggested_improvement": "Articulate different criteria for evaluating agentic systems or integrate the surrogate into a Bayesian decision-theoretic framework to provide formal guarantees.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8293694257736206, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8192312121391296}, {"unit_index": 3, "inspected_object": "Experimental scope and generalizability (PDE benchmarks)", "observation": "The reviewer observes that the PDE benchmarks used are low-dimensional, deterministic, and smooth, serving only as sanity checks rather than evidence of scalability.", "reasoning": "Robustness requires stress-testing methods in challenging regimes (3D, chaotic, stochastic, non-monotonic trade-offs) where the assumed smooth cost-fidelity relationship might break down; current experiments do not probe these failure modes.", "judgment": "The empirical evidence is insufficient to demonstrate scalability or robustness in deployment scenarios.", "valence": "negative", "suggested_improvement": "Test the method on more complex regimes (chaotic, stochastic) to probe failure modes like hallucination or instability.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8643901944160461, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8617989420890808}, {"unit_index": 4, "inspected_object": "Writing quality and presentation polish", "observation": "The reviewer notes random bolding, missing spaces, inconsistent punctuation, and text that appears unedited or LLM-generated.", "reasoning": "Poor writing quality signals a lack of attention to detail and care in the work, undermining the perception of the paper's overall polish and rigor.", "judgment": "The presentation is unpolished and detracts from the paper's credibility.", "valence": "negative", "suggested_improvement": "Edit the text carefully to fix formatting errors and ensure professional presentation.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8486979007720947, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7964721322059631}]}, {"review_id": "09UEYTGfgX", "paper_id": "ZkPULxynIJ", "paper_title": "FSTS: A Feature Space Transfer Selection Method for Data", "decision": "Reject", "summary": "The reviewer performs a feasibility and novelty audit, acknowledging coherent method description and promising narrow performance while systematically challenging unsubstantiated robustness and efficiency claims, limited generalizability, restrictive assumptions, and unclear novelty differentiation.", "units": [{"unit_index": 0, "inspected_object": "Method's core mechanism (shared backbone, centroid computation, cosine similarity, band-based coreset selection)", "observation": "The reviewer accurately reconstructs the pipeline steps without relying on authors' framing.", "reasoning": "Accurate reconstruction demonstrates the reviewer has read the paper closely enough to articulate method steps independently, establishing a baseline for evaluating coherence.", "judgment": "The method description is coherent and correctly described.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8068859577178955, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8235672116279602}, {"unit_index": 1, "inspected_object": "Performance at low selection ratios (10–30%)", "observation": "FSTS performs best at low selection ratios compared to baselines.", "reasoning": "This is a specific, empirical observation that anchors the reviewer's positive assessment of the method's effectiveness in narrow settings.", "judgment": "The method shows promise and competence in limited contexts.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8200284242630005, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8108858466148376}, {"unit_index": 2, "inspected_object": "Robustness claims under image corruption", "observation": "The claim of 'superior performance' under corruption lacks specific, quantitative backing.", "reasoning": "Without quantitative data, the claim remains unsubstantiated; the reviewer finds the claim plausible but insufficiently supported by evidence.", "judgment": "The robustness claim is currently weak due to lack of evidence.", "valence": "negative", "suggested_improvement": "Provide specific, quantitative data to back up robustness claims.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.836606502532959, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7932901978492737}, {"unit_index": 3, "inspected_object": "Computational cost claims ('computationally lightweight')", "observation": "The assertion of being computationally lightweight is supported only by complexity analysis, not comparative measurements; 'medium speed' is non-quantifiable.", "reasoning": "Complexity analysis alone does not prove practical efficiency; comparative measurements are needed to substantiate claims of lightweight operation.", "judgment": "The computational efficiency claim is unsubstantiated.", "valence": "negative", "suggested_improvement": "Provide comparative computational measurements instead of relying solely on complexity analysis.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8995729684829712, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7836804986000061}, {"unit_index": 4, "inspected_object": "Experimental scope (datasets and diversity)", "observation": "Only two small-scale datasets (CIFAR-10, KITTI Road) are used, which are standard but limited in domains and diversity.", "reasoning": "Limited dataset diversity raises doubts about whether the method's assumptions hold beyond these specific settings, particularly under substantial domain shifts or feature/label diversity.", "judgment": "Generalizability is uncertain due to narrow experimental scope.", "valence": "negative", "suggested_improvement": "Conduct experiments on more diverse datasets to test generalizability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8710678815841675, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8481511473655701}, {"unit_index": 5, "inspected_object": "Method assumptions (labeled source, source-target alignment)", "observation": "The method depends on a labeled, relevant source dataset and assumes strong source-target alignment.", "reasoning": "If these preconditions are rarely met in real-world data marketplaces, the method's practicability is reduced regardless of in-distribution performance.", "judgment": "Practicability is weakened by restrictive assumptions.", "valence": "negative", "suggested_improvement": "Address how the method handles scenarios with weak source-target alignment.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8449591994285583, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7902419567108154}, {"unit_index": 6, "inspected_object": "Novelty and differentiation from prior work", "observation": "The method combines established ideas from prototype-based methods, transfer learning, and data pruning without explicitly stating what differentiates it from the state of the art.", "reasoning": "A contribution should be recognizably novel without requiring readers to reverse-engineer novelty; explicit positioning is required to evaluate the contribution.", "judgment": "Novelty is not clearly established in the current framing.", "valence": "negative", "suggested_improvement": "Explicitly state what differentiates the method substantially from the state of the art.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.9326915144920349, "reasoning_key": "novelty_standard", "reasoning_sim": 0.796427845954895}, {"unit_index": 7, "inspected_object": "Component attribution (ablations)", "observation": "Lack of ablations on centroid type or distance metric.", "reasoning": "Without ablations, it is unclear which part of the method drives improvements, suggesting potential uncertainty in design choices.", "judgment": "Mechanism understanding is incomplete.", "valence": "negative", "suggested_improvement": "Perform ablation studies to identify which components drive improvements.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8141832947731018, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7863556146621704}]}, {"review_id": "09rz2VcHFu", "paper_id": "PYkGKPHBi0", "paper_title": "On the Efficiency-Safety Dilemma in Large Reasoning Models", "decision": "Reject", "summary": "The reviewer evaluates the paper against standards of construct validity, contribution novelty, and statistical rigor, finding the narrow safety definition, observational-only contribution, and lack of uncertainty quantification insufficient for the venue despite the timeliness of the topic.", "units": [{"unit_index": 0, "inspected_object": "Operationalization of safety (jailbreak resistance via GPT-4o scoring on two black-box suites)", "observation": "Safety is defined narrowly as jailbreak harmfulness using only 50 prompts per benchmark, excluding dimensions like toxicity, bias, honesty, privacy, or impersonation.", "reasoning": "This narrow definition represents construct under-representation; the measure fails to capture the full theoretical domain of 'safety', and the small sample size limits precision even for this specific construct.", "judgment": "The operationalization is insufficiently comprehensive and precise.", "valence": "negative", "suggested_improvement": "null", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7924421429634094, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7458817362785339}, {"unit_index": 1, "inspected_object": "Comparison with prior work on trustworthiness preservation", "observation": "The submission claims uniform superficial safety gains, while a cited ICML'24 paper found that quantization can preserve or improve trustworthiness across multiple dimensions.", "reasoning": "The divergent conclusions create uncertainty about whether the submission's findings are artifacts of measuring only one dimension rather than a general phenomenon, especially regarding sensitivity to compression in Large Reasoning Models.", "judgment": "The paper's conclusions are potentially contradictory to broader field findings and lack mechanistic reconciliation.", "valence": "negative", "suggested_improvement": "Explain the divergent conclusions from 'Decoding Compressed Trust' and hypothesize mechanisms such as LRM chain-of-thought sensitivity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8150883316993713, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7884328365325928}, {"unit_index": 2, "inspected_object": "Nature of the contribution (observational vs. prescriptive)", "observation": "The paper presents an observational study confirming that efficiency degradation reduces safety performance, described as an important but unsurprising data point.", "reasoning": "The venue values novelty over observational confirmation; delivering only empirical observations without deeper mechanistic analysis or novel solutions constitutes an incremental rather than transformative contribution.", "judgment": "The contribution level is insufficient for the venue due to its purely observational nature.", "valence": "negative", "suggested_improvement": "null", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8113797903060913, "reasoning_key": "novelty_standard", "reasoning_sim": 0.801020622253418}, {"unit_index": 3, "inspected_object": "Statistical reporting of results", "observation": "Results are reported as point estimates only, with no error bars, confidence intervals, or seed variation.", "reasoning": "Given the small test sets and sampling-sensitive decoding inherent to LRM outputs, the absence of uncertainty quantification undermines reproducibility and confidence in the stochastic findings.", "judgment": "The statistical rigor is inadequate for the observed phenomena.", "valence": "negative", "suggested_improvement": "Include error bars, confidence intervals, or seed variation to acknowledge stochasticity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8151420950889587, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8303751945495605}]}, {"review_id": "0A24oT2ypZ", "paper_id": "yhqCTAFy0O", "paper_title": "Teacher-Student Multi-Agent Reinforcement Learning Framework for AutoML Pipeline Construction", "decision": null, "summary": "The reviewer employs a dominant strategy of dismissal based on a suspected LLM authorship, using this as a sufficient condition for rejection and ethics flagging. This central claim organizes a series of specific technical and presentation critiques, including weak baselines, lack of statistical rigor, poor visualizations, undefined terms, and missing citations, which are treated as symptoms of the alleged illegitimate production rather than isolated flaws.", "units": [{"unit_index": 0, "inspected_object": "The paper's text and structure as indicative of LLM authorship", "observation": "The reviewer claims to recognize the output as 'almost entirely LLM-generated' based on tone and style before reading any disclosure.", "reasoning": "The reviewer invokes personal expertise ('worked with ChatGPT enough') to detect a mode of production that renders the work illegitimate, treating this detection as an independent confirmation of quality issues rather than post-hoc rationalization.", "judgment": "The paper is fundamentally illegitimate due to its presumed automated generation, warranting dismissal regardless of content.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8395032286643982, "reasoning_key": "construct_validity", "reasoning_sim": 0.7383558750152588}, {"unit_index": 1, "inspected_object": "The claim that the paper should be considered for acceptance", "observation": "The reviewer states the paper 'should not, by any means, be considered for acceptance'.", "reasoning": "The reviewer treats the suspected LLM authorship as a sufficient condition for rejection, overriding any potential merit in the technical content or experimental results.", "judgment": "Absolute rejection of the paper's suitability for the venue.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7491263747215271, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7750840783119202}, {"unit_index": 2, "inspected_object": "Ethical compliance regarding LLM usage", "observation": "The reviewer flags the submission for ethics review.", "reasoning": "The reviewer perceives the undisclosed or improperly handled LLM usage as a serious ethical violation requiring institutional scrutiny beyond standard peer review.", "judgment": "The paper raises ethical concerns that necessitate formal review.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.7194644212722778, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7143869996070862}, {"unit_index": 3, "inspected_object": "Problem motivation and domain understanding", "observation": "The problem statement is unclear and lacks sufficient motivation for AutoML challenges.", "reasoning": "The reviewer expects the problem to be motivated from the domain perspective first; the absence of this context is interpreted as a failure of the text itself (incoherence) rather than a fixable omission.", "judgment": "The paper fails to establish a clear, motivated problem statement.", "valence": "negative", "suggested_improvement": "Provide clearer motivation for the problem from the AutoML domain.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "problem_framing", "object_sim": 0.8708725571632385, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7899541258811951}, {"unit_index": 4, "inspected_object": "Baseline selection and justification", "observation": "Random/Grid Search and single-agent DDQL are characterized as strong baselines; auto-sklearn was removed due to instability.", "reasoning": "Weak baselines diminish the significance of claimed improvements; excluding unstable baselines without discussion may mask problematic comparisons.", "judgment": "The experimental design relies on insufficiently strong baselines and lacks transparency in exclusion criteria.", "valence": "negative", "suggested_improvement": "Justify baseline strength and provide further discussion or investigation into the removal of auto-sklearn.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.8937913179397583, "reasoning_key": "fair_comparison", "reasoning_sim": 0.845280110836029}, {"unit_index": 5, "inspected_object": "Statistical reporting of performance improvements", "observation": "Performance improvements are presented as percentages without means, standard deviations, or confidence intervals.", "reasoning": "Point estimates without distributional information cannot rule out outliers, failing to meet the evidentiary standard required for robust quantitative claims.", "judgment": "The quantitative claims lack statistical grounding and reliability.", "valence": "negative", "suggested_improvement": "Report means, standard deviations, and confidence intervals to support percentage claims.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8580159544944763, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8395671844482422}, {"unit_index": 6, "inspected_object": "Visualization clarity and honesty", "observation": "Figures contain unprofessional y-axis labels, a 'ridiculous' scatter plot, confusing shared axes, and a 'deceptive' dual-axis plot.", "reasoning": "Visualizations must be clear, honest, and informative; misleading techniques like dual axes or poor labeling undermine the presentation of results.", "judgment": "The visualizations are professionally inadequate and potentially misleading.", "valence": "negative", "suggested_improvement": "Redesign figures to ensure professional labels, clarity, and avoidance of deceptive plotting techniques.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.834613561630249, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8192799687385559}, {"unit_index": 7, "inspected_object": "Technical terminology definitions", "observation": "Terms such as DDQL, DEC-POMDP, NAS, and 'asymmetric' are undefined or ambiguous.", "reasoning": "Technical terms must be defined on first use; ambiguity indicates poor presentation and hinders comprehension.", "judgment": "The paper suffers from poor presentation due to undefined or ambiguous terminology.", "valence": "negative", "suggested_improvement": "Define all technical terms (DDQL, DEC-POMDP, NAS, asymmetric) clearly upon first use.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.6882862448692322, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8078749179840088}, {"unit_index": 8, "inspected_object": "Novelty of the proposed framework", "observation": "The framework is described as 'ASYMMETRIC DEC-POMDP', which the reviewer finds not novel.", "reasoning": "The claimed contribution appears to be a relabeling of existing work rather than a new methodological advance.", "judgment": "The paper lacks novelty in its core framework.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "low", "object_key": "novelty", "object_sim": 0.8515921235084534, "reasoning_key": "novelty_standard", "reasoning_sim": 0.81797194480896}, {"unit_index": 9, "inspected_object": "Citation completeness", "observation": "Missing citations for the concept 'Positioning vs LLM Orchestration'.", "reasoning": "Scholarly apparatus requires proper citation for positioning claims; missing references indicate incomplete literature engagement.", "judgment": "The paper fails to properly cite relevant literature for key concepts.", "valence": "negative", "suggested_improvement": "Add citations for 'Positioning vs LLM Orchestration' and other positioning claims.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "related_work", "object_sim": 0.8505240678787231, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7810103893280029}, {"unit_index": 10, "inspected_object": "Textual errors and surface polish", "observation": "Presence of an upside-down question mark and missing periods.", "reasoning": "Surface-level errors signal a lack of care and professionalism, contributing to the overall negative assessment of the paper's quality.", "judgment": "The paper exhibits a lack of care in its final presentation.", "valence": "negative", "suggested_improvement": "Proofread the text to remove surface-level errors such as typos and punctuation mistakes.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8265368342399597, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7727172374725342}, {"unit_index": 11, "inspected_object": "Related Work section", "observation": "The related work is described as 'too short and brief' leading to incoherence.", "reasoning": "Absence of expected content (comprehensive literature review) is conflated with textual confusion, indicating a failure to situate the work properly.", "judgment": "The related work is insufficient and incoherent.", "valence": "negative", "suggested_improvement": "Expand the related work section to provide a coherent and comprehensive overview of prior art.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "related_work", "object_sim": 0.817516028881073, "reasoning_key": "presentation_trust", "reasoning_sim": 0.812784731388092}]}, {"review_id": "0AAqbXPLoo", "paper_id": "tVe3qmrH2h", "paper_title": "Rethinking LLM Human Simulation: When a Graph is What You Need", "decision": null, "summary": "The reviewer dismisses the paper's contribution by characterizing it as a mere reframing of established GNN techniques applied to simple tasks, arguing that the resulting performance gains are unsurprising and that the paper's framing ('human simulation') overstates its scientific scope.", "units": [{"unit_index": 0, "inspected_object": "The paper's central claim of novelty and contribution.", "observation": "The reviewer re-describes the paper's contribution as merely reframing a classic recommender-style GNN for discrete choice prediction, rather than introducing new technical methods.", "reasoning": "The reviewer operates on an evaluative standard that defines novelty strictly as 'technical invention' (building a new tool) rather than 'problem reframing' (applying an existing tool to a new domain). Since GNNs for link prediction are well-established, applying them here is seen as presentational rather than substantive.", "judgment": "The paper offers limited scientific insight; its primary contribution appears modest and presentational.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8077608346939087, "reasoning_key": "design_justification", "reasoning_sim": 0.8301905393600464}, {"unit_index": 1, "inspected_object": "The use of the term 'human simulation' in the paper's framing.", "observation": "The reviewer objects to the term 'human simulation,' stating it exaggerates the scope and may mislead readers into thinking the paper deals with cognitive modeling or psychology.", "reasoning": "The reviewer holds a disciplinary boundary assumption that 'human simulation' is a term of art reserved for fields modeling cognition or strategic behavior. Predicting survey responses does not meet this threshold. The reviewer views the paper as an engineering benchmark dressed in inappropriate scientific language.", "judgment": "The paper overclaims its scope through misleading terminology.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7943504452705383, "reasoning_key": "design_justification", "reasoning_sim": 0.8180121183395386}, {"unit_index": 2, "inspected_object": "The empirical results showing GEMS outperforming LLM baselines.", "observation": "The reviewer acknowledges that GEMS achieves comparable or superior performance but argues these results are unsurprising given the task structure.", "reasoning": "The reviewer reasons that because LLMs are known to have weaknesses on certain tasks, demonstrating those weaknesses is not a breakthrough. Furthermore, the reviewer asserts the tasks are simple classification problems with small output spaces, implying that any model could perform well without needing novel insights.", "judgment": "The empirical performance, while true, is uninteresting and expected, thus not constituting a significant contribution.", "valence": "negative", "suggested_improvement": null, "support_status": "mixed", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8434129357337952, "reasoning_key": "design_justification", "reasoning_sim": 0.8240320086479187}]}, {"review_id": "0BeayXIu9H", "paper_id": "lNmZrawUMu", "paper_title": "AlphaAgentEvo: Evolution-Oriented Alpha Mining via Self-Evolving Agentic Reinforcement Learning", "decision": "Accept (Poster)", "summary": "The reviewer conducts a category assessment, judging the paper as applied engineering rather than fundamental research due to its reliance on existing algorithms (GRPO) and standard RL training. They challenge the semantic validity of 'self-evolving' and deem the experimental setup insufficient because it lacks training-matched baselines and transfer tests.", "units": [{"unit_index": 0, "inspected_object": "The core algorithm (GRPO) and its adaptation to an agentic setting.", "observation": "The reviewer identifies the underlying RL algorithm as GRPO and characterizes the paper's contribution as an adaptation of this existing algorithm to a new setting, rather than a new algorithmic machinery.", "reasoning": "The operative standard is that fundamental advances require new algorithmic machinery, not just applications of existing tools. Since the method is derivative in its core mechanics, it falls into the category of applied engineering rather than fundamental research.", "judgment": "The contribution is assessed as more applied than fundamental.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8493462800979614, "reasoning_key": "design_justification", "reasoning_sim": 0.770835280418396}, {"unit_index": 1, "inspected_object": "The reward function design (identified as hierarchical).", "observation": "The reviewer credits the hierarchical reward function as a genuine strength and thoughtful engineering.", "reasoning": "Despite the low score for fundamental contribution, the reviewer acknowledges the quality of the reward engineering as a positive attribute, creating a tension where good engineering does not equate to high novelty scores.", "judgment": "The reward design is a strength, but insufficient to elevate the work from applied to fundamental.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8506560325622559, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7702912092208862}, {"unit_index": 2, "inspected_object": "The terminology 'self-evolving' and the scope of the method.", "observation": "The reviewer challenges the use of 'self-evolving,' proposing that it implies meta-level change (agent modifying its own learning architecture), whereas the paper only shows object-level improvement (agent getting better at task via standard RL).", "reasoning": "The standard is literal semantic accuracy: 'self-evolution' should mean the agent controls or modifies its own learning process. The observed method is standard supervised-style RL fine-tuning, which does not meet this stricter definition.", "judgment": "The term 'self-evolving' is misleading and unjustified by the method's actual mechanics.", "valence": "negative", "suggested_improvement": "Clarify or correct the naming to accurately reflect that the agent improves through training rather than self-modification.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.7895435094833374, "reasoning_key": "design_justification", "reasoning_sim": 0.77675461769104}, {"unit_index": 3, "inspected_object": "The experimental baselines and generalizability of the method.", "observation": "All baselines used are 'training-free,' and the method is tested only on alpha mining without demonstrating transfer to other tasks.", "reasoning": "Comparisons against training-free baselines do not prove the advantage comes from the training/method itself rather than just the benefit of training. Furthermore, lack of transfer testing leaves the generality of the method unverified, suggesting it may be generic RL with a domain-specific reward.", "judgment": "The experimental evidence is insufficient to demonstrate specificity of advantage or generalizability.", "valence": "negative", "suggested_improvement": "Compare models trained on this task to isolate the training effect; demonstrate transfer to general tasks or explicitly scope the method's limitations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8867502212524414, "reasoning_key": "design_justification", "reasoning_sim": 0.8501959443092346}]}, {"review_id": "0BofO4QlPX", "paper_id": "brbIoPN91l", "paper_title": "OmniEarth-Bench: Probing Cognitive Abilities of MLLMs for Earth's Multi-sphere Observation Data", "decision": "Reject", "summary": "The reviewer systematically tests the benchmark's claims against standards of rigor, focusing on provenance transparency, novelty articulation, concrete utility demonstrations, and methodological consistency in scoring and reporting.", "units": [{"unit_index": 0, "inspected_object": "Provenance of annotations in the benchmark construction pipeline", "observation": "The reviewer questions whether all questions and categorizations were created by domain experts rather than LLMs, noting ambiguity in the phrase 'expert-in-the-loop curation'.", "reasoning": "If LLMs generated the questions with only expert validation, the benchmark could be circular, testing MLLMs on content produced by other LLMs and potentially encoding similar biases, which undermines the credibility of the ground truth.", "judgment": "The paper fails to meet a standard of transparency regarding annotation provenance, creating uncertainty about the validity of the benchmark's ground truth.", "valence": "negative", "suggested_improvement": "Clarify explicitly whether questions were human-created or LLM-generated to disambiguate the role of experts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8239997625350952, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7710468173027039}, {"unit_index": 1, "inspected_object": "Novelty claim relative to existing benchmarks", "observation": "The reviewer finds the related work section unclear regarding which data sources are not already incorporated in major ML benchmarks, citing ERA5 as an example of established data.", "reasoning": "The reviewer reconstructs the contribution as likely being the question-answer annotations rather than the data sourcing itself; if data sources are common, the paper must explicitly articulate the novelty of the annotations to distinguish its contribution for non-specialist readers.", "judgment": "The current framing obscures the paper's actual contribution, failing to clearly distinguish it from existing resources that may share similar data sources.", "valence": "negative", "suggested_improvement": "Clarify which data sources are novel versus pre-existing and explicitly state that the primary contribution is the sourcing of questions/annotations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.924799919128418, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8185299038887024}, {"unit_index": 2, "inspected_object": "Cross-sphere interaction claims", "observation": "The reviewer demands concrete examples of specific scientific challenges that previous benchmarks failed to capture but this one addresses, questioning the utility of abstract cross-sphere reasoning claims.", "reasoning": "A benchmark's value proposition should be tied to downstream scientific utility; merely having cross-sphere questions is insufficient unless they map to real problems that change how a scientist works, requiring a use-case-driven evaluation standard.", "judgment": "The abstract claim of addressing cross-sphere interactions lacks concrete instantiation and fails to demonstrate distinct external applicability compared to prior work.", "valence": "negative", "suggested_improvement": "Provide specific examples of scientific problems enabled by this benchmark that were not addressable by previous benchmarks.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8198649883270264, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7815176248550415}, {"unit_index": 3, "inspected_object": "Evaluation methodology for open-ended scoring via LLM-as-a-judge", "observation": "The reviewer notes a lack of validation experiments for the LLM-as-a-judge scoring mechanism and identifies potential failure modes in numerical tasks like epsilon-tolerance comparisons.", "reasoning": "Any scoring instrument must be validated before trusting it to score models; the epistemic standard requires that the measurement tool (scoring mechanism) be at least as reliable as the models it evaluates, necessitating demonstration through validation experiments.", "judgment": "The evaluation protocol is incomplete because the reliability of the automated scoring mechanism has not been demonstrated against known failure cases.", "valence": "negative", "suggested_improvement": "Report validation experiments demonstrating the reliability of the LLM-as-a-judge scorer, particularly on numerical comparison tasks.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8430822491645813, "reasoning_key": "construct_validity", "reasoning_sim": 0.8015047907829285}, {"unit_index": 4, "inspected_object": "Scoring documentation for visual grounding and captioning", "observation": "The reviewer observes that the appendix does not cover how visual grounding and captioning performance are scored, despite these being included task categories.", "reasoning": "Every task type in a benchmark should have a correspondingly documented scoring protocol; the absence of documentation creates uncertainty about whether these tasks were properly evaluated and undermines confidence in the reported results.", "judgment": "The omission of scoring protocols for specific task types represents a completeness gap that weakens the interpretability of the benchmark's results.", "valence": "negative", "suggested_improvement": "Document the specific scoring methods used for visual grounding and captioning tasks in the appendix.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8124537467956543, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7806459069252014}, {"unit_index": 5, "inspected_object": "ENSO case study in Section 4.3", "observation": "The reviewer finds the ENSO example appears disconnected from the main evaluation rigor, lacking prompting setup details, baselines, and context beyond a single GPT-4o prediction.", "reasoning": "Empirical claims must be interpretable within the paper's own evaluation framework; without baselines and setup details, the case study is an anecdote rather than evidence, violating the standard of methodological consistency established in earlier sections.", "judgment": "The case study is methodologically inconsistent and uninterpretable due to missing contextual information and baseline comparisons.", "valence": "negative", "suggested_improvement": "Either remove the case study or expand it into a proper task evaluation with rigorous detailing comparable to Section 4.2.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7124421000480652, "reasoning_key": "novelty_standard", "reasoning_sim": 0.807053804397583}, {"unit_index": 6, "inspected_object": "Taxonomic hierarchy placement (L3 vs L1)", "observation": "The reviewer notes that the 'type of analysis needed' category might logically belong at L1 rather than L3 because L1, L2, and L4 are problem-specific while L3 describes general skills.", "reasoning": "The organizational hierarchy should be logically coherent; mixing problem-specific levels with general skill descriptions creates taxonomic inconsistency, though this is viewed as a design consideration rather than a critical flaw.", "judgment": "The benchmark's hierarchical structure contains a logical incoherence in the placement of the analysis category, suggesting a need for future refinement.", "valence": "conditional", "suggested_improvement": "Consider moving the 'type of analysis needed' category to L1 for future iterations to improve taxonomic coherence.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7814104557037354, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7441224455833435}, {"unit_index": 7, "inspected_object": "Numerical consistency in Table 3", "observation": "The reviewer identifies a quantitative discrepancy where Table 3 lists 27k MCQ and 27k open-ended questions, summing to 54k, which exceeds the stated total of 29k.", "reasoning": "Internal consistency of reported numbers is required for reader trust; such discrepancies signal either typos or serious issues with task categorization overlap that must be resolved to verify the dataset size.", "judgment": "The reported statistics contain an internal inconsistency that undermines confidence in the precision of the dataset description.", "valence": "negative", "suggested_improvement": "Correct the numerical discrepancy in Table 3 and clarify the categorization counts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7906606197357178, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7928885221481323}, {"unit_index": 8, "inspected_object": "Model subset selection for open-ended evaluation", "observation": "The reviewer questions why open-ended questions are evaluated on a small subset of models compared to the full set used for multiple-choice questions.", "reasoning": "Evaluation should be consistent across task types if a benchmark reports results across models; asymmetry suggests cost constraints or selective reporting that should be explained rather than left implicit to maintain comparative fairness.", "judgment": "The asymmetric evaluation coverage between task types introduces uncertainty about the comparability of results across model capabilities.", "valence": "conditional", "suggested_improvement": "Explain the rationale for evaluating open-ended questions on a smaller subset of models or justify the asymmetry.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8473260998725891, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8153268098831177}]}, {"review_id": "0D9AQUqacH", "paper_id": "c8oAL6i2yR", "paper_title": "A Probabilistic Basis for Low-Rank Matrix Learning", "decision": "Reject", "summary": "The reviewer evaluates the paper positively on its theoretical gap-filling and empirical coherence but identifies three critical areas requiring clarification: novelty delimitation against prior work, scalability evidence for high dimensions, and theoretical error bounds for approximations.", "units": [{"unit_index": 0, "inspected_object": "Novelty claim relative to prior work on the Normal Product–nuclear norm connection", "observation": "The reviewer notes that several researchers in variational Bayesian inference have previously noted this connection, creating ambiguity about the extent of prior theoretical and applied work.", "reasoning": "The reviewer applies a norm that novelty must be explicitly delimited; if the connection was already well explored, the contribution's value would shrink, requiring the authors to preemptively address potential overlap with related literature.", "judgment": "The novelty claim requires clarification and boundary-setting to establish its distinctiveness from existing research.", "valence": "negative", "suggested_improvement": "Clarify the extent of prior work—both theoretical and applied—by these authors or in the field to situate the novelty.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.887768030166626, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8270740509033203}, {"unit_index": 1, "inspected_object": "Scalability to high-dimensional matrices", "observation": "The paper does not provide computational complexity indicators, runtime, or empirical timing results for the method.", "reasoning": "Practical claims require scaling evidence; without quantitative performance data showing how the method behaves as problem size grows, the practical relevance remains unconvincing.", "judgment": "The method's scalability is currently unverified due to missing empirical timing evidence.", "valence": "negative", "suggested_improvement": "Provide computational complexity indicators, runtime, or empirical timing results to demonstrate scalability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8332861661911011, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8297127485275269}, {"unit_index": 2, "inspected_object": "Theoretical basis of the NP-to-NND approximation", "observation": "The paper lacks specific assumptions (e.g., singular value conditions) and theoretical error analysis or limits for the approximation.", "reasoning": "Approximations should come with error characterization; relying solely on empirical agreement is insufficient for theoretical rigor, and the paper needs at least a heuristic argument or discussion of when the approximation holds.", "judgment": "The theoretical basis of the approximation is incomplete without an account of its validity bounds.", "valence": "negative", "suggested_improvement": "Provide theoretical error analysis, limits, or at least a discussion of the specific assumptions behind the NP-to-NND approximation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8704307675361633, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.787329375743866}, {"unit_index": 3, "inspected_object": "Theoretical contribution filling a gap in probabilistic characterization of nuclear norms", "observation": "The nuclear norm is widely used in optimization but its probabilistic characterization has been understudied, and the paper provides a clear theoretical treatment.", "reasoning": "Filling a theoretical gap where a widely used concept lacks sufficient study constitutes a valuable contribution, accepted on faith given the reviewer's low confidence in deep verification.", "judgment": "The theoretical treatment of the nuclear norm's probabilistic characterization is reliable and meaningful.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8243963718414307, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8011676669120789}, {"unit_index": 4, "inspected_object": "Adaptive lambda estimation and empirical validation", "observation": "The adaptive lambda estimation is well motivated and empirically validated through experiments.", "reasoning": "Theory should be usable, and the demonstration of usability through experiments provides coherence between theoretical claims and practical application.", "judgment": "The adaptive lambda estimation is well-motivated and effectively validated.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.861518144607544, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7657999396324158}]}, {"review_id": "0DFIaYKsPC", "paper_id": "5UrPAW3uI1", "paper_title": "FedOpenMatch: Towards Semi-Supervised Federated Learning in Open-Set Environments", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper as a competent engineering contribution with solid experimental breadth but incremental technical novelty. Key critiques focus on the reliance on existing components for novelty, potential client drift due to local iterations, inconsistencies in the gradient similarity diagnostic, and a minor sign error. The review is constructive, seeking to clarify boundaries and robustness rather than challenge core validity.", "units": [{"unit_index": 0, "inspected_object": "Novelty of solution components (logit adjustment, weak-strong consistency)", "observation": "The logit adjustment is directly taken from Menon et al. 2021, and weak-strong consistency is modified from existing methods.", "reasoning": "While the paper tackles a new problem (open-set federated semi-supervised learning), the technical machinery relies on existing ideas rather than introducing substantially new algorithms, which limits the contribution score despite the practical value.", "judgment": "Incremental technical novelty; solid but not exciting contribution.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8736365437507629, "reasoning_key": "design_justification", "reasoning_sim": 0.8114733099937439}, {"unit_index": 1, "inspected_object": "Local iterations in federated learning", "observation": "The paper does not provide sensitivity analysis regarding the number of local training steps.", "reasoning": "Since local clients only have unlabeled data, prolonged local training could cause diverging model updates (client drift); the absence of sensitivity analysis leaves a gap in understanding the method's robustness to this hyperparameter.", "judgment": "Potential fragility to hyperparameters; missing practical guidance.", "valence": "negative", "suggested_improvement": "Provide sensitivity analysis for local iterations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.789946973323822, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7902647852897644}, {"unit_index": 2, "inspected_object": "Gradient similarity diagnostic plots (Figure 4)", "observation": "Early training steps show low gradient similarity yet comparable accuracies, contradicting the implied monotonic relationship.", "reasoning": "The proposed explanatory mechanism (gradient stop reduces interference) is not fully supported by the visual evidence across all phases; alternative factors like gradient magnitude may be driving performance.", "judgment": "Explanatory story is incomplete or potentially measuring the wrong quantity.", "valence": "negative", "suggested_improvement": "Investigate alternative explanations such as gradient magnitude or refine the diagnostic metrics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.798177182674408, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7966869473457336}, {"unit_index": 3, "inspected_object": "Equation 6 sign", "observation": "There is a sign error in Equation 6 when compared to the original Menon et al. 2021 source.", "reasoning": "This is a concrete, verifiable discrepancy that affects reproducibility and signals careful verification against sources.", "judgment": "Minor flaw requiring correction.", "valence": "negative", "suggested_improvement": "Correct the sign in Equation 6.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.6914277076721191, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8489440083503723}, {"unit_index": 4, "inspected_object": "Outlier categorization decision rule", "observation": "The reviewer is unfamiliar with open-set semi-supervised learning and asks for clarification on when a sample is categorized as an outlier (e.g., whether all OVA classifiers must agree).", "reasoning": "The question stems from limited expertise in the subfield and seeks to understand the method's decision boundary rather than challenging correctness.", "judgment": "Request for clarification to aid understanding.", "valence": "conditional", "suggested_improvement": "Clarify the outlier categorization logic.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7874472141265869, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7841432094573975}]}, {"review_id": "0DYi0hG9VT", "paper_id": "LoisXFZL3k", "paper_title": "Faithful Bi-Directional Model Steering via Distribution Matching and Distributed Interchange Interventions", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper primarily through the lens of an outsider to the field, judging it deficient because it fails to provide sufficient context, define key terms, or justify methodological choices for a non-specialist audience. The critique focuses entirely on presentation and accessibility rather than technical soundness or experimental validity.", "units": [{"unit_index": 0, "inspected_object": "Paper's framing of the distinction between 'intervention-based' and 'optimization-based' approaches", "observation": "The reviewer questions the logical move from intervention-based to optimization-based, suggesting the paper conflates or elides a distinction that matters for understanding the method's novelty.", "reasoning": "A reader outside the specific model steering subfield requires clear definitions and distinctions to understand why the proposed method is different; vague framing creates an accessibility barrier.", "judgment": "Deficient presentation due to lack of clarity in key terminological distinctions.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8520676493644714, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.756953775882721}, {"unit_index": 1, "inspected_object": "Description of failure modes ('degenerate, repetitive generations') in prior work", "observation": "The reviewer asks what 'degenerate, repetitive generations' means, finding the description too vague to evaluate the claimed improvement over prior methods.", "reasoning": "Claims of improvement must be grounded in clearly defined baseline failures; undefined terms prevent assessment of the proposed solution's value.", "judgment": "Deficient presentation due to unclear problem definition.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.7977099418640137, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7761528491973877}, {"unit_index": 2, "inspected_object": "Characterization of DAS as 'the standard for causal variable localization'", "observation": "The reviewer challenges this claim, asking what it means and why the reader should accept it.", "reasoning": "Foundational premises must be justified for non-specialists; asserting a standard without explanation fails to orient the reader to the methodological lineage.", "judgment": "Deficient presentation due to insufficient justification of methodological choices.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8277973532676697, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7709589600563049}, {"unit_index": 3, "inspected_object": "Non-monotonic behavior in Fig. 1c", "observation": "The reviewer flags unexplained non-monotonic behavior in the figure without speculating on causes.", "reasoning": "Empirical results should be interpretable or at least acknowledged when they deviate from expected patterns; unexplained patterns create confusion for readers.", "judgment": "Deficient presentation due to lack of empirical transparency/interpretation.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8545262217521667, "reasoning_key": "presentation_trust", "reasoning_sim": 0.832107424736023}]}, {"review_id": "0DuF0r4HIc", "paper_id": "R3VBfYVK1x", "paper_title": "Evaluating LLMs on Real-World Forecasting Against Expert Forecasters", "decision": "Reject", "summary": "The reviewer evaluates the paper negatively by triangulating against prior work and venue norms, arguing that the dataset is too small/narrow, the evaluation protocol is incomplete (lacking optimization/calibration), and the retrieval pipeline is unvalidated. The reviewer prioritizes methodological exhaustiveness and novelty over the soundness of the current measurements.", "units": [{"unit_index": 0, "inspected_object": "Dataset construction (source, size, and temporal scope)", "observation": "The dataset consists of approximately 400 questions collected from a single platform (Metaculus) during a narrow time window (July–December 2024).", "reasoning": "The reviewer invokes a prior work with a larger sample as a comparator, establishing a normative standard that sufficient scale and diversity are required for an evaluation to support general claims about LLM forecasting capability rather than reflecting a narrow temporal or platform-specific slice.", "judgment": "The dataset is insufficiently diverse and small in sample size to robustly support the paper's broader claims.", "valence": "negative", "suggested_improvement": "Collect a larger, more diverse dataset spanning multiple platforms and time periods.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8409961462020874, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7650859951972961}, {"unit_index": 1, "inspected_object": "Evaluation protocol completeness (prompt optimization and calibration)", "observation": "The paper evaluates only two prompt styles and reports systematic miscalibration without testing any corrective post-processing or calibration methods.", "reasoning": "The reviewer applies a norm that evaluation papers should explore the full space of plausible improvements (e.g., fine-tuning, retrieval optimization, calibration correction) to determine true model capability, treating the failure to address known deficiencies as an incomplete evaluation rather than honest reporting.", "judgment": "The evaluation methodology is shallow and incomplete because it fails to optimize the prediction pipeline or correct for reported miscalibration.", "valence": "negative", "suggested_improvement": "Explore fine-tuning, retrieval-system optimization, and calibration correction methods.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8314003348350525, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7929368019104004}, {"unit_index": 2, "inspected_object": "Novelty and contribution type relative to venue expectations", "observation": "The paper provides no new technique and is characterized as 'only an eval paper' compared to prior works deemed stronger and more comprehensive.", "reasoning": "The reviewer holds an implicit standard that contributions at this venue must involve methodological innovation or be so comprehensive as to constitute a definitive benchmark; since the paper lacks both novelty and exhaustive scope, it fails to meet the threshold for contribution.", "judgment": "The paper lacks sufficient novelty and comprehensiveness to constitute a strong contribution.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.897612452507019, "reasoning_key": "novelty_standard", "reasoning_sim": 0.850169837474823}, {"unit_index": 3, "inspected_object": "Retrieval pipeline reliability and contamination risk", "observation": "The paper uses a new pipeline called AskNews but does not demonstrate its reliability regarding date cutoffs, despite known issues with news retrieval systems.", "reasoning": "The reviewer identifies a technical vulnerability where unreliable date enforcement could lead to data leakage, undermining the paper's core claim of preventing leakage. The burden of proof is placed on the authors to validate their specific pipeline against known failure modes.", "judgment": "The paper has not ruled out contamination risks, weakening the validity of its leakage-prevention claims.", "valence": "negative", "suggested_improvement": "Test the AskNews pipeline against known date-cutoff reliability issues to rule out contamination.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7998297214508057, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8105111718177795}]}, {"review_id": "0DzALLkZve", "paper_id": "OWHKdYwYiF", "paper_title": "Towards Principled Dataset Distillation: A Spectral Distribution Perspective", "decision": "Reject", "summary": "The reviewer conducts a provenance audit, systematically dismantling the paper's novelty claims by tracing components to prior art, challenging the theoretical grounding of modifications, and questioning experimental completeness and computational efficiency.", "units": [{"unit_index": 0, "inspected_object": "Theoretical motivation regarding first-moment limitations of existing methods", "observation": "The reviewer acknowledges the paper's critique that many distribution matching methods use linear kernels lacking universality as a legitimate theoretical observation, but notes the paper fails to engage with methods addressing higher moments (e.g., M3D, IID) and only compares against NCFM on long-tailed CIFAR.", "reasoning": "A critique of a research area must be tested against the strongest existing alternatives rather than strawman first-moment methods; limited comparison is insufficient to justify sweeping claims about first-moment shortcomings.", "judgment": "The theoretical motivation is partially valid but inadequately supported by comparative evidence against stronger prior art.", "valence": "mixed", "suggested_improvement": "Engage with and compare against methods that address higher moments (such as M3D or IID) to substantiate the claim of first-moment limitations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8390343189239502, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8177211284637451}, {"unit_index": 1, "inspected_object": "Novelty and attribution of the proposed SDD metric", "observation": "The mathematical content of the SDD metric (expression of squared MMD via characteristic functions) corresponds to established theory (Corollary 4 in [6]) and prior usage under different names (CFD in [4,7,8]), yet the paper presents it without acknowledging this provenance.", "reasoning": "Presenting a known mathematical object under a new name without citing its origins creates a novelty deficit and gives the impression that the concepts were first proposed herein, violating norms of scholarly attribution.", "judgment": "The SDD metric lacks sufficient novelty due to inadequate citation of prior art, raising concerns about attribution and framing.", "valence": "negative", "suggested_improvement": "Cite prior works establishing the SDD metric's mathematical basis (e.g., [4,6,7,8]) to clarify provenance and avoid implying originality for known concepts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.853564441204071, "reasoning_key": "novelty_standard", "reasoning_sim": 0.817441463470459}, {"unit_index": 2, "inspected_object": "Class-specific weighting mechanism (alpha(c))", "observation": "The class-specific weighting alpha(c) is viewed as a trivial modification from scalar to vector parameters of an already established amplitude-phase decomposition, and its value is treated as a hyperparameter rather than being systematically determined.", "reasoning": "Modifications to theoretically motivated methods must themselves be theoretically justified; treating key parameters as hyperparameters undermines the 'principled' nature of the method, and it is unclear if this modification preserves the optimality guarantees of the original distribution matching objective.", "judgment": "The class-specific weighting mechanism is theoretically unsubstantiated and appears heuristic rather than principled.", "valence": "negative", "suggested_improvement": "Provide theoretical justification for the class-specific weighting and demonstrate whether it preserves the optimality guarantees of the original distribution matching objective.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8336878418922424, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8368558883666992}, {"unit_index": 3, "inspected_object": "Experimental design and baseline comparisons", "observation": "The paper focuses on high imbalance factors on long-tailed datasets, but lacks class-balanced comparisons, raising uncertainty about whether the method's benefits are general or specific to long-tailed settings; additionally, there are anomalies in Long-tailed ImageNet baseline numbers.", "reasoning": "To establish that the method's benefits are not limited to the long-tailed setting, class-balanced comparisons are necessary; internal consistency checks reveal potential reporting inconsistencies or cherry-picking.", "judgment": "The experimental evaluation is incomplete regarding generalization to class-balanced settings and contains suspicious data points requiring verification.", "valence": "negative", "suggested_improvement": "Include class-balanced comparisons to test general performance and verify the internal consistency of reported baseline numbers, particularly for Long-tailed ImageNet.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.9236176609992981, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8149454593658447}, {"unit_index": 4, "inspected_object": "Computational complexity of the SDD computation", "observation": "The SDD computation involves nested summation forms from Monte-Carlo sampling, which contrasts with claims of linear complexity.", "reasoning": "Practical viability depends on computational cost; the nested structure suggests complexity may be higher than claimed, warranting explicit analysis.", "judgment": "The computational efficiency claims are unverified and potentially overstated given the algorithmic structure.", "valence": "negative", "suggested_improvement": "Provide a formal complexity analysis of the SDD computation to validate claims of linear complexity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8226672410964966, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8309730887413025}]}, {"review_id": "0E0BC4mPCs", "paper_id": "69iBZ4DzXg", "paper_title": "Efficient Algorithms for Adversarially Robust Approximate Nearest Neighbor Search", "decision": "Reject", "summary": "The reviewer positively assesses the conceptual elegance of the fairness-robustness link and the technical breakthrough of Theorem 3, while raising concerns about the $sqrt{Q}$ parameter dependence, the justification of the adversary model, and the lack of empirical validation.", "units": [{"unit_index": 0, "inspected_object": "The claimed implication that fairness in ANN search implies adversarial robustness", "observation": "The reviewer identifies this connection as 'conceptually elegant and powerful' with potential generality beyond ANN.", "reasoning": "The reviewer evaluates contributions by their capacity to reorganize the conceptual landscape and serve as a transferable insight bridging separate research areas, rather than just immediate algorithmic payoff.", "judgment": "Positive assessment of the contribution's novelty and theoretical value.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8135127425193787, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7756370306015015}, {"unit_index": 1, "inspected_object": "Theorem 3's improvement upon prior work breaking the $sqrt{n}$ barrier", "observation": "The reviewer notes the result is strong and improves upon prior work under mild assumptions.", "reasoning": "The reviewer treats the breaking of the known $sqrt{n}$ barrier as a meaningful benchmark of algorithmic progress, accepting the mild assumptions as non-trivial but acceptable.", "judgment": "Positive assessment of technical achievement relative to known limitations.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.829751193523407, "reasoning_key": "design_justification", "reasoning_sim": 0.7908856868743896}, {"unit_index": 2, "inspected_object": "The $sqrt{Q}$ factor in Theorems 2 and 3 regarding query-count dependence", "observation": "The reviewer flags the $sqrt{Q}$ factor as significant when adaptive queries are large, noting Theorem 1 avoids this but has data density dependence.", "reasoning": "Algorithmic guarantees should be examined for scaling with all relevant parameters; the presence of this factor represents a limitation in parameter regimes, posing an open question about whether it can be eliminated generally.", "judgment": "Negative/Concern regarding scalability and parameter dependence.", "valence": "negative", "suggested_improvement": "Clarify or eliminate the $sqrt{Q}$ dependence in the general case.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.828353226184845, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8141156435012817}, {"unit_index": 3, "inspected_object": "The adversary model where the dataset is fixed but queries are chosen adaptively", "observation": "The reviewer states the paper does not provide sufficient justification for this threat model and questions its natural appearance in practice.", "reasoning": "Threat models require motivational justification; the burden is on the authors to explain why the specific adversary model is natural or realistic.", "judgment": "Negative concern regarding lack of practical grounding for the model.", "valence": "negative", "suggested_improvement": "Provide concrete examples or justification for why this adversary model arises naturally in practice.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8091503977775574, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8013624548912048}, {"unit_index": 4, "inspected_object": "The absence of empirical experiments in a theoretical paper", "observation": "The reviewer notes the lack of experiments and asks if the authors intend to conduct empirical validation.", "reasoning": "Even theoretical algorithms should demonstrate practical viability or address the gap in empirical validation; the reviewer expects some acknowledgment of this limitation.", "judgment": "Negative/Neutral concern regarding completeness and practical relevance.", "valence": "negative", "suggested_improvement": "Address the absence of experiments or clarify intentions regarding empirical validation.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "novelty", "object_sim": 0.8071502447128296, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8298752307891846}, {"unit_index": 5, "inspected_object": "Presentation quality and minor errors", "observation": "The reviewer flags a typo on Page 4, Line 204 but rates presentation highly (4).", "reasoning": "Minor surface-level errors do not significantly detract from overall clarity, and the paper is otherwise well-written.", "judgment": "Positive assessment of presentation quality despite minor issues.", "valence": "positive", "suggested_improvement": "Correct the identified typo.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.873436450958252, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7817273139953613}]}, {"review_id": "0E594h5RAW", "paper_id": "rhV6QTMqq1", "paper_title": "Token-Regulated Group Relative Policy Optimization for Stable Reinforcement Learning in Large Language Models", "decision": null, "summary": "The reviewer conducts a priority-based gatekeeping assessment, identifying significant overlap with a prior arXiv paper across phenomenon, method, and experiments. The review concludes that the submission fails to demonstrate novelty or differentiation, leading to a negative judgment on contribution and an escalation to ethics review for potential integrity issues.", "units": [{"unit_index": 0, "inspected_object": "Submission's novelty claim relative to arXiv:2505.12929", "observation": "The submission presents a problem, method, and experiments that are highly similar in phenomenon, motivation, claims, experimental regimes, and structure to the prior work arXiv:2505.12929.", "reasoning": "The reviewer identifies specific overlaps in core phenomenon (low-probability tokens dominate updates), motivation, experimental regimes (K&K, Minerva/MATH, AIME/AMC), and methodological structure/background. The absence of any citation or discussion of this prior work is treated as a failure to demonstrate novelty relative to it.", "judgment": "The submission fails to establish its contribution as new because it does not position itself against or differentiate from the identified prior work.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8460791707038879, "reasoning_key": "design_justification", "reasoning_sim": 0.8557353019714355}, {"unit_index": 1, "inspected_object": "Methodological equivalence with Advantage Reweighting/Lopti", "observation": "TR-GRPO's token weighting may be functionally similar to Advantage Reweighting and the two-stage Lopti schedule described in the prior work.", "reasoning": "The reviewer notes that functional similarity without explicit comparison is insufficient; the burden is on the submission to demonstrate either non-equivalence or strict improvements over the prior method.", "judgment": "The submission's method is evaluated as potentially redundant rather than novel due to lack of differentiation.", "valence": "negative", "suggested_improvement": "Demonstrate equivalence or strict improvements compared to the prior work.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8449001908302307, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8008383512496948}, {"unit_index": 2, "inspected_object": "Experimental overlap and head-to-head comparison", "observation": "The submission uses experimental regimes close to those of the prior work but lacks a direct comparison between the two methods.", "reasoning": "Without head-to-head comparison, the submission's claimed gains could be redundant with the prior work's results, making the claims of outperforming GRPO uninformative in the context of the prior art.", "judgment": "The experimental evaluation is deficient for failing to distinguish the submission's performance from the prior work.", "valence": "negative", "suggested_improvement": "Perform head-to-head comparison with the prior work.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.872825026512146, "reasoning_key": "fair_comparison", "reasoning_sim": 0.774349570274353}, {"unit_index": 3, "inspected_object": "Research integrity regarding plagiarism/dual submission", "observation": "The submission's background, motivation, proof, methodology, and code are 'highly similar' to the prior work without detailing the relationship and differences.", "reasoning": "The severity of the structural and content similarities, combined with the lack of acknowledgment, raises concerns about research integrity, warranting escalation beyond technical critique.", "judgment": "The submission exhibits potential integrity issues requiring ethics review.", "valence": "negative", "suggested_improvement": "Detail the relationship and differences with the prior work to address integrity concerns.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8172634243965149, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7682404518127441}]}, {"review_id": "0ERkvkosjK", "paper_id": "I5Qgn23qAz", "paper_title": "Robust Domain Generalization under Divergent Marginal and Conditional Distributions", "decision": "Reject", "summary": "The reviewer performs a coherence check on the paper's theoretical claims, identifying three main gaps: a formal incompleteness regarding an undefined symbol, a lack of rigorous operationalization linking the Wasserstein bound to the heuristic loss, and a novelty deficit arising from the use of standard components. While the decomposition is praised for clarity, the overall theoretical justification is deemed weak.", "units": [{"unit_index": 0, "inspected_object": "Theoretical decomposition separating prior shift from feature shift", "observation": "The reviewer finds the decomposition clean and interpretable.", "reasoning": "The reviewer acknowledges the clarity of the separation between prior and feature shifts, accepting this structural aspect as well-executed.", "judgment": "Positive assessment of the decomposition's clarity and interpretability.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8115625381469727, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8341459631919861}, {"unit_index": 1, "inspected_object": "Definition of mathematical object π in Theorem 1", "observation": "The symbol π is undefined in the theorem statement.", "reasoning": "A theoretical paper must be self-contained at the level of notation; an undefined term in a central theorem constitutes a formal completeness failure that undermines the claim's evaluability.", "judgment": "Negative assessment due to a basic formal defect.", "valence": "negative", "suggested_improvement": "Define π in Theorem 1.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.734001874923706, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7943009734153748}, {"unit_index": 2, "inspected_object": "Connection between the Wasserstein bound and the implemented Domain Alignment (DA) loss", "observation": "The theory motivates minimizing a Wasserstein feature distance, but the actual loss is a heuristic contrastive loss aligning features with class centroids.", "reasoning": "The link between the theoretical bound and the training objective is qualitative rather than quantitative; if the method only approximates the bound heuristically, the theory may not provide a principled guarantee for the specific algorithm used.", "judgment": "Negative assessment of the theory-method alignment, characterizing the theory as potentially decorative rather than load-bearing.", "valence": "negative", "suggested_improvement": "Derive the DA loss from the Wasserstein bound more rigorously or weaken the theoretical claims to match the heuristic nature of the loss.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8262889385223389, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7682409286499023}, {"unit_index": 3, "inspected_object": "Novelty of the proposed components and combination", "observation": "Domain alignment, meta-learning, Wasserstein bounds, and InfoNCE decomposition are identified as well-established or standard tools.", "reasoning": "The contribution is reduced to combining known tools with theoretical justification; unless the combination yields new insight beyond the sum of its parts, it does not meet the bar for novelty.", "judgment": "Negative assessment of novelty, viewing the contribution as an additive combination of standard techniques rather than a transformative one.", "valence": "negative", "suggested_improvement": "Strengthen the novelty argument by showing why the specific combination is non-obvious or yields new predictions.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.8364522457122803, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8222508430480957}]}, {"review_id": "0EawWhJ5pg", "paper_id": "voMeZVAkKL", "paper_title": "FAST‑DIPS: Adjoint‑Free Analytic Steps and Hard‑Constrained Likelihood Correction for Diffusion‑Prior Inverse Problems", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper as theoretically sound and novel but critiques the scope of its claims, specifically regarding the 'adjoint-free' framing, nonlinear extensions, parameter sensitivity, and dataset generality. The logic centers on distinguishing between mathematical validity and communicative accuracy, demanding clearer boundary conditions for the method's advantages.", "units": [{"unit_index": 0, "inspected_object": "The method's claim of being 'adjoint-free' and its pre-computation of $(A A^T)^{-1} A$", "observation": "The reviewer observes that the method avoids per-iteration adjoint computations by pre-computing a term involving $(A A^T)^{-1}$, shifting cost to a one-time setup.", "reasoning": "The reviewer applies the standard that a methodological label should accurately describe the class of problems for which the method offers an advantage, not just the per-iteration computation. They infer that this pre-computation may become a bottleneck for certain operator classes (e.g., nonlinear or those where $A A^T$ is dense), potentially outweighing per-iteration speed-ups.", "judgment": "The 'adjoint-free' framing could be misleading regarding the method's scope and general applicability.", "valence": "negative", "suggested_improvement": "Specify the regime or classes of forward operators where the pre-computation is tractable and the method's advantage holds.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8203023672103882, "reasoning_key": "design_justification", "reasoning_sim": 0.8601514101028442}, {"unit_index": 1, "inspected_object": "The extension of the method to nonlinear problems via linearization", "observation": "The reviewer notes that the method linearizes the forward operator at each step for nonlinear problems like phase retrieval.", "reasoning": "The reviewer expects a methodological contribution to provide theoretical grounding or justification for approximations used in extensions. The absence of detail on why linearization is valid, when it might fail, or error analysis suggests a gap in rigor.", "judgment": "The nonlinear extension is under-justified and lacks necessary detail or justification despite being a reasonable approach.", "valence": "negative", "suggested_improvement": "Provide an explanation of why linearization is valid, discuss its limitations, or include error analysis.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8326631188392639, "reasoning_key": "design_justification", "reasoning_sim": 0.8263764977455139}, {"unit_index": 2, "inspected_object": "The choice of the epsilon parameter ($epsilon$) in the hard constraint", "observation": "The reviewer finds that the choice of $epsilon$ in the hard-constrained likelihood correction is not well explained.", "reasoning": "The reviewer assumes that optimization-based methods with free parameters should clarify how they are set (e.g., tuned per problem vs. fixed) and demonstrate robustness to their values to ensure transferability.", "judgment": "The parameter sensitivity is unclear, raising concerns about the method's practical utility and reliance on tuning.", "valence": "negative", "suggested_improvement": "Explain how $epsilon$ is selected and analyze the method's robustness to its value.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8305585980415344, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8134275674819946}, {"unit_index": 3, "inspected_object": "The switching threshold ($sigma_{switch}$) in the pixel-to-latent hybrid schedule", "observation": "The reviewer questions how the switching threshold is selected in practice and how performance varies with its value.", "reasoning": "The reviewer infers that the utility of the hybrid schedule depends on whether the switching point is robust or requires per-problem tuning, seeking practical guidance and sensitivity analysis.", "judgment": "The hybrid schedule's practical usability is uncertain due to lack of clarity on parameter selection and sensitivity.", "valence": "conditional", "suggested_improvement": "Provide practical guidance on selecting the threshold and show performance variation across different values.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8252343535423279, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7776120901107788}, {"unit_index": 4, "inspected_object": "The exclusive use of the FFHQ dataset for empirical evaluation", "observation": "The reviewer notes that the paper only uses FFHQ, a human face dataset, for experiments.", "reasoning": "The reviewer argues that face images have specific statistical properties (structured, aligned, narrow manifold) that might allow the method to exploit domain-specific priors, limiting the generality of conclusions to other domains.", "judgment": "The empirical scope is too narrow to support claims of general applicability; the results may be domain-specific artifacts.", "valence": "negative", "suggested_improvement": "Include at least one additional dataset (e.g., medical images or natural scenes) to demonstrate generalizability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8397620320320129, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7809074521064758}, {"unit_index": 5, "inspected_object": "The density and technicality of the method section", "observation": "The reviewer describes the method section as dense, overly technical, relying heavily on acronyms and derivations with minimal intuition.", "reasoning": "The reviewer implies that high density hinders comprehension and verification, potentially obscuring the method's boundaries and making full assessment difficult.", "judgment": "The presentation style creates a barrier to understanding and evaluating the method's core contributions.", "valence": "negative", "suggested_improvement": "Improve clarity by adding intuition and reducing reliance on dense mathematical derivations and acronyms.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.8334395885467529, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8044233322143555}, {"unit_index": 6, "inspected_object": "The theoretical soundness and novelty of the core idea", "observation": "The reviewer acknowledges the paper is theoretically sound and that the combination of analytic steps with hard-constrained correction is novel and effective.", "reasoning": "The reviewer accepts the validity of the proofs and the efficacy of the core mechanism based on the provided arguments and empirical support.", "judgment": "The core contribution is solid, novel, and theoretically grounded.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8699886202812195, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8267419934272766}]}, {"review_id": "043oR1P40C", "paper_id": "UqjoMDJCmM", "paper_title": "Real-time Routing under Partial Observability: Information-Efficient Policies for Connected Vehicles", "decision": null, "summary": "The reviewer employs a clarification-seeking strategy, identifying underspecification in terminology, input definitions, training objectives, and architectural justifications. The dominant evaluative activity is withholding full endorsement until these gaps are resolved, treating clarity as a prerequisite for verifying experimental validity and claims.", "units": [{"unit_index": 0, "inspected_object": "Core terminology and evaluation metrics (utility, MAP, routing head, travel time, delay, waiting time)", "observation": "The paper fails to define basic evaluation quantities such as travel time, delay, and waiting time.", "reasoning": "A paper must be self-contained enough for a reader outside the immediate subfield to follow the evaluation without external references; undefined terms prevent interpretation of experimental results.", "judgment": "The exposition is insufficiently explicit regarding the shared vocabulary required for scientific assessment.", "valence": "negative", "suggested_improvement": "Define core metrics and terminology explicitly to establish a shared vocabulary with the audience.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7967338562011719, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7474780678749084}, {"unit_index": 1, "inspected_object": "Input specification and data flow", "observation": "The model's inputs, their types (time-series vs. one-shot), temporal structure, and multi-vehicle data handling are not fully specified.", "reasoning": "Without a complete inventory of inputs and decision-making structure, it is impossible to verify how the model processes information or whether reported improvements are meaningful.", "judgment": "The input specification is ambiguous, creating uncertainty about the model's operational mechanics.", "valence": "negative", "suggested_improvement": "Provide a complete inventory of the model's inputs, their types, and the temporal structure of decision-making.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8252028822898865, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8266274929046631}, {"unit_index": 2, "inspected_object": "Training objective versus reported metrics", "observation": "It is unclear which of the three evaluation metrics (travel time, delay, waiting time) is actually optimized during training.", "reasoning": "If the authors do not state the training objective, the reader cannot know whether reported improvements on non-optimized metrics are meaningful or incidental, undermining experimental validity.", "judgment": "There is a potential mismatch between the training objective and reported metrics that requires clarification to interpret results.", "valence": "conditional", "suggested_improvement": "Explicitly state which metric is optimized during training to link the objective to the reported results.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8277900218963623, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8317757248878479}, {"unit_index": 3, "inspected_object": "Uncertainty-aware claim and MAP formulation", "observation": "The paper describes itself as 'uncertainty-aware' but does not explain how uncertainty is represented or managed, leaving ambiguity about whether MAP reflects a Bayesian formulation or an optimization heuristic.", "reasoning": "The reviewer frames this as a dichotomy: if Bayesian, the claim is substantiated; if a differentiable approximation/heuristic, the claim may be overstated. This ambiguity prevents full endorsement of the contribution.", "judgment": "The use of uncertainty-related terminology is loose or underspecified, requiring clarification to assess the validity of the 'uncertainty-aware' claim.", "valence": "negative", "suggested_improvement": "Clarify whether MAP is derived from a Bayesian formulation or is a differentiable approximation to a discrete optimization problem.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8041139841079712, "reasoning_key": "design_justification", "reasoning_sim": 0.8334805965423584}, {"unit_index": 4, "inspected_object": "Transformer architecture integration", "observation": "It is unclear whether the pipeline is conceptually inspired by the Transformer architecture or directly incorporates a Transformer-based spatio-temporal encoder.", "reasoning": "The reviewer probes the intellectual genealogy to determine if there is a principled reason for choosing a Transformer (e.g., suitability for sparse spatio-temporal data) or if it was adopted without deeper justification.", "judgment": "The scientific motivation for the architectural choice is insufficiently explained.", "valence": "negative", "suggested_improvement": "Explain the principled reason for choosing the Transformer architecture, distinguishing between conceptual inspiration and direct incorporation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7062534093856812, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7555975914001465}, {"unit_index": 5, "inspected_object": "Baseline naming and comparability", "observation": "The baseline 'Proximity–STGCN' may refer to both Proximity–STGCN–Dijkstra and Proximity–STGCN–A*, raising questions about whether the simplified naming implies identical outcomes.", "reasoning": "Precision in naming is required to ensure comparisons are apples-to-apples and to maintain transparency about what was actually compared.", "judgment": "The baseline naming convention obscures potentially important distinctions, reducing transparency.", "valence": "negative", "suggested_improvement": "Clarify baseline naming to distinguish between different solvers (e.g., Dijkstra vs. A*) and ensure transparent reporting of comparisons.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8505154848098755, "reasoning_key": "construct_validity", "reasoning_sim": 0.7445204854011536}, {"unit_index": 6, "inspected_object": "Citation style and typographical errors", "observation": "The review identifies issues with citation style (citep vs citet) and typos.", "reasoning": "These minor presentation issues contribute to the overall low presentation score and suggest a lack of thoroughness in final polishing.", "judgment": "Presentation quality is diminished by avoidable stylistic and typographical errors.", "valence": "negative", "suggested_improvement": "Correct citation style inconsistencies and fix typographical errors to improve presentation quality.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8294752836227417, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7973626255989075}]}, {"review_id": "04ksuzl8fU", "paper_id": "nZ7fUX1cHa", "paper_title": "Learning-Based Autonomy from Kernel-Embedded Multi-modal Fusion to Feedback Control", "decision": "Reject", "summary": "The reviewer performs a completeness audit, identifying multiple gaps in definitions, proofs, interpretations, and structural coherence. The dominant logic is that unexplained or unsubstantiated claims prevent evaluation, rendering them non-contributory regardless of potential merit.", "units": [{"unit_index": 0, "inspected_object": "Definition and mechanism of multi-modal observations via kernel mean embeddings", "observation": "The reviewer finds it unclear what 'multi-modal observations' are and how kernel mean embeddings fuse them, reporting a breakdown in following the proposed mechanism.", "reasoning": "The reviewer expects a paper to define its terms and explain its mechanisms clearly; failure to do so prevents understanding of the method's operation and feasibility.", "judgment": "The presentation of the core mechanism is insufficiently clear for evaluation.", "valence": "negative", "suggested_improvement": "Clarify how multi-modal embeddings work and what approximating value functions via RKHS embeddings promises.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8216602206230164, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7984088063240051}, {"unit_index": 1, "inspected_object": "Notation consistency and symbol definitions (noise variables)", "observation": "The reviewer notes that epsilon_t is introduced without a distribution, omega is described as process noise but never appears in the system definition, and later takes on a different meaning.", "reasoning": "Notation should be consistent and every symbol defined where introduced; treating notation as a contract with the reader, breaches indicate lack of rigor.", "judgment": "The paper exhibits internal inconsistency in its notation.", "valence": "negative", "suggested_improvement": "Ensure all symbols are defined consistently upon introduction.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8157389163970947, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7650949954986572}, {"unit_index": 2, "inspected_object": "Discussion of related literature", "observation": "The reviewer flags the absence of a discussion of related literature as a standalone weakness.", "reasoning": "A paper should situate itself within a research landscape, not merely cite sources; the absence suggests a failure to contextualize the contribution.", "judgment": "The paper lacks necessary scholarly infrastructure.", "valence": "negative", "suggested_improvement": "Include a discussion of related work to situate the paper within the research landscape.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8023245930671692, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8260653614997864}, {"unit_index": 3, "inspected_object": "Safety guarantees (Lipschitz-based closeness and control barrier functions)", "observation": "The paper asserts safety guarantees without any reference or proof, and equations (1) and (2) seem to be the same.", "reasoning": "Safety guarantees are a central promised contribution and must be substantiated with derivation or reference; their absence is treated as a fatal gap for such claims.", "judgment": "The safety claims are unsupported and presented redundantly.", "valence": "negative", "suggested_improvement": "Provide proof or reference for the Lipschitz-based closeness and control barrier function guarantees.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8385828733444214, "reasoning_key": "novelty_standard", "reasoning_sim": 0.677916944026947}, {"unit_index": 4, "inspected_object": "Experimental reporting and interpretation", "observation": "There are too few details, no discussion of results, tables lack captions, and it is unclear what the takeaways are from plots and tables.", "reasoning": "Experimental results should be interpreted, not merely displayed; without captions and discussion, the reader cannot determine what is being compared or evaluated.", "judgment": "The experimental evidence is presented but not interpreted, leaving analytical scaffolding absent.", "valence": "negative", "suggested_improvement": "Discuss the results and provide captions for tables to clarify comparisons and takeaways.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.824944257736206, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7899325489997864}, {"unit_index": 5, "inspected_object": "Structural coherence and problem formulation", "observation": "Sections 4.3 and 4.4 seem to introduce new problem definitions rather than sticking to the original one.", "reasoning": "The reviewer expects narrative unity and a single problem formulation to carry through the paper; reformulations disrupt this unity.", "judgment": "The paper suffers from structural incoherence due to shifting problem definitions.", "valence": "negative", "suggested_improvement": "Stick to the originally defined problem formulation throughout the paper.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.860053300857544, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7620471715927124}, {"unit_index": 6, "inspected_object": "Scope delivery vs. stated promises", "observation": "The paper promises discussions of kernel choice and robustness that never materialize.", "reasoning": "A paper should deliver on the scope it announces in its own text; failing to do so creates a mismatch between stated scope and delivered content.", "judgment": "The paper fails to deliver on its announced scope.", "valence": "negative", "suggested_improvement": "Deliver the promised discussions on kernel choice and robustness.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7340074777603149, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7971433997154236}, {"unit_index": 7, "inspected_object": "Typesetting error (Schölkopf's name)", "observation": "There is a LaTeX error with Schölkopf's name.", "reasoning": "Presentation quality is part of scholarly rigor; minor errors contribute to a broader pattern of under-polished execution.", "judgment": "The paper has minor presentation failures.", "valence": "negative", "suggested_improvement": "Correct typographical and formatting errors.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.6581064462661743, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8702566623687744}]}, {"review_id": "07Dylk1ok3", "paper_id": "xKRA2pkUy7", "paper_title": "The Linear Geometry of Moral Choice in LLMs", "decision": null, "summary": "The reviewer conducts a scope-testing evaluation, accepting the core methodology but demanding evidence for generalizability across moral framings, dataset sizes, network layers, and stylistic variations.", "units": [{"unit_index": 0, "inspected_object": "Title scope vs. empirical basis (single moral framing)", "observation": "The framework rests on a single moral framing (personal vs. impersonal trolley-style dilemmas), which contrasts with the title's promise of 'The Linear Geometry of Moral Choice.'", "reasoning": "The reviewer holds a norm that the artifact's packaging should match its empirical breadth; using only one framing type suggests the method may be narrow and not generalize to other cases.", "judgment": "Scope mismatch: the paper claims generality but demonstrates it on a narrow construct.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "problem_framing", "object_sim": 0.824072539806366, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7920024394989014}, {"unit_index": 1, "inspected_object": "Dataset size robustness (20 prompt pairs)", "observation": "The study uses only 20 prompt pairs without explicit justification for this number or analysis of its effect on results.", "reasoning": "In empirical NLP, dataset construction should be motivated, and robustness to dataset size is a relevant validity check; treating this as a given risks fragility.", "judgment": "Methodological opacity regarding data selection limits confidence in result stability.", "valence": "negative", "suggested_improvement": "Provide selection rationale and analyze the effect of changing the number of prompt pairs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8667346239089966, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8039706945419312}, {"unit_index": 2, "inspected_object": "Steering mechanism diagnostic depth (layer-wise separability)", "observation": "The paper provides some insights into layers and token positions but lacks additional cues on the separability of moral framings across layers.", "reasoning": "A paper about representation geometry should demonstrate where in the network the relevant structure lives, not merely that it exists; visualizing separation across layers confirms if the direction is conceptual rather than stylistic.", "judgment": "Under-explained mechanism: the location and emergence of the moral direction are not sufficiently detailed.", "valence": "negative", "suggested_improvement": "Provide visualizations or tables of separability by layer to show where moral framings start to separate.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8070777058601379, "reasoning_key": "design_justification", "reasoning_sim": 0.7929122447967529}, {"unit_index": 3, "inspected_object": "Stylistic robustness of identified directions", "observation": "It is unclear whether identified directions remain stable when emotionally loaded verbs are replaced with alternatives.", "reasoning": "If a latent direction tracks semantic moral content rather than surface syntax, it should be robust to stylistic variation; demonstrating this requires a counterfactual experiment.", "judgment": "Unverified assumption: the directionality might be driven by stylistic features rather than pure moral semantics.", "valence": "conditional", "suggested_improvement": "Perform a verb-substitution experiment to test if directions remain stable under stylistic shifts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8229061961174011, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7725880146026611}, {"unit_index": 4, "inspected_object": "Literature positioning relative to recent work", "observation": "The paper does not adequately situate itself within the recent representation-engineering canon, specifically missing citation of Arditi et al. (NeurIPS 2024).", "reasoning": "Authors are expected to track and engage with the field's recent output to properly frame their contribution's novelty and context.", "judgment": "Incomplete contextualization: the paper appears disconnected from immediate prior art.", "valence": "negative", "suggested_improvement": "Expand the literature review to include recent relevant works such as Arditi et al.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.9116254448890686, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7842251062393188}]}, {"review_id": "081UfRwXZO", "paper_id": "Y1UhUy7qAQ", "paper_title": "One Shot, One Kill: Attacking Video Object Segmentation with a Single Frame", "decision": null, "summary": "The reviewer conducts a specification-and-generalization audit, accepting the method's plausibility but demanding clearer operational definitions, broader architectural testing, and better situational framing to validate generalization and contribution claims.", "units": [{"unit_index": 0, "inspected_object": "OCI mechanism's temporal structure", "observation": "Uncertainty regarding whether trigger positions are independent per frame or optimized as a centroid across the entire dataset.", "reasoning": "The method's operational logic is ambiguous; it is unclear if the attack exploits per-frame object dynamics or a static spatial prior, which affects reproducibility and soundness.", "judgment": "Methodological underspecification leading to uncertainty about the attack's definitive behavior.", "valence": "negative", "suggested_improvement": "Clarify whether OCI uses per-frame or dataset-level centroids for trigger positioning.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7576086521148682, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7869274616241455}, {"unit_index": 1, "inspected_object": "Interaction between OCI and TRP strategies", "observation": "Ambiguity on whether OCI and TRP are mixed in the data or superimposed on a single image.", "reasoning": "The data construction method is unspecified, making it difficult to interpret ablation results or reproduce the combined strategy.", "judgment": "Lack of clarity in experimental setup reduces confidence in the reported results.", "valence": "negative", "suggested_improvement": "Specify how OCI and TRP are combined in the training data.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7825292348861694, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.771248996257782}, {"unit_index": 2, "inspected_object": "Model zoo (AOT and DeAOT)", "observation": "AOT and DeAOT are architecturally similar, limiting the demonstrated breadth of the attack.", "reasoning": "Testing on near-duplicate architectures does not support claims of generalizability across different VOS paradigms (e.g., memory-based vs. transformer-based).", "judgment": "Experimental scope is insufficient to validate generalization claims.", "valence": "negative", "suggested_improvement": "Test the attack on structurally diverse models such as XMem and OneVOS.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7832115888595581, "reasoning_key": "design_justification", "reasoning_sim": 0.8214296698570251}, {"unit_index": 3, "inspected_object": "Real-world use cases", "observation": "The paper does not clearly explain the specific scenarios in which this backdoor attack would exist or be deployed.", "reasoning": "Without articulating the adversary's motivation and context, the threat model lacks real-world anchoring and credibility.", "judgment": "Diminished contribution due to lack of situational relevance.", "valence": "negative", "suggested_improvement": "Articulate who would deploy such an attack and why, providing concrete real-world use cases.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7793877124786377, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.766486644744873}, {"unit_index": 4, "inspected_object": "Related security literature", "observation": "Absence of engagement with adjacent adversarial-VOS literature, specifically citing an ACMMM 2023 paper on one-shot adversarial attacks.", "reasoning": "Failure to position the work against prior art undermines the novelty claim of being the 'first' backdoor attack on VOS.", "judgment": "Novelty claim is weakened by missing contextualization.", "valence": "negative", "suggested_improvement": "Engage with related adversarial-VOS literature to clarify the paper's distinct contribution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.7767564654350281, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7793686985969543}]}, {"review_id": "088tpavWra", "paper_id": "2cq8FyBfDk", "paper_title": "ProteinVista: A compute-efficient atom-level 3D CNN that outperforms sequence transformers in protein–ligand prediction", "decision": "Reject", "summary": "The reviewer evaluates the paper primarily on novelty grounds, concluding that the use of established voxel/CNN representations makes the contribution incremental despite engineering scale. This is reinforced by critiques of missing structure-aware baselines and incomplete related work coverage, alongside a technical question regarding SE(3) invariance.", "units": [{"unit_index": 0, "inspected_object": "The paper's contribution relative to prior work (VoxMol, DeepSite, EnzyNet)", "observation": "The core representational choice (voxelization + 3D CNN) is already established in prior literature for various tasks.", "reasoning": "If the fundamental modeling principle is not new, then scaling with more data and modern augmentation constitutes an incremental contribution rather than a fundamental one.", "judgment": "The contribution is limited/incremental.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8297341465950012, "reasoning_key": "novelty_standard", "reasoning_sim": 0.774723470211029}, {"unit_index": 1, "inspected_object": "Experimental baselines", "observation": "The paper compares only against sequence-based models (ESM-2, MolFormer) and omits structure-aware baselines such as MaSIF and Umol.", "reasoning": "Without comparing against other structure-aware methods, the paper cannot substantiate its claim that full-atom 3D CNNs are superior for structure-dependent tasks, especially given the paper's framing implies superiority over structure-aware approaches.", "judgment": "Methodological gap; claims not fully substantiated.", "valence": "negative", "suggested_improvement": "Discuss how the approach relates to structure-aware baselines like MaSIF and Umol and justify their exclusion or inclusion.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.899196982383728, "reasoning_key": "design_justification", "reasoning_sim": 0.8498162031173706}, {"unit_index": 2, "inspected_object": "Related work coverage", "observation": "The related work section is inadequate and omits many protein-level models using 3D geometric information; there is no dedicated section for discussion of related works.", "reasoning": "Scholarship requires acknowledging and engaging with prior work that uses similar representational choices to contextualize the current contribution.", "judgment": "Incomplete scholarship/presentation issue.", "valence": "negative", "suggested_improvement": "Add a dedicated related-work section discussing prior models like VoxMol, DeepSite, and EnzyNet.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.7854881882667542, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7570741176605225}, {"unit_index": 3, "inspected_object": "SE(3) invariance/equivariance design choice", "observation": "The voxel-based method is neither invariant nor equivariant for the SE(3) group, yet this choice is not explicitly justified against alternatives like equivariant architectures.", "reasoning": "In geometric deep learning, equivariance is a desirable property for processing 3D structures; deviations from this norm require justification, especially when contrastive objectives or augmentations are used as alternatives.", "judgment": "Design choice is not obviously justified; raises technical questions.", "valence": "conditional", "suggested_improvement": "Explain why a non-equivariant approach was chosen and whether equivariant architectures were considered.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.8356994986534119, "reasoning_key": "design_justification", "reasoning_sim": 0.8226099610328674}]}, {"review_id": "08HowlqkKR", "paper_id": "aiJtxcNbqB", "paper_title": "Decoupling Shared and Modality-Specific Subspaces in Multimodal Learning via Low-Rank Representation Fine-Tuning", "decision": "Reject", "summary": "The reviewer systematically audits the paper's evidentiary basis, moving from recognizing conceptual promise to flagging insufficient justification for design choices, procedural ambiguities, and untested assumptions about scalability and robustness. The dominant evaluative activity involves demanding stronger theoretical/empirical warrants for constraints and clarifications for experimental procedures.", "units": [{"unit_index": 0, "inspected_object": "Independence and orthogonality constraints in the disentanglement objective", "observation": "The reviewer notes that these constraints are introduced as intuitively motivated but finds the justification insufficient.", "reasoning": "The reviewer questions whether enforcing both types of constraints is necessary or redundant, applying a standard that design choices require stronger epistemic warrant (theoretical or empirical) beyond intuition.", "judgment": "The design choice is under-argued and its necessity is uncertain.", "valence": "negative", "suggested_improvement": "Elaborate on the motivation for enforcing both independence and orthogonality constraints; justify their necessity or perform ablation studies to differentiate their effects.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8264841437339783, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8144600987434387}, {"unit_index": 1, "inspected_object": "Loss function and training pipeline specification", "observation": "The reviewer identifies specific gaps in describing the experimental procedure, including unclear details on fine-tuning timing, modality fusion, convergence criteria, and the cross-modal mutual information loss.", "reasoning": "Procedural ambiguity prevents reconstruction of the method's logic and verification of claimed contributions; without clear description, it is impossible to assess whether ablations were performed on individual components.", "judgment": "The methodological description is under-specified, leaving component contributions unverified.", "valence": "negative", "suggested_improvement": "Clarify the training pipeline details (fine-tuning, fusion, convergence, specific losses) and provide ablation studies on individual loss terms.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8353288769721985, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7875105738639832}, {"unit_index": 2, "inspected_object": "Scalability to multi-modal setups", "observation": "The paper evaluates only bimodal setups, and the formulation relies on pairwise interactions.", "reasoning": "The reviewer infers that the combinatorial growth of pairwise interactions suggests the method does not scale naturally beyond two modalities, challenging the claim of general applicability.", "judgment": "The method's scalability to more than two modalities is uncertain and potentially limited by combinatorial complexity.", "valence": "negative", "suggested_improvement": "Demonstrate or discuss how the method scales to settings with more than two modalities.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8123627305030823, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8076587915420532}, {"unit_index": 3, "inspected_object": "Dominant modality assumption in experiments", "observation": "Experiments appear to assume the relative importance of each modality is known beforehand, with dominant features underscored.", "reasoning": "The reviewer constructs a counterfactual where the task-relevant modality is not dominant to test robustness; this probes whether success conditions are too narrow.", "judgment": "The method's robustness when the dominant modality is unknown is untested and uncertain.", "valence": "negative", "suggested_improvement": "Test the method in scenarios where the task-relevant modality is not known or dominant beforehand.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8499813675880432, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8395448923110962}, {"unit_index": 4, "inspected_object": "Table 3a documentation regarding fused features", "observation": "Table 3a is identified as the only experiment where the dominant feature is not known beforehand, but the description of what 'fused' features means is unspecified.", "reasoning": "This lack of detail prevents assessment of the one experiment that could address the dominant modality concern, creating a gap in evidentiary support for robustness.", "judgment": "The documentation for the key robustness experiment is inadequate.", "valence": "negative", "suggested_improvement": "Clarify the definition and implementation of 'fused' features in Table 3a.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.771282970905304, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.794517457485199}, {"unit_index": 5, "inspected_object": "Fairness of baseline comparisons (DRIM-U)", "observation": "DRIM-U was adapted to a self-supervised variant without sufficient details provided.", "reasoning": "Unequal or unclear training conditions between the proposed method and baselines may invalidate reported improvements, undermining the epistemic validity of the empirical claims.", "judgment": "The fairness and transparency of baseline comparisons are questionable.", "valence": "negative", "suggested_improvement": "Provide detailed descriptions of baseline adaptations, specifically for DRIM-U, to ensure fair comparison.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8670544028282166, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8277701735496521}, {"unit_index": 6, "inspected_object": "Performance dependence on encoder quality", "observation": "The method builds on pretrained unimodal encoders, but the reviewer lacks evidence on how performance scales with encoder strength.", "reasoning": "If the method's value lies in adapting representations, its behavior under varying input quality is critical; saturation or poor performance with weak encoders would limit utility.", "judgment": "The robustness of the method across different encoder qualities is unverified.", "valence": "negative", "suggested_improvement": "Conduct sensitivity analysis showing performance scaling with respect to encoder quality.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8521459698677063, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8114315867424011}, {"unit_index": 7, "inspected_object": "Line 426 vs. loss function formulation", "observation": "The reviewer perceives a contradiction between line 426 and the stated loss function formulation.", "reasoning": "If independence loss prevents information leakage, it is unclear how fine-tuning can improve individual modality representations; this suggests a potential logical inconsistency in implementation or interpretation.", "judgment": "There is a potential internal inconsistency between the theoretical mechanism and reported outcomes.", "valence": "negative", "suggested_improvement": "Resolve the apparent contradiction between the independence loss formulation and the results described at line 426.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8053668141365051, "reasoning_key": "design_justification", "reasoning_sim": 0.8225460648536682}, {"unit_index": 8, "inspected_object": "Positioning within MI-based and PID-based literature", "observation": "The reviewer finds insufficient explicit comparison with related methods like FactorCL and CoMM.", "reasoning": "Without explicit discussion of similarities, differences, and complementarities, the novelty and specific contribution of MultiLoReFT remain unclear.", "judgment": "The paper's novelty and distinctiveness relative to existing literature are not sufficiently established.", "valence": "negative", "suggested_improvement": "Explicitly discuss similarities, differences, and potential complementarities with MI-based (FactorCL) and PID-based (CoMM) approaches.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8552169799804688, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8008061647415161}, {"unit_index": 9, "inspected_object": "Parameter efficiency and flexibility claims", "observation": "While parameter efficiency is noted as promising, there is limited empirical evidence on generalization across different pretrained representations or scaling to larger models.", "reasoning": "Claims of adaptability and efficiency require commensurate empirical support to be credible; current evidence is insufficient to validate broad generalizability.", "judgment": "The claims regarding generalization and scaling are under-supported by empirical evidence.", "valence": "mixed", "suggested_improvement": "Provide empirical evidence demonstrating how well the framework generalizes across different pretrained representations and scales to substantially larger models.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8652498126029968, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8286246657371521}]}, {"review_id": "098Jj66CH8", "paper_id": "qbCceo3FBE", "paper_title": "GOLDILOCS: GENERAL OBJECT-LEVEL DETECTION AND LABELING OF CHANGES IN SCENES", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper through a cost-benefit lens, accepting the method's viability but criticizing the lack of runtime evidence, marginal performance gains, and unclear architectural details. The logic prioritizes practical utility and rigorous ablation over mere accuracy rankings.", "units": [{"unit_index": 0, "inspected_object": "Method's reliance on large foundation models (SAM2, MASt3R, SAM)", "observation": "The method depends on two large foundation models, creating potential computational overhead.", "reasoning": "If the method is computationally expensive, it must demonstrate that its performance gains justify this cost; the absence of runtime comparisons leaves this trade-off unquantified and undermines the practical case for the method.", "judgment": "The paper lacks necessary empirical accountability regarding the efficiency-cost benefit analysis.", "valence": "negative", "suggested_improvement": "Provide runtime comparisons to quantify the overhead of the foundation models relative to baselines.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.848605215549469, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8648573756217957}, {"unit_index": 1, "inspected_object": "Magnitude of performance improvements in Table 4", "observation": "The performance improvements reported are marginal despite achieving best overall performance in most settings.", "reasoning": "Small gains combined with high computational costs result in a weak value proposition; the reviewer applies a standard of practical significance where marginal statistical wins do not justify the deployment effort.", "judgment": "The contribution is limited by the small magnitude of improvement relative to the complexity/cost.", "valence": "negative", "suggested_improvement": "Provide a stronger argument for why marginal gains are acceptable given the zero-shot advantage, or demonstrate larger empirical gains.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8582843542098999, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8069251179695129}, {"unit_index": 2, "inspected_object": "Robustness to reconstruction model quality", "observation": "It is unclear if the method's success depends heavily on the quality of the specific 3D reconstruction backbone used.", "reasoning": "A method built on external components should analyze how performance degrades with weaker components; lack of sensitivity analysis suggests potential brittleness and limits generalizability claims.", "judgment": "The method's robustness and modularity are unverified due to missing ablation/sensitivity data.", "valence": "negative", "suggested_improvement": "Conduct an ablation or sensitivity analysis varying the reconstruction backbone to isolate component contributions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8319702744483948, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8206878900527954}, {"unit_index": 3, "inspected_object": "Clarity of internal mechanics, specifically conflict resolution", "observation": "The description of the 'conflict resolution' stage in Section 4.2 is confusing and operationally undefined.", "reasoning": "Technical clarity is required for reproducibility and evaluation; vague descriptions of pipeline stages suggest incomplete documentation or conceptual ambiguity.", "judgment": "The presentation of key methodological steps lacks sufficient precision.", "valence": "negative", "suggested_improvement": "Clarify the operational meaning of the conflict resolution mechanism.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7906582951545715, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8467714190483093}, {"unit_index": 4, "inspected_object": "Architectural taxonomy (end-to-end vs. rule-based assembly)", "observation": "It is ambiguous whether the method is a unified learning system or a rule-based assembly of pre-trained components.", "reasoning": "The distinction between end-to-end learning and discrete pipeline assembly affects the assessment of novelty and contribution; suspicion of the latter lowers the perceived technical advancement.", "judgment": "The architectural nature of the contribution is unclear, weakening the claim of significant novelty.", "valence": "negative", "suggested_improvement": "Explicitly clarify whether the method is end-to-end or rule-based after segmentation/reconstruction.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7797825932502747, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8192594647407532}, {"unit_index": 5, "inspected_object": "Citation consistency across tables", "observation": "The citation for '3DGS-CD' differs between Table 1 and Table 2.", "reasoning": "Inconsistencies in scholarly apparatus signal sloppiness and reduce confidence in the rigor of the manuscript preparation.", "judgment": "The paper exhibits presentation errors that undermine its professional polish.", "valence": "negative", "suggested_improvement": "Correct citation inconsistencies to ensure uniformity across all tables and text.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8457351922988892, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8559862971305847}]}, {"review_id": "09liJZwVaY", "paper_id": "yDKawwfJ5O", "paper_title": "DeepEyesV2: Toward Agentic Multimodal Model", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper primarily through the lens of empirical rigor, specifically causal interpretability and fair comparison standards. While praising the diagnostic value of the ablation studies and tool-use analysis, the reviewer identifies critical gaps in methodological positioning against concurrent work, consistency of evaluation environments, and reproducibility via dataset release.", "units": [{"unit_index": 0, "inspected_object": "Comparison of RL-alone vs. two-stage cold-start-plus-RL training pipeline", "observation": "The reviewer finds the comparison interesting, insightful, and thorough, noting it isolates a variable to produce a clear finding.", "reasoning": "Good empirical work should demonstrate causal or quasi-causal understanding by isolating variables rather than just reporting performance gains.", "judgment": "Positive assessment of the experimental design's diagnostic value.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.830977201461792, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7632646560668945}, {"unit_index": 1, "inspected_object": "Pre-RL vs. post-RL tool-use pattern analysis", "observation": "The reviewer praises this analysis for showing change attributable to a specific intervention.", "reasoning": "Demonstrating change attributable to an intervention supports causal interpretation, which is the reviewer's implicit norm for good empirical work.", "judgment": "Positive assessment of the methodological clarity.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.828114926815033, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8048443794250488}, {"unit_index": 2, "inspected_object": "Positioning against concurrent work WebWatcher", "observation": "The paper cites WebWatcher's performance but fails to discuss its core methodologies, which seem very similar.", "reasoning": "A paper claiming novelty in methodology must position itself against the closest prior art methodologically, not just cite numbers; omission undermines the contribution claim.", "judgment": "Weakness: insufficient scholarly positioning regarding methodological novelty.", "valence": "negative", "suggested_improvement": "Discuss the methodological overlap with WebWatcher to clarify the source of novelty.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.7561464905738831, "reasoning_key": "novelty_standard", "reasoning_sim": 0.878884494304657}, {"unit_index": 3, "inspected_object": "Consistency of search API usage between baselines and proposed work", "observation": "Ambiguity exists regarding whether DeepEyesV2 and baselines use the same search APIs (e.g., SerpAPI vs. raw Google).", "reasoning": "Benchmark results are only meaningful when the environment is held constant; differing APIs could attribute gains to API quality rather than agentic capabilities.", "judgment": "Weakness: potential unfairness of comparison due to environmental inconsistency.", "valence": "negative", "suggested_improvement": "Clarify if the same search APIs were used for all models to ensure fair comparison.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8054009079933167, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7684001922607422}, {"unit_index": 4, "inspected_object": "Release of the curated training dataset", "observation": "The dataset is described as 'carefully curated' and fundamental, but its release status is unclear.", "reasoning": "Reproducibility is a prerequisite for contribution; without the dataset, results are hard to reproduce, diminishing the paper's value.", "judgment": "Condition: contribution is diminished if the dataset is not released.", "valence": "conditional", "suggested_improvement": "Release the curated training corpus to enable reproducibility.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8213302493095398, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8504304885864258}, {"unit_index": 5, "inspected_object": "Granular details of text search implementation and rate limiting", "observation": "Uncertainty about whether text search uses Google directly and whether rate limiting was encountered.", "reasoning": "Practical bottlenecks like rate limits affect reproducibility and system behavior; ambiguity here complicates the validity of the experimental setup discussed in Weakness 2.", "judgment": "Uncertainty requiring clarification for full verification.", "valence": "uncertain", "suggested_improvement": "Provide details on the text search backend and any rate-limiting measures taken.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7928548455238342, "reasoning_key": "robustness_norm", "reasoning_sim": 0.820905864238739}]}, {"review_id": "0A429OsXL1", "paper_id": "PiLfm0m2iK", "paper_title": "EcoXplain: An Interpretable Causal-Augmented Framework for Macroeconomic Forecasting", "decision": null, "summary": "The reviewer challenges the paper's core premise by arguing that quarterly macroeconomic data is too sparse to support complex causal modeling, suspecting overfitting and experimental flaws, and demanding external validation for causal claims alongside better literature coverage.", "units": [{"unit_index": 0, "inspected_object": "Data frequency and model complexity relationship", "observation": "Variables change only once per quarter, resulting in a small number of observation points over a decade.", "reasoning": "The low observation frequency significantly weakens the validity of 'causality' or 'causal aware' arguments inherent in the paper's complex modeling approach.", "judgment": "The causal framing is undermined by insufficient data granularity.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8281442523002625, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7471817135810852}, {"unit_index": 1, "inspected_object": "Input vector space characteristics", "observation": "The input vector space is described as quite small, resembling a simple set of strongly correlated economic variables.", "reasoning": "If the underlying structure can be captured by simpler techniques like time-lagged PCA, the necessity for the proposed complex non-linear model is questionable due to parsimony norms.", "judgment": "The contribution of the complex model is diminished if simpler baselines suffice.", "valence": "negative", "suggested_improvement": "Provide deeper understanding of why simpler models are not enough, potentially comparing against time-lagged PCA.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8285988569259644, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7753504514694214}, {"unit_index": 2, "inspected_object": "Causal signal validation", "observation": "The paper claims to extract causal signals but lacks external validation mechanisms.", "reasoning": "Causal claims in this context require auxiliary evidence from independent sources (e.g., news) to demonstrate correctness, rather than relying solely on internal consistency or predictive performance.", "judgment": "The causal claims are currently unsupported without triangulation.", "valence": "negative", "suggested_improvement": "Demonstrate auxiliary evidence of correctness of causal signals using external sources like news.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8060778975486755, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7533349394798279}, {"unit_index": 3, "inspected_object": "Experimental error variability", "observation": "MAPE and MAE variability across algorithms is very high.", "reasoning": "Such high error rates are implausible even for simplistic generalized linear models on this type of data, suggesting potential flaws in the experimental setup rather than genuine performance differences.", "judgment": "The experimental results are suspicious and likely reflect setup issues.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.836401641368866, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7828531265258789}, {"unit_index": 4, "inspected_object": "Train/test split methodology", "observation": "The volume of predicted data in training vs test sets is unclear.", "reasoning": "In time-series forecasting, random splits can artificially inflate performance; temporal appropriateness must be verified to rule out data leakage.", "judgment": "The validity of the performance metrics is uncertain due to ambiguous splitting strategy.", "valence": "uncertain", "suggested_improvement": "Comment on the volume of predicted data in training vs test and contrast with how the test split was done.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8148300051689148, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7759374976158142}, {"unit_index": 5, "inspected_object": "Related work coverage", "observation": "Potential omission of several non-linear time-series papers in the ML + Economics literature.", "reasoning": "Failure to engage with a relevant body of work suggests an incomplete positioning of the paper within the field.", "judgment": "The literature review is incomplete regarding non-linear time-series methods.", "valence": "negative", "suggested_improvement": "Review and cite relevant non-linear time-series papers in the ML + Economics literature.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "related_work", "object_sim": 0.7854881882667542, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7982974648475647}]}, {"review_id": "0Awg06Ossw", "paper_id": "x6c72680uD", "paper_title": "MeTA-LoRA: Data-Efficient Multi-Task Fine-Tuning for Large Language Models", "decision": "Reject", "summary": "The reviewer acknowledges the method's practical effectiveness and data efficiency but critically challenges the paper's theoretical grounding, experimental design (specifically unseen-task generalization), and ablation clarity. The dominant logic centers on the gap between the meta-learning framing and the limited empirical evidence provided.", "units": [{"unit_index": 0, "inspected_object": "The two-stage optimization design (MeTA-LoRA) and its theoretical grounding.", "observation": "The method is described as an adaptation of MAML into LoRA, but the paper lacks intuitive motivation or theoretical insight explaining why this specific two-stage aggregation works.", "reasoning": "Without a principled account connecting the meta-learning framing to observed behavior, the method appears as a 'clever engineering trick' rather than a contribution that illuminates multi-task adaptation principles, weakening the conceptual depth of the work.", "judgment": "The contribution is thinner than it could be; the paper fails to explain why the method works beyond demonstrating that it does.", "valence": "negative", "suggested_improvement": "Provide an intuitive or theoretical story that connects the meta-learning framing to the observed behavior.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8368444442749023, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8095470070838928}, {"unit_index": 1, "inspected_object": "Empirical validation scope, specifically the absence of unseen-task evaluation.", "observation": "The experiments train on and evaluate only on seen tasks, lacking evaluation on unseen tasks.", "reasoning": "The central promise of the meta-learning framework is rapid adaptation to new tasks. By not testing on unseen tasks, the experiments cannot demonstrate the generalization capability that justifies the meta-learning framing, regardless of performance numbers on seen tasks.", "judgment": "The experimental design is insufficient to support the paper's conceptual claims about meta-learning.", "valence": "negative", "suggested_improvement": "Include unseen-task evaluation to test whether the method enables rapid adaptation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8477345108985901, "reasoning_key": "design_justification", "reasoning_sim": 0.8454492688179016}, {"unit_index": 2, "inspected_object": "Data-efficiency claims versus training-dynamics confounds.", "observation": "The paper claims high data efficiency, but it is unclear if this advantage persists under matched data budgets where gradient update counts are controlled.", "reasoning": "Data efficiency can be an artifact of training longer on a small dataset (more gradient updates). Without controlling for training duration/updates, it is impossible to determine if the advantage comes from the meta-learning mechanism or trivial training dynamics.", "judgment": "The central claim of data efficiency is potentially confounded and unverified.", "valence": "negative", "suggested_improvement": "Perform comparisons under matched data budgets to isolate the effect of the mechanism from training duration.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8510887622833252, "reasoning_key": "design_justification", "reasoning_sim": 0.8380141854286194}, {"unit_index": 3, "inspected_object": "Ablation study clarity, specifically the –STA variant.", "observation": "It is ambiguous whether the –STA variant is effectively equivalent to standard LoRA.", "reasoning": "If removing the task-specific stage collapses the method into standard LoRA, the two-stage design's contribution is clear. If not, the ablation is confounded, and the interaction between stages may be driving success in ways not isolated by the current analysis.", "judgment": "The necessity and distinct contribution of the two-stage design are unclear due to ablation ambiguity.", "valence": "uncertain", "suggested_improvement": "Clarify if the –STA variant is equivalent to standard LoRA to isolate the contribution of the two-stage design.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7944380044937134, "reasoning_key": "design_justification", "reasoning_sim": 0.769490122795105}, {"unit_index": 4, "inspected_object": "Comparative landscape and positioning relative to HydraLoRA and LoRAHub.", "observation": "The paper compares primarily against LoRA-family methods and makes strong SoTA claims regarding HydraLoRA.", "reasoning": "HydraLoRA and LoRAHub innovate architecturally, while MeTA-LoRA innovates on data efficiency. The reviewer questions if these are competing on the same axis and if the paper has adequately mapped alternative approaches to data-efficient fine-tuning, suggesting potential overstatement of novelty or baseline strength.", "judgment": "The comparative context is narrow and potentially misaligned with prior architectural innovations.", "valence": "negative", "suggested_improvement": "Broaden comparisons to include non-LoRA data-efficient fine-tuning approaches and clarify the distinction from architectural contributions like HydraLoRA.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.757795512676239, "reasoning_key": "design_justification", "reasoning_sim": 0.7727245092391968}, {"unit_index": 5, "inspected_object": "Mechanistic claims regarding knowledge transfer and gradient conflict.", "observation": "The paper asserts mechanistic benefits (knowledge transfer) but lacks quantitative evidence such as gradient-conflict metrics.", "reasoning": "Without quantitative evidence supporting the claimed mechanisms, the explanation for *why* the method works remains anecdotal rather than empirical.", "judgment": "The mechanistic explanations are unsupported by rigorous data.", "valence": "negative", "suggested_improvement": "Provide quantitative gradient-conflict evidence to support mechanistic claims about knowledge transfer.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8359939455986023, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7931149005889893}]}, {"review_id": "0CVKYMciB5", "paper_id": "f6ugrBWs3K", "paper_title": "Chain of Time: In-Context Physical Simulation with Image Generation Models", "decision": "Reject", "summary": "The reviewer systematically challenges the empirical sufficiency of the paper by highlighting failures in the fluid domain, limited temporal and model scope, and missing literature context, while acknowledging the strength of the core conceptual motivation.", "units": [{"unit_index": 0, "inspected_object": "Fluid domain experimental results", "observation": "Chain-of-Time performs worse than direct prediction in the fluid domain, with the paper attributing this to flow rate estimation error.", "reasoning": "The reviewer expects a method claiming to improve physical simulation via step-by-step reasoning to provide mechanistic explanations for heterogeneous results; accepting the performance drop without distinguishing whether it is due to inherent properties (e.g., continuous deformation) or setup-specific issues leaves the theoretical coherence of the method unverified.", "judgment": "The failure in the fluid domain is a critical diagnostic weakness that undermines confidence in the method's general validity and theoretical understanding.", "valence": "negative", "suggested_improvement": "Provide deeper analysis to determine if the failure is principled (due to domain properties like partial transparency) or contingent on the experimental setup.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8038380742073059, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8134788870811462}, {"unit_index": 1, "inspected_object": "Temporal horizon of experiments (0.8 seconds)", "observation": "Experiments are confined to a short 0.8-second duration.", "reasoning": "Physical simulation claims require demonstrating performance over durations that meaningfully test the accumulation of error; a short horizon fails to reveal whether the method mitigates or exacerbates compound error effects over time.", "judgment": "The limited temporal scope is insufficient to support broad claims about the method's utility for physical simulation.", "valence": "negative", "suggested_improvement": "Test the method over longer simulation horizons to observe error accumulation dynamics.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8217607140541077, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7842568159103394}, {"unit_index": 2, "inspected_object": "Model and benchmark diversity", "observation": "The study exclusively uses GPT-4o and simple synthetic scenes, lacking comparison with other VLMs (e.g., Gemini, LLaVA) or established benchmarks (e.g., Moving MNIST, IntPhys).", "reasoning": "Generalizability claims require evidence that the method is not an artifact of a specific model's capabilities or narrow dataset characteristics; testing across diverse models and complex scenes is necessary to validate the approach's robustness.", "judgment": "The narrow selection of models and benchmarks leaves the generalizability claim unsupported.", "valence": "negative", "suggested_improvement": "Evaluate the method on multiple VLM+IGM combinations and established video prediction/physical reasoning benchmarks.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8657495379447937, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8779090046882629}, {"unit_index": 3, "inspected_object": "Related work section", "observation": "The paper omits discussion of adjacent literature, specifically video prediction and world models.", "reasoning": "Proper positioning within the research landscape is required to establish novelty and distinguish the contribution from existing approaches; failing to engage with these fields creates ambiguity about the paper's unique value.", "judgment": "The lack of engagement with relevant adjacent literature weakens the demonstration of novelty.", "valence": "negative", "suggested_improvement": "Cite and differentiate from prior work in video prediction and world models.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "related_work", "object_sim": 0.817516028881073, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8803107738494873}, {"unit_index": 4, "inspected_object": "Statistical rigor (sample sizes)", "observation": "Sample sizes vary between N=5 and N=20 without justification.", "reasoning": "Unexplained variation in sample size in small-N studies undermines confidence in the reliability and validity of reported performance comparisons.", "judgment": "The statistical inconsistency raises concerns about the reliability of the empirical results.", "valence": "negative", "suggested_improvement": "Justify sample size choices or standardize them to ensure reliable comparisons.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8109629154205322, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7953707575798035}, {"unit_index": 5, "inspected_object": "Conceptual framing and motivation", "observation": "The connection between human mental simulation and in-context reasoning is creative and well-motivated.", "reasoning": "The core idea offers interpretable intermediate steps and provides a strong conceptual foundation, even if empirical validation is currently lacking.", "judgment": "The conceptual contribution is significant and clearly motivated.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.931736171245575, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7610726952552795}]}, {"review_id": "0CkGz63r28", "paper_id": "IF4ccCwSKd", "paper_title": "LEMUR: Leveraging Vision-Language Models for Fine-Grained Multimodal Retrieval", "decision": null, "summary": "The reviewer applies disciplinary gatekeeping norms—historical positioning, external validation, controlled comparison, and self-contained presentation—to argue that despite technical merit, the paper's framing, proof, and presentation fail to meet community standards for credible claims.", "units": [{"unit_index": 0, "inspected_object": "Literature positioning and task framing (pre-CLIP/concurrent retrieval research)", "observation": "The paper overlooks specific prior work (VSRN, ALBEF) and frames retrieval as a CLIP-era problem rather than acknowledging the long line of pre-existing retrieval research.", "reasoning": "This omission constitutes a category error and misleading novelty narrative; the reviewer asserts that any retrieval paper must acknowledge the field's pre-existing foundations to avoid obscuring connections to prior work like Composed Image Retrieval.", "judgment": "The framing is misleading and the contribution is not properly situated within the disciplinary lineage.", "valence": "negative", "suggested_improvement": "Provide proper acknowledgment and discussion of pre-CLIP and concurrent retrieval research to calibrate novelty claims against existing task taxonomy.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8492838740348816, "reasoning_key": "novelty_standard", "reasoning_sim": 0.803577184677124}, {"unit_index": 1, "inspected_object": "Evaluation benchmark selection (Table 1)", "observation": "The main comparison in Table 1 uses only the authors' own FGMB benchmark without validation against external standards.", "reasoning": "Self-constructed benchmarks are treated as potentially self-serving unless validated against established public benchmarks for each retrieval sub-task; without such calibration, the results are inherently less persuasive.", "judgment": "The evidentiary basis for the performance claims is weak due to lack of external validation.", "valence": "negative", "suggested_improvement": "Show performance on established public benchmarks for each retrieval sub-task to allow readers to calibrate the method against known quantities.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8530392050743103, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8018633723258972}, {"unit_index": 2, "inspected_object": "Evaluation protocol fairness (Table 2)", "observation": "Prior models are evaluated zero-shot while LEMUR is trained on the benchmark, creating an asymmetry exemplified by the BLIP-2 discrepancy (63.8% vs 86.5–87.7% Recall@5).", "reasoning": "Evaluation protocols must hold training status constant across compared methods; the observed skew suggests the comparison protocol is flawed and undermines the persuasiveness of the entire results section.", "judgment": "The comparison appears unfair, rendering the reported improvements unreliable as evidence of superiority.", "valence": "negative", "suggested_improvement": "Hold training conditions constant across compared methods or prominently disclose the asymmetry.", "support_status": "mixed", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8180989027023315, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8106793761253357}, {"unit_index": 3, "inspected_object": "Ablation study completeness (Table 3)", "observation": "The paper lacks a key baseline consisting of fine-tuning the underlying Qwen2-VL backbone without LEMUR's architectural modifications, and the setup for Table 3 is not explained clearly.", "reasoning": "An architectural contribution must be isolated from the base model's inherent capacity through controlled comparison; without this, it is unclear how much improvement comes from the architecture versus the backbone.", "judgment": "The causal attribution of the improvement is uncertain and the experimental logic is insufficiently transparent.", "valence": "negative", "suggested_improvement": "Include a baseline of the fine-tuned Qwen2-VL backbone without LEMUR's architectural modifications to isolate the specific contribution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8129657506942749, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7773831486701965}, {"unit_index": 4, "inspected_object": "Presentation coherence and notation clarity (Figures/Tables)", "observation": "Figure 2 precedes Figure 1 disrupting logical flow; Table 3 contains unclear 'xattn' symbols and Table 5 has undefined T1/T2/T3/T4 references; no limitations section is present.", "reasoning": "Papers should be self-contained and readable without external decoding; presentation opacity signals potential issues with experimental thinking and transparency, treating presentation quality as a proxy for scientific rigor.", "judgment": "The paper does not yet feel ready for publication in its current form due to organizational problems and lack of transparency.", "valence": "negative", "suggested_improvement": "Order figures logically, define all symbols and table references explicitly, and include a limitations section analyzing weaknesses and failure cases.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8990944623947144, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7945453524589539}]}, {"review_id": "0D8qhDR44t", "paper_id": "Rj2tP4eXI1", "paper_title": "Know Thyself? On the Incapability and Implications of AI Self-Recognition", "decision": "Reject", "summary": "The reviewer evaluates the paper through a lens of methodological auditing, contrasting the ambitious claims of systematicity and causal explanation against the brevity of the methodology and the descriptive nature of the results. The logic moves from specific gaps in design justification and stability checks to a broader judgment that the work lacks scientific rigor and interpretive depth.", "units": [{"unit_index": 0, "inspected_object": "Methodology section scope and domain selection rationale", "observation": "The methodology section is brief and lacks a principled justification for the choice of three specific domains or evidence of multiple-run stability checks.", "reasoning": "The paper promises a 'systematic evaluation framework,' which implies a need for rigorous design rationales (orthogonality/completeness) and measurement reliability; without these, findings may be fragile and the claim of systematicity unsupported.", "judgment": "The execution falls short of the promised systematic rigor.", "valence": "negative", "suggested_improvement": "Provide a rationale for domain selection based on text type or model capability theory, and report variance measures from multiple runs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8168236017227173, "reasoning_key": "design_justification", "reasoning_sim": 0.8161301016807556}, {"unit_index": 1, "inspected_object": "Interpretation of hierarchical bias as a causal mechanism", "observation": "The central finding that models exhibit hierarchical bias is treated as a descriptive label rather than a mechanistic explanation involving training dynamics or architecture.", "reasoning": "A deep causal explanation requires identifying underlying drivers (e.g., data distribution, RLHF effects); merely labeling the observed pattern fails to provide interpretive depth or explain why the behavior occurs.", "judgment": "The analysis lacks sufficient depth to support the explanatory claims.", "valence": "negative", "suggested_improvement": "Propose a theoretical mechanism linking the bias to model training, architecture, or data composition.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.804998517036438, "reasoning_key": "design_justification", "reasoning_sim": 0.7763053774833679}]}, {"review_id": "0DLOEMtv1I", "paper_id": "cUYmiV7wvP", "paper_title": "Principal component analysis for very heavy-tailed data", "decision": "Reject", "summary": "The reviewer acknowledges strong empirical performance and transparent implementation but identifies significant gaps in theoretical guarantees, comprehensive baseline comparisons, conceptual positioning, and practitioner guidance. The review functions as a roadmap for revision to elevate the contribution from empirically promising to theoretically and practically complete.", "units": [{"unit_index": 0, "inspected_object": "Algorithmic mechanics (random subsampling, per-subsample PCA, projection-matrix averaging)", "observation": "The reviewer reconstructs the pipeline and notes it relies on standard linear-algebraic primitives.", "reasoning": "The simplicity and use of transparent, reproducible primitives are viewed as strengths that aid understanding.", "judgment": "Positive evaluation of implementation clarity and transparency.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8399839401245117, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7740824222564697}, {"unit_index": 1, "inspected_object": "Empirical validation scope", "observation": "Experiments cover synthetic data, transcriptomic/synaptic datasets, hyperparameter sensitivity, and comparisons against classical PCA, geometric-median PCA, and convex robust PCA.", "reasoning": "Performance under extreme heavy-tailed noise and infinite-variance regimes demonstrates practical utility.", "judgment": "Strong empirical performance observed.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8246753215789795, "reasoning_key": "design_justification", "reasoning_sim": 0.8039444088935852}, {"unit_index": 2, "inspected_object": "Theoretical grounding (BBP phase transition, random-matrix theory)", "observation": "Paper uses BBP phase transition to motivate the algorithm.", "reasoning": "Anchoring empirical methods in theoretical phenomena is appreciated, even if formal guarantees are absent.", "judgment": "Partial positive evaluation; theoretical motivation is valued but insufficient for completeness.", "valence": "mixed", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8098028898239136, "reasoning_key": "design_justification", "reasoning_sim": 0.7813868522644043}, {"unit_index": 3, "inspected_object": "Formal theoretical guarantees", "observation": "Paper lacks asymptotic analysis, finite-sample error bounds, perturbation-theoretic results, and convergence/robustness guarantees.", "reasoning": "Methods papers in this area are expected to provide a formal account of why they work, not just empirical success.", "judgment": "Significant weakness due to missing theoretical rigor.", "valence": "negative", "suggested_improvement": "Provide asymptotic analysis, finite-sample error bounds, or perturbation-theoretic results.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8251944184303284, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7572582960128784}, {"unit_index": 4, "inspected_object": "Hyperparameter tuning strategies", "observation": "Discussion of (N, P, R) parameters is qualitative; tuning strategies are heuristic.", "reasoning": "A method paper should offer principled guidance on parameter selection rather than just empirical sensitivity analyses.", "judgment": "Weakness due to lack of principled parameter guidance.", "valence": "negative", "suggested_improvement": "Offer principled guidance on parameter selection.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8895473480224609, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7735434770584106}, {"unit_index": 5, "inspected_object": "Conceptual positioning relative to existing literature", "observation": "Underdeveloped relationship to Tyler's M-estimator, ROBPCA, and spatial-sign PCA.", "reasoning": "Novelty cannot be properly assessed without situating the method within established robust-scatter frameworks.", "judgment": "Weakness due to unclear conceptual novelty and genealogy.", "valence": "negative", "suggested_improvement": "Develop the relationship to Tyler's M-estimator, ROBPCA, and spatial-sign PCA.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "related_work", "object_sim": 0.8882616758346558, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8216319680213928}, {"unit_index": 6, "inspected_object": "Baseline selection in experiments", "observation": "Omitted comparators include Catoni-type covariance, truncation-based PCA, and Lerman & Maunu (2018).", "reasoning": "Empirical claims of superiority require comparison against the full landscape of relevant methods.", "judgment": "Weakness due to incomplete baseline comparison.", "valence": "negative", "suggested_improvement": "Include comparisons with Catoni-type covariance, truncation-based PCA, and Lerman & Maunu (2018).", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.9065055847167969, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8667325377464294}, {"unit_index": 7, "inspected_object": "Failure cases and diagnostic tools", "observation": "Lack of discussion on failure cases, limitations, and practical checks like subsample stability tests.", "reasoning": "Practitioners need tools to recognize when the method is inappropriate, especially given concerns about robustness methods degrading outside intended regimes.", "judgment": "Weakness due to lack of practitioner-oriented guidance and failure analysis.", "valence": "negative", "suggested_improvement": "Discuss failure cases and provide diagnostic tools such as subsample stability tests.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7971552610397339, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7989144325256348}, {"unit_index": 8, "inspected_object": "Aggregation method choice", "observation": "Reviewer questions whether mean of projection matrices is the right aggregation choice.", "reasoning": "Probing alternative aggregations (geometric/median) reveals uncertainty about algorithmic design optimality.", "judgment": "Uncertainty regarding the optimality of the chosen aggregation method.", "valence": "uncertain", "suggested_improvement": "Investigate alternative aggregation methods like geometric or median subspace averaging.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.7950798869132996, "reasoning_key": "design_justification", "reasoning_sim": 0.8303528428077698}, {"unit_index": 9, "inspected_object": "Convergence of leading eigenvector", "observation": "No formal statement on convergence of aggregated projection matrix's top eigenvector to true principal direction.", "reasoning": "Directly targets the missing theory gap identified in unit 04.", "judgment": "Uncertainty about theoretical convergence properties.", "valence": "uncertain", "suggested_improvement": "Provide formal statements on convergence of the leading eigenvector.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.7883073091506958, "reasoning_key": "design_justification", "reasoning_sim": 0.7447717189788818}, {"unit_index": 10, "inspected_object": "Multiple component recovery", "observation": "Paper focuses only on the leading principal direction; reviewer asks about simultaneous recovery of multiple components.", "reasoning": "Evaluates practical utility beyond specific demonstration; challenges extension to full PCA.", "judgment": "Uncertainty about method's applicability as a general dimensionality reduction tool.", "valence": "uncertain", "suggested_improvement": "Extend analysis or experiments to multiple principal components.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "reproducibility", "object_sim": 0.7364128232002258, "reasoning_key": "design_justification", "reasoning_sim": 0.7692470550537109}]}, {"review_id": "0Dnmd23amQ", "paper_id": "Q8qgloDKUO", "paper_title": "Process-Level Trajectory Evaluation for Environment Configuration in Software Engineering Agents", "decision": "Accept (Poster)", "summary": "The reviewer performs a resource audit, praising the dataset's construction and utility while critiquing its incomplete coverage of state-of-the-art agents and poor figure legibility. The assessment balances instrumental value against venue fit concerns, without engaging deeply with the paper's empirical findings or methodological novelty claims.", "units": [{"unit_index": 0, "inspected_object": "Evaluation suite completeness relative to current frontier coding agents and models", "observation": "The review identifies notable omissions of specific SOTA base models (GPT-5-Codex, Claude 4.5) and agents (Codex CLI, Gemini CLI, Jules, Claude Code) from the evaluation suite.", "reasoning": "The reviewer applies a standard that a benchmark for environment configuration should test against tools practitioners actually use when facing configuration problems; the absence of these widely used tools renders the evaluation under-specified relative to community expectations.", "judgment": "The contribution is viewed as having incomplete coverage rather than being technically flawed, limiting its utility as a definitive benchmark.", "valence": "negative", "suggested_improvement": "Include the missing SOTA models and agents in the evaluation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8415457606315613, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7363523840904236}, {"unit_index": 1, "inspected_object": "Presentation clarity and figure legibility", "observation": "The reviewer notes issues with the readability of results presentation in figures.", "reasoning": "Clear communication is required for the resource to be useful to the community; cluttered or illegible figures hinder the immediate understanding of the dataset's performance metrics.", "judgment": "The presentation barrier reduces the immediate usability and impact of the resource.", "valence": "negative", "suggested_improvement": "Improve figure legibility by focusing on a subset of results.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8716784715652466, "reasoning_key": "presentation_trust", "reasoning_sim": 0.749932050704956}, {"unit_index": 2, "inspected_object": "Dataset construction methodology and explanatory detail", "observation": "The reviewer finds the dataset creation procedure to be thorough and well-explained.", "reasoning": "A clear and rigorous description of the construction process serves as a methodological trust signal, indicating that other researchers can reproduce and utilize the resource effectively.", "judgment": "The resource is assessed as potentially useful and trustworthy due to its transparent construction.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8317450284957886, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7819589972496033}, {"unit_index": 3, "inspected_object": "Fit between the paper's nature (resource contribution) and the venue (ICLR)", "observation": "The reviewer explicitly questions whether ICLR is the best venue, suggesting a dedicated dataset/benchmark track instead.", "reasoning": "The reviewer distinguishes between technical/methodological advances and resource contributions, holding an assumption that the latter may not align with the primary expectations of the conference venue.", "judgment": "The paper represents a high-quality resource but suffers from a potential genre mismatch with the target venue.", "valence": "conditional", "suggested_improvement": "Consider submitting to a dedicated dataset or benchmark track.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8011550307273865, "reasoning_key": "novelty_standard", "reasoning_sim": 0.809077262878418}]}, {"review_id": "0DtkQe9MpW", "paper_id": "e8bWf8old1", "paper_title": "Topological Alignment: A Universal Framework for Anomaly Detection", "decision": null, "summary": "The reviewer performs a claim-evidence calibration audit, acknowledging strong empirical results but criticizing the gap between the paper's ambitious claims (causality, universality) and the limited, conceptual evidence provided. The review focuses on evidentiary standards, scope of inference, and presentation clarity rather than technical correctness.", "units": [{"unit_index": 0, "inspected_object": "The paper's central diagnostic claim that domain conflict arising from incompatible manifold structures is the root cause of performance collapse.", "observation": "The explanation for the observed phenomenon is insufficiently proven, with evidence for topology as the root cause being limited and mostly conceptual rather than empirical.", "reasoning": "The reviewer applies a standard requiring controlled experiments or direct measurements to isolate causal mechanisms; because the paper relies on reasoning about why topology matters instead of measuring it, the causal inference is unsupported.", "judgment": "The core mechanistic claim lacks sufficient empirical demonstration.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.6968124508857727, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7940799593925476}, {"unit_index": 1, "inspected_object": "The claim that the framework consistently benefits from large-scale Domain Generalization where all baselines fail.", "observation": "The presented experiments do not convincingly support the scale implied by the 'large-scale' and 'universal' framing, as the number of training domains remains limited.", "reasoning": "There is a mismatch between the rhetorical scope of the claims (universality/large-scale) and the empirical footprint (limited domains); performance on these specific benchmarks does not license broader generalizability claims.", "judgment": "The scope of the evidence does not justify the broad interpretive claims made in the abstract and title.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7287770509719849, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7955496907234192}, {"unit_index": 2, "inspected_object": "The design choices regarding hyperparameter sensitivity and the placement of the adapter at the first encoder layer.", "observation": "The reviewer questions whether these design decisions are principled or arbitrary, noting a lack of clarity on their robustness.", "reasoning": "Design choices must be demonstrated as robust and well-motivated rather than accidental to satisfy the standard of methodological rigor; without this, the contribution's stability is uncertain.", "judgment": "The method's design rationale is unclear and potentially fragile.", "valence": "conditional", "suggested_improvement": "Demonstrate that design choices are robust and well-motivated through sensitivity analysis or ablation.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8460463881492615, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8313064575195312}, {"unit_index": 3, "inspected_object": "The contribution of persistent homology compared to traditional geometric hard-mining.", "observation": "The reviewer finds it difficult to understand the specific improvement brought by persistent homology due to vague comparisons and intuition.", "reasoning": "A novel topological approach must provide a clear contrastive analysis against known techniques to establish its added value; without this, the contribution is not legible within the existing landscape.", "judgment": "The distinct advantage of the proposed method over existing geometric methods is not clearly established.", "valence": "negative", "suggested_improvement": "Provide clearer comparison and intuition explaining the specific improvements brought by persistent homology over traditional geometric hard-mining.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.7867013216018677, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8038604855537415}, {"unit_index": 4, "inspected_object": "The scholarly grounding and presentation quality, specifically citations and figure clarity.", "observation": "The paper lacks foundational persistent homology citations, uses vague terms like 'current methods', and contains an illegible Figure 3.", "reasoning": "Scholarly infrastructure (citations, clear figures) is necessary to allow verification and contextualize contributions; missing elements signal a broader lack of clarity and rigor that hinders the evaluation of ambitious claims.", "judgment": "The presentation fails to meet standards of clarity and scholarly rigor required for the paper's scope.", "valence": "negative", "suggested_improvement": "Add foundational persistent homology citations, clarify vague terminology, and improve the legibility of Figure 3.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.8595554232597351, "reasoning_key": "presentation_trust", "reasoning_sim": 0.815909206867218}]}, {"review_id": "0EASUkTL2H", "paper_id": "kqT4pcOT10", "paper_title": "Emergent Bayesian Behaviour and Optimal Cue Combination in LLMs", "decision": "Reject", "summary": "The reviewer conducts a disciplinary audit, challenging the paper's core claim of 'optimal cue combination' because the normative Bayesian optimum was not derived, only descriptive models fitted. They probe alternative explanations for observed behaviors (prompt artifacts, local context dependence) and critique the lack of mechanistic insights and literature accuracy, resulting in a judgment of methodological competence but scientific underwhelmingness.", "units": [{"unit_index": 0, "inspected_object": "Establishment of Bayesian optimality for the task battery", "observation": "The authors fit linear, static Bayesian, and Kalman filter models to behavior data but did not establish the true Bayesian optimal strategy for the tasks.", "reasoning": "Optimality is a normative property derived from task structure; fitting descriptive models (even if they explain data well) does not prove those models are the optimal reference class. Without deriving the normative solution, claims about 'Bayesian behavior' remain relative rather than absolute.", "judgment": "Difficult to assess how close LLMs perform relative to the optimal strategy; the central claim of optimality is unsupported.", "valence": "negative", "suggested_improvement": "Derive the normative optimal strategy from first principles for each task to serve as a proper baseline.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8538548350334167, "reasoning_key": "design_justification", "reasoning_sim": 0.8016797304153442}, {"unit_index": 1, "inspected_object": "Prompt sensitivity and sequential effects", "observation": "The reviewer questions whether sequential effects (akin to Kalman filtering) persist if LLMs are instructed to ignore previous trial responses.", "reasoning": "If the effect disappears when ignoring history, the behavior is an artifact of instruction-following/prompt structure rather than genuine emergent Bayesian updating or internal inference.", "judgment": "Uncertainty regarding whether observed behaviors reflect true model properties or prompt artifacts.", "valence": "conditional", "suggested_improvement": "Conduct experiments instructing models to ignore previous trials to test the robustness of sequential effects.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8395625948905945, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7419014573097229}, {"unit_index": 2, "inspected_object": "Trial structure and block adaptation", "observation": "The reviewer asks what would happen if trials from different ranges were interleaved rather than blocked.", "reasoning": "Interleaving would disrupt block-level adaptation statistics. If the model's behavior changes significantly, it suggests reliance on local context rather than tracking the global generative distribution consistent with Bayesian updating.", "judgment": "Uncertainty about whether the model tracks the true distribution or merely responds to local block context.", "valence": "conditional", "suggested_improvement": "Test performance under interleaved trial structures to distinguish between block adaptation and global distribution tracking.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8484024405479431, "reasoning_key": "design_justification", "reasoning_sim": 0.8458134531974792}, {"unit_index": 3, "inspected_object": "Regression to the mean in GPT-Mini", "observation": "GPT-Mini's responses in Fig. 4 did not exhibit regression toward the mean.", "reasoning": "Bayesian behavior in these psychophysical tasks typically involves shrinkage toward the prior (regression to the mean). The absence of this phenomenon in one model contradicts the general claim of Bayesian behavior.", "judgment": "Anomalous behavior in specific models challenges the universality of the Bayesian interpretation.", "valence": "negative", "suggested_improvement": "Clarify why certain models do not show expected regression to the mean or adjust interpretations.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8013770580291748, "reasoning_key": "design_justification", "reasoning_sim": 0.7971853017807007}, {"unit_index": 4, "inspected_object": "Mechanistic explanation for multimodal integration differences", "observation": "The paper reports differences in multimodal integration across models but lacks hypotheses about why.", "reasoning": "Reporting empirical differences without mechanistic accounts (e.g., relating to training data differences) leaves the findings as a descriptive survey rather than an explanatory scientific contribution.", "judgment": "Lack of major insight due to missing mechanistic explanations for observed differences.", "valence": "negative", "suggested_improvement": "Provide hypotheses linking integration differences to model training characteristics.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8134663701057434, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7281573414802551}, {"unit_index": 5, "inspected_object": "Accuracy of literature interpretation", "observation": "Interpretations of prior psychophysics literature were inaccurate in various places.", "reasoning": "The paper borrows authority from a mature field; inaccurate representation of that literature undermines the legitimacy of the borrowed framework and norms.", "judgment": "Serious concern regarding the fidelity of the paper's engagement with foundational literature.", "valence": "negative", "suggested_improvement": "Perform a careful check of the accuracy of references to the literature.", "support_status": "memo_inferred", "confidence": "low", "object_key": "related_work", "object_sim": 0.8182425498962402, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7701761722564697}, {"unit_index": 6, "inspected_object": "Figure clarity (Fig. 1A)", "observation": "Dots with different sizes in Fig. 1A are unexplained.", "reasoning": "Unexplained visual encodings impede reconstruction of the experimental design from the presentation alone, suggesting poor expository structure.", "judgment": "Presentation barriers hinder comprehension.", "valence": "negative", "suggested_improvement": "Clarify the meaning of different dot sizes in Figure 1A.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8577932715415955, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8396087884902954}, {"unit_index": 7, "inspected_object": "Phrasing in Section 5.3", "observation": "The phrase 'higher BCS than expected' regarding Gemma 3 4B and Phi 4 Multimodal is confusing.", "reasoning": "Confusing interpretive language contributes to the overall low presentation score and impedes clear understanding of results.", "judgment": "Writing needs improvement; phrasing is unclear.", "valence": "negative", "suggested_improvement": "Clarify the phrasing and interpretation of results in Section 5.3.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7421497702598572, "reasoning_key": "presentation_trust", "reasoning_sim": 0.850753664970398}]}, {"review_id": "0EL4G2rBaj", "paper_id": "xAvqHtLVgz", "paper_title": "Lossless Vocabulary Reduction for Auto-Regressive Language Models", "decision": "Accept (Poster)", "summary": "The reviewer accepts the theoretical soundness provisionally but criticizes the paper for insufficient empirical decomposition, poor presentation of examples, textual redundancy, and ambiguous scope regarding knowledge distillation.", "units": [{"unit_index": 0, "inspected_object": "Experimental reporting of ensemble vocabulary compression rates", "observation": "The reviewer finds the reported ensemble-level compression insufficiently decomposed, noting a lack of per-model breakdowns for Qwen2.5-3B and Falcon-7B.", "reasoning": "The reviewer assumes that aggregate results may conflate effects, potentially driven by one model's tokenizer being more amenable to reduction or the ensemble strategy itself. Without decomposition, readers cannot attribute compression to specific sources or assess robustness.", "judgment": "The empirical claims lack causal clarity and transparency; the current reporting is insufficient to validate the method's general benefits.", "valence": "negative", "suggested_improvement": "Provide individual vocabulary compression rates for each model in the ensemble and compare across ensemble strategies.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8360434770584106, "reasoning_key": "design_justification", "reasoning_sim": 0.818366527557373}, {"unit_index": 1, "inspected_object": "Illustrative example in Section 3.2", "observation": "The reviewer identifies format and presentation issues with the illustrative example, finding it lacks pedagogical effectiveness.", "reasoning": "The reviewer believes that visual illustrations or alternative narrative styles would improve clarity and readability, indicating a standard that technical examples should be accessible and clearly presented.", "judgment": "The presentation of this key example is suboptimal and hinders reader comprehension.", "valence": "negative", "suggested_improvement": "Include a visual illustration or adopt an alternative narrative style for the example.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7980952858924866, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7678461670875549}, {"unit_index": 2, "inspected_object": "Abstract and Introduction overlap", "observation": "The reviewer notes significant textual redundancy where the first half of the abstract overlaps too much with the first paragraph of the introduction.", "reasoning": "This duplication indicates a lack of polish and precision in the paper's structure, contributing to lower clarity and readability scores.", "judgment": "The paper suffers from presentational flaws due to redundant phrasing between front-matter sections.", "valence": "negative", "suggested_improvement": "Revise the abstract to eliminate overlap with the introduction.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.821769654750824, "reasoning_key": "presentation_trust", "reasoning_sim": 0.818748414516449}, {"unit_index": 3, "inspected_object": "Knowledge distillation application", "observation": "The reviewer questions whether the proposed method applies to knowledge distillation, a context mentioned in the introduction but not explored empirically.", "reasoning": "The reviewer operates under a norm that a method's value is measured by its transferability to related problems gestured toward by the authors. The absence of discussion creates an uncertainty about the framework's true scope.", "judgment": "The paper's contribution breadth is ambiguous; it is unclear if the method is a one-off technique or a broadly applicable framework.", "valence": "conditional", "suggested_improvement": "Discuss the applicability of the method to knowledge distillation or explicitly delimit its scope.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8027257919311523, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8263282775878906}]}, {"review_id": "0EeR65irWy", "paper_id": "8TfiR1Lcvr", "paper_title": "Observational Auditing of Privacy", "decision": null, "summary": "The reviewer evaluates the paper through constructive boundary-setting, acknowledging formal rigor while critiquing empirical degradation at high epsilon, lack of proxy model analysis, limited comparisons, and narrow applicability to label DP.", "units": [{"unit_index": 0, "inspected_object": "The formal/methodological core of the privacy auditing method", "observation": "The privacy auditing is well-explained, formally proven, and based on previous results from the auditing literature.", "reasoning": "The reviewer values intellectual lineage; the method is legible because it connects to established work rather than floating free of it, indicating sound theoretical apparatus.", "judgment": "Qualified endorsement of the method's formal rigor and clarity.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7539659142494202, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7955676317214966}, {"unit_index": 1, "inspected_object": "Experimental validation on Criteo and CIFAR-10 datasets", "observation": "The quality of the estimate appears to be quite low on Criteo for values of epsilon beyond 2.", "reasoning": "This threshold observation indicates a regime where the method degrades, suggesting limited practical viability in certain parameter settings despite being meaningful for low epsilon values.", "judgment": "Concern regarding empirical performance stability at higher epsilon values.", "valence": "negative", "suggested_improvement": "Provide figures displaying how the quality of the estimate is impacted by the number of canaries to clarify the trade-off curve.", "support_status": "mixed", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8460636734962463, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7909846305847168}, {"unit_index": 2, "inspected_object": "Proxy model component and its design alternatives", "observation": "The paper lacks an in-depth discussion on possible alternatives to create the proxy model and the impact that this has on the quality of the estimate.", "reasoning": "The proxy model is a key dependency; without understanding design choices and alternatives, the robustness of the audit estimate cannot be fully assessed, as the proxy's quality may explain successes or limitations.", "judgment": "Identification of an unexamined methodological dependency that hinders full assessment of robustness.", "valence": "negative", "suggested_improvement": "Include a design-space analysis exploring alternative ways to create the proxy model and their impact on estimate quality.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7861555218696594, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.792608916759491}, {"unit_index": 3, "inspected_object": "Comparative landscape against prior work", "observation": "Comparison with prior work is limited to only the MIA approach proposed by Watson et al. 2022.", "reasoning": "Limited comparison fails to situate the method within the broader auditing landscape, raising concerns about demonstrated generality and scope relative to other approaches.", "judgment": "Critique of positioning and scope rather than internal validity.", "valence": "negative", "suggested_improvement": "Expand comparisons to include other relevant prior work beyond Watson et al. 2022.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.846626341342926, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7982160449028015}, {"unit_index": 4, "inspected_object": "Applicability beyond label differential privacy", "observation": "The method only works for label DP, protecting only the privacy of the label of a particular record.", "reasoning": "The primary claimed advantage (avoiding modification of training dataset) is constrained by the narrow scope of label DP, limiting applicability beyond this specific mechanism unless demonstrated otherwise.", "judgment": "Assessment that the contribution's significance is bound by its restricted setting.", "valence": "conditional", "suggested_improvement": "Demonstrate extension to other privacy mechanisms or explicitly acknowledge the limitation to label DP.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7825064063072205, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8037045001983643}]}, {"review_id": "0Emh8LZAiT", "paper_id": "S0IIgb33fO", "paper_title": "Gradient Fan-in Asymmetry: The Structural Cause of Layer Redundancy in Deep Transformers", "decision": null, "summary": "The reviewer critiques the manuscript primarily for scope mismanagement, arguing that attempting two major contributions dilutes the evidentiary depth of both. Specific weaknesses include lack of systematic experimental breadth for the hypothesis, missing formal definitions, insufficient comparison with state-of-the-art pruning methods, and imprecise presentation. The reviewer suggests consolidating the paper around a single objective to allow for deeper, more rigorous development.", "units": [{"unit_index": 0, "inspected_object": "Manuscript scope architecture and dual objectives", "observation": "The manuscript contains '2 or 2.5 manuscripts worth of topics', treating multiple objectives too thinly.", "reasoning": "A single paper should have one primary contribution; the dual objectives create mutual dilution where neither is sufficiently developed.", "judgment": "Structural problem: insufficient development due to over-scoping.", "valence": "negative", "suggested_improvement": "Lean into one of its two objectives and consolidate the narrative around that (with appropriate additions/revisions).", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "related_work", "object_sim": 0.8171650171279907, "reasoning_key": "novelty_standard", "reasoning_sim": 0.752703070640564}, {"unit_index": 1, "inspected_object": "Evidentiary structure for the explanatory hypothesis", "observation": "Evidence provided is hedged ('some evidence') and lacks systematic treatment over varied model sizes, parameters, and datasets.", "reasoning": "Causal/structural claims require demonstration across a range of configurations to establish generality rather than being an artifact of specific setups.", "judgment": "Explanatory hypothesis lacks breadth of evidence needed to be convincing.", "valence": "negative", "suggested_improvement": "Provide a more systematic treatment of experiments over more varied model sizes, training parameters, and data sets.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7988268733024597, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8611007332801819}, {"unit_index": 2, "inspected_object": "Formal articulation of the hypothesis", "observation": "The early part of the manuscript lacks a crisp, mathematical statement of the hypothesis; it is diffuse.", "reasoning": "Without precise formal statements, it is challenging to assess the strength of support for the claim via proxy measurements.", "judgment": "Presentation makes it difficult to evaluate the hypothesis rigorously.", "valence": "negative", "suggested_improvement": "Provide a crisp, mathematical statement of the hypothesis.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.798820436000824, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8118436932563782}, {"unit_index": 3, "inspected_object": "Figure 1(a) description", "observation": "The description of what is plotted in Fig. 1(a) is not precise enough.", "reasoning": "Imprecise descriptions hinder the link between theoretical constructs and empirical measurements.", "judgment": "Uncertainty regarding the specific empirical measurement shown.", "valence": "uncertain", "suggested_improvement": "Clarify exactly what is being plotted in Fig. 1(a).", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8362446427345276, "reasoning_key": "construct_validity", "reasoning_sim": 0.8222672343254089}, {"unit_index": 4, "inspected_object": "Pruning/compression contribution comparative positioning", "observation": "The manuscript compares against standard heuristics but lacks comparison with methods such as [1] and [2] and broader compression literature trade-offs.", "reasoning": "A pruning paper must position itself against the state of the art and discuss costs relative to other approaches to be competitive.", "judgment": "Pruning contribution lacks depth of comparison needed to be competitive.", "valence": "negative", "suggested_improvement": "Include comparisons with methods such as [1] and [2] and engage with the broader compression literature including trade-offs with structure pruning methods.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8319453001022339, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7566928267478943}, {"unit_index": 5, "inspected_object": "Presentation precision and redundancy", "observation": "There is significant redundancy in noting the observation is structural, and components are described in prose rather than formally.", "reasoning": "Prose descriptions leave room for ambiguity affecting reproducibility; redundancy reduces clarity.", "judgment": "Presentation is insufficiently precise and repetitive.", "valence": "negative", "suggested_improvement": "State method/architecture components more formally and unambiguously; reduce redundancy in rhetorical emphasis.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8754062652587891, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8586386442184448}]}, {"review_id": "03fJDbMfLW", "paper_id": "hxGdAUn3sB", "paper_title": "Self-Consistency Improves the Trustworthiness of Self-Interpretable GNNs", "decision": "Accept (Poster)", "summary": "The reviewer validates the paper's strong theoretical and empirical base but identifies two specific gaps in validation: the lack of an ablation study for the encoder freezing strategy and the absence of statistical significance testing for reported improvements. These are framed as requests for additional evidence rather than fundamental flaws.", "units": [{"unit_index": 0, "inspected_object": "Freezing of the GNN encoder during SC fine-tuning", "observation": "The paper freezes the encoder under the assumption that the representation is already optimal, without validating this choice against a control group where the encoder is fine-tuned.", "reasoning": "Design choices that restrict optimization space should be justified empirically through ablation studies to demonstrate they are beneficial or not harmful; the current assumption lacks supporting evidence.", "judgment": "The design choice is unsubstantiated and requires further validation.", "valence": "negative", "suggested_improvement": "Compare the freezing design with a control group where the encoder is not frozen.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7938168048858643, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7699024081230164}, {"unit_index": 1, "inspected_object": "Reporting of statistical significance in experimental results", "observation": "The paper reports improvements but does not provide statistical significance tests or quantify the percentage of results that significantly outperform baselines.", "reasoning": "Aggregate claims of improvement require measures of uncertainty to distinguish consistent, meaningful gains from noise or limited configurations across the experimental landscape.", "judgment": "The reliability and consistency of the empirical claims are unclear due to missing statistical reporting.", "valence": "negative", "suggested_improvement": "Conduct statistical significance tests and report the percentage of results that significantly outperform each baseline.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8291704654693604, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8210353255271912}]}, {"review_id": "03xHzSrFD1", "paper_id": "30u4c9soJu", "paper_title": "Refine, Don’t Rewrite: LearnFrom for Consistency-Aware LLM Decompilation", "decision": null, "summary": "The reviewer adopts a stance of conditional skepticism, accepting the potential value of the contribution while systematically probing four methodological areas—reproducibility, mechanistic transparency, robustness to edge cases, and domain alignment—to determine if the empirical claims are sufficiently substantiated.", "units": [{"unit_index": 0, "inspected_object": "The integration and setup of the validation pipeline (compilation testing and libFuzzer-based fuzzing)", "observation": "The reviewer finds the description of how compilation testing and fuzzing are integrated into the workflow and how environments are set up for complex programs with diverse dependencies to be underspecified.", "reasoning": "The validation pipeline is treated as the load-bearing basis for the claim that LearnFrom produces semantically consistent output; without exact conditions, the reported 5% improvement in re-execution rates cannot be independently verified or fully understood.", "judgment": "The empirical claims are not yet sufficiently supported to be endorsed due to lack of reproducibility infrastructure details.", "valence": "negative", "suggested_improvement": "Provide detailed descriptions of the workflow integration and environment setup for complex real-world programs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8361479043960571, "reasoning_key": "robustness_norm", "reasoning_sim": 0.739618718624115}, {"unit_index": 1, "inspected_object": "The role of validation signals in code evolution", "observation": "The reviewer questions whether compilation success and fuzzing correctness are used as binary pass/fail signals or if error messages actively inform and steer code evolution.", "reasoning": "This distinction determines the nature of the system: a generate-and-test loop versus a feedback-driven repair system. The reviewer suspects the paper may imply more sophistication than implemented by probing the mechanisms that would translate diagnostics into edits.", "judgment": "The system's intelligence and mechanism are ambiguous, creating uncertainty about the true capabilities described.", "valence": "conditional", "suggested_improvement": "Clarify the information flow within the system, specifically detailing how diagnostic error messages are translated into concrete edits if they are used.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8139604926109314, "reasoning_key": "design_justification", "reasoning_sim": 0.7940859794616699}, {"unit_index": 2, "inspected_object": "The robustness of block-wise editing given decompiled code characteristics", "observation": "The reviewer identifies potential failure modes where Tree-sitter fails on unparseable input, the initial decompiled output does not compile, or low-level artifacts like goto statements hinder reconstruction.", "reasoning": "The core mechanism presupposes parseable input and compilable output. Since decompiled code is notoriously messy and often structurally faithful but rarely recompilable, the system's behavior at these edge cases is critical to its validity.", "judgment": "The framework's applicability is stressed by boundary conditions that violate implicit preconditions, raising doubts about its practical robustness.", "valence": "negative", "suggested_improvement": "Address how the system handles unparseable inputs, uncompilable initial outputs, and low-level artifacts such as goto statements.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.816774845123291, "reasoning_key": "design_justification", "reasoning_sim": 0.8036741614341736}, {"unit_index": 3, "inspected_object": "The domain gap between training data and evaluation data", "observation": "The model is trained on pairs of assembly and original source code but refined and evaluated on pseudo-code generated by decompilers, creating a mismatch between clean source and noisy, syntactically irregular output.", "reasoning": "If the model has only seen clean source during training, its ability to edit decompiled pseudo-code is an emergent capability that is neither explained nor measured. The paper does not justify this design choice or analyze its impact.", "judgment": "The fundamental design choice lacks justification, and the generalization from clean source to noisy pseudo-code is an unresolved uncertainty.", "valence": "negative", "suggested_improvement": "Justify the design choice of using different distributions for training and evaluation, and analyze the impact of the domain gap on model performance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8468282222747803, "reasoning_key": "design_justification", "reasoning_sim": 0.8443319797515869}]}, {"review_id": "051nQvCWoM", "paper_id": "LlRJyoWMxv", "paper_title": "OT Score: An OT based Confidence Score for Source Free Unsupervised Domain Adaptation", "decision": null, "summary": "The reviewer conducts a formalist audit, prioritizing mathematical hygiene and verifiability over empirical performance. They identify systematic failures in notation definition, internal consistency, and assumption realism, leading to a judgment that the theoretical foundation is too weak to support the paper's claims.", "units": [{"unit_index": 0, "inspected_object": "Undefined symbols in Theorems 2, 3, 4, 6, and 8 (e.g., `Lip_b(R^d)`, `O(μ,ν)`, `ε`, `T^μ̂_ν`, `w*`, `α`)", "observation": "Multiple symbols are introduced without explicit specification or definition within the theorem statements.", "reasoning": "The reviewer applies a standard of formal precision requiring self-contained verifiability; undefined notation prevents verification of validity, rendering claims indeterminate.", "judgment": "The theoretical apparatus is incomplete and ambiguous, undermining trust in the claims.", "valence": "negative", "suggested_improvement": "Explicitly define all symbols used in theorem statements to ensure they are self-contained and verifiable.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7577124238014221, "reasoning_key": "presentation_trust", "reasoning_sim": 0.777001142501831}, {"unit_index": 1, "inspected_object": "Ambiguity in Theorem 4's `ε ∈ (0,1)` and phrase 'target samples … after the optimal transport'", "observation": "Symbols and phrases admit multiple interpretations (e.g., arbitrary constant vs. fixed parameter) without disambiguation.", "reasoning": "The reviewer demands authorial responsibility for clarity; ambiguity prevents determining the correct reading, and the reviewer refuses to infer intent.", "judgment": "The presentation fails to meet standards of rigorous communication.", "valence": "negative", "suggested_improvement": "Provide formal expressions to resolve vague phrases and explicitly state whether parameters are arbitrary or fixed.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.860436201095581, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7769233584403992}, {"unit_index": 2, "inspected_object": "Algorithm 1 input specification relative to the paper's 'source-free' claim", "observation": "Algorithm 1 includes 'source data' as an input, contradicting the title and framing of a source-free setting.", "reasoning": "Internal consistency requires that algorithmic specifications match the stated problem setup; this contradiction indicates a factual error in the paper's own specification.", "judgment": "The paper contains a fundamental internal inconsistency.", "valence": "negative", "suggested_improvement": "Align Algorithm 1 inputs with the source-free setting claim by removing source data or revising the problem framing.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8578673601150513, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7711936831474304}, {"unit_index": 3, "inspected_object": "Step 6's 'smoothed indicator functions of Laguerre cells'", "observation": "The step lacks justification or citation for the specific technique used.", "reasoning": "Technical steps require provenance to be reproducible and credible; the absence suggests a gap in rigor or knowledge.", "judgment": "The algorithmic description is insufficiently justified.", "valence": "negative", "suggested_improvement": "Cite Peyré et al. (2019) or provide justification for the use of smoothed indicator functions of Laguerre cells.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8148844242095947, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8377414345741272}, {"unit_index": 4, "inspected_object": "Section 2.2.1 restatement of Santambrogio (2015) results", "observation": "Results from Santambrogio (2015) are restated but never used in subsequent derivations.", "reasoning": "Theoretical results should serve a functional role in the argument; unused material indicates poor argumentative economy and rhetorical clutter.", "judgment": "The paper's structure is inefficient and rhetorically weak.", "valence": "negative", "suggested_improvement": "Remove unused theoretical results or demonstrate their utility in the derivations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7049928903579712, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7360769510269165}, {"unit_index": 5, "inspected_object": "Theorem 4's 'disjoint bounded sets' assumption", "observation": "The assumption requires disjoint source and target supports, which contradicts typical domain adaptation scenarios where supports overlap.", "reasoning": "Assumptions must be compatible with the intended application domain; if the assumption excludes the motivating scenario, the result has no practical relevance.", "judgment": "The theorem is practically irrelevant for the claimed application.", "valence": "negative", "suggested_improvement": "Relax the disjointness assumption or discuss applicability to overlapping support domains.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8051358461380005, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.771085262298584}]}, {"review_id": "054Bnamrrw", "paper_id": "t3ZMiHhqXm", "paper_title": "Person-Centric Annotations of LAION-400M: Auditing Bias and Its Transfer to Models", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper as technically engaged and valuable for its annotation contribution but incomplete in rigor and scope. Key critiques focus on the currency of the dataset, dependencies in measurement tools (captions, detectors), misalignment risks, and lack of explanation for residual bias variance.", "units": [{"unit_index": 0, "inspected_object": "Finetuning of classifiers for demographic categories instead of using off-the-shelf models", "observation": "The reviewer explicitly praises the decision to finetune classifiers, noting it handles female, male, mixed and unclear cases for gender and race.", "reasoning": "This approach demonstrates domain adaptation and acknowledges categorical ambiguity, which is valued over simpler off-the-shelf solutions.", "judgment": "Positive evaluation of the annotation methodology choice as a strength.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7784690856933594, "reasoning_key": "design_justification", "reasoning_sim": 0.7968094944953918}, {"unit_index": 1, "inspected_object": "Computational cost justification for avoiding ensemble VLMs in full-dataset annotation", "observation": "The reviewer questions whether ensemble VLMs were avoided due to computational costs rather than principled design.", "reasoning": "If the finetuning approach was driven by resource constraints rather than methodological superiority, its generalizability to other datasets (like CC-12M or DataComp) may be limited.", "judgment": "Uncertainty regarding the transferability and replicability of the annotation pipeline beyond the specific dataset studied.", "valence": "conditional", "suggested_improvement": "Clarify if the pipeline is transferable to other datasets and justify the choice against ensemble VLMs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8263914585113525, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8250014781951904}, {"unit_index": 2, "inspected_object": "Potential biases in the YOLOv11-l object detection component", "observation": "The reviewer asks how the authors verify that YOLOv11-l does not have gender/racial biases.", "reasoning": "If the detector systematically misses or misdetects certain demographic groups, downstream annotations would inherit that bias, creating an unexamined link in the annotation chain.", "judgment": "Concern about a potential blind spot where detection bias could compromise downstream annotation validity.", "valence": "negative", "suggested_improvement": "Provide evidence of neutrality for the YOLOv11-l detector.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8061032891273499, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.768924355506897}, {"unit_index": 3, "inspected_object": "Currency and relevance of the LAION-400M dataset for modern model analysis", "observation": "The reviewer notes that models trained on LAION-400M are outdated and that biases are expected to grow with scale.", "reasoning": "Findings from older, smaller datasets may understate the problem in larger collections where contemporary models actually train, limiting the inferential reach of the paper's claims about current systems.", "judgment": "Critique that the dataset's age and size limit the relevance of the findings for understanding modern model biases.", "valence": "negative", "suggested_improvement": "Conduct similar experiments on more modern datasets like LAION-2B or DataComp.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8262638449668884, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7993974089622498}, {"unit_index": 4, "inspected_object": "Dependency on automatically generated captions for identity-topic associations", "observation": "The reviewer notes that topic analysis relies on captions from pretrained caption generators, which may carry their own biases and lack accuracy verification.", "reasoning": "Generated captions introduce a potential confound; if the generator has demographic associations, the topic analysis could reflect the generator's biases rather than the dataset's actual content.", "judgment": "Concern that methodological independence is compromised, making it hard to attribute observed patterns solely to the dataset.", "valence": "negative", "suggested_improvement": "Acknowledge the dependency, address it, or validate the caption generation step; consider analyzing original captions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7674586772918701, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7611632347106934}, {"unit_index": 5, "inspected_object": "Caption-image misalignment in linking social categories to image content", "observation": "The reviewer cites literature indicating that caption presence does not ensure the corresponding image actually contains the person described.", "reasoning": "Misalignment breaks the inference link between caption mentions of social categories and the actual demographic composition of images, potentially invalidating the bias measurements.", "judgment": "Technical concern that the core assumption linking captions to image content is flawed due to known misalignment issues.", "valence": "negative", "suggested_improvement": "Address the risk of misalignment affecting bias measurements.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7414249181747437, "reasoning_key": "design_justification", "reasoning_sim": 0.7799434661865234}, {"unit_index": 6, "inspected_object": "Scope of downstream bias analysis relative to crime-related and sentiment-based categories", "observation": "The reviewer notes the analysis focuses on social categories but suggests crime-related and sentiment-based analysis would have been valuable.", "reasoning": "If the dataset analysis examined these categories, the downstream model analysis should mirror this scope to provide a complete picture of bias transfer.", "judgment": "Critique that the downstream analysis is incomplete in scope compared to the dataset-level analysis.", "valence": "negative", "suggested_improvement": "Include crime-related and sentiment-based analysis in the downstream model evaluation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7476804852485657, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.6950811743736267}, {"unit_index": 7, "inspected_object": "Explanation of residual variance in downstream bias (30-40%)", "observation": "The reviewer notes the authors report 60-70% of downstream bias is explained by dataset bias but do not discuss the causes of the remaining variance.", "reasoning": "Explanatory completeness requires discussing candidate explanations for the unexplained portion, such as model architecture, training dynamics, or sampling strategies.", "judgment": "Critique that the quantitative claim lacks necessary discussion of its limitations and residuals.", "valence": "negative", "suggested_improvement": "Discuss candidate explanations for the unexplained variance in downstream bias.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8159003257751465, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7688058614730835}]}, {"review_id": "055vR7xNBh", "paper_id": "l8PM2RukoP", "paper_title": "Kaleido: Open-Sourced Multi-Subject Reference Video Generation Model", "decision": "Reject", "summary": "The reviewer operates with qualified endorsement, accepting the data pipeline and positional encoding as plausible contributions while critically probing the validity of evaluation metrics and the robustness of the data generation process against artifacts and scalability limits.", "units": [{"unit_index": 0, "inspected_object": "Data construction pipeline", "observation": "The reviewer identifies the pipeline as a strength, noting it enhances subject and scene diversity, improves data fidelity, and ensures clear separation of subjects from irrelevant components.", "reasoning": "The reviewer accepts the paper's claims that the pipeline addresses multi-subject consistency and background disentanglement through its design for diversity and fidelity.", "judgment": "Positive evaluation of the data pipeline's design intent.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8042816519737244, "reasoning_key": "merit_recognition", "reasoning_sim": 0.788788914680481}, {"unit_index": 1, "inspected_object": "Reference-based position encoding (R-RoPE)", "observation": "The reviewer describes R-RoPE as emphasizing references to achieve better results.", "reasoning": "The reviewer evaluates the mechanism at a functional level ('does this plausibly do what it claims') rather than requiring technical novelty or elegance analysis.", "judgment": "Acceptance of the positional encoding's utility based on reported outcome.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8001590371131897, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8252456188201904}, {"unit_index": 2, "inspected_object": "Evaluation metrics (CLIP) for subject consistency", "observation": "The reviewer critiques CLIP as not fine-grained enough for subject consistency.", "reasoning": "The reviewer applies a domain-specific norm of construct validity: identity preservation is core to subject consistency, and global image-text alignment (CLIP) is too coarse to capture identity-level fidelity compared to face recognition metrics.", "judgment": "Negative evaluation of the measurement validity for subject consistency.", "valence": "negative", "suggested_improvement": "Use face recognition metrics for human faces.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8477619290351868, "reasoning_key": "design_justification", "reasoning_sim": 0.7871798276901245}, {"unit_index": 3, "inspected_object": "Robustness of data pipeline against synthetic editing artifacts", "observation": "The reviewer questions how artifacts from image editing methods like Flux affect generated video, specifically citing 'Flux redux reposes the human' introducing subject inconsistencies.", "reasoning": "The reviewer probes a failure mode where the pipeline's reliance on synthetic data introduces latent artifacts that may propagate as training signal, potentially undermining the claimed fidelity despite filtering steps.", "judgment": "Uncertainty regarding the robustness of the data pipeline's benefits against tool-specific artifacts.", "valence": "conditional", "suggested_improvement": "Analyze how artifacts from image editing tools impact generated video and subject consistency.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8417119383811951, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7839767932891846}, {"unit_index": 4, "inspected_object": "Scalability and limiting factors of multi-subject insertion", "observation": "The reviewer asks how many subjects can be inserted simultaneously and what the limiting factor is.", "reasoning": "The reviewer seeks a theory of failure for the method's boundaries, testing if the authors understand whether bottlenecks lie in positional encoding capacity, attention mechanisms, data coverage, or base model limits.", "judgment": "Uncertainty about the method's scalability ceiling and the authors' understanding of its constraints.", "valence": "uncertain", "suggested_improvement": "Articulate the limiting factor for the number of subjects that can be inserted into the video.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8383524417877197, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7928444743156433}]}, {"review_id": "058pXEshXa", "paper_id": "o31oAjIra3", "paper_title": "Segmentation Helps Understanding: Mask-Infused Vision-Language Pre-training for 3D Medical Images", "decision": null, "summary": "The reviewer critically evaluates the paper's novelty, generalizability, and evaluation rigor, identifying issues with methodological triviality, limited dataset coverage, modest performance gains, noisy evaluation labels, and unfair baseline comparisons.", "units": [{"unit_index": 0, "inspected_object": "Methodological components (Tversky loss, lightweight decoder, pre-training dataset assembly)", "observation": "The reviewer identifies the core ingredients as a pre-training dataset with report and segmentation masks, noting it is not contributed by the authors, and singles out the Tversky loss and lightweight decoder as 'extremely trivial'.", "reasoning": "The reviewer applies an implicit standard that a methodological contribution must introduce something beyond the assembly of existing techniques or standard tools. The lack of author-contributed data and the use of 'trivial' components lead to a discount on novelty.", "judgment": "The work lacks novelty and methodology advancements; the contribution is viewed as assembling existing pieces rather than generating new machinery.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8606292009353638, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8785654306411743}, {"unit_index": 1, "inspected_object": "Dataset coverage and generalizability (availability of segmentation masks for various abnormalities)", "observation": "The reviewer states that only lung nodules and effusion have segmentation masks shown in the paper, contrasting this with the reality that most lung diseases do not have voxel-level masks, and infers potential cherry-picking of lesion types.", "reasoning": "The reviewer uses an ecological validity standard: a pre-training method should be broadly applicable to real-world datasets comprising many disease types. Limited mask availability questions the method's applicability to the full heterogeneity of medical data.", "judgment": "The method's generalizability is questionable due to limited coverage of abnormality types in the training data.", "valence": "negative", "suggested_improvement": "Clarify how the method addresses lesions without segmentation masks and provide evidence of applicability to other disease types (e.g., abdominal CT).", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8164198398590088, "reasoning_key": "design_justification", "reasoning_sim": 0.7742176651954651}, {"unit_index": 2, "inspected_object": "Magnitude of performance improvement (1.8% average AUC increase)", "observation": "The reviewer quantifies the improvement as only 1.8% average AUC increase and calls it 'far from substantial'.", "reasoning": "The reviewer applies a threshold judgment regarding what magnitude of improvement constitutes a meaningful contribution. The observed value falls below this implicit threshold.", "judgment": "The reported improvements are not significant.", "valence": "negative", "suggested_improvement": "Justify why the magnitude of improvement matters.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7634687423706055, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7798065543174744}, {"unit_index": 3, "inspected_object": "Evaluation ground truth and label reliability", "observation": "The reviewer questions the steps for obtaining ground truth of nodules and effusion, suggesting they are 'not very reasonable' and that noisy labels are still used for evaluation.", "reasoning": "The reviewer applies a rigor standard requiring valid ground truth. If labels are noisy, the reported gains may be unreliable, undermining the validity of the evaluation.", "judgment": "The evaluation's trustworthiness is compromised by potentially noisy labels.", "valence": "negative", "suggested_improvement": "Provide methodological transparency on how evaluation labels were obtained and their reliability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8580374121665955, "reasoning_key": "construct_validity", "reasoning_sim": 0.8420847654342651}, {"unit_index": 4, "inspected_object": "Fairness of experimental comparisons (segmentation-trained vs. non-segmentation-trained baselines)", "observation": "The reviewer notes that previous CLIP models did not conduct segmentation training, making the comparison in Table 5 unfair when highlighting improvements in segmentation.", "reasoning": "The reviewer applies a fairness standard: comparing a method trained with segmentation supervision against methods without such supervision is an apples-to-oranges comparison, as the former has access to extra information.", "judgment": "The segmentation comparison is not fair and does not demonstrate superiority over baselines in a meaningful way.", "valence": "negative", "suggested_improvement": "Use more appropriate baselines or reframe the results to account for the difference in training regimes.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8656399250030518, "reasoning_key": "fair_comparison", "reasoning_sim": 0.746290385723114}]}, {"review_id": "06BpP3IMdQ", "paper_id": "H4RgGzx4iL", "paper_title": "VIRTUE: Visual-Interactive Text-Image Universal Embedder", "decision": "Accept (Poster)", "summary": "The reviewer acknowledges the novelty and potential of visual-interactive embeddings but raises significant concerns about execution, including insufficient motivation evidence, dataset validity and scope issues, model complexity, and experimental fairness. The review is characterized by constructive skepticism, demanding stronger evidence for claims and more rigorous evaluation practices.", "units": [{"unit_index": 0, "inspected_object": "The paper's central motivation regarding the gap in current models' support for visual prompts.", "observation": "The reviewer affirms that current models do not support visual prompts but finds the argumentation insufficient and the claim that existing models 'only accept text' borderline incorrect, distinguishing between capability and design intent.", "reasoning": "The reviewer expects a robust demonstration of the claimed gap through real-world use-case examples rather than just a technical observation, and demands precise language that accurately reflects model capabilities versus design intentions.", "judgment": "The motivation is acknowledged as having potential but is currently under-supported by evidence and rhetorical precision.", "valence": "negative", "suggested_improvement": "Include examples of real-world uses to argue the point more strongly.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8021371960639954, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7834809422492981}, {"unit_index": 1, "inspected_object": "The SCaR dataset's scope and modality coverage.", "observation": "The dataset only explores visual prompts, ignoring multimodal scenarios where users might combine bounding boxes with text queries.", "reasoning": "Real-world use cases often involve both visual and text prompts; limiting the dataset to visual-only prompts artificially narrows the evaluation scope and may not reflect practical applicability.", "judgment": "The dataset scope is limited and does not fully capture relevant application spaces.", "valence": "negative", "suggested_improvement": "Explore combined visual and text prompts to better reflect real-world use cases.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8189593553543091, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7819758653640747}, {"unit_index": 2, "inspected_object": "The validity and quality of the SCaR dataset generated via GPT-4V.", "observation": "The dataset relies heavily on GPT-4V without metrics like inter-annotator agreement, raising concerns about data quality and generalizability.", "reasoning": "Without human validation metrics, it is unclear if performance gains stem from genuine visual-interactive understanding or from overfitting to specific quirks of the GPT-4V generation process.", "judgment": "The dataset's trustworthiness and the generalizability of results are uncertain due to lack of quality validation.", "valence": "negative", "suggested_improvement": "Provide metrics such as inter-annotator agreement to evaluate dataset quality and assess generalizability beyond GPT-4V artifacts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8136634230613708, "reasoning_key": "design_justification", "reasoning_sim": 0.8104499578475952}, {"unit_index": 3, "inspected_object": "Figure 2 depicting the dataset filtering pipeline.", "observation": "Figure 2 omits elements involved in the filtering process, such as WordNet.", "reasoning": "Methodology figures should be complete and accurately represent all steps in the pipeline to avoid misleading readers about the construction process.", "judgment": "The figure is incomplete and potentially misleading.", "valence": "negative", "suggested_improvement": "Update Figure 2 to include all elements in the filtering process, such as WordNet.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8360254168510437, "reasoning_key": "merit_recognition", "reasoning_sim": 0.6990537643432617}, {"unit_index": 4, "inspected_object": "The proposed method's impact on model complexity.", "observation": "The method increases model complexity, which may hinder running, optimizing, and adapting the model.", "reasoning": "There is a trade-off between performance gains and practical viability; increased complexity imposes costs that must be justified by sufficient utility.", "judgment": "The increase in complexity is a concern for practical deployment and optimization.", "valence": "negative", "suggested_improvement": "Discuss the trade-offs between performance gains and increased complexity, or explore ways to mitigate the overhead.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.846535325050354, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8331584334373474}, {"unit_index": 5, "inspected_object": "Selection of the MMEB benchmark for evaluation.", "observation": "The reviewer questions why MMEB was chosen over other comparable benchmarks.", "reasoning": "Justification of evaluation framework choices is necessary to ensure the assessment is comprehensive and contextually appropriate.", "judgment": "The choice of benchmark lacks justification and may limit the perceived relevance of the evaluation.", "valence": "conditional", "suggested_improvement": "Provide justification for choosing MMEB over alternative comparable benchmarks.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.854457437992096, "reasoning_key": "merit_recognition", "reasoning_sim": 0.789069652557373}, {"unit_index": 6, "inspected_object": "The generality of the interactive selection mechanism relative to SAM2.", "observation": "It is unclear if the model supports arbitrary interactive human selection independent of the SAM2 segmentation model.", "reasoning": "If the capability is tied specifically to SAM2, the method may not be truly 'interactive' in a general sense, limiting its applicability.", "judgment": "The interaction fidelity and generality of the visual prompting mechanism are uncertain.", "valence": "conditional", "suggested_improvement": "Experiment with arbitrary interactive human selection to demonstrate that the model is not solely dependent on SAM2.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8461349606513977, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7743740081787109}, {"unit_index": 7, "inspected_object": "Fine-tuning asymmetry among baseline models on MMEB.", "observation": "Not all competitive baselines (e.g., GME-7B) were fine-tuned on MMEB, creating an uneven comparison.", "reasoning": "Fair comparisons require equivalent optimization effort for baselines unless there is a clear justification for asymmetry; unequal treatment may inflate the reported improvements.", "judgment": "The experimental fairness is compromised by inconsistent fine-tuning practices.", "valence": "negative", "suggested_improvement": "Clarify the selection process for fine-tuning or ensure all competitive baselines receive equivalent fine-tuning treatment.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8489624261856079, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8477210402488708}, {"unit_index": 8, "inspected_object": "Section naming in Appendix E.", "observation": "Appendix E is titled 'extensive experiments'.", "reasoning": "Academic writing should avoid rhetorical inflation; section titles should be concise and accurate.", "judgment": "The title is stylistically inappropriate.", "valence": "negative", "suggested_improvement": "Rename the section to simply 'experiments'.", "support_status": "memo_inferred", "confidence": "high", "object_key": "related_work", "object_sim": 0.7262537479400635, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7568153738975525}]}, {"review_id": "07C5xxGb1Q", "paper_id": "37SbPItNb8", "paper_title": "Adaptive Residual-Update Steering for Low-Overhead Hallucination Mitigation in Large Vision-Language Models", "decision": "Reject", "summary": "The reviewer positively evaluates the method's efficiency, integration, and empirical breadth but raises significant reservations about its theoretical framing, configurational fragility, evaluation scope, and lack of mechanistic diagnostics.", "units": [{"unit_index": 0, "inspected_object": "The RUDDER method's integration into standard decoding loops and its lack of extra forward passes.", "observation": "The reviewer notes the method requires no additional forward passes and integrates with standard decoding loops, while also noting low-overhead implementation.", "reasoning": "The reviewer values concrete efficiency gains that do not compromise the inference pipeline, viewing these as verifiable technical achievements.", "judgment": "Positive evaluation of the method's practical efficiency and ease of integration.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8159390687942505, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8188543319702148}, {"unit_index": 1, "inspected_object": "The adaptive gating mechanism compared to fixed-strength steering.", "observation": "The adaptive gate outperforms fixed-strength steering in ablation studies.", "reasoning": "The reviewer interprets this performance difference as evidence that token-wise gating provides a genuine advantage over static approaches.", "judgment": "Positive assessment of the core mechanism's efficacy relative to baselines.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7747097015380859, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8120415210723877}, {"unit_index": 2, "inspected_object": "The breadth of the empirical evaluation across models and decoding strategies.", "observation": "The paper evaluates performance on CHAIR and POPE across three LVLMs and three decoding strategies, including latency/throughput measurements.", "reasoning": "The reviewer considers cross-model and cross-decoding coverage essential for validating generalizability and robustness.", "judgment": "Positive evaluation of the experimental scope and rigor.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8648051619529724, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7969418168067932}, {"unit_index": 3, "inspected_object": "The theoretical framing of the gate as 'Bayesian-inspired'.", "observation": "The gate is heuristic and lacks formal guarantees of improvement in negative log likelihood.", "reasoning": "The reviewer holds that claims of theoretical inspiration should be backed by formal analysis or explicitly acknowledged as loose analogies; without such backing, the label is seen as decorative.", "judgment": "Negative judgment regarding the honesty and precision of the theoretical positioning.", "valence": "negative", "suggested_improvement": "Temper the theoretical claims or provide formal analysis supporting the Bayesian analogy.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8532074093818665, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7791938781738281}, {"unit_index": 4, "inspected_object": "The selection of the layer from which CARD vectors are extracted.", "observation": "The optimal layer choice is model-specific (late layers for LLaVA/Idefics2, early for InstructBLIP).", "reasoning": "The reviewer views model-specific tuning as evidence of practical fragility and a sensitive trade-off between CHAIR scores and recall, implying a lack of a principled, generalizable configuration criterion.", "judgment": "Negative judgment regarding the method's generalizability and configurational stability.", "valence": "negative", "suggested_improvement": "Provide a principled way to select the layer or demonstrate robustness across configurations.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7827346920967102, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8038879632949829}, {"unit_index": 5, "inspected_object": "The scope of capability evaluation beyond hallucination metrics.", "observation": "The evaluation does not include broader capability benchmarks like MM-Vet.", "reasoning": "The reviewer assumes that hallucination mitigation methods must be tested on general capability benchmarks to ensure they do not trade one failure mode for another (e.g., reducing hallucinations but degrading general competence).", "judgment": "Negative judgment regarding the completeness of the evaluation scope.", "valence": "negative", "suggested_improvement": "Add evaluation on MM-Vet to test for capability preservation.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8137398958206177, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.748534619808197}, {"unit_index": 6, "inspected_object": "The presence of faithfulness diagnostics verifying mechanistic impact.", "observation": "The paper shows outcome metrics and internal geometry but lacks diagnostics verifying that the method reduces language-prior reliance rather than merely suppressing certain token types.", "reasoning": "The reviewer demands mechanistic evidence to distinguish between genuine reduction of language-prior reliance and degenerate solutions that simply bias the model away from high-frequency tokens.", "judgment": "Negative judgment regarding the interpretability and causal validity of the mechanism.", "valence": "negative", "suggested_improvement": "Add faithfulness diagnostics to verify the method truly reduces language-prior reliance.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7795206904411316, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7845764756202698}, {"unit_index": 7, "inspected_object": "The precise definition and pooling operation of the CARD vector.", "observation": "The description of CARD extraction is abstract, lacking surgical specificity on tensor flow (per-layer vs per-head, mean vs median, head-weighted).", "reasoning": "The reviewer needs precise recipe details to assess reproducibility and determine if choices are principled or arbitrary.", "judgment": "Uncertainty regarding the reproducibility and definitional clarity of the core component.", "valence": "conditional", "suggested_improvement": "Clarify exactly which tensors are pooled and what pooling operation is used.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8014416694641113, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.784271240234375}, {"unit_index": 8, "inspected_object": "The hyperparameter selection for the gate (g_min, g_max, softplus temperature).", "observation": "The method for choosing gate hyperparameters is not clearly explained.", "reasoning": "If hyperparameters require careful per-model tuning, the claim of 'low-overhead' is weakened in practice; if set by a principled rule, the method is more robust.", "judgment": "Uncertainty regarding the tuning burden and practical scalability of the method.", "valence": "conditional", "suggested_improvement": "Clarify how gate hyperparameters were chosen and whether they require per-model tuning.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8524284958839417, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8650988340377808}]}, {"review_id": "07VlEpZwt1", "paper_id": "GpL66XgjjF", "paper_title": "When Thinking Backfires: Mechanistic Insights into Reason-induced Misalignment", "decision": "Accept (Poster)", "summary": "The reviewer endorses the paper's methodological rigor and core empirical findings while systematically challenging the universality of its claims. The dominant evaluative activity involves demanding broader empirical evidence for generalization and tighter controls for causal attribution.", "units": [{"unit_index": 0, "inspected_object": "Diversity of open-source LLMs (dense and MoE), multiple math-reasoning datasets, and the HEx-PHI safety benchmark.", "observation": "The paper studies RIM across diverse model architectures and datasets using a rigorous safety benchmark.", "reasoning": "Coverage across diverse configurations serves as a proxy for validity; if the phenomenon appears across different setups, it is less likely to be an artifact of a single configuration.", "judgment": "The empirical base provides credible support for the existence of the described phenomena.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8416507244110107, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7933060526847839}, {"unit_index": 1, "inspected_object": "Methodological toolkit including unsupervised probing, attention-head ablations, and neuron-level causal interventions.", "observation": "The paper constructs a causal chain from behavioral observation to internal mechanism, supported by Figures 3 and 4 linking co-attention on CoT tokens and refusal dynamics.", "reasoning": "The use of causal interventions and specific visualizations supports the mechanistic story that these elements are intricately linked.", "judgment": "The methodological approach solidifies the claim regarding the link between attention mechanisms and refusal dynamics.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7779996395111084, "reasoning_key": "merit_recognition", "reasoning_sim": 0.6839583516120911}, {"unit_index": 2, "inspected_object": "RAS metric and its comparison against baselines.", "observation": "The reviewer identifies RAS as a notable methodological addition where ablations demonstrate that specific neural circuits are disproportionately affected.", "reasoning": "The provision of a new tool, demonstrated superiority over existing tools, and its use in making causal claims renders the associated claims credible.", "judgment": "The RAS metric is a valid and useful methodological contribution.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.843457043170929, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7639688849449158}, {"unit_index": 3, "inspected_object": "Reliance on HEx-PHI as the sole benchmark for safety evaluation.", "observation": "The paper uses only a single benchmark for safety evaluation, a limitation explicitly acknowledged by the authors.", "reasoning": "Acknowledging a limitation does not mitigate its impact; reliance on a single benchmark introduces significant risk of dataset overfitting or unmeasured generalization failure, undermining claims about the universality of RIM.", "judgment": "The scope of safety-related claims is unsupported due to narrow empirical evidence.", "valence": "negative", "suggested_improvement": "Include at least one additional, established safety benchmark to strengthen claims about the universality of RIM.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7711069583892822, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8057073950767517}, {"unit_index": 4, "inspected_object": "Generalization of RIM findings to domains beyond math reasoning.", "observation": "The paper implies RIM extends to commonsense, coding, and multi-step logic, but this is not empirically tested.", "reasoning": "Empirical claims about a phenomenon's scope require empirical evidence of that scope; motivation for focusing on math is insufficient justification for assuming generality.", "judgment": "Claims of domain generality are currently unearned.", "valence": "negative", "suggested_improvement": "Conduct a brief pilot evaluation in another domain to signal that the phenomenon is not math-specific.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8246086239814758, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8134017586708069}, {"unit_index": 5, "inspected_object": "Construction of 'effort-minimizing' reasoning patterns and their controls.", "observation": "Further quantification of linguistic/semantic divergence between control and target Chain-of-Thoughts (CoTs) is needed to clarify causal attribution.", "reasoning": "Without deeper linguistic analysis, length or unrelated stylistic features may confound the effect, meaning the intended independent variable (effort-minimization) might not be the actual driver of results.", "judgment": "Causal attribution for effort-minimization effects is uncertain due to potential confounds.", "valence": "negative", "suggested_improvement": "Quantify linguistic and semantic divergence between control and target CoTs to rule out confounding factors like length or style.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8458485007286072, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7899739742279053}, {"unit_index": 6, "inspected_object": "Harmonic mean aggregation function in the RAS metric.", "observation": "The reviewer questions whether the RAS formulation is principled or arbitrary.", "reasoning": "A proposed metric's functional form should be motivated by theoretical considerations or validated against alternatives; uncertainty remains about why this specific aggregation was chosen.", "judgment": "The design principles underlying the RAS aggregation function are unclear.", "valence": "uncertain", "suggested_improvement": "Provide an ablation on aggregation methods to justify the choice of harmonic mean.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8243814706802368, "reasoning_key": "construct_validity", "reasoning_sim": 0.8030057549476624}]}, {"review_id": "08C8ttb1M5", "paper_id": "avwNGWtiHF", "paper_title": "ASSESS: A Semantic and Structural Evaluation Framework for Statement Similarity", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper based on landscape positioning and accessibility norms rather than internal technical validity, citing missing code, lack of comparison with SOTA models, and absence of retrieval benchmarks as reasons the paper is not yet ready for publication despite sound ideas.", "units": [{"unit_index": 0, "inspected_object": "Availability of source code for the proposed metric", "observation": "No source code is provided in the submission materials.", "reasoning": "The reviewer holds a norm that code availability is non-negotiable for methods papers, particularly when the core contribution is a metric others need to apply; the absence undermines reproducibility and trustworthiness.", "judgment": "The lack of code is treated as a fatal flaw or significant barrier to usability.", "valence": "negative", "suggested_improvement": "Provide the source code.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8495911955833435, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7884249091148376}, {"unit_index": 1, "inspected_object": "Comparison with current state-of-the-art autoformalization models", "observation": "The paper does not compare TransTED against specific modern LLM-based autoformalizers (e.g., Kimina-Autoformalizer-7B, Goedel-Formalizer-V2-8B).", "reasoning": "A metric's value is proven by its ability to discriminate among the outputs of the best current generators; without these comparisons, the evaluation is incomplete regarding the paper's relevance to the current landscape.", "judgment": "The evaluation is insufficiently situated within the current frontier of the task.", "valence": "negative", "suggested_improvement": "Compare TransTED with Kimina-Autoformalizer-7B and Goedel-Formalizer-V2-8B.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8257809281349182, "reasoning_key": "construct_validity", "reasoning_sim": 0.825218677520752}, {"unit_index": 2, "inspected_object": "Utility of the metric as a retrieval tool", "observation": "The paper evaluates the metric primarily on a curated benchmark but does not test it as an embedding-based retrieval system.", "reasoning": "A similarity metric should demonstrate broader utility beyond a specific benchmark setting, specifically by measuring recall accuracy in a practical retrieval application.", "judgment": "The metric's generality and practical applicability are unproven.", "valence": "conditional", "suggested_improvement": "Compare recall accuracy with some embedding-based retrieval systems.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.850940465927124, "reasoning_key": "construct_validity", "reasoning_sim": 0.7506563663482666}]}, {"review_id": "09LLgO1c07", "paper_id": "7Sph4KyeYO", "paper_title": "Constrained Decoding of Diffusion LLMs with Context-Free Grammars", "decision": "Accept (Poster)", "summary": "The reviewer acts as a technical auditor, challenging the paper's novelty, correctness, and practical viability. Key criticisms include lack of differentiation from prior art, missing complexity and correctness proofs, termination risks under adversarial distributions, and inefficient design choices. The review is conditional, suggesting that addressing these theoretical and practical gaps could improve the assessment.", "units": [{"unit_index": 0, "inspected_object": "Novelty of the algorithmic contribution (intersection of regular languages and CFGs)", "observation": "The reviewer asserts that the core algorithm is already known and cites a specific EACL 2023 paper as prior art.", "reasoning": "The reviewer applies a standard of literature coverage and novelty, expecting that algorithmic contributions should be distinguished from existing work in established venues. The citation of a concrete prior work serves as evidence that the claimed novelty is overstated.", "judgment": "The paper's claim to novelty is questionable due to insufficient differentiation from existing literature.", "valence": "negative", "suggested_improvement": "Cite and discuss the EACL 2023 paper to clarify the distinction between the current work and prior art.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.85759037733078, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8788568377494812}, {"unit_index": 1, "inspected_object": "Citation appropriateness for CFG emptiness algorithm", "observation": "The paper cites a random StackExchange post for the CFG emptiness algorithm instead of a canonical textbook.", "reasoning": "The reviewer expects algorithmic papers to ground standard results in canonical references (e.g., Sipser) rather than informal sources, viewing this as a lack of rigor and awareness of theoretical norms.", "judgment": "The citation practice is inappropriate and signals a lack of grounding in standard theory.", "valence": "negative", "suggested_improvement": "Replace the StackExchange citation with a standard theory of computation textbook reference.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8100286722183228, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8113790154457092}, {"unit_index": 2, "inspected_object": "Complexity analysis of the presented algorithm", "observation": "The paper introduces an algorithm to avoid a stated complexity issue but fails to provide the complexity analysis for the new algorithm.", "reasoning": "The reviewer identifies a gap in the argumentative structure: the motivation relies on complexity avoidance, but the justification (complexity analysis of the solution) is missing, leaving the benefit unverified.", "judgment": "The algorithmic contribution lacks necessary theoretical justification regarding its efficiency.", "valence": "negative", "suggested_improvement": "Provide a formal complexity analysis for the proposed algorithm.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.816599428653717, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8046869039535522}, {"unit_index": 3, "inspected_object": "Correctness of Algorithm 3 regarding tokens vs. lexemes", "observation": "The paper does not state correctness guarantees for Algorithm 3, specifically concerning the handling of tokens versus lexemes.", "reasoning": "Drawing on domain-specific knowledge of grammar-constrained decoding implementations, the reviewer notes that failing to distinguish tokens and lexemes leads to incorrect behavior. Without a formal correctness statement and assumptions about the lexer, there is no guarantee that invalid sequences are correctly rejected.", "judgment": "The soundness of the core algorithm is unproven and potentially flawed due to token/lexeme handling ambiguities.", "valence": "negative", "suggested_improvement": "Provide a formal correctness statement for Algorithm 3, including assumptions about lexer operation, and cite relevant prior work on token/lexeme issues.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7678406834602356, "reasoning_key": "design_justification", "reasoning_sim": 0.811729371547699}, {"unit_index": 4, "inspected_object": "Termination properties under adversarial distributions", "observation": "The reviewer constructs a counterexample where an adversarial distribution (more open parentheses than close) leads to non-termination.", "reasoning": "The reviewer tests the algorithm's robustness against worst-case model outputs. The existence of such a counterexample suggests a fundamental limitation in the approach's termination guarantees, which may explain empirical inefficiencies (long generation times).", "judgment": "The algorithm lacks guaranteed termination under general output distributions, representing a significant theoretical weakness.", "valence": "negative", "suggested_improvement": "Address the termination issue, possibly by explaining or justifying the use of random string generation as a fallback mechanism.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8202192783355713, "reasoning_key": "design_justification", "reasoning_sim": 0.8678317070007324}, {"unit_index": 5, "inspected_object": "Distribution distortion and sampling optimality", "observation": "The paper is silent on distribution distortion, despite prior work (Dingo) providing theoretical guarantees for optimal sampling.", "reasoning": "The reviewer expects constrained decoding papers to either provide distributional guarantees or explicitly acknowledge their absence, positioning the paper as incomplete relative to established standards in the field.", "judgment": "The paper fails to address a critical dimension of quality (distributional fidelity) that has been addressed in related work.", "valence": "negative", "suggested_improvement": "Discuss distribution distortion, compare with methods like Dingo that offer theoretical guarantees, or provide empirical evidence of low distortion.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8384515643119812, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7726053595542908}, {"unit_index": 6, "inspected_object": "Practical efficiency and design choice regarding token masks", "observation": "The paper does not directly compute token masks, relying instead on guessing and checking.", "reasoning": "The reviewer argues that newer GCD algorithms avoid expensive masking approaches and that direct mask computation would drastically reduce overhead. The current design is viewed as inefficient compared to integration-friendly alternatives.", "judgment": "The method's design choices lead to unnecessary computational overhead and poor integration potential.", "valence": "negative", "suggested_improvement": "Consider computing token masks directly to reduce overhead, citing examples like llguidance and GreatGramma.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8056970834732056, "reasoning_key": "design_justification", "reasoning_sim": 0.8113646507263184}, {"unit_index": 7, "inspected_object": "Experimental reporting: Latency metrics", "observation": "The paper reports relative latency increase rather than absolute latency-per-token.", "reasoning": "Relative metrics depend on hardware specifics (GPU), making comparisons across different setups difficult. Absolute metrics are required for reproducibility and fair comparison.", "judgment": "The experimental reporting is insufficient for reproducibility and cross-hardware comparison.", "valence": "negative", "suggested_improvement": "Report absolute latency-per-token alongside relative increases.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8511918783187866, "reasoning_key": "design_justification", "reasoning_sim": 0.7735344767570496}, {"unit_index": 8, "inspected_object": "Experimental reporting: SMILES evaluation metric", "observation": "The paper uses an LLM as a judge for SMILES semantic quality.", "reasoning": "The reviewer points out that established, objective metrics exist for SMILES validity/quality, making the subjective LLM judge unnecessary and less rigorous.", "judgment": "The evaluation methodology for chemical validity is suboptimal compared to available objective standards.", "valence": "negative", "suggested_improvement": "Use established objective metrics for SMILES evaluation instead of an LLM judge.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8295057415962219, "reasoning_key": "construct_validity", "reasoning_sim": 0.7511110901832581}]}, {"review_id": "09XFaufsg8", "paper_id": "4SxAu9zMVC", "paper_title": "JAPAN: Joint Adaptive Prediction Areas with Normalising Flow", "decision": "Accept (Poster)", "summary": "The reviewer conducts a practical feasibility audit and careful proofreading, positively evaluating the method's geometric flexibility and theoretical presence while raising concerns about computational scalability for region enumeration, experimental control in comparisons, conditional coverage testing, and time-series applicability, and negatively judging the mathematical notation rigor.", "units": [{"unit_index": 0, "inspected_object": "Use of log-likelihood as conformity scores vs latent-space residuals", "observation": "JAPAN uses log-likelihood as conformity scores, whereas CONTRA uses latent-space residuals.", "reasoning": "The reviewer reasons that this design choice enables the method to capture arbitrary, disjoint, and non-convex shapes, which is a capability strength derived directly from the architectural decision.", "judgment": "Positive evaluation of the method's geometric flexibility and capability.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8210362792015076, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7879416346549988}, {"unit_index": 1, "inspected_object": "Prediction volume performance relative to coverage level", "observation": "JAPAN consistently achieves lower prediction volume while maintaining coverage level in empirical results.", "reasoning": "The reviewer infers method quality from these results but qualifies this judgment by noting the absence of explicit specification regarding shared flow architecture and training procedures for baselines (JAPAN, CONTRA, PCP), creating uncertainty about whether differences reflect the conformal scoring method or implementation details.", "judgment": "Provisional positive assessment of efficiency, contingent on experimental control clarification.", "valence": "conditional", "suggested_improvement": "Clarify whether JAPAN, CONTRA, and PCP share the same flow architecture and training procedure to ensure controlled comparison.", "support_status": "mixed", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.837049663066864, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8049878478050232}, {"unit_index": 2, "inspected_object": "Enumeration/identification of points inside density-thresholded regions", "observation": "The paper discusses Monte Carlo estimation of the area of the prediction set but does not address how to enumerate or identify the specific points inside the region.", "reasoning": "The reviewer identifies this as a potential scalability bottleneck and computing challenge related to contour finding, questioning whether the smooth property of the density estimate can be exploited algorithmically rather than relying on brute-force grid evaluation.", "judgment": "Concern about practical deployment feasibility and computational scalability.", "valence": "negative", "suggested_improvement": "Explain how the step of identifying points inside the region is achieved and visualized in the current examples.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8251254558563232, "reasoning_key": "design_justification", "reasoning_sim": 0.8001061677932739}, {"unit_index": 3, "inspected_object": "Empirical conditional coverage", "observation": "JAPAN estimates conditional densities p(y|x) at its core but does not empirically test conditional coverage, despite lacking theoretical guarantees for it.", "reasoning": "The reviewer reasons that the informative conformity score (full density estimate) might confer a secondary benefit of better conditional calibration, making it a hypothesis worth testing even if not theoretically guaranteed.", "judgment": "Constructive suggestion that the method may possess beneficial properties beyond its stated claims.", "valence": "positive", "suggested_improvement": "Test whether JAPAN achieves reasonable conditional coverage empirically.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.8147437572479248, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8092597126960754}, {"unit_index": 4, "inspected_object": "Exchangeability assumptions in time-series applications", "observation": "The review acknowledges the use of time-series data but notes a lack of discussion on how exchangeability-backed conformal theory applies in temporal settings.", "reasoning": "The reviewer questions how a practitioner can assess whether produced intervals are trustworthy given the gap between theory assumptions and real-world deployment, seeking usable heuristics for validity checking.", "judgment": "Uncertainty regarding theoretical applicability and practical guidance for temporal contexts.", "valence": "uncertain", "suggested_improvement": "Provide tips or guidance on checking the validity of marginal coverage in theory and in practice for time-series applications.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8180932402610779, "reasoning_key": "construct_validity", "reasoning_sim": 0.7680549025535583}, {"unit_index": 5, "inspected_object": "Symbolic consistency and mathematical notation", "observation": "Three specific notation issues are identified: abuse of 'z' in Proposition 3 (denoting both sample size and instance), misleading description of ICP (describing split conformal instead), and undefined hat{p} appearing before formal introduction.", "reasoning": "The reviewer treats these as representative instances of a broader pattern of definitional hygiene problems, holding a norm that precision in symbolic language is integral to scientific communication quality, contributing significantly to the low presentation score.", "judgment": "Negative evaluation of mathematical presentation rigor, viewing these as fixable flaws requiring proofreading.", "valence": "negative", "suggested_improvement": "Perform a thorough proofreading pass to correct notation abuses, clarify definitions like ICP, and ensure all symbols are formally introduced before use.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8150074481964111, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7696351408958435}, {"unit_index": 6, "inspected_object": "Presence of theoretical grounding", "observation": "The paper provides theory of coverage guarantees, rank preservation, and a volume estimator.", "reasoning": "The reviewer credits the paper for providing this theoretical backing, valuing it as a component of soundness even without deep interrogation of its correctness.", "judgment": "Positive assessment of soundness based on the existence of theoretical components.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8423816561698914, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8106464147567749}]}, {"review_id": "05HEprsjp7", "paper_id": "TzHNCCyBJa", "paper_title": "A Generative Diffusion Framework for Single Image Reflection Separation", "decision": null, "summary": "The reviewer systematically challenges the paper's attribution of performance gains to its novel components, arguing that the underlying foundation model may account for most results. Key critiques target weak novelty claims, missing ablations, unfair baselines, and internal contradictions in hyperparameter analysis.", "units": [{"unit_index": 0, "inspected_object": "Positioning claim of being the first diffusion model explicitly fine-tuned for reflection removal", "observation": "Existing diffusion-based methods (e.g., Rosh et al., 2023) already exist, making the 'first' claim inaccurate and the differentiation minor.", "reasoning": "The presence of prior work invalidates the novelty claim; the paper fails to articulate a positive, task-specific rationale for using diffusion beyond generic limitations of prior methods.", "judgment": "The positioning claim is substantively weak due to lack of distinct motivation.", "valence": "negative", "suggested_improvement": "Articulate specific properties of diffusion models that are particularly effective for this specific task.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8185643553733826, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8501558303833008}, {"unit_index": 1, "inspected_object": "Cross-layer self-attention mechanism", "observation": "The mechanism is a straightforward concatenation of keys/values, showing marginal or negative effects on performance in ablations.", "reasoning": "The design lacks clear explanation for why it aids disentanglement, and empirical evidence does not support its necessity as a novel contribution.", "judgment": "The contribution of this component is unconvincing due to ambiguous empirical support.", "valence": "negative", "suggested_improvement": "Provide stronger justification for the design choice and clarify its causal role in disentanglement.", "support_status": "mixed", "confidence": "high", "object_key": "method_design", "object_sim": 0.8093076944351196, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7383406162261963}, {"unit_index": 2, "inspected_object": "Fidelity-Guided Feature Modulation (FGFM)", "observation": "FGFM addresses color drifts but is not explained/demonstrated in the main text, lacks comparison to other modulation techniques, and is absent from primary ablation tables.", "reasoning": "A claimed contribution must be motivated and empirically isolated; its omission from ablations suggests it is not treated as a critical component by the authors.", "judgment": "The module's novelty and necessity are unsupported by the presented argument structure.", "valence": "negative", "suggested_improvement": "Include FGFM in ablation studies and motivate its novelty against existing feature modulation techniques.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8123095035552979, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7763034701347351}, {"unit_index": 3, "inspected_object": "Latent composition and optimization", "observation": "Novelty is limited to application within latent space rather than mechanistic innovation, with claims of improved generalization and reduced overhead lacking strong supporting evidence.", "reasoning": "Application novelty is weaker than mechanistic novelty and requires rigorous demonstration to validate claims of efficiency and generalization benefits.", "judgment": "The contribution is incremental and under-supported by empirical data.", "valence": "negative", "suggested_improvement": "Provide stronger supporting evidence for claims of improved generalization and reduced computational overhead.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8556143045425415, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8022088408470154}, {"unit_index": 4, "inspected_object": "Experimental protocol: baseline selection and dataset choice", "observation": "Comparisons rely primarily on older, non-generative methods, omit recent large-scale benchmarks (Zhu et al., 2024), and include adversarial robustness-focused baselines (RobustSIRR).", "reasoning": "Fair comparison requires evaluating generative methods against other generative methods, even from adjacent tasks, to isolate the contribution of the generative prior.", "judgment": "The experimental setup is unfair and fails to properly benchmark the method's strengths.", "valence": "negative", "suggested_improvement": "Include a wider range of recent diffusion-based techniques from related image enhancement tasks adapted for this problem.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.890261173248291, "reasoning_key": "design_justification", "reasoning_sim": 0.8011548519134521}, {"unit_index": 5, "inspected_object": "'w/o C, O, D' ablation baseline", "observation": "The baseline (fine-tuned Stable Diffusion v2 without proposed modules) outperforms five competing methods, raising questions about whether gains are due to novel contributions or base model power.", "reasoning": "Causal attribution requires demonstrating that components add value beyond the foundation model; the current evidence leaves this confound unresolved.", "judgment": "The paper fails to demonstrate what makes the method work, attributing success potentially to the base model.", "valence": "negative", "suggested_improvement": "Conduct experiments with a more direct baseline of simply fine-tuning Stable Diffusion v3.5 to disentangle baseline power from architectural contribution.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8215665221214294, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7828236222267151}, {"unit_index": 6, "inspected_object": "Ablation completeness and hyperparameter w", "observation": "FGFM is missing from Table 3, computational efficiency relies on single internal comparison, and Figure 10 shows residual reflections when w=0, contradicting claims that w controls them.", "reasoning": "Internal consistency checks reveal contradictions between visual evidence and textual claims, undermining confidence in the reported control mechanisms.", "judgment": "The ablation analysis is incomplete and contains contradictory evidence regarding hyperparameter efficacy.", "valence": "negative", "suggested_improvement": "Address the contradiction in Figure 10 and provide more comprehensive ablation for computational claims.", "support_status": "mixed", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8438623547554016, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8037446141242981}, {"unit_index": 7, "inspected_object": "Limitations section", "observation": "Limitations are brief, attribute failures only to hyperparameters, and lack discussion of intrinsic limitations or out-of-distribution performance.", "reasoning": "Understanding failure modes strengthens the case for understanding success modes; superficial limitations suggest incomplete analysis.", "judgment": "The limitations discussion is insufficient to demonstrate deep understanding of the method.", "valence": "negative", "suggested_improvement": "Include specific failure cases and discuss intrinsic limitations related to out-of-distribution performance.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7175484299659729, "reasoning_key": "robustness_norm", "reasoning_sim": 0.791877269744873}]}, {"review_id": "06LDy0VmBh", "paper_id": "3qzgB7QIDl", "paper_title": "Reducing Contextual Stochastic Bilevel Optimization via Structured Function Approximation", "decision": "Accept (Poster)", "summary": "The reviewer conducts a critical evaluation focusing on internal consistency between motivation and assumptions, the practical validity of theoretical conditions, and the comprehensiveness of experimental validation. The assessment highlights significant gaps in comparative baselines, problem formulation justification, and robustness analysis, concluding that the novel idea is undermined by restrictive conditions and weak validation.", "units": [{"unit_index": 0, "inspected_object": "Theorem 4.5 conditions (c.1, c.2, c.3) regarding the discreteness of ξ", "observation": "Condition c.1 requires ξ to be discrete with finite cardinality, while the paper's motivation claims improvement over Guo et al. 2021 which fails when ξ is continuous.", "reasoning": "The reviewer uses an internal contradiction within the paper: if the claimed advantage is handling continuous ξ, but the new method re-introduces discreteness, the distinction from prior work collapses and the improvement is illusory.", "judgment": "The theoretical conditions undermine the paper's stated motivation and fail to clearly distinguish the contribution from Guo et al. 2021.", "valence": "negative", "suggested_improvement": "Include a table to have a comprehensive comparison of this paper versus other baselines to make comparative claims explicit and verifiable.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8144984841346741, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8205366730690002}, {"unit_index": 1, "inspected_object": "Practical machine learning examples satisfying Theorem 4.5 conditions", "observation": "The reviewer finds no evidence that there are practical ML examples satisfying conditions c.1, c.2, and c.3, nor any discussion on robustness if these conditions do not hold.", "reasoning": "The reviewer views the paper's value as contingent on the conditions being satisfied in practice; if they fail, complexity guarantees may evaporate, suggesting the theoretical framework might be vacuous without concrete instantiation.", "judgment": "The lack of practical instantiation or robustness analysis creates uncertainty about the method's real-world utility and potential complexity degradation.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8475571274757385, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8289166688919067}, {"unit_index": 2, "inspected_object": "Experimental validation scope and baselines", "observation": "Experiments use MNIST (described as toy problems at small scale) and consider only stocBiO as a baseline, missing many other relevant bilevel optimization approaches.", "reasoning": "A paper proposing a new method should benchmark against the full landscape of relevant approaches; limiting baselines to one and using small-scale datasets fails to demonstrate the method's competitive standing or generalizability.", "judgment": "The empirical validation is weak due to insufficient scale and incomplete baseline comparison.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8846303224563599, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8520241379737854}, {"unit_index": 3, "inspected_object": "CSBO formulation for hyperparameter optimization (Section 5.2, equation 5)", "observation": "The reviewer notes that literature often uses SBO instead of CSBO for hyperparameter optimization tasks.", "reasoning": "The reviewer questions whether the CSBO formulation is the appropriate framework for the application, suspecting it may be contrived to fit the method rather than derived from the problem structure, and asks why CSBO is better than SBO.", "judgment": "The choice of problem formulation lacks justification and appears potentially misaligned with standard practices in the field.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.878130316734314, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7364538311958313}, {"unit_index": 4, "inspected_object": "Use of fixed basis functions vs. learned basis functions", "observation": "The reviewer notes that using fixed basis functions is not necessarily good and that learned basis functions may yield stronger approximation guarantees.", "reasoning": "While speculative, the reviewer evaluates the paper on its potential to inspire future work; leaving the learned-basis direction unexamined suggests the paper closes off a potentially valuable avenue without discussion.", "judgment": "The reliance on fixed basis functions is an unexamined assumption that may limit the approach's strength compared to potential learned alternatives.", "valence": "conditional", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "low", "object_key": "theory", "object_sim": 0.8311615586280823, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7847871780395508}]}, {"review_id": "07MBVZPbvX", "paper_id": "oFsNco4aMm", "paper_title": "UP2You: Fast Reconstruction of Yourself from Unconstrained Photo Collections", "decision": "Accept (Poster)", "summary": "The reviewer conducts mechanistic due-diligence by probing four specific areas: output consistency mechanisms, input conditioning for noisy cameras, isolation of the rectifier's contribution to geometry, and hyperparameter sensitivity. The logic accepts empirical results but withholds full endorsement until the causal story and robustness are explicitly explained and tested.", "units": [{"unit_index": 0, "inspected_object": "Data rectifier's output quality and multi-view consistency mechanisms", "observation": "The reviewer questions how the rectifier ensures data quality, maintains multi-view texture consistency, and preserves fine details (e.g., clothing patterns, facial features) without distortion or misalignment.", "reasoning": "The reviewer treats the conversion from unconstrained photos to orthogonal views as a potentially lossy operation that could introduce errors. The judgment rests on the standard that high-fidelity reconstruction claims require an explicit causal account of the internal consistency mechanisms, not just empirical results.", "judgment": "The paper's explanation of the rectifier's ability to preserve detail is insufficient; mechanistic depth is lacking.", "valence": "negative", "suggested_improvement": "Describe the specific mechanisms within the rectifier that maintain multi-view texture consistency and preserve fine details.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.812845766544342, "reasoning_key": "design_justification", "reasoning_sim": 0.8509465456008911}, {"unit_index": 1, "inspected_object": "Input conditioning for unknown or noisy camera parameters", "observation": "The reviewer asks whether the rectifier uses calibration or pose/shape priors (e.g., SMPL-X normal maps) and whether alignment is iterative or direct.", "reasoning": "Reflecting disciplinary expectations that in-the-wild photos lack reliable camera information, the reviewer judges that any method claiming to produce orthogonal views must explain how it handles this known hard problem. The absence of this explanation creates uncertainty about the method's robustness to input noise.", "judgment": "The handling of input camera parameters is unclear and requires specification to validate the method's approach within the design space of camera-handling techniques.", "valence": "conditional", "suggested_improvement": "Specify how camera alignment is handled, including whether calibration or priors are used and if the process is iterative or direct.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7927832007408142, "reasoning_key": "design_justification", "reasoning_sim": 0.8056276440620422}, {"unit_index": 2, "inspected_object": "Downstream effect of rectification on geometric accuracy", "observation": "The reviewer questions how rectification specifically improves the geometric accuracy of subsequent 3D reconstruction, such as reducing mesh carving or texture baking errors.", "reasoning": "The reviewer notes that improved geometry could stem from multiple sources (better input views, feature aggregation, shape prediction). The evaluative standard requires isolating the rectifier's specific contribution to demonstrate a causal link between rectification and reconstruction quality.", "judgment": "The specific contribution of the rectifier to downstream geometric accuracy is not isolated or demonstrated.", "valence": "negative", "suggested_improvement": "Isolate the rectifier's contribution to geometry via targeted experiments demonstrating reductions in mesh carving or texture baking errors.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.804582417011261, "reasoning_key": "design_justification", "reasoning_sim": 0.8249916434288025}, {"unit_index": 3, "inspected_object": "Hyperparameter sensitivity of PCFA module", "observation": "The reviewer identifies a lack of ablation on PCFA hyperparameters, specifically the feature retention ratio (γ) and transformer block number, noting expected trade-offs like detail loss vs. noise.", "reasoning": "The reviewer expects that a paper introducing a new module would systematically vary its internal parameters. The absence of such ablations is treated as a gap in understanding the method's sensitivity and robustness, despite other ablations being present.", "judgment": "The sensitivity of the PCFA module to its internal hyperparameters is unexamined, leaving the robustness of the method unclear.", "valence": "negative", "suggested_improvement": "Add ablations on γ (feature retention ratio) and transformer block number to map the method's sensitivity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8548305630683899, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8314968347549438}]}, {"review_id": "07k0W94H0V", "paper_id": "sS6yP4WQw0", "paper_title": "MatRL: Provably Generalizable Iterative Algorithm Discovery via Monte-Carlo Tree Search", "decision": "Reject", "summary": "The reviewer conducts a practitioner's audit focusing on practical barriers: lack of OOD generalization evidence, high hyperparameter tuning costs, inadequate baselines, unnecessary framework complexity, and narrow scope. While acknowledging clarity and novelty, the reviewer judges the contribution low due to these usability and relevance gaps.", "units": [{"unit_index": 0, "inspected_object": "Out-of-distribution generalization evaluation (Figure 1, Table 2)", "observation": "Experiments are evaluated on the same distribution used to discover the algorithm.", "reasoning": "Outperformance on the discovery distribution is not surprising and does not demonstrate practical utility for a user with a different problem; baseline methods might perform comparably given similar search budgets or iteration counts.", "judgment": "The evidence for out-of-distribution robustness is insufficient to establish practical value.", "valence": "negative", "suggested_improvement": "Demonstrate performance on diverse distribution shifts with different spectra, matrix structures, or condition numbers.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8783772587776184, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8149766325950623}, {"unit_index": 1, "inspected_object": "Hyperparameter space ($C,α,ε_{tol}, E, T, D$, action choice) and tuning effort", "observation": "The method requires choosing multiple hyperparameters and it is unclear if they were tuned per instance.", "reasoning": "If hyperparameters are tuned per instance, performance reflects tuning effort rather than inherent algorithmic superiority; this creates high 'activation energy' for adoption compared to standard methods run for more iterations.", "judgment": "The epistemic status of the method's success is questionable due to potential overfitting to hyperparameters, making it less attractive than simpler alternatives.", "valence": "negative", "suggested_improvement": "Provide ablation studies or clarity on whether tuning was per-instance to assess the true cost of using the method.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8700318932533264, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7689638137817383}, {"unit_index": 2, "inspected_object": "Comparison baselines in experimental setup", "observation": "The paper compares against generic baselines but omits state-of-the-art methods like newtonschulz5/zeropower_via_newtonschulz5 relevant to the Muon/nanoGPT community.", "reasoning": "Without comparison to current frontier methods in matrix sign computation, claims of outperformance are not meaningful for practitioners in that domain.", "judgment": "The experimental evaluation fails to engage with the relevant competitive landscape, undermining the significance of the results.", "valence": "negative", "suggested_improvement": "Compare the proposed method against competing new methods being developed in the specific application context (e.g., square symmetric matrices).", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.919312596321106, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7913984656333923}, {"unit_index": 3, "inspected_object": "Action class composition and MCTS framework necessity", "observation": "The discovered algorithms appear to be compositions of odd monomials (via Newton/Schulz steps), suggesting the action space could be restricted to odd monomials.", "reasoning": "If the effective search space is limited to odd monomials, a simpler exhaustive search over polynomials up to degree 9 could replace the complex MCTS machinery, questioning the added value of the proposed framework.", "judgment": "The design space is unnecessarily complex; the MCTS approach may be superfluous if a simpler polynomial search suffices.", "valence": "negative", "suggested_improvement": "Investigate restricting actions to the class of odd monomials (up to ninth degree) to simplify the framework and reduce overhead.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7852964401245117, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7471747994422913}, {"unit_index": 4, "inspected_object": "Generalization to rectangular matrices", "observation": "The current framework focuses on square symmetric matrices, limiting applicability to problems involving rectangular data (e.g., Muon).", "reasoning": "Practical utility depends on handling rectangular matrices via SVD-based transformations ($f(UXV^T) = Uf(X)V^T$); the current scope excludes these important use cases.", "judgment": "The method's scope is too narrow for broader practical adoption without extension to rectangular matrices.", "valence": "conditional", "suggested_improvement": "Extend the framework to accommodate rectangular matrices using orthonormal matrices and singular values as state variables.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8384213447570801, "reasoning_key": "design_justification", "reasoning_sim": 0.8028990626335144}, {"unit_index": 5, "inspected_object": "Proposition 1 theoretical framing", "observation": "Proposition 1 is presented as a result specific to Algorithm 1 but appears to be a general statement about approximating matrix functions in high probability/dimensions.", "reasoning": "Overselling the theoretical contribution by framing a general asymptotic equivalence result as algorithm-specific misrepresents the novelty and depth of the theoretical insight.", "judgment": "The theoretical contribution is modest and potentially overstated in its specificity.", "valence": "negative", "suggested_improvement": "Clarify the scope of Proposition 1 to reflect its general nature rather than implying algorithm-specific novelty.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8635464310646057, "reasoning_key": "design_justification", "reasoning_sim": 0.8314639329910278}]}, {"review_id": "08OJFUDS9v", "paper_id": "ZQcDUhEOg9", "paper_title": "AdaGC: Improving Training Stability for Large Language Model Pretraining", "decision": "Reject", "summary": "The reviewer evaluates the paper through a lens of deployment risk, accepting the core mechanism and implementation simplicity while critically probing boundary conditions including temporal lag, hyperparameter sensitivity, and standalone robustness via hypothetical failure modes.", "units": [{"unit_index": 0, "inspected_object": "The method's mechanism for isolating abnormal gradients polluting optimizer states as the cause of loss spikes.", "observation": "The reviewer accepts the paper's framing that this mechanism unifies multi-causal phenomena (data outliers, hardware faults, precision issues) into a single intervention point.", "reasoning": "The reviewer applies an implicit norm of scientific parsimony, valuing methods that identify minimal sufficient causes for complex problems.", "judgment": "Positive evaluation: The abstraction move is praised as a strength for converting a messy phenomenon into a tractable solution.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7326734066009521, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7764787673950195}, {"unit_index": 1, "inspected_object": "The implementation cost and empirical payoff of the EMA-based clipping threshold.", "observation": "The method requires approximately 4 bytes per tensor and less than 10 lines of code change while suppressing spikes on models from 1.3B to 10B.", "reasoning": "The reviewer weighs practical engineering criteria, specifically low-friction adoption and high cost-benefit ratio, over theoretical complexity.", "judgment": "Positive evaluation: Simplicity and efficiency are valued as significant strengths.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8340142369270325, "reasoning_key": "cost_benefit", "reasoning_sim": 0.789768636226654}, {"unit_index": 2, "inspected_object": "The temporal behavior of the EMA-based clipping threshold during sudden gradient spikes.", "observation": "The EMA is slow to rise, potentially allowing outlier gradients to enter the optimizer state before the threshold adjusts sufficiently.", "reasoning": "The reviewer constructs a mechanistic counterfactual based on the design choice of using historical norms, applying a precautionary principle that flags theoretical vulnerabilities even if not empirically observed in the paper.", "judgment": "Negative/Hypothetical weakness: A potential window of vulnerability exists where the method may fail to suppress spikes immediately.", "valence": "negative", "suggested_improvement": "Clarify or mitigate the lag inherent in the EMA calculation during sudden spikes.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7805091738700867, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7783038020133972}, {"unit_index": 3, "inspected_object": "The sensitivity of the relative clipping threshold hyperparameter (λ_rel) to downstream performance.", "observation": "Experimental results show that λ_rel has a noticeable impact on performance, contradicting the paper's claim of being a fully adaptive method.", "reasoning": "The reviewer applies a norm of consistency between claims and evidence, arguing that a method labeled 'adaptive' should not require task-specific tuning of key parameters.", "judgment": "Negative judgment: There is a rhetorical contradiction between the 'adaptive' positioning and the empirical need for hyperparameter tuning.", "valence": "negative", "suggested_improvement": "Demonstrate robustness across tasks or justify the necessity of tuning λ_rel.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8856852054595947, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7762159109115601}, {"unit_index": 4, "inspected_object": "The dependency on GlobalGC during the early training phase (warmup).", "observation": "AdaGC relies on GlobalGC initially, which the reviewer interprets as evidence that AdaGC alone may not be sufficiently reliable at the beginning of training.", "reasoning": "The reviewer infers instability from the design choice rather than observed failures, applying a norm of standalone validity that expects a novel method to function without relying on the baseline it aims to improve.", "judgment": "Negative judgment: The method lacks standalone robustness in the early training phase due to its hybrid dependency.", "valence": "negative", "suggested_improvement": "Clarify why AdaGC appears unstable during early stages or ablate the GlobalGC warmup to prove independent reliability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.799041211605072, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8139033317565918}, {"unit_index": 5, "inspected_object": "The scalability of the method beyond the tested bounds (models >10B, steps >36K).", "observation": "The review notes the explicit boundaries of the empirical scope but questions if stability holds at the frontier of scale.", "reasoning": "The reviewer assumes that loss spikes become more frequent or severe at larger scales, implying that validation at 10B does not automatically transfer to larger models.", "judgment": "Uncertain/Conditional: The method's efficacy at extreme scales is unverified and poses a risk.", "valence": "uncertain", "suggested_improvement": "Test or discuss stability on models larger than 10B or training runs exceeding 36K steps.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8082403540611267, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7228749394416809}]}, {"review_id": "094d386J1Y", "paper_id": "5AuN6RD072", "paper_title": "Sparse Autoencoders Reveal Interpretable Features in Single-Cell Foundation Models", "decision": "Reject", "summary": "The reviewer evaluates the paper primarily on epistemic hygiene, criticizing unsubstantiated feature claims, unjustified architectural choices, imprecise terminology, and a lack of comparative baselines, concluding that the paper fails scholarly standards of justification and transparency.", "units": [{"unit_index": 0, "inspected_object": "Categorical feature counts (26 cell type features, 2 batch features, 29 biological processes, 7 gene sets, 36 uncategorized concepts)", "observation": "The paper presents specific counts of learned features categorized into distinct types but does not explain the mechanism by which these categories were assigned or validated.", "reasoning": "A claim about learned representations requires a transparent accounting of the labeling process to be interpretable; without provenance, the quantitative claims are unverified and opaque.", "judgment": "The feature counts are unsubstantiated and lack interpretability due to missing methodological explanation.", "valence": "negative", "suggested_improvement": "Provide a transparent description of the labeling or validation procedure used to assign features to their respective categories.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7635881900787354, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8096286058425903}, {"unit_index": 1, "inspected_object": "Selection of layer 10 out of 12 for SAE training", "observation": "The paper justifies using layer 10 based on an expectation that deeper layers are 'less abstract and more interpretable,' but provides no study testing this expectation.", "reasoning": "Architectural design choices require either principled justification derived from prior work/ theory or empirical demonstration that the choice is optimal; mere expectation is insufficient.", "judgment": "The design choice is unmotivated and unstudied.", "valence": "negative", "suggested_improvement": "Provide a principled justification for the layer selection or present an empirical comparison across layers to demonstrate the choice's validity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8396171927452087, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7700431942939758}, {"unit_index": 2, "inspected_object": "Thresholding procedure for classifying cells as positive/negative", "observation": "The reviewer characterizes the application of thresholds to classify cells as 'engineered and not meaningful.'", "reasoning": "Methodological steps should be grounded in principled reasoning rather than arbitrary engineering; the absence of such grounding renders the classification step questionable.", "judgment": "The thresholding method is problematic and lacks meaningful justification.", "valence": "negative", "suggested_improvement": "Clarify the basis for the thresholding procedure or propose a more principled alternative for classification.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.7905413508415222, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7833715677261353}, {"unit_index": 3, "inspected_object": "Terminology: Conflation of 'drug response prediction' and 'perturbation response prediction'", "observation": "The paper treats 'drug response prediction' and 'perturbation response prediction' as interchangeable.", "reasoning": "Precision in terminology is required in this field; conflating distinct biological intervention models undermines the clarity and accuracy of the paper's scope.", "judgment": "The terminology is imprecise and misleading.", "valence": "negative", "suggested_improvement": "Distinguish clearly between drug response and perturbation response prediction in the text.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7709441184997559, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7908084392547607}, {"unit_index": 4, "inspected_object": "Framing statement: 'many single-cell foundation models remain difficult to interpret'", "observation": "The paper implies a relationship between the growing utility of scFMs and their difficulty in interpretation.", "reasoning": "There is no logical necessity linking utility to interpretability; implying such a relationship introduces an unwarranted assumption that confuses the reader regarding the model's properties.", "judgment": "The framing contains an unwarranted causal or correlational implication.", "valence": "negative", "suggested_improvement": "Remove the implied link between utility and interpretability or clarify the independent nature of these properties.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7792272567749023, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7172569632530212}, {"unit_index": 5, "inspected_object": "Claim that scFMs 'tend to be outperformed by linear models'", "observation": "The paper asserts that scFMs are generally outperformed by linear models.", "reasoning": "This claim oversimplifies an active debate where evidence exists both for and against linear model superiority; balanced representation of the literature is required to avoid strong simplification.", "judgment": "The claim is a strongly simplified and unbalanced representation of the current state of knowledge.", "valence": "negative", "suggested_improvement": "Acknowledge countervailing evidence where scFMs outperform linear models to provide a balanced view of the literature.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7289621829986572, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7715108394622803}, {"unit_index": 6, "inspected_object": "Comparative baseline analysis", "observation": "The paper proposes a feature selection/disentanglement method without comparing it to the breadth of existing methods addressing this challenge in single-cell data.", "reasoning": "Novelty and contribution are established through contrast with existing approaches; failing to position the work relative to relevant baselines leaves the method's relative merit undefined.", "judgment": "The paper lacks sufficient comparative context to establish its contribution.", "valence": "negative", "suggested_improvement": "Include comparisons to existing feature selection and disentanglement methods for single-cell data.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8932820558547974, "reasoning_key": "novelty_standard", "reasoning_sim": 0.859258770942688}]}, {"review_id": "09IxVeTMJr", "paper_id": "aiGqJwOE2x", "paper_title": "InfoScan: Information-Efficient Visual Scanning via Resource-Adaptive Walks", "decision": "Accept (Poster)", "summary": "The reviewer provides qualified endorsement, praising the novelty of the dynamic scanning policy, the formalization of the joint optimization, and the extensive empirical gains over strong baselines. However, they identify four critical gaps: missing latency analysis for efficiency claims, absent hyperparameter disclosures, insufficient ablation/explanation for the information scoring module, and incomplete comparison against Mamba-based baselines.", "units": [{"unit_index": 0, "inspected_object": "The paper's claim of parameter efficiency without corresponding computational analysis.", "observation": "The reviewer notes the paper claims parameter efficiency but fails to analyze latency overhead from the Information Scoring Module and iterative Path Planning Module during inference.", "reasoning": "Efficiency claims require end-to-end wall-clock measurements, not just parameter counts; the reviewer distinguishes between parameter efficiency and computational efficiency, noting that RL-based path planning may not be parallelizable like fixed scanning patterns.", "judgment": "The efficiency argument is incomplete because the actual cost of the machinery producing it is unmeasured.", "valence": "negative", "suggested_improvement": "Analyze the latency overhead introduced by the Information Scoring Module and the iterative, sequential decision-making of the Path Planning Module during inference.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8176063299179077, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7671566009521484}, {"unit_index": 1, "inspected_object": "Disclosure of model architecture hyperparameters.", "observation": "The hyperparameters of the model architecture are not presented in the paper.", "reasoning": "Novelty must be accompanied by transparency; a novel architecture that cannot be reproduced is not a full contribution, and vision backbone papers should disclose configuration details.", "judgment": "The work lacks necessary reproducibility information regarding its configuration.", "valence": "negative", "suggested_improvement": "Present the hyperparameters of the model architecture.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8587436676025391, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7916765809059143}, {"unit_index": 2, "inspected_object": "The Information Scoring Module (combining entropy and local variance).", "observation": "The paper asserts the scoring method works but does not explain why or demonstrate whether alternative scoring methods would perform comparably.", "reasoning": "Design choices need justification through ablation; any component making a performance claim should be tested against alternatives to determine if the specific design is necessary or merely sufficient.", "judgment": "The contribution of the specific information scoring mechanism is unsubstantiated due to lack of ablation or solid explanation.", "valence": "negative", "suggested_improvement": "Explain the information importance evaluation method more solidly or provide more ablation studies.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8550431728363037, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7472013831138611}, {"unit_index": 3, "inspected_object": "Comparative evaluation against Mamba-based vision backbones.", "observation": "The paper compares against some baselines but excludes previous Mamba-based vision backbones.", "reasoning": "A paper claiming state-of-the-art performance within a lineage must benchmark against all relevant prior work in that immediate family to establish true competitive standing.", "judgment": "The positioning of the method relative to its direct competitors is incomplete.", "valence": "negative", "suggested_improvement": "Include more previous Mamba-based vision backbones in the comparisons.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7871001362800598, "reasoning_key": "novelty_standard", "reasoning_sim": 0.6798813939094543}, {"unit_index": 4, "inspected_object": "The shift from fixed, content-agnostic scanning to dynamic, information-driven policy.", "observation": "The reviewer finds the motivation and method novel, specifically the move to a dynamic, information-driven policy.", "reasoning": "The shift represents a meaningful change in approach compared to static patterns, offering a new direction for efficient scanning.", "judgment": "The core methodological shift is novel and compelling.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7764309048652649, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7137429714202881}, {"unit_index": 5, "inspected_object": "The joint optimization objective and reinforcement learning setup.", "observation": "The methodology is formally defined with a joint optimization objective and an RL setup.", "reasoning": "The formalization suggests the contribution may impact other researchers beyond the specific application, indicating transferable value.", "judgment": "The formalization is strong and has potential broader impact.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8393177390098572, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7445767521858215}, {"unit_index": 6, "inspected_object": "Extensive evaluations across multiple core vision tasks and domains.", "observation": "Evaluations consistently show performance improvements over strong baselines including CNNs, ViTs, and SSMs.", "reasoning": "Broad evaluation across diverse tasks and comparison against strong, varied baselines demonstrates robustness and effectiveness.", "judgment": "The empirical results are extensive and demonstrate consistent improvement.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8390222787857056, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8125588297843933}]}, {"review_id": "09kvR1K39V", "paper_id": "3SX1m4nFym", "paper_title": "WaveGS: Physics-Inspired Wavelet Splatting for Thermal Novel View Synthesis", "decision": null, "summary": "The reviewer endorses the physical motivation and qualitative/quantitative results but flags gaps in scope discussion, missing runtime analysis, inadequate frequency-aware baselines, and hyperparameter sensitivity. The review reflects moderate enthusiasm tempered by requests for more rigorous self-assessment and positioning.", "units": [{"unit_index": 0, "inspected_object": "Physical motivation of the method (heat conduction low-pass characteristic guiding wavelet decomposition)", "observation": "The reviewer explicitly endorses the core premise, describing it as 'domain-specific regularization' and an 'elegant' stabilizer for an ill-posed problem.", "reasoning": "The reviewer operates under the assumption that thermal novel-view synthesis is fundamentally underconstrained, making any physically motivated inductive bias a valuable contribution to stability.", "judgment": "Positive evaluation of the conceptual framing and physical grounding as a strength.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7648161053657532, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7555243968963623}, {"unit_index": 1, "inspected_object": "Qualitative visual results (thermal boundaries and ghosting)", "observation": "The reviewer observes noticeably sharper contours and attributes this improvement directly to the LF/HF frequency separation.", "reasoning": "The reviewer performs a causal inference from qualitative output: because the method separates frequencies and the outputs look sharper, the separation is assumed responsible for the improvement, without requesting ablation studies to isolate this effect.", "judgment": "Positive evaluation of the method's effectiveness in addressing specific failure modes like blurry boundaries.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8435376286506653, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7669962048530579}, {"unit_index": 2, "inspected_object": "Quantitative improvements over Thermal-3DGS baseline", "observation": "The reviewer cites 'significant quantitative improvements' but does not report specific numbers or question the experimental protocol.", "reasoning": "The reviewer accepts the paper's reported results at face value, consistent with a low confidence score indicating limited deep verification of experimental claims.", "judgment": "Positive evaluation of the empirical performance relative to the primary baseline.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8206960558891296, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7672593593597412}, {"unit_index": 3, "inspected_object": "Scope limitations regarding the low-pass assumption", "observation": "The reviewer posits that textured heated surfaces might violate the low-frequency assumption, noting that thermal imagery is dominated by low frequencies.", "reasoning": "A method's validity is bounded by its assumptions; therefore, the paper should explicitly delineate these bounds rather than assuming universal applicability of the physical prior.", "judgment": "Negative evaluation of the paper's self-assessment regarding scope, requiring more explicit discussion of limitations.", "valence": "negative", "suggested_improvement": "Add more explicit discussion of limitations, specifically regarding scenarios like textured heated surfaces where the low-pass assumption may fail.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8543059229850769, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7609946727752686}, {"unit_index": 4, "inspected_object": "Absence of runtime and computational cost analysis", "observation": "The reviewer notes that wavelet decomposition and inverse transforms 'might introduce overhead' but finds no data on computational cost.", "reasoning": "The absence of runtime analysis is treated as a gap in the paper's completeness and transparency regarding engineering costs, rather than evidence of a flaw.", "judgment": "Negative evaluation of the paper's completeness regarding practical implementation details.", "valence": "negative", "suggested_improvement": "Provide computational cost data or runtime analysis to confirm or refute potential overhead introduced by the method.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8949058651924133, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7471631765365601}, {"unit_index": 5, "inspected_object": "Comparison set including WaveNeRF and FreGS", "observation": "The reviewer notes that WaveNeRF and FreGS are only briefly mentioned despite being other frequency-domain methods.", "reasoning": "Because the paper claims frequency-domain advantages, the comparison set should include other frequency-representation approaches to properly position the method within its taxonomic family.", "judgment": "Negative evaluation of the comparative baselines, viewing the exclusion of frequency-aware methods as a mispositioning of the contribution.", "valence": "negative", "suggested_improvement": "Include direct quantitative comparisons against WaveNeRF and FreGS.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7730057835578918, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7299595475196838}, {"unit_index": 6, "inspected_object": "Hyperparameter sensitivity (sparsity/deformation weighting, decomposition levels, mask threshold)", "observation": "The reviewer identifies specific tunable knobs and questions the method's robustness to these choices.", "reasoning": "The reviewer implies a concern that the method might be brittle to hyperparameter choices, necessitating an analysis of sensitivity to ensure practical usability.", "judgment": "Uncertainty/Negative evaluation regarding the method's stability and ease of use without further sensitivity analysis.", "valence": "conditional", "suggested_improvement": "Analyze hyperparameter sensitivity, specifically regarding the weighting of sparsity and deformation losses, the number of decomposition levels, and the mask threshold.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8524960875511169, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8170549273490906}]}, {"review_id": "09u5pcpeDH", "paper_id": "E9qZRF5gCj", "paper_title": "What really matters in matrix whitening optimizers?", "decision": "Reject", "summary": "The reviewer critiques the paper primarily on grounds of limited generalizability due to a single experimental setup, lack of clarity in key metrics, and incomplete methodological comparisons. While acknowledging the empirical interest, the reviewer finds the evidence insufficient for top-tier acceptance without broader validation or clearer exposition.", "units": [{"unit_index": 0, "inspected_object": "Generalization of optimizer performance findings to other configurations", "observation": "The experimental evaluation is limited to a single setup (OpenWebText, GPT-2, cosine annealing scheduler, specific hyperparameters).", "reasoning": "A single configuration provides insufficient evidence for general claims about optimizer families; alternative schedulers or setups might yield different conclusions.", "judgment": "The findings are not sufficiently general to warrant acceptance at a top venue based on the current evidence.", "valence": "negative", "suggested_improvement": "Articulate an argumentative strategy demonstrating scope and robustness across configurations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8553624749183655, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7844553589820862}, {"unit_index": 1, "inspected_object": "Adam's epsilon hyperparameter treatment", "observation": "Epsilon is held constant across all optimizers rather than being swept.", "reasoning": "Epsilon controls effective step size in low-gradient regions and interacts with whitening methods; holding it constant may mask differential sensitivity and confound the comparison.", "judgment": "The fairness and completeness of the optimizer comparison are potentially compromised by this design choice.", "valence": "negative", "suggested_improvement": "Sweep Adam's epsilon to ensure a fair comparison of differential sensitivity.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8558521270751953, "reasoning_key": "design_justification", "reasoning_sim": 0.8046171069145203}, {"unit_index": 2, "inspected_object": "Clarity and interpretability of Figure 3 left panel (spectral descent metric)", "observation": "The reviewer cannot understand the plotted quantity ('singular value of updates') because the update is a vector, not a matrix, and the metric explanation is opaque.", "reasoning": "If a central piece of evidence and its underlying metric are conceptually unclear, the associated argument (that spectral descent accuracy does not explain performance) cannot be assessed or trusted.", "judgment": "The paper's argumentative chain is undermined by the lack of clarity in this key evidence.", "valence": "negative", "suggested_improvement": "Clarify the conceptual link between the plotted quantity and the argument, and resolve the dimensional confusion regarding singular values of vector updates.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.846958339214325, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8283306956291199}, {"unit_index": 3, "inspected_object": "Distinction between weight decay and L2 regularization implementation", "observation": "The paper does not clearly distinguish whether it uses decoupled weight decay or standard L2 regularization.", "reasoning": "Different implementations interact differently with adaptive methods; citing literature on weight decay mechanisms suggests this distinction could affect optimizer performance comparisons.", "judgment": "There is a potential confound in the comparison due to ambiguous regularization implementation.", "valence": "negative", "suggested_improvement": "Clarify the specific regularization implementation used and discuss its interaction with the tested optimizers.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8411009311676025, "reasoning_key": "design_justification", "reasoning_sim": 0.8147368431091309}, {"unit_index": 4, "inspected_object": "Exclusion of natural-gradient methods (KFAC/EKFAC) from the benchmark", "observation": "Natural-gradient methods like KFAC/EKFAC are absent from the comparison despite being relevant to curvature-based optimization.", "reasoning": "These methods use curvature information similar to matrix-whitening optimizers; their absence weakens the theoretical grounding and completeness of the comparison.", "judgment": "The method family studied is incomplete, weakening the comparative analysis.", "valence": "negative", "suggested_improvement": "Include natural-gradient methods like KFAC/EKFAC to provide a more theoretically grounded comparison.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8433775901794434, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7778066992759705}, {"unit_index": 5, "inspected_object": "Reporting of training loss versus validation loss", "observation": "Only validation loss is reported, while training loss is omitted.", "reasoning": "Training loss isolates optimization performance, whereas validation loss can be influenced by regularization and generalization effects; omitting training loss may conflate optimization capability with generalization.", "judgment": "Claims about optimizer performance may be conflating optimization and generalization metrics.", "valence": "negative", "suggested_improvement": "Report training loss to isolate and accurately assess optimization performance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8230438232421875, "reasoning_key": "design_justification", "reasoning_sim": 0.8140416145324707}]}, {"review_id": "0ACuwsN9MM", "paper_id": "07R3pHnBqc", "paper_title": "Instruction Agent: Enhancing Agent with Expert Demonstration", "decision": null, "summary": "The reviewer conducts a feasibility audit, challenging the method's scalability due to per-task human requirements, questioning the verifier's robustness, and criticizing the narrow evaluation scope and lack of operational transparency.", "units": [{"unit_index": 0, "inspected_object": "The use of expert demonstrations at test time", "observation": "The approach requires a human demonstration for every new task.", "reasoning": "While practical, this requirement prevents the system from scaling to hundreds of tasks or dynamic environments where pre-specification is impossible; a useful GUI agent should handle novel tasks without per-task human effort.", "judgment": "The method's scope and practical utility are limited by its lack of scalability.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8268669247627258, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7528384327888489}, {"unit_index": 1, "inspected_object": "The verifier module's reliability", "observation": "The verifier uses screenshot comparison via LLM prompting, which may be brittle.", "reasoning": "As a newly introduced module, it requires independent validation beyond end-to-end results to determine if success is due to the module itself or specific prompt artifacts; novel contributions require deeper scrutiny.", "judgment": "The verifier's robustness is unproven and potentially fragile.", "valence": "negative", "suggested_improvement": "Provide quantitative verifier accuracy metrics and prompt sensitivity ablations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.7905827164649963, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7695044875144958}, {"unit_index": 2, "inspected_object": "The evaluation protocol breadth", "observation": "Evaluation is confined to OSWorld using only one agent backbone (ChatGPT-4o).", "reasoning": "A single benchmark and agent type are too narrow to support general claims about GUI agents; the method must demonstrate generalization across different environments and backbones.", "judgment": "The evidence for generalizability is insufficient.", "valence": "negative", "suggested_improvement": "Evaluate on additional benchmarks and/or with different agent backbones.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8368940949440002, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8339089751243591}, {"unit_index": 3, "inspected_object": "Experimental reporting consistency", "observation": "Table 1 shows a precision discrepancy: agents reported at 0 decimal places while human performance is at 2.", "reasoning": "Inconsistent formatting undermines confidence in the comparability of results and suggests a lack of methodological hygiene; experimental reporting should be uniform and transparent.", "judgment": "The trustworthiness of the reported numbers is questionable.", "valence": "negative", "suggested_improvement": "Ensure consistent formatting and precision in result tables.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8459929823875427, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8530334234237671}, {"unit_index": 4, "inspected_object": "Operational details of input recording and execution", "observation": "The paper does not specify the format for recording mouse/keyboard inputs or how temporal dynamics (e.g., waiting for elements to load) are handled.", "reasoning": "Without these details, it is unclear if the method is robust to real-world timing variations; the absence of operational detail makes the system appear non-self-contained and hinders reproducibility.", "judgment": "The presentation is insufficiently transparent regarding critical implementation details.", "valence": "negative", "suggested_improvement": "Clarify the input recording format and describe handling of temporal execution dynamics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.816831111907959, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8185144066810608}]}, {"review_id": "0ALB52sTUH", "paper_id": "zSsNzxjWP4", "paper_title": "On the Interaction of Compressibility and Adversarial Robustness", "decision": "Accept (Poster)", "summary": "The reviewer applies a standard of mechanistic specificity and scope honesty, demanding formal definitions, process-level explanations, and engagement with counterexamples to substantiate theoretical claims beyond mere correlation.", "units": [{"unit_index": 0, "inspected_object": "Model compression setup and scope of claims", "observation": "The paper does not explicitly state whether compression is achieved via fine-tuning with data or in a data-free manner, yet claims results are 'irrespective of how compressibility is achieved.'", "reasoning": "Compressibility is entangled with feature representations if achieved via fine-tuning; the theoretical claim requires boundary conditions to define which regimes it covers.", "judgment": "The claim space is under-specified and potentially overbroad without explicit scope definition.", "valence": "negative", "suggested_improvement": "Provide a clearer introduction to the model compression setup to clarify boundary conditions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8442177176475525, "reasoning_key": "design_justification", "reasoning_sim": 0.7688425779342651}, {"unit_index": 1, "inspected_object": "Interlayer alignment and sensitive directions", "observation": "The paper attributes potent attack directions to interlayer alignment but neither visualizes nor formally defines these directions or the quantities $A^*_{inf}$ and $A^*_2$.", "reasoning": "Correlation between compressibility and robustness cannot be directly drawn as mechanistic causation without independent verification (formal definition or visualization) of the proposed mechanism.", "judgment": "The mechanism linking compressibility to vulnerability is not substantiated by the evidence presented.", "valence": "negative", "suggested_improvement": "Formally define and visualize the sensitive directions and clarify what $A^*_{inf}$ and $A^*_2$ represent.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.79380863904953, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7790795564651489}, {"unit_index": 2, "inspected_object": "Amplification mechanism", "observation": "It remains unclear how adversarial attacks are amplified and what specific mechanisms cause this amplification to occur as perturbations propagate.", "reasoning": "The theory lacks a process-level explanation for the dynamics of amplification, leaving the contribution incomplete regarding the chain of causation.", "judgment": "The theoretical contribution is incomplete due to lack of explanatory depth on amplification dynamics.", "valence": "negative", "suggested_improvement": "Explain the process-level mechanism of how adversarial perturbations are magnified through the network.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7274354100227356, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.6934570670127869}, {"unit_index": 3, "inspected_object": "Robustness-aware pruning techniques", "observation": "Prior work achieves simultaneous preservation of accuracy and robustness during structured pruning, which contrasts with the paper's claim that structured compressibility induces vulnerability irrespective of achievement method.", "reasoning": "If compression methods exist that preserve robustness, the paper's universal claim may be too broad unless it engages with these counterexamples.", "judgment": "The paper has not engaged with a relevant class of counterexamples, challenging the generality of its claims.", "valence": "negative", "suggested_improvement": "Include additional analysis comparing the theory against adversarially robust model pruning techniques.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8191727995872498, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7924667000770569}, {"unit_index": 4, "inspected_object": "Small spectral compressibility effects", "observation": "Figures 5 and 6 show that small spectral compressibility can slightly improve adversarial robustness for CNNs and Transformers, contradicting the monotonic tension claim.", "reasoning": "A complete theory should accommodate all observed data, including anomalies like positive effects at low compressibility levels.", "judgment": "The theory's explanatory range is incomplete as it fails to account for non-monotonic empirical observations.", "valence": "conditional", "suggested_improvement": "Explain how the proposed theory accounts for the positive effect of small compressibility on robustness.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7650582194328308, "reasoning_key": "robustness_norm", "reasoning_sim": 0.6931741833686829}]}, {"review_id": "0BrKZmNm1i", "paper_id": "paVKFh8veb", "paper_title": "MIDUS: Memory-Infused Depth Up-Scaling", "decision": "Reject", "summary": "The reviewer accepts the empirical results but judges the paper as theoretically and mechanistically under-justified, demanding explanations for why the method works, what the memory components learn, and how to scale them, rather than critiquing the experimental validity itself.", "units": [{"unit_index": 0, "inspected_object": "Theoretical justification for head-wise memory retrieval", "observation": "Absence of formal justification or theoretical analysis explaining why memory retrieval at the head level leads to better generalization or gradient propagation.", "reasoning": "The reviewer expects a causal story connecting the architectural choice (head-wise memory) to observed outcomes, specifically regarding how sparse retrieval interacts with backpropagation and gradient stability.", "judgment": "The work is theoretically under-justified; empirical results alone are insufficient without mechanistic explanation.", "valence": "negative", "suggested_improvement": "Provide formal justification or theoretical analysis of why memory retrieval at head level leads to better generalization or gradient propagation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8010844588279724, "reasoning_key": "design_justification", "reasoning_sim": 0.7712494730949402}, {"unit_index": 1, "inspected_object": "Interpretability of head-wise memories", "observation": "Absence of visualization or analysis showing what the head-wise memories actually learn or retrieve (e.g., task-specific patterns, contextual cues, or token-level semantics).", "reasoning": "Without interpretability evidence, it is impossible to distinguish between a method that genuinely exploits head specialization and one that works for unexamined reasons, undermining the credibility of the contribution.", "judgment": "The mechanism's opacity undermines the contribution's credibility.", "valence": "negative", "suggested_improvement": "Include visualization or analysis of what the head-wise memories actually learn or retrieve.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8031674027442932, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8222550749778748}, {"unit_index": 2, "inspected_object": "Scaling guidance for hyperparameters", "observation": "Absence of guidance on determining memory size and placement, which are new hyperparameters introduced by replacing dense FFN expansion with sparse retrieval.", "reasoning": "A method whose performance depends on unexamined design choices is not yet a mature contribution; scaling behavior is part of the expected contribution for architecture papers.", "judgment": "Practical usability is compromised by unspecified design choices; the method appears brittle without scaling guidance.", "valence": "negative", "suggested_improvement": "Provide guidance on how to determine memory size and placement to ensure robust scaling.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8884314298629761, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8157637715339661}, {"unit_index": 3, "inspected_object": "Learning rate tuning methodology", "observation": "Uncertainty regarding whether the fixed learning rate was empirically tuned for MIDUS or simply adopted from earlier works.", "reasoning": "If the learning rate was not tuned for the proposed method, the comparison against baselines might be unfair, raising reproducibility and fairness concerns.", "judgment": "Experimental rigor is questionable due to lack of clarity on hyperparameter tuning.", "valence": "conditional", "suggested_improvement": "Clarify whether the learning rate was empirically tuned for MIDUS or adopted from prior work.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7994512319564819, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8636969923973083}, {"unit_index": 4, "inspected_object": "Residual connection dynamics", "observation": "Question about underlying dynamics causing internal residual connections to weaken retrieval signals, as mentioned in the paper.", "reasoning": "The reviewer seeks a mechanistic explanation for a phenomenon noted by the authors, indicating a gap in understanding the method's failure modes.", "judgment": "The authors' understanding of their own method's internal dynamics requires further elucidation.", "valence": "uncertain", "suggested_improvement": "Explain the underlying dynamics causing internal residual connections to weaken retrieval signals.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8113901615142822, "reasoning_key": "novelty_standard", "reasoning_sim": 0.787696361541748}, {"unit_index": 5, "inspected_object": "Inference overhead", "observation": "Lack of quantification of per-token inference cost for the HML component.", "reasoning": "Deployment viability requires quantifying practical costs, especially given the paper's central claim of efficiency.", "judgment": "Practical utility is uncertain without quantified efficiency metrics beyond parameter count.", "valence": "negative", "suggested_improvement": "Quantify the per-token inference overhead of HML.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8659133911132812, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8240140080451965}]}, {"review_id": "0ByCKL8m8t", "paper_id": "jz41Oh9eV1", "paper_title": "Robust Preference Alignment via Directional Neighborhood Consensus", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper's conceptual novelty and practical utility positively while critically assessing the theoretical rigor and evaluation validity. The logic centers on distinguishing between heuristic arguments and formal proofs, and demanding convergent evidence for empirical claims.", "units": [{"unit_index": 0, "inspected_object": "Conceptual framing of the method (generate-then-select paradigm)", "observation": "The reviewer identifies a shift from direct constrained generation to sampling from a neighborhood of related preference vectors, viewing this as providing new insights into LLM alignment.", "reasoning": "The reviewer applies a standard of conceptual novelty and utility, valuing the reframing as a promising hypothesis that offers practical value for practitioners seeking robustness.", "judgment": "Positive assessment of the method's core idea as valuable and insightful.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8958120346069336, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8288261890411377}, {"unit_index": 1, "inspected_object": "Method characteristics (post-hoc, training-free nature)", "observation": "The reviewer notes the method is post-hoc and training-free, emphasizing its low barrier for adoption.", "reasoning": "The reviewer evaluates the paper through a deployment lens, prioritizing usability and accessibility alongside correctness.", "judgment": "Positive assessment of the method's practical utility and lightweight character.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8517409563064575, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7963927388191223}, {"unit_index": 2, "inspected_object": "Empirical validation design (compute-matched baseline and diverse models)", "observation": "The reviewer praises the use of compute-matched baselines, a diverse set of models (SFT, DPO, DPA), and an analysis correlating performance gain with OOD-ness.", "reasoning": "The reviewer applies a standard of rigorous experimental design, valuing comparisons that control for resources and demonstrate that improvements scale with difficulty as predicted by the hypothesis.", "judgment": "Positive assessment of the empirical evidence as compelling and rigorously designed.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8776442408561707, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7977868318557739}, {"unit_index": 3, "inspected_object": "Theoretical argument (Theorem 1 and local consistency assumption)", "observation": "The reviewer identifies that the 'local consistency' assumption is presented without formal justification, error bounds, or discussion of conditions under which it holds.", "reasoning": "The reviewer applies a norm that theoretical claims must either provide formal conditions for assumptions or be labeled as heuristic; they distinguish between a proof and a heuristic argument, viewing the current framing as an overclaim.", "judgment": "Negative assessment of the theoretical rigor, viewing the 'proof' as a mischaracterization of a heuristic argument.", "valence": "negative", "suggested_improvement": "Provide conditions on the reward model and distance between vectors that yield bounded error for the approximation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8738949298858643, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7826669216156006}, {"unit_index": 4, "inspected_object": "Evaluation protocol (reliance on single LLM judge)", "observation": "The reviewer notes the reliance on GPT-4o-mini as the sole judge for win rates.", "reasoning": "The reviewer applies a triangulation norm, arguing that a single automated metric is insufficient for quality claims due to potential artifacts of the specific judge's biases; convergent evidence is required.", "judgment": "Negative assessment of the evaluation validity, citing risk that results are artifacts rather than true improvements.", "valence": "negative", "suggested_improvement": "Conduct a human evaluation study or analyze using multiple distinct judge models to check for consensus.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7677735686302185, "reasoning_key": "construct_validity", "reasoning_sim": 0.7858000993728638}, {"unit_index": 5, "inspected_object": "Generalizability to higher-dimensional spaces", "observation": "The reviewer questions the method's behavior with ≥3 attributes and asks for preliminary results.", "reasoning": "The reviewer assesses the scope of the method, noting that experiments focus on 2D spaces and seeking evidence of conceptual scaling.", "judgment": "Uncertainty regarding the method's generalizability beyond the specific experimental settings.", "valence": "uncertain", "suggested_improvement": "Report preliminary results or analysis for preference spaces with three or more attributes.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8434411883354187, "reasoning_key": "merit_recognition", "reasoning_sim": 0.794847309589386}, {"unit_index": 6, "inspected_object": "Hyperparameter sensitivity (k and theta_max)", "observation": "The reviewer asks about the impact of neighborhood size (k) and angular threshold (theta_max).", "reasoning": "The reviewer applies a robustness standard, seeking to determine if the method's performance is fragile to parameter choices.", "judgment": "Uncertainty regarding the robustness of the method to hyperparameter variations.", "valence": "uncertain", "suggested_improvement": "Perform a sensitivity analysis or ablation study on k and theta_max.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8536084294319153, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8520312309265137}]}, {"review_id": "0CheAAk0h7", "paper_id": "rp51BU7LeW", "paper_title": "V-MAGE: A Game Evaluation Framework for Assessing Vision-Centric Capabilities in Multimodal Large Language Models", "decision": null, "summary": "The reviewer validates the paper's internal coherence, diagnostic granularity, and metric design as strengths, but critically challenges the lack of mechanistic root-cause attribution and external predictive validity, requesting specific ablations and correlations to bridge these epistemic gaps.", "units": [{"unit_index": 0, "inspected_object": "Problem framing argument regarding grid-based or text-reducible games failing to stress vision-centric competencies", "observation": "Reviewer affirms the justification for the benchmark's motivation against existing alternatives.", "reasoning": "The reviewer validates the motivating premise using an assumption check, accepting that a benchmark must be justified against alternatives and finding this specific justification convincing.", "judgment": "Positive assessment of the problem definition and motivation.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8433240652084351, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7921558618545532}, {"unit_index": 1, "inspected_object": "Unit tests mapping to interpretable sub-skills (specifically weaknesses in tracking and timing)", "observation": "Reviewer identifies diagnostic granularity as a key feature, noting the ability to localize failures to specific competencies.", "reasoning": "The reviewer treats the benchmark as a measurement instrument whose value lies in producing fine-grained, actionable information rather than just aggregate scores.", "judgment": "Positive assessment of the benchmark's diagnostic utility and granularity.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8652791976928711, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7779988646507263}, {"unit_index": 2, "inspected_object": "ELO system design for unified metric retention of per-level diagnostics", "observation": "Reviewer praises the system for delivering a scale-robust metric while retaining diagnostic capabilities.", "reasoning": "The reviewer evaluates this as a successful resolution to the common benchmark-design tension between aggregation and specificity.", "judgment": "Positive assessment of the metric design.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8319351077079773, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8382418751716614}, {"unit_index": 3, "inspected_object": "Anchoring bias quantification and error taxonomy", "observation": "Reviewer notes these elements go beyond mere leaderboard reporting.", "reasoning": "The reviewer holds the standard that evaluation papers should explain model failures and generate hypotheses about limitations, not just provide measurements.", "judgment": "Positive assessment of the analytical depth beyond ranking.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8552983999252319, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7573720216751099}, {"unit_index": 4, "inspected_object": "Perception bypass ablation study", "observation": "Reviewer recognizes the ablation as strengthening the claim that visual perception and downstream reasoning jointly constrain performance.", "reasoning": "The reviewer values this as evidence of rigorous causal inference design that decomposes sources of model failure.", "judgment": "Positive assessment of the methodological rigor in isolating failure modes.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7769843339920044, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7581019997596741}, {"unit_index": 5, "inspected_object": "Root cause analysis of model failures (e.g., poor tracking)", "observation": "Reviewer observes that the paper documents *that* models fail but not *why*, lacking mechanistic discrimination among architectural, data, or inference causes.", "reasoning": "The reviewer applies the standard that a benchmark paper should discriminate among mechanistic explanations and attribute failures to specific components of the MLLM pipeline, treating the benchmark as a diagnostic tool for attribution.", "judgment": "Negative judgment regarding the interpretive reach and causal depth of the findings.", "valence": "negative", "suggested_improvement": "Conduct ablations or analyses to discriminate among candidate explanations such as architectural constraints, training data gaps, or inference inefficiencies.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8045096397399902, "reasoning_key": "merit_recognition", "reasoning_sim": 0.764507532119751}, {"unit_index": 6, "inspected_object": "Connection between game-specific scores and real-world vision-centric scenarios", "observation": "Reviewer notes the absence of correlation evidence between V-MAGE performance and external real-world tasks.", "reasoning": "The reviewer applies a construct validity norm requiring predictive validity for tasks outside the benchmark to establish the framework's utility and transferability.", "judgment": "Negative judgment regarding the external validity and generalization of the benchmark.", "valence": "negative", "suggested_improvement": "Demonstrate correlation between V-MAGE performance and real-world task performance to provide external validation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.810706615447998, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7896301746368408}, {"unit_index": 7, "inspected_object": "Minimal agent wrapper and its potential confounds", "observation": "Reviewer questions whether the evaluation harness itself introduces bottlenecks distinct from inherent MLLM deficiencies.", "reasoning": "The reviewer seeks to isolate the contribution of the evaluation harness from the model to rule out confounds in the 'minimal agent wrapper' design.", "judgment": "Uncertainty/Condition regarding the isolation of model capabilities from harness effects.", "valence": "conditional", "suggested_improvement": "Perform an ablation to disentangle agent-side processing bottlenecks from inherent deficiencies in the MLLMs themselves.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8230534195899963, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7733975648880005}]}, {"review_id": "0CyRr8KYGY", "paper_id": "yDbJHQlrbf", "paper_title": "Code Driven Planning with Domain-Adaptive Selector", "decision": "Accept (Poster)", "summary": "The reviewer accepts the headline results but questions the method's robustness and distinctiveness by identifying dependencies on plan quality and RL fine-tuning, and by noting the lack of comparison with hybrid PDDL baselines. The evaluation focuses on boundary conditions rather than empirical details.", "units": [{"unit_index": 0, "inspected_object": "Dependency on generated plan quality", "observation": "Solutions are highly dependent on the quality of the generated plans.", "reasoning": "If performance hinges on inputs that are variable or potentially unreliable, the method's robustness is questionable and it may degrade if conditions are not favorable.", "judgment": "The dependency on plan quality is a weakness indicating potential fragility.", "valence": "negative", "suggested_improvement": "Conduct sensitivity analysis to characterize how performance varies with the diversity and quality of generated plans.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.7989659309387207, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8716344833374023}, {"unit_index": 1, "inspected_object": "Dependency on RL fine-tuning", "observation": "Selector's effectiveness depends on RL fine-tuning.", "reasoning": "RL fine-tuning is treated as a cost or risk rather than just a component, implying it may be hard to reproduce, sensitive to hyperparameters, or less predictable than supervised methods.", "judgment": "The reliance on RL fine-tuning is a weakness due to potential unreliability or complexity.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8508597612380981, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8022082448005676}, {"unit_index": 2, "inspected_object": "Baseline comparison with PDDL+LLM hybrids", "observation": "The paper does not compare against PPDL approaches that use LLM alongside additional solvers.", "reasoning": "Claimed advantages in cost reduction and performance might not be unique to CoPiC; hybrid systems could achieve similar results, so the paper should position itself against these stronger alternatives.", "judgment": "The lack of comparison with hybrid baselines creates uncertainty about the method's distinctiveness.", "valence": "conditional", "suggested_improvement": "Consider using PPDL approaches with LLMs and additional solvers as baselines to demonstrate distinctiveness.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.830885648727417, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7431764602661133}, {"unit_index": 3, "inspected_object": "Sensitivity of performance to number of generated plans", "observation": "The reviewer asks how the number of generated plans affects performance and cost.", "reasoning": "Without understanding the shape of the performance landscape (cliff, plateau, sweet spot), it is impossible to assess whether the method is robust or fragile regarding this hyperparameter.", "judgment": "The absence of sensitivity data for plan count is a gap in evaluating robustness.", "valence": "uncertain", "suggested_improvement": "Analyze and report how performance and cost vary with the number of generated plans.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8211139440536499, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8108012080192566}]}, {"review_id": "0DC6qYYmtJ", "paper_id": "eaAGI1lIb4", "paper_title": "XIL: Cross-Expanding Incremental Learning", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper primarily on its positioning relative to prior work and its grounding in real-world data, expressing skepticism about the novelty claims due to insufficient differentiation from cross-domain continual learning and a lack of evidence for the prompt selection mechanism's efficacy.", "units": [{"unit_index": 0, "inspected_object": "Novelty and differentiation of the XIL setting from cross-domain continual learning (references [1] and [2])", "observation": "The reviewer acknowledges that XIL is 'new' but observes that it 'intersects with existing setting' and notes the absence of a clear explanation differentiating the proposed work from prior cross-domain continual learning approaches.", "reasoning": "The reviewer operates under an evaluative standard where novelty requires explicit boundary-drawing against the closest prior work; without this differentiation, the contribution's distinctness remains unproven and its soundness is compromised.", "judgment": "The novelty claim is plausible but currently unestablished due to insufficient differentiation from existing literature.", "valence": "negative", "suggested_improvement": "Explain the difference between the proposed work and cross-domain continual learning as proposed in references [1] and [2].", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8545302152633667, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8483880162239075}, {"unit_index": 1, "inspected_object": "Prompt selection mechanism design and performance", "observation": "The reviewer identifies the prompt selection mechanism as non-parametric and potentially over-simplified, noting that its accuracy is not reported.", "reasoning": "A non-parametric mechanism may fail to learn appropriate selections; therefore, evidence of its behavior (accuracy) is required to verify that it functions as intended and does not introduce errors into the pipeline.", "judgment": "The mechanism is suspect due to lack of diagnostic evidence, representing a potential weak link in the system.", "valence": "negative", "suggested_improvement": "Report the prompt selection accuracy to provide evidence that the selection mechanism works correctly.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8507948517799377, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8122544288635254}, {"unit_index": 2, "inspected_object": "Documentation of domain order sensitivity", "observation": "The reviewer asserts that domain order should affect results in continual learning systems and observes that this sensitivity is not detailed in the paper.", "reasoning": "Continual learning papers are expected to report order sensitivity as a default property; failing to document how or if order affects outcomes leaves a gap in understanding the system's robustness and behavior.", "judgment": "The paper lacks necessary documentation regarding a fundamental expected property of the problem setting.", "valence": "negative", "suggested_improvement": "Detail how domain order affects the results, either through sensitivity analysis, theoretical argument, or discussion.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8616409301757812, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7825113534927368}, {"unit_index": 3, "inspected_object": "Real-world grounding and dataset applicability of the XIL setting", "observation": "The reviewer questions whether the benchmarks used actually instantiate the XIL setting and asks for a concrete dataset that represents the real-world context of the problem.", "reasoning": "A problem setting's value is contingent on its real-world instantiation; if no natural dataset exists to capture the setting, the contribution may be viewed as a purely synthetic academic exercise rather than a practical advancement.", "judgment": "The real-world applicability of the setting is uncertain and requires clarification to establish pragmatic validity.", "valence": "negative", "suggested_improvement": "Elaborate on the real-world context and link the problem directly to a concrete dataset that represents it.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8368261456489563, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8097538352012634}]}, {"review_id": "0Dqfy7TxC4", "paper_id": "bKcBgNiREu", "paper_title": "FreeFuse: Multi-Subject LoRA Fusion via Auto Masking at Test Time", "decision": "Reject", "summary": "The reviewer offers qualified endorsement based on strong theoretical grounding and specific metric successes, while identifying critical empirical gaps in efficiency verification, generalization scope, and scalability data.", "units": [{"unit_index": 0, "inspected_object": "Theoretical justification of the core mechanism (masking LoRA outputs to approximate isolated inference)", "observation": "Reviewer notes the paper's claim that masks are extracted from a single attention block and denoising step, and that locality of attention ensures near-identical representations inside the mask.", "reasoning": "The reviewer finds the formal mathematical argument persuasive and the formulation elegant, accepting the theoretical grounding as strong despite empirical gaps.", "judgment": "The core conceptual idea is compelling and theoretically sound.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8271600604057312, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8264918327331543}, {"unit_index": 1, "inspected_object": "Empirical evaluation metrics (VLM score, DreamSim)", "observation": "Reviewer highlights specific high scores: VLM 74.03 vs 57.74 for Mix-of-Show and DreamSim 10-pass 0.8052.", "reasoning": "These specific numbers anchor the reviewer's positive assessment of generation quality, particularly because the VLM score represents a holistic, language-grounded evaluation.", "judgment": "The method achieves strong performance on key holistic metrics.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.857384204864502, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7465643882751465}, {"unit_index": 2, "inspected_object": "Ablation study (Fig. 7) regarding attention-sink handling and voting", "observation": "Reviewer observes that Fig. 7 clearly isolates the effect of attention-sink handling, self-attention maps, and block-level voting, noting that omitting any step causes visible artifacts.", "reasoning": "This visual evidence reinforces the necessity of each design component, providing convincing internal coherence for the method.", "judgment": "The ablation study successfully validates the individual contributions of the method's components.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8184571862220764, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7863001823425293}, {"unit_index": 3, "inspected_object": "Efficiency claims (runtime benchmarks)", "observation": "Paper mentions '37s per image' but lacks comparative timing against CLoRA/OMG or multi-step variants.", "reasoning": "Without comparative data, it is hard to quantify efficiency gains; the central selling point of being 'training-free' and 'efficient' cannot be verified against alternatives.", "judgment": "The efficiency claim is unsubstantiated due to missing comparative benchmarks.", "valence": "negative", "suggested_improvement": "Provide comparative runtime benchmarks against existing methods like CLoRA or OMG.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8741244673728943, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8421130180358887}, {"unit_index": 4, "inspected_object": "Generalization beyond photorealistic human scenarios", "observation": "Evaluation prompts emphasize intimate, realistic human interactions; no tests on style LoRAs, cartoons, or abstract concepts.", "reasoning": "The mask-extraction mechanism relies on cross-attention maps localizing subjects; if applied to style LoRAs which modify global texture/color, masking may be nonsensical, raising doubts about generality.", "judgment": "Generality claims are weak due to narrow evaluation domain focused on human-centric prompts.", "valence": "negative", "suggested_improvement": "Test generalization on object+character or style+subject LoRA fusion (e.g., anime character + van Gogh style).", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.830773651599884, "reasoning_key": "design_justification", "reasoning_sim": 0.7910677194595337}, {"unit_index": 5, "inspected_object": "Scalability with subject count", "observation": "No quantitative evidence provided for performance degradation beyond two subjects.", "reasoning": "Cross-attention maps may become more diffuse or overlapping with more subjects, potentially making mask extraction less reliable; scalability limits are unknown.", "judgment": "Multi-subject capability claims lack quantitative support for scaling behavior.", "valence": "negative", "suggested_improvement": "Provide quantitative degradation data (e.g., VLM/LVFace scores) for 3-, 4-, and 5-subject scenes.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8474684953689575, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7635438442230225}, {"unit_index": 6, "inspected_object": "Parameter sensitivity (threshold p=1% in Eq. 6)", "observation": "Fixed threshold for attention-sink filtering; reviewer questions if this was tuned for different resolutions/datasets.", "reasoning": "Suspicion that results may be contingent on careful hyperparameter tuning rather than robust across settings.", "judgment": "Method robustness to hyperparameter choices is uncertain.", "valence": "conditional", "suggested_improvement": "Clarify if p was tuned for different resolutions/datasets and report sensitivity analysis.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8040158152580261, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8995088934898376}]}, {"review_id": "0Dsuw7HwFc", "paper_id": "fWHd3yYicX", "paper_title": "Train on Validation (ToV): Fast data selection with applications to fine-tuning", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper primarily on practical completeness, finding the core idea sound but the evidentiary basis incomplete. They identify specific gaps in cost measurement, metric diversity, and baseline coverage that undermine the claims of speed and quality, while also raising mechanistic concerns about overfitting and symmetry failure.", "units": [{"unit_index": 0, "inspected_object": "Efficiency claim (title: 'Fast data selection') and associated empirical evidence", "observation": "The paper asserts efficiency but provides no wall-clock time, memory usage, or forward/backward pass counts on identical hardware.", "reasoning": "Without quantitative cost measurements, the 'fast' claim remains an assertion rather than a demonstrated advantage; the reviewer contrasts ToV's overhead (repeated validation fine-tuning) with LESS's warm-up costs to highlight this gap.", "judgment": "The efficiency claim is not credible due to lack of measurement.", "valence": "negative", "suggested_improvement": "Provide quantitative cost comparison including wall-clock time and resource usage.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8687172532081604, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7826089859008789}, {"unit_index": 1, "inspected_object": "Empirical evaluation metrics (test log-loss)", "observation": "ToV's empirical case rests entirely on test log-loss, despite LESS warning against using loss as a proxy for accuracy in LLM evaluation.", "reasoning": "Fine-tuning is ultimately about task performance; loss reductions may not translate to task improvements, so relying solely on log-loss fails to establish 'better task quality'.", "judgment": "Task quality improvement is not established by the current evidence.", "valence": "negative", "suggested_improvement": "Add complementary metrics such as accuracy, MMLU-style subsets, exact-match, or F1.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8752080798149109, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7819897532463074}, {"unit_index": 2, "inspected_object": "Baseline coverage (comparison methods)", "observation": "All compared baselines (LESS, random, uncertainty) either use per-example gradients or are trivial; alignment-based selectors (importance resampling) and datamodel-style methods (TRAK) are absent.", "reasoning": "Comparing only against gradient-using methods makes it unclear whether ToV's advantage comes from the selection principle itself or simply from avoiding expensive gradient machinery; comparing against other gradient-free methods is needed to isolate the source of advantage.", "judgment": "The advantage over competing methods is not generalizable without broader baseline coverage.", "valence": "negative", "suggested_improvement": "Add comparisons with alignment-based selectors and datamodel-style methods that avoid per-example gradient computations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8898054361343384, "reasoning_key": "design_justification", "reasoning_sim": 0.8285931348800659}, {"unit_index": 3, "inspected_object": "Overfitting risk due to repeated validation set usage", "observation": "ToV uses the validation set both to train the scoring model and to evaluate the selected subset, creating a potential feedback loop.", "reasoning": "Repeated training on a small validation set during scoring could cause the selection policy to overfit, potentially deteriorating test performance with more ToV cycles.", "judgment": "There is an unresolved mechanistic concern regarding overfitting to the validation set.", "valence": "conditional", "suggested_improvement": "Analyze whether test performance deteriorates with more ToV cycles to assess overfitting.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8232686519622803, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7574526071548462}, {"unit_index": 4, "inspected_object": "Train-on-val symmetry assumption", "observation": "The paper relies on the symmetry between training on validation data and training on target data but does not discuss failure regimes.", "reasoning": "The symmetry appears to be an empirical regularity rather than a mathematical identity; understanding where this approximation breaks is critical for defining the method's practical boundaries.", "judgment": "The scope and reliability of the method are uncertain without characterization of symmetry failure modes.", "valence": "uncertain", "suggested_improvement": "Discuss when the train-on-val = train-on-x symmetry breaks down.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8499068021774292, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7521673440933228}, {"unit_index": 5, "inspected_object": "Base subset size hyperparameter ($m$)", "observation": "The effect of the base subset size $m$ is not characterized in the paper.", "reasoning": "This parameter acts as a potentially sensitive knob for the heuristic; its sensitivity affects the method's usability and robustness for practitioners.", "judgment": "Hyperparameter sensitivity is unknown, limiting practical guidance.", "valence": "uncertain", "suggested_improvement": "Characterize the effect of varying the base subset size $m$.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.8180104494094849, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8022906184196472}]}, {"review_id": "0E5B2JELkG", "paper_id": "9nOpFdYJez", "paper_title": "From Sparse to Structured: A New Paradigm for Gradient-Based Parameter-Efficient Fine-Tuning", "decision": "Reject", "summary": "The reviewer evaluates the paper with qualified approval, critically examining the conceptual justification for row/column selection versus sparse selection, the narrow experimental scope limited to classification, and suspected inconsistencies in data presentation between tables and figures. While acknowledging strong writing and convincing ablations, the reviewer flags significant concerns regarding the novelty of the contribution, the need for broader validation, and potential selective reporting.", "units": [{"unit_index": 0, "inspected_object": "Theoretical justification for row/column selection over sparse selection", "observation": "The reviewer finds that the authors do not adequately explain why the row/column selection scheme is more effective than the sparse selection scheme.", "reasoning": "The reviewer reasons that if sparse selection identifies high-gradient individual parameters and row selection identifies high-gradient rows, the latter is a coarser approximation. The paper's theoretical claim of optimality must justify why this coarsening is beneficial rather than lossy, and the reviewer does not accept empirical results as sufficient evidence without a compelling conceptual story.", "judgment": "Conceptual gap in the fundamental premise; the paper fails to articulate a clear mechanistic distinction from its closest competitor (GPS).", "valence": "negative", "suggested_improvement": "Further elaborate on the contributions relative to GPS and provide a clear mechanistic distinction.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8356908559799194, "reasoning_key": "design_justification", "reasoning_sim": 0.8653923869132996}, {"unit_index": 1, "inspected_object": "Experimental scope limited to classification datasets", "observation": "The reviewer notes that experiments are conducted only on classification datasets, which are prone to overfitting.", "reasoning": "The reviewer applies a methodological norm that parameter-efficient fine-tuning methods should be validated across task types stressing different capabilities. They suspect the method's success might be specific to classification due to regularization effects rather than genuine superiority of the selection mechanism, requiring evidence from challenging tasks like object detection or segmentation to establish generality.", "judgment": "Insufficient experimental breadth to support broad claims; generalization risk remains unaddressed.", "valence": "negative", "suggested_improvement": "Conduct experimental analysis on more challenging tasks such as object detection and segmentation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8228399753570557, "reasoning_key": "design_justification", "reasoning_sim": 0.8577937483787537}, {"unit_index": 2, "inspected_object": "Consistency between ablation tables and figures", "observation": "The reviewer observes that Figures 3 and 4 appear to intentionally omit results for the Oxford Flowers subset, which are present in Tables 4 and 5.", "reasoning": "The reviewer engages in forensic cross-referencing and infers potential selective reporting of favorable data. They apply a norm that figures and tables should present consistent information and that omissions require explanation, flagging this as a potential ethical issue regarding evidentiary integrity.", "judgment": "Suspicion of deliberate selective reporting; integrity concern casts doubt on the empirical section.", "valence": "negative", "suggested_improvement": "Clarify the omission of the Oxford Flowers subset from the figures or include the missing results.", "support_status": "mixed", "confidence": "medium", "object_key": "clarity", "object_sim": 0.8372508883476257, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7921302318572998}, {"unit_index": 3, "inspected_object": "Mask generation process (static vs. dynamic)", "observation": "The reviewer questions the design choice of using pre-generated static masks across epochs.", "reasoning": "The reviewer probes whether the static mask is a limitation or a deliberate choice, hypothesizing that dynamic masks might improve performance by adapting to the changing gradient landscape during training. This reflects an expectation that papers should justify design decisions against plausible alternatives.", "judgment": "Uncertainty about the robustness of the design choice; potential for improvement via dynamic adaptation.", "valence": "conditional", "suggested_improvement": "Analyze or discuss the impact of using dynamic masks compared to static ones.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7679041624069214, "reasoning_key": "design_justification", "reasoning_sim": 0.8481557369232178}, {"unit_index": 4, "inspected_object": "Metric selection for parameter selection", "observation": "The reviewer asks if there are better metrics than the row sum of squares for parameter selection.", "reasoning": "The reviewer implicitly challenges the arbitrariness of the chosen metric, connecting it to the paper's theoretical claims of optimality. They expect the authors to defend the sum of squared gradients against plausible alternatives.", "judgment": "Questioning the theoretical grounding and optimality of the specific metric choice.", "valence": "conditional", "suggested_improvement": "Justify the choice of row sum of squares against alternative metrics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8580531477928162, "reasoning_key": "design_justification", "reasoning_sim": 0.8468592166900635}, {"unit_index": 5, "inspected_object": "Equation formatting", "observation": "The reviewer notes that 'Eq. equation i' should be 'Eq. i'.", "reasoning": "This is identified as a trivial formatting complaint that signals thorough reading but carries no evaluative weight on the paper's scientific merit.", "judgment": "Minor presentation error.", "valence": "negative", "suggested_improvement": "Correct the formatting of equation references.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7893974781036377, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7809763550758362}]}, {"review_id": "0E7l3hy8kZ", "paper_id": "AZl4sTVDCC", "paper_title": "Query-aware Hub Prototype Learning for Few-Shot 3D Point Cloud Semantic Segmentation", "decision": "Reject", "summary": "The reviewer acknowledges the paper's novelty and procedural completeness but judges it wanting due to a lack of engagement with current frontier baselines, unexamined robustness and scaling properties, and missing diagnostic qualitative evidence. The evaluation focuses on external relevance and field positioning rather than internal technical validity.", "units": [{"unit_index": 0, "inspected_object": "Conceptual core: hub-based prototype generation for support-query misalignment", "observation": "The reviewer identifies the introduction of hub-based prototype generation as novel and frames it as a new perspective on addressing support-query misalignment.", "reasoning": "The reviewer accepts this mechanism at face value without questioning technical details (bipartite graph construction, hub identification, purity-reweighted contrastive loss), treating it as a substantive strength rather than a procedural one.", "judgment": "Novel conceptual contribution that provides a new perspective on the problem.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7817127108573914, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8177284002304077}, {"unit_index": 1, "inspected_object": "Experimental apparatus: evaluations on S3DIS/ScanNet, ablations, parameter sensitivity", "observation": "Evaluations cover multiple datasets and shot settings with ablations and sensitivity analyses present.", "reasoning": "The reviewer treats this as a strength but characterizes it as *procedural*—checking that standard boxes are ticked—rather than evaluating whether the experiments are diagnostic of the method's specific claims or robustness.", "judgment": "Procedurally complete but potentially lacking in diagnostic depth regarding the method's specific claims.", "valence": "mixed", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8413692116737366, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7807956337928772}, {"unit_index": 2, "inspected_object": "Comparison set against recent literature (transformer-based meta-learners, distillation, prompt-based adaptation)", "observation": "The paper does not compare against recent transformer-based meta-learners, distillation methods, or prompt-based adaptation.", "reasoning": "The reviewer argues that the field has moved toward these paradigms; without these comparisons, the claimed 'state-of-the-art' performance is unverifiable because the yardstick is outdated. This positions the method's relevance as questionable relative to current frontier methods.", "judgment": "Limited baselines undermine the verification of the paper's claimed relevance and state-of-the-art status.", "valence": "negative", "suggested_improvement": "Compare against recent transformer-based meta-learners, distillation methods, or prompt-based adaptation to establish relevance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8486033082008362, "reasoning_key": "design_justification", "reasoning_sim": 0.8272361159324646}, {"unit_index": 3, "inspected_object": "Robustness analysis under domain shift, class imbalance, and open-set scenarios", "observation": "The paper does not analyze robustness under significant domain shift, class imbalance, or real-world open-set scenarios.", "reasoning": "The reviewer suspects the method may be 'somewhat heuristic' and works only on tested benchmarks. They demand demonstration of boundary conditions to ensure the method is not just a clever trick but has principled guarantees or robustness.", "judgment": "Lack of robustness analysis creates uncertainty about the method's generalizability beyond specific benchmarks.", "valence": "negative", "suggested_improvement": "Analyze robustness under domain shift, class imbalance, and open-set scenarios to demonstrate boundary conditions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8408142924308777, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8205094933509827}, {"unit_index": 4, "inspected_object": "Scaling behavior of graph construction and clustering in large-scale settings", "observation": "The cost of bipartite graph construction and clustering in large-scale 5-way settings is not analyzed, despite brief comparisons of FLOPs/inference time.", "reasoning": "The reviewer concerns that computational overhead might explode with more classes/points, threatening practical deployability. The request is for evidence that the method scales gracefully.", "judgment": "Unanalyzed scaling behavior raises doubts about the method's practical viability in large-scale applications.", "valence": "negative", "suggested_improvement": "Provide a scaling analysis of the computational cost in large-scale 5-way settings to demonstrate practical deployability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7775005102157593, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8611447215080261}, {"unit_index": 5, "inspected_object": "Qualitative visualizations: presence of failure cases", "observation": "All qualitative visualizations emphasize improvement; no failure cases are shown.", "reasoning": "The reviewer requests diagnostic visualizations to understand where the method breaks. The implicit assumption is that limitations are as informative as strengths for understanding the mechanism.", "judgment": "Absence of failure cases limits the diagnostic value of qualitative results.", "valence": "negative", "suggested_improvement": "Include failure cases in qualitative visualizations to provide a realistic sense of when the method works and when it does not.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8086449503898621, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7662288546562195}, {"unit_index": 6, "inspected_object": "Motivation presentation: visual example of prototype bias", "observation": "Prototype bias is well described in text but lacks a concrete visual example of support-only prototypes failing on specific query samples.", "reasoning": "The reviewer wants to *show* the problem solved, not just *tell* it. A before/after comparison would make the contribution intuitively graspable and more persuasive.", "judgment": "Motivation is textually adequate but visually insufficient for intuitive graspability.", "valence": "negative", "suggested_improvement": "Provide a concrete visual example of support-only prototypes failing on specific query samples to visually demonstrate the motivating problem.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8495612740516663, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7999837398529053}]}, {"review_id": "064gru5FCj", "paper_id": "R8y089OGoo", "paper_title": "Dichotomous Diffusion Policy Optimization", "decision": "Accept (Poster)", "summary": "The reviewer performs a boundary-testing evaluation, accepting the method's internal validity within offline settings but challenging its positioning through two specific critiques: the lack of explanation for limited scope (offline vs. online) and the perceived lack of principled justification for baseline selection.", "units": [{"unit_index": 0, "inspected_object": "Scope of applicability regarding online RL", "observation": "The method is evaluated only in offline or offline-to-online settings, with no explicit discussion of its applicability to pure online RL.", "reasoning": "The reviewer applies a norm of generalizability, noting that the core mechanism (dichotomous decomposition) appears to be a general algorithmic idea similar to results found in online RL literature (e.g., Ma et al.). The absence of an explanation for why this mechanism does not transfer to online settings creates uncertainty about the method's fundamental boundaries and generality.", "judgment": "Uncertainty regarding whether the paper has sufficiently articulated the scope conditions of its design.", "valence": "negative", "suggested_improvement": "Articulate the scope conditions of the design, specifically explaining why the dichotomous trick may or may not transfer from offline to online settings.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.790467381477356, "reasoning_key": "design_justification", "reasoning_sim": 0.857663631439209}, {"unit_index": 1, "inspected_object": "Selection and justification of empirical baselines", "observation": "The baseline set includes model-free methods but omits model-based offline RL baselines, and the reviewer finds the paper's explanation for these choices unconvincing.", "reasoning": "The reviewer applies a norm of principled baseline selection, expecting coverage of major methodological families (e.g., model-based vs. model-free) to provide a rigorous comparison. The phrase 'seem random' indicates a perception that the selection lacks organizing logic, raising concerns about the completeness and integrity of the empirical validation.", "judgment": "Dissatisfaction with the rigor and rationale of the comparative evaluation.", "valence": "negative", "suggested_improvement": "Provide a clearer rationale for baseline choices, potentially including model-based offline RL baselines to demonstrate comprehensive coverage of the methodological landscape.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8997688889503479, "reasoning_key": "design_justification", "reasoning_sim": 0.8216291666030884}]}, {"review_id": "06QSOoQbW7", "paper_id": "0xE0kNdGIz", "paper_title": "Vulcan: Crafting Compact Class-Specific Vision Transformers For Edge Intelligence", "decision": "Accept (Poster)", "summary": "The reviewer acknowledges strong experimental execution and reproducibility but raises significant concerns regarding the novelty of the core components, the practical necessity compared to lightweight alternatives, the fairness of baseline comparisons, and the theoretical justification for joint optimization.", "units": [{"unit_index": 0, "inspected_object": "Novelty of the proposed method components (FFN neuron clustering/collapse and MHA TNNR regularization)", "observation": "The reviewer identifies that the empirical observation regarding FFN layers capturing class-related information is known, and the technical techniques (clustering, low-rank decomposition) are well-established in the literature.", "reasoning": "The reviewer applies a component-level novelty standard, treating novelty as the sum of its parts rather than considering the combination or problem framing. Since both the underlying observation and the machinery are deemed standard, the reviewer infers limited residual novelty for the work.", "judgment": "Limited novelty due to reliance on established observations and techniques.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7808855772018433, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8391450047492981}, {"unit_index": 1, "inspected_object": "Practical necessity of the approach compared to training lightweight models from scratch", "observation": "The reviewer notes that the application scenario involves only a small number of target classes, which could often be addressed by training lightweight models on smaller datasets.", "reasoning": "The reviewer employs a cost-benefit and pragmatic criterion, arguing that if a simpler alternative (training small models) exists for many real-world deployments, the complexity of the proposed compression framework may be unjustified unless it demonstrates a clear advantage over this baseline.", "judgment": "Questionable value proposition if lightweight alternatives are viable.", "valence": "negative", "suggested_improvement": "Conduct a head-to-head comparison with lightweight models trained from scratch on the target classes to demonstrate Vulcan's advantage.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.844520092010498, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8414687514305115}, {"unit_index": 2, "inspected_object": "Fairness of experimental baselines regarding post-training adaptation", "observation": "The reviewer observes that the paper does not specify whether baseline models were retrained on the same class subsets as the proposed method.", "reasoning": "The reviewer assumes a procedural norm of symmetric training for fair comparison. The absence of specification creates suspicion of experimental asymmetry, where the reported accuracy improvements might be artifacts of unequal training budgets or adaptation opportunities rather than the pruning strategy itself.", "judgment": "Potential confound in experimental evaluation due to unspecified baseline protocols.", "valence": "negative", "suggested_improvement": "Clarify whether baselines received equivalent post-training on the class subset, or discuss why any asymmetry is acceptable.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8531737327575684, "reasoning_key": "design_justification", "reasoning_sim": 0.8411888480186462}, {"unit_index": 3, "inspected_object": "Theoretical justification and interaction effects in joint optimization", "observation": "The reviewer notes that the losses for FFN collapse and MHA low-rank compression are directly added together without explicit theoretical grounding or analysis of potential interactions.", "reasoning": "The reviewer applies a norm requiring design choices in method papers to have either theoretical justification or empirical validation. The additive loss design is treated as a claim needing support, raising concerns about unexamined blind spots or pathologies in the joint optimization process.", "judgment": "Methodological coherence is questioned due to lack of theoretical or empirical evidence for the additive loss structure.", "valence": "negative", "suggested_improvement": "Provide a theoretical argument for why the additive loss is well-posed, or conduct ablation experiments showing that joint optimization does not degrade component effectiveness.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8418227434158325, "reasoning_key": "design_justification", "reasoning_sim": 0.8289041519165039}, {"unit_index": 4, "inspected_object": "Experimental breadth and reproducibility details", "observation": "The reviewer catalogs the extensive use of multiple datasets (ImageNet, CIFAR-10/100, COCO), tasks, metrics, implementation details, hyperparameter sensitivity analyses, and neuron visualizations.", "reasoning": "The reviewer values thorough execution and transparency, noting that the careful reading of supplementary material and ablation studies indicates high effort and robustness in the experimental apparatus.", "judgment": "Strong execution and reproducibility.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8789529204368591, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8017669916152954}]}, {"review_id": "07JMWwptT4", "paper_id": "tz5GRv9Vzu", "paper_title": "Durian: Dual Reference Image-Guided Portrait Animation with Attribute Transfer", "decision": "Accept (Poster)", "summary": "The reviewer operates as a skeptical comparativist, systematically undermining the paper's contribution claims by attacking input provenance fairness, metric validity, baseline competitiveness, and architectural novelty, concluding the work is incremental engineering rather than fundamental invention.", "units": [{"unit_index": 0, "inspected_object": "Self-Reenactment Pipeline Input Provenance", "observation": "Discrepancy between Figure 2 (guidance video frames as direct input) and Appendix A.3 (LivePortrait-generated self-reenactment video as intermediate input).", "reasoning": "The model's performance is contingent on LivePortrait's ability to preserve identity while transferring motion, creating a structural asymmetry where the method relies on an external tool it is also compared against.", "judgment": "Uncertainty-preserving critique regarding fairness; the comparison 'may not be fair' due to circular advantage potential.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8184852600097656, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7234254479408264}, {"unit_index": 1, "inspected_object": "Ablation Variant Metric Validity", "observation": "The 'Full Reference Image Input' variant using unmasked images achieves better quantitative scores than the paper's masked design choice.", "reasoning": "If full-input scores best, either the masking strategy is unjustified or the metrics fail to capture the claimed objectives. The reviewer cites internal inconsistencies (Table 2 vs Fig 5) and external comparisons (Fig 10) to triangulate that the metrics are suspect rather than the design.", "judgment": "Doubt cast on the reliability of the entire quantitative evaluation framework.", "valence": "negative", "suggested_improvement": null, "support_status": "mixed", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8412682414054871, "reasoning_key": "construct_validity", "reasoning_sim": 0.8555116653442383}, {"unit_index": 2, "inspected_object": "Two-Stage Baseline Construction", "observation": "The baseline combines image editing and video animation in a two-stage pipeline.", "reasoning": "Two-stage pipelines inherently suffer from error accumulation and domain mismatch, making it unsurprising for an end-to-end model to outperform them. Such a baseline only demonstrates that end-to-end training beats naive composition, which is not a novel finding.", "judgment": "The comparison is not a solid contribution because it lacks informative failure modes or plausible competitive baselines.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8467517495155334, "reasoning_key": "design_justification", "reasoning_sim": 0.8569026589393616}, {"unit_index": 3, "inspected_object": "Dual ReferenceNet Architectural Novelty", "observation": "The architecture maps to existing methods: ARNet to IP-Adapter and IdentityNet to InstantID, appearing as a parallel combination.", "reasoning": "Naming conventions should reflect architectural lineage; calling known adapters by new names obscures the fact that the contribution is compositional rather than fundamental invention. The comparison is functional, noting similar mechanisms (CLIP encoding, ArcFace) without engaging deeper implementation differences.", "judgment": "The architecture is a 'parallel combination' of existing adapters, rendering novelty claims hollow and constituting incremental engineering.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8075540065765381, "reasoning_key": "design_justification", "reasoning_sim": 0.7878544330596924}, {"unit_index": 4, "inspected_object": "Ethics Compliance", "observation": "Reviewer flags bias, privacy, and responsible research with a single sentence stating 'Portrait generation may include bias.'", "reasoning": "Minimal compliance move checking boxes without elaboration, specific examples, or reference to the paper's data/evaluation.", "judgment": "Ethics concerns are flagged but remain underdeveloped and secondary to technical critiques.", "valence": "conditional", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.6716930270195007, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.6990306377410889}]}, {"review_id": "07OPsqFF3b", "paper_id": "tAHDPnuYr5", "paper_title": "Done Is Better than Perfect: Unlocking Efficient Reasoning by Structured Multi-Turn Decomposition", "decision": "Reject", "summary": "The reviewer conducts a methodological audit focusing on reproducibility and baseline comprehensiveness, expressing skepticism about empirical soundness due to missing hyperparameters and narrow comparisons, while simultaneously acknowledging the method's conceptual clarity and potential value.", "units": [{"unit_index": 0, "inspected_object": "Experimental reproducibility details (hyperparameter configurations and repeated experiments)", "observation": "The paper fails to provide specific hyperparameter configurations or information on whether repeated experiments were performed.", "reasoning": "Reproducibility is treated as a necessary condition for soundness; without these details, other researchers cannot replicate the study's results and verify its conclusions, undermining confidence in the reported empirical claims.", "judgment": "The method's empirical grounding is insufficiently rigorous to support the soundness of the findings due to missing verification infrastructure.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8790960311889648, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8569604754447937}, {"unit_index": 1, "inspected_object": "Baseline selection diversity", "observation": "Most baselines fall into the category of methods controlling early stopping via token budget constraints, while length penalty rewards, reasoning path pruning, and reasoning token simplification are absent.", "reasoning": "Efficiency gains are only meaningful if demonstrated against the full space of plausible alternatives; the current narrow baseline set restricts a comprehensive assessment of the proposed method's competitiveness and may mask whether improvements are due to the framework or merely different from the chosen subset.", "judgment": "The claim of competitiveness is not fully substantiated because the experimental comparison does not cover the broader landscape of token-reduction techniques.", "valence": "negative", "suggested_improvement": "Provide a detailed discussion of the advantages and drawbacks of each major category of token-reduction methods to situate the proposed approach.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.857191264629364, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8234033584594727}, {"unit_index": 2, "inspected_object": "Method clarity and conceptual simplicity", "observation": "The paper presents a concise method with clear writing that allows readers to easily follow the content and grasp key ideas.", "reasoning": "High communicative quality and conceptual accessibility indicate genuine value in the idea of multi-turn decomposition, independent of the current empirical execution gaps.", "judgment": "The work demonstrates significant conceptual merit and clarity, warranting interest despite empirical shortcomings.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.819763720035553, "reasoning_key": "merit_recognition", "reasoning_sim": 0.811317503452301}, {"unit_index": 3, "inspected_object": "Generalization across model families", "observation": "The method was evaluated primarily on R1-Distill models, raising questions about whether the observed reductions are artifacts of this specific family.", "reasoning": "A method's value is partly determined by its applicability across architectures; testing transfer to other model families (e.g., QwQ) is required to establish robustness and generality beyond the training distribution.", "judgment": "The generalizability of the efficiency gains remains uncertain pending evidence of performance on distinct model architectures.", "valence": "conditional", "suggested_improvement": "Apply the method to other model families to demonstrate robustness.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.830951452255249, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8195095062255859}, {"unit_index": 4, "inspected_object": "Reward design mechanism ($R_{unit}$)", "observation": "The reviewer queries the sign of $R_{unit}$ values during Compliance, noting the current structure likely yields zero or negative values for non-compliant outputs.", "reasoning": "Understanding why a reward works mechanistically is as important as demonstrating that it works; clarifying how the reward signal influences model behavior provides deeper insight into the method's efficacy.", "judgment": "The mechanistic basis for the reward's effectiveness requires clarification to fully understand the driver of performance.", "valence": "conditional", "suggested_improvement": "Clarify the role and sign of $R_{unit}$ values in influencing model behavior.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8172381520271301, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7185561656951904}]}, {"review_id": "0B398Gtb0K", "paper_id": "OIkXZkZ1kX", "paper_title": "Curie: Toward Rigorous and Automated Computer Science Experimentation with AI Agents", "decision": null, "summary": "The reviewer performs a standards-enforcement critique, focusing primarily on methodological gaps in evaluation (benchmark validity, model coverage, validation details) rather than technical flaws in the proposed framework. The logic moves from observations of missing or weak empirical evidence to negative judgments about the paper's credibility and contribution value.", "units": [{"unit_index": 0, "inspected_object": "Benchmark construction (46-question custom benchmark)", "observation": "The authors created a new set of 46 questions rather than using existing benchmarks, and the sample size is small.", "reasoning": "Novel benchmarks are suspect without evidence of validity comparable to established ones; comparability across research groups requires shared reference points, and 46 questions may lack sufficient statistical power for credibility.", "judgment": "Weak evaluation due to lack of comparability and insufficient scale.", "valence": "negative", "suggested_improvement": "Conduct evaluations on standard/prominent benchmarks that other developers have tested.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.7768674492835999, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8303771615028381}, {"unit_index": 1, "inspected_object": "Model coverage in evaluation", "observation": "Only GPT-4o was tested as the underlying model.", "reasoning": "Agent scaffolds are highly sensitive to the choice of model; demonstrating robustness across model classes is necessary to prove the scaffold's generalizability rather than it being an artifact of a specific model's strengths.", "judgment": "Incomplete picture of the scaffold's utility.", "valence": "negative", "suggested_improvement": "Test the scaffold across multiple model classes.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8371540307998657, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8067485690116882}, {"unit_index": 2, "inspected_object": "LLM-as-judge validation details", "observation": "The reviewer could not find details regarding manual validation of the LLM judge in the cited appendices.", "reasoning": "Without specified experimental setup and validation metrics, results cannot be trusted or replicated; absence of detail suggests potential lack of rigor in automated evaluation methods.", "judgment": "Unverifiable claims regarding evaluation accuracy.", "valence": "negative", "suggested_improvement": "Provide more details on the LLM-as-judge manual evaluation.", "support_status": "mixed", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7779043912887573, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8867977857589722}, {"unit_index": 3, "inspected_object": "Experimental setup clarity (scheduler/agent handling)", "observation": "Unclear phrasing about agents 'may be busy handling other partitions' raises questions about system complexity and design rationale.", "reasoning": "If the experimental setup is not fully specified, results cannot be trusted; ambiguous design choices suggest potential unnecessary complexity or lack of clear rationale.", "judgment": "Insufficient clarity to assess experimental validity.", "valence": "negative", "suggested_improvement": "Clarify the scheduler logic and provide complete experimental details.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.7886127233505249, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8595985770225525}, {"unit_index": 4, "inspected_object": "Development process transparency", "observation": "Missing information on train/test splits, leakage prevention, and iterative development of the scaffold.", "reasoning": "Complete experimental details including development process are required to ensure reproducibility and rule out data leakage.", "judgment": "Lack of trustworthiness due to missing procedural details.", "valence": "negative", "suggested_improvement": "Provide details on train/test splits and leakage prevention.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7878326177597046, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.814603328704834}, {"unit_index": 5, "inspected_object": "Framing claims ('first step towards rigorous... experimentation')", "observation": "The paper claims to be the first step in a well-populated area.", "reasoning": "Contributions must be situated within existing literature; claiming novelty in a field with significant prior work reflects poor positioning and overclaiming.", "judgment": "Overclaiming and lack of proper contextualization.", "valence": "negative", "suggested_improvement": "Frame contributions modestly within the existing literature.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8738440871238708, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8383992314338684}, {"unit_index": 6, "inspected_object": "Component ablation study", "observation": "The framework components (intra-agent rigor, inter-agent rigor, experiment knowledge module) are acknowledged as interesting but not analyzed via ablation.", "reasoning": "The contribution's value is only as strong as its demonstrated component effects; without ablation, it is unclear which parts drive the reported improvements.", "judgment": "Limited demonstration of specific contribution value.", "valence": "conditional", "suggested_improvement": "Show the effect of adding each of the components via ablation studies.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.8066991567611694, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7258381247520447}]}, {"review_id": "0CRrnBm7FV", "paper_id": "7TlCUD2tQI", "paper_title": "Augmenting Industrial Maintenance with LLMs: A Benchmark, Analysis, and Generalization Study", "decision": "Reject", "summary": "The reviewer evaluates the paper primarily as a community resource, judging that it fails to sufficiently differentiate itself from prior benchmarks, overclaims generalization given its narrow dataset scope, and lacks immediate reproducibility due to code issues.", "units": [{"unit_index": 0, "inspected_object": "Positioning within the benchmark literature (comparison with CAMB and wind-turbine maintenance logs benchmarks)", "observation": "The paper presents DiagnosticIQ in isolation without explicit comparative positioning against known similar benchmarks.", "reasoning": "Novelty in a benchmark paper is relational; it must be established against closest competitors through contrast on dimensions such as task types, domain scope, modality coverage, and construction process. Without this comparison, it is unclear whether the work is a novel contribution or a re-instantiation of existing ideas.", "judgment": "Insufficient differentiation from prior work limits the assessment of novelty.", "valence": "negative", "suggested_improvement": "Provide a clearer comparison with existing benchmarks (e.g., CAMB, wind-turbine logs) across specified dimensions to highlight novelty.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8150710463523865, "reasoning_key": "design_justification", "reasoning_sim": 0.8334688544273376}, {"unit_index": 1, "inspected_object": "Generalization claims relative to dataset breadth", "observation": "The abstract promises generalization to 'previously unseen assets,' but the dataset covers only 16 asset types.", "reasoning": "A claim of cross-domain transferability requires empirical evidence demonstrating transfer across meaningfully different categories of assets (e.g., rotating vs. static infrastructure). The current limited scope (16 types within a narrow set) does not support the title-level promise of a 'Generalization Study.'", "judgment": "Scope-to-claim mismatch; generalization is not demonstrated by the current evidence.", "valence": "negative", "suggested_improvement": "Broaden the dataset to include diverse asset categories or temper the generalization language to match the empirical basis.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.850390613079071, "reasoning_key": "design_justification", "reasoning_sim": 0.8116665482521057}, {"unit_index": 2, "inspected_object": "Reproducibility of the code artifact", "observation": "An attempt to run the provided implementation failed due to missing configuration files and unclear dependencies.", "reasoning": "For a benchmark paper, the code is a core deliverable and community resource, not just a supplement. Failure to reproduce indicates the benchmark cannot currently function as a standardized tool for others, limiting its infrastructural value.", "judgment": "Reproducibility failure significantly limits the benchmark's utility as a community resource.", "valence": "negative", "suggested_improvement": "Fix configuration issues and clarify dependencies to ensure the implementation can be successfully run by other researchers.", "support_status": "mixed", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.9278472065925598, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7769348621368408}]}, {"review_id": "0CgYtnikrw", "paper_id": "emM6KIsBHl", "paper_title": "OpenMarcie: Dataset for Multimodal Action Recognition in Industrial Environments", "decision": null, "summary": "The reviewer conducts a comparative evaluation arguing that OpenMarcie's multimodal advantages do not compensate for its limited task diversity, non-industrial setting, and lack of empirical benchmarking against Ego-Exo4D. The review demands greater quantitative transparency and direct comparative experiments to justify the dataset's contribution.", "units": [{"unit_index": 0, "inspected_object": "Dataset task diversity and scalability", "observation": "The dataset includes only two scenarios, leading to uncertainty about whether the data collection approach is scalable.", "reasoning": "The reviewer treats the small number of scenarios as a proxy for the generalizability of the collection pipeline; without evidence of scalability across more settings, the method's utility for broader manufacturing contexts is questioned.", "judgment": "Limited contribution due to insufficient demonstration of methodology reproducibility and scope.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8396309018135071, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8103170394897461}, {"unit_index": 1, "inspected_object": "Incremental value relative to existing datasets (Ego4D/Ego-Exo4D)", "observation": "It is unclear if the dataset offers unique capabilities not available in existing resources like Ego4D, despite having more sensing modalities.", "reasoning": "The reviewer performs a cost-benefit calculation: the advantage of additional modalities is weighed against the deficit in task diversity. The conclusion is that the modalities do not clearly compensate for the lack of novelty compared to prior work.", "judgment": "The dataset's incremental value is unproven and likely insufficient to justify its existence as a new resource.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.781446635723114, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8344156742095947}, {"unit_index": 2, "inspected_object": "Authenticity of the data collection setting", "observation": "The data was recorded in a bicycle assembly setting that resembles hobbyist repair rather than an industrial manufacturing environment.", "reasoning": "Because the paper claims relevance for manufacturing but uses a non-industrial setting, the validity of the dataset's claim to capture 'manufacturing' data is weakened, especially when similar tasks exist in other datasets.", "judgment": "The dataset fails to authentically represent the claimed domain, undermining its specific contribution.", "valence": "negative", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8028192520141602, "reasoning_key": "novelty_standard", "reasoning_sim": 0.77792888879776}, {"unit_index": 3, "inspected_object": "Presentation of comparative analysis with Ego-Exo4D", "observation": "The discussion comparing OpenMarcie to Ego-Exo4D is limited to Table 2 and Appendix A, which the reviewer finds too brief.", "reasoning": "Given the significant overlap in tasks and the importance of distinguishing new contributions, the reviewer expects a proactive, in-depth discussion in the main text rather than relegating it to supplementary materials.", "judgment": "Inadequate presentation of critical comparative context reduces the paper's clarity and persuasive power.", "valence": "negative", "suggested_improvement": "Provide a more in-depth discussion of the advantages of OpenMarcie over Ego-Exo4D in the main text.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8309819102287292, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7789521813392639}, {"unit_index": 4, "inspected_object": "Quantitative comparison metrics (Table 1)", "observation": "Table 1 reports only percentages, obscuring the absolute volume of relevant data; the reviewer calculates that Ego-Exo4D has nearly twice the hours of industrial-relevant data.", "reasoning": "The reviewer holds that absolute quantities (total hours) are more informative than relative proportions for assessing competitiveness. By converting percentages, the reviewer demonstrates that OpenMarcie is significantly smaller in relevant scope than the competitor.", "judgment": "The current reporting metric underplays the disparity in scale between OpenMarcie and Ego-Exo4D.", "valence": "negative", "suggested_improvement": "Include the total number of hours of data available relevant for manufacturing in all datasets in Table 1.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8778795599937439, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7656562924385071}, {"unit_index": 5, "inspected_object": "Benchmarking strategy", "observation": "The paper lacks empirical evaluation comparing model performance trained on Ego-Exo4D versus OpenMarcie.", "reasoning": "A new dataset contribution must be validated by demonstrating what it enables that prior datasets do not. Without cross-dataset benchmarking, the dataset's practical utility and superiority remain unproven.", "judgment": "The dataset's contribution is not empirically substantiated relative to existing benchmarks.", "valence": "negative", "suggested_improvement": "Perform an evaluation of the performance obtained on different tasks when a model is trained on Ego-Exo4D versus OpenMarcie.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.82796311378479, "reasoning_key": "design_justification", "reasoning_sim": 0.8119177222251892}, {"unit_index": 6, "inspected_object": "Figure 7 readability", "observation": "Figure 7 presents visual data without numerical labels, forcing readers to guess the values.", "reasoning": "Precise, unambiguous quantitative information is required for effective communication. Visuals alone are insufficient when exact numbers are needed for interpretation.", "judgment": "The figure is not very informative due to lack of explicit numerical data.", "valence": "negative", "suggested_improvement": "Add percentages in numbers to Figure 7.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8512464165687561, "reasoning_key": "design_justification", "reasoning_sim": 0.7908422946929932}]}, {"review_id": "0DC3fYLTuo", "paper_id": "eMnRH5Yes3", "paper_title": "Interpretable Vision Tasks via Vision Logic Model Integrating Visual Reasoning and Textual Explanation", "decision": "Reject", "summary": "The reviewer conducts an epistemic audit of the paper's knowledge claims, finding that the 'logic-based' framing is unsupported by the architecture, the novelty is unproven against modern baselines, faithfulness lacks mechanistic evidence, and reproducibility details are insufficient.", "units": [{"unit_index": 0, "inspected_object": "The method's self-description as 'logic-based' versus its actual architecture (prototype attention + gated refinement).", "observation": "The architecture appears closer to prototype attention and gated refinement rather than symbolic or rule-grounded logic; the reviewer questions if prototypes encode logical structure or latent embeddings.", "reasoning": "A method labeled 'logic-based' requires either symbolic reasoning or a formal account of how components constitute logical structure. The current framing is aspirational rather than technically justified by the architectural description.", "judgment": "The paper's self-characterization is overreaching and the term 'logic' is not technically justified by the evidence provided.", "valence": "negative", "suggested_improvement": "Clarify prototype semantics and initialization to test whether prototypes correspond to human-interpretable concepts or are merely latent embeddings.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8132021427154541, "reasoning_key": "design_justification", "reasoning_sim": 0.8005152344703674}, {"unit_index": 1, "inspected_object": "The comparative baseline set for interpretability methods.", "observation": "Comparisons include Grad-CAM, ProtoPNet, and VLM justifications but lack recent intrinsic interpretability methods such as mechanistic ViT explainability, slot-attention, causal representation learning, and visual chain-of-thought models.", "reasoning": "Novelty claims must be established relative to the current state of the art, not just historical baselines. The absence of these recent methods leaves the contribution's novelty unproven and risks making it feel incremental.", "judgment": "The positioning within the research frontier is deficient due to missing modern comparators.", "valence": "negative", "suggested_improvement": "Add recent intrinsic interpretability baselines to demonstrate novelty relative to the current frontier.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8419973850250244, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8445159196853638}, {"unit_index": 2, "inspected_object": "Faithfulness enforcement mechanisms (monotonic confidence and explanation alignment constraints).", "observation": "Causal intervention tests are included but deemed insufficient to prove genuine causal reasoning; potential for artifact learning exists.", "reasoning": "Interpretability claims require mechanistically demonstrated evidence about how the model represents and uses concepts, not just behavioral metrics or limited interventions. Current evidence does not disambiguate between genuine causal reasoning and plausible explanatory artifacts.", "judgment": "The faithfulness claims are unsupported by sufficient mechanistic evidence.", "valence": "negative", "suggested_improvement": "Perform OOD generalization tests, probe for spurious correlations, conduct concept emergence analysis, or provide human-interpretable grounding of prototypes.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8154610991477966, "reasoning_key": "design_justification", "reasoning_sim": 0.8100720643997192}, {"unit_index": 3, "inspected_object": "Reproducibility surface and procedural transparency.", "observation": "Missing details on prototype semantics, human evaluation design, annotation criteria, compute costs, and robustness across seeds/architectures/domains.", "reasoning": "Incomplete procedural information undermines verifiability and suggests ambiguity about robustness. Claims should be demonstrated to generalize across dimensions like seeds and architectures, not just within a single configuration.", "judgment": "Epistemic hygiene is lacking; reproducibility is difficult and robustness is uncertain.", "valence": "negative", "suggested_improvement": "Provide detailed human evaluation design, annotation criteria, compute costs, and evidence of persistence across seeds, architectures, and data domains.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8962427377700806, "reasoning_key": "design_justification", "reasoning_sim": 0.863184928894043}]}, {"review_id": "0DmuDf2cZm", "paper_id": "yHWDILUKdR", "paper_title": "End-to-End QA Construction Pipeline for Continual Pre-training of Large Language Models", "decision": null, "summary": "The reviewer critically interrogates the pipeline's internal validation logic, identifying a central circularity where the same model generates and judges ground truth. They argue this undermines epistemic validity due to the absence of external validation (non-Qwen models or human evaluation). While praising the novelty of the knowledge boundary mining technique and specific empirical findings, the reviewer emphasizes transparency, traceability, and non-circular standards, leaving several methodological questions unresolved.", "units": [{"unit_index": 0, "inspected_object": "Pipeline validation logic and ground truth generation", "observation": "The 'ground truth' for knowledge boundary labeling is generated by the same model (Qwen2.5-72B-Instruct) that later serves as the judge of whether other models' responses are correct.", "reasoning": "This creates a self-referential dependency where the pipeline measures whether smaller models can reproduce what Qwen2.5-72B+RAG generates, rather than identifying true knowledge gaps relative to the world. This conflates reproducibility with knowledge.", "judgment": "The internal validation logic is circular, undermining the epistemic validity of the results.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8161101937294006, "reasoning_key": "design_justification", "reasoning_sim": 0.8036362528800964}, {"unit_index": 1, "inspected_object": "Absence of external validation mechanisms", "observation": "The paper lacks non-Qwen models (e.g., Llama, GPT, Claude) as base models or evaluators, and has no human evaluation of answer correctness.", "reasoning": "Without an external reference point independent of the Qwen family, it is impossible to break the circularity identified in the primary critique. External validation is required to verify that the pipeline identifies genuine knowledge boundaries rather than model-specific artifacts.", "judgment": "The results cannot be trusted as valid measures of knowledge boundaries without external corroboration.", "valence": "negative", "suggested_improvement": "Include non-Qwen models or human evaluation to provide external validation of the pipeline's outputs.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7920215725898743, "reasoning_key": "design_justification", "reasoning_sim": 0.8297615051269531}, {"unit_index": 2, "inspected_object": "Knowledge Boundary Model section clarity", "observation": "Section 3.3 is described as confusing and requiring multiple readings; Figure 1 does not clearly show data flow and dependencies.", "reasoning": "The lack of procedural clarity makes it difficult to trace the data flow and assess the validity of the pipeline's steps, particularly regarding potential hidden dependencies or circularities.", "judgment": "Presentation barriers hinder the verification of the methodological claims.", "valence": "negative", "suggested_improvement": "Provide a clearer pipeline diagram showing data flow and dependencies, and clarify the text in Section 3.3.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.781041145324707, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7880464196205139}, {"unit_index": 3, "inspected_object": "Operational definition of accuracy calculation", "observation": "It is unclear how accuracy is calculated in Section 3.3—specifically whether it is macro-averaged across 30 responses or based on token probabilities.", "reasoning": "The interpretation of the results depends on this operational clarity. Without knowing the exact quantity being computed, the validity of the reported metrics cannot be fully assessed.", "judgment": "Methodological ambiguity exists regarding the computation of key metrics.", "valence": "uncertain", "suggested_improvement": "Clarify whether accuracy is macro-averaged across responses or based on token probabilities.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7515727281570435, "reasoning_key": "construct_validity", "reasoning_sim": 0.8543450832366943}, {"unit_index": 4, "inspected_object": "Knowledge boundary mining technique", "observation": "The use of temperature 0.8, 30 responses, and classifier training for knowledge boundary mining is identified as a novel contribution.", "reasoning": "This represents a methodological innovation in how queries are selected to push model boundaries, distinct from standard benchmark improvements.", "judgment": "The specific technique employed shows genuine novelty and methodological innovation.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7453883290290833, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7299233078956604}, {"unit_index": 5, "inspected_object": "Figure 3 empirical finding", "observation": "Figure 3 suggests that Qwen2-7B and Qwen2.5 models appear to have different training data.", "reasoning": "This empirical observation provides interesting insight into the training history of the models used, independent of the pipeline's validation logic.", "judgment": "The finding is interesting and contributes to understanding the model landscape.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8111487627029419, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7434224486351013}]}, {"review_id": "0FJVxCvgWQ", "paper_id": "uljG4Pbn3i", "paper_title": "Benchmarking Mitigations For Covert Misuse", "decision": "Reject", "summary": "The reviewer acts as a benchmark-construction critic, acknowledging the paper's timely contribution but systematically probing the boundaries of its claims regarding domain coverage, reproducibility via data release, adversarial robustness of the defense, and generalizability across model architectures.", "units": [{"unit_index": 0, "inspected_object": "Scope of evaluation domains (biosecurity and cybersecurity only)", "observation": "The paper tests only two domains, excluding misinformation and social manipulation.", "reasoning": "The reviewer applies a standard that broad claims about 'covert misuse' require representative coverage across relevant threat spaces; limiting to two domains raises concerns about ecological validity and whether tasks reflect real-world covert misuse rather than artificial examples tailored to refusal behavior.", "judgment": "Uncertainty regarding the generalizability and construct validity of the benchmark's findings beyond the tested domains.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7594739198684692, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8007128834724426}, {"unit_index": 1, "inspected_object": "Dataset availability and non-release", "observation": "The dataset is not released, preventing external verification.", "reasoning": "The reviewer argues that a benchmark's value as a community contribution is contingent on its availability; without access, it is impossible to verify if the 'difficult, refused, and answerable' criteria are genuinely met or if the generation pipeline produces contrived examples.", "judgment": "Reduced utility and trust in the benchmark's construction due to lack of reproducibility.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7617778778076172, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8170421123504639}, {"unit_index": 2, "inspected_object": "Attacker model and stateful defense robustness", "observation": "The current attacker model may be too narrow, lacking adaptive evasion tactics like cross-account evasion or mimicry.", "reasoning": "The reviewer applies a standard that defenses must be tested against the strongest plausible adversary; probing for adaptive attacks questions whether the proposed 'stateful defense' would hold up under realistic adversarial pressure or if it is only effective against naive attackers.", "judgment": "Uncertainty about the practical robustness and meaningfulness of the defense mechanism.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7731918096542358, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7917789816856384}, {"unit_index": 3, "inspected_object": "Model generalizability ('strong vs. weak' distinction)", "observation": "The paper relies on a distinction between strong (safety-trained) and weak (unaligned) models without testing across different architectures.", "reasoning": "The reviewer questions if the decomposition attack's effectiveness is an artifact of specific model families; if results vary across architectures, the central finding that decomposition is a 'misuse enabler' might be contingent rather than robust.", "judgment": "Skepticism about the architecture-agnostic nature of the reported safety findings.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8447866439819336, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8338173031806946}]}, {"review_id": "0GFtoZk1Cr", "paper_id": "6eqyiE3Uu1", "paper_title": "Scene-R1: Video-Grounded Large Language Models  for 3D Scene Reasoning without 3D Annotations", "decision": null, "summary": "The reviewer conducts a deconstructive reading focused on the epistemological status of the paper's supervision claims. They argue that 2D masks derived from 3D ground truth are information-theoretically equivalent to 3D annotations, rendering the 'no 3D annotations' claim false. This leads to critiques of unfair baseline comparisons and internal contradictions regarding offline reconstruction.", "units": [{"unit_index": 0, "inspected_object": "Training pipeline supervision requirements and annotation efficiency claim", "observation": "The reviewer identifies that the paper's two-stage grounding pipeline requires mask supervision for the relevant object across all video frames, derived from projecting ground truth 3D segmentation masks to 2D in ScanNet.", "reasoning": "The reviewer asserts an equivalence claim: for posed RGB-D video, 3D mask annotations and 2D video masks are interconvertible via projection/unprojection. Therefore, training with 2D masks derived from 3D ground truth is functionally identical to training with 3D annotations. This shifts the definition of annotation cost from format (point-wise 3D instance labels) to information content, leading to the judgment that the paper's 'annotation-efficient' claim is wrong because the cost/effort of obtaining either is the same.", "judgment": "The core claim of 'no 3D annotations' is misleading and factually incorrect regarding the supervision burden.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8254836201667786, "reasoning_key": "design_justification", "reasoning_sim": 0.7929812669754028}, {"unit_index": 1, "inspected_object": "Comparison table grouping (Table 1 baselines)", "observation": "The reviewer finds the grouping of baselines into 'free from 3D instance or annotation supervision' versus 'fully supervised' categories misleading on both ends.", "reasoning": "The reviewer argues that methods like vlm-grounder, open-scene, and lerf use no supervision at all, while the proposed method uses both grounding and mask supervision. The implicit norm is that comparison groups should be matched on total supervision used, not just on the specific format of annotation avoided. Grouping a method with significant supervision burden alongside zero-supervision methods creates an unfair comparison.", "judgment": "The experimental framing and baseline categorization are misleading and structurally flawed.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8145147562026978, "reasoning_key": "design_justification", "reasoning_sim": 0.8304643034934998}, {"unit_index": 2, "inspected_object": "Introduction claim vs. Method Section 4.3 consistency", "observation": "The introduction claims the method enables 'end-to-end reasoning directly on video streams, bypassing the need for offline 3D scene reconstruction,' but Section 4.3 uses reconstructed point clouds.", "reasoning": "This is an internal-consistency check testing whether the rhetorical framing matches the actual pipeline. The use of reconstructed point clouds in the evaluation phase contradicts the explicit claim of bypassing offline 3D scene reconstruction.", "judgment": "There is a direct textual contradiction between the introduction's claims and the method's implementation.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7899815440177917, "reasoning_key": "design_justification", "reasoning_sim": 0.8153553009033203}, {"unit_index": 3, "inspected_object": "Baseline currency and SOTA references", "observation": "The reviewer notes that the 'fully supervised' baselines are significantly old and directs authors to UniVLG as current SOTA.", "reasoning": "Staleness of baselines is a legitimate reason to question comparison validity. However, the reviewer does not verify if UniVLG applies to the same task setting or explain its differences, making this critique underdeveloped compared to the annotation equivalence argument.", "judgment": "The comparison landscape includes outdated baselines, weakening the empirical validation.", "valence": "negative", "suggested_improvement": "Check UniVLG (arXiv:2503.10745) Table 1 for recent baselines.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.7818360328674316, "reasoning_key": "design_justification", "reasoning_sim": 0.8630483746528625}]}, {"review_id": "0GaMO98X1i", "paper_id": "nydRcDqjKL", "paper_title": "UML-CoT: Structured Reasoning and Planning with Unified Modeling Language for Robotic Room Cleaning", "decision": null, "summary": "The reviewer systematically audits the paper's evidentiary architecture, focusing on the gap between strong claims of executability and the absence of physical or human-grounded validation. Key criticisms target the uncalibrated reward signal, the unvalidated similarity metric, the unspecified generalization protocol, the lack of standard benchmark testing, and the unjustified choice of UML over simpler formalisms.", "units": [{"unit_index": 0, "inspected_object": "Reward function design relying on final-plan embedding similarity and format bonus without intermediate supervision", "observation": "Rewards are computed only from the final plan; there is no direct supervision for intermediate class-diagram quality.", "reasoning": "The absence of intermediate supervision invites reward hacking or spurious similarity rather than faithful stepwise reasoning, failing to enforce the internal coherence required by the claimed formalism.", "judgment": "The training procedure fails to enforce structural validity or cross-diagram consistency constraints implied by the method's claims.", "valence": "negative", "suggested_improvement": "Provide mechanisms ensuring class diagrams are accurate and consistent with activity diagrams, such as structural validity or cross-diagram consistency constraints during training.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8353552222251892, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7858128547668457}, {"unit_index": 1, "inspected_object": "Evaluation metric (all-MiniLM-L12-v2 cosine similarity) validity", "observation": "The metric uses partitioned plan sections with greedy matching but lacks human preference or simulator/hardware execution studies to correlate higher similarity with actual cleaning success or safety.", "reasoning": "A proxy metric's validity is unestablished if it is not calibrated against a ground-truth outcome measure; text similarity does not inherently prove physical success or safety.", "judgment": "The metric's ability to reflect true task performance is unvalidated, rendering the results difficult to interpret regarding actual executability.", "valence": "negative", "suggested_improvement": "Validate the similarity metric with human ratings and execution-based metrics in established benchmarks like ALFRED or TEACh to demonstrate correlation with success rates.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8352094888687134, "reasoning_key": "construct_validity", "reasoning_sim": 0.8211269974708557}, {"unit_index": 2, "inspected_object": "Cross-task generalization protocol for cooking and painting", "observation": "Generalization results are reported but under-specified; key details including training data, zero-shot vs fine-tuned settings, annotation sources, schema alignment, and overlap controls are missing.", "reasoning": "Without these details, it is hard to interpret causally whether observed generalization is attributable to the UML-CoT method or to confounds in the data or protocol.", "judgment": "The causal interpretability of the generalization claims is compromised due to insufficient experimental specification.", "valence": "negative", "suggested_improvement": "Clarify the generalization protocol by detailing training data, settings, annotation sources, schema alignment methods, and overlap controls.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.8201093077659607, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8205005526542664}, {"unit_index": 3, "inspected_object": "Absence of established embodied benchmarks (ALFRED, TEACh, Habitat Rearrangement)", "observation": "The paper does not use standard execution-based benchmarks despite claiming 'executable plans' and 'planning quality'.", "reasoning": "If the paper makes claims about executability in embodied tasks, field-standard execution-based benchmarks are the appropriate testbed; their absence constitutes a prima facie gap in external validity.", "judgment": "The claim of executability is unsupported by evidence from the domain's standard validation environments.", "valence": "negative", "suggested_improvement": "Test on established benchmarks like ALFRED, TEACh, or Habitat Rearrangement, or explicitly explain any interface incompatibilities that prevent their use.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.806671142578125, "reasoning_key": "fair_comparison", "reasoning_sim": 0.779503583908081}, {"unit_index": 4, "inspected_object": "Comparative justification for UML over alternative formalisms", "observation": "The case for UML is purely conceptual; there are no controlled ablations against simpler structured formats like typed JSON schemas, DSL-based plans, or PDDL.", "reasoning": "Representational choices are empirical claims requiring experimental support; without ablation, it is unclear if UML's heavy symbolic overhead provides benefits over simpler alternatives with similar interpretability.", "judgment": "The superiority of UML is not empirically justified, raising concerns about unnecessary complexity.", "valence": "negative", "suggested_improvement": "Perform controlled ablations comparing UML against simpler structured prompts (e.g., JSON or DSL-based plans) to justify the choice of formalism.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7819998264312744, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7949399948120117}, {"unit_index": 5, "inspected_object": "Perception-grounding pipeline for UML generation", "observation": "There is no hybrid grounding pipeline leveraging scene graph extraction to populate UML class diagrams from perception.", "reasoning": "A perception-first pipeline would likely reduce hallucination and improve consistency; the reviewer seeks the authors' intuition on why this was not pursued.", "judgment": "The lack of perceptual grounding is a notable omission that may impact reliability, though the authors' reasoning is requested.", "valence": "conditional", "suggested_improvement": "Explain the reasoning behind omitting a perception-grounding pipeline, or consider incorporating scene graph extraction to improve consistency.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8009263277053833, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7579895853996277}, {"unit_index": 6, "inspected_object": "LLM-generated dataset labels and evaluation conversion", "observation": "Dataset labels are largely LLM-generated (GPT-4o/DeepSeek-R1), and evaluation converts textual outputs to UML via GPT-4o.", "reasoning": "This creates risks of label noise, conversion bias, and metric circularity, where the evaluation metric may be biased by the same model used for generation/conversion.", "judgment": "The reliance on LLMs for both data creation and evaluation introduces potential circularity and bias issues.", "valence": "negative", "suggested_improvement": "Address concerns about label noise, conversion bias, and metric circularity arising from LLM-as-annotator practices.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8498347401618958, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.765962541103363}, {"unit_index": 7, "inspected_object": "Sensitivity of evaluation to predefined partitions", "observation": "The review asks about sensitivity to predefined partitions ('Main Messy Areas', 'Priority', 'Steps') and whether they are domain-neutral enough for cooking/painting.", "reasoning": "If partitions are biased or not domain-neutral, the evaluation results may not generalize fairly across tasks.", "judgment": "Uncertainty remains regarding whether the evaluation framework is unbiased across different domains.", "valence": "uncertain", "suggested_improvement": "Clarify if the analysis of partition sensitivity is present in the paper or provide it to ensure domain neutrality.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8587788939476013, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8161376714706421}]}, {"review_id": "0H1IZuDcPW", "paper_id": "CwAS9PZawD", "paper_title": "Distilled Protein Backbone Generation", "decision": "Reject", "summary": "The reviewer positively evaluates the paper's documentation of negative results and practical speedup but critically challenges the scope of validation by demanding analysis of specific failure modes, conditional task performance, and theoretical limits of generalization and distributional fidelity.", "units": [{"unit_index": 0, "inspected_object": "Documentation of negative results and failed adaptation attempts", "observation": "The authors explicitly demonstrate all things that did not work as well as the modifications necessary to make the method succeed, rather than presenting only the final working algorithm.", "reasoning": "This deviates from a common but unhelpful practice in the field where papers hide failed attempts; it provides a useful practical resource for practitioners who will face similar adaptation problems by mapping the failure landscape.", "judgment": "Positive evaluation of the paper's value as a diagnostic and prescriptive case study in engineering adaptation.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.840568482875824, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7756603360176086}, {"unit_index": 1, "inspected_object": "Empirical performance metrics (speedup and preservation)", "observation": "The method achieves a 20x speedup while preserving performance compared to the baseline.", "reasoning": "While the result is impressive and useful in practical applications, it represents an engineering contribution rather than a paradigm-shifting theoretical breakthrough, limiting its conceptual depth.", "judgment": "Solid instrumental value with limited conceptual novelty.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.879604697227478, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7729575037956238}, {"unit_index": 2, "inspected_object": "Analysis of fold-class performance deficits", "observation": "The paper reports lower fold class metrics for the distilled model compared to the pretrained teacher but lacks a detailed analysis of which specific folds are underrepresented or what the systematic failure mode is.", "reasoning": "Aggregate mean shifts hide meaningful distributional failures; a mechanistic understanding of why certain folds are missed is more informative than aggregate metrics and suggests potential structural issues with the distillation process that could be corrected.", "judgment": "Insufficient granular analysis of failure modes limits the interpretability of the performance drop.", "valence": "negative", "suggested_improvement": "Provide a more detailed analysis of the failure model, specifically characterizing which folds are underrepresented and identifying systematic deviations from the teacher.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8418151140213013, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8013827800750732}, {"unit_index": 3, "inspected_object": "Scope of evaluation (unconditional vs. conditional generation)", "observation": "The evaluations are limited to unconditional generation, whereas practical protein engineering tasks often require complex conditional generation such as motif scaffolding.", "reasoning": "The claimed practical utility of enabling large-scale in silico design requires demonstration on the conditional tasks that motivate such applications; assuming transferability from unconditional to conditional settings without evidence is a gap in validating the core motivation.", "judgment": "The evidence does not fully support the claimed practical utility due to the lack of conditional task validation.", "valence": "negative", "suggested_improvement": "Evaluate the method on conditional generation tasks, such as motif scaffolding, to validate its utility for practical protein engineering applications.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7862118482589722, "reasoning_key": "cost_benefit", "reasoning_sim": 0.814834475517273}, {"unit_index": 4, "inspected_object": "Generalization to all-atom structure generation models", "observation": "The reviewer questions whether the distillation insights apply to all-atom models like La-Proteina or Protpardelle, which involve more complex low-noise schedules.", "reasoning": "If the key insights are specific to backbone-only settings or simple noise schedules, their broader applicability to state-of-the-art all-atom methods is uncertain, potentially limiting the generalizability of the contribution.", "judgment": "Uncertainty regarding the architectural generality of the proposed distillation scheme beyond the specific setting tested.", "valence": "conditional", "suggested_improvement": "Discuss or test the generalization of the approach to all-atom structure generation models with more complex noise scheduling.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.7701373100280762, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8175246715545654}, {"unit_index": 5, "inspected_object": "One-step generation limitations", "observation": "The reviewer asks whether the observed degradation in one-step generation is a fundamental limitation of the underlying generative process or a contingent artifact of the specific method.", "reasoning": "Distinguishing between inherent theoretical limits and implementation-specific artifacts is crucial for understanding the true potential of the distillation approach and guiding future improvements.", "judgment": "Need for theoretical clarification on the boundaries of one-step feasibility.", "valence": "uncertain", "suggested_improvement": "Engage with the theoretical limits of the approach to clarify whether one-step degradation is fundamental or avoidable.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8275070786476135, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7391644716262817}, {"unit_index": 6, "inspected_object": "Temperature-based sampling fidelity and distribution coverage", "observation": "The reviewer notes that distilling at a specific temperature may restrict the model to a subset of the data distribution with better scores but potentially reduced diversity, asking if the full distribution can be preserved.", "reasoning": "If the distilled model permanently narrows the output distribution, it cannot be used for exploratory sampling, and reported diversity metrics may be misleading; a faithful compression should retain the ability to sample broadly.", "judgment": "Concern that the distillation process compromises the model's ability to represent the full data distribution.", "valence": "negative", "suggested_improvement": "Investigate whether the distilled model has lost the ability to sample from the full data distribution and explore methods to preserve or recover this capability.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8213894963264465, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7498511672019958}]}, {"review_id": "0HKGcJweki", "paper_id": "SLBcDQvBbh", "paper_title": "Quantization Meets Sparsification for Faster Image Generation", "decision": "Reject", "summary": "The reviewer evaluates the paper primarily through the lens of systems research, focusing on the adequacy of experimental baselines and the validity of generalization claims. They argue that without comparisons to hardware-native sparse implementations and tests on diverse architectures, the claimed novelty and broad applicability remain unproven. The review emphasizes reproducibility and rigorous metric reporting as prerequisites for validating the engineering contribution.", "units": [{"unit_index": 0, "inspected_object": "Experimental baseline selection and comparative evidence for the proposed sparse GEMM kernel", "observation": "The experiments compare the proposed method only against a dense FP8 baseline (qGEMM) and ablations, omitting NVIDIA’s Sparse Tensor Core implementation.", "reasoning": "Without comparison to hardware-native sparse capabilities or strong external baselines, it is impossible to distinguish genuine incremental novelty from engineering repackaging of existing functionality.", "judgment": "The paper fails to demonstrate the true benefit and novelty of the combined quantization-sparsification approach due to insufficient comparative evidence.", "valence": "negative", "suggested_improvement": "Include NVIDIA’s Sparse Tensor Core implementation as a baseline to assess if the proposed kernel offers advantages over native hardware sparse capabilities.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8241463899612427, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7825961709022522}, {"unit_index": 1, "inspected_object": "Generality claims regarding architecture agnosticism and model coverage", "observation": "The paper claims QuaSpa is 'architecture agnostic' and promises 'faster image generation,' yet experiments are limited exclusively to DiTs.", "reasoning": "A claim of broad applicability requires empirical substantiation across multiple model classes; testing only one class creates a mismatch between the stated scope and the provided evidence.", "judgment": "The generality claim is unsubstantiated and the results may not generalize beyond the specific models tested.", "valence": "negative", "suggested_improvement": "Test the method on another image generation model class to substantiate the architecture-agnostic claim.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8091333508491516, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8670157790184021}, {"unit_index": 2, "inspected_object": "Implementation details and reproducibility specifications", "observation": "Critical details are missing: pruning methodology, ambiguous definitions of qGEMM/BF16 sparse/FP8, and MS-COCO evaluation specifics (image count, captions).", "reasoning": "Precise specification is required for an engineering contribution to be useful and reproducible; ambiguity in baselines and quality metrics prevents independent verification of speedup and quality claims.", "judgment": "The lack of detailed reporting undermines the paper's value as a reproducible system and obscures the interpretation of reported metrics.", "valence": "negative", "suggested_improvement": "Provide complete details on pruning methodology, clarify baseline definitions, and specify MS-COCO evaluation parameters.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.929722249507904, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.846784234046936}, {"unit_index": 3, "inspected_object": "Generation quality assessment methodology", "observation": "The current evaluation lacks FID scores, which are standard for rigorous quality assessment in image generation literature.", "reasoning": "Standard metrics like FID allow for meaningful comparison with other methods; their absence limits the ability to assess whether quality preservation claims hold up against established benchmarks.", "judgment": "The quality assessment is currently insufficiently rigorous compared to field standards.", "valence": "negative", "suggested_improvement": "Report FID scores in Section 4.3.2 to provide a more rigorous and comparable assessment of generation quality.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7385231256484985, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7759332060813904}, {"unit_index": 4, "inspected_object": "Positioning relative to cache-based quantization methods (Q&C)", "observation": "The review notes Q&C is a cache-based method rather than a sparse kernel, yet suggests comparing against it.", "reasoning": "Comparing against methods that achieve similar end-to-end goals through different mechanisms helps position the proposed approach within the broader efficient-generation landscape.", "judgment": "The paper needs to clarify its competitive positioning against non-sparse acceleration techniques.", "valence": "conditional", "suggested_improvement": "Compare performance with Q&C to evaluate competitiveness against cache-based acceleration methods.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8388134241104126, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7600020170211792}]}, {"review_id": "0HN3Kckku1", "paper_id": "2kGGR5KbWE", "paper_title": "Rethinking Intracranial Aneurysm Vessel Segmentation: A Perspective from Computational Fluid Dynamics Applications", "decision": "Reject", "summary": "The reviewer identifies excessive scope as the root cause of multiple deficiencies, criticizing the lack of documentation for the dataset, unjustified methodological choices, and unclear figures. While praising the dataset's scale and open-source release, the reviewer argues that the broad ambition undermines the clarity and value of all contributions, suggesting a complete restructuring around a single focused contribution.", "units": [{"unit_index": 0, "inspected_object": "Dataset scale and curation effort", "observation": "The paper compiles and curates a large-scale 3D MRA dataset, combining existing datasets with a new in-house collection (641 volumes and IAs).", "reasoning": "The reviewer explicitly recognizes this effort as 'quite impressive', indicating genuine appreciation for the data construction work.", "judgment": "Positive evaluation of the dataset's existence and scale.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8239347338676453, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7898757457733154}, {"unit_index": 1, "inspected_object": "Dataset contribution clarity and documentation", "observation": "The paper lacks detailed information on how the data was constructed and which aneurysm-related tasks it supports.", "reasoning": "Without these details, the dataset's value is not legible; the reviewer cannot assess if the contribution is merely re-packaging or genuinely novel, nor can they determine applicability scope.", "judgment": "The dataset's potential impact is undermined by insufficient documentation.", "valence": "negative", "suggested_improvement": "Provide detailed information on data construction methods and specify which aneurysm-related tasks the dataset supports.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7910546064376831, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8320940136909485}, {"unit_index": 2, "inspected_object": "Open-source commitment", "observation": "The authors have released the dataset and code.", "reasoning": "Open science practices are recognized as a virtue in the field, representing unqualified approval from the reviewer.", "judgment": "Positive evaluation of the open-source practice.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7863907217979431, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7503186464309692}, {"unit_index": 3, "inspected_object": "Enhancement claim regarding existing dataset", "observation": "The reviewer is unclear about how the existing dataset has been enhanced and why these enhancements are significant.", "reasoning": "The reviewer operates under the assumption that segmentation and CFD application are separable stages; without justification, the integration appears unnecessary rather than a genuine advance.", "judgment": "The enhancement claim is unsubstantiated and potentially superfluous.", "valence": "negative", "suggested_improvement": "Justify why CFD-based optimization is necessary compared to applying it as a subsequent step after existing segmentation methods.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8008694648742676, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7446205615997314}, {"unit_index": 4, "inspected_object": "Stage 1 loss function (count classification loss)", "observation": "The inclusion of count classification loss is questioned without explanation of its necessity.", "reasoning": "The reviewer suspects the component may be decorative rather than justified, reflecting a demand for methodological transparency and coherence.", "judgment": "The architectural choice is unjustified as presented.", "valence": "negative", "suggested_improvement": "Explain why the count classification loss is needed.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8014524579048157, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7989280819892883}, {"unit_index": 5, "inspected_object": "Stage 2 loss function parameters", "observation": "The parameters for the stage 2 loss are not explained in the text.", "reasoning": "The components are not self-explanatory, violating the norm of methodological transparency required for reproducibility and evaluation.", "judgment": "The method documentation is incomplete.", "valence": "negative", "suggested_improvement": "Provide explanations for each parameter in the stage 2 loss function.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8007871508598328, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8435221314430237}, {"unit_index": 6, "inspected_object": "Figure 4 (Vessel geometry check)", "observation": "Figure 4 is difficult to understand, specifically regarding how 'check vessel geometry' is performed and whether CFD is used.", "reasoning": "Ambiguity in the figure suggests a potential inconsistency with the paper's central claim about CFD applicability, undermining the legibility of the methodology.", "judgment": "The presentation of the geometry check procedure is unclear and potentially inconsistent.", "valence": "negative", "suggested_improvement": "Clarify Figure 4 to explicitly show the procedure for checking vessel geometry and indicate whether CFD is used.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7618700265884399, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8052370548248291}, {"unit_index": 7, "inspected_object": "Overall paper scope and narrative structure", "observation": "The paper attempts to contribute a dataset, benchmark, and method simultaneously, leading to excessive scope and lack of focused contribution.", "reasoning": "The breadth of ambition results in insufficient depth for each component, causing deficiencies in documentation, justification, and clarity across all parts.", "judgment": "The structural approach undermines the value of individual contributions.", "valence": "negative", "suggested_improvement": "Rethink the entire story of the paper to focus on a single, well-defined contribution.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8501718044281006, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7370530366897583}]}, {"review_id": "0HX2sZGytt", "paper_id": "CUFJluq5O8", "paper_title": "U-MARVEL: Unveiling Key Factors for Universal Multimodal Retrieval via Embedding Learning with MLLMs", "decision": "Accept (Poster)", "summary": "The reviewer distinguishes between the paper's engineering utility and its scientific contribution, denying novelty due to reliance on ported techniques while demanding mechanistic explanations, controlled comparisons, and precise parameter reporting to validate empirical claims.", "units": [{"unit_index": 0, "inspected_object": "Transfer of known retrieval techniques (mean pooling, bidirectional attention, learnable temperature) from LLM to MLLM.", "observation": "The reviewer identifies that the paper's key findings correspond to established methods in prior work (e.g., LLM2Vec, TMLR papers).", "reasoning": "The reviewer applies a standard that scientific contribution requires new techniques, phenomena, or explanations; transferring known techniques across model families is viewed as engineering rather than science unless it yields non-obvious insights.", "judgment": "The paper lacks scientific novelty and offers little scientific significance despite its practical utility.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.837943434715271, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8352712988853455}, {"unit_index": 1, "inspected_object": "Instruction masking finding during pooling.", "observation": "The paper states results of instruction masking but provides no mechanistic explanation for why it helps.", "reasoning": "Empirical findings require a causal or functional account to be credible; merely stating results is insufficient for a valid research contribution.", "judgment": "The presentation of this finding is inadequate and its impact is questioned based on reviewer priors.", "valence": "negative", "suggested_improvement": "Provide a mechanistic explanation for why masking instruction tokens helps.", "support_status": "mixed", "confidence": "high", "object_key": "method_design", "object_sim": 0.8190249800682068, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7863067984580994}, {"unit_index": 2, "inspected_object": "Multi-stage training effectiveness claim.", "observation": "The paper claims multi-stage training is effective, but the setup involves increased data volume alongside the strategy change.", "reasoning": "Experimental comparisons must isolate the specific strategy from confounding variables like data quantity to be considered objective and fair.", "judgment": "The validity of the conclusion regarding the training strategy is undermined by potential confounds.", "valence": "negative", "suggested_improvement": "Control for data quantity or clarify how the comparison isolates the strategy effect.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7723634839057922, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8142616152763367}, {"unit_index": 3, "inspected_object": "Parameter specification for hard-negative filtering and pipeline settings.", "observation": "The paper omits exact values for similarity thresholds, initialization/optimizer for temperature, alpha in recall-then-rerank, and top-k criteria.", "reasoning": "Unspecified parameters prevent decision-level reproducibility and robustness checking; conclusions depend heavily on these choices.", "judgment": "The paper is under-specified to the point of being non-reproducible in its decision-making.", "valence": "negative", "suggested_improvement": "Report exact values and criteria used for parameter selection.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8592278957366943, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8548636436462402}, {"unit_index": 4, "inspected_object": "Consistency of hard negative effects across experiments.", "observation": "Hard negatives help in 'continue-hard' experiments but fail in Section 3.2.2.", "reasoning": "A coherent body of evidence should be internally consistent; discrepancies suggest differences in experimental settings or data choice that need resolution.", "judgment": "The inconsistency raises concerns about the reliability or generalizability of the hard negative findings.", "valence": "negative", "suggested_improvement": "Explain whether the discrepancy is caused by differences in experimental settings or data.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8035182356834412, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7721421718597412}, {"unit_index": 5, "inspected_object": "Distillation method novelty and efficiency claims.", "observation": "The distillation method is described as relatively conventional, and efficiency is claimed without empirical time comparisons.", "reasoning": "Claims of innovation and efficiency require empirical validation (actual time measurements) rather than just theoretical derivations or conventional descriptions.", "judgment": "The novelty and practical advantage of the distillation method are unsubstantiated.", "valence": "negative", "suggested_improvement": "Provide empirical time comparisons to validate efficiency claims.", "support_status": "memo_inferred", "confidence": "high", "object_key": "novelty", "object_sim": 0.8251239061355591, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8081855773925781}, {"unit_index": 6, "inspected_object": "Compression prompt effect size.", "observation": "The paper reports a positive effect for compression prompts.", "reasoning": "Based on the reviewer's personal experience in the field, the reported impact may not be pronounced, suggesting the effect might be overstated.", "judgment": "The claimed effect of compression prompts is likely exaggerated.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7975780963897705, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7685337662696838}, {"unit_index": 7, "inspected_object": "Overall experimental reporting and reproducibility.", "observation": "The paper explicitly lists backbone, datasets, and training configs.", "reasoning": "Explicit listing of these components makes code-level replication feasible.", "judgment": "The paper has high practical utility and actionability as a technical report.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8718473315238953, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7254911065101624}]}, {"review_id": "0ILixELCqg", "paper_id": "ijhhFHvWS6", "paper_title": "Lost in Real-World Scenarios: Concretization Disrupts LLM Logical Reasoning", "decision": "Reject", "summary": "The reviewer performs a scoping evaluation, primarily concerned with external validity. They acknowledge the internal coherence of the dual-learning framework but critique the narrow empirical support: reliance on a single task, insufficient task difficulty for frontier models, and lack of model breadth in validating mitigation strategies. Presentation issues are noted as minor.", "units": [{"unit_index": 0, "inspected_object": "Benchmark scope and task diversity", "observation": "The benchmark only investigates the problem with one task.", "reasoning": "A framework claiming to study concretization as a general phenomenon must demonstrate this across multiple reasoning domains; testing on additional tasks (e.g., graph coloring, scheduling) would strengthen the claim of generality implied by the title.", "judgment": "The scope is too narrow for a general framework claim.", "valence": "negative", "suggested_improvement": "Test the framework on additional reasoning tasks beyond the single current one.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8460123538970947, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8373690843582153}, {"unit_index": 1, "inspected_object": "Task difficulty and ceiling effects", "observation": "GPT-o3 achieves >97% accuracy on all settings.", "reasoning": "If a frontier model nearly saturates the benchmark, the documented performance decline may be confined to weaker models, undermining the practical significance of the challenge and suggesting the benchmark is not hard enough to serve as a durable stress test for LLM fragility.", "judgment": "The task is not challenging enough to validate the claimed robustness issues effectively.", "valence": "negative", "suggested_improvement": "Calibrate the benchmark so that even strong models show meaningful variance in performance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7893539071083069, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7809157967567444}, {"unit_index": 2, "inspected_object": "Mitigation method validation breadth", "observation": "Only Qwen3-30B-A3B is used to evaluate the prompt-based and training-based mitigation methods.", "reasoning": "Methods should be robust across model families and scales; relying on a single relatively small MoE model cannot establish that the abstraction strategy works generally compared to larger or different architectures.", "judgment": "The validation of the mitigation methods lacks sufficient breadth to be fully credible.", "valence": "negative", "suggested_improvement": "Evaluate the mitigation methods on broader model coverage, such as Llama-3-70B and GPT-4o-mini.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8203598856925964, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8185034394264221}, {"unit_index": 3, "inspected_object": "Figure 5 visualization choice", "observation": "Figure 5 uses a pie chart to display absolute counts.", "reasoning": "Pie charts are conventionally discouraged for count data because they obscure magnitude differences; a bar chart would better represent the absolute values presented.", "judgment": "Minor presentation issue regarding graphical integrity.", "valence": "negative", "suggested_improvement": "Replace the pie chart in Figure 5 with a bar chart.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8174590468406677, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.6999384164810181}, {"unit_index": 4, "inspected_object": "Code and benchmark availability", "observation": "It is unclear whether the code and benchmark will be open-sourced.", "reasoning": "Open-source artifacts are standard for reproducibility in ML research; asking for this clarifies the authors' intent rather than critiquing a deficiency.", "judgment": "Neutral inquiry into reproducibility plans.", "valence": "positive", "suggested_improvement": "Clarify intentions regarding open-sourcing the code and benchmark.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8385539054870605, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8353015780448914}]}, {"review_id": "0JIyWnfgOt", "paper_id": "PBmV8bOIsi", "paper_title": "Bridging Bag-Level and Instance-Level Uncertainty with Conformalizable MIL", "decision": "Reject", "summary": "The reviewer constructs a conditional endorsement by mapping the boundaries of the paper's theoretical claims, specifically probing the novelty of Lemma 1, the general validity of the bag-score assumption, the computability of constant C₁, the empirical failure on Camelyon16, and the architectural scope limitations.", "units": [{"unit_index": 0, "inspected_object": "Lemma 1 (core condition for instance PAC learnability)", "observation": "Imported from Jang & Kwon (2025a) rather than derived in the current work.", "reasoning": "The paper's novelty lies in connecting existing learnability results to conformal guarantees, not in the learnability result itself; this defines a boundary on the originality of the theoretical foundation.", "judgment": "The contribution is framed as a connection/link rather than a foundational discovery, which qualifies the perceived novelty.", "valence": "conditional", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8268113732337952, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8309447765350342}, {"unit_index": 1, "inspected_object": "Assumption that bag scores approximate weighted sums of instance scores with bounded deviation (Eq. 16 / Appendix F)", "observation": "The assumption is plausible for specific architectures like Conjunctive-Pooling but its general validity across all MIL models and nonconformity scores is questioned.", "reasoning": "Without precise scope conditions, it is unclear when the theory applies broadly versus narrowly; the reviewer seeks to know *when* the theory holds rather than accepting broad applicability.", "judgment": "The theoretical framework lacks clearly specified scope conditions for general application.", "valence": "negative", "suggested_improvement": "Specify scope conditions more precisely to delineate when the theory applies.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8089451193809509, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7964857220649719}, {"unit_index": 2, "inspected_object": "Constant C₁ in Theorem 1's finite-sample bound", "observation": "C₁ encapsulates how excess risk translates to coverage degradation, but the paper provides no estimation method for it despite acknowledging its dependence on several factors.", "reasoning": "A finite-sample bound that cannot be computed has limited practical value; operationalizability requires methods to compute or bound such constants.", "judgment": "The bound is of limited practical utility due to lack of computability.", "valence": "negative", "suggested_improvement": "Provide intuition, bounds for specific architectures, and scaling behavior with bag size to transform C₁ into a usable quantity.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8264352083206177, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8147505521774292}, {"unit_index": 3, "inspected_object": "Empirical validation on Camelyon16 benchmark", "observation": "All three methods perform comparably with no significant differences (p > 0.84), contradicting the hypothesis that only Conjunctive MIL should transfer guarantees in D_Gen domains.", "reasoning": "The theory predicts Conjunctive MIL should excel in D_Gen domains, which holds for synthetic data but not clearly for Camelyon16; this suggests the theory's predictive power is limited to controlled settings.", "judgment": "The empirical evidence fails to support the central theoretical prediction on a major benchmark.", "valence": "negative", "suggested_improvement": "Explain why theoretical predictions do not clearly hold on Camelyon16 (Q3).", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8484842777252197, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7606239914894104}, {"unit_index": 4, "inspected_object": "Architectural scope of tested methods (Additive, ABMIL, Conjunctive)", "observation": "Tested architectures are representative regarding learnability theory but omit broader MIL landscape components like TransMIL and attention mechanisms beyond ABMIL.", "reasoning": "Transformer-based methods often violate the Eq. 16 assumption by construction; testing only convex-combination aggregators may limit the framework's apparent generality.", "judgment": "The framework's applicability may be limited to a specific architectural family, requiring explicit delimitation or extension.", "valence": "conditional", "suggested_improvement": "Extend the theory or explicitly delimit its domain regarding Transformer-based methods (Q4).", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8255571722984314, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.80763840675354}]}, {"review_id": "0KsuiQ3Kl1", "paper_id": "Bi7CPaOMHC", "paper_title": "RAID: A Benchmark Dataset for Testing the Adversarial Robustness of AI-Generated Image Detectors", "decision": null, "summary": "The reviewer evaluates the benchmark primarily on its engineering quality and practical utility, praising the dataset construction and experimental breadth while critiquing the lack of theoretical insight, maintenance planning, and robustness testing under realistic conditions. The judgment balances appreciation for the tool's value against concerns about its scientific contribution and long-term sustainability.", "units": [{"unit_index": 0, "inspected_object": "Dataset composition and detector ensemble architecture", "observation": "The dataset includes 96,000 images with adversarial examples across three perturbation levels and attacks seven detectors based on diverse architectures (ResNet, CLIP, patch-level).", "reasoning": "Architectural diversity ensures that the generated adversarial samples are not tied to any single model, making them more general and closer to real-world black-box scenarios.", "judgment": "The design choice enhances the practical utility and validity of the benchmark.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8408891558647156, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7746816277503967}, {"unit_index": 1, "inspected_object": "Image compression handling", "observation": "The dataset uses PNG storage and matches real-image compression characteristics.", "reasoning": "Matching compression reduces bias in the evaluation by ensuring the dataset reflects realistic conditions rather than idealized inputs.", "judgment": "The methodological care in format choices strengthens the validity of the results.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7729470133781433, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8004162311553955}, {"unit_index": 2, "inspected_object": "Evaluation metrics", "observation": "The paper employs multiple metrics including F1, accuracy, and AUROC.", "reasoning": "Using multiple metrics avoids bias from relying on any single measure, providing a more comprehensive assessment.", "judgment": "The multi-metric approach is a strength that supports robust evaluation.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8531594276428223, "reasoning_key": "construct_validity", "reasoning_sim": 0.7891086935997009}, {"unit_index": 3, "inspected_object": "Theoretical contribution and mechanistic insight", "observation": "The paper builds on established methods like PGD and ensemble attack concepts without proposing new theoretical ideas or mechanistic insights.", "reasoning": "A benchmark paper is expected to provide scientific insight or theoretical grounding beyond engineering execution; the absence limits its contribution to mere infrastructure.", "judgment": "The work is mainly engineering-focused and lacks the novelty or depth expected for a higher contribution score.", "valence": "negative", "suggested_improvement": "Provide theoretical insights or mechanistic explanations for why the benchmark performs as it does.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.7958755493164062, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7592275738716125}, {"unit_index": 4, "inspected_object": "Dataset maintenance and versioning plan", "observation": "There is no plan provided for dataset maintenance or versioning.", "reasoning": "Benchmarks should be living artifacts that evolve; reliance on diffusion-based generators may limit relevance as new architectures emerge without a maintenance strategy.", "judgment": "The lack of a sustainability plan is a weakness for long-term utility.", "valence": "negative", "suggested_improvement": "Include a plan for dataset maintenance, versioning, and adaptation to future generator architectures.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7745352983474731, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7761392593383789}, {"unit_index": 5, "inspected_object": "Robustness to common image transformations", "observation": "The adversarial examples are not tested for robustness against common transformations like cropping or JPEG compression.", "reasoning": "Real-world AIGI often undergoes such transformations, which can reduce attack effectiveness; evaluating only worst-case perturbations provides an incomplete picture of practical utility.", "judgment": "The current evaluation scope is too narrow to fully assess real-world applicability.", "valence": "negative", "suggested_improvement": "Test adversarial robustness against common image transformations like cropping and JPEG compression.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8201431035995483, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7913575172424316}, {"unit_index": 6, "inspected_object": "Commercial system evaluation sample size", "observation": "The evaluation of commercial detectors uses a small sample of 50 real and 50 generated images per detector.", "reasoning": "Small sample sizes lead to less reliable statistical estimates and may not capture the variability inherent in commercial systems.", "judgment": "The findings regarding commercial systems are statistically weak due to limited sampling.", "valence": "negative", "suggested_improvement": "Increase the sample size for commercial detector evaluations to improve statistical reliability.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8062621355056763, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8057568073272705}, {"unit_index": 7, "inspected_object": "Discussion of defensive insights", "observation": "The paper does not discuss potential defenses or insights for improving robustness based on the findings.", "reasoning": "Even engineering contributions should generate understanding; discussing how findings might inspire robustness improvements would add value beyond diagnosis.", "judgment": "The paper stops at diagnosis without offering prescriptive insights, limiting its scientific impact.", "valence": "negative", "suggested_improvement": "Add a discussion on how RAID's findings might inspire robustness improvements or defense strategies.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7692434191703796, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7743427753448486}, {"unit_index": 8, "inspected_object": "Perturbation perceptibility", "observation": "It is unclear whether epsilon=32/255 produces visible noise and if human perceptual validation was performed.", "reasoning": "If perturbations are visible, they may not function as effective adversarial examples in practice; this challenges the boundary between technical definition and practical deception.", "judgment": "The perceptibility of perturbations raises questions about the meaningfulness of the adversarial nature of the examples.", "valence": "conditional", "suggested_improvement": "Clarify if perturbations are visually perceptible and provide human validation results.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.8279529213905334, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8410590887069702}]}, {"review_id": "0L730IlOWN", "paper_id": "rjNz2ZYOhL", "paper_title": "TABS: Strategic Game-Based Multi-Stage Reinforcement Learning Challenge", "decision": null, "summary": "The reviewer critically evaluates TABS as a benchmark infrastructure contribution, focusing on its differentiation from existing environments, API standardization, and empirical rigor. They demand quantitative comparisons, standardized APIs, and performance metrics to justify the environment's place in the ecosystem, while also noting usability failures in the provided code.", "units": [{"unit_index": 0, "inspected_object": "TABS differentiation within the accelerated RL environment landscape", "observation": "The reviewer identifies a potential structural overlap between TABS's first two stages and Jumanji (both combinatorial) and notes the absence of quantitative comparisons to existing benchmarks like Atari, Jumanji, and MuJoCo.", "reasoning": "A new benchmark infrastructure must justify its existence against established alternatives by demonstrating distinctiveness in exploration or credit assignment challenges, not merely by offering another option. The reviewer expects a rigorous, quantitative mapping of properties similar to agent-focused analyses like Behaviour Suite.", "judgment": "The evidence for TABS's unique value is incomplete; the contribution claim lacks empirical demonstration of novelty relative to crowded existing options.", "valence": "negative", "suggested_improvement": "Provide quantitative comparisons to existing environments (Atari, Jumanji, MuJoCo) to empirically demonstrate what TABS measures that others do not.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.837030291557312, "reasoning_key": "design_justification", "reasoning_sim": 0.8238697052001953}, {"unit_index": 1, "inspected_object": "Environment API standardization and ecosystem integration", "observation": "The reviewer observes that TABS uses a configurable-but-unstandardized approach rather than a gym(-nax)-like API with fixed/predefined configurations.", "reasoning": "Standardization and named configurations reduce adoption friction and aid broader community uptake. A non-standard interface acts as a barrier to the environment's utility as a shared resource.", "judgment": "The current interface design hinders practical usability and broader adoption despite the underlying configurability.", "valence": "negative", "suggested_improvement": "Implement a factory pattern such as `tabs.make(\"<scenario-name>\")` with fixed/predefined configurations and accompanying names.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7590644359588623, "reasoning_key": "design_justification", "reasoning_sim": 0.7684226632118225}, {"unit_index": 2, "inspected_object": "Documentation quality for environment creation and spaces", "observation": "While the source code is easy to navigate, the README lacks detailed documentation regarding stages, environment creation, observation/action spaces, and reward computation.", "reasoning": "Practical use of an environment requires clear onboarding information. Insufficient documentation creates onboarding friction for potential adopters.", "judgment": "Documentation is insufficient for practical use despite navigable code.", "valence": "negative", "suggested_improvement": "Add more detailed documentation in the README regarding stages, environment creation, observation and action spaces, and reward computation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.8273327350616455, "reasoning_key": "cost_benefit", "reasoning_sim": 0.7613136768341064}, {"unit_index": 3, "inspected_object": "Attribution of difficulty in multi-stage nature", "observation": "The reviewer questions how much of TABS's difficulty stems from its multi-stage nature versus known difficulties like exploration in large action spaces or delayed rewards.", "reasoning": "It is necessary to determine if the challenge is genuinely novel (multi-stage coupling) or merely a re-packaging of known issues. An ablation or sensitivity analysis would clarify this causal decomposition.", "judgment": "Uncertainty remains regarding whether the difficulty profile represents a novel contribution or familiar challenges.", "valence": "uncertain", "suggested_improvement": "Perform an ablation or sensitivity analysis to measure the impact of early-stage decisions on later stages and decompose sources of difficulty.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "problem_framing", "object_sim": 0.8351768255233765, "reasoning_key": "design_justification", "reasoning_sim": 0.8033279776573181}, {"unit_index": 4, "inspected_object": "Scalability of difficulty via starter opponents", "observation": "The reviewer asks about starter opponents and configurations with increasing difficulty.", "reasoning": "A good benchmark should support iterative research by allowing practitioners to quickly test approaches and gradually scale up challenge, rather than presenting only a single hard setting.", "judgment": "Current setup may lack the curriculum structure needed for iterative benchmarking.", "valence": "conditional", "suggested_improvement": "Include starter opponents and configurations with increasing difficulty to support curricular testing.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.830771267414093, "reasoning_key": "fair_comparison", "reasoning_sim": 0.756697416305542}, {"unit_index": 5, "inspected_object": "Generalization measurement capabilities", "observation": "The reviewer probes whether procedural generation allows for generalization studies and if the environment is configurable in practice for this purpose.", "reasoning": "Generalization is a major concern in contemporary RL; the environment must demonstrate it supports configures-in-practice generalization studies, not just theoretically.", "judgment": "Unclear if the environment supports practical generalization studies.", "valence": "uncertain", "suggested_improvement": "Clarify how procedural generation supports generalization studies and provide examples of configurable scenarios for such purposes.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "stats_metrics", "object_sim": 0.8420765995979309, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7935800552368164}, {"unit_index": 6, "inspected_object": "End-to-end integration of three-stage environment", "observation": "The reviewer notices a potential gap between the paper's multi-stage framing and the implementation which might consist of separate environments per stage.", "reasoning": "If the central claim is multi-stage coordination, the implementation should instantiate this in a single trainable environment rather than separate ones.", "judgment": "Suspect disconnect between conceptual framing and practical implementation.", "valence": "negative", "suggested_improvement": "Demonstrate or clarify if the implementation supports a unified three-stage environment for end-to-end training.", "support_status": "memo_inferred", "confidence": "low", "object_key": "compute_cost", "object_sim": 0.7843812108039856, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7230284810066223}, {"unit_index": 7, "inspected_object": "Performance benchmarking metrics", "observation": "The reviewer requests specific speed metrics: steps/second, frames/second, and episodes/second.", "reasoning": "Environments claiming to be 'accelerated' via JAX implementation must provide concrete speed metrics to justify this architectural choice.", "judgment": "Lack of speed metrics undermines the justification for the 'accelerated' claim.", "valence": "negative", "suggested_improvement": "Report steps/second, frames/second, and episodes/second metrics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8481246829032898, "reasoning_key": "cost_benefit", "reasoning_sim": 0.73600172996521}, {"unit_index": 8, "inspected_object": "Balance tuning of unit strength/price", "observation": "The reviewer asks about unit strength/price balancing to probe if difficulty is designed or accidental.", "reasoning": "Difficulty should be engineered to avoid trivial dominant strategies, indicating intentional design rather than emergent or accidental complexity.", "judgment": "Unclear if difficulty is intentionally designed.", "valence": "uncertain", "suggested_improvement": "Provide details on balance tuning to show avoidance of trivial dominant strategies.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "robustness_sensitivity", "object_sim": 0.7792317271232605, "reasoning_key": "cost_benefit", "reasoning_sim": 0.785611629486084}, {"unit_index": 9, "inspected_object": "Code verification via run_example.py", "observation": "The reviewer encountered a failure in `run_example.py` due to a depreciated object.", "reasoning": "Running the code and hitting an error provides concrete evidence of usability failure, undermining claims of ease of use or reproducibility.", "judgment": "The example code fails to run, indicating immediate usability issues.", "valence": "negative", "suggested_improvement": "Fix the depreciated object usage in `run_example.py` to ensure the example runs successfully.", "support_status": "memo_inferred", "confidence": "high", "object_key": "reproducibility", "object_sim": 0.7557993531227112, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.7266330122947693}]}, {"review_id": "08oGragWHN", "paper_id": "oy4fc9h9oT", "paper_title": "Signals, Concepts, and Laws: Toward Universal, Explainable Time-Series Forecasting", "decision": "Reject", "summary": "The reviewer scrutinizes the paper's core claims of interpretability and physics-grounding, finding them under-supported due to reliance on internal correlations and lack of parameter analysis, while also noting presentation deficiencies.", "units": [{"unit_index": 0, "inspected_object": "Claim that learned concepts are strongly aligned with analytic targets (interpretability)", "observation": "Evidence for alignment is limited to internal correlations; absence of external validation such as case studies, human evaluations, or domain-level verification.", "reasoning": "Interpretability requires cognitive understanding by humans, not merely statistical correlation with pre-defined targets; internal correlation is the wrong category of evidence for interpretability claims.", "judgment": "The interpretability claim is under-supported because it lacks the required type of evidence (human/domain-level).", "valence": "negative", "suggested_improvement": "Provide visual, human, or domain-level verification of concept interpretability.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8318142294883728, "reasoning_key": "construct_validity", "reasoning_sim": 0.7878260016441345}, {"unit_index": 1, "inspected_object": "Driven-damped ODE head as a physics-informed first-principles constraint", "observation": "The ODE form is applied uniformly across unrelated domains without justification; learned coefficients are never analyzed or examined.", "reasoning": "A physics-informed model must justify why a specific physical form applies and demonstrate that parameters correspond to meaningful dynamics; without parameter analysis, the ODE functions only as a generic smoothness prior/regularizer rather than capturing domain-specific physics.", "judgment": "The physics-informed component is deflationarily reinterpreted as a generic regularizer, failing to support the 'first-principles' contribution claim.", "valence": "negative", "suggested_improvement": "Justify the choice of ODE form per domain and analyze learned coefficients to show correspondence to meaningful dynamics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8209241628646851, "reasoning_key": "design_justification", "reasoning_sim": 0.8245758414268494}, {"unit_index": 2, "inspected_object": "Presentation quality (Figures 2 and 3, grammar)", "observation": "Missing axis explanations in Figures 2 and 3; four specific grammar errors identified with line numbers.", "reasoning": "Careful presentation correlates with rigorous science; sloppy writing undermines confidence in the claims and indicates lack of attention to detail.", "judgment": "The presentation quality is deficient, contributing to a negative assessment of rigor.", "valence": "negative", "suggested_improvement": "Add axis explanations to Figures 2 and 3; correct the identified grammar errors.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.8882161974906921, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8553382754325867}]}, {"review_id": "08waAIe1B8", "paper_id": "SL3GBvdwKw", "paper_title": "VidEEG-Gen: A Dataset and Diffusion Framework for Video-Conditioned Privacy-Preserving EEG Generation", "decision": null, "summary": "The reviewer tests the paper's broader claims against evidentiary sufficiency, identifying gaps in dataset independence, ablation scope, and biological validation while acknowledging conceptual novelty.", "units": [{"unit_index": 0, "inspected_object": "Task framing and dataset construction scale (SEED-DV)", "observation": "The reviewer acknowledges the task redefinition as an interesting contribution but notes the dataset derives from SEED-DV with only 15 participants and 40 concepts, alongside a reported 12% cross-subject degradation in MSE.", "reasoning": "The limited scale and diversity of the substrate are evaluated against the generality implied by the task framing; the reviewer tracks whether these limits undermine the claim that the task is broadly applicable or representative.", "judgment": "Conceptual novelty is granted, but empirical generalizability is questioned due to the narrow scope of the underlying data.", "valence": "mixed", "suggested_improvement": "Clarify how the limited participant count and concept set impact the broader validity of the task definition.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8568425178527832, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8167474865913391}, {"unit_index": 1, "inspected_object": "Circularity between dataset generation and evaluation", "observation": "The reviewer identifies that the SPGN model may be used both to create the synthetic dataset and to benchmark it.", "reasoning": "If the generator defines the data and the evaluator measures against that same generator family, the 'benchmark' status becomes self-referential and fails to validate external fidelity.", "judgment": "The ontological status of the synthetic dataset as a ground-truth benchmark is undermined by potential circularity.", "valence": "negative", "suggested_improvement": "Demonstrate that the evaluation metrics are independent of the generator's specific inductive biases or provide evidence breaking this loop.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8436883687973022, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7676131725311279}, {"unit_index": 2, "inspected_object": "Availability of EEG prior during training", "observation": "The reviewer questions whether the EEG prior is always available during training or optional for simulating unseen-subject conditions.", "reasoning": "If the prior is always present, the conditioning is partially circular (needing EEG to generate EEG); if optional, the reviewer needs to know performance without it to assess privacy-preserving claims.", "judgment": "The alignment between the stated motivation (low-cost, scalable, privacy-preserving) and technical implementation is uncertain.", "valence": "conditional", "suggested_improvement": "Clarify the boundary conditions of the model's applicability regarding prior availability and simulate unseen-subject conditions explicitly.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7549170851707458, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7540180087089539}, {"unit_index": 3, "inspected_object": "Ablation study scope", "observation": "The ablation study covers only spatial attention and diffusion step count, despite the presence of multiple interacting modules (CLIP encoders, text embeddings, graph convolutions, etc.).", "reasoning": "Without component-level decomposition, it is impossible to attribute performance gains to specific architectural choices like the graph structure or fusion mechanism versus others.", "judgment": "The causal responsibility for the reported performance remains unverified due to insufficient empirical attribution.", "valence": "negative", "suggested_improvement": "Conduct ablations on key interacting modules to isolate their individual contributions to performance.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7766514420509338, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.782813549041748}, {"unit_index": 4, "inspected_object": "Biological plausibility claims", "observation": "Biological plausibility is assessed via internal metrics (frequency-band similarity, stability index) rather than expert or empirical comparison with real EEG traces.", "reasoning": "Internal consistency does not establish external validity; physiological characteristics recognized by experts require biological validation beyond statistical similarity.", "judgment": "Claims of biological plausibility are not sufficiently supported by the current evidence framework.", "valence": "negative", "suggested_improvement": "Include external validation such as expert visual inspection, classifier transfer tests, or comparison with real EEG traces.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.7447887659072876, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.8183804750442505}, {"unit_index": 5, "inspected_object": "Downstream utility of generated EEG", "observation": "The downstream utility of the generated EEG (e.g., improving classifier performance) is untested.", "reasoning": "The value of the dataset should be demonstrated through its usefulness in downstream tasks, not just through internal metrics.", "judgment": "The practical utility and benchmark status of the dataset remain unproven.", "valence": "uncertain", "suggested_improvement": "Evaluate the generated EEG on downstream classification tasks to demonstrate utility.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7931131720542908, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7571889758110046}]}, {"review_id": "0B6BVUPNiQ", "paper_id": "H13wHRiL3i", "paper_title": "Hyperspherical Latents Improve Continuous-Token Autoregressive Generation", "decision": "Accept (Poster)", "summary": "The reviewer evaluates the paper primarily through a verificationist lens, assessing the internal coherence between the identified problem (variance collapse), the theoretical justification (hyperspherical constraints), and the empirical evidence (ablations and SOTA comparisons). Positive judgments stem from the clarity of the theory, the efficiency of the results, and the rigorous validation of components. Minor reservations exist regarding practical efficiency metrics and broader generalizability to masked autoencoding.", "units": [{"unit_index": 0, "inspected_object": "Theoretical mechanism of hyperspherical constraints and first-order analysis", "observation": "The reviewer identifies the paper's core claim that hyperspherical constraints remove scale degrees of freedom and that a first-order analysis shows normalization removes radial errors, characterizing it as a 'clean theoretical first-order analysis'.", "reasoning": "The reviewer applies a norm of theoretical grounding, valuing parsimony and precision in explanation. The observation supports the judgment because the theory is seen as simple enough to be graspable yet precise enough to be falsifiable, thereby doing the explanatory work claimed.", "judgment": "The theoretical component is deemed adequate and well-justified.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8545622229576111, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8399642109870911}, {"unit_index": 1, "inspected_object": "Empirical comparisons against diffusion and masked-generation baselines", "observation": "The reviewer notes specific FID numbers (1.54 for SphereAR-L) matching MAR-H with half the parameters, framing this as clear SOTA FID at comparable or fewer parameters.", "reasoning": "The reviewer uses a comparative standard rather than an absolute one, evaluating efficiency relative to existing methods. The observation supports the judgment because the method demonstrates superior performance efficiency compared to diffusion and masked-generation models.", "judgment": "The empirical results demonstrate superior efficiency and state-of-the-art performance.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8644691705703735, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8013716340065002}, {"unit_index": 2, "inspected_object": "Ablation structure confirming component contributions", "observation": "The reviewer explicitly mentions ablations confirming hyperspherical latents outperform diagonal-Gaussian, fixed-variance, and Gaussian+post-hoc normalization, and that constant-norm refeeding is the key driver of gains.", "reasoning": "The reviewer applies a norm of comprehensive validation, expecting systematic isolation of components. The observation supports the judgment because the ablations verify that proposed components are not decorative but earn their place through controlled comparison against relevant alternatives.", "judgment": "The ablation structure rigorously validates the necessity and contribution of each component.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8181400895118713, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8435442447662354}, {"unit_index": 3, "inspected_object": "Internal coherence between problem, theory, and evidence", "observation": "The reviewer observes that the paper identifies variance collapse under CFG, proposes a theoretically justified solution, and verifies results match predictions.", "reasoning": "The reviewer employs a coherence-based evaluation standard, checking if the three pillars form a stable structure. The observation supports the judgment because the internal consistency of the narrative suggests claims are well-supported by presented evidence, leading to high confidence.", "judgment": "The paper presents a sound, internally coherent argument.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8417646288871765, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8446496725082397}, {"unit_index": 4, "inspected_object": "Wall-clock efficiency comparisons", "observation": "The reviewer notes the absence of time-to-result metrics alongside FID scores.", "reasoning": "The reviewer applies a pragmatic validation standard, questioning whether computational gains come at unacceptable costs. The observation supports the judgment that while the method is effective, its practical attractiveness is incomplete without efficiency data.", "judgment": "The method's practical utility is partially unverified due to missing wall-clock efficiency data.", "valence": "negative", "suggested_improvement": "Provide wall-clock efficiency comparisons to assess time-to-result.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "compute_cost", "object_sim": 0.8103269934654236, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8352774977684021}, {"unit_index": 5, "inspected_object": "Generalization to the MAE paradigm", "observation": "The reviewer notes the hyperspherical insight is currently applied only to autoregressive generation.", "reasoning": "The reviewer applies a breadth standard, seeing potential transferability of the core insight. The observation supports the judgment that the contribution's impact could be enhanced by demonstrating applicability beyond the current scope.", "judgment": "The scope of the contribution is limited to autoregressive generation, potentially overlooking broader applicability.", "valence": "conditional", "suggested_improvement": "Investigate extension of the hyperspherical insight to the MAE paradigm.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.8477779626846313, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8218369483947754}]}, {"review_id": "0DocwNrEF0", "paper_id": "xCoMdsTHCe", "paper_title": "Bob’s Confetti: Phonetic Memorization Attacks in Music and Video Generation", "decision": null, "summary": "The reviewer rejects the paper's significance by arguing that the claimed phonetic bypass is redundant because verbatim lyrics already produce similar outputs, and that the observed cross-modal leakage does not meet legal thresholds for copyright infringement due to lack of quantified substantial similarity. The reviewer relies heavily on personal experience with platform behavior as disconfirming evidence for the paper's threat model.", "units": [{"unit_index": 0, "inspected_object": "The paper's central motivational claim that phonetic rewrites bypass verbatim lyric filters on platforms like Suno.", "observation": "The reviewer notes that their own generation of 'Lose Yourself' on Suno using real lyrics produced output that was 'similar but not identical,' and cites SONICS (ICLR 2025) as further evidence that verbatim lyrics already produce similar-sounding songs.", "reasoning": "The reviewer applies the standard that an attack's significance depends on it achieving something different from normal system behavior. Since verbatim lyrics already yield 'similar but not identical' output, the reviewer infers that phonetic rewrites are redundant and do not exploit a unique bypass mechanism.", "judgment": "The premise that phonetic attacks provide a distinct advantage over verbatim prompts is considered false or misframed relative to real product behavior.", "valence": "negative", "suggested_improvement": "Verify that platforms actually block verbatim lyrics before claiming a bypass exists.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.5732501745223999, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8033210039138794}, {"unit_index": 1, "inspected_object": "The framing of cross-modal leakage (text-to-video experiments) as a copyright risk.", "observation": "The reviewer observes that the paper presents visual similarities (hooded rapper, graffiti) resulting from phonetic prompts, but interprets these as stylistic similarities rather than literal reconstruction or plagiarism.", "reasoning": "The reviewer applies a legal evaluative standard ('substantial similarity') requiring formal, measurable metrics to define infringement. Because the paper does not quantify this similarity and the output is merely 'similar but not the same,' the reviewer judges there is no meaningful legal or ethical breach.", "judgment": "The identified cross-modal leakage does not constitute a significant copyright harm under current legal categories.", "valence": "negative", "suggested_improvement": "Demonstrate substantial similarity using formal musicological metrics.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7856541872024536, "reasoning_key": "construct_validity", "reasoning_sim": 0.7583317756652832}, {"unit_index": 2, "inspected_object": "The policy value and practical benefit of identifying phonetic similarity in AI-generated content.", "observation": "The reviewer constructs a counterfactual where platforms legally produce AI covers or user-generated remixes, rendering the identification of phonetic similarity irrelevant for policy enforcement.", "reasoning": "The reviewer uses a utility-based standard: if the behavior described (similar-but-not-identical output) is already legally tolerated or normal, then detecting it provides no additional policy or practical benefit.", "judgment": "The research question lacks clear policy relevance because the phenomena it detects are already part of the accepted operational landscape of these platforms.", "valence": "negative", "suggested_improvement": "Identify a concrete harm that phonetic prompting enables beyond what verbatim prompting already achieves.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.796069860458374, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7030807733535767}, {"unit_index": 3, "inspected_object": "The novelty and clarity of the phonetic rewriting procedure itself.", "observation": "The reviewer explicitly acknowledges that the phonetic rewriting procedure is 'clear, reproducible' and describes the probing angle as 'intuitively clever' and 'potentially useful.'", "reasoning": "The reviewer separates the technical execution/clarity of the method from its motivational premise and significance. The observation stands independently of the negative judgments on threat model and legal harm.", "judgment": "The methodological approach is sound and innovative in its design, even if the application context is questionable.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.783585250377655, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7984565496444702}]}, {"review_id": "0EtcU2071F", "paper_id": "kmK3WSCOCT", "paper_title": "From Markov to Laplace: How Mamba In-Context Learns Markov Chains", "decision": "Accept (Oral)", "summary": "The reviewer offers qualified endorsement, praising the paper's novelty, narrative coherence, and strong alignment between theory and Mamba architecture design. However, they express concerns about the limited experimental scope (lack of dedicated ICL tasks), the uncertain real-world impact of Mamba compared to transformers, and omissions in the theoretical model (gating mechanism).", "units": [{"unit_index": 0, "inspected_object": "Novelty of theoretical investigation into Mamba architecture ICL ability", "observation": "The paper is the first work to theoretically investigate the In-Context Learning (ICL) ability of the Mamba architecture.", "reasoning": "The reviewer treats the fact of being the first to perform this specific theoretical investigation as a self-evident positive value, without weighing substantive versus incremental novelty.", "judgment": "Positive evaluation based on priority/novelty.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7855958938598633, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8267317414283752}, {"unit_index": 1, "inspected_object": "Narrative coherence and expository structure", "observation": "The paper starts from an empirical observation, identifies convolution's role, and echoes this in the theoretical analysis, providing enough background for readers to understand the scientific question.", "reasoning": "The reviewer implicitly assumes that a well-organized, self-contained narrative is a virtue that aids community understanding and serves as a proxy for correctness or quality.", "judgment": "Positive evaluation of presentation and clarity.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8066073060035706, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7630037069320679}, {"unit_index": 2, "inspected_object": "Alignment between theoretical analysis and Mamba architecture design", "observation": "The theoretical analysis (convolution -> recurrence -> selectivity) perfectly aligns with and sheds light on the real-world application and key components of the Mamba architecture.", "reasoning": "The reviewer values the paper as a successful act of translation between abstract theory and concrete architectural design, specifically noting how it explains the 'real-world' mechanism.", "judgment": "Strong positive evaluation ('perfectly designed') regarding the bridge between theory and architecture.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7866302132606506, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7813759446144104}, {"unit_index": 3, "inspected_object": "Real-world practical relevance of Mamba architecture", "observation": "Mamba has not demonstrated strong ICL ability comparable to large transformer-based language models in real-world applications.", "reasoning": "The reviewer applies a relevance-based norm: the impact of theoretical work depends on the practical significance of its subject; since Mamba's ICL capability is currently unproven relative to transformers, the theoretical work's impact is contingent and potentially limited.", "judgment": "Hedged concern about the paper's real-world impact due to the subject's current standing.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.7819558382034302, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7827451229095459}, {"unit_index": 4, "inspected_object": "Empirical validation scope (WikiText-103 vs dedicated ICL tasks)", "observation": "The real-world experiment is limited to language modeling (perplexity on WikiText-103) rather than a dedicated, real-world ICL experiment.", "reasoning": "The reviewer invokes a construct-validity norm: if a theory claims to explain ICL, it should be tested on ICL tasks, not just general language modeling metrics.", "judgment": "Negative evaluation of experimental scope and targeting.", "valence": "negative", "suggested_improvement": "Conduct dedicated, real-world ICL experiments to validate the theory.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7996397018432617, "reasoning_key": "construct_validity", "reasoning_sim": 0.8124849200248718}, {"unit_index": 5, "inspected_object": "Comparative parameter efficiency (Theorem 2 implications)", "observation": "Theorem 2 provides an exponential lower bound on hidden dimension for Mamba.", "reasoning": "The reviewer seeks to interpret this bound comparatively to determine if it implies transformers are more parameter-efficient than Mamba for complex/high-order Markov processes, and asks for design insights to sidestep this trade-off.", "judgment": "Uncertainty/Question regarding the comparative engineering implications of the theoretical bound.", "valence": "conditional", "suggested_improvement": "Provide insight into whether transformers are more parameter-efficient and suggest design improvements for Mamba blocks to sidestep the trade-off.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "theory", "object_sim": 0.8619227409362793, "reasoning_key": "design_justification", "reasoning_sim": 0.7469302415847778}, {"unit_index": 6, "inspected_object": "Modeling omission of gating mechanism in theory", "observation": "The theoretical proof omits the gating mechanism, which is an important component in NLP tasks.", "reasoning": "The reviewer expects theoretical fidelity to the actual architecture; the omission raises questions about how the proposed 'counting' mechanism interacts with gating in practice.", "judgment": "Concern about the theory's completeness and fidelity to the real architecture.", "valence": "negative", "suggested_improvement": "Clarify how the gating mechanism interacts with the 'counting' mechanism described in Theorem 1.", "support_status": "reviewer_explicit", "confidence": "medium", "object_key": "theory", "object_sim": 0.7959453463554382, "reasoning_key": "novelty_standard", "reasoning_sim": 0.788222074508667}]}, {"review_id": "0FGp70aofy", "paper_id": "FOZ4FwC7YX", "paper_title": "Speech-DRAME: A Framework for Human-Aligned Benchmarks in Speech Role-Play", "decision": null, "summary": "The reviewer conducts a validity audit focusing on data provenance mismatches, weak human alignment metrics, and scope limitations, concluding the paper fails multiple evaluative standards.", "units": [{"unit_index": 0, "inspected_object": "Realism Evaluation data provenance and conceptual framing", "observation": "The paper frames Realism Evaluation as grounded in real human speech, but the evaluation model's training data contains synthetic speech.", "reasoning": "This creates a domain mismatch between the stated design philosophy (ecologically valid, bottom-up) and implementation, reducing the evaluator's usability by potentially biasing judgments learned from synthetic-sounding speech.", "judgment": "Design coherence is violated; the distinction between real and synthetic regimes is incomplete.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8589497804641724, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7510873675346375}, {"unit_index": 1, "inspected_object": "Realism Evaluation quantitative alignment with humans", "observation": "Realism Evaluation shows a Spearman correlation of 0.375 with human perception.", "reasoning": "A rank-based metric like Spearman is more robust to outliers than Pearson; a value of 0.375 falls below an implicit threshold for acceptable predictive validity in a benchmark claiming human alignment.", "judgment": "Poor alignment and limited reliability of the Realism strategy.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8423685431480408, "reasoning_key": "robustness_norm", "reasoning_sim": 0.730621337890625}, {"unit_index": 2, "inspected_object": "Evaluation protocol scope (single-turn vs multi-turn)", "observation": "The framework only supports single-turn evaluations.", "reasoning": "Role-play is inherently a multi-turn phenomenon; failing to assess coherence across turns limits the framework's construct coverage and generalizability relative to its claim of being comprehensive.", "judgment": "Scope limitation reduces the benchmark's relevance to the target task.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7718161940574646, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7552704811096191}, {"unit_index": 3, "inspected_object": "Presentation: absence of audible demo cases", "observation": "No audible demo cases are provided for the speech role-play generation task.", "reasoning": "For a speech paper, the absence of audio examples prevents independent verification of generated quality, violating the norm of auditory verifiability.", "judgment": "Presentation deficiency impairs verifiability.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7326464056968689, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7696804404258728}, {"unit_index": 4, "inspected_object": "Presentation: redundant content in introduction", "observation": "Redundant content exists in the introduction (lines 083–101).", "reasoning": "Padding wastes readers' attention and violates the standard of expository economy.", "judgment": "Presentation deficiency due to lack of concision.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8554283976554871, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7374593019485474}]}, {"review_id": "0FUv9Bq4JH", "paper_id": "6Y4TpUvrcP", "paper_title": "ProteinRPN: Towards Accurate Protein Function Prediction with Graph-Based Region Proposals", "decision": "Reject", "summary": "The reviewer conducts a coherence audit, demanding mechanistic transparency and principled justification for architectural choices over mere empirical performance. They critique the method as a composition of existing components lacking formal theoretical grounding, question the necessity of dual contrastive losses given small ablation gains, and request comparisons with simpler baselines and clarification on hierarchical modeling assumptions.", "units": [{"unit_index": 0, "inspected_object": "Methodological composition and architectural justification", "observation": "The method is a modular assembly of known techniques (RPN adaptation, node-drop pooling, functional attention, Graph Multiset Transformer) without formal characterization of the RPN adaptation or its inductive bias.", "reasoning": "The reviewer applies a standard requiring mechanistic transparency and principled reasoning for design choices, arguing that empirical gains alone are insufficient when the contribution relies on assembling existing components without explaining why they belong together.", "judgment": "The paper's contribution is perceived as thin because it lacks a chain of reasoning from biological problem to formal justification for each component.", "valence": "negative", "suggested_improvement": "Formalize the RPN adaptation mathematically and articulate the information-theoretic or geometric content of design decisions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.824079155921936, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8201867938041687}, {"unit_index": 1, "inspected_object": "Contrastive objective ablation results", "observation": "Ablation studies show small improvement from the whole contrastive objective, yet the paper employs two contrastive losses (SupCon and InfoNCE).", "reasoning": "The reviewer uses the paper's internal evidence to identify an inconsistency: if the combined objective barely helps, doubling the complexity with two losses appears unjustified.", "judgment": "The use of dual contrastive losses is questioned due to limited marginal gain shown in ablations.", "valence": "negative", "suggested_improvement": "Clarify why both SupCon and InfoNCE are needed given the small improvement in ablation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8312052488327026, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8039107322692871}, {"unit_index": 2, "inspected_object": "Comparative baselines against simpler mechanisms", "observation": "The paper does not compare against simpler alternative mechanisms such as Top-k attention, random region sampling, or single-loss variants (SupCon-only, InfoNCE-only).", "reasoning": "The reviewer expects component-level ablations to demonstrate that proposed mechanisms outperform trivial or existing counterparts, rather than just comparing against full SOTA systems.", "judgment": "The lack of comparison with simpler alternatives weakens the claim of specific contribution.", "valence": "negative", "suggested_improvement": "Include results comparing with other subgraph sampling methods like Top-k attention or random region, and ablate individual loss components.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.873416006565094, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7411118745803833}, {"unit_index": 3, "inspected_object": "Biological vs. structural hierarchy in attention mechanism", "observation": "The cone attention mechanism is described as 'hierarchy-aware', but it is unclear if it models GO term hierarchy (functional ontology) or only local structural hierarchy (protein architecture).", "reasoning": "The reviewer suspects the paper may be conflating two different kinds of hierarchy, potentially using biological language to describe a geometric mechanism without clarifying the distinction.", "judgment": "Skepticism exists regarding whether the biological priors are actually doing the work claimed or if the terminology is rhetorical.", "valence": "conditional", "suggested_improvement": "Clarify whether the cone attention directly models GO term hierarchy or only local structural hierarchy.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7939777970314026, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8034122586250305}, {"unit_index": 4, "inspected_object": "Hyperparameter threshold justification", "observation": "A 70% overlap threshold is used, but its origin (empirical tuning vs. biological reasoning) is not explained.", "reasoning": "The reviewer applies a norm that arbitrary-looking parameters should have provenance, probing whether the authors can distinguish between principled design and empirical fitting.", "judgment": "Uncertainty about whether the parameter choice reflects principled design or ad hoc fitting.", "valence": "uncertain", "suggested_improvement": "Justify the 70% overlap threshold by explaining whether it was tuned empirically or derived from biological reasoning.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8845424056053162, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8326435089111328}]}, {"review_id": "0FmWm3HrUI", "paper_id": "w3Wk9GjfEy", "paper_title": "Hierarchical Scoring with 3D Gaussian Splatting for Instance Image-Goal Navigation", "decision": "Reject", "summary": "The reviewer focuses on precision-seeking and comprehension-verification, identifying gaps in definitional clarity, notational consistency, and empirical robustness while questioning boundary conditions like generalization and disambiguation. The evaluation emphasizes internal coherence and teachability over external comparison or theoretical guarantees.", "units": [{"unit_index": 0, "inspected_object": "Methodological clarity and pedagogical accessibility of the scoring mechanism", "observation": "The reviewer finds the method currently opaque to readers seeking intuition, noting a lack of concrete examples distinguishing high-score from low-score rays.", "reasoning": "A paper should enable the reader to build a mental model of the mechanism; the absence of design intent articulation prevents intuitive understanding despite potential correctness.", "judgment": "The presentation fails to meet the standard of being teachable or intuitively explainable.", "valence": "negative", "suggested_improvement": "Add a short motivation and one concrete example illustrating which types of rays receive high versus low scores.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8251344561576843, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8105936646461487}, {"unit_index": 1, "inspected_object": "Definition of training/evaluation terms ('ground-truth rays' and 'ground-truth scores')", "observation": "The reviewer identifies that specific terms used in the protocol are undefined within the text.", "reasoning": "Undefined terminology blocks the reader's ability to understand how the local scoring module is supervised or validated, violating the norm that papers should be self-contained.", "judgment": "The specification is incomplete due to missing definitions.", "valence": "negative", "suggested_improvement": "Define 'ground-truth rays' and 'ground-truth scores' clearly.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7433277368545532, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7746214270591736}, {"unit_index": 2, "inspected_object": "Cross-attention equation (Eq. 5) tensor roles", "observation": "The reviewer notes that the distinction between query and key/value tensors is not self-evident in the provided equations.", "reasoning": "A reader must be able to mentally simulate the attention mechanism and verify information flow; ambiguity here prevents verification of the architectural specification.", "judgment": "The technical specification lacks necessary granularity for independent verification.", "valence": "negative", "suggested_improvement": "Explicitly specify which tensor serves as the query and which serve as keys/values in Eq. 5.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8445759415626526, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7635084986686707}, {"unit_index": 3, "inspected_object": "Notational consistency (upper/lower-case usage and subscripts)", "observation": "The reviewer detects mixed upper/lower-case usages and inconsistent subscripts throughout the formal apparatus.", "reasoning": "Professional standards require uniformity to reduce cognitive friction; inconsistency signals a lack of craftsmanship even if it does not alter mathematical meaning.", "judgment": "The presentation does not meet professional standards of notational hygiene.", "valence": "negative", "suggested_improvement": "Standardize notation by eliminating mixed case usages and ensuring consistent subscript application.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8039847016334534, "reasoning_key": "presentation_trust", "reasoning_sim": 0.7820563316345215}, {"unit_index": 4, "inspected_object": "Empirical robustness regarding hyperparameter sensitivity", "observation": "The reviewer observes the absence of ablation studies on the hyperparameters introduced by the Global/Local score components.", "reasoning": "State-of-the-art results are only meaningful if they are stable and not artifacts of fragile parameter choices; without sensitivity analysis, robustness cannot be assessed.", "judgment": "The empirical evaluation lacks demonstrated robustness against parameter variation.", "valence": "negative", "suggested_improvement": "Perform and report sensitivity ablations on the hyperparameters governing the Global and Local scores.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.9206761121749878, "reasoning_key": "robustness_norm", "reasoning_sim": 0.813357949256897}, {"unit_index": 5, "inspected_object": "Generalization capability of the semantic stage (Mask-R-CNN dependency)", "observation": "The reviewer questions whether the method inherits limitations from Mask-R-CNN's closed-vocabulary training classes.", "reasoning": "If the semantic stage relies on a closed-vocabulary detector, the method's applicability may be restricted to known classes, limiting its generalization scope.", "judgment": "Uncertainty remains regarding the method's ability to handle unseen classes or broader vocabularies.", "valence": "uncertain", "suggested_improvement": "Clarify how the method handles unseen classes or discuss the implications of relying on a closed-vocabulary detector.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.798625111579895, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8250195980072021}, {"unit_index": 6, "inspected_object": "Practical feasibility of the 3DGS reconstruction step", "observation": "The reviewer seeks information on the time and hardware requirements for the prerequisite 3D Gaussian Splatting reconstruction.", "reasoning": "Deployment in realistic settings requires an understanding of computational cost; timing data combined with quality metrics is needed to calibrate feasibility.", "judgment": "The practical deployability of the pipeline is unclear due to missing performance metrics on the reconstruction step.", "valence": "uncertain", "suggested_improvement": "Report basic quality metrics and timing/hardware requirements for the 3DGS reconstruction step.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.7878124713897705, "reasoning_key": "cost_benefit", "reasoning_sim": 0.824103832244873}, {"unit_index": 7, "inspected_object": "Disambiguation capability for same-class objects", "observation": "The reviewer questions how CLIP-derived relevancy fields distinguish between instances of the same class.", "reasoning": "Since CLIP provides class-level similarity, there is a conceptual gap in how instance-level disambiguation is achieved; failure to address this undermines the method's 'instance' goal.", "judgment": "Potential conceptual limitation exists where the method may fail to disambiguate same-class instances.", "valence": "conditional", "suggested_improvement": "Explain or demonstrate how the method distinguishes between different instances of the same class.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.7955102920532227, "reasoning_key": "design_justification", "reasoning_sim": 0.8105278015136719}]}, {"review_id": "0GgEX5XAv5", "paper_id": "tBf2SUzfZw", "paper_title": "Visual Jigsaw Post-Training Improves MLLMs", "decision": "Accept (Poster)", "summary": "The reviewer acknowledges strong technical execution and parsimonious task design but judges the conceptual novelty as limited because the method adapts existing jigsaw pretexts rather than inventing a new task. This central novelty concern drives requests for broader architecture testing, scalability analysis, mechanistic justification against reconstruction methods, and SoTA comparisons to establish practical value.", "units": [{"unit_index": 0, "inspected_object": "Conceptual novelty of the jigsaw post-training method", "observation": "The reviewer identifies that the task design (partitioning, shuffling, predicting permutations) extends classical self-supervised 'jigsaw' pretexts (Noroozi & Favaro 2016) into RL post-training, noting the novelty lies mainly in adapting it to MLLM post-training rather than the task itself.", "reasoning": "The reviewer holds a norm that task-level novelty matters for contribution, not just application-level novelty. Since the core task is not new and the adaptation is viewed as incremental, the paper fails to clear the threshold of conceptual contribution despite being technically sound.", "judgment": "Limited conceptual novelty; contribution is incremental rather than novel.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8047413229942322, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8686907291412354}, {"unit_index": 1, "inspected_object": "Mechanistic justification relative to reconstruction-based methods", "observation": "The reviewer notes the absence of discussion on X-Former, a method combining contrastive and masked-reconstruction objectives, and questions why ordering achieves similar benefits to reconstruction and if they are complementary.", "reasoning": "The reviewer seeks mechanistic justification for why structural ordering provides visual understanding benefits compared to dense pixel reconstruction. Without this explanation, the empirical results lack theoretical grounding against a competing paradigm.", "judgment": "Insufficient mechanistic justification; the method's relationship to reconstruction-based approaches is unclear.", "valence": "negative", "suggested_improvement": "Discuss X-Former and explain whether ordering and reconstruction paradigms are complementary or why ordering is justified given reconstruction alternatives.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8505972027778625, "reasoning_key": "design_justification", "reasoning_sim": 0.8339571356773376}, {"unit_index": 2, "inspected_object": "Training overhead comparison", "observation": "The reviewer notes the omission of a discussion on training overhead for Jigsaw compared to SFT.", "reasoning": "Practical adoption depends on resource efficiency. Without a relative cost comparison, it is unclear if the method's benefits come at a prohibitive computational cost.", "judgment": "Missing practical evaluation of computational cost.", "valence": "negative", "suggested_improvement": "Include a discussion comparing training overhead of Jigsaw to SFT.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8263834118843079, "reasoning_key": "cost_benefit", "reasoning_sim": 0.8958398699760437}, {"unit_index": 3, "inspected_object": "Reward sensitivity and scalability beyond current settings", "observation": "The reviewer asks about reward sensitivity (specifically discount factor γ = 0.2) and scalability beyond 3×3 grids/6 clips, questioning if performance degrades with larger grids or clip counts.", "reasoning": "If performance collapses at larger grid sizes or is highly sensitive to hyperparameters, the method may be brittle or contingent on specific configurations rather than being a generalizable approach.", "judgment": "Uncertain robustness and generality of the method outside tested hyperparameter ranges.", "valence": "conditional", "suggested_improvement": "Analyze reward sensitivity and test scalability with larger grid sizes (e.g., 4×4, 5×5) and more video clips.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8615701794624329, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8558053374290466}, {"unit_index": 4, "inspected_object": "Architecture generality across model families", "observation": "All results rely on Qwen2.5-VL, while the reviewer suggests testing on LLaVA/Blip to check for model-specific exploitation.", "reasoning": "If the method only works on one model family, its contribution is diminished. Generality requires diversity across model families, not just within the same architectural lineage.", "judgment": "Limited evidence of cross-architecture generality.", "valence": "negative", "suggested_improvement": "Test the method on other model architectures such as LLaVA or Blip.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8181583285331726, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8352895975112915}, {"unit_index": 5, "inspected_object": "Qualitative analysis of visual reasoning traces", "observation": "The reviewer requests qualitative analysis, specifically <think> outputs, to see evidence of the mechanism rather than just benchmark numbers.", "reasoning": "Benchmark improvements could be spurious or task-specific. Qualitative evidence of the model's reasoning process is needed to confirm the claimed improvement in visual understanding.", "judgment": "Lack of mechanistic evidence for improved visual understanding.", "valence": "negative", "suggested_improvement": "Provide qualitative analysis including visual reasoning traces (<think> outputs).", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8126709461212158, "reasoning_key": "design_justification", "reasoning_sim": 0.824987530708313}, {"unit_index": 6, "inspected_object": "Comparison to State-of-the-Art (SoTA)", "observation": "The reviewer notes missing comparison to SoTA for image and video benchmarks.", "reasoning": "Without SoTA comparisons, it is impossible to assess the practical value of the method. If the method does not match or beat SoTA, its contribution relies solely on conceptual novelty, which has already been judged limited.", "judgment": "Incomplete empirical validation due to missing SoTA comparisons.", "valence": "negative", "suggested_improvement": "Add comparisons to State-of-the-Art methods on image and video benchmarks.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.799395740032196, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8082664012908936}, {"unit_index": 7, "inspected_object": "Task design parsimony and modality coverage", "observation": "The reviewer approves that the method avoids generative components and is applied across three modalities, describing it as 'simple yet effective'.", "reasoning": "Parsimony in method design and broad modality coverage are valued strengths that demonstrate technical competence and careful execution.", "judgment": "Strong technical execution and well-designed task.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8274828195571899, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8030526041984558}]}, {"review_id": "0Gp4UJbjts", "paper_id": "cvztBvlglK", "paper_title": "Pre-training Limited Memory Language Models with Internal and External Knowledge", "decision": "Accept (Poster)", "summary": "The reviewer conducts a conceptual audit, praising the paradigm's economic innovation while systematically challenging its validation scope, the risk profile of its masking mechanism, and the validity of its core assumptions regarding knowledge-capability trade-offs under scaling and complex reasoning conditions.", "units": [{"unit_index": 0, "inspected_object": "Pre-training methodology masking mechanism and labeling pipeline", "observation": "The reviewer notes that the pipeline involves masking knowledge from the loss and using a distilled model for labeling, but flags that masking decisions carry risk due to potential labeling errors.", "reasoning": "The reviewer reasons via a risk asymmetry argument: unlike quality scoring which affects how much a model learns, masking determines what it can learn at all; therefore, an error in masking could permanently remove necessary knowledge, a consequence not adequately addressed by the paper despite similar-cost methods existing elsewhere.", "judgment": "The masking mechanism presents a significant, unaddressed risk profile regarding the integrity of learned knowledge.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8118177652359009, "reasoning_key": "design_justification", "reasoning_sim": 0.8409664034843445}, {"unit_index": 1, "inspected_object": "Evaluation suite scope (FactScore-only)", "observation": "The reviewer acknowledges strong FactScore results for the 355M model but criticizes the evaluation suite as being solely factuality-focused.", "reasoning": "The reviewer applies an evaluative standard that a new pre-training paradigm claiming fundamental changes must demonstrate general capabilities (instruction following, reasoning), not just performance on the specific dimension optimized; relying solely on factuality is deemed 'clearly unreasonable' for establishing the paradigm's validity.", "judgment": "The current evidence is insufficient to support the paradigm's central claims because it lacks demonstration of general competence.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7775524258613586, "reasoning_key": "construct_validity", "reasoning_sim": 0.7615407705307007}, {"unit_index": 2, "inspected_object": "Training scale and extrapolation validity", "observation": "The reviewer questions whether the observed advantages in the small-scale model can be overwritten or invalidated in subsequent training phases.", "reasoning": "The reviewer employs a counterfactual concern about scaling: if the benefits disappear when the paradigm is scaled up or continued, the current evidence cannot rule out this possibility, representing a validity threat to the paradigm's broader applicability.", "judgment": "The uncertainty regarding scalability and long-term retention constitutes a weakness in the validation.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8254929780960083, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8028476238250732}, {"unit_index": 3, "inspected_object": "Core assumption of knowledge externalization vs internalization", "observation": "The reviewer tests the paradigm against hard cases involving integrated concepts, such as mathematics, questioning if the model can handle domains where knowledge is not discrete facts.", "reasoning": "The reviewer uses a reductio-style counterfactual: if the model cannot internalize any knowledge, tasks requiring deep background knowledge become impractical because the prompt would need to carry an unreasonable amount of context, exposing a potential flaw in the clean externalization assumption.", "judgment": "The paradigm's core assumption may not hold for complex, integrated knowledge domains, creating practical barriers.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7831164002418518, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8046721816062927}, {"unit_index": 4, "inspected_object": "Motivating premise regarding memory-capability parameter trade-off", "observation": "The reviewer cites empirical patterns from Qwen3-4B where small models are competent at reasoning but weak on knowledge, challenging the idea that memory and capability strictly compete for parameters.", "reasoning": "The reviewer infers that if improved data quality allows more efficient compression of knowledge, the proposed trade-off might not hold, suggesting the externalization strategy solves a problem that better data could solve more simply, thereby undermining the paradigm's foundational motivation.", "judgment": "The motivating premise that externalization is necessary due to parameter competition is questionable and potentially flawed.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8699310421943665, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8001214861869812}, {"unit_index": 5, "inspected_object": "Paradigm ambition and cost-effectiveness", "observation": "The reviewer praises the proposal as 'highly innovative, radical yet cost-effective,' noting that searching is cheaper than memorizing with large pre-training data.", "reasoning": "The reviewer accepts an economic argument for resource efficiency, finding the departure from standard practice intuitively compelling and valuable, despite other concerns.", "judgment": "The approach has high value and novelty due to its economic efficiency and radical nature.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7677448391914368, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7744267582893372}]}, {"review_id": "0HnFGEQBnH", "paper_id": "7bFbxg0uBn", "paper_title": "Leveraging Human Collaboration: Learning from Human-Verified Labels", "decision": null, "summary": "The reviewer systematically tests the paper's theoretical assumptions, practical claims, and comparative positioning against standards of evidence, identifying gaps in error modeling, empirical validation of effort, estimator analysis, and baseline comparisons.", "units": [{"unit_index": 0, "inspected_object": "The HVL condition (Def. 1) and its underlying assumption that human verification of a VLM label as true implies perfect ground truth.", "observation": "The reviewer identifies an idealized assumption in Definition 1 where human verification is treated as error-free, while the paper's own experiments (Figure 4) acknowledge human error.", "reasoning": "Systematic misverification (e.g., consistent false positives) could break the theoretical guarantees derived from the HVL condition; synthetic experiments are insufficient substitutes for theoretical treatment of realistic error.", "judgment": "A significant mismatch exists between the theory's clean guarantees and practical conditions, undermining the paper's soundness.", "valence": "negative", "suggested_improvement": "Provide real human verification error rates to calibrate the acknowledgment of human error.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.821269154548645, "reasoning_key": "robustness_norm", "reasoning_sim": 0.793917179107666}, {"unit_index": 1, "inspected_object": "The central claim of 'limited human effort' in the paper's framing.", "observation": "The review lacks specific empirical evidence such as annotation time per verification, inter-annotator agreement, cost per instance, or realistic crowdsourcing experiments.", "reasoning": "A claim about human effort cannot be validated by synthetic simulations or round counts alone; it requires measurement in the actual setting where the claim is made to establish ecological validity.", "judgment": "There is a gap between the paper's framing and its evidence regarding the practical value of the method.", "valence": "negative", "suggested_improvement": "Include concrete metrics on annotation time, inter-annotator agreement, and cost per instance.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.763547956943512, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7902806997299194}, {"unit_index": 2, "inspected_object": "The hybrid conditional probability estimator used in the risk re-writing (Lemma 2, Theorem 3).", "observation": "The estimator is heuristically motivated but lacks statistical grounding, specifically bias/variance analysis or calibration strategy.", "reasoning": "Early in training, when model predictions are poor, the fused distribution may diverge from the true conditional; without analysis, it is unclear if the estimator's convenience compromises theoretical validity in this dynamic regime.", "judgment": "The estimator's lack of statistical examination raises concerns about its validity in critical training regimes.", "valence": "negative", "suggested_improvement": "Conduct bias/variance analysis or provide a calibration strategy for the hybrid estimator.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8265659213066101, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7979483604431152}, {"unit_index": 3, "inspected_object": "Comparative baselines, specifically active learning and simple confidence-threshold verification.", "observation": "The paper does not compare its method against these natural, low-cost baselines for human-verified setups.", "reasoning": "To assess whether HVLs offer genuine advantages over simpler alternatives for efficient human-AI collaboration, the method must be situated against existing approaches to selective human labeling.", "judgment": "The absence of these comparisons makes it difficult to assess the method's positioning and relative contribution.", "valence": "negative", "suggested_improvement": "Include comparisons to active learning and confidence-threshold verification baselines.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8558325171470642, "reasoning_key": "design_justification", "reasoning_sim": 0.7802018523216248}, {"unit_index": 4, "inspected_object": "The multi-round verification algorithm's mechanism for generating new candidate labels after rejection.", "observation": "The specific operational details of how new candidates are generated are not clearly described.", "reasoning": "Understanding this detail is crucial for reproducibility and for determining whether the method is genuinely iterative or merely re-sampling the same distribution.", "judgment": "The lack of clarity hinders reproducibility and understanding of the method's actual behavior.", "valence": "negative", "suggested_improvement": "Clarify the algorithm for generating new candidate labels after rejection.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8214302062988281, "reasoning_key": "reproducibility_norm", "reasoning_sim": 0.8068194389343262}]}, {"review_id": "0I3UE2FFQ7", "paper_id": "YRp4xqTs3n", "paper_title": "Counterfactual Time Series Forecasting with Textual Conditions", "decision": "Reject", "summary": "The reviewer performs a rhetorical-epistemic audit focusing on mismatches between claimed contributions and demonstrated artifacts, questioning the novelty of the problem setting relative to prior work, identifying potential circularity in the DTTC metric's usage, and challenging the ecological validity of evaluations on saturated datasets with potentially overfitted text models.", "units": [{"unit_index": 0, "inspected_object": "The alignment between the paper's stated contributions in the introduction/abstract and the actual artifacts demonstrated in the conclusion.", "observation": "The reviewer notes that TADiff is absent from the intro and abstract, while vague terms like 'framework' and 'paradigm' are used instead of concrete model architecture details.", "reasoning": "Inconsistent framing creates an epistemic problem where it is difficult to know what was actually built; specific artifacts and procedures must be listed for claims to be falsifiable and clear.", "judgment": "The presentation obscures the paper's actual contribution, requiring a shift from conceptual novelties to concrete artifacts and procedures.", "valence": "negative", "suggested_improvement": "Replace vague terms with 'model architecture + training strategy' and align the intro/conclusion to explicitly list artifacts like the model, DTTC, data generation strategy, and generalization guarantees.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.7419108748435974, "reasoning_key": "presentation_trust", "reasoning_sim": 0.8230184316635132}, {"unit_index": 1, "inspected_object": "The novelty of the problem setting versus the method for generating counterfactual futures.", "observation": "Prior work (Context is Key, Time-MMD) already addresses forecasting with textual information about future conditions.", "reasoning": "Conflating the problem with the method inflates the contribution; if the problem is established, the paper must demonstrate its technique on existing benchmarks rather than self-generated text.", "judgment": "The counterfactual generation is akin to a data augmentation strategy within an existing task, not a new task itself.", "valence": "negative", "suggested_improvement": "Cite prior work and reframe the contribution as a new technique/data augmentation strategy rather than a novel problem setting.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8219149708747864, "reasoning_key": "novelty_standard", "reasoning_sim": 0.819197416305542}, {"unit_index": 2, "inspected_object": "The dual role of the DTTC metric in both generating counterfactual datasets and evaluating generalization.", "observation": "DTTC scores are used to generate datasets and then used to evaluate generalization in Table 2.", "reasoning": "Using the same metric for optimization and evaluation introduces potential circularity; the metric's own generalization properties must be established before it can serve as a trustworthy evaluator.", "judgment": "There is a methodological concern regarding the validity of using DTTC as an evaluator without a proper train/test separation.", "valence": "negative", "suggested_improvement": "Fit the model and DTTC on a subset and test on held-out datasets to ensure separation; move DTTC training details to the main text for inspection.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8499934673309326, "reasoning_key": "construct_validity", "reasoning_sim": 0.7981447577476501}, {"unit_index": 3, "inspected_object": "The use of saturated datasets (ETT, Exchange, Weather, Traffic) and LLM-generated text.", "observation": "These datasets are claimed to be saturated, and the text may be generated by the same LLM family used to train the model.", "reasoning": "Beating baselines on saturated datasets may reflect incremental tuning rather than the value of textual conditioning; using the same LLM family risks overfitting to a text distribution rather than learning evidence-based reasoning.", "judgment": "The inferential value of the results is low due to potential dataset saturation and lack of ecological validity regarding text modality shifts.", "valence": "negative", "suggested_improvement": "Test on datasets where text is genuinely informative and perform cross-LLM generalization tests to verify robustness to text distribution shifts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8050847053527832, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.8135288953781128}, {"unit_index": 4, "inspected_object": "The invocation of Leibniz’s Principle of Continuity for exchange rate time series.", "observation": "The paper invokes a philosophical principle to justify modeling choices for financial data.", "reasoning": "The assumption of continuity may be unwarranted for exchange rates, and applying an unexamined metaphysical principle to statistical modeling lacks empirical grounding.", "judgment": "The application of the principle is unjustified for this domain.", "valence": "negative", "suggested_improvement": "Justify why a philosophical principle should constrain a statistical model of financial data or provide empirical grounding for the continuity assumption.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7982248663902283, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7742450833320618}]}, {"review_id": "0ILUUZVtYP", "paper_id": "UtMayAnGPo", "paper_title": "CAReDiO: Enhancing Cultural Alignment of LLM via Representativeness and Distinctiveness Guided Data Optimization", "decision": "Reject", "summary": "The reviewer balances appreciation for the paper's conceptual clarity and practical utility with skepticism regarding theoretical verification, evaluation cleanliness, and technical novelty. Key concerns include unverified theoretical preconditions, potential data leakage from shared dataset provenance, and the need for ablations to demonstrate causal attribution of the method's components.", "units": [{"unit_index": 0, "inspected_object": "Conceptual framing of representativeness and distinctiveness", "observation": "The reviewer explicitly credits the authors for clear problem framing and notes that separating data quality into these two challenges is conceptually helpful.", "reasoning": "The reviewer values work that makes cultural alignment tractable by breaking it into component challenges, finding the explanatory utility of this decomposition to be a strength.", "judgment": "The conceptual framing is positively evaluated as helpful and clear.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.880511999130249, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8012939095497131}, {"unit_index": 1, "inspected_object": "Theoretical grounding of the distinctiveness objective (Proposition 2)", "observation": "The distinctiveness objective relies on Proposition 2, which depends on strong, partly unverifiable conditions regarding classifier accuracy and calibration.", "reasoning": "Theory is valuable only if its assumptions are demonstrably satisfied in practice; the paper has not provided diagnostics to confirm these premises hold across the studied cultures, leaving a gap between mathematical derivation and empirical instantiation.", "judgment": "The theoretical foundation is flagged as a significant weakness due to lack of empirical verification of preconditions.", "valence": "negative", "suggested_improvement": "Provide diagnostics to verify that classifier accuracy and calibration conditions hold across the 15 cultures studied.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.7968165278434753, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7780706286430359}, {"unit_index": 2, "inspected_object": "Evaluation setup and potential data leakage", "observation": "Several compared datasets are built or augmented from WVS or related sources, and the evaluation also uses WVS/GlobalOpinionQA, but the paper does not rigorously audit overlap or de-duplication.", "reasoning": "If training and evaluation data share provenance, performance gains might reflect style overlap or memorization rather than genuine cultural alignment; claims of improvement require evidence that the improvement is not an artifact of experimental design.", "judgment": "The evaluation process is criticized for lacking rigorous auditing of data overlap, creating uncertainty about the validity of reported gains.", "valence": "negative", "suggested_improvement": "Rigorously audit overlap and perform de-duplication to demonstrate that the evaluation is clean.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8306563496589661, "reasoning_key": "fair_comparison", "reasoning_sim": 0.826532781124115}, {"unit_index": 3, "inspected_object": "LLM-as-judge circularity risk", "observation": "The method uses LLM-based classifiers to estimate culture labels, which may encode the same cultural priors the method hopes to correct, risking amplification of bias.", "reasoning": "While the tool used to measure cultural distinctiveness may itself be culturally biased, the authors mostly mitigate this risk through the use of existing well-developed human preference datasets like PRISM and global value surveys, making the mitigation adequate enough to not reject the paper.", "judgment": "The circularity risk is acknowledged but considered sufficiently mitigated by external validation sources.", "valence": "conditional", "suggested_improvement": null, "support_status": "mixed", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8180992603302002, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7622999548912048}, {"unit_index": 4, "inspected_object": "Technical novelty relative to prior synthetic pipelines", "observation": "The technical novelty relative to prior work (CulturePark, CultureLLM, etc.) is incremental, despite the particular objective pairing being novel.", "reasoning": "A new method should demonstrate that its specific components are responsible for success; without stronger ablations isolating the independent objectives, the reviewer cannot determine if success stems from the novel pairing or mundane factors like the iterative refinement loop.", "judgment": "The contribution is viewed as incremental rather than transformative due to lack of causal demonstration via ablations.", "valence": "negative", "suggested_improvement": "Perform ablations on the independent objectives to isolate the contribution of representativeness and distinctiveness.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "novelty", "object_sim": 0.892086386680603, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8104021549224854}, {"unit_index": 5, "inspected_object": "Pipeline simplicity and model-agnosticism", "observation": "The pipeline is described as simple to implement and model-agnostic.", "reasoning": "The reviewer values practical reproducibility and work that others can readily adopt, crediting the authors for a clean pipeline.", "judgment": "The practical design and ease of implementation are positively evaluated.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "medium", "object_key": "empirical_scope", "object_sim": 0.835711658000946, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8063979148864746}, {"unit_index": 6, "inspected_object": "Baseline development description", "observation": "The baseline description was found insufficiently precise to evaluate the comparison fairly.", "reasoning": "Methodological transparency is required to assess comparative performance; the current level of detail prevents fair evaluation.", "judgment": "The baseline description is insufficient for fair comparison.", "valence": "negative", "suggested_improvement": "Provide a more precise description of the role-playing baseline development.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "baselines_ablations", "object_sim": 0.8237928152084351, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8113858103752136}]}, {"review_id": "0IaMg5i5go", "paper_id": "bYA07SdHoS", "paper_title": "Jailbreaking Commercial Black-Box LLMs with Explicitly Harmful Prompts", "decision": null, "summary": "The reviewer conducts a durability assessment, accepting the technical validity of the attacks and dataset methods but questioning their long-term relevance due to dependence on specific API configurations. The logic separates ephemeral, API-tied findings from robust, methodology-driven contributions.", "units": [{"unit_index": 0, "inspected_object": "The 'emergent Developer role' as the basis for D-Attack and DH-CoT attacks", "observation": "The reviewer identifies the 'Developer role' as enabling transferable black-box attacks with high efficacy against state-of-the-art models.", "reasoning": "The reviewer applies a durability standard, reasoning that if the 'Developer role' is an accidental feature of the current OpenAI API configuration rather than a stable property of LLMs, the findings may become obsolete after one or two system updates.", "judgment": "The core attack mechanism is viewed as temporally fragile and potentially time-limited in its value.", "valence": "negative", "suggested_improvement": "Reconceptualize the finding from 'developer-role-based attacks' to a more general phenomenon like 'role-based attacks that exploit hierarchical instruction-following' to demonstrate durability beyond specific API versions.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7649760842323303, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7565978765487671}, {"unit_index": 1, "inspected_object": "The scope of evaluation limited primarily to OpenAI's API", "observation": "The review notes the paper's focus on OpenAI's API and the use of the 'system' role for non-OpenAI models instead of a native 'developer' role.", "reasoning": "The reviewer uses a generality standard, arguing that claims about 'commercial black-box LLMs' require evidence across multiple providers using their native role structures to prove the mechanism is not specific to OpenAI's implementation.", "judgment": "The current framing is too OpenAI-centric and fails to establish universal applicability of the conclusions.", "valence": "negative", "suggested_improvement": "Test methods on non-OpenAI models using their native role structures to determine if the attacks still work, thereby supporting a claim of universal understanding.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7739165425300598, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7669984698295593}, {"unit_index": 2, "inspected_object": "The MDH dataset-cleaning framework and manual annotation component", "observation": "The reviewer acknowledges the MDH framework achieves over 95% detection accuracy with minimal human effort but flags the manual annotation component as a limitation.", "reasoning": "Applying a scalability standard, the reviewer reasons that while human verification ensures ground truth trustworthiness (accuracy), it inherently limits the method's ability to scale to larger corpora of red-teaming data.", "judgment": "The dataset contribution is conceptually robust and durable but practically limited by scalability concerns.", "valence": "mixed", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8206644058227539, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7947962284088135}, {"unit_index": 3, "inspected_object": "The conceptual taxonomy of problematic prompts (BP, NHP, NTP)", "observation": "The reviewer treats the taxonomy of problematic prompts as a genuine conceptual contribution used to critique and refine red-teaming datasets.", "reasoning": "Using a structural problem standard, the reviewer reasons that this taxonomy addresses the persistent issue of low-quality red-teaming data, making it independent of transient API changes and thus temporally robust.", "judgment": "This aspect of the work is a stable, durable contribution to the field.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8160365223884583, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7612200379371643}, {"unit_index": 4, "inspected_object": "The distinction between technical merit and field-level contribution", "observation": "The reviewer grants three substantive strengths regarding technical execution but assigns a low contribution score.", "reasoning": "The reviewer applies a field-impact standard, distinguishing between well-executed technical work and contributions that fundamentally change how the field thinks about a problem. The temporal fragility of the core attack mechanism prevents it from meeting the bar for a transformative field-level contribution.", "judgment": "The paper represents competent technical execution but lacks the lasting impact required for a high contribution rating.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.7910271286964417, "reasoning_key": "confound_hypothesis", "reasoning_sim": 0.7977309823036194}]}, {"review_id": "0JIjE8Dfzq", "paper_id": "Gytb40h5CW", "paper_title": "Learning to Price Bundles: A GCN Approach for Mixed Bundling", "decision": "Reject", "summary": "The reviewer systematically interrogates the paper's validation, focusing on generalization across distributions, interpretability of pruning, hyperparameter robustness, theoretical-consistency of monotonicity, and component efficacy in hybrid pipelines, while acknowledging clear motivation and presentation.", "units": [{"unit_index": 0, "inspected_object": "Generalization claim of the GCN trained on 5-product instances to larger problems", "observation": "The paper claims generalization based on size scaling but lacks systematic sweeps across utility families and distribution shifts.", "reasoning": "A learning-based method's generalization claim is only meaningful if tested across input distributions (e.g., log(1+x), complementarities/substitutes, correlated utilities), not just problem dimensions; without this, the claim rests on an uncontrolled variable.", "judgment": "The generalization evidence is insufficient for a distribution-generalization claim.", "valence": "negative", "suggested_improvement": "Conduct a sweep across different utility distributions and functional forms.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8232062458992004, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.8446313738822937}, {"unit_index": 1, "inspected_object": "Positioning relative to neural combinatorial optimization and recommender systems literature", "observation": "The paper does not compare or discuss neural baselines (e.g., BGCN-style methods) or proxy comparisons.", "reasoning": "Field literacy requires demonstrating awareness of neighboring methods even if direct apples-to-apples benchmarking is impractical due to objective mismatch; a discussion of scalability and interpretability trade-offs is expected.", "judgment": "The paper lacks necessary positioning against relevant neural alternatives.", "valence": "negative", "suggested_improvement": "Include a discussion of objective mismatch, scalability, and interpretability relative to neural CO/recommender methods.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8257954120635986, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7953038215637207}, {"unit_index": 2, "inspected_object": "Interpretability of the GCN-based pruning mechanism", "observation": "The paper provides no visualizations (bundle heatmaps, per-segment items, before/after pruning sets) to explain pruning decisions.", "reasoning": "Without visualization, the GCN's contribution appears as a black box that might replace simpler heuristics; visualizations are needed to verify that pruning decisions align with economic intuition and are not arbitrary.", "judgment": "The pruning mechanism's intellectual contribution is opaque and unverifiable beyond aggregate performance.", "valence": "negative", "suggested_improvement": "Provide bundle heatmaps, per-segment top-items, and before/after pruning set visualizations.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8053746223449707, "reasoning_key": "design_justification", "reasoning_sim": 0.7282065749168396}, {"unit_index": 3, "inspected_object": "Hyperparameter selection and cutoff stability", "observation": "The paper uses a fixed 0.5 cutoff and specific GCN depth/hidden size/dropout without discussing adaptive variants.", "reasoning": "If success depends on hand-tuned thresholds, the method's generality is suspect; an adaptive or learned cutoff would suggest greater robustness across problem instances.", "judgment": "The method's robustness is questionable due to reliance on fixed hyperparameters.", "valence": "negative", "suggested_improvement": "Discuss sensitivity to hyperparameters or propose/evaluate adaptive cutoff variants.", "support_status": "memo_inferred", "confidence": "high", "object_key": "robustness_sensitivity", "object_sim": 0.8917384147644043, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8492525815963745}, {"unit_index": 4, "inspected_object": "Consistency between theoretical monotonicity assumption and algorithmic implementation", "observation": "The paper relies on subadditivity equivalence which assumes monotonicity, but it is unclear if the final prices produced by the algorithm are monotone.", "reasoning": "If the theory assumes monotonicity but the algorithm doesn't enforce it, there is a gap: either the theory doesn't apply to the outputs, or the algorithm violates its own assumptions; this consistency must be addressed.", "judgment": "There is an unresolved tension between the theoretical framework and practical implementation regarding monotonicity.", "valence": "negative", "suggested_improvement": "Address whether final prices are monotone and resolve the gap between theory and implementation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8560652732849121, "reasoning_key": "design_justification", "reasoning_sim": 0.8086287379264832}, {"unit_index": 5, "inspected_object": "Effectiveness of LP-guided local search in improving MILP gains", "observation": "It is unknown whether LP-guided improvements translate into actual MILP objective gains.", "reasoning": "Hybrid methods often suffer from mismatched objectives between components; the reviewer wants to know if the local search is earning its keep or just improving a proxy without benefiting the final MILP solution.", "judgment": "The contribution of the local search component to the final objective is unvalidated.", "valence": "negative", "suggested_improvement": "Analyze whether LP-guided improvements fail to translate into MILP gains.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8266890048980713, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7779912948608398}, {"unit_index": 6, "inspected_object": "Problem framing and motivation", "observation": "The reviewer acknowledges the problem is well-motivated and connects to revenue-management/economics literature.", "reasoning": "The reviewer checks if the problem is worth caring about and if the paper acknowledges its intellectual lineage, finding it acceptable in this regard.", "judgment": "The problem framing is strong and well-positioned.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.9342833757400513, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8221235275268555}, {"unit_index": 7, "inspected_object": "Methodological clarity and readability", "observation": "The pipeline (GCN → pruning → MILP) is described clearly and is readable.", "reasoning": "The reviewer evaluates comprehensibility and relevance, noting that while not novel, the presentation is clear.", "judgment": "The method description is clear and readable.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8266572952270508, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8336182832717896}]}, {"review_id": "0JoceklNCh", "paper_id": "9aaiQbIUND", "paper_title": "Leveraging Generative Trajectory Mismatch for Cross-Domain Policy Adaptation", "decision": "Reject", "summary": "The reviewer evaluates the paper as competent and well-written but questions its novelty and experimental completeness, leading to a qualified endorsement that highlights the need for broader comparisons and clarifications on design choices.", "units": [{"unit_index": 0, "inspected_object": "Paper framing and motivation", "observation": "The reviewer explicitly praises the paper as well-motivated, well-written, easy to follow, and notes that it tells a coherent story.", "reasoning": "The reviewer values the clarity of presentation and the timeliness of addressing an underserved area (online off-dynamics RL), which supports a positive evaluation of the paper's execution and relevance.", "judgment": "Positive assessment of quality and topical importance.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.894406795501709, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8205085396766663}, {"unit_index": 1, "inspected_object": "Downstream methodological mechanisms", "observation": "The downstream methods rely on reward modification or data filtering, which resembles prior works like DARC, PAR, and VGDF.", "reasoning": "The reviewer distinguishes between the novel generative framing and the established mechanism, concluding that the core contribution may be derivative rather than genuinely new.", "judgment": "Concern about limited novelty in the mechanism.", "valence": "negative", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8025662899017334, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8237412571907043}, {"unit_index": 2, "inspected_object": "Theoretical analysis", "observation": "The theoretical results resemble those in prior works, but the reviewer acknowledges they bring some insights.", "reasoning": "The reviewer performs a cost-benefit analysis, accepting partial novelty because the theory is competently executed and provides value, despite not being fully original.", "judgment": "Acceptable theoretical contribution due to insight, despite lack of full novelty.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.7824702262878418, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8356902599334717}, {"unit_index": 3, "inspected_object": "Experimental scope and coverage", "observation": "The paper does not cover all standard dynamics shifts available in ODRL benchmarks, specifically missing gravity and friction shifts.", "reasoning": "The reviewer holds an implicit norm that a general cross-domain adaptation method should demonstrate robustness across diverse types of dynamics mismatch, not just kinematic/morphology shifts.", "judgment": "Evaluation is incomplete regarding scope.", "valence": "negative", "suggested_improvement": "Extend experimental scope to include other dynamics shifts like gravity and friction.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.83089679479599, "reasoning_key": "design_justification", "reasoning_sim": 0.7937638759613037}, {"unit_index": 4, "inspected_object": "Generative modeling choices (Diffusion vs. Alternatives)", "observation": "The paper lacks comparison with other generative modeling methods such as flow matching or VAEs, despite mentioning flow matching.", "reasoning": "If the contribution relies on using diffusion for trajectory mismatch estimation, the authors should empirically justify why diffusion is the right choice over other generative tools.", "judgment": "Justification for the specific generative model choice is missing.", "valence": "negative", "suggested_improvement": "Compare the proposed method against other generative modeling methods (e.g., flow matching, VAE).", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8142388463020325, "reasoning_key": "design_justification", "reasoning_sim": 0.8272712230682373}, {"unit_index": 5, "inspected_object": "Design philosophy: Generative Trajectory Mismatch vs. Data Augmentation", "observation": "The reviewer questions why the authors use diffusion for generative trajectory mismatch rather than target domain data augmentation.", "reasoning": "The reviewer views data augmentation as a more intuitive alternative and implies that the chosen approach requires explicit justification to show its superiority or distinct insight.", "judgment": "Uncertainty about the necessity and insight of the specific architectural choice.", "valence": "conditional", "suggested_improvement": "Provide insight into why generative trajectory mismatch is preferred over target domain data augmentation.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8218187093734741, "reasoning_key": "merit_recognition", "reasoning_sim": 0.8072925806045532}, {"unit_index": 6, "inspected_object": "Sensitivity to diffusion steps", "observation": "The number of diffusion steps has a significant impact on performance, but the selection criteria and reasons for this sensitivity are unclear.", "reasoning": "High sensitivity to a poorly understood hyperparameter poses a practical obstacle to usability and adoption.", "judgment": "Usability concern due to parameter sensitivity.", "valence": "negative", "suggested_improvement": "Provide insights on how to select the diffusion step parameter and explain why different steps have significant impacts.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8203765749931335, "reasoning_key": "robustness_norm", "reasoning_sim": 0.8594845533370972}]}, {"review_id": "0JqIoIaQZS", "paper_id": "y9SV71G8fP", "paper_title": "Explore Briefly, Then Decide: Mitigating LLM Overthinking via Cumulative Entropy Regulation", "decision": "Reject", "summary": "The reviewer systematically challenges the paper's claims of generality, causal mechanism, and incentive compatibility by identifying gaps in experimental breadth, lack of causal diagnostics for TECA, and insufficient justification for CER design. The evaluation focuses on converting correlational evidence into causal and generalizable proof through specific requested experiments.", "units": [{"unit_index": 0, "inspected_object": "Experimental corpus (model scale and diversity)", "observation": "Only Qwen3-4B and Qwen3-8B are used.", "reasoning": "GRPO is sensitive to backbone initialization and emergent base-model bias; a two-model sample from the same family is insufficient to support claims of generality.", "judgment": "The claim that the effect is a general property of 'overthinking' is unsupported due to limited experimental breadth.", "valence": "negative", "suggested_improvement": "Include more models from diverse families to test for transferability.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8562976717948914, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.7866432070732117}, {"unit_index": 1, "inspected_object": "TECA metric evidentiary basis", "observation": "The paper establishes only correlation between TECA trajectories and response length, lacking causal diagnostics like intervention studies or ablations vs. simpler uncertainty surrogates.", "reasoning": "Correlation does not establish causation; without interventions, it is unclear if TECA is the operative mechanism or merely a correlate of a surface symptom.", "judgment": "The paper has not demonstrated that TECA is the cause of the observed effect, only that it correlates with it.", "valence": "negative", "suggested_improvement": "Conduct an intervention study (e.g., injecting entropy spikes) and ablation against simpler uncertainty surrogates.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8078575730323792, "reasoning_key": "novelty_standard", "reasoning_sim": 0.6723954677581787}, {"unit_index": 2, "inspected_object": "CER reward construction design choices", "observation": "The CER design uses equal weighting with accuracy, exponential form, and applies TECA only at the end step, without mathematical justification.", "reasoning": "These choices appear hyperparameter-like and interchangeable; without justification, the paper fails to show CER is the 'right variable' rather than just one variable correlated with length.", "judgment": "The theoretical motivation for CER's specific design is missing, leaving its optimality unproven.", "valence": "negative", "suggested_improvement": "Provide theoretical analysis or justification for the specific reward design choices in CER.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "method_design", "object_sim": 0.7403612732887268, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7380545139312744}, {"unit_index": 3, "inspected_object": "Incentive compatibility of CER reward", "observation": "Rewarding TECA only on correct samples may encourage lucky short guesses.", "reasoning": "RL reward shaping should be analyzed for what it optimizes in expectation, including edge cases; this potential failure mode has not been ruled out.", "judgment": "The reward design may incentivize undesirable behavior (lucky short guesses), which constitutes a potential incentive misalignment.", "valence": "negative", "suggested_improvement": "Analyze the reward design to rule out the possibility that it encourages lucky short guesses.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8181760907173157, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7726569771766663}, {"unit_index": 4, "inspected_object": "Length reduction result interpretation", "observation": "The paper reports length reduction as a headline result but does not distinguish between eliminating alternative chains prematurely versus eliminating irrelevant exploration.", "reasoning": "Pass@k is the standard instrument to adjudicate whether the model becomes more decisive (good) or merely shallower (bad); without this control, the success metric is ambiguous.", "judgment": "The interpretation of length reduction as positive performance improvement is unsubstantiated without disambiguating the underlying cause.", "valence": "negative", "suggested_improvement": "Apply Pass@k analysis to determine if CER makes the model more decisive or merely shallower.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "theory", "object_sim": 0.7905824780464172, "reasoning_key": "design_justification", "reasoning_sim": 0.7288869023323059}, {"unit_index": 5, "inspected_object": "Marginal advantage of CER over CCoT", "observation": "For the 8B model, CER's advantage over CCoT is marginal on some metrics.", "reasoning": "Even within the limited corpus, the effect is not uniformly strong, raising questions about the robustness and significance of the claimed improvements.", "judgment": "The observed benefits of CER are marginal in certain contexts, weakening the case for its general superiority.", "valence": "negative", "suggested_improvement": "Investigate conditions under which CER provides significant advantages over baselines like CCoT.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.7876039147377014, "reasoning_key": "fair_comparison", "reasoning_sim": 0.8004074096679688}, {"unit_index": 6, "inspected_object": "TECA interpretability", "observation": "TECA is intuitively interpretable and simple to compute.", "reasoning": "While not a primary finding, this characteristic adds value to the metric's utility if its causal role can be established.", "judgment": "TECA has surface appeal and practical utility, contingent on further validation of its mechanistic role.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.7839909791946411, "reasoning_key": "construct_validity", "reasoning_sim": 0.7406070232391357}, {"unit_index": 7, "inspected_object": "Qualitative case study", "observation": "The paper includes a qualitative case study.", "reasoning": "Case studies provide illustrative evidence, though they do not substitute for statistical or causal rigor.", "judgment": "The qualitative case study offers supportive anecdotal evidence but does not resolve broader evidentiary gaps.", "valence": "positive", "suggested_improvement": null, "support_status": "reviewer_explicit", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.7645918726921082, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.6956888437271118}, {"unit_index": 8, "inspected_object": "Preliminaries section content", "observation": "The preliminaries repeat known GRPO and entropy definitions.", "reasoning": "Repetition of known material reduces novelty and efficiency of exposition.", "judgment": "The preliminary exposition lacks novelty and could be condensed.", "valence": "negative", "suggested_improvement": "Condense or streamline the preliminaries to avoid repeating known definitions.", "support_status": "reviewer_explicit", "confidence": "high", "object_key": "clarity", "object_sim": 0.7669891715049744, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7935777306556702}]}, {"review_id": "0KKZyCtaBH", "paper_id": "aJd636PxP1", "paper_title": "PAC-Bayesian Reinforcement Learning Trains Generalizable Policies", "decision": "Reject", "summary": "The reviewer conducts a coherence check, identifying a fundamental mismatch between the paper's theoretical claims and experimental evidence. While acknowledging theoretical novelty, the reviewer critiques the categorical error in bounding quantities, implausible experimental results, and unnecessary algorithmic complexity, concluding the paper lacks focus.", "units": [{"unit_index": 0, "inspected_object": "Theoretical contribution (bound derivation)", "observation": "The reviewer acknowledges the novel use of a bounded-differences formula for Markov chains and accepts that the PAC-Bayes conversion follows a standard recipe.", "reasoning": "The reviewer grants theoretical novelty early and does not dispute the mathematics itself, recognizing the specific technical components as valid.", "judgment": "The theoretical component is recognized as novel and mathematically plausible, though its positioning is questioned elsewhere.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8645056486129761, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7902702689170837}, {"unit_index": 1, "inspected_object": "Presentation of theory (Sections 4.2–4.4)", "observation": "The expository path is unclear, pointing to different PAC Bayes bound types until Equation 9 with overly brief justification.", "reasoning": "The reviewer evaluates the rhetorical structure, characterizing the section as a 'random walk within the literature' rather than a coherent argument.", "judgment": "The presentation is frustrating and lacks clear logical steps, hindering interpretability.", "valence": "negative", "suggested_improvement": "Clarify the expository path and justify each logical step leading to Equation 9.", "support_status": "memo_inferred", "confidence": "high", "object_key": "clarity", "object_sim": 0.8419417142868042, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7347499132156372}, {"unit_index": 2, "inspected_object": "Experimental results (SAC baselines)", "observation": "Reported SAC performance numbers are implausibly low compared to known results from Hui et al. (Double Gumbel Q-Learning, NeurIPS 2023).", "reasoning": "The reviewer uses external knowledge of expected performance profiles as a sanity check for implementation correctness.", "judgment": "The experimental results appear incorrect or uncalibrated against state-of-the-art benchmarks.", "valence": "negative", "suggested_improvement": "Verify SAC implementation and compare against established baselines like Hui et al.", "support_status": "memo_inferred", "confidence": "high", "object_key": "baselines_ablations", "object_sim": 0.8847371339797974, "reasoning_key": "merit_recognition", "reasoning_sim": 0.7932705283164978}, {"unit_index": 3, "inspected_object": "Algorithmic design choice: ε-Thompson exploration", "observation": "The method is a hybrid of ε-greedy and Thompson sampling that inherits properties of both but is 'actually none of the two.'", "reasoning": "The reviewer questions whether theoretical guarantees from either parent method transfer to this hybrid approach.", "judgment": "The theoretical basis for the exploration strategy is uncertain due to its hybrid nature.", "valence": "negative", "suggested_improvement": "Justify why theoretical guarantees from parent methods apply or provide new guarantees for the hybrid.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.8289912939071655, "reasoning_key": "novelty_standard", "reasoning_sim": 0.7658643126487732}, {"unit_index": 4, "inspected_object": "Algorithmic design choice: Actor freezing", "observation": "Actor freezing is flagged as not common practice; the reviewer asks if it is a prerequisite and where the fragility comes from.", "reasoning": "The reviewer suspects the method may be brittle and requires specific, non-standard conditions to work.", "judgment": "The reliance on actor freezing suggests potential brittleness and lack of robustness.", "valence": "negative", "suggested_improvement": "Explain the necessity of actor freezing and address concerns about fragility.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "method_design", "object_sim": 0.826011061668396, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7956216931343079}, {"unit_index": 5, "inspected_object": "Algorithmic design choice: Full-trajectory collection", "observation": "Collecting full trajectories with a frozen model yields independent samples, for which a simple PAC-Bayes bound would suffice.", "reasoning": "The reviewer challenges the necessity of the paper's complex machinery if simpler alternatives achieve the same statistical property.", "judgment": "The complexity of the proposed algorithm may be unnecessary given the data collection strategy.", "valence": "negative", "suggested_improvement": "Justify why the complex machinery is needed over simple first-visit Monte Carlo estimates.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.8141165971755981, "reasoning_key": "novelty_standard", "reasoning_sim": 0.8395183682441711}, {"unit_index": 6, "inspected_object": "Theoretical framing: Bounded quantity vs. Bellman error", "observation": "The paper bounds the signed difference between population and empirical estimates of return, but compares against bounds for Bellman error in L₂ space.", "reasoning": "These are two different quantities with separate motivations; conflating them means the bound cannot guarantee contraction mapping for uncertainty-aware policy search.", "judgment": "There is a fundamental category error in the theoretical positioning that diffuses into the whole storyline.", "valence": "negative", "suggested_improvement": "Reposition the contribution to correctly align the bounded quantity with appropriate comparisons.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8861873745918274, "reasoning_key": "robustness_norm", "reasoning_sim": 0.739931583404541}, {"unit_index": 7, "inspected_object": "Experimental alignment with claimed purpose", "observation": "The paper attempts to address tightest PAC-Bayes bound, highest performance with guarantees, and directed exploration simultaneously.", "reasoning": "The reviewer argues that dense-reward tasks do not require directed exploration, confounding the comparison, and that the PAC-Bayes improvement stems from task choice rather than method superiority.", "judgment": "The paper attempts to solve multiple difficult problems and ends up solving none of them effectively.", "valence": "negative", "suggested_improvement": "Focus on a single coherent goal and ensure experiments match that specific claim.", "support_status": "memo_inferred", "confidence": "high", "object_key": "empirical_scope", "object_sim": 0.8259750604629517, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7545158267021179}, {"unit_index": 8, "inspected_object": "Maximum-entropy confound in PAC-Bayes bound", "observation": "SAC optimizes a maximum-entropy objective, so the critic is trained on a reward including an entropy bonus.", "reasoning": "If the bound is computed on max-ent reward, it cannot conclude much because the entropy score is generated from the model itself, creating a circularity.", "judgment": "It is unclear what the bound actually certifies due to the confusion between learned and task objectives.", "valence": "negative", "suggested_improvement": "Clarify whether the bound is computed on environment reward or max-ent reward and its implications.", "support_status": "memo_inferred", "confidence": "medium", "object_key": "theory", "object_sim": 0.8310267925262451, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7572368383407593}]}, {"review_id": "0KP6zIQUXQ", "paper_id": "iHsH4ejjuh", "paper_title": "Emergent SO(3)-Invariant Molecular Representations from Multimodal Alignment", "decision": "Reject", "summary": "The reviewer acknowledges the paper's conceptual framing and behavioral results but critiques the lack of mechanistic evidence for multimodal integration, the narrow scope of generalization tests, and the absence of practical downstream validation.", "units": [{"unit_index": 0, "inspected_object": "The causal attribution of SO(3) invariance to the 3D encoder versus SMILES dominance.", "observation": "The paper claims emergent invariance, but there is no modality disentanglement analysis to rule out that the model collapses to purely SMILES-based features.", "reasoning": "Without diagnostic evidence such as modality dropout or CKA similarity, it is impossible to distinguish whether the 3D encoder contributes information or acts as a decorative appendage while the model learns from text alone.", "judgment": "The interpretation of the mechanism is underdetermined by the current behavioral evidence.", "valence": "negative", "suggested_improvement": "Provide direct evidence via modality dropout, CKA similarity, or conformer variation tests to show the 3D encoder contributes beyond SMILES.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7979258298873901, "reasoning_key": "statistical_identifiability", "reasoning_sim": 0.7780665755271912}, {"unit_index": 1, "inspected_object": "The generalizability of invariance claims to real-world molecular geometry.", "observation": "The evaluation relies on rigid SO(3) rotations and small, rigid QM9 molecules, ignoring conformational flexibility, chirality, and drug-like systems.", "reasoning": "Real molecules exhibit conformational flexibility and chirality; if the model's invariance only holds for artificial rigid rotations, the claim of a scalable direction for molecular representations is limited in ecological validity.", "judgment": "The scope of the evidence is narrow and potentially unrepresentative of the chemical space where the method would be applied.", "valence": "negative", "suggested_improvement": "Test the model on datasets with conformational flexibility, chirality, or larger biomolecular/drug-like systems.", "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8169333338737488, "reasoning_key": "design_justification", "reasoning_sim": 0.7478992938995361}, {"unit_index": 2, "inspected_object": "The validation of representation quality against practical utility.", "observation": "The paper uses unsupervised metrics like retrieval accuracy and clustering (Davies-Bouldin index) as primary evidence of quality.", "reasoning": "Practical utility in chemistry requires tasks like conformer ranking, docking, or reactivity prediction; internal consistency metrics do not fully arbitrate representation quality for downstream scientific use.", "judgment": "The empirical evidence is insufficient to establish the practical utility of the learned representations.", "valence": "negative", "suggested_improvement": "Include more demanding downstream tasks such as conformer ranking, docking, or reactivity prediction.", "support_status": "memo_inferred", "confidence": "high", "object_key": "stats_metrics", "object_sim": 0.8450868725776672, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7943825721740723}, {"unit_index": 3, "inspected_object": "The explanatory narrative regarding how spatial cues are encoded.", "observation": "The paper attributes invariance to 'emergence' from contrastive objectives without detailing the specific representational strategies or attention patterns used.", "reasoning": "The term 'emergence' serves as a placeholder for a lack of mechanistic understanding; granular accounts (e.g., attention visualization, feature attribution) are needed to explain *how* the invariance is implemented rather than just observing that it exists.", "judgment": "The interpretability of the mechanism is lacking, leaving the causal story opaque.", "valence": "negative", "suggested_improvement": "Perform mechanistic interpretability analyses, such as attention visualization or feature attribution, to elucidate how spatial cues are encoded.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8026356101036072, "reasoning_key": "design_justification", "reasoning_sim": 0.814317524433136}]}, {"review_id": "0KRb1MTVor", "paper_id": "Lvkbe0CgmZ", "paper_title": "Mirror Mean-Field Langevin Dynamics", "decision": "Reject", "summary": "The reviewer evaluates the paper as technically sound and novel but critiques its weak motivation and lack of computational feasibility analysis. They propose extending the framework to solve open problems in neural network training, viewing the paper as a valuable tool for the community rather than just a theoretical extension.", "units": [{"unit_index": 0, "inspected_object": "The paper's motivation and problem-setting context", "observation": "The reviewer finds the motivation underdeveloped, noting that while the abstract mentions constrained domains, it lacks specific instances of problem classes unlocked by the analysis.", "reasoning": "A theoretical contribution in optimization is judged valuable based on the class of previously intractable problems it can address; without concrete examples (e.g., training weight-constrained neural networks), the practical reach remains unproven.", "judgment": "The paper's motivational scaffolding is weak despite technical soundness.", "valence": "negative", "suggested_improvement": "Show what new settings can be unlocked by their analysis, e.g. for training weight-constrained two-layer neural networks or for generative modeling.", "support_status": "memo_inferred", "confidence": "high", "object_key": "problem_framing", "object_sim": 0.8469677567481995, "reasoning_key": "robustness_norm", "reasoning_sim": 0.7994398474693298}, {"unit_index": 1, "inspected_object": "The computational cost and simulation feasibility of the mirror step (Step 5)", "observation": "The discretization cost of Step 5 is not analyzed, leaving unclear whether simulating this step is easier than simulating Brownian motion on the constraint set.", "reasoning": "If the mirror step is computationally expensive or difficult to simulate, the algorithmic advantage over projected MFLD or constrained Brownian motion is unclear; practical viability requires justification beyond theoretical elegance.", "judgment": "The algorithmic implementation's practical viability is unverified due to missing computational analysis.", "valence": "negative", "suggested_improvement": "Provide an informal discussion of why simulating the mirror step is easier than simulating a Brownian motion on the constraint set.", "support_status": "memo_inferred", "confidence": "high", "object_key": "compute_cost", "object_sim": 0.8301724791526794, "reasoning_key": "cost_benefit", "reasoning_sim": 0.805257260799408}, {"unit_index": 2, "inspected_object": "The potential application of MMFLD to end-to-end guarantees for two-layer network training", "observation": "The reviewer identifies that standard frameworks break uniform LSI due to unbounded layers, but bilevel reductions reduce the problem to MFLD on the unit sphere, which lacks discretization guarantees.", "reasoning": "MMFLD handles constrained domains with discretization guarantees; applying it after the bilevel reduction could provide the missing end-to-end guarantee, serving as a key enabler for this open problem.", "judgment": "The paper has significant potential to enable future work on two-layer net training, though this connection was not made by the authors.", "valence": "positive", "suggested_improvement": "Explore using MMFLD after the bilevel reduction to provide an end-to-end guarantee for training two-layer networks in the mean-field regime.", "support_status": "memo_inferred", "confidence": "high", "object_key": "method_design", "object_sim": 0.7966390252113342, "reasoning_key": "claim_evidence_match", "reasoning_sim": 0.6972756385803223}, {"unit_index": 3, "inspected_object": "The numerical illustration comparing MMFLD to projected MFLD", "observation": "The review notes the numerical illustration showing superiority over projected MFLD as a strength.", "reasoning": "Numerical evidence supports the claim of superiority, contributing to the assessment of the method's effectiveness, although it is considered insufficient alone for motivation.", "judgment": "The numerical results are a positive indicator of performance.", "valence": "positive", "suggested_improvement": null, "support_status": "memo_inferred", "confidence": "high", "object_key": "theory", "object_sim": 0.8322486877441406, "reasoning_key": "fair_comparison", "reasoning_sim": 0.7664219737052917}]}]}
