[{"paper": {"id": "2608.04505", "title": "K-EXAONE 2.0 Technical Report", "authors": ["Eunbi Choi", "Kibong Choi", "Sehyun Chun", "Seokhee Hong", "Junwon Hwang", "Hyojin Jeon", "Ahra Jo", "Hyunjik Jo", "Yeonsik Jo", "Minhyeok Jung", "Doyoung Kim", "Heegyu Kim", "Joonkee Kim", "Seonghwan Kim", "Soyeon Kim", "Sunkyoung Kim", "Yireun Kim", "Yongil Kim", "Byungoh Ko", "Changhun Lee", "Dohaeng Lee", "Haeju Lee", "Jinsik Lee", "Kyungmin Lee", "Minwoo Lee", "Wonkee Lee", "Sangha Park", "Sungjune Park", "Kwangrok Ryoo", "Kijung Seo", "Minju Seo", "Yongwoo Song", "Sejong Yang", "Heuiyeen Yeen", "Stanley Jungkyu Choi", "Yemuk Choi", "Yongchan Chun", "Jiwon Ham", "Dasol Hong", "Sujeong Im", "Kijeong Jeon", "Gerrard Jeongwon Jo", "Hyeongjun Jo", "Yujin Jo", "Jiyeon Jung", "Naeun Kang", "Daeseong Kim", "Euisoon Kim", "Hayeon Kim", "Hyosang Kim", "Myoungshin Kim", "Unsol Kim", "Youchul Kim", "Chaeeun Lee", "ChaeYoon Lee", "Edward Hwayoung Lee", "Honglak Lee", "Hwansoo Lee", "Minkyung Lee", "Sangeun Lee", "Solji Lim", "Woohyung Lim", "Chanwoo Moon", "Jueun Mun", "Jimin Park", "Seojeong Park", "Yongmin Park", "Hyerin Seo", "Donghyeon Shin", "Donghyun Son", "Eunyong Son", "Kaehyun Um", "Sihoon Yang", "Chang En Yea", "Sihyuk Yi", "Kyungjae Yoo", "Chansik Yoon"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04505", "github": "", "website": ""}}, "category": "foundation_models", "tags": ["model"], "summary": "K-EXAONE 2.0 is LG AI Research's open-weight multilingual MoE foundation model (750B total/37B active params, 256K context) upcycled from its predecessor, with training aimed at strengthening reasoning, agentic coding, multilingual capability, and safety. It shows its largest gains in agentic coding and long-context understanding.", "reason": "A general-purpose flagship foundation model with agentic coding as one of several broad capabilities, not tied to a single downstream task.", "source": "crawl"}, {"paper": {"id": "2409.18048", "title": "Augmenting software engineering with AI - The ai4se taxonomy and its use", "authors": ["Ina K. Schieferdecker"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2409.18048", "github": "", "website": ""}}, "category": "studies", "tags": ["survey"], "summary": "Surveys the integration of generative and agentic AI into model-driven software engineering, introducing the 'ai4se' taxonomy to classify AI applications in SE and proposing a 'big models' vision plus a pair-modelling human-AI collaboration paradigm.", "reason": "A field-wide survey whose object is AI-augmented software engineering practice itself, not a proposed task-performing agent.", "source": "crawl"}, {"paper": {"id": "2510.22254", "title": "Ten Simple Rules for AI-Assisted Coding in Science", "authors": ["Eric W. Bridgeford", "Iain Campbell", "Zijao Chen", "Zhicheng Lin", "Harrison Ritz", "Joachim Vandekerckhove", "Russell A. Poldrack"], "venue": "arXiv 2025/10", "category": "", "published": "2025-10-31", "links": {"paper": "https://arxiv.org/abs/2510.22254", "github": "", "website": ""}}, "category": "studies", "tags": ["position"], "summary": "This paper offers ten practical guidelines for using AI coding assistants responsibly in scientific software development, covering problem framing, context management, testing, and code quality.", "reason": "A position/opinion piece about the practice of AI-assisted coding, not a proposed agent or task, matching the studies leaf.", "source": "crawl"}, {"paper": {"id": "2608.04148", "title": "AgentForge: An Immersive Role-Playing Platform for Learning Agentic Software Engineering", "authors": ["Zihan Fang", "Yueke Zhang", "Yu Huang"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2608.04148", "github": "", "website": ""}}, "category": "studies", "tags": ["empirical"], "summary": "AgentForge is an immersive learning platform where novice developers take on one role (planner, patch author, reviewer, tester) in a multi-agent code-repair workflow alongside AI agents performing the rest, studied with 37 novices.", "reason": "An empirical human-agent interaction study of learning to collaborate with agentic SE tools, not a task-performing agent method, fits studies.", "source": "crawl"}, {"paper": {"id": "2608.04661", "title": "An Exploratory Study of Agent Plans for Agentic AI Coding Tools in Open-Source Software", "authors": ["Muhammad Auwal Abubakar", "Seyedmoein Mohsenimofidi", "Jai Lal Lulla", "Jie M. Zhang", "Christoph Treude", "Sebastian Baltes", "Matthias Galster"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04661", "github": "", "website": ""}}, "category": "studies", "tags": ["empirical"], "summary": "An exploratory empirical study mining 36,710 GitHub repositories to identify and characterize 'Agent Plan' Markdown files used to guide agentic AI coding tools like Claude Code. Finds these task-oriented artifacts support maintenance, design, and testing work with concrete implementation guidance.", "reason": "Object of study is the agent artifacts/practices themselves, not a proposed agent performing a task, so it fits the studies/empirical off-axis branch.", "source": "crawl"}, {"paper": {"id": "2608.05116", "title": "Characterizing Visual Accessibility Issues in AI Developer Tools: An Empirical Study", "authors": ["Sabrina Haque", "Christoph Csallner"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.05116", "github": "", "website": ""}}, "category": "studies", "tags": ["empirical"], "summary": "An empirical study analyzing GitHub issues and forum discussions across five AI coding tool ecosystems (Copilot, Cursor, Claude Code, Codex, OpenCode) to characterize visual accessibility barriers faced by blind, low-vision, and color-vision-deficient developers.", "reason": "An empirical/behavioral study of AI developer tool ecosystems rather than a task-performing agent, so it belongs to studies.", "source": "crawl"}, {"paper": {"id": "2509.25465", "title": "BloomAPR: A Bloom's Taxonomy-based Framework for Assessing the Capabilities of LLM-Powered APR Solutions", "authors": ["Yinghang Ma", "Jiho Shin", "Leuson Da Silva", "Zhen Ming", "Jiang", "Song Wang", "Foutse Khomh", "Shin Hwei Tan"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2509.25465", "github": "", "website": ""}}, "category": "software_debugging", "tags": ["benchmark"], "summary": "BloomAPR is a Bloom's Taxonomy-based dynamic evaluation framework that assesses LLM-powered automated program repair solutions across progressively complex reasoning levels using Defects4J.", "reason": "Evaluates automated program repair agents, which is issue resolution/APR, fitting software_debugging plus a benchmark contribution.", "source": "crawl"}, {"paper": {"id": "2608.04682", "title": "Active-SWE: Benchmarking Coding Agents for Proactive Bug Fixing without Issue Reports", "authors": ["Haobin Li", "Ping Deng", "Weizhong Qian", "Liang Jiang", "Zhenyu Huang", "Mouxing Yang", "Xi Peng"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04682", "github": "", "website": ""}}, "category": "software_debugging", "tags": ["benchmark"], "summary": "Active-SWE is a benchmark of 1,663 tasks evaluating coding agents on proactively discovering and repairing multiple bugs in codebases without pre-existing issue reports, across six bug categories and eight languages. It finds state-of-the-art coding agents struggle with this proactive bug-fixing setting.", "reason": "Task is repairing real bugs in code (issue resolution/repair), matching software_debugging despite the proactive discovery framing.", "source": "crawl"}, {"paper": {"id": "2608.04804", "title": "Scrouting: Cost-Aware Routing of Coding Agents by Scouting the Repository First", "authors": ["Ishaan Bhola", "Adithyan Krishnan", "Mukunda NS"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04804", "github": "", "website": ""}}, "category": "software_debugging", "tags": ["model"], "summary": "SuperScout is a cost-aware router for repository-level issue-resolution agents that first has a 7B searcher scout the repository and produce a sandbox-verified handoff, then routes tasks to one of four frontier fixer models. On SWE-bench Pro it matches the best single model's solve rate at roughly a fifth of the cost.", "reason": "Serves SWE-bench-style issue resolution, so software_debugging per benchmark routing.", "source": "crawl"}, {"paper": {"id": "2608.05144", "title": "Argus: A General-Purpose Agentic Runtime for Long-Horizon Reasoning", "authors": ["Boxiu Li", "Zimo Wen", "Yijia Fan", "Junxiang Lei", "Sufeng Guo", "Jiaao Wu", "Ruize Tang", "Mukai Li", "Yifei Shen", "Xiaoyu Chen", "Wanbo Zhang", "Runjing Gu", "Yifei Gao", "Yuheng Wu", "Xuyao Huang", "Zelong Zhao", "Jiachen Zhang", "Shibo Hu", "Hangxi Guo", "Yilin Chen", "Yuzhe Zhang", "Fan Yang", "Chuan Wen", "Xian Zhang", "Xuanhe Zhou", "Zhijie Deng"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.05144", "github": "", "website": ""}}, "category": "software_debugging", "tags": [], "summary": "Argus is a persistent, self-evolving agentic runtime with Manager/Planner/Engineer/Reviewer roles that performs long-horizon missions with verification-gated self-evolution, achieving strong results on SWE-Bench Pro among other benchmarks like math synthesis and GPU-kernel tasks.", "reason": "Its most detailed and headline evaluation is SWE-Bench Pro issue resolution, so it routes to software_debugging despite broader generalist framing.", "source": "crawl"}, {"paper": {"id": "2601.03808", "title": "From Brute Force to Semantic Insight: Performance-Guided Data Transformation Design with LLMs", "authors": ["Usha Shrestha", "Dmitry Ignatov", "Radu Timofte"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2601.03808", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["training-data"], "summary": "NNGPT fine-tunes LLMs via LoRA on a repository of over 6,000 empirically-scored PyTorch data-augmentation functions to autonomously generate performance-optimal code transformations without brute-force search.", "reason": "The task's ultimate purpose is producing performance-tuned code (augmentation functions), placing it in software_code_generation.", "source": "crawl"}, {"paper": {"id": "2604.17529", "title": "Automated Logging Is Language-Sensitive: A Multilingual Benchmark and Empirical Study of LLMs", "authors": ["Renyi Zhong", "Yichen Li", "Yulun Wu", "Jinxi Kuang", "Yintong Huo", "Michael R. Lyu"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2604.17529", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["benchmark", "empirical"], "summary": "MultiLogBench is a multilingual benchmark and empirical study of automated logging-statement generation across six programming languages, covering site localization, API/severity selection, message generation, and variable recovery. It shows logging quality is highly language-sensitive and model rankings are unstable except at the top tier.", "reason": "Generating logging statements at specific code sites is a repo-aware code-completion task, matching software_code_generation.", "source": "crawl"}, {"paper": {"id": "2605.12153", "title": "CIDR: A Large-Scale Industrial Source Code Dataset for Software Engineering Research", "authors": ["Vladislav Savenkov"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2605.12153", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["training-data"], "summary": "CIDR is a large-scale curated dataset of 4,225 industrial repositories (832M raw LOC) across 75 languages with version history and engineering metadata, intended to support code intelligence research. The authors demonstrate its use by fine-tuning a 3B code LM and measuring gains on enterprise code.", "reason": "A resource dataset whose downstream purpose is training/improving code-generating models, so it follows the code-generation task it serves.", "source": "crawl"}, {"paper": {"id": "2606.27733", "title": "BashCoder-R1: Towards Robust and Explainable Bash Code Generation with Robustness-Aware Group Relative Policy Optimization", "authors": ["Lei Yu", "Peng Wang", "Jia Xu", "Jingyuan Zhang", "Xin Wang", "Jiajia Ma", "Li Yang", "Changzhi Deng", "Zenghua Wang", "Fengjun Zhang"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2606.27733", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["benchmark", "model"], "summary": "BashCoder-R1 combines continual pretraining, chain-of-thought SFT, and a robustness-aware GRPO reinforcement learning stage to generate syntactically correct and robust Bash scripts, evaluated on the new BashBench benchmark of 952 real-world tasks.", "reason": "The task is producing Bash code from specification, which is code generation/completion at script level, not execution as agentic action.", "source": "crawl"}, {"paper": {"id": "2608.04336", "title": "COMPAS: Difficulty-Aware Joint Search for Optimizing Code Generation", "authors": ["Jingzhi Gong", "Jie M. Zhang", "Gunel Jahangirova", "Dong Huang", "Mohammad Reza Mousavi", "Mark Harman"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04336", "github": "", "website": ""}}, "category": "software_code_generation", "tags": [], "summary": "COMPAS jointly optimizes model choice, prompt, and decoding settings for code generation by learning difficulty-aware quality-cost fronts and routing tasks to the matching configuration. It raises pass@1 on LiveCodeBench while cutting cost, and also improves SWE-bench resolution rates.", "reason": "Optimizes LLM configurations for the code-generation task, with LiveCodeBench as the primary benchmark, per software_code_generation.", "source": "crawl"}, {"paper": {"id": "2608.04975", "title": "SciCode-Verified: How Benchmark Defects Underestimated the Scientific-Coding Ability of Language Models", "authors": ["Sihan Hu", "Lyuhan Huang", "Youjin Deng", "Kun Chen"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04975", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["benchmark"], "summary": "SciCode-Verified is a corrected version of the SciCode benchmark, fixing 263 audited defects that caused correct scientific-code solutions implementing research-level physics/math theory to be wrongly rejected. Re-evaluation on the corrected benchmark shows frontier models are far more proficient at scientific coding than previously measured.", "reason": "The benchmark measures LLMs' ability to generate working numerical code from scientific specifications, a code-generation task.", "source": "crawl"}, {"paper": {"id": "2602.06709", "title": "Using Large Language Models to Support Automation of Failure Management in CI/CD Pipelines: A Case Study in SAP HANA", "authors": ["Duong Bui", "Stefan Grintz", "Alexander Berndt", "Thomas Bach"], "venue": "arXiv 2026/02", "category": "", "published": "2026-02-06", "links": {"paper": "https://arxiv.org/abs/2602.06709", "github": "", "website": ""}}, "category": "software_infrastructure", "tags": ["empirical"], "summary": "A case study evaluating an LLM-based system for automating CI/CD pipeline failure management at SAP HANA, testing its ability to localize errors and propose exact fixes. Domain knowledge from historical failures substantially boosts accuracy, reaching 92.1% exact solutions.", "reason": "CI/CD failure diagnosis and repair is enabling work in service of the codebase, matching software_infrastructure.", "source": "crawl"}, {"paper": {"id": "2608.04270", "title": "CURATE: Leveraging LLM Agents to Compose, Catalog, and Deploy Reproducible Workflows", "authors": ["Nolan Cutler", "Chia-Chen Kuo", "Nanda Velugoti", "Kathryn Newhart", "Renato Figueiredo"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2608.04270", "github": "", "website": ""}}, "category": "software_infrastructure", "tags": [], "summary": "CURATE is a human-in-the-loop multi-agent LLM system that composes, catalogs, reuses, and deploys reproducible computational workflows across their full lifecycle, demonstrated on scientific and applied workflow benchmarks.", "reason": "The agent's focus is enabling deployment, cataloging, and reuse of code modules, which matches the environment/CI-CD enabling activity.", "source": "crawl"}, {"paper": {"id": "2512.02567", "title": "Feedback Loops and Code Perturbations in LLM-based Software Engineering: A Case Study on a C-to-Rust Translation System", "authors": ["Martin Weiss", "Jesko Hecking-Harbusch", "Jochen Quante", "Matthias Woehrle"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2512.02567", "github": "", "website": ""}}, "category": "software_maintenance", "tags": ["empirical"], "summary": "A case study on a generate-and-check C-to-Rust translation system, examining how automated feedback loops, LLM choice, and behavior-preserving code perturbations affect translation success and robustness.", "reason": "Studies one specific code-translation (migration) system's design variables, so it routes to software_maintenance plus empirical tag rather than field-wide studies.", "source": "crawl"}, {"paper": {"id": "2604.07341", "title": "ReCodeAgent: A Multi-agent Workflow for Language-Agnostic Translation and Validation of Large-Scale Repositories", "authors": ["Ali Reza Ibrahimzada", "Brandon Paulsen", "Daniel Kroening", "Reyhaneh Jabbarvand"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2604.07341", "github": "", "website": ""}}, "category": "software_maintenance", "tags": [], "summary": "ReCodeAgent is a multi-agent system that autonomously translates and validates entire repositories across programming languages, using language-specific tools per PL. It outperforms prior neuro-symbolic and agentic baselines on 118 real-world projects across 6 languages.", "reason": "Cross-language repository translation is behavior-preserving code evolution, matching software_maintenance.", "source": "crawl"}, {"paper": {"id": "2608.04611", "title": "The Order Is the Guarantee: Verifier-Budgeted Code Deletion with Static-First Learned Proposals", "authors": ["Ruitong Li", "Binjie Guo", "Aisheng Mo", "Guowei Su", "Han Wang", "Jie Li", "Ru Zhang"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04611", "github": "", "website": ""}}, "category": "software_maintenance", "tags": [], "summary": "Formulates redundant-code deletion as proposal scheduling, where a ranker orders single-statement deletion candidates and an execution suite verifies behavior preservation under a bounded verifier budget. Evaluated on MBPP/MBPP+, it improves verified dead-code removal while auditably bounding verifier calls.", "reason": "Behavior-preserving removal of redundant code is refactoring/technical-debt reduction, matching software_maintenance.", "source": "crawl"}, {"paper": {"id": "2605.13138", "title": "A Comprehensive Evaluation of Code Language Models for Security Patch Detection", "authors": ["Nils Loose", "Joseph Bienhüls", "Kristoffer Hempel", "Felix Mächtle", "Thomas Eisenbarth"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2605.13138", "github": "", "website": ""}}, "category": "software_security", "tags": ["benchmark", "empirical"], "summary": "A rigorous re-evaluation of code language models (125M–80B parameters) for detecting vulnerability-fixing commits, consolidating 20 datasets under a unified, leakage-controlled framework. Finds model capacity gives limited gains and every fine-tuned model misses most fixes at strict false-positive budgets.", "reason": "Vulnerability/security-fix detection on code is the served task, matching software_security; released unified evaluation framework counts as a benchmark.", "source": "crawl"}, {"paper": {"id": "2608.04217", "title": "Neuro-Symbolic Proof-of-Vulnerability Generation with Open-Weight Models", "authors": ["Yu Nong", "Haipeng Cai"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2608.04217", "github": "", "website": ""}}, "category": "software_security", "tags": [], "summary": "POVGEN is a neuro-symbolic framework that localizes vulnerable code, performs reachability analysis, and uses LLM-guided constraint reasoning with an SMT solver to automatically generate Proof-of-Vulnerability inputs. It outperforms fuzzing and symbolic execution on benchmarks and real CVEs.", "reason": "Automated multi-step vulnerability triggering/validation pipeline serves the security-assurance activity on code.", "source": "crawl"}, {"paper": {"id": "2608.04783", "title": "RepoProbe: Benchmarking Architecture-Aware Repository Comprehension with Checklists", "authors": ["Yuexi Yang", "Alyssa Wu", "Ji Luo", "Richeng Xuan", "Zhichao Hu", "Yuhong Liu", "Zhen Qin"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04783", "github": "", "website": ""}}, "category": "software_comprehension", "tags": ["benchmark"], "summary": "RepoProbe is a benchmark for repository-level code comprehension that uses GitHub Discussions-derived open-ended architectural Q&A instead of bug reports, paired with a checklist-based verification protocol to reduce LLM-judge variance. It reveals models exhibit 'edit bias', prematurely proposing code changes over genuine architectural understanding.", "reason": "Serves the code-comprehension lifecycle activity (repo QA) via a dedicated benchmark, per software_comprehension leaf.", "source": "crawl"}, {"paper": {"id": "2608.05141", "title": "OctoLong: Mid-Training On Cross-Repository Code Contexts Enhances Long-Context Modeling", "authors": ["Indraneil Paul", "Falko Helm", "Goran Glavaš", "Iryna Gurevych"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.05141", "github": "", "website": ""}}, "category": "software_comprehension", "tags": ["training-data", "model"], "summary": "OctoLong is a pipeline using AST parsing, language servers, and package managers to curate dependency-rich, cross-repository code contexts for mid-training, producing long-context LMs that improve repository-level code understanding and downstream agentic coding tasks.", "reason": "A resource paper (training data + model) serving repository-level code comprehension for coding agents.", "source": "crawl"}]
Reply with commands, one per line:
/approve all·/approve 1,3-5·/reject 2·/edit 3 category=world_terminal tags=benchmark(edit implies approve;tags=-clears tags). Valid category keys: see taxonomy.json.Auto-skipped 107 out-of-scope; 0 failed (retried next run).
1. K-EXAONE 2.0 Technical Report
Eunbi Choi, Kibong Choi, Sehyun Chun, et al. · arXiv 2026/08 · paper
proposed:
foundation_models· tags:model2. Augmenting software engineering with AI - The ai4se taxonomy and its use
Ina K. Schieferdecker · arXiv 2026/08 · paper
proposed:
studies· tags:survey3. Ten Simple Rules for AI-Assisted Coding in Science
Eric W. Bridgeford, Iain Campbell, Zijao Chen, et al. · arXiv 2025/10 · paper
proposed:
studies· tags:position4. AgentForge: An Immersive Role-Playing Platform for Learning Agentic Software Engineering
Zihan Fang, Yueke Zhang, Yu Huang · arXiv 2026/08 · paper
proposed:
studies· tags:empirical5. An Exploratory Study of Agent Plans for Agentic AI Coding Tools in Open-Source Software
Muhammad Auwal Abubakar, Seyedmoein Mohsenimofidi, Jai Lal Lulla, et al. · arXiv 2026/08 · paper
proposed:
studies· tags:empirical6. Characterizing Visual Accessibility Issues in AI Developer Tools: An Empirical Study
Sabrina Haque, Christoph Csallner · arXiv 2026/08 · paper
proposed:
studies· tags:empirical7. BloomAPR: A Bloom's Taxonomy-based Framework for Assessing the Capabilities of LLM-Powered APR Solutions
Yinghang Ma, Jiho Shin, Leuson Da Silva, et al. · arXiv 2026/08 · paper
proposed:
software_debugging· tags:benchmark8. Active-SWE: Benchmarking Coding Agents for Proactive Bug Fixing without Issue Reports
Haobin Li, Ping Deng, Weizhong Qian, et al. · arXiv 2026/08 · paper
proposed:
software_debugging· tags:benchmark9. Scrouting: Cost-Aware Routing of Coding Agents by Scouting the Repository First
Ishaan Bhola, Adithyan Krishnan, Mukunda NS · arXiv 2026/08 · paper
proposed:
software_debugging· tags:model10. Argus: A General-Purpose Agentic Runtime for Long-Horizon Reasoning
Boxiu Li, Zimo Wen, Yijia Fan, et al. · arXiv 2026/08 · paper
proposed:
software_debugging· tags: none11. From Brute Force to Semantic Insight: Performance-Guided Data Transformation Design with LLMs
Usha Shrestha, Dmitry Ignatov, Radu Timofte · arXiv 2026/08 · paper
proposed:
software_code_generation· tags:training-data12. Automated Logging Is Language-Sensitive: A Multilingual Benchmark and Empirical Study of LLMs
Renyi Zhong, Yichen Li, Yulun Wu, et al. · arXiv 2026/08 · paper
proposed:
software_code_generation· tags:benchmarkempirical13. CIDR: A Large-Scale Industrial Source Code Dataset for Software Engineering Research
Vladislav Savenkov · arXiv 2026/08 · paper
proposed:
software_code_generation· tags:training-data14. BashCoder-R1: Towards Robust and Explainable Bash Code Generation with Robustness-Aware Group Relative Policy Optimization
Lei Yu, Peng Wang, Jia Xu, et al. · arXiv 2026/08 · paper
proposed:
software_code_generation· tags:benchmarkmodel15. COMPAS: Difficulty-Aware Joint Search for Optimizing Code Generation
Jingzhi Gong, Jie M. Zhang, Gunel Jahangirova, et al. · arXiv 2026/08 · paper
proposed:
software_code_generation· tags: none16. SciCode-Verified: How Benchmark Defects Underestimated the Scientific-Coding Ability of Language Models
Sihan Hu, Lyuhan Huang, Youjin Deng, et al. · arXiv 2026/08 · paper
proposed:
software_code_generation· tags:benchmark17. Using Large Language Models to Support Automation of Failure Management in CI/CD Pipelines: A Case Study in SAP HANA
Duong Bui, Stefan Grintz, Alexander Berndt, et al. · arXiv 2026/02 · paper
proposed:
software_infrastructure· tags:empirical18. CURATE: Leveraging LLM Agents to Compose, Catalog, and Deploy Reproducible Workflows
Nolan Cutler, Chia-Chen Kuo, Nanda Velugoti, et al. · arXiv 2026/08 · paper
proposed:
software_infrastructure· tags: none19. Feedback Loops and Code Perturbations in LLM-based Software Engineering: A Case Study on a C-to-Rust Translation System
Martin Weiss, Jesko Hecking-Harbusch, Jochen Quante, et al. · arXiv 2026/08 · paper
proposed:
software_maintenance· tags:empirical20. ReCodeAgent: A Multi-agent Workflow for Language-Agnostic Translation and Validation of Large-Scale Repositories
Ali Reza Ibrahimzada, Brandon Paulsen, Daniel Kroening, et al. · arXiv 2026/08 · paper
proposed:
software_maintenance· tags: none21. The Order Is the Guarantee: Verifier-Budgeted Code Deletion with Static-First Learned Proposals
Ruitong Li, Binjie Guo, Aisheng Mo, et al. · arXiv 2026/08 · paper
proposed:
software_maintenance· tags: none22. A Comprehensive Evaluation of Code Language Models for Security Patch Detection
Nils Loose, Joseph Bienhüls, Kristoffer Hempel, et al. · arXiv 2026/08 · paper
proposed:
software_security· tags:benchmarkempirical23. Neuro-Symbolic Proof-of-Vulnerability Generation with Open-Weight Models
Yu Nong, Haipeng Cai · arXiv 2026/08 · paper
proposed:
software_security· tags: none24. RepoProbe: Benchmarking Architecture-Aware Repository Comprehension with Checklists
Yuexi Yang, Alyssa Wu, Ji Luo, et al. · arXiv 2026/08 · paper
proposed:
software_comprehension· tags:benchmark25. OctoLong: Mid-Training On Cross-Repository Code Contexts Enhances Long-Context Modeling
Indraneil Paul, Falko Helm, Goran Glavaš, et al. · arXiv 2026/08 · paper
proposed:
software_comprehension· tags:training-datamodelmachine payload (do not edit)
[{"paper": {"id": "2608.04505", "title": "K-EXAONE 2.0 Technical Report", "authors": ["Eunbi Choi", "Kibong Choi", "Sehyun Chun", "Seokhee Hong", "Junwon Hwang", "Hyojin Jeon", "Ahra Jo", "Hyunjik Jo", "Yeonsik Jo", "Minhyeok Jung", "Doyoung Kim", "Heegyu Kim", "Joonkee Kim", "Seonghwan Kim", "Soyeon Kim", "Sunkyoung Kim", "Yireun Kim", "Yongil Kim", "Byungoh Ko", "Changhun Lee", "Dohaeng Lee", "Haeju Lee", "Jinsik Lee", "Kyungmin Lee", "Minwoo Lee", "Wonkee Lee", "Sangha Park", "Sungjune Park", "Kwangrok Ryoo", "Kijung Seo", "Minju Seo", "Yongwoo Song", "Sejong Yang", "Heuiyeen Yeen", "Stanley Jungkyu Choi", "Yemuk Choi", "Yongchan Chun", "Jiwon Ham", "Dasol Hong", "Sujeong Im", "Kijeong Jeon", "Gerrard Jeongwon Jo", "Hyeongjun Jo", "Yujin Jo", "Jiyeon Jung", "Naeun Kang", "Daeseong Kim", "Euisoon Kim", "Hayeon Kim", "Hyosang Kim", "Myoungshin Kim", "Unsol Kim", "Youchul Kim", "Chaeeun Lee", "ChaeYoon Lee", "Edward Hwayoung Lee", "Honglak Lee", "Hwansoo Lee", "Minkyung Lee", "Sangeun Lee", "Solji Lim", "Woohyung Lim", "Chanwoo Moon", "Jueun Mun", "Jimin Park", "Seojeong Park", "Yongmin Park", "Hyerin Seo", "Donghyeon Shin", "Donghyun Son", "Eunyong Son", "Kaehyun Um", "Sihoon Yang", "Chang En Yea", "Sihyuk Yi", "Kyungjae Yoo", "Chansik Yoon"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04505", "github": "", "website": ""}}, "category": "foundation_models", "tags": ["model"], "summary": "K-EXAONE 2.0 is LG AI Research's open-weight multilingual MoE foundation model (750B total/37B active params, 256K context) upcycled from its predecessor, with training aimed at strengthening reasoning, agentic coding, multilingual capability, and safety. It shows its largest gains in agentic coding and long-context understanding.", "reason": "A general-purpose flagship foundation model with agentic coding as one of several broad capabilities, not tied to a single downstream task.", "source": "crawl"}, {"paper": {"id": "2409.18048", "title": "Augmenting software engineering with AI - The ai4se taxonomy and its use", "authors": ["Ina K. Schieferdecker"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2409.18048", "github": "", "website": ""}}, "category": "studies", "tags": ["survey"], "summary": "Surveys the integration of generative and agentic AI into model-driven software engineering, introducing the 'ai4se' taxonomy to classify AI applications in SE and proposing a 'big models' vision plus a pair-modelling human-AI collaboration paradigm.", "reason": "A field-wide survey whose object is AI-augmented software engineering practice itself, not a proposed task-performing agent.", "source": "crawl"}, {"paper": {"id": "2510.22254", "title": "Ten Simple Rules for AI-Assisted Coding in Science", "authors": ["Eric W. Bridgeford", "Iain Campbell", "Zijao Chen", "Zhicheng Lin", "Harrison Ritz", "Joachim Vandekerckhove", "Russell A. Poldrack"], "venue": "arXiv 2025/10", "category": "", "published": "2025-10-31", "links": {"paper": "https://arxiv.org/abs/2510.22254", "github": "", "website": ""}}, "category": "studies", "tags": ["position"], "summary": "This paper offers ten practical guidelines for using AI coding assistants responsibly in scientific software development, covering problem framing, context management, testing, and code quality.", "reason": "A position/opinion piece about the practice of AI-assisted coding, not a proposed agent or task, matching the studies leaf.", "source": "crawl"}, {"paper": {"id": "2608.04148", "title": "AgentForge: An Immersive Role-Playing Platform for Learning Agentic Software Engineering", "authors": ["Zihan Fang", "Yueke Zhang", "Yu Huang"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2608.04148", "github": "", "website": ""}}, "category": "studies", "tags": ["empirical"], "summary": "AgentForge is an immersive learning platform where novice developers take on one role (planner, patch author, reviewer, tester) in a multi-agent code-repair workflow alongside AI agents performing the rest, studied with 37 novices.", "reason": "An empirical human-agent interaction study of learning to collaborate with agentic SE tools, not a task-performing agent method, fits studies.", "source": "crawl"}, {"paper": {"id": "2608.04661", "title": "An Exploratory Study of Agent Plans for Agentic AI Coding Tools in Open-Source Software", "authors": ["Muhammad Auwal Abubakar", "Seyedmoein Mohsenimofidi", "Jai Lal Lulla", "Jie M. Zhang", "Christoph Treude", "Sebastian Baltes", "Matthias Galster"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04661", "github": "", "website": ""}}, "category": "studies", "tags": ["empirical"], "summary": "An exploratory empirical study mining 36,710 GitHub repositories to identify and characterize 'Agent Plan' Markdown files used to guide agentic AI coding tools like Claude Code. Finds these task-oriented artifacts support maintenance, design, and testing work with concrete implementation guidance.", "reason": "Object of study is the agent artifacts/practices themselves, not a proposed agent performing a task, so it fits the studies/empirical off-axis branch.", "source": "crawl"}, {"paper": {"id": "2608.05116", "title": "Characterizing Visual Accessibility Issues in AI Developer Tools: An Empirical Study", "authors": ["Sabrina Haque", "Christoph Csallner"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.05116", "github": "", "website": ""}}, "category": "studies", "tags": ["empirical"], "summary": "An empirical study analyzing GitHub issues and forum discussions across five AI coding tool ecosystems (Copilot, Cursor, Claude Code, Codex, OpenCode) to characterize visual accessibility barriers faced by blind, low-vision, and color-vision-deficient developers.", "reason": "An empirical/behavioral study of AI developer tool ecosystems rather than a task-performing agent, so it belongs to studies.", "source": "crawl"}, {"paper": {"id": "2509.25465", "title": "BloomAPR: A Bloom's Taxonomy-based Framework for Assessing the Capabilities of LLM-Powered APR Solutions", "authors": ["Yinghang Ma", "Jiho Shin", "Leuson Da Silva", "Zhen Ming", "Jiang", "Song Wang", "Foutse Khomh", "Shin Hwei Tan"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2509.25465", "github": "", "website": ""}}, "category": "software_debugging", "tags": ["benchmark"], "summary": "BloomAPR is a Bloom's Taxonomy-based dynamic evaluation framework that assesses LLM-powered automated program repair solutions across progressively complex reasoning levels using Defects4J.", "reason": "Evaluates automated program repair agents, which is issue resolution/APR, fitting software_debugging plus a benchmark contribution.", "source": "crawl"}, {"paper": {"id": "2608.04682", "title": "Active-SWE: Benchmarking Coding Agents for Proactive Bug Fixing without Issue Reports", "authors": ["Haobin Li", "Ping Deng", "Weizhong Qian", "Liang Jiang", "Zhenyu Huang", "Mouxing Yang", "Xi Peng"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04682", "github": "", "website": ""}}, "category": "software_debugging", "tags": ["benchmark"], "summary": "Active-SWE is a benchmark of 1,663 tasks evaluating coding agents on proactively discovering and repairing multiple bugs in codebases without pre-existing issue reports, across six bug categories and eight languages. It finds state-of-the-art coding agents struggle with this proactive bug-fixing setting.", "reason": "Task is repairing real bugs in code (issue resolution/repair), matching software_debugging despite the proactive discovery framing.", "source": "crawl"}, {"paper": {"id": "2608.04804", "title": "Scrouting: Cost-Aware Routing of Coding Agents by Scouting the Repository First", "authors": ["Ishaan Bhola", "Adithyan Krishnan", "Mukunda NS"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04804", "github": "", "website": ""}}, "category": "software_debugging", "tags": ["model"], "summary": "SuperScout is a cost-aware router for repository-level issue-resolution agents that first has a 7B searcher scout the repository and produce a sandbox-verified handoff, then routes tasks to one of four frontier fixer models. On SWE-bench Pro it matches the best single model's solve rate at roughly a fifth of the cost.", "reason": "Serves SWE-bench-style issue resolution, so software_debugging per benchmark routing.", "source": "crawl"}, {"paper": {"id": "2608.05144", "title": "Argus: A General-Purpose Agentic Runtime for Long-Horizon Reasoning", "authors": ["Boxiu Li", "Zimo Wen", "Yijia Fan", "Junxiang Lei", "Sufeng Guo", "Jiaao Wu", "Ruize Tang", "Mukai Li", "Yifei Shen", "Xiaoyu Chen", "Wanbo Zhang", "Runjing Gu", "Yifei Gao", "Yuheng Wu", "Xuyao Huang", "Zelong Zhao", "Jiachen Zhang", "Shibo Hu", "Hangxi Guo", "Yilin Chen", "Yuzhe Zhang", "Fan Yang", "Chuan Wen", "Xian Zhang", "Xuanhe Zhou", "Zhijie Deng"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.05144", "github": "", "website": ""}}, "category": "software_debugging", "tags": [], "summary": "Argus is a persistent, self-evolving agentic runtime with Manager/Planner/Engineer/Reviewer roles that performs long-horizon missions with verification-gated self-evolution, achieving strong results on SWE-Bench Pro among other benchmarks like math synthesis and GPU-kernel tasks.", "reason": "Its most detailed and headline evaluation is SWE-Bench Pro issue resolution, so it routes to software_debugging despite broader generalist framing.", "source": "crawl"}, {"paper": {"id": "2601.03808", "title": "From Brute Force to Semantic Insight: Performance-Guided Data Transformation Design with LLMs", "authors": ["Usha Shrestha", "Dmitry Ignatov", "Radu Timofte"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2601.03808", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["training-data"], "summary": "NNGPT fine-tunes LLMs via LoRA on a repository of over 6,000 empirically-scored PyTorch data-augmentation functions to autonomously generate performance-optimal code transformations without brute-force search.", "reason": "The task's ultimate purpose is producing performance-tuned code (augmentation functions), placing it in software_code_generation.", "source": "crawl"}, {"paper": {"id": "2604.17529", "title": "Automated Logging Is Language-Sensitive: A Multilingual Benchmark and Empirical Study of LLMs", "authors": ["Renyi Zhong", "Yichen Li", "Yulun Wu", "Jinxi Kuang", "Yintong Huo", "Michael R. Lyu"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2604.17529", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["benchmark", "empirical"], "summary": "MultiLogBench is a multilingual benchmark and empirical study of automated logging-statement generation across six programming languages, covering site localization, API/severity selection, message generation, and variable recovery. It shows logging quality is highly language-sensitive and model rankings are unstable except at the top tier.", "reason": "Generating logging statements at specific code sites is a repo-aware code-completion task, matching software_code_generation.", "source": "crawl"}, {"paper": {"id": "2605.12153", "title": "CIDR: A Large-Scale Industrial Source Code Dataset for Software Engineering Research", "authors": ["Vladislav Savenkov"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2605.12153", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["training-data"], "summary": "CIDR is a large-scale curated dataset of 4,225 industrial repositories (832M raw LOC) across 75 languages with version history and engineering metadata, intended to support code intelligence research. The authors demonstrate its use by fine-tuning a 3B code LM and measuring gains on enterprise code.", "reason": "A resource dataset whose downstream purpose is training/improving code-generating models, so it follows the code-generation task it serves.", "source": "crawl"}, {"paper": {"id": "2606.27733", "title": "BashCoder-R1: Towards Robust and Explainable Bash Code Generation with Robustness-Aware Group Relative Policy Optimization", "authors": ["Lei Yu", "Peng Wang", "Jia Xu", "Jingyuan Zhang", "Xin Wang", "Jiajia Ma", "Li Yang", "Changzhi Deng", "Zenghua Wang", "Fengjun Zhang"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2606.27733", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["benchmark", "model"], "summary": "BashCoder-R1 combines continual pretraining, chain-of-thought SFT, and a robustness-aware GRPO reinforcement learning stage to generate syntactically correct and robust Bash scripts, evaluated on the new BashBench benchmark of 952 real-world tasks.", "reason": "The task is producing Bash code from specification, which is code generation/completion at script level, not execution as agentic action.", "source": "crawl"}, {"paper": {"id": "2608.04336", "title": "COMPAS: Difficulty-Aware Joint Search for Optimizing Code Generation", "authors": ["Jingzhi Gong", "Jie M. Zhang", "Gunel Jahangirova", "Dong Huang", "Mohammad Reza Mousavi", "Mark Harman"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04336", "github": "", "website": ""}}, "category": "software_code_generation", "tags": [], "summary": "COMPAS jointly optimizes model choice, prompt, and decoding settings for code generation by learning difficulty-aware quality-cost fronts and routing tasks to the matching configuration. It raises pass@1 on LiveCodeBench while cutting cost, and also improves SWE-bench resolution rates.", "reason": "Optimizes LLM configurations for the code-generation task, with LiveCodeBench as the primary benchmark, per software_code_generation.", "source": "crawl"}, {"paper": {"id": "2608.04975", "title": "SciCode-Verified: How Benchmark Defects Underestimated the Scientific-Coding Ability of Language Models", "authors": ["Sihan Hu", "Lyuhan Huang", "Youjin Deng", "Kun Chen"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04975", "github": "", "website": ""}}, "category": "software_code_generation", "tags": ["benchmark"], "summary": "SciCode-Verified is a corrected version of the SciCode benchmark, fixing 263 audited defects that caused correct scientific-code solutions implementing research-level physics/math theory to be wrongly rejected. Re-evaluation on the corrected benchmark shows frontier models are far more proficient at scientific coding than previously measured.", "reason": "The benchmark measures LLMs' ability to generate working numerical code from scientific specifications, a code-generation task.", "source": "crawl"}, {"paper": {"id": "2602.06709", "title": "Using Large Language Models to Support Automation of Failure Management in CI/CD Pipelines: A Case Study in SAP HANA", "authors": ["Duong Bui", "Stefan Grintz", "Alexander Berndt", "Thomas Bach"], "venue": "arXiv 2026/02", "category": "", "published": "2026-02-06", "links": {"paper": "https://arxiv.org/abs/2602.06709", "github": "", "website": ""}}, "category": "software_infrastructure", "tags": ["empirical"], "summary": "A case study evaluating an LLM-based system for automating CI/CD pipeline failure management at SAP HANA, testing its ability to localize errors and propose exact fixes. Domain knowledge from historical failures substantially boosts accuracy, reaching 92.1% exact solutions.", "reason": "CI/CD failure diagnosis and repair is enabling work in service of the codebase, matching software_infrastructure.", "source": "crawl"}, {"paper": {"id": "2608.04270", "title": "CURATE: Leveraging LLM Agents to Compose, Catalog, and Deploy Reproducible Workflows", "authors": ["Nolan Cutler", "Chia-Chen Kuo", "Nanda Velugoti", "Kathryn Newhart", "Renato Figueiredo"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2608.04270", "github": "", "website": ""}}, "category": "software_infrastructure", "tags": [], "summary": "CURATE is a human-in-the-loop multi-agent LLM system that composes, catalogs, reuses, and deploys reproducible computational workflows across their full lifecycle, demonstrated on scientific and applied workflow benchmarks.", "reason": "The agent's focus is enabling deployment, cataloging, and reuse of code modules, which matches the environment/CI-CD enabling activity.", "source": "crawl"}, {"paper": {"id": "2512.02567", "title": "Feedback Loops and Code Perturbations in LLM-based Software Engineering: A Case Study on a C-to-Rust Translation System", "authors": ["Martin Weiss", "Jesko Hecking-Harbusch", "Jochen Quante", "Matthias Woehrle"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2512.02567", "github": "", "website": ""}}, "category": "software_maintenance", "tags": ["empirical"], "summary": "A case study on a generate-and-check C-to-Rust translation system, examining how automated feedback loops, LLM choice, and behavior-preserving code perturbations affect translation success and robustness.", "reason": "Studies one specific code-translation (migration) system's design variables, so it routes to software_maintenance plus empirical tag rather than field-wide studies.", "source": "crawl"}, {"paper": {"id": "2604.07341", "title": "ReCodeAgent: A Multi-agent Workflow for Language-Agnostic Translation and Validation of Large-Scale Repositories", "authors": ["Ali Reza Ibrahimzada", "Brandon Paulsen", "Daniel Kroening", "Reyhaneh Jabbarvand"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2604.07341", "github": "", "website": ""}}, "category": "software_maintenance", "tags": [], "summary": "ReCodeAgent is a multi-agent system that autonomously translates and validates entire repositories across programming languages, using language-specific tools per PL. It outperforms prior neuro-symbolic and agentic baselines on 118 real-world projects across 6 languages.", "reason": "Cross-language repository translation is behavior-preserving code evolution, matching software_maintenance.", "source": "crawl"}, {"paper": {"id": "2608.04611", "title": "The Order Is the Guarantee: Verifier-Budgeted Code Deletion with Static-First Learned Proposals", "authors": ["Ruitong Li", "Binjie Guo", "Aisheng Mo", "Guowei Su", "Han Wang", "Jie Li", "Ru Zhang"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04611", "github": "", "website": ""}}, "category": "software_maintenance", "tags": [], "summary": "Formulates redundant-code deletion as proposal scheduling, where a ranker orders single-statement deletion candidates and an execution suite verifies behavior preservation under a bounded verifier budget. Evaluated on MBPP/MBPP+, it improves verified dead-code removal while auditably bounding verifier calls.", "reason": "Behavior-preserving removal of redundant code is refactoring/technical-debt reduction, matching software_maintenance.", "source": "crawl"}, {"paper": {"id": "2605.13138", "title": "A Comprehensive Evaluation of Code Language Models for Security Patch Detection", "authors": ["Nils Loose", "Joseph Bienhüls", "Kristoffer Hempel", "Felix Mächtle", "Thomas Eisenbarth"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2605.13138", "github": "", "website": ""}}, "category": "software_security", "tags": ["benchmark", "empirical"], "summary": "A rigorous re-evaluation of code language models (125M–80B parameters) for detecting vulnerability-fixing commits, consolidating 20 datasets under a unified, leakage-controlled framework. Finds model capacity gives limited gains and every fine-tuned model misses most fixes at strict false-positive budgets.", "reason": "Vulnerability/security-fix detection on code is the served task, matching software_security; released unified evaluation framework counts as a benchmark.", "source": "crawl"}, {"paper": {"id": "2608.04217", "title": "Neuro-Symbolic Proof-of-Vulnerability Generation with Open-Weight Models", "authors": ["Yu Nong", "Haipeng Cai"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-04", "links": {"paper": "https://arxiv.org/abs/2608.04217", "github": "", "website": ""}}, "category": "software_security", "tags": [], "summary": "POVGEN is a neuro-symbolic framework that localizes vulnerable code, performs reachability analysis, and uses LLM-guided constraint reasoning with an SMT solver to automatically generate Proof-of-Vulnerability inputs. It outperforms fuzzing and symbolic execution on benchmarks and real CVEs.", "reason": "Automated multi-step vulnerability triggering/validation pipeline serves the security-assurance activity on code.", "source": "crawl"}, {"paper": {"id": "2608.04783", "title": "RepoProbe: Benchmarking Architecture-Aware Repository Comprehension with Checklists", "authors": ["Yuexi Yang", "Alyssa Wu", "Ji Luo", "Richeng Xuan", "Zhichao Hu", "Yuhong Liu", "Zhen Qin"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.04783", "github": "", "website": ""}}, "category": "software_comprehension", "tags": ["benchmark"], "summary": "RepoProbe is a benchmark for repository-level code comprehension that uses GitHub Discussions-derived open-ended architectural Q&A instead of bug reports, paired with a checklist-based verification protocol to reduce LLM-judge variance. It reveals models exhibit 'edit bias', prematurely proposing code changes over genuine architectural understanding.", "reason": "Serves the code-comprehension lifecycle activity (repo QA) via a dedicated benchmark, per software_comprehension leaf.", "source": "crawl"}, {"paper": {"id": "2608.05141", "title": "OctoLong: Mid-Training On Cross-Repository Code Contexts Enhances Long-Context Modeling", "authors": ["Indraneil Paul", "Falko Helm", "Goran Glavaš", "Iryna Gurevych"], "venue": "arXiv 2026/08", "category": "", "published": "2026-08-05", "links": {"paper": "https://arxiv.org/abs/2608.05141", "github": "", "website": ""}}, "category": "software_comprehension", "tags": ["training-data", "model"], "summary": "OctoLong is a pipeline using AST parsing, language servers, and package managers to curate dependency-rich, cross-repository code contexts for mid-training, producing long-context LMs that improve repository-level code understanding and downstream agentic coding tasks.", "reason": "A resource paper (training data + model) serving repository-level code comprehension for coding agents.", "source": "crawl"}]