{"success":true,"type":"trending","date_or_query":"2026-09-10","count":10,"papers":[{"id":"2608.12564","title":"Scaling Automatic Research Agents via World Models","arxiv_id":"2608.12564","upvotes":437,"published_at":"2026-08-28T20:00:00.000Z","authors":["Xiyuan Yang","Sheikh Sarwar","Jingru Cheng"],"url":"https://huggingface.co/papers/2608.12564"},{"id":"2609.10522","title":"Show-Harness: Just a VLM Agent Can Play Robots","arxiv_id":"2609.10522","upvotes":134,"published_at":"2026-09-08T20:00:00.000Z","authors":["Yanzhe Chen","Zechen Bai","Zhijun Cao"],"url":"https://huggingface.co/papers/2609.10522"},{"id":"2609.08572","title":"AgentGrad: Intervention-guided Prompt Optimization for Multi Agent Systems","arxiv_id":"2609.08572","upvotes":89,"published_at":"2026-09-07T20:00:00.000Z","authors":["Jaewon Chu","Jinwoo Seo","Jaewon Cho"],"url":"https://huggingface.co/papers/2609.08572"},{"id":"2609.10540","title":"Programmable World Model","arxiv_id":"2609.10540","upvotes":85,"published_at":"2026-09-08T20:00:00.000Z","authors":["Zheng-Hui Huang","Guixu Lin","Jiacheng Lin"],"url":"https://huggingface.co/papers/2609.10540"},{"id":"2609.11042","title":"T1: Terminal Agent Reinforcement Learning for Long-Horizon Tasks","arxiv_id":"2609.11042","upvotes":46,"published_at":"2026-09-09T20:00:00.000Z","authors":["Junyao Yang","Yucheng Shi","Zhongzhi Li"],"url":"https://huggingface.co/papers/2609.11042"},{"id":"2609.05405","title":"WearableQA: A Benchmark for Health Reasoning over Real-World Wearable Data","arxiv_id":"2609.05405","upvotes":31,"published_at":"2026-09-03T20:00:00.000Z","authors":["Ji Soo Lee","Xilun Chen","Pierce Chuang"],"url":"https://huggingface.co/papers/2609.05405"},{"id":"2609.08149","title":"SWE-Bench Pro Verified: A Reliable Benchmark for Software Engineering Agents","arxiv_id":"2609.08149","upvotes":21,"published_at":"2026-09-07T20:00:00.000Z","authors":["Pujun Zheng","Zixin Shang","Shufan Jiang"],"url":"https://huggingface.co/papers/2609.08149"},{"id":"2609.06702","title":"PARSER: Read in Parallel, Reason in Depth for Long-Context LLM Agents","arxiv_id":"2609.06702","upvotes":17,"published_at":"2026-09-05T20:00:00.000Z","authors":["Kun Li","Zexuan Qiu","Tianhua Zhang"],"url":"https://huggingface.co/papers/2609.06702"},{"id":"2609.09113","title":"SAEScientist-Bench: Can AI Agents Conduct Autonomous SAE Interpretability Research?","arxiv_id":"2609.09113","upvotes":16,"published_at":"2026-09-07T20:00:00.000Z","authors":["Yuqiao Tan","Shizhu He","Jun Zhao"],"url":"https://huggingface.co/papers/2609.09113"},{"id":"2609.09219","title":"Scores Alone Do Not Prove Discovery: The Discovery Certification Protocol for Auditing AI Research Agents","arxiv_id":"2609.09219","upvotes":16,"published_at":"2026-09-06T20:00:00.000Z","authors":["Jingjie Ning","Shanshan Zhong","Xiaochuan Li"],"url":"https://huggingface.co/papers/2609.09219"}],"source":"HuggingFace Papers"}