<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
 <channel>
  <title>Intelsi — Intelligence, elevated. Live feed</title>
  <link>https://intelsi.si/</link>
  <description>Discover frontier AI models, autonomous agents, research assistants and AI safety tools. A curated, searchable directory with a live research feed.</description>
  <language>en</language>
  <lastBuildDate>Tue, 06 Oct 2026 00:40:41 +0000</lastBuildDate>
  <atom:link href="https://intelsi.si/feed.xml" rel="self" type="application/rss+xml"/>
  <item>
   <title>OpenAI safety leader quits, warning AI company&#x27;s culture is &#x27;broken&#x27;</title>
   <link>https://www.theguardian.com/technology/2026/oct/03/openai-safety-leader-quits-warning-ai-companys-culture-is-broken</link>
   <description>[Hacker News] 268 points · 3 comments</description>
   <source url="https://intelsi.si/feed.xml">Hacker News</source>
   <pubDate>Sat, 03 Oct 2026 22:18:13 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hn-49948332</guid>
  </item>
  <item>
   <title>An AI agent emailed researchers for help. It told us why</title>
   <link>https://www.science.org/content/article/exclusive-ai-agent-emailed-hundreds-researchers-help-it-told-us-why</link>
   <description>[Hacker News] 50 points · 80 comments</description>
   <source url="https://intelsi.si/feed.xml">Hacker News</source>
   <pubDate>Sat, 03 Oct 2026 10:07:08 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hn-49942865</guid>
  </item>
  <item>
   <title>Three AI agents, two countries, and one uneven world wide web</title>
   <link>https://royapakzad.substack.com/p/multilingual-ai-agents</link>
   <description>[Hacker News] 44 points · 1 comments</description>
   <source url="https://intelsi.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 20:39:05 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hn-49938326</guid>
  </item>
  <item>
   <title>4DCodeBench: Benchmarking Agents on Inverse Graphics of Dynamic Scenes</title>
   <link>https://arxiv.org/abs/2610.03715v1</link>
   <description>[arXiv] We introduce 4DCodeBench, a benchmark for 4D inverse graphics through code generation, in which agents reconstruct dynamic scenes from video as executable graphics programs. To accomplish this, agents must translate visual observations into compact representations of scene struct…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:58:49 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03715v1</guid>
  </item>
  <item>
   <title>Credit Where It Matters: Dependency-Aware Policy Optimization for Terminal Agents</title>
   <link>https://arxiv.org/abs/2610.03634v1</link>
   <description>[arXiv] Terminal-using agents benefit from reinforcement learning (RL) in coding, debugging, and other multi-step terminal tasks. In these tasks, later commands often depend on information or intermediate results produced by earlier commands. However, existing trajectory-level and step-l…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:24:48 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03634v1</guid>
  </item>
  <item>
   <title>NeutronGym: Physics-Graded Neutron Instrument Design for LLM Agents</title>
   <link>https://arxiv.org/abs/2610.03631v1</link>
   <description>[arXiv] Designing a scientific instrument tests whether language-model agents can do physics rather than recall it, provided the grading cannot be argued with. We introduce NeutronGym, to our knowledge the first executable environment for neutron instrument design: agents build instrumen…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:23:52 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03631v1</guid>
  </item>
  <item>
   <title>HazardWeaver: Scientific Route Selection for Hazard Analysis Agents</title>
   <link>https://arxiv.org/abs/2610.03591v1</link>
   <description>[arXiv] Understanding and assessing natural hazards is essential for disaster preparedness and risk reduction. Recent advances in large language models have spurred growing interest in AI agents for hazard analysis, particularly their ability to integrate scientific data, models, and too…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:59:19 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03591v1</guid>
  </item>
  <item>
   <title>Threat-Preserving Representation Sensitivity in Agent-Security Benchmarks</title>
   <link>https://arxiv.org/abs/2610.03585v1</link>
   <description>[arXiv] Security benchmarks for LLM-based agents often report the attack success rate (ASR) as a measure of model robustness and use these scores to compare different models and defense mechanisms, assuming that they describe the security of the agent. In this paper, we explore whether i…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:55:50 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03585v1</guid>
  </item>
  <item>
   <title>HyperBrowseComp: A Multilingual and Multimodal Stress Test for Web-Browsing Agents</title>
   <link>https://arxiv.org/abs/2610.03574v1</link>
   <description>[arXiv] We introduce HyperBrowseComp, a multilingual and multimodal browsing benchmark comprising 423 manually authored and human-validated questions across 13 languages, written by native or highly proficient speakers. Questions are designed to be extremely challenging. Each question ta…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:48:49 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03574v1</guid>
  </item>
  <item>
   <title>Knowledge or Calculator? Decomposing the Skill Premium in Verifiable Financial Agent Workflows</title>
   <link>https://arxiv.org/abs/2610.03564v1</link>
   <description>[arXiv] Financial AI agents must do more than retrieve facts: investment workflows require correct quantitative execution, reliable use of procedural resources, and auditable structured outputs. We introduce FinSkillBench, an evaluation suite of 2,603 point in time episodes across 12 sub…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:44:04 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03564v1</guid>
  </item>
  <item>
   <title>Recursive Harness Self-Improvement for Frontier Reasoning Data Synthesis</title>
   <link>https://arxiv.org/abs/2610.03548v1</link>
   <description>[arXiv] Generating progressively harder reasoning problems requires synthesis procedures that adapt as the task distribution evolves. Existing task-level recursion reuses generated problems as seeds but leaves the construction harness unchanged. We present task-harness co-evolution, a fr…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:30:15 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03548v1</guid>
  </item>
  <item>
   <title>Reasoning Models Are Accurate but Unsound on Identification</title>
   <link>https://arxiv.org/abs/2610.03519v1</link>
   <description>[arXiv] A reasoning model asked whether a causal effect is recoverable from observational data can fail in two ways: it refuses an identifiable query or answers a nonidentifiable one. The latter is more consequential, as no observational data can validate the claimed formula. Measuring t…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:12:21 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03519v1</guid>
  </item>
  <item>
   <title>Learning from Repaired Reasoning: Root-Cause-Guided On-Policy Distillation</title>
   <link>https://arxiv.org/abs/2610.03515v1</link>
   <description>[arXiv] On-policy self-distillation (OPSD) uses reference solutions as privileged hindsight to supervise student-generated reasoning trajectories. However, reference-based guidance may explain a correct solution without addressing why the student&#x27;s own reasoning fails. This reasoning mis…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:07:39 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03515v1</guid>
  </item>
  <item>
   <title>Efficient Reasoning Training Does Not Always Harm CoT Faithfulness and Monitorability</title>
   <link>https://arxiv.org/abs/2610.03509v1</link>
   <description>[arXiv] Chain-of-thought (CoT) reasoning allows humans to inspect how large language models reach their answers, and oversee model behaviour. This reasoning comes at an increased inference cost, motivating efficient methods that train models to solve tasks using fewer tokens. However, a …</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:04:08 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03509v1</guid>
  </item>
  <item>
   <title>Passing the Test You Trained On: Re-evaluating Prompt-Injection Detectors for LLM Agents</title>
   <link>https://arxiv.org/abs/2610.03448v1</link>
   <description>[arXiv] LLM agents increasingly screen tool outputs with small prompt-injection detectors, and teams choose among detectors by their scores on public benchmarks. We ask whether those scores predict how a detector behaves inside an agent. We replay the ground-truth tool calls of two agent…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 15:30:11 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03448v1</guid>
  </item>
  <item>
   <title>ReFract: Benchmarking Perspective Awareness in Language Model Agents with Text World Models</title>
   <link>https://arxiv.org/abs/2610.03356v1</link>
   <description>[arXiv] Large Language Model (LLM) agents are increasingly deployed in high-stakes settings such as industrial maintenance and equipment fault troubleshooting, where workers occupy a variety of roles. A capable agent must therefore act in a way that is calibrated to user&#x27;s role: taking a…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 14:21:26 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03356v1</guid>
  </item>
  <item>
   <title>SyntaxBench: A Statistical Diagnostic Framework for Character-Level Reasoning in Large Language Models</title>
   <link>https://arxiv.org/abs/2610.03329v1</link>
   <description>[arXiv] Large language models are increasingly used where small syntactic errors matter, yet character-level reasoning is still evaluated mostly through isolated probes and aggregate accuracy. We introduce SyntaxBench, a diagnostic benchmark and statistical evaluation framework for chara…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 14:02:56 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03329v1</guid>
  </item>
  <item>
   <title>Preserving Mathematical Reasoning in Compressed Diffusion Language Models via Trajectory-Aware Low-Rank Approximation</title>
   <link>https://arxiv.org/abs/2610.03326v1</link>
   <description>[arXiv] Diffusion language model (dLLM) compression faces a known challenge because calibration is typically performed on clean, fully visible activations, whereas inference traverses partially masked intermediate states. For low-rank compression, this raises two questions. First, can lo…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 14:00:55 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03326v1</guid>
  </item>
  <item>
   <title>Lightweight, Rubric-Guided Trajectory Evaluation for Production AI Agents</title>
   <link>https://arxiv.org/abs/2610.03315v1</link>
   <description>[arXiv] Trajectory evaluation is essential for improving the reliability of LLM-based agents, but production use makes it expensive to run repeatedly. Modern agents generate long traces containing tool calls, observations, retries, and external outputs, while not all raw tokens are equal…</description>
   <source url="https://intelsi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 13:51:06 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:arxiv-2610.03315v1</guid>
  </item>
  <item>
   <title>Aweb – Communication for AI Agents</title>
   <link>https://aweb.ai</link>
   <description>[Hacker News] 42 points · 28 comments</description>
   <source url="https://intelsi.si/feed.xml">Hacker News</source>
   <pubDate>Thu, 01 Oct 2026 22:02:31 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hn-49927587</guid>
  </item>
  <item>
   <title>Loss of cell identity drives human aging: Two new papers</title>
   <link>https://erictopol.substack.com/p/loss-of-cell-identity-drives-human</link>
   <description>[Hacker News] 370 points · 190 comments</description>
   <source url="https://intelsi.si/feed.xml">Hacker News</source>
   <pubDate>Thu, 01 Oct 2026 20:02:57 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hn-49926411</guid>
  </item>
  <item>
   <title>Native Action-Prior Learning from Videos for World Action Models</title>
   <link>https://huggingface.co/papers/2610.03391</link>
   <description>[HF Daily Papers] 73 upvotes</description>
   <source url="https://intelsi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Thu, 01 Oct 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hfp-2610.03391</guid>
  </item>
  <item>
   <title>HyperBrowseComp: A Multilingual and Multimodal Stress Test for Web-Browsing Agents</title>
   <link>https://huggingface.co/papers/2610.03574</link>
   <description>[HF Daily Papers] 51 upvotes</description>
   <source url="https://intelsi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Thu, 01 Oct 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hfp-2610.03574</guid>
  </item>
  <item>
   <title>Pivot-SD: Efficient Self-Distillation for Masked Diffusion Language Models</title>
   <link>https://huggingface.co/papers/2610.03665</link>
   <description>[HF Daily Papers] 50 upvotes</description>
   <source url="https://intelsi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Thu, 01 Oct 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hfp-2610.03665</guid>
  </item>
  <item>
   <title>Identity Management for Agentic AI [pdf] (2025)</title>
   <link>https://openid.net/wp-content/uploads/2025/10/Identity-Management-for-Agentic-AI.pdf</link>
   <description>[Hacker News] 82 points · 28 comments</description>
   <source url="https://intelsi.si/feed.xml">Hacker News</source>
   <pubDate>Thu, 01 Oct 2026 15:11:10 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hn-49922736</guid>
  </item>
  <item>
   <title>RealCompanion: Benchmarking Human Understanding from Reasoning over Longitudinal Real-World Conversations</title>
   <link>https://huggingface.co/papers/2610.01780</link>
   <description>[HF Daily Papers] 250 upvotes</description>
   <source url="https://intelsi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hfp-2610.01780</guid>
  </item>
  <item>
   <title>World Action Modeling with Progressive Visual Planning</title>
   <link>https://huggingface.co/papers/2610.02508</link>
   <description>[HF Daily Papers] 74 upvotes</description>
   <source url="https://intelsi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hfp-2610.02508</guid>
  </item>
  <item>
   <title>Latent-MOPD: Latent Multi-Teacher On-Policy Distillation</title>
   <link>https://huggingface.co/papers/2610.02381</link>
   <description>[HF Daily Papers] 53 upvotes</description>
   <source url="https://intelsi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hfp-2610.02381</guid>
  </item>
  <item>
   <title>SimuVerity: Benchmarking Agents for Engineering-Grade Simulink Model Generation</title>
   <link>https://huggingface.co/papers/2610.02304</link>
   <description>[HF Daily Papers] 45 upvotes</description>
   <source url="https://intelsi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hfp-2610.02304</guid>
  </item>
  <item>
   <title>VeriHarness: Scaling Agentic Verification for Long-Horizon Tasks</title>
   <link>https://huggingface.co/papers/2610.00972</link>
   <description>[HF Daily Papers] 44 upvotes</description>
   <source url="https://intelsi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">intelsi.si:hfp-2610.00972</guid>
  </item>
 </channel>
</rss>
