<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
 <channel>
  <title>Axiomi — Axioms of intelligence. Live feed</title>
  <link>https://axiomi.si/</link>
  <description>A curated directory of open-weight models, ML frameworks, fine-tuning tools, interpretability libraries, datasets and benchmarks — with live arXiv and Hugging Face feeds.</description>
  <language>en</language>
  <lastBuildDate>Tue, 06 Oct 2026 01:26:45 +0000</lastBuildDate>
  <atom:link href="https://axiomi.si/feed.xml" rel="self" type="application/rss+xml"/>
  <item>
   <title>Dust: Pretraining Transformers Without Backpropagation</title>
   <link>https://qlabs.sh/research/dust</link>
   <description>[Hacker News] 96 points · 15 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Mon, 05 Oct 2026 21:15:07 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49970871</guid>
  </item>
  <item>
   <title>Beam: Reflection&#x27;s 501B open-weight model</title>
   <link>https://reflection.ai/blog/introducing-beam</link>
   <description>[Hacker News] 302 points · 78 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Mon, 05 Oct 2026 19:16:35 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49969183</guid>
  </item>
  <item>
   <title>&quot;Torturing&quot; LLMs in a Robot Prison Has Triggered the Dumbest Debate in AI Yet</title>
   <link>https://www.404media.co/someone-torturing-llms-in-a-robot-prison-has-triggered-the-dumbest-debate-in-ai-yet/</link>
   <description>[Hacker News] 46 points · 102 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Sun, 04 Oct 2026 08:02:33 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49951684</guid>
  </item>
  <item>
   <title>Aleph Alpha Kolibri: How the sovereign German LLM works</title>
   <link>https://tej.as/blog/aleph-alpha-kolibri</link>
   <description>[Hacker News] 420 points · 12 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Sat, 03 Oct 2026 10:43:51 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49943034</guid>
  </item>
  <item>
   <title>Kolibri: A Sovereign Open-Weight Model</title>
   <link>https://aleph-alpha.com/en/blog/kolibri-has-landed-a-sovereign-open-weight-model/</link>
   <description>[Hacker News] 695 points · 335 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Sat, 03 Oct 2026 09:36:04 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49942706</guid>
  </item>
  <item>
   <title>From the creator of Redis; run LLM locally with ds4</title>
   <link>https://dwarfstar.sh/</link>
   <description>[Hacker News] 362 points · 104 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 18:01:16 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49936575</guid>
  </item>
  <item>
   <title>Language Models that Play Chess and Explain Their Moves</title>
   <link>https://arxiv.org/abs/2610.03695v1</link>
   <description>[arXiv] Modern chess engines are silent experts: they play at a superhuman level, but do not offer explanations for their play. On the other hand, language models (LMs) can generate plausible-sounding explanations, but their weak playing strength limits the utility of their explanations.…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:54:22 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03695v1</guid>
  </item>
  <item>
   <title>FrugalEvo: Towards Cost-Aware LLM-Guided Program Evolution</title>
   <link>https://arxiv.org/abs/2610.03675v1</link>
   <description>[arXiv] LLM-guided evolutionary methods, such as AlphaEvolve, have emerged as powerful approaches for challenging computational optimization problems, such as circle packing. However, prior work typically optimizes performance gain over a fixed number of iterations. We argue that practic…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:44:05 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03675v1</guid>
  </item>
  <item>
   <title>Pivot-SD: Efficient Self-Distillation for Masked Diffusion Language Models</title>
   <link>https://arxiv.org/abs/2610.03665v1</link>
   <description>[arXiv] Masked diffusion language models (dLMs) offer a promising parallel alternative to autoregressive models for complex reasoning. However, they face a distinct credit-assignment challenge, since a few commitments during denoising sharply reduce the uncertainty over the remaining mas…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:37:51 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03665v1</guid>
  </item>
  <item>
   <title>Writerslogic at the CLEF 2026 SimpleText Track: Multi-Candidate LLM Simplification and Stacked Complexity Spotting</title>
   <link>https://arxiv.org/abs/2610.03567v1</link>
   <description>[arXiv] We describe the Writerslogic team&#x27;s participation in the CLEF 2026 SimpleText shared task, addressing Task 1 (text simplification) and Task 2 (complexity spotting). For Task 1, we develop a multi-candidate generation pipeline using GPT-4o-mini that produces five simplification ca…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:44:53 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03567v1</guid>
  </item>
  <item>
   <title>Objects Without Morphisms: What LLMs for Mathematics Do Not Represent</title>
   <link>https://arxiv.org/abs/2610.03551v1</link>
   <description>[arXiv] Large language models (LLMs) have reached expert-level performance on competition mathematics largely through the volume of search placed around them: candidate solutions are sampled in quantity and retained only when an external criterion accepts them. Such a procedure improves …</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:33:32 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03551v1</guid>
  </item>
  <item>
   <title>Author Representation Strategies for Zero-Shot Authorship Attribution: A Comparative Study of LLM-Based and Embedding-Based Approaches</title>
   <link>https://arxiv.org/abs/2610.03531v1</link>
   <description>[arXiv] Authorship Attribution (AA) requires capturing fine-grained stylistic characteristics, making it particularly challenging in zero-shot (ZS) settings where no task-specific supervision is available. In this work, we investigate the effect of author representations on ZS AA by eval…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:16:47 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03531v1</guid>
  </item>
  <item>
   <title>Single-Pass Uncertainty Heads for Claim-Level Hallucination Detection in Persian Medical Language Models</title>
   <link>https://arxiv.org/abs/2610.03482v1</link>
   <description>[arXiv] Hallucination detection is particularly important for medical language models, but repeated-sampling approaches are expensive and existing uncertainty-head resources do not directly transfer to a new backbone and language. We adapt the LLM Uncertainty Head (LUH) framework to Aya-…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 15:50:46 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03482v1</guid>
  </item>
  <item>
   <title>Generalization of Transformer-Based Neural Quantum States via In-Context Learning</title>
   <link>https://arxiv.org/abs/2610.03463v1</link>
   <description>[arXiv] Neural quantum states based on modern deep learning architectures have emerged as powerful representations for quantum many-body systems. In particular, Transformer-based neural quantum states provide expressive models capable of capturing long-range correlations, and their empir…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 15:38:10 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03463v1</guid>
  </item>
  <item>
   <title>Passing the Test You Trained On: Re-evaluating Prompt-Injection Detectors for LLM Agents</title>
   <link>https://arxiv.org/abs/2610.03448v1</link>
   <description>[arXiv] LLM agents increasingly screen tool outputs with small prompt-injection detectors, and teams choose among detectors by their scores on public benchmarks. We ask whether those scores predict how a detector behaves inside an agent. We replay the ground-truth tool calls of two agent…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 15:30:11 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03448v1</guid>
  </item>
  <item>
   <title>From Patching to Pruning Visual Computation in Vision Language Models</title>
   <link>https://arxiv.org/abs/2610.03389v1</link>
   <description>[arXiv] Vision language models (VLMs) incur substantial inference cost because every visual token is processed by the attention and MLP projections of every decoder layer, even when token-specific visual computation is unnecessary at many depths. We introduce Patch-to-Prune (P2P), inspir…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 14:37:53 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03389v1</guid>
  </item>
  <item>
   <title>SyntaxBench: A Statistical Diagnostic Framework for Character-Level Reasoning in Large Language Models</title>
   <link>https://arxiv.org/abs/2610.03329v1</link>
   <description>[arXiv] Large language models are increasingly used where small syntactic errors matter, yet character-level reasoning is still evaluated mostly through isolated probes and aggregate accuracy. We introduce SyntaxBench, a diagnostic benchmark and statistical evaluation framework for chara…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 14:02:56 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03329v1</guid>
  </item>
  <item>
   <title>Decision models like Jev don&#x27;t beat LLM-as-a-judge or traditional classifiers</title>
   <link>https://developers.redhat.com/articles/2026/10/02/benchmarking-ai-decision-models-against-traditional-guardrails</link>
   <description>[Hacker News] 153 points · 70 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 13:47:41 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49933476</guid>
  </item>
  <item>
   <title>Don&#x27;t be fooled–LLMs don&#x27;t reason</title>
   <link>https://www.technologyreview.com/2026/10/02/1145639/dont-be-fooled-llms-dont-reason/</link>
   <description>[Hacker News] 76 points · 159 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 13:45:42 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49933459</guid>
  </item>
  <item>
   <title>JOVE: Joint Execution and Verification for Resource-Aware LLM Task Graphs</title>
   <link>https://arxiv.org/abs/2610.03296v1</link>
   <description>[arXiv] Complex reasoning queries can be decomposed into directed acyclic task graphs and distributed across heterogeneous LLMs, reducing latency through parallelism and enabling smaller models to solve complex tasks. In practice, however, the suitability of an LLM for a given subtask ma…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 13:37:59 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03296v1</guid>
  </item>
  <item>
   <title>Wrong Organ, Right Physics: Transferring Echocardiography Pretraining to Lung Ultrasound for Tuberculosis Screening</title>
   <link>https://arxiv.org/abs/2610.03290v1</link>
   <description>[arXiv] Lung ultrasound (LUS) is attractive for tuberculosis (TB) screening at primary-care level, but labelled cohorts are small. Echocardiography carries no such constraint, while sharing the same underlying ultrasound imaging physics, signal processing and B-mode appearance as LUS. We…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 13:33:18 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03290v1</guid>
  </item>
  <item>
   <title>SPEAR: A Spectral-Disentangled MoE Neural Operator with Knowledge-Guided Expert Aggregation for Large-Scale PDE Pretraining</title>
   <link>https://arxiv.org/abs/2610.03265v1</link>
   <description>[arXiv] Large-scale pre-training has improved the generalization of neural operators across diverse PDEs. However, existing PDE foundation models still struggle with heterogeneous dynamics, where shared representations may cause knowledge interference, while mixture-of-experts (MoE) arch…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 13:10:09 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03265v1</guid>
  </item>
  <item>
   <title>D2K-Bench: Can LLM Agents Turn Expert Designs into Efficient GPU Kernels?</title>
   <link>https://arxiv.org/abs/2610.03226v1</link>
   <description>[arXiv] GPU kernels generated by large language model (LLM) agents can remain less efficient than expert implementations, but runtime alone does not reveal how the gap relates to design discovery and implementation. We introduce D2K-Bench, a diagnostic benchmark of 26 tasks and 85 worklo…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 12:42:12 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03226v1</guid>
  </item>
  <item>
   <title>Predicting and Repairing Merge Collapse in Large Language Models</title>
   <link>https://arxiv.org/abs/2610.03199v1</link>
   <description>[arXiv] Large language models fine-tuned from a shared base can be merged by averaging their task vectors, but some merges collapse far below the base model, and common merge operators give no warning before evaluation. We show that one statistic of the specialists&#x27; task vectors both pre…</description>
   <source url="https://axiomi.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 12:11:43 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:arxiv-2610.03199v1</guid>
  </item>
  <item>
   <title>Greg Kroah-Hartman – Security in the LLM Age [video]</title>
   <link>https://www.youtube.com/watch?v=NnV_cWeoo5Q</link>
   <description>[Hacker News] 338 points · 127 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 02:51:27 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49929391</guid>
  </item>
  <item>
   <title>Native Action-Prior Learning from Videos for World Action Models</title>
   <link>https://huggingface.co/papers/2610.03391</link>
   <description>[HF Daily Papers] 75 upvotes</description>
   <source url="https://axiomi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Thu, 01 Oct 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hfp-2610.03391</guid>
  </item>
  <item>
   <title>HyperBrowseComp: A Multilingual and Multimodal Stress Test for Web-Browsing Agents</title>
   <link>https://huggingface.co/papers/2610.03574</link>
   <description>[HF Daily Papers] 51 upvotes</description>
   <source url="https://axiomi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Thu, 01 Oct 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hfp-2610.03574</guid>
  </item>
  <item>
   <title>Pivot-SD: Efficient Self-Distillation for Masked Diffusion Language Models</title>
   <link>https://huggingface.co/papers/2610.03665</link>
   <description>[HF Daily Papers] 50 upvotes</description>
   <source url="https://axiomi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Thu, 01 Oct 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hfp-2610.03665</guid>
  </item>
  <item>
   <title>Clef: Open-weight decision models, and new RL fine-tuning platform</title>
   <link>https://blog.cloudflare.com/clef-decision-models/</link>
   <description>[Hacker News] 640 points · 217 comments</description>
   <source url="https://axiomi.si/feed.xml">Hacker News</source>
   <pubDate>Thu, 01 Oct 2026 16:18:57 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hn-49923692</guid>
  </item>
  <item>
   <title>RealCompanion: Benchmarking Human Understanding from Reasoning over Longitudinal Real-World Conversations</title>
   <link>https://huggingface.co/papers/2610.01780</link>
   <description>[HF Daily Papers] 250 upvotes</description>
   <source url="https://axiomi.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">axiomi.si:hfp-2610.01780</guid>
  </item>
 </channel>
</rss>
