<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
 <channel>
  <title>TrueIQ — Measure what matters. Live feed</title>
  <link>https://trueiq.si/</link>
  <description>Find LLM evaluation frameworks, leaderboards, observability platforms and AI-powered learning and assessment tools. Curated, searchable and refreshed daily.</description>
  <language>en</language>
  <lastBuildDate>Tue, 06 Oct 2026 01:26:45 +0000</lastBuildDate>
  <atom:link href="https://trueiq.si/feed.xml" rel="self" type="application/rss+xml"/>
  <item>
   <title>RAG Evaluation Metrics: Ragas vs DeepEval vs TruLens</title>
   <link>https://trueiq.si/guides/rag-evaluation-metrics.html</link>
   <description>RAG evaluation metrics explained: faithfulness, relevancy, context precision and recall, which need a reference, and how Ragas, DeepEval and TruLens differ.</description>
   <category>Explainers</category>
   <pubDate>Tue, 06 Oct 2026 12:00:00 +0000</pubDate>
   <guid isPermaLink="true">https://trueiq.si/guides/rag-evaluation-metrics.html</guid>
  </item>
  <item>
   <title>LLM Leaderboards Compared: Which Ones to Trust</title>
   <link>https://trueiq.si/guides/llm-leaderboards-compared.html</link>
   <description>Every major LLM leaderboard compared: how Arena, Artificial Analysis, LiveBench, SEAL, Epoch AI and Vellum score models, and where each one can mislead you.</description>
   <category>Comparison</category>
   <pubDate>Tue, 06 Oct 2026 12:00:00 +0000</pubDate>
   <guid isPermaLink="true">https://trueiq.si/guides/llm-leaderboards-compared.html</guid>
  </item>
  <item>
   <title>LLM Benchmarks Explained: What the Scores Mean</title>
   <link>https://trueiq.si/guides/llm-benchmarks-explained.html</link>
   <description>LLM benchmarks explained: what MMLU, GPQA Diamond, Humanity&#x27;s Last Exam, SWE-bench and ARC-AGI measure, why scores saturate, and how to spot contamination.</description>
   <category>Explainer</category>
   <pubDate>Tue, 06 Oct 2026 12:00:00 +0000</pubDate>
   <guid isPermaLink="true">https://trueiq.si/guides/llm-benchmarks-explained.html</guid>
  </item>
  <item>
   <title>LLM as a Judge: How It Works and When to Trust It</title>
   <link>https://trueiq.si/guides/llm-as-a-judge.html</link>
   <description>LLM as a judge explained: how model graders work, the biases that skew them, how to validate one against human labels, and which open-source tools to use.</description>
   <category>Explainers</category>
   <pubDate>Tue, 06 Oct 2026 12:00:00 +0000</pubDate>
   <guid isPermaLink="true">https://trueiq.si/guides/llm-as-a-judge.html</guid>
  </item>
  <item>
   <title>Langfuse vs LangSmith: Which Should You Use?</title>
   <link>https://trueiq.si/guides/langfuse-vs-langsmith.html</link>
   <description>Langfuse vs LangSmith in 2026: open-source license, self-hosting, pricing, retention, framework fit and migration tips, so you can pick the right LLM tracer.</description>
   <category>Comparison</category>
   <pubDate>Tue, 06 Oct 2026 12:00:00 +0000</pubDate>
   <guid isPermaLink="true">https://trueiq.si/guides/langfuse-vs-langsmith.html</guid>
  </item>
  <item>
   <title>How to Evaluate LLM Apps: A 2026 Playbook</title>
   <link>https://trueiq.si/guides/how-to-evaluate-llm-applications.html</link>
   <description>How to evaluate LLM apps step by step: error analysis, code checks, calibrated LLM judges, CI gates and production sampling, with tools for each stage.</description>
   <category>Guides</category>
   <pubDate>Tue, 06 Oct 2026 12:00:00 +0000</pubDate>
   <guid isPermaLink="true">https://trueiq.si/guides/how-to-evaluate-llm-applications.html</guid>
  </item>
  <item>
   <title>Best LLM Observability Tools in 2026, Compared</title>
   <link>https://trueiq.si/guides/best-llm-observability-tools.html</link>
   <description>The best LLM observability tools in 2026 compared: Langfuse, LangSmith, Phoenix, Opik, Weave and more, with licenses, self-hosting and who now owns what.</description>
   <category>Best of</category>
   <pubDate>Tue, 06 Oct 2026 12:00:00 +0000</pubDate>
   <guid isPermaLink="true">https://trueiq.si/guides/best-llm-observability-tools.html</guid>
  </item>
  <item>
   <title>AI Tutoring with Khanmigo in a Two-Year School Experiment</title>
   <link>https://edworkingpapers.com/ai26-1551</link>
   <description>[Hacker News] 23 points · 12 comments</description>
   <source url="https://trueiq.si/feed.xml">Hacker News</source>
   <pubDate>Tue, 06 Oct 2026 00:00:45 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hn-49972419</guid>
  </item>
  <item>
   <title>Leaderboards and speedrun.com&#x27;s new terms of service</title>
   <link>https://therun.gg/blog/leaderboards-speedruncom</link>
   <description>[Hacker News] 34 points · 11 comments</description>
   <source url="https://trueiq.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 18:07:05 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hn-49936641</guid>
  </item>
  <item>
   <title>4DCodeBench: Benchmarking Agents on Inverse Graphics of Dynamic Scenes</title>
   <link>https://arxiv.org/abs/2610.03715v1</link>
   <description>[arXiv] We introduce 4DCodeBench, a benchmark for 4D inverse graphics through code generation, in which agents reconstruct dynamic scenes from video as executable graphics programs. To accomplish this, agents must translate visual observations into compact representations of scene struct…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:58:49 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03715v1</guid>
  </item>
  <item>
   <title>RNADyn: A Benchmark for Generating and Understanding RNA Dynamics</title>
   <link>https://arxiv.org/abs/2610.03712v1</link>
   <description>[arXiv] Ribonucleic acid (RNA) functions through conformational changes that are not fully captured by static structures. However, large-scale standardized RNA dynamics data remain limited, and existing approaches typically treat trajectory generation and dynamics understanding as separa…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:58:08 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03712v1</guid>
  </item>
  <item>
   <title>Forecasting from Counterfactual Simulator Rollouts: A Sim2Real Evaluation</title>
   <link>https://arxiv.org/abs/2610.03662v1</link>
   <description>[arXiv] Deploying a new decision policy creates a cold-start problem for prediction models whose targets depend on the policy&#x27;s actions: historical observations reflect earlier policies, while real observations under the new policy are not yet available. Simulation offers a way to addres…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:37:14 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03662v1</guid>
  </item>
  <item>
   <title>Do Large Language Models Know Colombian Law? A Reliability Benchmark for the Colombian Legal System</title>
   <link>https://arxiv.org/abs/2610.03639v1</link>
   <description>[arXiv] Large language models (LLMs) are increasingly used to support legal practice, education, and research, yet their reliability in national legal systems outside the United States remains largely undocumented. We introduce an expert-validated benchmark for evaluating LLM reliability…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:27:41 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03639v1</guid>
  </item>
  <item>
   <title>World Embedding Benchmark</title>
   <link>https://arxiv.org/abs/2610.03632v1</link>
   <description>[arXiv] Physical fidelity has received increasing attention in world models and video generation, yet how video representations encode physical information remains less understood. We introduce the World Embedding Benchmark, comprising 8,000 controlled simulation cases from 80 families s…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 17:24:25 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03632v1</guid>
  </item>
  <item>
   <title>Threat-Preserving Representation Sensitivity in Agent-Security Benchmarks</title>
   <link>https://arxiv.org/abs/2610.03585v1</link>
   <description>[arXiv] Security benchmarks for LLM-based agents often report the attack success rate (ASR) as a measure of model robustness and use these scores to compare different models and defense mechanisms, assuming that they describe the security of the agent. In this paper, we explore whether i…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:55:50 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03585v1</guid>
  </item>
  <item>
   <title>Beyond Trained Models: Compiling GNNs for a Sound Explainer Benchmark</title>
   <link>https://arxiv.org/abs/2610.03526v1</link>
   <description>[arXiv] Explainers for Graph Neural Networks (GNNs) are commonly evaluated by their plausibility, i.e., how well their explanations recover a predefined ground truth, such as a motif planted in the data. This protocol implicitly assumes that a GNN trained on such data relies on the inten…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:14:19 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03526v1</guid>
  </item>
  <item>
   <title>From Benchmarks to Production: A Text-to-SQL System for Complex Financial Data</title>
   <link>https://arxiv.org/abs/2610.03524v1</link>
   <description>[arXiv] General-purpose Text-to-SQL systems achieve strong performance on academic benchmarks like Spider and BIRD, where schemas are relatively shallow and column values are often human readable. In production financial databases, where concepts are stored as opaque integer keys rather …</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 16:13:40 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03524v1</guid>
  </item>
  <item>
   <title>Below what training size do deep tabular generators stop beating trivial baselines? A preregistered benchmark on a size ladder of clinical and standard datasets</title>
   <link>https://arxiv.org/abs/2610.03500v1</link>
   <description>[arXiv] Deep tabular generative models are benchmarked on datasets with tens of thousands of rows; clinical datasets have hundreds. We preregistered and ran a size-ladder benchmark to find where the two regimes diverge: 8 public datasets subsampled from 200 to 20,000 training rows, seven…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 15:58:41 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03500v1</guid>
  </item>
  <item>
   <title>Beyond Random Splits: Evaluating Drug-Target Affinity Models Under Chemically and Biologically Motivated Distribution Shifts Copy</title>
   <link>https://arxiv.org/abs/2610.03456v1</link>
   <description>[arXiv] Drug-target affinity (DTA) prediction is widely used to prioritize candidate compounds before costly experimental screening. DTA models are often compared under a single data split, even though deployment may require extrapolation to new chemical series, new protein targets, or b…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 15:33:20 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03456v1</guid>
  </item>
  <item>
   <title>Passing the Test You Trained On: Re-evaluating Prompt-Injection Detectors for LLM Agents</title>
   <link>https://arxiv.org/abs/2610.03448v1</link>
   <description>[arXiv] LLM agents increasingly screen tool outputs with small prompt-injection detectors, and teams choose among detectors by their scores on public benchmarks. We ask whether those scores predict how a detector behaves inside an agent. We replay the ground-truth tool calls of two agent…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 15:30:11 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03448v1</guid>
  </item>
  <item>
   <title>Benchmarking Candidate Coverage in Typed Decision Models</title>
   <link>https://arxiv.org/abs/2610.03387v1</link>
   <description>[arXiv] Typed decision models return choices or distributions over answer options supplied at request time. Accuracy with complete options does not establish whether a model recognizes that a reference answer is missing or avoids rejecting valid candidates. We present a paired candidate-…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 14:36:46 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03387v1</guid>
  </item>
  <item>
   <title>Fixing GRPO&#x27;s credit assignment problem without evaluating every step</title>
   <link>https://arxiv.org/abs/2609.36178</link>
   <description>[Hacker News] 23 points · 3 comments</description>
   <source url="https://trueiq.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 14:36:04 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hn-49934012</guid>
  </item>
  <item>
   <title>ReFract: Benchmarking Perspective Awareness in Language Model Agents with Text World Models</title>
   <link>https://arxiv.org/abs/2610.03356v1</link>
   <description>[arXiv] Large Language Model (LLM) agents are increasingly deployed in high-stakes settings such as industrial maintenance and equipment fault troubleshooting, where workers occupy a variety of roles. A capable agent must therefore act in a way that is calibrated to user&#x27;s role: taking a…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 14:21:26 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03356v1</guid>
  </item>
  <item>
   <title>To Jev or Not? Evaluating the Accuracy and Efficiency of Structured Decision Models for Hate-Speech Moderation</title>
   <link>https://arxiv.org/abs/2610.03324v1</link>
   <description>[arXiv] The scale of online content makes hate-speech moderation challenging, while Large Language Models (LLMs) enable harmful material to be produced and adapted more easily. Moderation therefore requires efficient classifiers that can accommodate different definitions of hate speech. …</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 13:59:21 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03324v1</guid>
  </item>
  <item>
   <title>Lightweight, Rubric-Guided Trajectory Evaluation for Production AI Agents</title>
   <link>https://arxiv.org/abs/2610.03315v1</link>
   <description>[arXiv] Trajectory evaluation is essential for improving the reliability of LLM-based agents, but production use makes it expensive to run repeatedly. Modern agents generate long traces containing tool calls, observations, retries, and external outputs, while not all raw tokens are equal…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 13:51:06 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03315v1</guid>
  </item>
  <item>
   <title>Decision models like Jev don&#x27;t beat LLM-as-a-judge or traditional classifiers</title>
   <link>https://developers.redhat.com/articles/2026/10/02/benchmarking-ai-decision-models-against-traditional-guardrails</link>
   <description>[Hacker News] 153 points · 70 comments</description>
   <source url="https://trueiq.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 13:47:41 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hn-49933476</guid>
  </item>
  <item>
   <title>Benchmarking retrieval for agents on messy real-world company knowledge</title>
   <link>https://www.kapa.ai/blog/company-knowledge-bench</link>
   <description>[Hacker News] 26 points · 3 comments</description>
   <source url="https://trueiq.si/feed.xml">Hacker News</source>
   <pubDate>Fri, 02 Oct 2026 13:37:18 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hn-49933381</guid>
  </item>
  <item>
   <title>PaMIR: Open Benchmark of Public Credit-Default Datasets</title>
   <link>https://arxiv.org/abs/2610.03259v1</link>
   <description>[arXiv] We release PaMIR (Public Arrival-ordered Measurement for Inference in Risk), an open benchmark for credit-default prediction when labels are scarce and arrive late. The field&#x27;s reference benchmark studies use eight datasets each, only two or four of them public. PaMIR brings toge…</description>
   <source url="https://trueiq.si/feed.xml">arXiv</source>
   <pubDate>Fri, 02 Oct 2026 13:04:45 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:arxiv-2610.03259v1</guid>
  </item>
  <item>
   <title>DoGBench: The first user-facing docs generation benchmark. No model scores &gt;50%</title>
   <link>https://dogbench.ai/</link>
   <description>[Hacker News] 18 points · 3 comments</description>
   <source url="https://trueiq.si/feed.xml">Hacker News</source>
   <pubDate>Thu, 01 Oct 2026 23:10:32 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hn-49928118</guid>
  </item>
  <item>
   <title>HyperBrowseComp: A Multilingual and Multimodal Stress Test for Web-Browsing Agents</title>
   <link>https://huggingface.co/papers/2610.03574</link>
   <description>[HF Daily Papers] 51 upvotes</description>
   <source url="https://trueiq.si/feed.xml">HF Daily Papers</source>
   <pubDate>Thu, 01 Oct 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hfp-2610.03574</guid>
  </item>
  <item>
   <title>4DCodeBench: Benchmarking Agents on Inverse Graphics of Dynamic Scenes</title>
   <link>https://huggingface.co/papers/2610.03715</link>
   <description>[HF Daily Papers] 19 upvotes</description>
   <source url="https://trueiq.si/feed.xml">HF Daily Papers</source>
   <pubDate>Thu, 01 Oct 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hfp-2610.03715</guid>
  </item>
  <item>
   <title>RealCompanion: Benchmarking Human Understanding from Reasoning over Longitudinal Real-World Conversations</title>
   <link>https://huggingface.co/papers/2610.01780</link>
   <description>[HF Daily Papers] 250 upvotes</description>
   <source url="https://trueiq.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hfp-2610.01780</guid>
  </item>
  <item>
   <title>SimuVerity: Benchmarking Agents for Engineering-Grade Simulink Model Generation</title>
   <link>https://huggingface.co/papers/2610.02304</link>
   <description>[HF Daily Papers] 45 upvotes</description>
   <source url="https://trueiq.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hfp-2610.02304</guid>
  </item>
  <item>
   <title>EditHero: A Benchmark for Long-Horizon Part-Level 3D Editing and Vibe Modeling</title>
   <link>https://huggingface.co/papers/2610.02298</link>
   <description>[HF Daily Papers] 44 upvotes</description>
   <source url="https://trueiq.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hfp-2610.02298</guid>
  </item>
  <item>
   <title>From Retrieval to Typed Decisions: Calibrated System One Models from Biomedical Sentence Encoders</title>
   <link>https://huggingface.co/papers/2610.02486</link>
   <description>[HF Daily Papers] 11 upvotes</description>
   <source url="https://trueiq.si/feed.xml">HF Daily Papers</source>
   <pubDate>Wed, 30 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hfp-2610.02486</guid>
  </item>
  <item>
   <title>Science or Slop?: Benchmarking and Mitigating Scientific Slop in AI-Generated Papers</title>
   <link>https://huggingface.co/papers/2610.00531</link>
   <description>[HF Daily Papers] 47 upvotes</description>
   <source url="https://trueiq.si/feed.xml">HF Daily Papers</source>
   <pubDate>Tue, 29 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hfp-2610.00531</guid>
  </item>
  <item>
   <title>JEPA-TTT: Persistent Test-Time Training of Latent World Models for Planning under Dynamics Shifts</title>
   <link>https://huggingface.co/papers/2610.00722</link>
   <description>[HF Daily Papers] 2 upvotes</description>
   <source url="https://trueiq.si/feed.xml">HF Daily Papers</source>
   <pubDate>Tue, 29 Sep 2026 20:00:00 +0000</pubDate>
   <guid isPermaLink="false">trueiq.si:hfp-2610.00722</guid>
  </item>
 </channel>
</rss>
