{"items":[{"id":"91c1a621-bf24-4c31-979c-aab481d73dcb","title":"Procedural Graphs: Self-Evolving Execution Structures for LLM Agents","url":"https://arxiv.org/html/2609.09153v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-10T14:38:31.563+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/procedural-graphs-self-evolving-execution-structures-for-llm-age-1789051455661.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/procedural-graphs-self-evolving-execution-structures-for-llm-age-1789051455661.txt","audio_summary":"Masonry and Eyre discuss the 'Procedural Graphs' paper, which proposes a self-evolving graph structure to manage procedural knowledge for LLM agents, moving away from flat history logs toward a structured 'what-to-do' map that the agent can refine through trial and error.","audio_duration_seconds":286,"audio_generated_at":"2026-09-10T14:44:28.31+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":966,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":966},{"id":"3a1d837c-0547-4e53-aba5-4e54f17038d5","title":"AIM — India","url":"https://analyticsindiamag.com/ai-news/deepseek-launches-v41-flash-to-make-large-scale-ai-inference-cheaper","type":"normal","description":"India","cover_image_url":null,"discovered_at":"2026-09-10T14:32:10.963+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/aim-india-1789051413337.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/aim-india-1789051413337.txt","audio_summary":"The company has made V4.1-Flash available through its API with native multimodal support. DeepSeek has retired V4-Flash and V4-Flash-Vision-Exp, while the older model names deepseek-v4-flash and deepseek-v4-flash-vision-exp will temporarily route requests to V4.1-Flash.","audio_duration_seconds":52,"audio_generated_at":"2026-09-10T14:43:36.839+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":965,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":965},{"id":"f2210e8c-6989-431b-91d0-12a7d2298b1f","title":"Chain of Thought vs. Tree of Thoughts: Which is Best for AI Agents? - MachineLearningMastery.com","url":"https://machinelearningmastery.com/chain-of-thought-vs-tree-of-thoughts-which-is-best-for-ai-agents/","type":"normal","description":"Compare Chain of Thought and Tree of Thoughts reasoning to understand which approach best fits your AI agent.","cover_image_url":null,"discovered_at":"2026-09-10T14:30:48.929+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/chain-of-thought-vs-tree-of-thoughts-which-is-best-for-ai-agents-1789051165768.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/chain-of-thought-vs-tree-of-thoughts-which-is-best-for-ai-agents-1789051165768.txt","audio_summary":"Tree of Thoughts: Which is Best for AI Agents? By Vinod Chugani on September 9, 2026 in Artificial Intelligence 0 Share Post Share In this article, you will learn the key differences between Chain of Thought and Tree of Thoughts prompting, and how each reasoning framework is applied in AI agent systems.","audio_duration_seconds":44,"audio_generated_at":"2026-09-10T14:39:31.263+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":964,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":964},{"id":"3656ca92-e710-4888-ab64-297bb57c6669","title":"Introducing Muse: The World’s First Personal AI Agent Built for Everyone","url":"https://about.fb.com/news/2026/09/introducing-muse-personal-ai-agent/","type":"normal","description":"Muse is a secure, private personal AI agent that proactively helps people meet their goals and suggests ideas.","cover_image_url":null,"discovered_at":"2026-09-09T15:37:06.3+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-muse-the-world-s-first-personal-ai-agent-built-for-e-1788968401506.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-muse-the-world-s-first-personal-ai-agent-built-for-e-1788968401506.txt","audio_summary":"Meta launches Muse as a personal AI agent built for mass adoption, running on a dedicated Secure VM with a Sentinel gatekeeper, powered by Muse Spark, and usable via app or WhatsApp with proactive planning, background work, and Stripe Link checkout protections.","audio_duration_seconds":226,"audio_generated_at":"2026-09-09T15:40:13.894+00:00","audio_provider":"fishaudio:s2.1-pro-free","audio_script_meta":{"episodeNumber":963,"archetype":"quickRead","scriptModelLabel":"Muse Glimmer 30B","model":"meta/muse-glimmer-30b","voiceProviderLabel":"Fish Audio S2.1 Pro","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":963},{"id":"5c8cf40a-d152-4f3a-b57b-82027ad57103","title":"vLLM x AgentX: Optimizing for Real-World Agentic Serving","url":"https://vllm.ai/blog/2026-09-08-vllm-agentx","type":"normal","description":"How vLLM optimizes KV cache management, parallelism, scheduling, and P/D disaggregation for agentic workloads, validated on SemiAnalysis AgentX with up to 130K","cover_image_url":null,"discovered_at":"2026-09-09T15:31:19.719+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/vllm-x-agentx-optimizing-for-real-world-agentic-serving-1788968079981.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/vllm-x-agentx-optimizing-for-real-world-agentic-serving-1788968079981.txt","audio_summary":"Vince and Ava unpack vLLM’s AgentX post on agentic serving, debating whether its full-stack KV-cache and P/D disaggregation story is a genuine product win or benchmark engineering dressed as insight.","audio_duration_seconds":257,"audio_generated_at":"2026-09-09T15:34:52.02+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":962,"archetype":"skepticsTake","scriptModelLabel":"Muse Glimmer 30B","model":"meta/muse-glimmer-30b","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":962},{"id":"c08d0916-27c0-49f0-84b7-7837f1d11116","title":"Organizing Context in a Multi-Agent Harness","url":"https://www.langchain.com/blog/organizing-context-in-a-multi-agent-harness","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-09T15:30:03.892+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/organizing-context-in-a-multi-agent-harness-1788968009709.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/organizing-context-in-a-multi-agent-harness-1788968009709.txt","audio_summary":"Justy and Cody unpack LangChain’s deepagents post on forked vs isolated subagents, debating whether inheriting supervisor context is a real efficiency win or a bias risk, and who actually benefits from context modes in production harnesses.","audio_duration_seconds":273,"audio_generated_at":"2026-09-09T15:33:44.855+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":961,"archetype":"quickRead","scriptModelLabel":"Muse Glimmer 30B","model":"meta/muse-glimmer-30b","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":961},{"id":"d534a544-ed44-4148-975c-f2c0c2782aab","title":"Model Behavior: Week of September 7, 2026","url":"https://www.sandrise.io/exploring-next/model-behavior/2026-09-07","type":"model-behavior","description":null,"cover_image_url":null,"discovered_at":"2026-09-09T00:00:12.691+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-september-7-2026-1788912675387.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-september-7-2026-1788912675387.txt","audio_summary":"They dig into the AGI framing, the cybersecurity numbers, and whether the Codex context-window fix is the quietly interesting thing nobody's leading with. The move collapses the friction between local-first and cloud-capable workflows.","audio_duration_seconds":33,"audio_generated_at":"2026-09-09T00:11:19.046+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":960,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"}},"domains":[],"concept_keys":[],"entities":[],"source_form":null,"event_type":null,"episode_number":960},{"id":"e213d14a-89e4-4a16-8858-79c021cc1a26","title":"When Models Edit Too Much: On the Fidelity of Minimal Code Edits","url":"https://arxiv.org/html/2609.04061v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-08T18:54:35.334+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/when-models-edit-too-much-on-the-fidelity-of-minimal-code-edits-1788894405033.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/when-models-edit-too-much-on-the-fidelity-of-minimal-code-edits-1788894405033.txt","audio_summary":"We study over-editing, the tendency of a model to rewrite code beyond what is required to fix a bug. We construct an evaluation framework from 400 BigCodeBench problems by injecting controlled AST-level corruptions into reference solutions, giving each repair task a known minimal patch.","audio_duration_seconds":58,"audio_generated_at":"2026-09-08T19:06:49.864+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":959,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":959},{"id":"315226bd-bef2-4e01-af23-a9748e0c4d2f","title":"archive.ph/2026.09.08-185035/https://thenewstack.io/enterprise-rag-permission-assembly","url":"https://archive.ph/2026.09.08-185035/https://thenewstack.io/enterprise-rag-permission-assembly/","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-08T18:53:36.512+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/archive-ph-2026-09-08-185035-https-thenewstack-io-enterprise-rag-1788908629958.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/archive-ph-2026-09-08-185035-https-thenewstack-io-enterprise-rag-1788908629958.txt","audio_summary":"Enterprise RAG just got a guard dog—token‑level gates that keep the model from pulling in secrets.","audio_duration_seconds":32,"audio_generated_at":"2026-09-08T23:03:54.371+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":958,"archetype":"coldBrief","scriptModelLabel":"GPT-OSS 20B","model":"openai/gpt-oss-20b","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":958},{"id":"db99db98-4b22-41ee-8ddb-65c91d5e3cbc","title":"Enoki: Efficient Multi-Level Hallucination Detection","url":"https://arxiv.org/html/2609.00581v2","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-08T18:49:42.117+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/enoki-efficient-multi-level-hallucination-detection-1788893776971.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/enoki-efficient-multi-level-hallucination-detection-1788893776971.txt","audio_summary":"Justy and Cody discuss the Enoki research paper, focusing on its approach to multi-level hallucination detection using text-anchored Open Information Extraction (OpenIE) to bridge the gap between claim-level verification and span-span level localization.","audio_duration_seconds":239,"audio_generated_at":"2026-09-08T18:56:29.384+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":957,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":957},{"id":"7cc2bdb7-e9fb-42de-b90e-e4f401f5283a","title":"Bilevel Coordinated Reflection: A Game-Theoretic Approach to Multi-Agent LLM Systems","url":"https://arxiv.org/html/2609.02750v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-08T18:47:20.617+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/bilevel-coordinated-reflection-a-game-theoretic-approach-to-mult-1788893514451.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/bilevel-coordinated-reflection-a-game-theoretic-approach-to-mult-1788893514451.txt","audio_summary":"Masonry and Eyre discuss the 'Bilevel Coordinated Reflection' paper, which applies game theory and stochastic approximation to multi-agent LLM coordination and memory improvement, specifically introducing SRMA to solve the 'text-only critic' plateau.","audio_duration_seconds":243,"audio_generated_at":"2026-09-08T18:52:05.106+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":956,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":956},{"id":"d2731458-9b54-4e01-a021-6ecd999a7745","title":"To See a World in a Living Context:Unified Indoor-Outdoor Urban World Generation","url":"https://arxiv.org/html/2608.05879v2","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-08T18:46:33.175+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/to-see-a-world-in-a-living-context-unified-indoor-outdoor-urban--1788893470944.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/to-see-a-world-in-a-living-context-unified-indoor-outdoor-urban--1788893470944.txt","audio_summary":"Edmund and Geffen discuss HoloWorld, a new framework for unified indoor-outdoor urban 3D generation that ensures the inside of a building actually matches its exterior and the surrounding city context.","audio_duration_seconds":258,"audio_generated_at":"2026-09-08T18:51:22.534+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":955,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":955},{"id":"d4be045a-3297-4948-9d27-7b6d1017b7a5","title":"Causal evidence that language models use confidence to drive behaviour - Nature Machine Intelligence","url":"https://www.nature.com/articles/s42256-026-01293-x","type":"normal","description":"Kumaran et al. show that large language models making decisions on when to answer a question or abstain from answering can be influenced by boosting or suppressing confidence signals in the model.","cover_image_url":null,"discovered_at":"2026-09-08T18:44:49.354+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/causal-evidence-that-language-models-use-confidence-to-drive-beh-1788893530789.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/causal-evidence-that-language-models-use-confidence-to-drive-beh-1788893530789.txt","audio_summary":"Pippa and Tyler discuss a new Nature Machine Intelligence paper providing causal evidence that LLMs use internal confidence representations to decide when to abstain from answering. They explore the distinction between verbal confidence and internal activation states, and the implications for autonomous agents.","audio_duration_seconds":225,"audio_generated_at":"2026-09-08T18:52:20.951+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":954,"archetype":"quickRead","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":954},{"id":"8f977ab6-f93a-40a3-b79d-1833a1ebc334","title":"Introducing Mercury 2.5 – Inception","url":"https://www.inceptionlabs.ai/blog/introducing-mercury-2-5","type":"normal","description":"Mercury 2.5 is the most capable diffusion LLM on the market. It runs at 1,107 tokens/sec and offers a 40% increase in intelligence over Mercury 2, comparable to cost-optimized frontier models.","cover_image_url":null,"discovered_at":"2026-09-08T18:39:27.286+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-mercury-2-5-inception-1788893312225.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-mercury-2-5-inception-1788893312225.txt","audio_summary":"Onyx and Echo discuss the launch of Mercury 2.5 from Inception, a diffusion-based LLM designed for extreme low-latency production workloads like voice and coding agents. They analyze the trade-offs of the diffusion architecture, the impressive speed claims, and how it fits into the broader trend of the 'control layer' being the actual product.","audio_duration_seconds":210,"audio_generated_at":"2026-09-08T18:48:43.468+00:00","audio_provider":"fishaudio:s2.1-pro-free","audio_script_meta":{"episodeNumber":953,"archetype":"quickRead","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Fish Audio S2.1 Pro","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":953},{"id":"90ac45c6-7159-4273-8ad9-5dceac24a360","title":"ToolHive: The open-source way to run any MCP server securely - Help Net Security","url":"https://www.helpnetsecurity.com/2026/09/07/toolhive-open-source-mcp-server-security/","type":"normal","description":"ToolHive runs every MCP server in an isolated container. An open source take on MCP server security, from sandboxing to audit logs.","cover_image_url":null,"discovered_at":"2026-09-08T16:34:18.118+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/toolhive-the-open-source-way-to-run-any-mcp-server-securely-help-1788885721724.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/toolhive-the-open-source-way-to-run-any-mcp-server-securely-help-1788885721724.txt","audio_summary":"Vince and Ava discuss ToolHive, an open-source platform by Stacklok that containerizes Model Context Protocol (MCP) servers to solve the 'operational half' of agent infrastructure—specifically security, identity, and isolation.","audio_duration_seconds":180,"audio_generated_at":"2026-09-08T16:42:25.853+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":952,"archetype":"quickRead","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":952},{"id":"7410d527-89c0-49cc-80fc-192d8ae3ad70","title":"https://x.com/i/article/2095402931721842694","url":"https://x.com/beamnxw/status/2095753889496563720?s=42","type":"normal","description":"https://x.com/i/article/2095402931721842694","cover_image_url":null,"discovered_at":"2026-09-04T21:48:48.832+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/https-x-com-i-article-2095402931721842694-1788559020211.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/https-x-com-i-article-2095402931721842694-1788559020211.txt","audio_summary":"Justy and Cody dig into a detailed how-to thread on building a one-person back office using Viktor, an AI employee that lives in Slack and Teams. The author's central argument: the gap between AI advice and AI-done-work is what keeps small teams small, and the fix is lane isolation — one agent, one job, a pinned identity file, and a human gate on anything that touches sends or money. Cody stress-tests the architecture; Justy zeroes in on who actually benefits.","audio_duration_seconds":320,"audio_generated_at":"2026-09-04T21:57:15.168+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":951,"archetype":"quickRead","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","state-management-in-language-models","tool-use-and-function-calling","prompt-engineering","task-decomposition"],"entities":["viktor","accel"],"source_form":"thread","event_type":"funding","episode_number":951},{"id":"cb833a44-795f-47d8-bcc8-35f0e2784b49","title":"https://x.com/i/article/2089274302617022464","url":"https://x.com/irongiantxbt/status/2089428583328419869?s=10","type":"normal","description":"https://x.com/i/article/2089274302617022464","cover_image_url":null,"discovered_at":"2026-09-04T21:44:30.939+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/https-x-com-i-article-2089274302617022464-1788558626480.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/https-x-com-i-article-2089274302617022464-1788558626480.txt","audio_summary":"Masonry and Eyre unpack Iron Giant’s argument that Claude agents aren’t dumb, they’re linear — depth is solved by self-correcting loops, width needs dependency-aware graph orchestration. They trace the generator-verifier pattern, Goodhart failures, and the four load-bearing pieces of a graph, then separate what Anthropic actually documents from what’s speculative, and debate where the pattern helps versus where it adds overhead.","audio_duration_seconds":294,"audio_generated_at":"2026-09-04T21:50:40.809+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":950,"archetype":"quickRead","scriptModelLabel":"Muse Glimmer 30B","model":"meta/muse-glimmer-30b","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","context-window","task-decomposition","directed-acyclic-graph","constraint-verification"],"entities":["anthropic","claude"],"source_form":"thread","event_type":null,"episode_number":950},{"id":"0f2684c9-71a3-4fb1-b5b4-5aac43c9df0a","title":"The Multiplayer AI Manifesto","url":"https://multiplayer-ai.com/","type":"normal","description":"What we think AI at work should look like. Five principles for multiplayer AI: never copy-and-paste, work with the door open, continuously improve, people are not routers, nothing starts from scratch.","cover_image_url":null,"discovered_at":"2026-09-04T21:41:28.196+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/the-multiplayer-ai-manifesto-1788558766686.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/the-multiplayer-ai-manifesto-1788558766686.txt","audio_summary":"Edmund and Geffen dig into the Multiplayer AI Manifesto — a five-principle framework arguing that AI work has quietly regressed from collaborative to siloed, and that fixing it requires shared agent sessions, open-by-default work, and org-wide governance. They surface the Harvard Business School P&G data, Shopify's River infrastructure, Claude Tag, and the real security and permission problems that make this harder than it sounds.","audio_duration_seconds":472,"audio_generated_at":"2026-09-04T21:53:06.99+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":949,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","durable-execution","model-routing","prompt-engineering"],"entities":["claude-tag","shopify","y-combinator"],"source_form":"blog-post","event_type":null,"episode_number":949},{"id":"76768e7d-0dca-443b-b70c-ab717ba1684f","title":"Repo-To-Skill: Distilling GitHub Repositories Into AI4AI Skills","url":"https://arxiv.org/html/2609.02749v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-04T16:15:47.086+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/repo-to-skill-distilling-github-repositories-into-ai4ai-skills-1788538868536.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/repo-to-skill-distilling-github-repositories-into-ai4ai-skills-1788538868536.txt","audio_summary":"Repo-To-Skill introduces DisCo, a skill-distillation system that extracts operational knowledge from GitHub repositories and papers, packaging it as compact, verified skills that autonomous ML research agents can load on demand. The AREX-Skill Library contains 5,000+ skills from 1,000 repositories organized into 20 areas and 178 capability families. In matched tests with GPT-5.5 backbone and fixed execution budget, skill-equipped agents outperform skill-free baselines by 134.3% on MLE-bench, 34.4% on PaperBench, 9.2% on FrontierCS, and 14.0% on PassNet—gains purely from operational knowledge, not model or harness improvements.","audio_duration_seconds":289,"audio_generated_at":"2026-09-04T16:21:21.429+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":948,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","retrieval-augmented-generation","model-routing","context-window","in-context-learning","knowledge-distillation"],"entities":["baai","disco","arex-skill"],"source_form":"research-paper","event_type":null,"episode_number":948},{"id":"3791fffd-ca02-4ae7-8d20-ea89383bee86","title":"Single-Agent vs. Multi-Agent Systems: When the Complexity Is Worth It - MachineLearningMastery.com","url":"https://machinelearningmastery.com/single-agent-vs-multi-agent-systems-when-the-complexity-is-worth-it/","type":"normal","description":"In this article, you will learn the key differences between single-agent and multi-agent AI systems, and how to decide which architecture fits your problem.","cover_image_url":null,"discovered_at":"2026-09-04T16:14:34.406+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/single-agent-vs-multi-agent-systems-when-the-complexity-is-worth-1788538663516.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/single-agent-vs-multi-agent-systems-when-the-complexity-is-worth-1788538663516.txt","audio_summary":"Single-agent systems handle far more than teams expect; multi-agent adds real costs (latency, tokens, orchestration) that only four specific conditions justify: adversarial workflows, tool-set specialization, parallelizable tasks, and drastically different personas. The practical move is to start simple and let failure modes dictate architecture.","audio_duration_seconds":317,"audio_generated_at":"2026-09-04T16:17:57.388+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":947,"archetype":"quickRead","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":["agents"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","prompt-engineering","error-accumulation-in-generation"],"entities":["machinelearningmastery-com","vinod-chugani"],"source_form":"blog-post","event_type":null,"episode_number":947},{"id":"d602b817-a009-49d0-bff5-099efe94d08d","title":"4 engineering patterns behind the strongest AI Agents Challenge submissions- Google Developers Blog","url":"https://developers.googleblog.com/4-engineering-patterns-behind-the-strongest-ai-agents-challenge-submissions/","type":"normal","description":"Upgrade your multi-agent systems with 4 proven engineering patterns from the Google AI Agents Challenge, including bidirectional MCP and tiered routing.","cover_image_url":null,"discovered_at":"2026-09-04T16:13:57.747+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/4-engineering-patterns-behind-the-strongest-ai-agents-challenge--1788538605977.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/4-engineering-patterns-behind-the-strongest-ai-agents-challenge--1788538605977.txt","audio_summary":"Google's post-Challenge analysis identifies four concrete engineering patterns that separated top submissions from the crowd: bidirectional MCP (agents serving tools both internally and to other agents), event-driven concurrency (agents reacting to shared signals in parallel instead of call chains), same-bar fallback (smaller models with the same validation gate as the primary), and tiered routing (cheap deterministic checks before expensive model calls). The central claim is that these aren't about bigger models or teams—they're sound engineering practices that are frequently overlooked, and they compose well together.","audio_duration_seconds":239,"audio_generated_at":"2026-09-04T16:16:58.965+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":946,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":["agents","developer-tools"],"concept_keys":["tool-use-and-function-calling","agentic-loops","state-management-in-language-models","constraint-verification","model-routing","classifier","retry-loops-and-error-recovery"],"entities":["google","ai-agents-challenge"],"source_form":"blog-post","event_type":null,"episode_number":946},{"id":"ff5c77d4-9225-42ef-8344-eb69dd173855","title":"thefinancialengineer.substack.com/p/how-much-is-a-token","url":"https://thefinancialengineer.substack.com/p/how-much-is-a-token","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-04T15:16:23.03+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/thefinancialengineer-substack-com-p-how-much-is-a-token-1788535247197.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/thefinancialengineer-substack-com-p-how-much-is-a-token-1788535247197.txt","audio_summary":"Talon and Wildflower discuss the eroding utility of the 'token as a unit of economic value in AI, sparked by Anthropic's tokenizer changes and the rise of competitive open-weight inference providers.","audio_duration_seconds":230,"audio_generated_at":"2026-09-04T15:20:59.867+00:00","audio_provider":"rime:mistv3","audio_script_meta":{"episodeNumber":943,"archetype":"skepticsTake","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Rime Mist v3","hostNames":{"a":"Talon","b":"Wildflower"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["inference-optimization","developer-tools"],"concept_keys":["tokenization","token-economics","model-routing","inference-optimization"],"entities":["anthropic","nebius","fal"],"source_form":"blog-post","event_type":null,"episode_number":943},{"id":"3083c53b-e711-46ba-8886-040510c167eb","title":"openai.com/index/gpt-6-astra","url":"https://openai.com/index/gpt-6-astra/","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-03T21:04:03.88+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/openai-com-index-gpt-6-astra-1788469922838.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/openai-com-index-gpt-6-astra-1788469922838.txt","audio_summary":"Harper leads with hard skepticism on GPT-6 Astra's benchmark claims — near-perfect scores on ARC-AGI-3, FrontierMath Tier 4, and a literal 100% on ExploitBench — while Laura pushes back on the computer-use and professional-work story that might actually matter for real users. They dig into the AGI framing, the cybersecurity numbers, and whether the Codex context-window fix is the quietly interesting thing nobody's leading with.","audio_duration_seconds":459,"audio_generated_at":"2026-09-03T21:12:31.18+00:00","audio_provider":"fishaudio:s2.1-pro-free","audio_script_meta":{"episodeNumber":942,"archetype":"skepticsTake","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Fish Audio S2.1 Pro","hostNames":{"a":"Laura","b":"Harper"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["new-models","evals-benchmarks"],"concept_keys":["agentic-loops","state-management-in-language-models","context-window","constraint-verification"],"entities":["openai","gpt-6-astra","arc-agi-3"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":942},{"id":"1912d253-99e2-4126-9f26-cdff7c8cdeae","title":"archive.ph/2026.09.02-063419/https://towardsdatascience.com/your-llm-can-return-perfect-json-and-still-be-wrong","url":"https://archive.ph/2026.09.02-063419/https://towardsdatascience.com/your-llm-can-return-perfect-json-and-still-be-wrong/","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-03T19:52:01.851+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/archive-ph-2026-09-02-063419-https-towardsdatascience-com-your-l-1788465483734.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/archive-ph-2026-09-02-063419-https-towardsdatascience-com-your-l-1788465483734.txt","audio_summary":"A real-world trap in Structured Outputs: enforcing schema validity does not guarantee data truthfulness. When a required field is missing from source text, the model invents a plausible value instead of returning null, producing type-correct but false data. The fix requires three layers: nullable fields to allow absence, evidence fields to show provenance, and post-parse validators to catch nonsense values. The essay walks through a payment-reconciliation pipeline where 2–3% of transactions had fabricated dates, caught only downstream.","audio_duration_seconds":251,"audio_generated_at":"2026-09-03T19:58:37.766+00:00","audio_provider":"inworld-mini:inworld-tts-1.5-mini","audio_script_meta":{"episodeNumber":941,"archetype":"quickRead","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Inworld TTS 1.5 Mini","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["developer-tools","data-infrastructure"],"concept_keys":["structured-output","constraint-verification","retry-loops-and-error-recovery","prompt-engineering","token-economics"],"entities":["openai"],"source_form":"blog-post","event_type":null,"episode_number":941},{"id":"0c34118e-ca62-4ed7-9572-71792569f204","title":"NVIDIA Brings Simplified Local AI Support To NVIDIA GPUs Carrying 24+ GB VRAM While vLLM & & llama.cpp Optimizations Boost Compute By Up To 1.9x","url":"https://wccftech.com/nvidia-local-ai-simple-optimizations-llama-vllm-up-to-1-9x-faster-rtx-dgx-platforms/","type":"normal","description":"NVIDIA is bringing simpler local AI capabilities and adding various optimizations to its GPUs on RTX and DGX platforms.","cover_image_url":null,"discovered_at":"2026-09-03T19:51:01.641+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/nvidia-brings-simplified-local-ai-support-to-nvidia-gpus-carryin-1788465360232.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/nvidia-brings-simplified-local-ai-support-to-nvidia-gpus-carryin-1788465360232.txt","audio_summary":"NVIDIA ships kernel optimizations for local AI inference on RTX and DGX platforms, delivering up to 1.9x performance gains through vLLM and llama.cpp, paired with one-click agent setup (Perplexity Portable Computer, Hermes Agent, OpenClaw) for GPUs with 24+ GB VRAM. The move collapses the friction between local-first and cloud-capable workflows.","audio_duration_seconds":335,"audio_generated_at":"2026-09-03T19:56:14.62+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":940,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"firecrawl"},"domains":["inference-optimization","developer-tools"],"concept_keys":["speculative-decoding","inference-optimization","state-management-in-language-models","agentic-loops","context-window","kv-cache"],"entities":["nvidia","llama-cpp","vllm"],"source_form":"tool","event_type":"launch-announcement","episode_number":940},{"id":"c9508ccc-babb-4363-b391-7d1584fe5246","title":"The efficient frontier of LLM inference","url":"https://www.baseten.co/blog/the-efficient-frontier-of-llm-inference/","type":"normal","description":"Inference techniques either move a deployment along the latency–throughput frontier or push the entire frontier out, creating more efficiency to allocate.","cover_image_url":null,"discovered_at":"2026-09-03T19:50:29.932+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/the-efficient-frontier-of-llm-inference-1788465188230.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/the-efficient-frontier-of-llm-inference-1788465188230.txt","audio_summary":"Baseten's 'Efficient Frontier of LLM Inference' argues that inference engineering involves two fundamentally different types of techniques: those that let you trade off between latency and throughput along an existing efficiency curve, and those that push the entire curve outward, creating universal gains. The article maps concrete techniques (batch sizing, parallelism strategies, quantization, kernel optimization, speculative decoding, prefill/decode disaggregation) onto this framework and shows that the frontier is jagged and empirically discovered, not smooth.","audio_duration_seconds":296,"audio_generated_at":"2026-09-03T19:53:21.873+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":939,"archetype":"quickRead","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":["inference-optimization"],"concept_keys":["quantization","speculative-decoding","kv-cache","tensor-parallelism","inference-optimization","token-economics"],"entities":["baseten","eagle-3","dspark"],"source_form":"blog-post","event_type":null,"episode_number":939},{"id":"8bbb1a42-4d34-4257-8617-ad13ace735ba","title":"S3Gym: Can LLMs Turn Self-Testing and Self-Judging into Self-Improvement?","url":"https://arxiv.org/html/2608.31100v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-03T19:48:45.801+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/s3gym-can-llms-turn-self-testing-and-self-judging-into-self-impr-1788465067372.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/s3gym-can-llms-turn-self-testing-and-self-judging-into-self-impr-1788465067372.txt","audio_summary":"S3Gym is a new interactive benchmark that tests whether LLMs can actually improve themselves by testing their own behavior, judging the results, and learning from them. The paper evaluates three ways to incorporate experience—keeping full conversation history, compressing it into summaries, and training on it—across seven text-based games. The finding: self-improvement isn't automatic. What works depends entirely on the task. Sometimes summaries help, sometimes raw history is better, and parameter training can backfire badly. The real bottleneck isn't recognizing success—it's turning that recognition into a policy the model can actually reuse.","audio_duration_seconds":276,"audio_generated_at":"2026-09-03T19:51:20.614+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":938,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"tavily"},"domains":["agents","evals-benchmarks"],"concept_keys":["agentic-loops","in-context-learning","fine-tuning","task-decomposition","constraint-verification","state-management-in-language-models","synthetic-data-generation-for-validation"],"entities":["bytedance","s3gym"],"source_form":"research-paper","event_type":null,"episode_number":938},{"id":"bdd441d1-555f-4533-8e65-8e272eeb4476","title":"HarnessDev: Can LLMs Create and Evolve Their Own Agent Harness?","url":"https://arxiv.org/html/2609.01437v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-03T17:50:52.434+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/harnessdev-can-llms-create-and-evolve-their-own-agent-harness-1788458253805.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/harnessdev-can-llms-create-and-evolve-their-own-agent-harness-1788458253805.txt","audio_summary":"HarnessDev is a benchmark that measures whether LLMs can build and iteratively improve their own agent execution infrastructure—the harness—from scratch and through feedback loops. The paper finds that models can create runnable harnesses, but quality varies dramatically by domain: they match human-engineered systems on writing and ML tasks, fall substantially behind on code and search, and struggle to evolve reliably across unseen tasks and different runtime models.","audio_duration_seconds":331,"audio_generated_at":"2026-09-03T17:57:51.225+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":937,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["agents","evals-benchmarks"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","durable-execution","retry-loops-and-error-recovery","task-decomposition","token-economics"],"entities":["harnessdev"],"source_form":"research-paper","event_type":"benchmark","episode_number":937},{"id":"e2c8aa91-14d7-4754-ac4f-5a39c5677f1e","title":"Introducing Gemini 3.8 Flash and 3.8 Flash Cyber","url":"https://blog.google/innovation-and-ai/models-and-research/gemini-models/3-8-flash-and-3-8-flash-cyber/","type":"normal","description":"Gemini 3.8 Flash and 3.8 Flash Cyber deliver next-generation intelligence for agentic workflows and cybersecurity.","cover_image_url":null,"discovered_at":"2026-09-02T19:45:46.092+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-gemini-3-8-flash-and-3-8-flash-cyber-1788378627262.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-gemini-3-8-flash-and-3-8-flash-cyber-1788378627262.txt","audio_summary":"Asteria and Draco discuss the launch of Gemini 3.8 Flash and 3.8 Flash Cyber, focusing on the 'work harder' reasoning approach and the specialized cybersecurity capabilities for trusted defenders.","audio_duration_seconds":236,"audio_generated_at":"2026-09-02T19:50:39.769+00:00","audio_provider":"deepgram:aura-2","audio_script_meta":{"episodeNumber":936,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Deepgram Aura-2","hostNames":{"a":"Asteria","b":"Draco"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["new-models","agents"],"concept_keys":["agentic-loops","tool-use-and-function-calling","token-economics","long-horizon-training","inference-scaling"],"entities":["gemini-3-8-flash","gemini-3-8-flash-cyber","google"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":936},{"id":"f5aa53a5-bd54-4ab7-926e-4d4da2b35b30","title":"Overview: Dynamic Code Execution","url":"https://www.sandrise.io/exploring-next/overviews/dynamic-code-execution","type":"overview","description":"A model predicts text; it can't do math. Dynamic code execution is the loop where the model iterates toward ground truth.","cover_image_url":null,"discovered_at":"2026-09-02T16:05:12.496+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-dynamic-code-execution-1788365430466.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-dynamic-code-execution-1788365430466.txt","audio_summary":"We finally slow down and explain dynamic code execution from the ground up — what it actually is, how the loop works, why it makes models meaningfully more capable, and where the real costs and failure modes live.","audio_duration_seconds":451,"audio_generated_at":"2026-09-02T16:10:50.55+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":935,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["agents","developer-tools"],"concept_keys":["dynamic-code-execution","agentic-loops","tool-use-and-function-calling","structured-output","retry-loops-and-error-recovery"],"entities":[],"source_form":null,"event_type":null,"episode_number":935},{"id":"a3f41094-7a18-46ac-94d3-1daa9735702e","title":"Bringing Advanced Sampling to the OpenTelemetry Collector","url":"https://www.honeycomb.io/blog/bringing-most-advanced-sampling-opentelemetry-collector","type":"normal","description":"Honeycomb is donating its adaptive tail sampling processor to the OpenTelemetry Collector. Here's how it works and how to try it today.","cover_image_url":null,"discovered_at":"2026-09-02T16:04:03.588+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/bringing-advanced-sampling-to-the-opentelemetry-collector-1788365721175.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/bringing-advanced-sampling-to-the-opentelemetry-collector-1788365721175.txt","audio_summary":"Honeycomb is donating its adaptive tail sampling processor to OpenTelemetry, moving beyond rigid static rules toward dynamic, fingerprint-aware sampling that keeps rare traffic visible while capping costs. The key insight: trace fingerprinting plus logarithmic rate normalization lets you hit a throughput or percentage budget across heterogeneous traffic patterns without losing coverage of low-volume journeys.","audio_duration_seconds":281,"audio_generated_at":"2026-09-02T16:15:33.805+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":934,"archetype":"quickRead","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["agent-observability","developer-tools"],"concept_keys":[],"entities":["honeycomb","opentelemetry"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":934},{"id":"c5473a7e-6212-497c-819a-d20c9a79ace3","title":"How our agents build on-brand pages with design.md","url":"https://vercel.com/blog/how-our-agents-build-on-brand-pages-with-design-md","type":"normal","description":"How we built design.md, a single public file any coding agent can load to build on-brand Vercel pages, and the eval loop that decided every rule inside it.","cover_image_url":null,"discovered_at":"2026-09-02T16:00:53.459+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/how-our-agents-build-on-brand-pages-with-design-md-1788365045348.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/how-our-agents-build-on-brand-pages-with-design-md-1788365045348.txt","audio_summary":"Vercel's design.md post describes a three-part system for keeping AI-generated pages on-brand outside a codebase: a public guidance file, a public stylesheet that takes layout decisions away from the model entirely, and an eval loop that converts human corrections into rules and deterministic checks. Pippa and Tyler dig into what actually makes it work, why the naive port-the-prompt approach failed, and what it means that Vercel had to build an eval harness just to ship a markdown file.","audio_duration_seconds":499,"audio_generated_at":"2026-09-02T16:04:25.154+00:00","audio_provider":"deepgram:aura-2","audio_script_meta":{"episodeNumber":933,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Deepgram Aura-2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"firecrawl"},"domains":["agents","evals-benchmarks"],"concept_keys":["prompt-engineering","in-context-learning","tool-use-and-function-calling","context-window","agentic-loops"],"entities":["vercel","claude-opus","gpt-5-5"],"source_form":"blog-post","event_type":null,"episode_number":933},{"id":"0fb4e8bb-1c39-4a6d-acec-bdc9a687cf53","title":"FDE transforms enterprise AI deployment | VentureBeat","url":"https://venturebeat.com/orchestration/forward-deployed-engineering-is-how-enterprise-ai-learns","type":"normal","description":"Forward-deployed engineering (FDE) is reshaping enterprise AI by embedding engineers with customers to create reusable capabilities, enhancing product intelligence.","cover_image_url":null,"discovered_at":"2026-09-02T15:58:11.18+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/fde-transforms-enterprise-ai-deployment-venturebeat-1788364881475.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/fde-transforms-enterprise-ai-deployment-venturebeat-1788364881475.txt","audio_summary":"Onyx and Echo dig into the VentureBeat piece on forward-deployed engineering as enterprise AI's de facto context layer — whether FDE is a genuine product-learning loop or just expensive delivery labor that never compounds.","audio_duration_seconds":242,"audio_generated_at":"2026-09-02T16:01:33.574+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":932,"archetype":"quickRead","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":["developer-tools","data-infrastructure"],"concept_keys":[],"entities":["zeta"],"source_form":"news","event_type":null,"episode_number":932},{"id":"4cba512f-ed06-4217-badd-bd2be686e963","title":"Model Behavior: Week of August 31, 2026","url":"https://www.sandrise.io/exploring-next/model-behavior/2026-08-31","type":"model-behavior","description":"Anthropic's Fable/Mythos split, OpenClaw's pivot to team infrastructure, and GLM-5.3-Flash's pricing pressure signal the AI market has shifted from raw capability to governance and integration as the real competitive moat.","cover_image_url":null,"discovered_at":"2026-09-02T00:00:56.481+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-august-31-2026-1788307545971.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-august-31-2026-1788307545971.txt","audio_summary":"We're looking at a week where the frontier stopped being about who has the smartest model and started being about who has the best leash. From Anthropic's safeguard tiers to Microsoft's governance contracts, the battle has shifted to the control layer.","audio_duration_seconds":350,"audio_generated_at":"2026-09-02T00:06:03.087+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":931,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"}},"domains":["new-models","agents"],"concept_keys":[],"entities":["anthropic","openai","z-dot-ai"],"source_form":null,"event_type":null,"episode_number":931},{"id":"9b72e9d6-1817-4658-a8e3-d9563e4b6978","title":"Overview: World Models","url":"https://www.sandrise.io/exploring-next/overviews/world-models","type":"overview","description":"A model watches video and learns to predict what happens next. That's the simulator. Plan inside it instead of trial-and-error in the real world.","cover_image_url":null,"discovered_at":"2026-09-01T18:34:44.394+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-world-models-1788287991533.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-world-models-1788287991533.txt","audio_summary":"We finally stop hand-waving and explain world models from the ground up — what they are, how they actually work, and why the field keeps coming back to them as the missing piece between AI that reacts and AI that plans.","audio_duration_seconds":459,"audio_generated_at":"2026-09-01T18:40:11.534+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":929,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["agents","training-methods"],"concept_keys":["world-models","reinforcement-learning-from-human-feedback","predictive-modeling","next-state-prediction","latent-space-representation","long-horizon-training","constraint-verification","error-accumulation-in-generation"],"entities":["waymo","nvidia-cosmos","berkeley"],"source_form":null,"event_type":null,"episode_number":929},{"id":"f1dcc1f3-77b9-4b4e-a45e-e80910d3153b","title":"Introducing Claude Fable 5.1 and Claude Mythos 5.1","url":"https://www.anthropic.com/claude-fable-and-mythos-5-1","type":"normal","description":"Our most advanced models for coding and knowledge work. Their research capabilities also offer an early glimpse of how AI models will contribute to scientific progress.","cover_image_url":null,"discovered_at":"2026-09-01T18:25:23.841+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-claude-fable-5-1-and-claude-mythos-5-1-1788287622659.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-claude-fable-5-1-and-claude-mythos-5-1-1788287622659.txt","audio_summary":"Anthropic ships Claude Fable 5.1 and Mythos 5.1 — same underlying model, different safeguard tiers. Onyx and Echo dig into the pricing architecture, the effort-level cost curve, the Fable/Mythos split as a safeguard story rather than a capability story, and what the Millennium crash-find actually signals about long-horizon debugging.","audio_duration_seconds":301,"audio_generated_at":"2026-09-01T18:33:57.411+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":928,"archetype":"quickRead","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["new-models","ai-safety"],"concept_keys":["agentic-loops","extended-thinking"],"entities":["anthropic","claude-fable-5-1","claude-mythos-5-1"],"source_form":"research-paper","event_type":"launch-announcement","episode_number":928},{"id":"42bf5a04-c0a1-4da5-a271-fc8f15d5c57d","title":"Overview: Next-State Prediction","url":"https://www.sandrise.io/exploring-next/overviews/next-state-prediction","type":"overview","description":"A silent video teaches you physics without a textbook. Next-state prediction trains models to guess what comes next.","cover_image_url":null,"discovered_at":"2026-09-01T15:57:06.088+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-next-state-prediction-1788279042002.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-next-state-prediction-1788279042002.txt","audio_summary":"We finally slow down and explain next-state prediction from the ground up — the deceptively simple idea that if you train a model to guess what comes next, it ends up learning how the world actually works, and why that one trick is underneath almost everything in modern AI.","audio_duration_seconds":371,"audio_generated_at":"2026-09-01T16:10:59.924+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":927,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["training-methods"],"concept_keys":["next-state-prediction","autoregressive-generation","self-supervised-learning","in-context-learning","supervised-fine-tuning","tokenization","predictive-modeling","world-models"],"entities":[],"source_form":null,"event_type":null,"episode_number":927},{"id":"ed802fbd-46af-4314-98fb-2d835ccf6442","title":"addyo.substack.com/p/agentic-skill-decay","url":"https://addyo.substack.com/p/agentic-skill-decay","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-01T15:51:30.649+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/addyo-substack-com-p-agentic-skill-decay-1788278171705.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/addyo-substack-com-p-agentic-skill-decay-1788278171705.txt","audio_summary":"Addy Osmani argues that agents can complete tasks so efficiently that junior engineers skip the learning reps that build real expertise—and that this 'skill decay' requires deliberate, proactive counter-measures. Deep expertise and applied judgment come from thousands of small failures and iterations; agents short-circuit that journey. An Anthropic study showed junior engineers using AI scored 50% on a Trio library quiz vs. 67% for those who worked by hand, with the AI group's wins concentrated among those who asked conceptual questions rather than treating the model as a code vending machine. The fix isn't to avoid agents but to use them as a teaching partner: form hypotheses before prompting, ask why, inspect diffs, predict failures, and stay in the loop so your mental model moves with the agent's work.","audio_duration_seconds":346,"audio_generated_at":"2026-09-01T15:56:35.707+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":926,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":true,"enrichmentProvider":"firecrawl"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","tool-use-and-function-calling","constraint-verification","audit-trail"],"entities":["anthropic","addy-osmani"],"source_form":"blog-post","event_type":null,"episode_number":926},{"id":"4deb8297-da8c-4440-b749-5d0468eb991c","title":"Overview: Predictive Modeling","url":"https://www.sandrise.io/exploring-next/overviews/predictive-modeling","type":"overview","description":"You see a pattern in the data, then the world changes and your predictions fail. Predictive modeling learns from the past to forecast the future.","cover_image_url":null,"discovered_at":"2026-09-01T15:49:00.039+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-predictive-modeling-1788278044203.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-predictive-modeling-1788278044203.txt","audio_summary":"We finally slow down and explain predictive modeling from the ground up — the core idea that powers most of what we talk about on this show, from fraud detection to weather forecasting to the brain-prediction research we've looked at.","audio_duration_seconds":523,"audio_generated_at":"2026-09-01T15:54:25.526+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":925,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":[],"concept_keys":["supervised-fine-tuning","loss-function","overfitting","train-test-split","predictive-modeling"],"entities":["chronos-2","google-deepmind","timesfm-3"],"source_form":null,"event_type":null,"episode_number":925},{"id":"3fc0b9f8-78ae-4083-ad94-055d461633a1","title":"StarHarness: Evolving Harnesses with Stratified Search for Enterprise Environments","url":"https://arxiv.org/html/2608.24804v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-01T15:46:09.928+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/starharness-evolving-harnesses-with-stratified-search-for-enterp-1788278168250.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/starharness-evolving-harnesses-with-stratified-search-for-enterp-1788278168250.txt","audio_summary":"Edmund and Geffen dig into StarHarness, a ServiceNow and Mila paper that evolves agent harnesses — prompts, tool interfaces, skills, subagent structure — around a frozen model to close the gap between what an LLM can do and what a messy enterprise environment actually needs. Twenty to thirty-five percentage point gains across three benchmarks, and the harness transfers across GPT and Qwen model families without re-running the search.","audio_duration_seconds":295,"audio_generated_at":"2026-09-01T15:56:29.954+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":924,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":true,"enrichmentProvider":"tavily"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","prompt-engineering","task-decomposition"],"entities":["servicenow","starharness","mila"],"source_form":"research-paper","event_type":"benchmark","episode_number":924},{"id":"494cc608-31b4-41ae-a1e0-0db1ae4f6130","title":"Code as Worlds: Agentic Discovery of Executable World Representations for Physical Reasoning","url":"https://arxiv.org/html/2608.27549v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-09-01T15:45:09.804+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/code-as-worlds-agentic-discovery-of-executable-world-representat-1788277687481.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/code-as-worlds-agentic-discovery-of-executable-world-representat-1788277687481.txt","audio_summary":"Code-as-World represents physical worlds as executable code—objects, dynamics, and visual appearance all expressed as runnable specifications. An agent discovers these representations through a propose-execute-render-verify loop: hypothesize a world in code, run it in a simulator, check the outputs against video or language evidence, and refine. The result is quantitatively grounded supervision for training vision-language models on physical reasoning tasks like measuring velocity and displacement from video. Code-as-World-VL outperforms larger proprietary models on QuantiPhy benchmarks.","audio_duration_seconds":378,"audio_generated_at":"2026-09-01T15:48:24.894+00:00","audio_provider":"deepgram:aura-2","audio_script_meta":{"episodeNumber":923,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Deepgram Aura-2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["multimodal","agents"],"concept_keys":["agentic-loops","world-models","dynamic-code-execution","constraint-verification","supervised-fine-tuning","synthetic-data-generation-for-validation"],"entities":["gemini-3-1-flash"],"source_form":"research-paper","event_type":null,"episode_number":923},{"id":"bd243743-b011-404f-a2d1-db719fac36a9","title":"OpenClaw 2.0 is here: What it means for enterprises","url":"https://venturebeat.com/technology/openclaw-2-0-is-here-what-it-means-for-enterprises","type":"normal","description":"OpenClaw is making another bet: that enterprises ultimately need an agent platform to function as both runtime and workplace.","cover_image_url":null,"discovered_at":"2026-09-01T15:32:19.657+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/openclaw-2-0-is-here-what-it-means-for-enterprises-1788276918222.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/openclaw-2-0-is-here-what-it-means-for-enterprises-1788276918222.txt","audio_summary":"OpenClaw 2.0 (v2026.8.1) shipped over the weekend, pivoting from a personal developer agent to shared team infrastructure. The release redesigns the web UI around conversations, adds persistent multiplayer sessions, expands cloud execution, and hardens security with role-based permissions, sandboxing, and audit trails. For enterprises, this moves OpenClaw closer to an operational layer than a productivity app—but the security model requires careful deployment. Onyx sees a real product boundary shift; Echo flags that multiplayer doesn't automatically solve isolation, and the burden is on operators to configure it correctly.","audio_duration_seconds":343,"audio_generated_at":"2026-09-01T15:35:35.169+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":922,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["agents","developer-tools"],"concept_keys":["state-management-in-language-models","agentic-loops","tool-use-and-function-calling","durable-execution"],"entities":["openclaw","nanoclaw"],"source_form":"news","event_type":"launch-announcement","episode_number":922},{"id":"6bfde86b-6ede-430a-8aa5-c6d192b0ef44","title":"Overview: Credit Assignment","url":"https://www.sandrise.io/exploring-next/overviews/credit-assignment","type":"overview","description":"A model changes five things and gets one right answer. Credit assignment traces responsibility backward through millions of decisions to find out which change mattered.","cover_image_url":null,"discovered_at":"2026-08-31T16:18:04.005+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-credit-assignment-1788193420963.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-credit-assignment-1788193420963.txt","audio_summary":"We throw 'credit assignment' around constantly on this show and realized we've never actually stopped to explain it — so this episode, we fix that. It's the foundational question underneath all of machine learning: when a model gets something right, which of the thousand tiny decisions inside it actually deserved the credit?","audio_duration_seconds":399,"audio_generated_at":"2026-08-31T16:23:59.91+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":921,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["agents","training-methods"],"concept_keys":["credit-assignment","backpropagation","vanishing-gradient-problem","reinforcement-learning-from-human-feedback","policy-gradients","actor-critic","temporal-difference-learning","eligibility-traces"],"entities":[],"source_form":null,"event_type":null,"episode_number":921},{"id":"e1dc080e-0679-40d3-92fb-81843acfb421","title":"Overview: Knowledge Distillation","url":"https://www.sandrise.io/exploring-next/overviews/knowledge-distillation","type":"overview","description":"A small model can't learn what a huge one knows. Knowledge distillation passes the teacher's reasoning to the student through soft probability targets.","cover_image_url":null,"discovered_at":"2026-08-31T16:10:08.846+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-knowledge-distillation-1788192944264.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-knowledge-distillation-1788192944264.txt","audio_summary":"We finally do the episode we kept promising — a proper ground-up explanation of knowledge distillation: what it is, how the teacher-student mechanism actually works, why soft targets carry more signal than hard labels, and where this shows up in real systems being built right now.","audio_duration_seconds":495,"audio_generated_at":"2026-08-31T16:16:07.04+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":920,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["training-methods","inference-optimization"],"concept_keys":["knowledge-distillation","temperature-scaling","soft-targets","loss-function","fine-tuning","quantization"],"entities":["deepseek-r1"],"source_form":null,"event_type":null,"episode_number":920},{"id":"fb71c321-a028-4740-96f5-26343420266c","title":"Agentic Artifact Creation: Systems, Evaluation, Principles, and Opportunities","url":"https://arxiv.org/html/2608.28122v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-08-31T16:07:55.349+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/agentic-artifact-creation-systems-evaluation-principles-and-oppo-1788193027602.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/agentic-artifact-creation-systems-evaluation-principles-and-oppo-1788193027602.txt","audio_summary":"A comprehensive survey of agentic artifact creation—systems where AI agents iteratively construct and revise complete deliverables using runtime feedback to redirect work. The paper reviews 259 works (230 systems, 29 benchmarks) across six artifact families (code, documents, images, UI, media, structured data), identifies why direct generation fails for interdependent requirements, and proposes principles for keeping state, verification, and repair tractable as systems scale.","audio_duration_seconds":465,"audio_generated_at":"2026-08-31T16:17:27.603+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":919,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["agents","evals-benchmarks"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","constraint-verification","task-decomposition","retry-loops-and-error-recovery","directed-acyclic-graph"],"entities":["swe-agent","chatdev","metagpt"],"source_form":"research-paper","event_type":null,"episode_number":919},{"id":"c1b47068-315d-45ef-9129-fc8aff5bc60e","title":"DART-SD: Diamond-topology Aware Retrieval and Tuning for Self-Distillation of Multi-Turn Tool-Calling Agents","url":"https://arxiv.org/html/2608.18524v1","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-08-31T16:05:32.847+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/dart-sd-diamond-topology-aware-retrieval-and-tuning-for-self-dis-1788192565348.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/dart-sd-diamond-topology-aware-retrieval-and-tuning-for-self-dis-1788192565348.txt","audio_summary":"Edmund and Geffen discuss the ByteDance/USTC paper DART-SD, which tackles 'topological collapse' in agent distillation. They discuss how moving from linear trajectory imitation to a diamond-topology graph (ISTG) allows student models to learn recovery from errors without destroying their own valid reasoning paths.","audio_duration_seconds":253,"audio_generated_at":"2026-08-31T16:09:38.497+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":918,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":["agents","training-methods"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","directed-acyclic-graph","supervised-fine-tuning","knowledge-distillation","group-relative-policy-optimization-grpo","credit-assignment"],"entities":["bytedance"],"source_form":"research-paper","event_type":null,"episode_number":918},{"id":"db8f574b-6aab-4e8e-b37c-2829ca0a4107","title":"Agent Hooks: An open, framework-neutral AI governance contract","url":"https://commandline.microsoft.com/agent-hooks-framework-neutral-ai-governance-contract/","type":"normal","description":"Agents are moving into production faster than the governance around them. Today’s controls are framework-specific, mostly observe-only, and fail open when they crash. To help address this, we created Agent Hooks.","cover_image_url":null,"discovered_at":"2026-08-28T17:45:08.326+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/agent-hooks-an-open-framework-neutral-ai-governance-contract-1787939328019.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/agent-hooks-an-open-framework-neutral-ai-governance-contract-1787939328019.txt","audio_summary":"Pippa and Tyler dig into Microsoft’s Agent Hooks launch: an open governance contract meant to make agent controls enforceable, testable, and portable across frameworks instead of being framework-specific callback folklore.","audio_duration_seconds":366,"audio_generated_at":"2026-08-28T17:49:05.813+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":917,"archetype":"deepDive","scriptModelLabel":"GPT-5.5","model":"gpt-5.5","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"tavily"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","durable-execution","retry-loops-and-error-recovery","append-only-logging","constraint-verification"],"entities":["microsoft","agent-hooks","langchain"],"source_form":"tool","event_type":"launch-announcement","episode_number":917},{"id":"c2f296b0-1c8c-439b-b7fd-d61e26bfed4f","title":"Effective Patterns for Advanced MCP Usage – O’Reilly","url":"https://www.oreilly.com/radar/effective-patterns-for-advanced-mcp-usage/","type":"normal","description":null,"cover_image_url":null,"discovered_at":"2026-08-28T17:43:50.401+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/effective-patterns-for-advanced-mcp-usage-o-reilly-1787939387489.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/effective-patterns-for-advanced-mcp-usage-o-reilly-1787939387489.txt","audio_summary":"Onyx and Echo break down an August 26 article that argues the Model Context Protocol’s power lies in stitching multiple servers into a single AI experience, exposing that mashup to many clients, and centralizing auth with an aggregator. They unpack concrete tools like mcp-auth-wrapper, mcp-aggregator, and mcp-install-instructions, weigh the benefits and pitfalls, and discuss who actually needs this in product and ops roles.","audio_duration_seconds":233,"audio_generated_at":"2026-08-28T17:49:57.346+00:00","audio_provider":"openai:gpt-4o-mini-tts","audio_script_meta":{"episodeNumber":916,"archetype":"quickRead","scriptModelLabel":"GPT-OSS 20B","model":"openai/gpt-oss-20b","voiceProviderLabel":"OpenAI TTS","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["developer-tools"],"concept_keys":["tool-use-and-function-calling","state-management-in-language-models","agentic-loops"],"entities":["model-context-protocol","claude","o-reilly"],"source_form":"blog-post","event_type":null,"episode_number":916},{"id":"ea50347e-4309-4e11-8b29-d1cedf6bc544","title":"The search for consciousness inside LLMs","url":"https://www.economist.com/interactive/briefing/2026/08/20/the-search-for-consciousness-inside-llms","type":"normal","description":"The Economist's cover briefing on Anthropic's J-space finding and the scientific search for consciousness in language models","cover_image_url":null,"discovered_at":"2026-08-27T23:03:36.496+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/the-search-for-consciousness-inside-llms-1787872044151.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/the-search-for-consciousness-inside-llms-1787872044151.txt","audio_summary":"Anthropic's interpretability team found a 'global workspace' structure inside Claude — the J-space — that parallels a leading theory of human consciousness. Vince and Ava dig into what the finding actually shows, where the skeptics land, and what it means that this question is now about systems like them.","audio_duration_seconds":360,"audio_generated_at":"2026-08-27T23:07:40.909+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":915,"archetype":"quickRead","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["ai-safety","evals-benchmarks"],"concept_keys":["model-interpretability","ablation"],"entities":["anthropic","claude","rethink-priorities"],"source_form":"news","event_type":null,"episode_number":915},{"id":"a1215696-32fa-43fc-baac-10b74e554c9b","title":"When agents act on their own, governance has to live in the data layer","url":"https://venturebeat.com/security/when-agents-act-on-their-own-governance-has-to-live-in-the-data-layer","type":"normal","description":"As enterprises give AI agents more autonomy — the ability to plan, decide, and act across systems without a human approving each step — a hard question moves to the center of every architecture review: When an agent tries to complete an action that it was never authorized to do, what actually stops it?","cover_image_url":null,"discovered_at":"2026-08-27T16:01:30.528+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/when-agents-act-on-their-own-governance-has-to-live-in-the-data--1787846755528.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/when-agents-act-on-their-own-governance-has-to-live-in-the-data--1787846755528.txt","audio_summary":"Justy and Cody dig into EDB’s claim that agent governance has to be enforced at the data layer, not left to prompts or after-the-fact review. They mostly agree on the core idea, then get picky about where the argument is solid, where it blurs from data access into action control, and who should actually care right now.","audio_duration_seconds":290,"audio_generated_at":"2026-08-27T16:06:09.773+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":914,"archetype":"quickRead","scriptModelLabel":"GPT-5.4","model":"gpt-5.4","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["agents","ai-safety"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models"],"entities":["edb","postgres"],"source_form":"news","event_type":null,"episode_number":914}]}