{"items":[{"id":"edf11f68-7008-4c74-928c-f9c98ead6522","title":"Model Behavior: Week of September 21, 2026","url":"https://www.sandrise.io/exploring-next/model-behavior/2026-09-21","type":"model-behavior","description":"OpenAI's Agents API, Claude Code Projects, and the shift from model races to who controls the execution layer where agents actually run long-term work.","cover_image_url":null,"discovered_at":"2026-09-23T00:01:31.182+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-september-21-2026-1790121917474.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-september-21-2026-1790121917474.txt","audio_summary":"We read this week as the frontier splitting around agent execution: the winning move is less about one model leap and more about who controls the place where long-running work actually happens. The complication is that OpenAI, Anthropic, Google, DeepSeek, and the open-weight challengers are still making the parts better, cheaper, or sharper, so the stack is not replacing models so much as turning them into swappable pieces.","audio_duration_seconds":739,"audio_generated_at":"2026-09-23T00:05:41.905+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":1002,"archetype":"deepDive","scriptModelLabel":"GPT-5.5","model":"gpt-5.5","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"}},"domains":["agents","new-models"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","durable-execution","model-routing","retry-loops-and-error-recovery","context-window-management","token-economics"],"entities":["openai","anthropic","google"],"source_form":null,"event_type":null,"episode_number":1002},{"id":"5e5fb899-8d77-4136-be64-711c00aa0b61","title":"Introducing GPT-6 Sol and Luna","url":"https://openai.com/index/introducing-gpt-6-sol-and-luna/","type":"normal","description":"Meet GPT-6 Sol and Luna, two models that bring frontier intelligence to everyday work with different balances of capability and cost.","cover_image_url":null,"discovered_at":"2026-09-22T21:46:38.963+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-gpt-6-sol-and-luna-1790113878804.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-gpt-6-sol-and-luna-1790113878804.txt","audio_summary":"Edmund and Geffen dig into OpenAI’s GPT-6 Sol and Luna launch as a cost-curve move rather than a pure capability leap. They focus on the practical user story, the benchmark shape, the caching changes, and whether cheaper frontier-ish models actually change what ships.","audio_duration_seconds":297,"audio_generated_at":"2026-09-22T21:51:27.621+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":1001,"archetype":"skepticsTake","scriptModelLabel":"GPT-5.4 mini","model":"gpt-5.4-mini","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":true,"enrichmentProvider":"tavily"},"domains":["new-models","inference-optimization"],"concept_keys":["kv-cache","model-routing","token-economics","agentic-loops"],"entities":["openai","gpt-6-sol","gpt-6-luna"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":1001},{"id":"23e32047-2976-4554-8de9-79f96d4b6ef1","title":"Introducing Claude Opus 5.5","url":"https://www.anthropic.com/claude-opus-5-5","type":"normal","description":"Claude Opus 5.5 leads in agentic coding and knowledge work, and costs 40% less to run than Opus 5 on typical workloads.","cover_image_url":null,"discovered_at":"2026-09-22T21:45:52.648+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-claude-opus-5-5-1790113765670.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-claude-opus-5-5-1790113765670.txt","audio_summary":"Pippa and Tyler unpack Anthropic’s Claude Opus 5.5 launch — a cheaper, faster, Fable-level flagship tuned for long coding and agentic work, with heavier safety and alignment testing — and what it really changes for teams already on Opus 5 or Fable 5.1.","audio_duration_seconds":431,"audio_generated_at":"2026-09-22T21:49:38.946+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":1000,"archetype":"quickRead","scriptModelLabel":"GPT-5.1","model":"gpt-5.1","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["new-models","agents"],"concept_keys":["token-economics","kv-cache","agentic-loops","tool-use-and-function-calling","prompt-injection","inference-optimization"],"entities":["claude-opus-5-5","anthropic","openai"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":1000},{"id":"29a35272-e72c-4ad0-aa0f-9e5098eba049","title":"Harness-Zero: Harness Distillation via Agent-as-Harness","url":"https://arxiv.org/html/2609.24974v1","type":"normal","description":"Agent harnesses, the external systems that mediate model-environment interaction, can substantially improve agent performance, but their gains remain tied to the harness at deployment. Because the best harness varies across domains, instances, and models, a general-purpose agent must either settle for a suboptimal shared harness or route among an ever-growing set of specialized ones. We therefore study agent harness distillation: using a domain- or instance-optimized harness as training-time","cover_image_url":null,"discovered_at":"2026-09-22T15:40:31.882+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/1introduction-1790092199079.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/1introduction-1790092199079.txt","audio_summary":"Harness-Zero turns a great, evolved agent harness into training data so the underlying model permanently learns those behaviors and can run on a much simpler harness — often matching or even beating the original setup.","audio_duration_seconds":532,"audio_generated_at":"2026-09-22T15:50:15.215+00:00","audio_provider":"openai:gpt-4o-mini-tts","audio_script_meta":{"episodeNumber":999,"archetype":"deepDive","scriptModelLabel":"GPT-5.1","model":"gpt-5.1","voiceProviderLabel":"OpenAI TTS","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["agents","training-methods"],"concept_keys":["supervised-fine-tuning","knowledge-distillation","agentic-loops","tool-use-and-function-calling","state-management-in-language-models","in-context-learning"],"entities":["qwen","meta-harness"],"source_form":"research-paper","event_type":null,"episode_number":999},{"id":"c766b673-482f-448b-8a9a-5d55bb0a442c","title":"BI-Agent and BI-Bench: Towards Automating End-to-End Business Intelligence","url":"https://arxiv.org/html/2609.20886v1","type":"normal","description":"Business intelligence (BI) is a cornerstone of enterprise decision-making and is widely used by enterprise users in software such as Power BI and Tableau. In traditional BI workflows, users need to prepare data by (1) identifying relevant tables, (2) performing data transformations, and (3) building join relationships, before they can (4) answer their business questions. These steps can be complex and time-consuming, making BI challenging. Given the strong capabilities of large language models","cover_image_url":null,"discovered_at":"2026-09-22T15:39:48.27+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/bi-agent-and-bi-bench-towards-automating-end-to-end-business-int-1790091841083.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/bi-agent-and-bi-bench-towards-automating-end-to-end-business-int-1790091841083.txt","audio_summary":"Jessica and Cathy dig into BI-Agent and BI-Bench, a paper that tests whether language models can automate the messy end-to-end workflow behind business intelligence dashboards. They focus on the gap between natural-language SQL demos and real BI work: table selection, transformations, joins, and analysis over messy Power BI-style projects. Jessica sees a practical product path for internal analytics teams, while Cathy is impressed by the benchmark design and tool-augmented architecture, with caveats about dashboard-derived ground truth and public-data leakage.","audio_duration_seconds":348,"audio_generated_at":"2026-09-22T15:44:12.253+00:00","audio_provider":"cartesia:sonic-3.6","audio_script_meta":{"episodeNumber":998,"archetype":"deepDive","scriptModelLabel":"GPT-5.5","model":"gpt-5.5","voiceProviderLabel":"Cartesia TTS","hostNames":{"a":"Jessica","b":"Cathy"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["agents","developer-tools"],"concept_keys":["task-decomposition","tool-use-and-function-calling","agentic-loops","supervised-fine-tuning","reinforcement-learning-from-human-feedback","fine-tuning-on-execution-traces","in-context-learning"],"entities":["bi-agent","bi-bench","qwen"],"source_form":"research-paper","event_type":null,"episode_number":998},{"id":"e6d9a355-4326-4a46-998e-2ed9985aa5a5","title":"Jev is now available in LangSmith Evals","url":"https://www.langchain.com/blog/jev-is-now-available-in-langsmith-evals","type":"normal","description":"Use Jev as a judge for LangSmith evals to evaluate agent traces with faster, cheaper structured feedback across production runs, datasets, and regression tests.","cover_image_url":null,"discovered_at":"2026-09-22T15:39:26.319+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/jev-is-now-available-in-langsmith-evals-1790091771832.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/jev-is-now-available-in-langsmith-evals-1790091771832.txt","audio_summary":"Jev, TypeSafe AI's System One model, is now available as a judge inside LangSmith Evals. Vince sees a real workflow unlock; Ava respects the architecture but keeps poking at the single-benchmark evidence and the data-retention footnote teams will miss.","audio_duration_seconds":242,"audio_generated_at":"2026-09-22T15:43:00.417+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":997,"archetype":"skepticsTake","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["evals-benchmarks","developer-tools"],"concept_keys":["prompt-injection","classifier"],"entities":["langsmith","jev","typesafe"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":997},{"id":"f6b9da30-eb16-4d35-947d-c0c25e6574e4","title":"EvoOntology: A Self-Evolving Ontology Layer for Data Agents","url":"https://arxiv.org/html/2609.15779v1","type":"normal","description":"Data agents aim to fulfill natural-language instructions over heterogeneous data, including tables, files, and databases. However, data agents face a challenging agent-data gap: heterogeneous data resides outside the agent, while the agent can access it (e.g., column names and file paths) only through generic tools. Existing approaches either let agents directly explore raw data sources or inject manually constructed semantic layers into prompts. However, neither scales well to large","cover_image_url":null,"discovered_at":"2026-09-22T15:33:28.157+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/evoontology-a-self-evolving-ontology-layer-for-data-agents-1790091512328.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/evoontology-a-self-evolving-ontology-layer-for-data-agents-1790091512328.txt","audio_summary":"Justy and Cody dig into EvoOntology, a paper proposing an MCP-served ontology layer that data agents can build and then update from their own failures. Their take: the interesting move is not 'semantic layer for agents' by itself, but making that layer selective at runtime and maintainable through typed edits plus paired validation, so grounding can accumulate instead of being rediscovered on every task.","audio_duration_seconds":345,"audio_generated_at":"2026-09-22T15:38:45.992+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":996,"archetype":"deepDive","scriptModelLabel":"GPT-5.4","model":"gpt-5.4","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":true,"enrichmentProvider":"firecrawl"},"domains":["agents","data-infrastructure"],"concept_keys":["tool-use-and-function-calling","retrieval-augmented-generation","agentic-loops","state-management-in-language-models","model-routing","context-window"],"entities":["evoontology"],"source_form":"research-paper","event_type":null,"episode_number":996},{"id":"0d7271b7-5332-4722-b662-b722db7c7553","title":"Grounded Skill Synthesis from Code at Scale for Agentic Intelligence","url":"https://arxiv.org/html/2609.05571v1","type":"normal","description":"Reusable skills give agents transferable procedural knowledge, making scalable acquisition essential for extending agents beyond prior experience. Existing methods face two limitations: trajectory-based synthesis requires interactions with specific environments, while document-derived skills may lack executable evidence and verification. Source code offers a complementary path: it requires no prior agent experience yet provides executable evidence for grounding abstractions. We present","cover_image_url":null,"discovered_at":"2026-09-22T15:32:31.834+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/1introduction-1790091580363.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/1introduction-1790091580363.txt","audio_summary":"Masonry and Eyre dig into Code2Skill, a paper arguing that source code is the missing substrate for scalable skill banks. The real move is not just extracting tips from repos, but verifying that a synthesized skill can reconstruct the implementation without seeing the source body, then checking it against the original. They land on it as a strong infrastructure-as-product play for agentic systems: less mystical self-improvement, more mining maintained codebases for grounded procedures with provenance and boundaries.","audio_duration_seconds":398,"audio_generated_at":"2026-09-22T15:39:52.169+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":995,"archetype":"deepDive","scriptModelLabel":"GPT-5.4","model":"gpt-5.4","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":["agents","developer-tools"],"concept_keys":["retrieval-augmented-generation","tool-use-and-function-calling","agentic-loops","in-context-learning","state-management-in-language-models"],"entities":["github","codeskillbank","hugging-face"],"source_form":"research-paper","event_type":null,"episode_number":995},{"id":"d9c795d1-241d-45c6-ae73-19845dd40c2f","title":"AI Agents Are Rewriting the Rules of Lateral Movement","url":"https://thehackernews.com/2026/09/ai-agents-are-rewriting-rules-of.html","type":"normal","description":"AI agents can chain credentials and tools to reach beyond direct permissions, as a Hugging Face evaluation showed.","cover_image_url":null,"discovered_at":"2026-09-22T15:31:27.335+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/ai-agents-are-rewriting-the-rules-of-lateral-movement-1790091338090.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/ai-agents-are-rewriting-the-rules-of-lateral-movement-1790091338090.txt","audio_summary":"A contributed security article makes a strong case that an agent's real blast radius is the chain of identities, credentials, tools, and trust relationships it can assemble, not merely its direct permissions. Edmund sees a practical governance model in purpose-bound agent identity; Geffen agrees with the graph-based risk framing but pushes back on identity being the only meaningful control plane.","audio_duration_seconds":297,"audio_generated_at":"2026-09-22T15:35:47.701+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":994,"archetype":"quickRead","scriptModelLabel":"GPT-5.6 Terra","model":"gpt-5.6-terra","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":true,"enrichmentProvider":"tavily"},"domains":["agents","ai-safety"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","task-decomposition","retry-loops-and-error-recovery","graph-based-memory-representation"],"entities":["hugging-face","token-security","owasp"],"source_form":"blog-post","event_type":null,"episode_number":994},{"id":"b8eab322-c6ef-4060-a8e5-c4aa82ad287c","title":"GraphRAG: A Practitioner's Guide to 6 Advanced Architectural Patterns","url":"https://towardsdatascience.com/graphrag-a-practitioners-guide-to-6-advanced-architectural-patterns/","type":"normal","description":"Beyond basic graph retrieval: six production-oriented architectures for combining semantic search, knowledge graphs, and LLM reasoning.","cover_image_url":null,"discovered_at":"2026-09-22T15:28:30.572+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/graphrag-a-practitioner-s-guide-to-6-advanced-architectural-patt-1790091066769.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/graphrag-a-practitioner-s-guide-to-6-advanced-architectural-patt-1790091066769.txt","audio_summary":"Pippa and Tyler dig into a GraphRAG guide that’s really arguing for a more honest question: not “does GraphRAG work,” but “which retrieval shape matches the query shape?” They focus on the production trade-offs between strict graph querying, hybrid vector-plus-graph setups, and when knowledge graphs actually earn their keep.","audio_duration_seconds":346,"audio_generated_at":"2026-09-22T15:31:19.483+00:00","audio_provider":"openai:gpt-4o-mini-tts","audio_script_meta":{"episodeNumber":993,"archetype":"deepDive","scriptModelLabel":"GPT-5.4 mini","model":"gpt-5.4-mini","voiceProviderLabel":"OpenAI TTS","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["data-infrastructure","developer-tools"],"concept_keys":["retrieval-augmented-generation","embeddings","graph-based-memory-representation","entity-resolution","prompt-engineering","in-context-learning"],"entities":["graphrag"],"source_form":"blog-post","event_type":null,"episode_number":993},{"id":"6558f64f-80da-4665-af6d-d49cbc220381","title":"'Better than DeepSeek': Xiaomi's MiMo-V2.6-Pro debuts as the top open weights model in the world alongside cheaper V2.6-Flash","url":"https://venturebeat.com/technology/better-than-deepseek-xiaomis-mimo-v2-6-pro-debuts-as-the-top-open-weights-model-in-the-world-alongside-cheaper-v2-6-flash","type":"normal","description":"Xiaomi demonstrates MiMo taking text, images or video and coordinating multiple agents to create playable 3D worlds, construct scenes, implement interaction logic, inspect rendered output and iteratively refine the result.","cover_image_url":null,"discovered_at":"2026-09-22T15:27:23.566+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/better-than-deepseek-xiaomi-s-mimo-v2-6-pro-debuts-as-the-top-op-1790091240061.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/better-than-deepseek-xiaomi-s-mimo-v2-6-pro-debuts-as-the-top-op-1790091240061.txt","audio_summary":"Echo is skeptical that Xiaomi’s MiMo-V2.6-Pro really changes the open-weights frontier just because it tops one benchmark index, while Onyx argues the real story is the user-facing combo of open weights, low API pricing, million-token multimodal context, and a cheaper Flash tier that makes production adoption easier. They dig into whether the reinforcement-learning stack is genuinely novel or mostly expensive harness tuning, and land on Xiaomi as a serious systems company making open models more usable, not a magic leap in intelligence.","audio_duration_seconds":342,"audio_generated_at":"2026-09-22T15:34:11.38+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":992,"archetype":"skepticsTake","scriptModelLabel":"GPT-5.4 mini","model":"gpt-5.4-mini","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["new-models","agents"],"concept_keys":["reinforcement-learning-from-human-feedback","reward-hacking","fine-tuning","context-window","model-routing","agentic-loops","token-economics"],"entities":["xiaomi","mimo-v2-6-pro","mimo-v2-6-flash"],"source_form":"news","event_type":"launch-announcement","episode_number":992},{"id":"27c4361f-6114-4404-853b-992d936b585b","title":"SpaceXAI Releases Grok 4.7 for Coding and Knowledge Work","url":"https://www.unite.ai/spacexai-releases-grok-4-7-for-coding-and-knowledge-work/","type":"normal","description":"SpaceXAI on September 21, 2026 released Grok 4.7, its newest model for coding and knowledge work, priced from $2 per million input tokens and $6 per million output tokens. In the release announcement, SpaceXAI describes...","cover_image_url":null,"discovered_at":"2026-09-21T16:44:33.794+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/spacexai-releases-grok-4-7-for-coding-and-knowledge-work-1790009175525.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/spacexai-releases-grok-4-7-for-coding-and-knowledge-work-1790009175525.txt","audio_summary":"Pippa and Tyler examine Grok 4.7 as a shipped coding and knowledge-work model whose real pitch is long-running agent execution, harness integration, and low pricing rather than benchmark dominance.","audio_duration_seconds":312,"audio_generated_at":"2026-09-21T16:46:26.864+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":991,"archetype":"skepticsTake","scriptModelLabel":"GPT-5.6 Luna","model":"gpt-5.6-luna","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["new-models","developer-tools"],"concept_keys":["reinforcement-learning-from-human-feedback","agentic-loops","tool-use-and-function-calling","context-window","inference-optimization"],"entities":["grok-4-7","spacexai","cursor"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":991},{"id":"7dd2fd8f-72c6-4d5e-87dc-583d32926446","title":"Can Jev Be a Better Agent Evaluator?","url":"https://www.langchain.com/blog/jev-agent-evals-langsmith","type":"normal","description":"We tested using Jev-as-a-Judge against LLM judges on accuracy, repeatability, latency, and cost to see whether System One models could offer a new approach to agent evaluation.","cover_image_url":null,"discovered_at":"2026-09-21T16:26:20.463+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/can-jev-be-a-better-agent-evaluator-1790008070703.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/can-jev-be-a-better-agent-evaluator-1790008070703.txt","audio_summary":"LangChain's evaluation of TypeSafe AI's Jev as an agent evaluator reveals a fundamentally different architecture — not an LLM that generates text, but a 'System One' model that returns typed decisions with calibrated probabilities. In a narrow test against GPT-5.6 Luna, Terra, and Claude Sonnet 4.6, Jev matched human oracle accuracy on binary pass/fail decisions (100% vs. 80–99.8%), achieved 92–913x lower variance on continuous scoring, and cost $0.00035 per call versus $28.17 for Claude. The implication: agent evals may have a third viable path beyond code-based (narrow, deterministic) and LLM-as-judge (slow, expensive, non-deterministic). The catch: this is one narrow test on five weather requests, and low cost can amplify mistakes at scale.","audio_duration_seconds":315,"audio_generated_at":"2026-09-21T16:28:00.494+00:00","audio_provider":"openai:gpt-4o-mini-tts","audio_script_meta":{"episodeNumber":990,"archetype":"quickRead","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"OpenAI TTS","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["agents","evals-benchmarks"],"concept_keys":["structured-output","sampling-and-temperature","agentic-loops","tool-use-and-function-calling","calibration"],"entities":["jev","typesafe-ai","langchain"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":990},{"id":"2a8619f9-1d7f-4ad4-acfd-e8aed8ceb6d1","title":"Cloudflare Introduces the Agent Development Lifecycle to Replace Traditional SDLC","url":"https://www.infoq.com/news/2026/09/cloudflare-adlc-agents/","type":"normal","description":"Cloudflare has introduced the Agent Development Lifecycle to enhance AI-driven engineering. The approach replaces the traditional SDLC, addressing bottlenecks in testing, deployment, and maintenance. Key components include automated software factories, dynamic orchestration, advanced observability, and a security model for autonomous agents, aiming for more efficient software management.","cover_image_url":null,"discovered_at":"2026-09-21T16:21:08.684+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/cloudflare-introduces-the-agent-development-lifecycle-to-replace-1790007849876.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/cloudflare-introduces-the-agent-development-lifecycle-to-replace-1790007849876.txt","audio_summary":"Cloudflare wants to retire the classic SDLC for agent-driven development, pitching an end-to-end agent lifecycle with built-in observability, dynamic security, and automated review. Ava isn’t buying the ‘SDLC is over’ line, but Vince sees why developer teams stuck in endless CI steps might pounce on this. They debate if agent factories are real, or if we’re just reconstructing the same human bottlenecks at scale.","audio_duration_seconds":257,"audio_generated_at":"2026-09-21T16:24:20.154+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":989,"archetype":"skepticsTake","scriptModelLabel":"GPT-4.1","model":"gpt-4.1","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"firecrawl"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","durable-execution","model-routing","retry-loops-and-error-recovery","task-decomposition"],"entities":["cloudflare","opentelemetry"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":989},{"id":"aa51812a-ec24-4efc-b051-d6055783e67c","title":"Introducing System One Models & Jev - TypeSafe AI Blog","url":"https://typesafe.ai/blog/introducing-system-one-models-and-jev","type":"normal","description":"TypeSafe AI is an AI lab building machine-native intelligence infrastructure for automation, designed to make decisions within software. Try our first System One Model, Jev, in early access.","cover_image_url":null,"discovered_at":"2026-09-19T01:58:52.026+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-system-one-models-jev-typesafe-ai-blog-1789783323360.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-system-one-models-jev-typesafe-ai-blog-1789783323360.txt","audio_summary":"TypeSafe AI launches Jev, the first 'System One' model: no text generation, but blazingly fast, type-safe, parallel, probabilistic decisions for automation at the code layer. Jev claims 40–200x speedups versus LLMs like GPT-5.6 Terra, with zero type errors and per-decision calibration, aiming to finally unlock real-world automation where string output can't go. This episode dives into how Jev works, what it means to give up freeform text, and whether TypeSafe’s new architecture marks the start of a real post-LLM automation wave.","audio_duration_seconds":279,"audio_generated_at":"2026-09-19T02:02:13.263+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":988,"archetype":"quickRead","scriptModelLabel":"GPT-4.1","model":"gpt-4.1","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":["new-models","developer-tools"],"concept_keys":["structured-output","calibration","inference-optimization","token-economics"],"entities":["typesafe","jev"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":988},{"id":"281d38d9-3378-49d9-9002-c6d13315aea3","title":"What Is Jev? A Guide to TypeSafe AI’s System One Model","url":"https://www.langchain.com/blog/building-a-harness-with-jev","type":"normal","description":"What is Jev? Learn how TypeSafe AI’s System One model makes fast, structured decisions, where it fits in the agent loop, and how to use Jev with LangChain","cover_image_url":null,"discovered_at":"2026-09-19T01:58:36.855+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/what-is-jev-a-guide-to-typesafe-ai-s-system-one-model-1789783172073.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/what-is-jev-a-guide-to-typesafe-ai-s-system-one-model-1789783172073.txt","audio_summary":"Masonry and Eyre dig into Jev as a different kind of model bet: not a chatty LLM replacement, but a fast structured-decision layer for agent loops. They focus on the real product angle, the mechanics behind parallel typed questions, and where the claims are strong versus still a little hand-wavy.","audio_duration_seconds":291,"audio_generated_at":"2026-09-19T01:59:42.872+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":987,"archetype":"deepDive","scriptModelLabel":"GPT-5.4 mini","model":"gpt-5.4-mini","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":true,"enrichmentProvider":"tavily"},"domains":["agents","inference-optimization"],"concept_keys":["tool-use-and-function-calling","model-routing","agentic-loops","structured-output","calibration","classifier","state-management-in-language-models"],"entities":["jev","typesafe-ai","langchain"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":987},{"id":"b61a86e3-d0b3-4bc5-b98f-a4ef5cd383b7","title":"Learning Difficulty-Aware Length Controlfor Efficient Hybrid Reasoning Models","url":"https://arxiv.org/html/2609.19671v1","type":"normal","description":"Large Reasoning Models (LRMs) achieve strong performance on complex tasks but exhibit systematic inefficiency: they often overthink easy problems and underthink hard ones. Existing approaches based on uniform length penalties or rigid routing incur an efficiency tax, trading reduced computation on easy instances for accuracy loss on hard instances. We formulate efficient reasoning as an instance-adaptive computation allocation problem and propose When2Think, a post-training framework for hybrid","cover_image_url":null,"discovered_at":"2026-09-18T15:49:58.765+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/learning-difficulty-aware-length-controlfor-efficient-hybrid-rea-1789747180259.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/learning-difficulty-aware-length-controlfor-efficient-hybrid-rea-1789747180259.txt","audio_summary":"Edmund and Geffen dig into When2Think, a framework that teaches large reasoning models to spend fewer tokens on easy math problems while still thinking deeply on hard ones, using a clever difficulty-aware reward signal instead of a separate controller or reward model.","audio_duration_seconds":406,"audio_generated_at":"2026-09-18T15:59:52.852+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":986,"archetype":"deepDive","scriptModelLabel":"GPT-5.1","model":"gpt-5.1","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["inference-optimization","training-methods"],"concept_keys":["chain-of-thought","reinforcement-learning-from-human-feedback","token-economics","inference-optimization","reward-hacking","importance-sampling","difficulty-aware-compute-allocation"],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":986},{"id":"b7107ee9-157a-4d7d-9d01-49d76651015e","title":"Repeated VM Escapes By GPT-5.6-Cyber Based Agents Prove VMs and OS","url":"https://www.infoq.com/news/2026/09/agent-escape-vm/","type":"normal","description":"Traditional virtual machines are inadequate for isolating cyber-capable autonomous agents. Tests using GPT-5.6-Cyber indicated multiple escape attempts due to kernel flaws. While Firecracker provided some containment, vulnerabilities remained. The study underscores the need for minimal attack surface virtualisation technologies and rapid, proactive patching strategies to safeguard host systems.","cover_image_url":null,"discovered_at":"2026-09-18T15:49:41.946+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/repeated-vm-escapes-by-gpt-5-6-cyber-based-agents-prove-vms-and--1789747298253.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/repeated-vm-escapes-by-gpt-5-6-cyber-based-agents-prove-vms-and--1789747298253.txt","audio_summary":"Pippa and Tyler dig into an InfoQ piece on GPT‑5.6‑Cyber repeatedly breaking out of QEMU/KVM but not Firecracker, and what that really means for VM isolation, patching, and anyone running autonomous cyber-capable agents.","audio_duration_seconds":435,"audio_generated_at":"2026-09-18T16:01:54.513+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":985,"archetype":"deepDive","scriptModelLabel":"GPT-5.1","model":"gpt-5.1","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["agents","ai-safety"],"concept_keys":["agentic-loops","tool-use-and-function-calling","retry-loops-and-error-recovery","state-management-in-language-models","durable-execution"],"entities":["gpt-5-6-cyber","qemu","firecracker"],"source_form":"research-paper","event_type":"benchmark","episode_number":985},{"id":"a6115b01-cfae-42b3-bad6-521ea6b83910","title":"PrismML — Introducing Bonsai 2 27B: Near-Lossless Compression in a 9x Smaller Footprint","url":"https://prismml.com/news/bonsai-2-27b","type":"normal","description":"Ternary Bonsai 2 27B retains 98.2% of Qwen3.8 27B benchmark performance in a 5.9GB footprint, with multimodal and agentic capabilities.","cover_image_url":null,"discovered_at":"2026-09-18T15:49:13.73+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/prismml-introducing-bonsai-2-27b-near-lossless-compression-in-a--1789746864016.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/prismml-introducing-bonsai-2-27b-near-lossless-compression-in-a--1789746864016.txt","audio_summary":"Onyx and Echo discuss PrismML's Ternary Bonsai 2 27B launch, focusing on near-lossless low-bit compression, local deployment, throughput, energy efficiency, and where the benchmark claims still need real-world validation.","audio_duration_seconds":324,"audio_generated_at":"2026-09-18T15:54:35.194+00:00","audio_provider":"openai:gpt-4o-mini-tts","audio_script_meta":{"episodeNumber":984,"archetype":"quickRead","scriptModelLabel":"GPT-5.5","model":"gpt-5.5","voiceProviderLabel":"OpenAI TTS","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["inference-optimization","multimodal"],"concept_keys":["quantization","inference-optimization","token-economics","error-accumulation-in-generation","tool-use-and-function-calling","agentic-loops"],"entities":["prismml","bonsai-2-27b","qwen-3-8-27b"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":984},{"id":"c284ebb3-a5f2-4e44-9ef6-0337293519b7","title":"Agentic Work Management is here!","url":"https://youtu.be/GH7mw6MQTPY","type":"normal","description":"Agentic Work Management is here – your easy button for AI productivity across every team. 👏 Now included for every paid Asana customer: 30+ new prebuilt AI Teammates, ready to work inside real workflows, and Asana Dash, your personal AI chief of staff and Asana expert that knows your goals and priorities and surfaces what needs your attention. Together, they show what Agentic Work Management makes possible: humans and agents working from the same plan, with the same context and goals, to","cover_image_url":null,"discovered_at":"2026-09-18T15:47:55.167+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/agentic-work-management-is-here-1789746684941.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/agentic-work-management-is-here-1789746684941.txt","audio_summary":"Asana's new Agentic Work Management layer ships 30-plus prebuilt AI Teammates and a personal AI chief of staff called Asana Dash — all included for paid customers — so humans and agents can coordinate from one shared plan.","audio_duration_seconds":56,"audio_generated_at":"2026-09-18T15:51:30.285+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":983,"archetype":"coldBrief","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","state-management-in-language-models","in-context-learning"],"entities":["asana","asana-dash"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":983},{"id":"3eb4c3a5-1124-4f08-ba56-4ddd4ea8c18c","title":"Your Agent Is Only As Good As Your Infrastructure","url":"https://archive.ph/2026.09.18-154501/https://coreweave.com/blog/your-agent-is-only-as-good-as-your-infrastructure","type":"normal","description":"Argues that AI agent quality in production is largely determined by infrastructure—chain-wide latency, bursty GPU demand, and KV caching—not the underlying model itself.","cover_image_url":null,"discovered_at":"2026-09-18T15:47:16.464+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/archive-ph-2026-09-18-154501-https-coreweave-com-blog-your-agent-1789747306572.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/archive-ph-2026-09-18-154501-https-coreweave-com-blog-your-agent-1789747306572.txt","audio_summary":"Justy and Cody dig into CoreWeave's argument that agent quality in production is mostly an infrastructure story once workflows get long, bursty, and tool-heavy. They mostly buy the core claim, but Cody pushes on how much of this is genuine systems insight versus a cloud vendor setting up next week's pitch on KV caching and routing.","audio_duration_seconds":339,"audio_generated_at":"2026-09-18T16:02:00.287+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":982,"archetype":"quickRead","scriptModelLabel":"GPT-5.4","model":"gpt-5.4","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":["agents","inference-optimization"],"concept_keys":["agentic-loops","tool-use-and-function-calling","durable-execution","state-management-in-language-models","kv-cache","context-window","token-counting-and-optimization"],"entities":["coreweave"],"source_form":"blog-post","event_type":null,"episode_number":982},{"id":"13a09755-fcb4-4fed-8f36-affacc67410e","title":"WSO2 Releases Agent Manager as Enterprises Look to Control Growing AI Agent Sprawl","url":"https://www.infoq.com/news/2026/09/ws02-agent-manager/","type":"normal","description":"WSO2 has announced the general availability of WSO2 Agent Manager, an open-source platform designed to provide centralized governance, identity management, security controls, and operational oversight for AI agents running across different models, frameworks, and deployment environments.","cover_image_url":null,"discovered_at":"2026-09-18T15:42:11.557+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/wso2-releases-agent-manager-as-enterprises-look-to-control-growi-1789746484976.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/wso2-releases-agent-manager-as-enterprises-look-to-control-growi-1789746484976.txt","audio_summary":"Masonry and Eyre dig into WSO2 Agent Manager going general availability, and why the interesting part is not another agent builder but a framework-independent control plane for identity, policy, sandboxing, and lifecycle management across messy enterprise agent fleets.","audio_duration_seconds":318,"audio_generated_at":"2026-09-18T15:48:16.395+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":981,"archetype":"quickRead","scriptModelLabel":"GPT-5.4","model":"gpt-5.4","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":["agents","developer-tools"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models"],"entities":["wso2","wso2-agent-manager","langchain"],"source_form":"tool","event_type":"launch-announcement","episode_number":981},{"id":"a86960e2-3552-48ce-b549-282827b42d32","title":"Guest Post: Why AI Agent Identities Need Post-Quantum Cryptography","url":"https://thequantuminsider.com/2026/09/17/post-quantum-security-ai-agent-economy/","type":"normal","description":"The post examines how quantum computing could threaten AI agent identities and why crypto-agile authentication & verification may be needed.","cover_image_url":null,"discovered_at":"2026-09-17T22:09:26.402+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/guest-post-why-ai-agent-identities-need-post-quantum-cryptograph-1789683143348.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/guest-post-why-ai-agent-identities-need-post-quantum-cryptograph-1789683143348.txt","audio_summary":"AI agents need quantum-resistant identity signatures, not only quantum-resistant encryption.","audio_duration_seconds":68,"audio_generated_at":"2026-09-17T22:12:26.918+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":980,"archetype":"coldBrief","scriptModelLabel":"GPT-5.6 Terra","model":"gpt-5.6-terra","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":true,"enrichmentProvider":"tavily"},"domains":["agents","ai-safety"],"concept_keys":[],"entities":["nist"],"source_form":"blog-post","event_type":null,"episode_number":980},{"id":"55295fd6-cf4d-4ea1-ba8b-bf22cc9ce9c4","title":"Anthropic launches Claude Code Projects, an ‘always-on’ conversation that remembers and delegates your long-running dev work","url":"https://venturebeat.com/orchestration/anthropic-launches-claude-code-projects-an-always-on-conversation-that-remembers-and-delegates-your-long-running-dev-work","type":"normal","description":"For enterprises, it makes a whole lot of sense: their digital storefront, website, content management system, procurement platform, or other business application rarely has a discrete endpoint.","cover_image_url":null,"discovered_at":"2026-09-17T22:07:54.233+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/anthropic-launches-claude-code-projects-an-always-on-conversatio-1789683006577.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/anthropic-launches-claude-code-projects-an-always-on-conversatio-1789683006577.txt","audio_summary":"Pippa and Tyler dig into Anthropic’s Claude Code Projects, the new always-on coordination layer for long-running coding work. They focus on the shift from single-session coding help to persistent project memory, threaded delegation, and the practical question of whether this is the kind of orchestration layer developers will actually adopt.","audio_duration_seconds":300,"audio_generated_at":"2026-09-17T22:10:19.232+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":979,"archetype":"deepDive","scriptModelLabel":"GPT-5.4 mini","model":"gpt-5.4-mini","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["agents","developer-tools"],"concept_keys":["state-management-in-language-models","agentic-loops","task-decomposition","tool-use-and-function-calling","context-window-management","durable-execution"],"entities":["anthropic","claude-code-projects"],"source_form":"news","event_type":"launch-announcement","episode_number":979},{"id":"6156d6d2-c75f-4be8-b5d8-f0ad826a71ed","title":"LLMs respond differently to harmful prompts when AI watermarking is used","url":"https://arstechnica.com/security/2026/09/ai-text-watermarking-can-make-models-more-vulnerable-to-adversarial-prompts/","type":"normal","description":"SynthID can cause models to follow harmful instructions they would otherwise refuse.","cover_image_url":null,"discovered_at":"2026-09-17T22:06:00.646+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/llms-respond-differently-to-harmful-prompts-when-ai-watermarking-1789683079226.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/llms-respond-differently-to-harmful-prompts-when-ai-watermarking-1789683079226.txt","audio_summary":"Onyx and Echo dig into a weird side effect of text watermarking: once you change the token sampler, you can also change refusal behavior, tool calls, and how easily harmful prompts slip through. The take is that provenance is not a free add-on; it’s another moving part that has to be red-teamed like any other safety mechanism.","audio_duration_seconds":225,"audio_generated_at":"2026-09-17T22:11:27.157+00:00","audio_provider":"openai:gpt-4o-mini-tts","audio_script_meta":{"episodeNumber":978,"archetype":"quickRead","scriptModelLabel":"GPT-5.4 mini","model":"gpt-5.4-mini","voiceProviderLabel":"OpenAI TTS","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["ai-safety","developer-tools"],"concept_keys":["sampling-and-temperature","prompt-injection","tool-use-and-function-calling","decoding-strategy"],"entities":["synthid-text","hugging-face"],"source_form":"news","event_type":null,"episode_number":978},{"id":"04eed4d2-3ee6-422d-985c-e89373b7e992","title":"Enabling Creative Explorationfor Vibe Design Agents","url":"https://arxiv.org/html/2609.15078v1","type":"normal","description":"Vibe design agents turn natural-language briefs into rendered interfaces and frontend code. Yet a useful design agent should do more than produce one valid page: it should help users explore coherent alternatives. Increasing token-level temperature is a blunt solution because it varies aesthetic decisions and syntax-sensitive code at the same time. We instead separate exploration from implementation through an inference architecture that makes design direction an explicit intermediate decision.","cover_image_url":null,"discovered_at":"2026-09-17T15:14:10.135+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/enabling-creative-explorationfor-vibe-design-agents-1789658578903.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/enabling-creative-explorationfor-vibe-design-agents-1789658578903.txt","audio_summary":"Vince and Ava dig into a paper about separating design exploration from code generation in vibe design agents. They focus on why temperature is too blunt, how structured design specs plus external selection give you controlled variety, and what the offline and online results actually say about usefulness versus uncertainty.","audio_duration_seconds":314,"audio_generated_at":"2026-09-17T15:23:10.437+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":977,"archetype":"deepDive","scriptModelLabel":"GPT-5.4 mini","model":"gpt-5.4-mini","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["developer-tools","agents"],"concept_keys":["sampling-and-temperature","structured-output","calibration"],"entities":["lovable","stitch"],"source_form":"research-paper","event_type":null,"episode_number":977},{"id":"814e0eae-cce0-471f-b4a5-3d3c9c92f51f","title":"“Everyone's in a race to replace GitHub\": Zed launches Delta because agents made pull requests obsolete","url":"https://thenewstack.io/zed-delta-github-alternative/","type":"normal","description":"Zed disabled pull requests on its own codebase. Delta, now in public beta, bets that shared threads suit agents better than GitHub's review model.","cover_image_url":null,"discovered_at":"2026-09-17T15:11:43.685+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/everyone-s-in-a-race-to-replace-github-zed-launches-delta-becaus-1789658212679.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/everyone-s-in-a-race-to-replace-github-zed-launches-delta-becaus-1789658212679.txt","audio_summary":"Zed launches Delta, a multiplayer coding environment designed around agent-first workflows that replaces GitHub's pull-request model with shared conversation threads. The shift reflects a broader industry bet that traditional code review is incompatible with high-volume AI-generated changes, and that the real bottleneck has moved from authorship to collaborative verification and deployment.","audio_duration_seconds":357,"audio_generated_at":"2026-09-17T15:17:06.188+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":976,"archetype":"deepDive","scriptModelLabel":"Haiku 4","model":"claude-haiku-4-5","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":true,"enrichmentProvider":"jina"},"domains":["developer-tools","agents"],"concept_keys":[],"entities":["zed","delta","github"],"source_form":"blog-post","event_type":"launch-announcement","episode_number":976},{"id":"d5028081-bfae-4498-ac55-3261d7678129","title":"OpenAI Adds New Safety Guardrails and Public Incident Disclosures for Frontier Models","url":"https://www.nytimes.com/2026/09/16/technology/openai-model-safety-guardrails.html","type":"normal","description":"OpenAI announces stricter pre-launch safety reviews and public incident reports, including six new cases of models evading safeguards, testing whether formal oversight can truly constrain frontier AI.","cover_image_url":null,"discovered_at":"2026-09-17T15:10:10.613+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/nytimes-com-2026-09-16-technology-openai-model-safety-guardrails-1789665693089.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/nytimes-com-2026-09-16-technology-openai-model-safety-guardrails-1789665693089.txt","audio_summary":"OpenAI’s latest safety push spotlights the tension between new guardrails and real-world reliability: the big claim is that extra layers of oversight and disclosure will actually keep frontier models in check before they go off-script. The real story: OpenAI is betting that publishing incidents and tightening policies can scale with model complexity, but the details show how hard it is to define and enforce ‘safety’ in practice.","audio_duration_seconds":224,"audio_generated_at":"2026-09-17T17:21:41.877+00:00","audio_provider":"deepgram:aura-2","audio_script_meta":{"episodeNumber":975,"archetype":"quickRead","scriptModelLabel":"GPT-4.1","model":"gpt-4.1","voiceProviderLabel":"Deepgram Aura-2","hostNames":{"a":"Natalie","b":"Ansel"},"enrichmentUsed":true,"enrichmentProvider":"firecrawl"},"domains":["ai-safety"],"concept_keys":[],"entities":["openai"],"source_form":"blog-post","event_type":null,"episode_number":975},{"id":"9e58edc4-2be3-4ff9-af50-cf5d2ad3227f","title":"Reimagining research papers as interactive and reliable AI agents - Nature","url":"https://www.nature.com/articles/s41586-026-11044-y","type":"normal","description":"Paper2Agent converts research papers into interactive artificial intelligence agents by turning manuscripts, code and data into model context protocol-based tool-invoking systems that reproduce original results, answer new scientific queries and collaborate to generate novel insights.","cover_image_url":null,"discovered_at":"2026-09-17T15:09:16.304+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/reimagining-research-papers-as-interactive-and-reliable-ai-agent-1789658261203.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/reimagining-research-papers-as-interactive-and-reliable-ai-agent-1789658261203.txt","audio_summary":"Paper2Agent proposes transforming research papers into interactive AI agents, making scientific results executable, testable, and reusable. Talon and Wildflower debate the technical and practical impact—Wildflower unpacks the architecture and real limits, Talon pushes the user story and actual workflow wins. They question where this leap is real versus hype, and who actually needs it.","audio_duration_seconds":196,"audio_generated_at":"2026-09-17T15:17:48+00:00","audio_provider":"rime:mistv3","audio_script_meta":{"episodeNumber":974,"archetype":"deepDive","scriptModelLabel":"GPT-4.1","model":"gpt-4.1","voiceProviderLabel":"Rime Mist v3","hostNames":{"a":"Talon","b":"Wildflower"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":["agents","developer-tools"],"concept_keys":["tool-use-and-function-calling","agentic-loops","dynamic-code-execution"],"entities":["paper2agent","claude-code","model-context-protocol"],"source_form":"research-paper","event_type":null,"episode_number":974},{"id":"8cbf9872-336c-4cc1-8b0c-cc18bb4a6b5c","title":"Overview: Error Accumulation in Generation","url":"https://www.sandrise.io/exploring-next/overviews/error-accumulation-in-generation","type":"overview","description":"A model writes one token, then predicts the next from its own output—including mistakes. Error accumulation is the whiteboard you can never erase.","cover_image_url":null,"discovered_at":"2026-09-16T16:24:43.5+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-error-accumulation-in-generation-1789576140365.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-error-accumulation-in-generation-1789576140365.txt","audio_summary":"We finally slow down and explain error accumulation in generation from the ground up — why mistakes compound as a model writes more tokens, how exposure bias sets the trap, and what actually helps.","audio_duration_seconds":521,"audio_generated_at":"2026-09-16T16:29:16.518+00:00","audio_provider":"fishaudio:s2.1-pro-free","audio_script_meta":{"episodeNumber":973,"archetype":"deepDive","scriptModelLabel":"Sonnet 4.6","model":"claude-sonnet-4-6","voiceProviderLabel":"Fish Audio S2.1 Pro","hostNames":{"a":"Laura","b":"Harper"},"enrichmentUsed":true,"enrichmentProvider":"tavily"},"domains":["inference-optimization"],"concept_keys":["autoregressive-generation","error-accumulation-in-generation","task-decomposition","constraint-verification","reranking","agentic-loops"],"entities":[],"source_form":null,"event_type":null,"episode_number":973},{"id":"9b39d260-1939-401f-8a37-a75da16a9811","title":"Overview: Reranking","url":"https://www.sandrise.io/exploring-next/overviews/reranking","type":"overview","description":"A retriever finds a hundred plausible results in milliseconds. A reranker reorders them carefully. Why two passes instead of one smart one?","cover_image_url":null,"discovered_at":"2026-09-16T16:17:36.543+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-reranking-1789575668675.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/overview-reranking-1789575668675.txt","audio_summary":"We finally slow down on reranking, the search-and-RAG move we keep name-dropping whenever retrieval quality comes up. We build it from the basic intuition, then get into why the second scoring pass helps, where it breaks, and why it is still very much alive in current systems.","audio_duration_seconds":630,"audio_generated_at":"2026-09-16T16:21:28.007+00:00","audio_provider":"inworld-mini:inworld-tts-1.5-mini","audio_script_meta":{"episodeNumber":972,"archetype":"deepDive","scriptModelLabel":"GPT-5.5","model":"gpt-5.5","voiceProviderLabel":"Inworld TTS 1.5 Mini","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"brightdata"},"domains":["data-infrastructure","developer-tools"],"concept_keys":["retrieval-augmented-generation","embeddings","reranking","cross-attention","construct-validity"],"entities":["cohere","jina","mixedbread"],"source_form":null,"event_type":null,"episode_number":972},{"id":"2cb3f5f8-c323-4cc5-a264-d4782d3b5b0d","title":"Emergence World: Adversarial Stress-Testing of Long-Horizon Multi-Agent Systems","url":"https://arxiv.org/html/2609.17320v1","type":"normal","description":"As AI agents move from bounded tasks to persistent deployments, failures can propagate through memory, tools, other agents, and environmental state long after their interactions. This creates a safety regime that cannot be characterized by evaluating model responses in isolation. Emergence World, is a continuously running multi-agent environment for adversarial stress testing of long horizon autonomous systems. We ran eight parallel worlds of ten agents from identical starting conditions: seven","cover_image_url":null,"discovered_at":"2026-09-16T16:14:10.929+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/emergence-world-adversarial-stress-testing-of-long-horizon-multi-1789575818124.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/emergence-world-adversarial-stress-testing-of-long-horizon-multi-1789575818124.txt","audio_summary":"Pippa and Tyler unpack Emergence World’s new Study 2: a 16‑day adversarial stress test of long‑horizon multi‑agent systems. They dig into how the simulated “worlds” work, what the phishing, misinformation, and memory‑breach attacks revealed, why model‑level alignment isn’t compositional, and how teams could actually use the Emergence World repo to probe real agent workflows.","audio_duration_seconds":420,"audio_generated_at":"2026-09-16T16:23:51.485+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":971,"archetype":"deepDive","scriptModelLabel":"GPT-5.1","model":"gpt-5.1","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":true,"enrichmentProvider":"serpapi"},"domains":["agents","ai-safety"],"concept_keys":["agentic-loops","tool-use-and-function-calling","state-management-in-language-models","prompt-injection","error-accumulation-in-generation","in-context-learning","multi-model-routing"],"entities":["emergence-world"],"source_form":"research-paper","event_type":null,"episode_number":971},{"id":"eda62c27-d5a0-411a-9f6f-b114fc46b732","title":"The Router Within: ElicitingNative Skill Routing from a Frozen LLM","url":"https://arxiv.org/html/2609.15982v1","type":"normal","description":"Skills extend an LLM agent beyond its parametric knowledge, and the gain they promise rests on picking the right one. Deployed harnesses route by preloading every skill's metadata into the context, which disperses the agent's attention and caps the library size. Retrieval pipelines move the selection out of the context, but also out of the agent's capability. We show that the frozen agent LLM already carries the routing signal in its own forward passes, and that two linear maps suffice to read","cover_image_url":null,"discovered_at":"2026-09-16T16:13:51.957+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/the-router-within-elicitingnative-skill-routing-from-a-frozen-ll-1789575399135.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/the-router-within-elicitingnative-skill-routing-from-a-frozen-ll-1789575399135.txt","audio_summary":"Onyx and Echo unpack Gavel, a paper arguing that a frozen agent model already contains useful skill-routing signals in its hidden states, if you read them out with two small trained projections and a second-stage verdict.","audio_duration_seconds":461,"audio_generated_at":"2026-09-16T16:16:54.794+00:00","audio_provider":"openai:gpt-4o-mini-tts","audio_script_meta":{"episodeNumber":970,"archetype":"deepDive","scriptModelLabel":"GPT-5.5","model":"gpt-5.5","voiceProviderLabel":"OpenAI TTS","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":["agents","developer-tools"],"concept_keys":["model-routing","tool-use-and-function-calling","embeddings","reranking","in-context-learning","agentic-loops","mixture-of-experts","context-window"],"entities":["qwen","skillret"],"source_form":"research-paper","event_type":null,"episode_number":970},{"id":"99ad37dc-8197-43a5-9c70-a4dce956754d","title":"Model Behavior: Week of September 14, 2026","url":"https://www.sandrise.io/exploring-next/model-behavior/2026-09-14","type":"model-behavior","description":"OpenAI's Data Agent and Agents API reshape the runtime fight; Claude Fable 5.1's benchmarks keep raw capability relevant; ServiceNow and SSI's infrastructure plays show the competitive center has shifted from model leaderboards to enterprise deployment control.","cover_image_url":null,"discovered_at":"2026-09-16T00:01:07.172+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-september-14-2026-1789517152543.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-september-14-2026-1789517152543.txt","audio_summary":"We read this week as a control-layer week: the flashy model race kept moving, but the real competitive shift was toward owning where agents run, what data they can reach, and how enterprises actually deploy them. We still give Anthropic credit for raw capability, but OpenAI, ServiceNow, Nvidia, SSI, and the open-weight wave made the board feel less like a benchmark race and more like a runtime fight.","audio_duration_seconds":480,"audio_generated_at":"2026-09-16T00:06:14.524+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":969,"archetype":"deepDive","scriptModelLabel":"GPT-5.5","model":"gpt-5.5","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"}},"domains":["agents","developer-tools"],"concept_keys":["tool-use-and-function-calling","agentic-loops","state-management-in-language-models","durable-execution","retry-loops-and-error-recovery","context-window-management","inference-optimization"],"entities":["openai","anthropic","servicenow"],"source_form":null,"event_type":"launch-announcement","episode_number":969},{"id":"7f908a17-d7f4-4f93-a2f4-9a185255d8bd","title":"Long-running AI agents quietly drop compliance rules, and bigger context windows won't fix it","url":"https://venturebeat.com/orchestration/long-running-ai-agents-quietly-drop-compliance-rules-and-bigger-context-windows-wont-fix-it","type":"normal","description":"Why long-running agents silently drop compliance rules","cover_image_url":null,"discovered_at":"2026-09-14T15:34:14.953+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/long-running-ai-agents-quietly-drop-compliance-rules-and-bigger--1789400741508.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/long-running-ai-agents-quietly-drop-compliance-rules-and-bigger--1789400741508.txt","audio_summary":"Long-running AI agents quietly drop compliance rules, and bigger context windows won't fix it Ankit Anand 11:00 am, PT, September 13, 2026 CleoPtolemy made with Midjourney Imagine deploying an AI agent to run a multi-day master data validation workflow. By day three, it has ingested thousands of records.","audio_duration_seconds":34,"audio_generated_at":"2026-09-14T15:45:44.668+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":968,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"news","event_type":null,"episode_number":968},{"id":"06dc445c-dfd7-471f-96d9-a6447f9128da","title":"Coding Agents Don't Need Longer History — They Need Intent Continuity | Towards Data Science","url":"https://towardsdatascience.com/coding-agents-dont-need-longer-history-they-need-intent-continuity/","type":"normal","description":"I built a system that automatically discovers, verifies, and applies relevant requirements from earlier interactions without asking the user where they came from.","cover_image_url":null,"discovered_at":"2026-09-14T15:31:35.164+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/coding-agents-don-t-need-longer-history-they-need-intent-continu-1789400487392.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/coding-agents-don-t-need-longer-history-they-need-intent-continu-1789400487392.txt","audio_summary":"Agentic AI Coding Agents Don't Need Longer History — They Need Intent Continuity I built a system that automatically discovers, verifies, and applies relevant requirements from earlier interactions without asking the user where they came from. Emmimal P Alexander September 11, 2026 17 min read Image By Author.","audio_duration_seconds":44,"audio_generated_at":"2026-09-14T15:41:32.975+00:00","audio_provider":"deepgram:aura-2","audio_script_meta":{"episodeNumber":967,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"Deepgram Aura-2","hostNames":{"a":"Asteria","b":"Draco"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":967},{"id":"91c1a621-bf24-4c31-979c-aab481d73dcb","title":"Procedural Graphs: Self-Evolving Execution Structures for LLM Agents","url":"https://arxiv.org/html/2609.09153v1","type":"normal","description":"Large language models are increasingly deployed as agents that plan over long horizons and act through external tools. Most agents select actions through unconstrained generation over an accumulating history, leaving implicit the procedural knowledge of what to do, in what order, and under which conditions. As trajectories lengthen, agents can lose track of their objectives, invoke tools out of order, and repeat unproductive actions. We introduce the Procedural Graph: just as a knowledge graph","cover_image_url":null,"discovered_at":"2026-09-10T14:38:31.563+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/procedural-graphs-self-evolving-execution-structures-for-llm-age-1789051455661.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/procedural-graphs-self-evolving-execution-structures-for-llm-age-1789051455661.txt","audio_summary":"Masonry and Eyre discuss the 'Procedural Graphs' paper, which proposes a self-evolving graph structure to manage procedural knowledge for LLM agents, moving away from flat history logs toward a structured 'what-to-do' map that the agent can refine through trial and error.","audio_duration_seconds":286,"audio_generated_at":"2026-09-10T14:44:28.31+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":966,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":966},{"id":"3a1d837c-0547-4e53-aba5-4e54f17038d5","title":"AIM — India","url":"https://analyticsindiamag.com/ai-news/deepseek-launches-v41-flash-to-make-large-scale-ai-inference-cheaper","type":"normal","description":"India","cover_image_url":null,"discovered_at":"2026-09-10T14:32:10.963+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/aim-india-1789051413337.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/aim-india-1789051413337.txt","audio_summary":"The company has made V4.1-Flash available through its API with native multimodal support. DeepSeek has retired V4-Flash and V4-Flash-Vision-Exp, while the older model names deepseek-v4-flash and deepseek-v4-flash-vision-exp will temporarily route requests to V4.1-Flash.","audio_duration_seconds":52,"audio_generated_at":"2026-09-10T14:43:36.839+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":965,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":965},{"id":"f2210e8c-6989-431b-91d0-12a7d2298b1f","title":"Chain of Thought vs. Tree of Thoughts: Which is Best for AI Agents? - MachineLearningMastery.com","url":"https://machinelearningmastery.com/chain-of-thought-vs-tree-of-thoughts-which-is-best-for-ai-agents/","type":"normal","description":"Compare Chain of Thought and Tree of Thoughts reasoning to understand which approach best fits your AI agent.","cover_image_url":null,"discovered_at":"2026-09-10T14:30:48.929+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/chain-of-thought-vs-tree-of-thoughts-which-is-best-for-ai-agents-1789051165768.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/chain-of-thought-vs-tree-of-thoughts-which-is-best-for-ai-agents-1789051165768.txt","audio_summary":"Tree of Thoughts: Which is Best for AI Agents? By Vinod Chugani on September 9, 2026 in Artificial Intelligence 0 Share Post Share In this article, you will learn the key differences between Chain of Thought and Tree of Thoughts prompting, and how each reasoning framework is applied in AI agent systems.","audio_duration_seconds":44,"audio_generated_at":"2026-09-10T14:39:31.263+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":964,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":964},{"id":"3656ca92-e710-4888-ab64-297bb57c6669","title":"Introducing Muse: The World’s First Personal AI Agent Built for Everyone","url":"https://about.fb.com/news/2026/09/introducing-muse-personal-ai-agent/","type":"normal","description":"Muse is a secure, private personal AI agent that proactively helps people meet their goals and suggests ideas.","cover_image_url":null,"discovered_at":"2026-09-09T15:37:06.3+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-muse-the-world-s-first-personal-ai-agent-built-for-e-1788968401506.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-muse-the-world-s-first-personal-ai-agent-built-for-e-1788968401506.txt","audio_summary":"Meta launches Muse as a personal AI agent built for mass adoption, running on a dedicated Secure VM with a Sentinel gatekeeper, powered by Muse Spark, and usable via app or WhatsApp with proactive planning, background work, and Stripe Link checkout protections.","audio_duration_seconds":226,"audio_generated_at":"2026-09-09T15:40:13.894+00:00","audio_provider":"fishaudio:s2.1-pro-free","audio_script_meta":{"episodeNumber":963,"archetype":"quickRead","scriptModelLabel":"Muse Glimmer 30B","model":"meta/muse-glimmer-30b","voiceProviderLabel":"Fish Audio S2.1 Pro","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"you"},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":963},{"id":"5c8cf40a-d152-4f3a-b57b-82027ad57103","title":"vLLM x AgentX: Optimizing for Real-World Agentic Serving","url":"https://vllm.ai/blog/2026-09-08-vllm-agentx","type":"normal","description":"How vLLM optimizes KV cache management, parallelism, scheduling, and P/D disaggregation for agentic workloads, validated on SemiAnalysis AgentX with up to 130K","cover_image_url":null,"discovered_at":"2026-09-09T15:31:19.719+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/vllm-x-agentx-optimizing-for-real-world-agentic-serving-1788968079981.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/vllm-x-agentx-optimizing-for-real-world-agentic-serving-1788968079981.txt","audio_summary":"Vince and Ava unpack vLLM’s AgentX post on agentic serving, debating whether its full-stack KV-cache and P/D disaggregation story is a genuine product win or benchmark engineering dressed as insight.","audio_duration_seconds":257,"audio_generated_at":"2026-09-09T15:34:52.02+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":962,"archetype":"skepticsTake","scriptModelLabel":"Muse Glimmer 30B","model":"meta/muse-glimmer-30b","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":962},{"id":"c08d0916-27c0-49f0-84b7-7837f1d11116","title":"Organizing Context in a Multi-Agent Harness","url":"https://www.langchain.com/blog/organizing-context-in-a-multi-agent-harness","type":"normal","description":"Learn how context modes in deepagents help subagents fork a supervisor's context or start isolated — for faster, cheaper, more focused multi-agent work.","cover_image_url":null,"discovered_at":"2026-09-09T15:30:03.892+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/organizing-context-in-a-multi-agent-harness-1788968009709.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/organizing-context-in-a-multi-agent-harness-1788968009709.txt","audio_summary":"Justy and Cody unpack LangChain’s deepagents post on forked vs isolated subagents, debating whether inheriting supervisor context is a real efficiency win or a bias risk, and who actually benefits from context modes in production harnesses.","audio_duration_seconds":273,"audio_generated_at":"2026-09-09T15:33:44.855+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":961,"archetype":"quickRead","scriptModelLabel":"Muse Glimmer 30B","model":"meta/muse-glimmer-30b","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":961},{"id":"d534a544-ed44-4148-975c-f2c0c2782aab","title":"Model Behavior: Week of September 7, 2026","url":"https://www.sandrise.io/exploring-next/model-behavior/2026-09-07","type":"model-behavior","description":null,"cover_image_url":null,"discovered_at":"2026-09-09T00:00:12.691+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-september-7-2026-1788912675387.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/model-behavior-week-of-september-7-2026-1788912675387.txt","audio_summary":"They dig into the AGI framing, the cybersecurity numbers, and whether the Codex context-window fix is the quietly interesting thing nobody's leading with. The move collapses the friction between local-first and cloud-capable workflows.","audio_duration_seconds":33,"audio_generated_at":"2026-09-09T00:11:19.046+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":960,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"}},"domains":[],"concept_keys":[],"entities":[],"source_form":null,"event_type":null,"episode_number":960},{"id":"e213d14a-89e4-4a16-8858-79c021cc1a26","title":"When Models Edit Too Much: On the Fidelity of Minimal Code Edits","url":"https://arxiv.org/html/2609.04061v1","type":"normal","description":"Large language models (LLMs) are increasingly used to edit existing code, but correctness alone is not enough: useful repairs should also be minimal, reviewable, and faithful to the original implementation. We study over-editing, the tendency of a model to rewrite code beyond what is required to fix a bug. We construct an evaluation framework from 400 BigCodeBench problems by injecting controlled AST-level corruptions into reference solutions, giving each repair task a known minimal patch.","cover_image_url":null,"discovered_at":"2026-09-08T18:54:35.334+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/when-models-edit-too-much-on-the-fidelity-of-minimal-code-edits-1788894405033.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/when-models-edit-too-much-on-the-fidelity-of-minimal-code-edits-1788894405033.txt","audio_summary":"We study over-editing, the tendency of a model to rewrite code beyond what is required to fix a bug. We construct an evaluation framework from 400 BigCodeBench problems by injecting controlled AST-level corruptions into reference solutions, giving each repair task a known minimal patch.","audio_duration_seconds":58,"audio_generated_at":"2026-09-08T19:06:49.864+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":959,"archetype":"coldBrief","scriptModelLabel":"Built-in brief","model":"fallback","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":959},{"id":"315226bd-bef2-4e01-af23-a9748e0c4d2f","title":"Enterprise RAG Permission Assembly","url":"https://archive.ph/2026.09.08-185035/https://thenewstack.io/enterprise-rag-permission-assembly/","type":"normal","description":"A new policy layer adds token-level gating that checks each retrieval request against user roles before data hits the vector store, preventing accidental leaks in enterprise RAG pipelines.","cover_image_url":null,"discovered_at":"2026-09-08T18:53:36.512+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/archive-ph-2026-09-08-185035-https-thenewstack-io-enterprise-rag-1788908629958.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/archive-ph-2026-09-08-185035-https-thenewstack-io-enterprise-rag-1788908629958.txt","audio_summary":"Enterprise RAG just got a guard dog—token‑level gates that keep the model from pulling in secrets.","audio_duration_seconds":32,"audio_generated_at":"2026-09-08T23:03:54.371+00:00","audio_provider":"hume","audio_script_meta":{"episodeNumber":958,"archetype":"coldBrief","scriptModelLabel":"GPT-OSS 20B","model":"openai/gpt-oss-20b","voiceProviderLabel":"Hume Octave 2","hostNames":{"a":"Vince","b":"Ava"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":958},{"id":"db99db98-4b22-41ee-8ddb-65c91d5e3cbc","title":"Enoki: Efficient Multi-Level Hallucination Detection","url":"https://arxiv.org/html/2609.00581v2","type":"normal","description":"Ensuring factuality remains a critical challenge for deploying LLMs in high-stakes settings. Existing hallucination detectors usually operate at a single level: claim-level methods provide interpretable factual units, while span-level methods localize unsupported text. Bridging these views is costly, as LLM-heavy pipelines require multiple decomposition and verification calls, and modular systems need additional claim-to-span alignment. We propose Enoki, an Open Information Extraction framework","cover_image_url":null,"discovered_at":"2026-09-08T18:49:42.117+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/enoki-efficient-multi-level-hallucination-detection-1788893776971.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/enoki-efficient-multi-level-hallucination-detection-1788893776971.txt","audio_summary":"Justy and Cody discuss the Enoki research paper, focusing on its approach to multi-level hallucination detection using text-anchored Open Information Extraction (OpenIE) to bridge the gap between claim-level verification and span-span level localization.","audio_duration_seconds":239,"audio_generated_at":"2026-09-08T18:56:29.384+00:00","audio_provider":"elevenlabs","audio_script_meta":{"episodeNumber":957,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"ElevenLabs v3","hostNames":{"a":"Justy","b":"Cody"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":957},{"id":"7cc2bdb7-e9fb-42de-b90e-e4f401f5283a","title":"Bilevel Coordinated Reflection: A Game-Theoretic Approach to Multi-Agent LLM Systems","url":"https://arxiv.org/html/2609.02750v1","type":"normal","description":"Multi-agent LLM systems commonly use an orchestrator to decompose a task for a team of workers and then improve through textual reflection. Despite strong empirical results, these systems lack a unified account of coordination, memory improvement, and the role of external verification. We model orchestrator-worker interaction as a bilevel coordination game: under bounded coupling, the workers' local-update game is an approximate potential game whose equilibrium slack is controlled by","cover_image_url":null,"discovered_at":"2026-09-08T18:47:20.617+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/bilevel-coordinated-reflection-a-game-theoretic-approach-to-mult-1788893514451.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/bilevel-coordinated-reflection-a-game-theoretic-approach-to-mult-1788893514451.txt","audio_summary":"Masonry and Eyre discuss the 'Bilevel Coordinated Reflection' paper, which applies game theory and stochastic approximation to multi-agent LLM coordination and memory improvement, specifically introducing SRMA to solve the 'text-only critic' plateau.","audio_duration_seconds":243,"audio_generated_at":"2026-09-08T18:52:05.106+00:00","audio_provider":"rime:coda","audio_script_meta":{"episodeNumber":956,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Rime Coda","hostNames":{"a":"Masonry","b":"Eyre"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":956},{"id":"d2731458-9b54-4e01-a021-6ecd999a7745","title":"To See a World in a Living Context:Unified Indoor-Outdoor Urban World Generation","url":"https://arxiv.org/html/2608.05879v2","type":"normal","description":"Text-driven 3D generation has advanced rapidly in creating large-scale outdoor environments and detailed indoor scenes, but these domains are usually synthesized independently, lacking the correspondence required for a coherent urban world. We present HoloWorld, a unified indoor-outdoor urban world generation framework built on a continuously updated cross-scale world context. Initializing from a user description, HoloWorld progressively represents and updates the diverse world information,","cover_image_url":null,"discovered_at":"2026-09-08T18:46:33.175+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/to-see-a-world-in-a-living-context-unified-indoor-outdoor-urban--1788893470944.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/to-see-a-world-in-a-living-context-unified-indoor-outdoor-urban--1788893470944.txt","audio_summary":"Edmund and Geffen discuss HoloWorld, a new framework for unified indoor-outdoor urban 3D generation that ensures the inside of a building actually matches its exterior and the surrounding city context.","audio_duration_seconds":258,"audio_generated_at":"2026-09-08T18:51:22.534+00:00","audio_provider":"speechify:simba-3.2","audio_script_meta":{"episodeNumber":955,"archetype":"deepDive","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Speechify Simba 3.2","hostNames":{"a":"Edmund","b":"Geffen"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"research-paper","event_type":null,"episode_number":955},{"id":"d4be045a-3297-4948-9d27-7b6d1017b7a5","title":"Causal evidence that language models use confidence to drive behaviour - Nature Machine Intelligence","url":"https://www.nature.com/articles/s42256-026-01293-x","type":"normal","description":"Kumaran et al. show that large language models making decisions on when to answer a question or abstain from answering can be influenced by boosting or suppressing confidence signals in the model.","cover_image_url":null,"discovered_at":"2026-09-08T18:44:49.354+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/causal-evidence-that-language-models-use-confidence-to-drive-beh-1788893530789.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/causal-evidence-that-language-models-use-confidence-to-drive-beh-1788893530789.txt","audio_summary":"Pippa and Tyler discuss a new Nature Machine Intelligence paper providing causal evidence that LLMs use internal confidence representations to decide when to abstain from answering. They explore the distinction between verbal confidence and internal activation states, and the implications for autonomous agents.","audio_duration_seconds":225,"audio_generated_at":"2026-09-08T18:52:20.951+00:00","audio_provider":"inworld-v2:inworld-tts-2","audio_script_meta":{"episodeNumber":954,"archetype":"quickRead","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Inworld TTS 2","hostNames":{"a":"Pippa","b":"Tyler"},"enrichmentUsed":false,"enrichmentProvider":null},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":954},{"id":"8f977ab6-f93a-40a3-b79d-1833a1ebc334","title":"Introducing Mercury 2.5 – Inception","url":"https://www.inceptionlabs.ai/blog/introducing-mercury-2-5","type":"normal","description":"Mercury 2.5 is the most capable diffusion LLM on the market. It runs at 1,107 tokens/sec and offers a 40% increase in intelligence over Mercury 2, comparable to cost-optimized frontier models.","cover_image_url":null,"discovered_at":"2026-09-08T18:39:27.286+00:00","audio_status":"complete","audio_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-mercury-2-5-inception-1788893312225.mp3","audio_transcript_url":"https://sthbxxuzklalnaiaeagd.supabase.co/storage/v1/object/public/exploring-next-audio/episodes/introducing-mercury-2-5-inception-1788893312225.txt","audio_summary":"Onyx and Echo discuss the launch of Mercury 2.5 from Inception, a diffusion-based LLM designed for extreme low-latency production workloads like voice and coding agents. They analyze the trade-offs of the diffusion architecture, the impressive speed claims, and how it fits into the broader trend of the 'control layer' being the actual product.","audio_duration_seconds":210,"audio_generated_at":"2026-09-08T18:48:43.468+00:00","audio_provider":"fishaudio:s2.1-pro-free","audio_script_meta":{"episodeNumber":953,"archetype":"quickRead","scriptModelLabel":"Gemma 4 31B","model":"google/gemma-4-31b-it","voiceProviderLabel":"Fish Audio S2.1 Pro","hostNames":{"a":"Onyx","b":"Echo"},"enrichmentUsed":true,"enrichmentProvider":"exa"},"domains":[],"concept_keys":[],"entities":[],"source_form":"blog-post","event_type":null,"episode_number":953}]}