{
 "generatedAt": "2026-08-28T01:08:00.398Z",
 "count": 711,
 "posts": [
  {
   "title": "Fast, fault-tolerant PyTorch training on AI Runtime",
   "url": "https://www.databricks.com/blog/fast-fault-tolerant-pytorch-training-ai-runtime",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": null,
   "summary": "At scale, your training efficiency is determined by a single metric: \"goodput\", the...",
   "firstSeen": "2026-08-28T01:08:00.398Z"
  },
  {
   "title": "Run Claude Managed Agents with Chat SDK",
   "url": "https://vercel.com/changelog/claude-managed-agents-with-chat-sdk",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-28T00:00:00.000Z",
   "summary": "You can now run Claude Managed Agents with Chat SDK . Claude Managed Agents handles the agent loop server-side, including the model, tools, session state, and sandboxed web research. That means you can ship a Slack research bot built on Claude Managed Agents and Chat SDK, with one persistent session per thread and streamed briefs with sources. What you get Token-by-token streaming : Replies render as the model writes them, over a single streamed response. Live activity feed : Tool calls and model requests are available during the turn, so you can surface a trace in the chat. No database to run",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ExtractBench: The Most Comprehensive Extraction Benchmark",
   "url": "https://www.llamaindex.ai/blog/introducing-extractbench",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "The most comprehensive document extraction benchmark: 14 systems scored on accuracy, completeness, grounding, and cost across 370 enterprise documents.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OCR for KYC: Why Standard Text Extraction Falls Short",
   "url": "https://www.llamaindex.ai/blog/ocr-for-kyc",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn why standard OCR falls short for KYC compliance and how agentic document extraction delivers the field-level accuracy AML regulations require.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Mortgage Document Automation: Transforming Loan Processing",
   "url": "https://www.llamaindex.ai/blog/mortgage-document-automation",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn how mortgage document automation transforms loan processing, from document ingestion and extraction to validation and system integration.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "KYC Automation: How to Replace Manual Verification at Scale",
   "url": "https://www.llamaindex.ai/blog/kyc-automation",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn how KYC automation replaces manual verification with scalable, compliant workflows that cut costs, reduce errors, and speed up onboarding.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Why Deep Extraction is Superior to Single-Pass Pipelines",
   "url": "https://www.llamaindex.ai/blog/deep-extraction",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn why single-pass extraction fails and how deep extraction uses agentic verification to deliver production-grade accuracy on complex documents.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AI Document Classification: A Practical Guide",
   "url": "https://www.llamaindex.ai/blog/ai-document-classification",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn how AI document classification automates document sorting and routing at scale, and what separates systems that perform on real documents from those that don",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OCR Accuracy Explained: How to Improve It",
   "url": "https://www.llamaindex.ai/blog/ocr-accuracy",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn what impacts OCR accuracy, how it’s measured, and how to improve performance in real-world document processing workflows.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Unstructured Data Extraction: Turn Documents into Insights",
   "url": "https://www.llamaindex.ai/blog/unstructured-data-extraction",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn how unstructured data extraction turns documents, PDFs, and text into structured insights using AI, NLP, and LLMs for scalable data processing.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Agentic Document Extraction Improves Accuracy and Automation",
   "url": "https://www.llamaindex.ai/blog/agentic-document-extraction",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Agentic document extraction uses AI reasoning and visual grounding to accurately process complex documents without templates. Learn how it works.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agentic Document Processing: How AI Agents Automate Workflows",
   "url": "https://www.llamaindex.ai/blog/agentic-document-processing",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Agentic document processing uses AI agents to autonomously handle document workflows end to end. Learn how it works and where to start.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OCR for Tables: How to Extract Structured Data from Documents",
   "url": "https://www.llamaindex.ai/blog/ocr-for-tables",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "OCR for tables converts complex document layouts into structured, machine-readable data. Learn how LlamaParse preserves table integrity.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agentic OCR for Receipts: Why Traditional Pipelines Break",
   "url": "https://www.llamaindex.ai/blog/ocr-for-receipts",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "OCR for receipts breaks when layouts vary and rules pile up. Discover how agentic OCR reconstructs line items, totals, and structured data for automation.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OCR for Images: Top AI Software for Image-to-Text Conversion",
   "url": "https://www.llamaindex.ai/blog/ocr-for-images",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "OCR for images helps convert photos, labels, and screenshots into structured text. Compare the top AI OCR tools and learn what makes a reliable image-to-text system.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OCR for Invoices: How to Extract Data with Accuracy and Speed",
   "url": "https://www.llamaindex.ai/blog/ocr-for-invoices",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Discover how OCR for invoices streamlines finance operations, automates data extraction, reduces errors, and speeds up accounts payable with LlamaParse.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What Is Agentic OCR? The Next Evolution of Intelligent Document Automation",
   "url": "https://www.llamaindex.ai/blog/agentic-ocr",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Discover how agentic OCR transforms document processing with multimodal reasoning, self-correction loops, and template-free automation.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OCR for Accounts Payable: Benefits, Challenges, and Best Practices",
   "url": "https://www.llamaindex.ai/blog/ocr-for-accounts-payable",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn how OCR for accounts payable automates invoice processing, improves accuracy, reduces costs, and integrates structured data into ERP systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Exploring Static Embedding Retrieval",
   "url": "https://www.llamaindex.ai/blog/exploring-static-embedding-retrieval",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "LlamaIndex is a simple, flexible framework for building knowledge assistants using LLMs connected to your enterprise data.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How LlamaIndex Uses Temporal to Scale Reliable Document Orchestration",
   "url": "https://www.llamaindex.ai/blog/temporal-scale-document-orchestration",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "LlamaIndex is a simple, flexible framework for building knowledge assistants using LLMs connected to your enterprise data.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OCR Automation: Demo vs. Production",
   "url": "https://www.llamaindex.ai/blog/ocr-automation",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "OCR that works in a demo often stalls in production. How modern document pipelines handle real corpora, and how to hit 90%+ straight-through rates.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Intelligent OCR: Production Document AI",
   "url": "https://www.llamaindex.ai/blog/intelligent-ocr",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Intelligent OCR turns documents into validated, structured data, not just text. How production pipelines handle parsing, extraction, and confidence.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Document OCR is Not Getting Commoditized",
   "url": "https://www.llamaindex.ai/blog/document-ocr-is-not-getting-commoditized",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "LlamaIndex is a simple, flexible framework for building knowledge assistants using LLMs connected to your enterprise data.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LLM OCR: Why the Errors Got Harder to Spot",
   "url": "https://www.llamaindex.ai/blog/llm-ocr",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "LLM OCR lowered the error rate but changed what an error looks like. Why fluent output hides silent substitutions, and what has to sit around the model.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Parse Gateway: Smart, Page-Level Document Parsing Routing",
   "url": "https://www.llamaindex.ai/blog/parse-gateway-smart-page-level-document-parser-routing",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Cut parsing costs without losing accuracy. Parse Gateway routes each PDF page to the right tier based on complexity, from free LiteParse to advanced OCR.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Real Alternative to Template OCR Isn",
   "url": "https://www.llamaindex.ai/blog/alternative-to-template-ocr",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Discover why the future of OCR isn",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Mortgage Banking Document Automation: Common Workflow Gaps",
   "url": "https://www.llamaindex.ai/blog/mortgage-banking-document-automation",
   "source": "LlamaIndex",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Discover why mortgage banking document automation fails at workflow handoffs and how AI document extraction improves speed, accuracy, and compliance.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we contain Claude across products",
   "url": "https://www.anthropic.com/engineering/how-we-contain-claude",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "An update on recent Claude Code quality reports",
   "url": "https://www.anthropic.com/engineering/april-23-postmortem",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling Managed Agents: Decoupling the brain from the hands",
   "url": "https://www.anthropic.com/engineering/managed-agents",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we built Claude Code auto mode: a safer way to skip permissions",
   "url": "https://www.anthropic.com/engineering/claude-code-auto-mode",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Harness design for long-running application development",
   "url": "https://www.anthropic.com/engineering/harness-design-long-running-apps",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Eval awareness in Claude Opus 4.6’s BrowseComp performance",
   "url": "https://www.anthropic.com/engineering/eval-awareness-browsecomp",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Quantifying infrastructure noise in agentic coding evals",
   "url": "https://www.anthropic.com/engineering/infrastructure-noise",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building a C compiler with a team of parallel Claudes",
   "url": "https://www.anthropic.com/engineering/building-c-compiler",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Demystifying evals for AI agents",
   "url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Demystifying evals for AI agents",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Effective harnesses for long-running agents",
   "url": "https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing advanced tool use on the Claude Developer Platform",
   "url": "https://www.anthropic.com/engineering/advanced-tool-use",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Claude can now discover, learn, and execute tools dynamically to enable agents that take action in the real world. Here’s how.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Code execution with MCP: building more efficient AI agents",
   "url": "https://www.anthropic.com/engineering/code-execution-with-mcp",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn how code execution with the Model Context Protocol enables agents to handle more tools while using fewer tokens, reducing context overhead by up to 98.7%.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Making Claude Code more secure and autonomous with sandboxing",
   "url": "https://www.anthropic.com/engineering/claude-code-sandboxing",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Learn how Claude Code's new sandboxing feature protects developers with filesystem and network isolation, reducing permission prompts and increasing user safety.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Equipping agents for the real world with Agent Skills",
   "url": "https://www.anthropic.com/engineering/equipping-agents-for-the-real-world-with-agent-skills",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Discover how Anthropic builds AI agents with practical capabilities through modular skills, enabling them to handle complex real-world tasks more effectively and reliably.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Effective context engineering for AI agents",
   "url": "https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Anthropic is an AI safety and research company that's working to build reliable, interpretable, and steerable AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A postmortem of three recent issues",
   "url": "https://www.anthropic.com/engineering/a-postmortem-of-three-recent-issues",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "This is a technical report on three bugs that intermittently degraded responses from Claude. Below we explain what happened, why it took time to fix, and what we're changing.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Writing effective tools for AI agents—using AI agents",
   "url": "https://www.anthropic.com/engineering/writing-tools-for-agents",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Writing effective tools for AI agents—using AI agents",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Claude Desktop Extensions: One-click MCP server installation for Claude Desktop",
   "url": "https://www.anthropic.com/engineering/desktop-extensions",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Claude Desktop Extensions: One-click MCP server installation for Claude Desktop",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we built our multi-agent research system",
   "url": "https://www.anthropic.com/engineering/multi-agent-research-system",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "On the the engineering challenges and lessons learned from building Claude's Research system",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Best practices for Claude Code - Claude Code Docs",
   "url": "https://www.anthropic.com/engineering/claude-code-best-practices",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Tips and patterns for getting the most out of Claude Code, from configuring your environment to scaling across parallel sessions.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The \"think\" tool: Enabling Claude to stop and think",
   "url": "https://www.anthropic.com/engineering/claude-think-tool",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "A blog post for developers, describing a new method for complex tool-use situations",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Claude SWE-Bench Performance",
   "url": "https://www.anthropic.com/engineering/swe-bench-sonnet",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Explore Claude's breakthrough performance on SWE-Bench, demonstrating advanced software engineering capabilities and code generation accuracy. Learn about our technical evaluation methods.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building Effective AI Agents",
   "url": "https://www.anthropic.com/engineering/building-effective-agents",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Discover how Anthropic approaches the development of reliable AI agents. Learn about our research on agent capabilities, safety considerations, and technical framework for building trustworthy AI.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Contextual Retrieval in AI Systems",
   "url": "https://www.anthropic.com/engineering/contextual-retrieval",
   "source": "Anthropic Engineering",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Explore how Anthropic enhances AI systems through advanced contextual retrieval methods. Learn about our approach to improving information access and relevance in large language models.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Decoding cosmic signals with deep learning and Keras",
   "url": "https://developers.googleblog.com/decoding-cosmic-signals-with-deep-learning-and-keras",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Astroparticle physics sits at the exciting intersection of astrophysics and particle physics and stu...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Enterprise-Grade Precision for Long-Context Multimodal Embedding Inference on Cloud TPU",
   "url": "https://developers.googleblog.com/enterprise-grade-precision-for-long-context-multimodal-embedding-inference-on-cloud-tpu",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Google Cloud has natively integrated TPU support into the vLLM serving engine, allowing developers to elastically scale high-demand embedding pipelines using Google Kubernetes Engine (GKE). To handle massive 15K+ token contexts for models like Qwen3-Embedding-8B, the engineering team implemented TPU-specific optimizations such as hardware-safe tensor alignment, JAX/XLA compilation pre-warming, and a hybrid StepPool architecture for chunked prefill management. These enhancements achieve near-perfect numerical parity with reference GPU baselines, and developers can immediately leverage the open-",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to Evaluate Live & Voice Agents in ADK",
   "url": "https://developers.googleblog.com/how-to-evaluate-live-voice-agents-in-adk",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Moving live voice agents from demo to production requires rigorous, automated testing to handle the unpredictability of real multi-turn conversations. ADK now provides native live evaluation, allowing developers to test graph-based agent workflows against LLM-driven simulated users that generate actual audio via Gemini TTS. By defining evaluation scenarios and natural-language rubrics, you can automatically score audio responses and tool executions, inspect the resulting transcripts in ADK Web, or run the CLI directly in your CI/CD pipeline.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Build zero-trust AI agents with Google's Agent Development Kit",
   "url": "https://developers.googleblog.com/build-zero-trust-ai-agents-with-googles-agent-development-kit",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Building autonomous AI agents that mutate production state requires moving beyond soft system prompts to a robust zero-trust architecture. To secure Google Agent Development Kit (ADK) workflows against prompt injections and malicious execution, developers must implement hardware-backed cryptographic signatures for database writes, kernel-level sandboxing with gVisor for dynamic code, and deterministic semantic gateways for I/O validation. By enforcing these hard security boundaries at the infrastructure level, you can safely deploy multi-tool AI agents without risking unauthorized data manipul",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "HeyGen x Google Cloud: Bringing Avatar IV to TPUs",
   "url": "https://developers.googleblog.com/heygen-x-google-cloud-bringing-avatar-iv-to-tpus",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "HeyGen ported their 18B+ parameter Avatar IV video generation model to Google Cloud's Trillium (v6e) TPUs via torchax and XLA, utilizing FSDP and Ulysses sequence parallelism across an eight-chip mesh. To achieve a 1.86x speedup for real-time streaming, the engineering team pipelined exposed all-to-all collectives, aligned sparse attention block sizes to eliminate mask padding, and bypassed softmax serial dependencies using a precomputed Cauchy-Schwarz upper bound. These custom Pallas kernel and compiler optimizations were deployed only after passing rigorous two-tier quality gates to guarante",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Credentio: Open Source C++ Library for C2PA Content Credentials from Google",
   "url": "https://developers.googleblog.com/introducing-credentio-open-source-c-library-for-c2pa-content-credentials-from-google",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Credentio is a newly released, open-source C++ library from Google that allows developers to integrate high-performance, local-first validation of C2PA Content Credentials into their client and server applications. By processing assets entirely locally with a highly optimized memory footprint, the library delivers instant validation verdicts for multi-gigabyte media files without incurring cloud latency, bandwidth costs, or data privacy risks. The library currently features deep manifest parsing alongside configurable trust list integration, and is available now on Google Source with future pl",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Mastering Edge AI on Raspberry Pi with LiteRT and Gemma",
   "url": "https://developers.googleblog.com/mastering-edge-ai-on-raspberry-pi-with-litert-and-gemma",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Deploying secure, real-time Edge AI on Raspberry Pi is now simplified using LiteRT and lightweight Gemma open models. LiteRT optimizes CPU and GPU performance, delivering fast token speeds for models like Gemma4, enabling real-time local reasoning for robotics. Developers can quickly convert, quantize, and run these models using the lightweight LiteRT CLI tool. Support for Hailo AI accelerators is also coming very soon.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Why Go is an Ideal Language for AI-Assisted Software Engineering",
   "url": "https://developers.googleblog.com/why-go-is-an-ideal-language-for-ai-assisted-software-engineering",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "As AI coding assistants shift the developer's primary role from writing boilerplate to reviewing and maintaining systems, language choice becomes critical for long-term architectural integrity. Go directly addresses this new paradigm by utilizing its strict compiler, integrated toolchain, and uncompromising readability to provide deterministic guardrails that help AI models self-correct and generate highly standardized code. By enforcing ecosystem-wide consistency and strict backward compatibility, the Go platform empowers engineering teams to efficiently verify, optimize, and maintain high-ve",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agent Plugins package your skills, tools, and more",
   "url": "https://developers.googleblog.com/agent-plugins-package-your-skills-tools-and-more",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Agent Plugins 1.0.0 is a new, vendor-neutral directory specification—backed by Google, Amazon, Microsoft, and others—for packaging Agent Skills and MCP servers into a single portable unit. By standardizing the manifest (plugin.json) and utilizing a fixed directory layout, it eliminates the need for developers to maintain separate wrappers or configurations to support different AI coding agents and IDEs. Google has officially joined as a Core Maintainer and already rolled out support in the Agents CLI and Data Agent Kit, allowing developers to start building and distributing interoperable plugi",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling AI Agent Infrastructure with the MCP Stateless updates",
   "url": "https://developers.googleblog.com/scaling-ai-agent-infrastructure-with-the-mcp-stateless-updates",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "The 2026-07-28 Model Context Protocol (MCP) specification replaces legacy stateful constraints with a fully stateless core, enabling cloud-native horizontal scaling, serverless deployments, and standard round-robin load balancing. This architectural shift introduces standardized HTTP headers for efficient routing without deep packet inspection, caching controls, and Multi Round-Trip Requests (MRTR) to handle interactive and long-running tasks without blocking connections. Developers can immediately begin migrating their agentic applications to this highly scalable infrastructure using the newl",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Model routing with Google Cloud API Gateway",
   "url": "https://developers.googleblog.com/a-unified-api-for-ai-model-routing",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Google Cloud API Gateway now offers a model routing feature in Public Preview, allowing developers to dynamically route traffic to models like Gemini, Claude, or OpenAI OSS-GPT without hardcoding endpoints or managing open-source proxies. Developers can easily configure these routing rules directly within their OpenAPI 3.x specifications by mapping virtual model names to specific backend targets on a shared host. Once deployed, the Gateway acts as a serverless ingress layer that accepts standard OpenAI-compatible requests, automatically transcodes the payload to the native schema of the target",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling real-time AI agents with session-aware load balancing",
   "url": "https://developers.googleblog.com/scaling-real-time-ai-agents-with-session-aware-load-balancing",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Real-time AI agents break traditional request-response load balancing paradigms because they rely on long-lived, stateful bidirectional streams that obscure true server capacity. To solve this, developers must implement application-level session tracking directly within the runtime to accurately measure the committed concurrent workload of active conversations. By feeding these precise session counts alongside standard CPU utilization metrics into a hybrid routing algorithm, infrastructure can effectively distribute stateful AI traffic and prevent individual backend bottlenecks.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Enable on-demand expertise with Agent Skills in Genkit Go",
   "url": "https://developers.googleblog.com/enable-on-demand-expertise-with-agent-skills-in-genkit-go",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "To prevent context window bloat and reduce token consumption, Genkit Go introduces Agent Skills based on a progressive disclosure architecture. Developers can package specialized instructions, scripts, and references into modular SKILL.md bundles where only the frontmatter metadata is initially exposed to the agent's system prompt. When a task matches the skill's description, Genkit's middleware dynamically loads the full instruction body and associated assets, ensuring the model accesses precise workflows exactly when needed.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agent and Model Evaluations in Gemini Enterprise Agent Platform are now GA",
   "url": "https://developers.googleblog.com/agent-and-model-evaluations-in-gemini-enterprise-agent-platform-are-now-ga",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Agent Platform's evaluation service is now generally available, providing developers with a unified engine to measure agent quality consistently across local development experiments and live production traffic. You can evaluate agents using over 20 pre-built metrics, DeepMind-backed adaptive rubrics, or custom code-based and LLM-as-a-judge metrics stored in a centralized, versioned registry. The service integrates directly into existing workflows via the Agent Platform SDK, agents-cli, and ADK, offering built-in user and environment simulators to automate complex multi-turn testing and streaml",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to use Google microbenchmarks for evaluating TPU performance",
   "url": "https://developers.googleblog.com/how-to-use-google-microbenchmarks-for-evaluating-tpu-performance",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Google's open-source TPU microbenchmark suite provides developers with granular performance metrics across Network, Compute, HBM, Host Transfer, and Attention components to validate real-world hardware capabilities. By leveraging these benchmarks to establish a Roofline model, engineers can accurately diagnose whether their machine learning workloads are compute-, memory-, or network-bound. This empirical baseline directly guides targeted software optimizations—such as kernel tuning, mesh sharding, and rematerialization—to maximize hardware utilization for large-scale model deployments.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Run Ray on TPU, Part 2: Ray AI libraries",
   "url": "https://developers.googleblog.com/run-ray-on-tpu-part-2-ray-ai-libraries",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "This second installment explores how Ray’s higher-level libraries—Serve, Data, and Train—abstract the complexities of running AI workloads on Google's TPU slices. Ray Serve uses a simple topology configuration to correctly gang-schedule large multi-host models, while Ray Data eliminates data-loading bottlenecks by feeding accelerators directly with native JAX batches. Finally, JaxTrainer streamlines distributed training across TPUs by automatically handling cross-slice coordination, checkpointing, and fault tolerance.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling Agentic RL: High-Throughput Agentic Training with Tunix",
   "url": "https://developers.googleblog.com/scaling-agentic-rl-high-throughput-agentic-training-with-tunix",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Tunix is Google’s new JAX-native post-training library designed to eliminate TPU idling bottlenecks when training multi-turn, tool-using LLM reasoning agents. It maximizes hardware throughput by combining highly concurrent, asynchronous rollouts with a decoupled producer-consumer pipeline, ensuring the trainer is constantly fed even while agents wait on network I/O or environment steps. Additionally, Tunix provides plug-and-play abstractions and continuous macro-level profiling, allowing developers to easily integrate custom open-source environments and optimize complex distributed workflows w",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Run Ray on TPU, Part 1: The foundations",
   "url": "https://developers.googleblog.com/run-ray-on-tpu-part-1-the-foundations",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Ray 2.55 introduces official, first-class support for Google Cloud TPUs, enabling developers to run distributed Python workloads on Google's accelerators using the familiar Ray task-and-actor APIs. To handle the strict networking requirement of keeping multi-host TPU \"slices\" together over their Inter-Chip Interconnect (ICI), the KubeRay Operator on GKE automatically provisions and labels the underlying hardware layout. Ray Core utilizes these labels via its slice_placement_group() primitive to atomically reserve complete slices, allowing developers to deploy jobs through KubeRay, Ray Train, o",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building scalable AI agents with modular prompt transpilation",
   "url": "https://developers.googleblog.com/building-scalable-ai-agents-with-modular-prompt-transpilation",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "To resolve the scaling bottlenecks and runtime errors caused by monolithic system prompts, engineering teams should treat prompts as build artifacts by modularizing instructions into reusable templates. By running these modular \"skill files\" through a transpiler, developers can enforce static validation, catch missing dependencies at build time, and integrate prompt generation directly into their CI/CD pipelines. This deterministic approach prevents code drift and ultimately establishes a safe framework where agents can propose updates to their own logic via standard pull requests.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Evolving Spec-Driven Development: Conductor Now Supports Antigravity",
   "url": "https://developers.googleblog.com/evolving-spec-driven-development-conductor-now-supports-antigravity",
   "source": "Google Developers Blog",
   "group": "Agent frameworks and runtimes",
   "published": null,
   "summary": "Conductor has evolved from a Gemini CLI extension into a portable plugin, bringing conversational Spec-Driven Development (SDD) to ecosystems like Antigravity CLI and Claude. Rather than relying on strict command sequences, developers can now chat naturally with their AI assistant while it dynamically manages persistent markdown artifacts (like spec.md and plan.md) in the background. This update eliminates workflow friction while ensuring your repository remains a version-controlled, single source of truth for your project's architecture and state across different AI tools.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The two AI gateway patterns in production inference",
   "url": "https://www.baseten.co/blog/ai-gateways-production-inference",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "Access gateways and serving gateways explained, and how they handle identity, tenancy, limits, and metering for teams using and serving AI models.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to run any open model inside DeepSeek Harness",
   "url": "https://www.baseten.co/blog/how-to-run-any-open-model-inside-deepseek-harness",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "Learn how to power DeepSeek Harness with Baseten Model APIs and run open models like Kimi K3, GLM 5.2, and DeepSeek V4 Pro in under 5 minutes.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How leading platforms ensure observability for LLM inference",
   "url": "https://www.baseten.co/blog/how-leading-platforms-ensure-observability-for-llm-inference",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "Learn how metrics, logs, and traces work together to catch slow responses, errors, and failed deployments before users do.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Inference engineering for DeepSeek V4 Pro 0813",
   "url": "https://www.baseten.co/blog/inference-engineering-for-deepseek-v4-pro-0813",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "DeepSeek V4 Pro 0813 is a 1.7T-parameter open frontier model under the MIT license and is available for inference today on Baseten model APIs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Baseten delivers open-source inference for You.com",
   "url": "https://www.baseten.co/blog/baseten-delivers-open-source-inference-for-youcom",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "Baseten is proud to power the inference behind You.com's search and answer stack.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing NVIDIA Nemotron 3.5 Lightning",
   "url": "https://www.baseten.co/blog/introducing-nemotron-35-lightning",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "NVIDIA’s Nemotron 3.5 Lightning, now on Baseten, delivers high-throughput, efficient reasoning for faster and more accurate agentic workflows.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing NVIDIA Nemotron 3.5 ASR Streaming",
   "url": "https://www.baseten.co/blog/introducing-nvidia-nemotron-35-asr-streaming",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "Deploy NVIDIA Nemotron 3.5 ASR for low-latency, production-ready speech recognition with 6x higher throughput and multilingual support.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Laguna S 2.1 goes Greek: a repository-scale game transformation",
   "url": "https://www.baseten.co/blog/laguna-s-21-goes-greek",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "We put Poolside’s new Laguna S 2.1 model to the test, tasking it with a repository-scale transformation of the open-source game Hypersomnia.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Fine-tuning Qwen3-TTS for high-quality voice cloning",
   "url": "https://www.baseten.co/blog/fine-tuning-qwen3-tts-for-high-quality-voice-cloning",
   "source": "Baseten",
   "group": "Inference companies",
   "published": null,
   "summary": "A guide to fine-tuning Qwen3-TTS for high-quality voice cloning, comparing ICL, speaker-embedding-only, and fine-tuning approaches with a full training recipe.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Cerebras CS-4: The Fastest AI Gets Faster",
   "url": "https://www.cerebras.ai/blog/introducing-cerebras-cs-4",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Cerebras CS-4 delivers up to 30x faster AI inference than GPUs, with a modular rack-scale architecture built for hyperscale AI deployment.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Ultrafast Frontier Inference | Cerebras Hot Chips 2026",
   "url": "https://www.cerebras.ai/blog/ultrafast-frontier-inference-cerebras-deep-dive-at-hot-chips-2026",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "At Hot Chips 2026, Cerebras details CS-4 and Nexus, plus the CS-5 and CS-6 roadmap for faster, more efficient frontier AI inference.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Accelerating GPT-5.6 Sol Ultrafast with OpenAI",
   "url": "https://www.cerebras.ai/blog/accelerating-gpt-5-6-sol-ultrafast-with-openai",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Cerebras powers OpenAI’s GPT-5.6 Sol Ultrafast in the OpenAI API, delivering frontier intelligence at real-time speeds for critical AI work.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Faster AI Inference Strengthens Cybersecurity",
   "url": "https://www.cerebras.ai/blog/ai-inference-cybersecurity",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Discover how faster AI inference helps cybersecurity teams improve threat detection, AI security, validation, and response without sacrificing speed.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cerebras and Upstage Bring Fast AI to Korea",
   "url": "https://www.cerebras.ai/blog/cerebras-and-upstage-bring-ultra-fast-ai-to-korea",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Cerebras and Upstage bring ultra-fast AI inference to Korea, delivering up to 2,000 tokens per second for real-time enterprise AI applications.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AI-Native Engineering Interviews at Cerebras",
   "url": "https://www.cerebras.ai/blog/hiring-engineers-for-an-ai-native-world",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Learn how Cerebras redesigned engineering interviews for the AI era, evaluating AI collaboration, engineering judgment, verification, and real-world skills.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Gemma 4 on Cerebras: Fast Multimodal AI",
   "url": "https://www.cerebras.ai/blog/first-look-gemma-4-on-cerebras-3-fast-multimodal-apps-we-built",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Build lightning-fast multimodal AI apps with Gemma 4 on Cerebras. Learn image understanding, vision workflows, and high-speed inference for developers.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Gemma 4 on Cerebras—The Fastest Inference is Now Multimodal",
   "url": "https://www.cerebras.ai/blog/gemma-4-on-cerebras-the-fastest-inference-is-now-multimodal",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Gemma 4 on Cerebras delivers the fastest multimodal inference—1,500+ TPS for real-time image understanding, agentic workflows, and document AI.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Never Loop Without Verifiers | Cerebras Blog",
   "url": "https://www.cerebras.ai/blog/never-loop-without-verifiers",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Discover why AI loops require verification to avoid spiraling. See how Cerebras runs Gemma 4 at 1,500 tokens/sec for fast, autonomous visual loops.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Economics of AI Reasoning",
   "url": "https://www.cerebras.ai/blog/the-economics-of-ai-reasoning",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Reasoning boosts AI accuracy, but at a steep cost. Explore test-time compute, agent performance, speed tradeoffs, and when thinking hurts.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Which is faster: Gemini 3.5 Flash or Kimi K2.6 on Cerebras",
   "url": "https://www.cerebras.ai/blog/which-is-faster-gemini-3-5-flash-or-kimi-k2-6-on-cerebras",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Kimi K2.6 on Cerebras matches Gemini 3.5 Flash in intelligence while delivering 5× faster output, lower latency, and open-weight flexibility.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Why the AI Race Shifted to Speed​​​​",
   "url": "https://www.cerebras.ai/blog/why-the-ai-race-shifted-to-speed",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Cerebras powers the world's fastest AI inference on the biggest wafer chip. Cerebras CS-4 delivers up to 30x faster inference than GPUs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to stop your autoresearch loop from cheating",
   "url": "https://www.cerebras.ai/blog/how-to-stop-your-autoresearch-loop-from-cheating",
   "source": "Cerebras",
   "group": "Inference companies",
   "published": null,
   "summary": "Stop autoresearch loops from “cheating” by enforcing strict evaluation, isolating experiments, and designing metrics that prevent shortcuts and false gains.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Build agentic creative workflows with Amazon Quick and fal",
   "url": "https://aws.amazon.com/blogs/machine-learning/build-agentic-creative-workflows-with-amazon-quick-and-fal",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-27T23:04:22.000Z",
   "summary": "Creative teams produce more assets than ever, but fragmented tools and manual context transfer slow production. This post shows how to build a reusable agent harness with Amazon Quick and fal, connected through the Model Context Protocol (MCP), using two hands-on workflows: an eight-panel storyboard and a music-video concept prototype.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Breaking Claude Code Opus 5 Auto Mode",
   "url": "https://simonwillison.net/2026/Aug/27/breaking-claude-code-opus-5-auto-mode",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-27T22:50:25.000Z",
   "summary": "Breaking Claude Code Opus 5 Auto Mode Anthropic are putting a great deal of faith in Claude Code's auto mode for protecting their coding agent users against prompt injection attacks. They recently made that the default and have made bold claims about its effectiveness. Johann Rehberger is one of the most credible prompt injection researchers active today. He found an attack against auto mode which he claims works 80% of the time, by tricking Claude Code into downloading and uncompressing a zip archive, then executing code that imports base64 without noticing that this will import and execute a",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing OpenAI models on Amazon Bedrock for in-country inferencing in India",
   "url": "https://aws.amazon.com/blogs/machine-learning/introducing-openai-models-on-amazon-bedrock-for-in-country-inferencing-in-india",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-27T18:36:08.000Z",
   "summary": "Amazon Bedrock now supports the OpenAI GPT-5.6 models, Terra and Luna, in India with India geographic cross-Region inference. If you have local data processing requirements, you can now use these models at scale while Amazon Bedrock keeps inference requests and data within India.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "So, is ClickHouse winning the observability wars?",
   "url": "https://clickhouse.com/blog/is-clickhouse-winning-the-observability-wars",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-27T18:24:53.000Z",
   "summary": "After claims that ClickHouse is “winning the observability wars” sparked debate, we reflect on why it has become a leading storage and query engine, where it still falls short, and why winning the database layer isn’t the same as winning observability.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we saved 100 terabytes of memory by optimizing 1.1.1.1’s DNS cache",
   "url": "https://blog.cloudflare.com/dns-cache-memory-optimization-1111",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-27T17:02:35.000Z",
   "summary": "Five Rust-level memory optimizations to the DNS cache layout of Big Pineapple cut per-entry memory by 56%, freeing approximately 100 TB of memory across Cloudflare's fleet.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Deepgram deepens Amazon SageMaker AI observability with Enhanced Metrics",
   "url": "https://aws.amazon.com/blogs/machine-learning/deepgram-deepens-amazon-sagemaker-ai-observability-with-enhanced-metrics",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-27T16:11:27.000Z",
   "summary": "Self-hosted speech AI carries an observability trade-off: the numbers that drive capacity planning and cost management stay locked inside the vendor container. Deepgram closes that gap on Amazon SageMaker AI with two capabilities that land billing, usage, and per-GPU metrics directly in your own Amazon CloudWatch account.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Reduce ASR inference costs by 75% with NVIDIA MPS on Amazon EC2",
   "url": "https://aws.amazon.com/blogs/machine-learning/reduce-asr-inference-costs-by-75-with-nvidia-mps-on-amazon-ec2",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-27T16:05:10.000Z",
   "summary": "Serving automatic speech recognition (ASR) models at scale is costly when each request uses only a fraction of a GPU. Learn how NVIDIA CUDA Multi-Process Service (MPS) with NVIDIA Triton Inference Server on Amazon EC2 GPU instances cuts GPU infrastructure by 75% while holding sub-second latency at 92.1 requests per second per GPU.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OpenClaw went viral. Meet the maintainers building and securing it.",
   "url": "https://github.blog/open-source/maintainers/openclaw-went-viral-meet-the-maintainers-building-and-securing-it",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-27T16:00:00.000Z",
   "summary": "OpenClaw is the fastest-growing project in GitHub history. Peter Steinberger and several maintainers share what they learned in the project's first six months. The post OpenClaw went viral. Meet the maintainers building and securing it. appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Reimagining work: How Pythian’s internal AI playbook delivers customer ROI",
   "url": "https://cloud.google.com/blog/topics/startups/how-pythians-internal-ai-playbook-delivers-customer-roi",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-27T16:00:00.000Z",
   "summary": "When Pythian rolled out Google Cloud’s Gemini Enterprise across our 500-person company in 27 countries, the goal was simple: use our own company as a proving ground to discover how enterprise AI actually delivers ROI. What we found changed our strategy entirely. Since the rollout of Gemini Enterprise and our previous enterprise AI deployments, Pythian observed firsthand why so many enterprise AI initiatives stall out or fail. Most organizations trap themselves in a tool-centric mindset — buying licenses, making tools broadly available, and assuming value will naturally follow. They get stuck c",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cursor is now available in the AI SDK harness layer",
   "url": "https://vercel.com/changelog/cursor-ai-sdk-harness-adapter",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-27T14:47:00.000Z",
   "summary": "The AI SDK harness layer now supports Cursor through the official @ai-sdk/harness-cursor adapter. The harness layer lets your application run different coding agents through the same HarnessAgent interface, so you can switch agents without changing your application code. Pass cursor to HarnessAgent : Under the hood, the adapter uses @ai-sdk/harness-acp to connect Cursor to HarnessAgent through the Agent Client Protocol (ACP). Supported harnesses now include, in addition to Cursor, Claude Code, Cline, Codex, Deep Agents, Grok Build, OpenCode, and Pi, with more coming soon. Read the Cursor harne",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Tailscale PAM beta: Manage connectivity and privileged access in one place",
   "url": "https://tailscale.com/blog/tailscale-pam-beta",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-08-27T14:00:00.000Z",
   "summary": "Just-in-time access, resource policies, and session auditing, right where you need them.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building a New Rust Driver for ScyllaDB’s DynamoDB API – with 58% More Throughput",
   "url": "https://www.scylladb.com/2026/08/27/new-rust-driver-for-scylladbs-dynamodb-api",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-08-27T13:10:38.000Z",
   "summary": "How our new Rust driver load-balances DynamoDB-style requests across a ScyllaDB cluster, and how we extended Latte to measure its performance",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Find deployments faster with redesigned filters",
   "url": "https://vercel.com/changelog/find-deployments-faster-with-redesigned-filters",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-27T13:00:00.000Z",
   "summary": "The Deployments page now has redesigned filters that make it faster to find a specific deployment. You can: Apply suggested filters for common searches. Type to find an option instead of scrolling through the list. Describe what you’re looking for in natural language, and Vercel automatically applies the matching filters. Open any project's Deployments page to try the new filters. Read more",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Better answers, broader thinking: What students gain from ChatGPT and critical-thinking training",
   "url": "https://openai.com/index/what-students-gain-from-chatgpt-critical-thinking-training",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-27T09:00:00.000Z",
   "summary": "A randomized study of more than 1,000 students examines ChatGPT, critical thinking, originality, and student performance on a real-world university assignment.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Object Storage + WAL: Lakebase Postgres for the agentic era",
   "url": "https://www.databricks.com/blog/object-storage-wal-lakebase-postgres-agentic-era",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-27T07:18:59.000Z",
   "summary": "Agents that interact with a traditional OLTP database often create bottlenecks at the storage layer. New deployments...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The best workflow engine is a programming language",
   "url": "https://vercel.com/blog/the-best-workflow-engine-is-a-programming-language",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-27T07:00:00.000Z",
   "summary": "The idea of orchestrating long-running, stateful logic on top of unreliable, stateless infrastructure isn't new. We've had message queues, job runners, microservice choreographies, and full-blown workflow engines for a long time. What we never had was a version of it that felt good to write. I'd spent about six months working on a fork of Temporal, mostly on weekends, trying to turn it into a serverless answer with DX that felt more Vercel-native. Eventually it dawned on me that to ship that experience, I'd need to own the execution environment too. So I dropped the fork, joined Vercel, and st",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Hot Chips 2026: Intel’s Crescent Island",
   "url": "https://chipsandcheese.com/p/hot-chips-2026-intels-crescent-island",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-27T05:12:31.000Z",
   "summary": "Hello you fine Internet folks,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "August 2026: LangChain Newsletter — Managed Deep Agents, LLM Gateway, and More",
   "url": "https://www.langchain.com/blog/august-2026-langchain-newsletter",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-27T04:26:36.000Z",
   "summary": "Managed Deep Agents and LLM Gateway hit public beta, plus Deep Agents v0.7, Tuned Evaluators, Bring Your Own Cloud on AWS, and LangSmith Engine upgrades.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Expanding OpenAI’s presence in Brazil",
   "url": "https://openai.com/index/expanding-our-presence-in-brazil",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-27T03:00:00.000Z",
   "summary": "OpenAI is expanding its presence in Brazil, deepening engagement with developers, businesses, and communities to support AI adoption across the country.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] NVIDIA buys HuggingFace for $13B, as OpenAI publishes their HF incident retro",
   "url": "https://www.latent.space/p/ainews-nvidia-buys-huggingface-for",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-27T01:50:54.000Z",
   "summary": "Open Source wins!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] Hot Chips: OpenAI’s Jalapeño, Cerebras CS-5, Groq 3 LPX, Apple M6",
   "url": "https://www.latent.space/p/ainews-hot-chips-openais-jalapeno",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-27T01:31:22.000Z",
   "summary": "The conference with hot chips and even hotter companies",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Ling 3.0 Flash Fin now available on AI Gateway for free",
   "url": "https://vercel.com/changelog/ling-3-0-flash-fin-now-available-on-ai-gateway-for-free",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-27T00:00:00.000Z",
   "summary": "Ling 3.0 Flash Fin from Inclusion AI is now available on AI Gateway and free to use through September 25. Ling 3.0 Flash Fin is a finance-focused version of Ling 3.0 Flash . It has a 256K-token context window, produces up to 32K output tokens, and supports reasoning and function calling. The model is designed for financial research and analysis, including multi-step workflows that user multiple tool calls before producing an answer. Choose a model ID based on what should happen after the free period: Continue after September 25: Use inclusionai/ling-3.0-flash-fin . Requests are free during the",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to Choose an Open Source LLM: A Decision Framework",
   "url": "https://www.openhands.dev/blog/how-to-choose-open-source-llm",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-27T00:00:00.000Z",
   "summary": "Choosing an open source LLM means weighing license terms, hardware limits, and your real workload before you ship.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Linear for Project Management: Setup, Workflows, and Best Practices",
   "url": "https://www.openhands.dev/blog/linear-for-project-management",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-27T00:00:00.000Z",
   "summary": "A Linear project management guide covering setup, cycles, triage rules, and automating tickets into merged pull requests",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ROCm 10.0: A Decade of Open Compute, Built for the Age of Agentic AI",
   "url": "https://rocm.blogs.amd.com/ecosystems-and-partners/rocm-x-blog/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-27T00:00:00.000Z",
   "summary": "AMD shipped ROCm 1.0 in April 2016: an open-source GPU compute stack built around a C++ compiler and a GPU programming language called HIP, aimed at high-performance computing. A decade later, the same platform trains and serves frontier AI models across industries and enables nearly every compute application benefiting from GPUs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Qwen3.8-Flash-Next",
   "url": "https://simonwillison.net/2026/Aug/26/qwen38-flash-next",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-26T23:52:58.000Z",
   "summary": "Qwen3.8-Flash-Next Another open weights model from Qwen. This one is \"a multimodal MoE model that also serves as an early preview of the architecture used in Qwen4\". It's pretty big: 125B tokens, but only 6B active which means it gets a significant performance boost. I've been trying it out on a DGX Spark using these Unsloth quantized models . I'm still exploring the model - so far I've tried the 72.5GB UD-IQ1_S one (producing these pelicans ) and the 78.9GB UD-Q2_K_XL (producing these ). My favorite so far was this xhigh reasoning effort one from UD-Q2_K_XL: Via Hacker News Tags: ai , generat",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "NVIDIA NVLink Fusion Brings NVHBM to Next-Generation AI Infrastructure",
   "url": "https://developer.nvidia.com/blog/nvidia-nvlink-fusion-brings-nvhbm-to-next-generation-ai-infrastructure",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-26T21:06:58.000Z",
   "summary": "AI factories must support increasingly large models and more complex reasoning workloads. To keep up with the insatiable compute demands of AI workloads,...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GitHub Copilot app for Beginners: Automate Dependabot pull request triage",
   "url": "https://github.blog/ai-and-ml/github-copilot/github-copilot-app-for-beginners-automate-dependabot-pull-request-triage",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T20:12:53.000Z",
   "summary": "Managing library updates can be tedious at times. Learn how the GitHub Copilot app can handle this type of repetitive task. The post GitHub Copilot app for Beginners: Automate Dependabot pull request triage appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to Train a Cross-Embodiment Robot Navigation Policy with AI Agents",
   "url": "https://developer.nvidia.com/blog/how-to-train-a-cross-embodiment-robot-navigation-policy-with-ai-agents",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-26T20:05:06.000Z",
   "summary": "Navigation enables a robot to turn perception and motion into purposeful autonomy. Unlike locomotion, which produces stable movement, navigation must be used to...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Evaluate any agent framework with Amazon Bedrock AgentCore Evaluations",
   "url": "https://aws.amazon.com/blogs/machine-learning/evaluate-any-agent-framework-with-amazon-bedrock-agentcore-evaluations",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-26T19:13:35.000Z",
   "summary": "Amazon Bedrock AgentCore Evaluations decouples agent evaluation from the framework you build on. As long as your agent emits OpenTelemetry telemetry, the service can score it, whether you use LangGraph, LlamaIndex, the OpenAI Agents SDK, Google ADK, the Claude Agent SDK, or Strands Agents. This post explains how the framework-agnostic contract works.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Experiment with Qwen3.8-Flash-Next on NVIDIA GB300 NVL72 for Agentic Coding",
   "url": "https://developer.nvidia.com/blog/experiment-with-qwen3-8-flash-next-on-nvidia-gb300-nvl72-for-agentic-coding",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-26T17:07:12.000Z",
   "summary": "Alibaba released the model weights for Qwen3.8-Flash-Next as a preview of the upcoming Qwen4 architecture for developers to experiment with and evaluate. It’s...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LangSmith LLM Gateway: Runtime Controls for Agents",
   "url": "https://www.langchain.com/blog/langsmith-llm-gateway-runtime-controls-for-production-agents",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "LangSmith LLM Gateway is in public beta: spend caps, rate limits, model fallbacks and PII redaction for production agents, without provider lock-in.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How LangSmith and LangChain OSS Help You Meet EU AI Act Requirements",
   "url": "https://www.langchain.com/blog/langsmith-langchain-oss-eu-ai-act",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "The EU AI Act compliance deadline is August 2, 2026. Learn what the EU AI Act requires, and how LangSmith and LangChain OSS products help you meet each requirement.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Factory used LangSmith to automate their feedback loop and improve iteration speed by 2x",
   "url": "https://www.langchain.com/blog/customers-factory",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "How Factory AI uses LangSmith to debug issues and close the product feedback loop, resulting in a 2x improvement in iteration speed.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Aligning LLM-as-a-Judge with Human Preferences",
   "url": "https://www.langchain.com/blog/aligning-llm-as-a-judge-with-human-preferences",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "Deep dive into self-improving evaluators in LangSmith, motivated by the rise of LLM-as-a-Judge evaluators plus research on few-shot learning and aligning human preferences.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing LangGraph v0.1 & LangGraph Cloud: Running agents at scale, reliably",
   "url": "https://www.langchain.com/blog/langgraph-cloud",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "Our new infrastructure for running agents at scale, LangGraph Cloud, is available in beta. We also have a new stable release of LangGraph.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Podium optimized agent behavior and reduced engineering intervention by 90% with LangSmith",
   "url": "https://www.langchain.com/blog/customers-podium",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "See how Podium tests across the lifecycle development of their AI employee agent, using LangSmith for dataset curation and finetuning. They improved agent F1 response quality to 98% and reduced the need for engineering intervention by 90%.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Pushing LangSmith to new limits with Replit Agent's complex workflows",
   "url": "https://www.langchain.com/blog/customers-replit",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "See how Replit built their agents atop LangGraph and integrated LangSmith to pinpoint issues, improve the performance of their agents, and enable human-in-the-loop workflows.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LangChain's Second Birthday",
   "url": "https://www.langchain.com/blog/langchain-second-birthday",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "Reflections on how LangChain has evolved — including our products, ecosystem, and community — over the past two years, and where we're headed next.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LangSmith: Redesigned product homepage and Resource Tags for better organization",
   "url": "https://www.langchain.com/blog/langsmith-homepage-redesign-and-resource-tags",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "LangSmith's homepage is now organized into Observability, Evaluation, and Prompt Engineering. Learn why we organized the homepage like this. Plus, see our latest Resource Tags updates.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LangChain State of AI 2024 Report",
   "url": "https://www.langchain.com/blog/langchain-state-of-ai-2024",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "Dive into LangSmith product usage patterns that show how the AI ecosystem and the way people are building LLM apps is evolving.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Interrupt: The AI Agent Conference by LangChain",
   "url": "https://www.langchain.com/blog/introducing-interrupt-langchain-conference",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "Join us this May at Interrupt, LangChain’s inaugural conference where the future of AI agents takes center stage.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Reflections on Three Years of Building LangChain",
   "url": "https://www.langchain.com/blog/three-years-langchain",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "Harrison Chase on LangChain's 3-year journey from open source to $1.25B company, announcing LangChain 1.0, LangSmith expansion, and $125M funding.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Why We Rebuilt LangChain’s Chatbot and What We Learned",
   "url": "https://www.langchain.com/blog/rebuilding-chat-langchain",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "Learn how LangChain rebuilt their chatbot using Deep Agents and subgraphs for sub-15-second responses with precise citations—and what you can apply.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "January 2026: LangChain Newsletter",
   "url": "https://www.langchain.com/blog/january-2026-langchain-newsletter",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:53:53.000Z",
   "summary": "Read about the latest product updates, events, and content from the LangChain team",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How GoDaddy transformed its analytics with Amazon Quick",
   "url": "https://aws.amazon.com/blogs/machine-learning/how-godaddy-transformed-its-analytics-with-amazon-quick",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-26T16:51:53.000Z",
   "summary": "In this post, you will learn how GoDaddy migrated from their legacy business intelligence (BI) tool to Amazon Quick. This was a two-year transformation that delivered results across every dimension of the business: 15,000 hours saved annually, 50% reduction in dashboard count, rendering times cut to under 5 seconds, and AI-powered self-service analytics now accessible to every employee.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Natera’s intelligent appointment scheduling with Amazon Bedrock AgentCore",
   "url": "https://aws.amazon.com/blogs/machine-learning/nateras-intelligent-appointment-scheduling-with-amazon-bedrock-agentcore",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-26T16:36:35.000Z",
   "summary": "Learn how Natera built an automated voice agent on Amazon Bedrock AgentCore that lets patients book mobile phlebotomy appointments through natural conversation. The post covers the dual-WebSocket bridge, event-driven latency masking, and progressive-trust authentication behind 100% tool-calling accuracy and sub-7-second latency.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Bring your own model with Amazon SageMaker AI: Script mode in SDK v3",
   "url": "https://aws.amazon.com/blogs/machine-learning/bring-your-own-model-with-amazon-sagemaker-ai-script-mode-in-sdk-v3",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-26T16:31:32.000Z",
   "summary": "The SageMaker Python SDK v3 redesigns script mode with unified ModelTrainer and ModelBuilder classes. This post walks through two end-to-end examples, a scikit-learn Random Forest and a multi-GPU Stable Diffusion 3.5 LoRA fine-tune, showing how SourceCode syncs your local code into any container at runtime so you can iterate without rebuilding Docker images.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Preparing data for supervised fine-tuning Part 2: Advanced data strategies",
   "url": "https://aws.amazon.com/blogs/machine-learning/preparing-data-for-supervised-fine-tuning-part-2-advanced-data-strategies",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-26T16:24:05.000Z",
   "summary": "The advanced side of supervised fine-tuning data prep. This second post in a two-part series covers evaluating data readiness with learning curves, selecting high-value data subsets, augmenting data with synthetic and distilled examples, and mixing data sources to prevent catastrophic forgetting.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Preparing data for supervised fine-tuning Part 1: Formatting and quality",
   "url": "https://aws.amazon.com/blogs/machine-learning/preparing-data-for-supervised-fine-tuning-part-1-formatting-and-quality",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-26T16:24:02.000Z",
   "summary": "Data preparation determines the ceiling of any supervised fine-tuning project. This first post in a two-part series covers the foundations of SFT data prep: quality checks, conversational (JSONL) formatting, reasoning and tool-calling schemas, and a representative train/evaluation split.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Lovable CTO: The Future of SaaS Is Apps That Agents Can Use",
   "url": "https://www.latent.space/p/lovable-future-of-saas",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-26T16:16:25.000Z",
   "summary": "Lovable is branching out from AI-powered web app creation and into MCP-powered ‘capabilities’. We talk to CTO Fabian Hedin.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Evaluating OpenWiki with WikiBench",
   "url": "https://www.langchain.com/blog/evaluating-openwiki-with-wikibench",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:15:01.000Z",
   "summary": "We built WikiBench to test whether generated wikis help coding agents. Pairing a wiki with source code scored higher than source alone, at lower cost.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LangChain and NVIDIA Launch NemoClaw Deep Agents Blueprint",
   "url": "https://www.langchain.com/blog/langchain-and-nvidia-launch-the-nemoclaw-deep-agents-blueprint",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:15:01.000Z",
   "summary": "LangChain and NVIDIA launch the NemoClaw Deep Agents blueprint, combining Deep Agents Code, Nemotron 3 Ultra, and OpenShell for open, governed enterprise agents.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AI Agent Latency 101: How do I speed up my AI agent?",
   "url": "https://www.langchain.com/blog/how-do-i-speed-up-my-agent",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:15:01.000Z",
   "summary": "Learn proven strategies to speed up your AI agent: reduce latency, optimize LLM calls, enable parallelism, and improve UX. Expert tips from LangChain.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LangGraph Platform is now Generally Available: Deploy & manage long-running, stateful Agents",
   "url": "https://www.langchain.com/blog/langgraph-platform-ga",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:15:01.000Z",
   "summary": "LangGraph Platform, our infrastructure for deploying and managing agents at scale, is now generally available. Learn how to deploy",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LangChain raises $125M to build the platform for agent engineering",
   "url": "https://www.langchain.com/blog/series-b",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:15:01.000Z",
   "summary": "We raised $125M at a $1.25B valuation to build the platform for agent engineering.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LangChain Announces Enterprise Agentic AI Platform Built with NVIDIA",
   "url": "https://www.langchain.com/blog/nvidia-enterprise",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T16:15:01.000Z",
   "summary": "Build, deploy, and monitor production-grade AI agents at scale with LangChain's enterprise agentic AI platform integrated with NVIDIA.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Aperture GA: Building a home(lab) for agentic AI",
   "url": "https://tailscale.com/blog/aperture-ga",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-08-26T16:00:00.000Z",
   "summary": "Adding model tokens, chat Projects, MCP controls, and more.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Connect Amazon Bedrock AgentCore to cross-account knowledge bases",
   "url": "https://aws.amazon.com/blogs/machine-learning/connect-amazon-bedrock-agentcore-to-cross-account-knowledge-bases",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-26T15:48:10.000Z",
   "summary": "Learn how Amazon Bedrock AgentCore agents in one account can generate answers from an Amazon Bedrock knowledge base backed by Amazon Redshift Serverless in another account, without copying source data. This post covers the architecture, security boundary, and two orchestration models: a code-based Strands agent and a declarative AgentCore harness.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "🔬“We have foundation models for language, not for physics” — Anima Anandkumar, Bren Professor of Computing",
   "url": "https://www.latent.space/p/anima",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-26T15:15:39.000Z",
   "summary": "Anima Anandkumar has spent two decades in AI, from classical math to deep learning and back. Now she's using it to model the physical world, from weather to fusion reactors.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Proving Agentic AI ROI in Financial Services | LangChain",
   "url": "https://www.langchain.com/blog/proving-the-roi-of-agentic-ai-in-financial-services",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T15:02:51.000Z",
   "summary": "How to prove agentic AI ROI in financial services: business KPIs, cost tracking and governance for RFP and AML use cases, using LangSmith and Pay-i together.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Wiki Memory: File-Based Memory for AI Agents | LangChain",
   "url": "https://www.langchain.com/blog/wiki-memory",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T15:02:51.000Z",
   "summary": "Wiki memory uses an agent to compress raw data into a persistent, file-based knowledge base. How it differs from RAG, real examples, and when to use it.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Candidly Built State-Aware Agent Harnesses in LangSmith",
   "url": "https://www.langchain.com/blog/how-candidly-built-state-aware-agent-harnesses-with-langsmith",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T15:02:51.000Z",
   "summary": "Candidly's agent Cait reads partial traces to infer user state mid-conversation and steer replies, using a LangSmith labeling pipeline at 92.3% human agreement.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Build an Auditable VC Research Agent with Perplexity",
   "url": "https://www.langchain.com/blog/build-an-auditable-vc-research-agent-with-the-perplexity-agent-api-langgraph-and-langsmith",
   "source": "LangChain",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T15:02:51.000Z",
   "summary": "Build a VC research agent that drafts a cited investment memo in about 90 seconds for $0.40, using the Perplexity Agent API, LangGraph, and LangSmith evals.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "FinOps for the AI era: New flexible billing and cost controls for agents",
   "url": "https://cloud.google.com/blog/products/ai-machine-learning/flexible-billing-and-cost-controls-for-agents-on-google-cloud",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-26T13:30:00.000Z",
   "summary": "Editor's note: A product image was updated after initial publication. As AI takes on more complex work, business leaders face a new challenge: enabling rapid innovation using agents while protecting their margins and budgets. To get a real return on AI, financial operations (FinOps) and cost management must evolve alongside technology, giving you clear visibility, proactive cost controls, and flexible payment models that fit your needs. That’s why today we’re introducing expanded billing flexibility and new cost management tools for agent workloads across Gemini Enterprise and developer tools ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Dynamic capacity management for AI infrastructure",
   "url": "https://cloud.google.com/blog/topics/ai-infrastructure/best-practices-for-dynamic-capacity-management",
   "source": "Google Cloud Infrastructure",
   "group": "Large-scale production systems",
   "published": "2026-08-26T13:30:00.000Z",
   "summary": "The internet connected billions of people and mobile devices, putting computers in every hand. Now, we’re in the middle of the next big technology shift, deploying millions of autonomous AI agents to work alongside employees and end users. Today, we announced new FinOps controls for Gemini Enterprise to help organizations manage project-level AI spend and eliminate token shock. But the sheer scale of the agentic era is placing new constraints at every layer of the stack, including infrastructure. AI workloads are notoriously difficult to architect, resource-intensive, and bursty, which can als",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Bringing ChatGPT for Teachers to more U.S. school districts",
   "url": "https://openai.com/index/bringing-chatgpt-for-teachers-to-more-us-school-districts",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T10:00:00.000Z",
   "summary": "ChatGPT for Teachers is expanding to 55 U.S. school systems, bringing secure AI tools, training, and support to over 100,000 more educators and staff.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Learning never stops: How AI makes learning continuous",
   "url": "https://openai.com/index/learning-never-stops",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T10:00:00.000Z",
   "summary": "OpenAI’s new report explores how students and educators use ChatGPT to make learning more continuous, with support that extends beyond the classroom.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Seeing Through the Stack: End-to-End and Fine-Grained Tracing in llm-d",
   "url": "https://llm-d.ai/blog/end-to-end-and-fine-grained-tracing-in-llm-d",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-08-26T09:00:00.000Z",
   "summary": "How llm-d stitches Gateway, EPP, KV-cache, P/D proxy, and vLLM into one OpenTelemetry trace, and instruments the scheduling decisions that metrics alone cannot explain.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Quoting Paul Dix",
   "url": "https://simonwillison.net/2026/Aug/26/paul-dix",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-26T08:07:55.000Z",
   "summary": "The fact that AI wrote 1M LOC and then refined it over the course of the next couple of months to produce a reliable piece of software that is currently running on millions of developer machines is absolutely mind blowing. And you can say, “well it’s not that impressive because they had an oracle to compare against, so it was simple to go from one language to another”, but I think that’s selling this entire thing short. If you can build a verification system and give proper direction, AI can produce a highly complex, highly sophisticated piece of software and it can continue to refine it until",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Hot Chips 2026: Fujitsu’s Monaka CPU",
   "url": "https://chipsandcheese.com/p/hot-chips-2026-fujitsus-monaka-cpu",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-26T01:20:30.000Z",
   "summary": "Building on a long history of HPC-focused cores, and looking beyond HPC",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Hugging Face incident and the road ahead",
   "url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "OpenAI shares findings from the Hugging Face security incident and the steps we’re taking to strengthen AI model security, monitoring, and alignment.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How loveholidays is making everyone a builder with Codex",
   "url": "https://openai.com/index/loveholidays",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "Discover how loveholidays uses OpenAI Codex to make software development accessible across the business, helping teams turn ideas into products faster.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Muse Image now available on AI Gateway",
   "url": "https://vercel.com/changelog/muse-image-now-available-on-ai-gateway",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "Muse Image from Meta Superintelligence Labs is now available on AI Gateway. It is their first image model and a separate family from Muse Spark, returning images rather than text. Send a prompt and get an image back, or send an image with an instruction and get it changed. One model does both, so you don't switch models to move from generating to editing. To use Muse Image, set model to meta/muse-image-1.0 and call generateImage from the AI SDK : To steer the result toward art you already have, pass reference images in prompt.images alongside the text, and the model blends them into what it dr",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Gemini 3.5 Transcribe now available on AI Gateway",
   "url": "https://vercel.com/changelog/gemini-3-5-transcribe-now-available-on-ai-gateway",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "Gemini 3.5 Transcribe from Google is now available on AI Gateway for recorded and live audio: google/gemini-3.5-transcribe transcribes a complete audio file in one request. google/gemini-3.5-transcribe-live transcribes audio over a WebSocket and returns text as the audio arrives. Both models automatically detect more than 85 languages, including when a speaker switches languages. You can also provide custom vocabulary to improve the transcription of names, technical terms, and uncommon spellings. The model for complete recordings can also identify speakers and return word-level timestamps. Str",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Qwen 3.8 Flash now available on AI Gateway",
   "url": "https://vercel.com/changelog/qwen-3-8-flash-now-available-on-ai-gateway",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "Qwen 3.8 Flash from Alibaba is now available on AI Gateway. It takes text and images as input, serves a context window of 1 million tokens, and can return up to 65k tokens in a response. Alibaba recommends it for coding, tool use, and multi-step agent workflows. To use Qwen3.8-Flash, set model to alibaba/qwen3.8-flash in the AI SDK : To use it in a coding agent, see the coding agents guide , then run vercel ai-gateway coding-agents setup to connect agents like Claude Code, Codex, OpenCode, Cursor, Pi, and more and select alibaba/qwen3.8-flash inside the agent. Try Qwen3.8-Flash in the model pl",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GLM 5.3 Flash now available on AI Gateway",
   "url": "https://vercel.com/changelog/glm-5-3-flash-now-available-on-ai-gateway",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "GLM 5.3 Flash from Z.ai is now available on AI Gateway as zai/glm-5.3-flash . GLM 5.3 Flash is a multimodal coding model with a 1M-token context window. It accepts both text and image inputs and supports function calling, structured output, and streaming. To include images, pass them with text in a message. A request can include multiple images using URLs, Base64 data URLs, or binary data: To use GLM 5.3 Flash with a coding agent, run vercel ai-gateway coding-agents setup , then select zai/glm-5.3-flash as your agent's model. Try GLM 5.3 Flash in the model playground , or browse all language m",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Vercel Security Dashboard is now generally available",
   "url": "https://vercel.com/changelog/vercel-security-dashboard-is-now-generally-available",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "The Vercel Security Dashboard is now generally available on all plans, giving you one place to see your security posture across every account and project. You can access the Security Dashboard in the UI or run vercel security check in the Vercel CLI. As teams grow and coding agents make it faster to spin up projects, small misconfigurations add up quietly. The Security Dashboard automatically flags issues like: Team members without 2FA Long-lived credentials that can be replaced with OIDC Public preview deployments Non-sensitive and stale environment variables The Security Dashboard UI Misconf",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Python projects now support routing rules",
   "url": "https://vercel.com/changelog/python-projects-now-support-routing-rules",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "Python projects can now use routing rules to set response headers or rewrite requests to internal paths, including apps built with FastAPI, Django, and Flask. The Vercel CDN evaluates rules before requests reach your application, so changes apply without a new deployment. For example, this FastAPI app serves a /new route: To send requests for /old to that route, create a rewrite from the CDN tab in your project dashboard or with the Vercel CLI : Published rules take effect immediately across all regions, and you can roll back to a previous version from the History tab. You can also manage rule",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Qwen3.8-Flash-Next: Day-0 Support in SGLang",
   "url": "https://lmsys.org/blog/2026-08-26-qwen-flash-next",
   "source": "SGLang and LMSYS",
   "group": "Serving engines and runtimes",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "Today, the Qwen team open-sourced Qwen3.8-Flash-Next, a multimodal MoE model and an early preview of the Qwen4 architecture. It plays the same role for Qwen4 that Qwen3-Next played for Qwen3.5. The Ga...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Training and Finetuning Multi-Vector Embedding Models with Sentence Transformers",
   "url": "https://huggingface.co/blog/train-multi-vector-encoder",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Post-training Kimi K3 with Harvey for long-horizon legal work",
   "url": "https://fireworks.ai/blog/post-training-kimi-k3-with-harvey-for-long-horizon-legal-work",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "Harvey recently announced its first model, Tenet, post-trained in collaboration with Fireworks for long-horizon legal work.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What's new in the ClickHouse .NET Driver: the road from 1.0 to 1.3",
   "url": "https://clickhouse.com/blog/whats-new-clickhouse-net-driver-10-to-13",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-26T00:00:00.000Z",
   "summary": "Discover how ClickHouse .NET Driver 1.1–1.3 adds type-safe POCO workflows, extensible serialization, broader type support, performance improvements, and official ecosystem integrations.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "EVE Online: The Move to Python 3 Begins!",
   "url": "https://simonwillison.net/2026/Aug/25/eve-online-move-to-python-3",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-25T22:59:30.000Z",
   "summary": "EVE Online: The Move to Python 3 Begins! EVE Online has been one of the most interesting case studies in Python at scale for over twenty years now. They've been running on Stackless Python since their launch in 2003, and their last major upgrade was 16 years ago, to Stackless Python 2.7 in 2010 . Their upgrade to Python 3 will start using the futurize script against 2.4 million lines of code, followed by careful manual review of the ~20,000 places where Python 2 and 3 behavior differ - for example 1 / 2 is 0 in Python 2 but is 0.5 in Python 3. There's nothing in this announcement about how the",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to evaluate LLMs before production",
   "url": "https://github.blog/ai-and-ml/llms/how-to-evaluate-llms-before-production",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-25T21:35:11.000Z",
   "summary": "These are the lessons we learned evaluating LLMs for real-world secret scanning. The post How to evaluate LLMs before production appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Restore LLM Inference Capacity in Seconds with Shadow Engine Recovery in NVIDIA Dynamo",
   "url": "https://developer.nvidia.com/blog/restore-llm-inference-capacity-in-seconds-with-shadow-engine-recovery-in-nvidia-dynamo",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-25T20:57:54.000Z",
   "summary": "When an LLM engine process fails, the standard recovery path involves a cold restart. This requires loading weights into HBM from storage, compiling kernels,...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agentic observability with Amazon OpenSearch Service MCP Apps",
   "url": "https://aws.amazon.com/blogs/machine-learning/agentic-observability-with-amazon-opensearch-service-mcp-apps",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-25T19:00:09.000Z",
   "summary": "Amazon OpenSearch Service now supports MCP Apps, which return interactive visualizations alongside your AI agent's text responses. Learn how a single, locally run MCP server lets your agent move from alert to trace to logs to root cause in one conversation, and how you can verify every step inline without leaving your IDE.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Suprema Gaming made its data platform agent-ready with ClickHouse Cloud",
   "url": "https://clickhouse.com/blog/suprema-gaming-agent-ready-platform",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-25T18:52:53.000Z",
   "summary": "Suprema Gaming migrated its analytics platform from Snowflake to ClickHouse Cloud to power a company-wide shift toward agentic operations.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Vercel applications are protected from Next.js August 2026 security vulnerabilities",
   "url": "https://vercel.com/changelog/nextjs-august-2026-security-release",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T16:39:17.904Z",
   "summary": "Summary Two vulnerabilities affecting Next.js were disclosed in the August 2026 Security Release. Next.js applications hosted on Vercel are protected and require no customer action. Next.js August 2026 vulnerabilities Next.js disclosed the following critical vulnerabilities: GHSA-2xp9-vwfh-vxw4 originates in the upstream libheif dependency and can lead to unauthenticated remote code execution when Image Optimization processes a crafted AVIF input. CVE-2026-75604 ( GHSA-p293-qw3h-jr36 ) can lead to unauthenticated remote code execution on Windows-hosted Next.js servers in applications using the",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Governed reports with Amazon Quick Desktop and Amazon FSx for NetApp ONTAP",
   "url": "https://aws.amazon.com/blogs/machine-learning/governed-reports-with-amazon-quick-desktop-and-amazon-fsx-for-netapp-ontap",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-25T16:35:01.000Z",
   "summary": "Build a governed weekly reporting workflow with Amazon Quick Desktop and Amazon FSx for NetApp ONTAP. An Amazon S3 access point exposes an approved folder to a Quick knowledge base, and a custom skill drafts cited weekly reports and Slack summaries with human review before anything is shared.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Speed Insights now has a free tier",
   "url": "https://vercel.com/changelog/speed-insights-free-tier",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T16:00:00.000Z",
   "summary": "Speed Insights now has a free tier that gives you a high-level performance overview from your real users. The new free tier: Is available on every plan, for any number of projects Includes 10,000 events per team, every 30 days Install Speed Insights: Previously, free Speed Insights was limited to a single project on Hobby, and upgrading to Pro meant losing access unless you paid for the add-on. Now your free tier carries over, and you only pay if you upgrade. The paid product, now called Speed Insights Plus, includes deeper diagnostics and historical data. If you were already paying for Speed ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Granite 4.2 LLMs: How They're Built",
   "url": "https://huggingface.co/blog/ibm-granite/granite-4-2",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-25T15:14:14.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "CUDA Python 1.0: Stable APIs, One Foundation, Full Platform Access",
   "url": "https://developer.nvidia.com/blog/cuda-python-1-0-stable-apis-one-foundation-full-platform-access",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-25T15:00:00.000Z",
   "summary": "For years, a Python developer who needed a GPU had two realistic choices: Learn NVIDIA CUDA C++ well enough to write an extension, set up a build toolchain, and...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Read your writes: WAIT FOR in PostgreSQL 19",
   "url": "https://clickhouse.com/blog/postgresql-19-wait-for-read-your-writes",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-25T12:36:23.000Z",
   "summary": "PostgreSQL 19's new `WAIT FOR` command enables read-your-writes consistency on asynchronous replicas by letting individual reads wait for a specific WAL position.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ClickGap: Autonomous QA for ClickHouse",
   "url": "https://clickhouse.com/blog/clickgap-autonomous-qa-for-clickhouse",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-25T12:19:59.000Z",
   "summary": "ClickGap reviews merged ClickHouse changes, executes reproducers, rejects false positives, and attributes regressions to specific commits.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Now introducing Gemini Enterprise for Legal",
   "url": "https://cloud.google.com/blog/products/ai-machine-learning/introducing-gemini-enterprise-for-legal",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-25T12:00:00.000Z",
   "summary": "Few professions are as exacting as the practice of law. A team reviewing a contract or building a case works inside strictly privileged information, firm-specific playbooks, and a body of law that changes constantly. The work thrives on nuanced, professional judgment — and the systems supporting it inherit real obligations: ethical walls that cannot be crossed, matter permissions that cannot be flattened, and a duty of confidentiality that does not bend for convenience. General-purpose AI, however capable, does not meet that standard on its own. Foundational model intelligence is necessary. Fo",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Now introducing Gemini Enterprise for Financial Services",
   "url": "https://cloud.google.com/blog/products/ai-machine-learning/introducing-gemini-enterprise-for-financial-services",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-25T12:00:00.000Z",
   "summary": "Protecting capital in today's markets requires immense speed and precision. A financial analyst preparing a deal memo works across licensed market data, internal models, and confidential client files. General-purpose AI lacks the real-time accuracy, verifiable data lineage, and strict security that financial institutions demand. While model intelligence is necessary, without deep integration into trusted financial systems, it is not sufficient. Making AI genuinely useful inside an industry requires four things, together: domain expertise encoded into reusable skills, secure connections to the ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Quantization-Aware Healing: a compressed, 4-bit model that outperforms its full-precision original",
   "url": "https://huggingface.co/blog/MultiverseComputingCAI/quantization-aware-healing",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-25T11:39:24.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling Ray for AI workloads to 10k node clusters",
   "url": "https://anyscale.com/blog/how-we-scaled-ray-from-batch-inference-to-10000-node-training-clusters",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Optimizing LLM Serving Efficiency: Moving Beyond KV Cache Reuse to Token-Load Awareness with Ray Serve LLM",
   "url": "https://anyscale.com/blog/llm-kv-token-aware-routing",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Ray History Server: Post-Mortem Observability for Ray on Kubernetes",
   "url": "https://anyscale.com/blog/ray-history-server",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Learning Loops: The Path to Owning Your Intelligence",
   "url": "https://anyscale.com/blog/learning-loops",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "FP8 Reinforcement Learning in SkyRL: Preserving Policy Consistency Across Training and Rollout",
   "url": "https://anyscale.com/blog/fp8-reinfinforcement-learning-in-skyrl",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GPU-Native Operators in Ray Data",
   "url": "https://anyscale.com/blog/gpu-native-operators-in-ray-data",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Anyscale GPU Health Observability: From app to hardware",
   "url": "https://anyscale.com/blog/anyscale-gpu-health-observability",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing Native Sandboxing in Ray",
   "url": "https://anyscale.com/blog/announcing-native-sandboxing-in-ray",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing KubeRay v1.6 and v1.7",
   "url": "https://anyscale.com/blog/kuberay-v1-7",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Anyscale KubeRay Connect: Doubling down on Kubernetes",
   "url": "https://anyscale.com/blog/announcing-anyscale-connect-for-kuberay",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Shuffle V2 in Ray Data: Faster, Fault-Tolerant Joins and Aggregations",
   "url": "https://anyscale.com/blog/ray-data-shuffle-v2",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-25T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The full stack behind abundant intelligence",
   "url": "https://openai.com/index/the-full-stack-behind-abundant-intelligence",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-25T07:05:00.000Z",
   "summary": "OpenAI CFO Sarah Friar explains how advances across chips, compute, models, and products compound to deliver more useful intelligence at greater scale and lower cost.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Jalapeño’s first results show industry-leading speed and efficiency in AI inference",
   "url": "https://openai.com/index/jalapeno-first-results",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-25T07:00:00.000Z",
   "summary": "Jalapeño is a custom inference chip from OpenAI that delivers faster, more power-efficient AI inference, with higher throughput and lower latency for modern models.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Hot Chips 2026: Intel’s Diamond Rapids",
   "url": "https://chipsandcheese.com/p/hot-chips-2026-intels-diamond-rapids",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-25T06:20:03.000Z",
   "summary": "Hello you fine Internet folks,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Run SDK: secure eval for your agents",
   "url": "https://vercel.com/blog/introducing-run",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T04:00:00.000Z",
   "summary": "Agents increasingly write TypeScript programs to coordinate tools and process their results. Once those programs touch real applications, some steps require authentication, while others need human approval. Executing that code with eval gives it the same access as the application around it, including its secrets and internal services, and leaves no durable way to pause at those boundaries. Today, we're releasing the Run SDK , a package for executing untrusted JavaScript and TypeScript without giving it direct access to your application or system. Applications expose narrow host functions and c",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The end of credential sprawl for agents",
   "url": "https://vercel.com/blog/the-end-of-credential-sprawl-for-agents",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T04:00:00.000Z",
   "summary": "Every useful agent reaches beyond your codebase. It posts to Slack, opens pull requests, queries Snowflake, or calls an internal API. That reach is what makes it valuable, and it's also where the risk lives, because for years, granting it meant provisioning a long-lived token and hoping it never leaked. Vercel Connect replaces long-lived tokens with ones your code requests at runtime, scoped to the task and expiring on their own. During the public beta , we've grown the ecosystem past 100 connectors, unified how they work, and added the governance capabilities teams need in production. Today, ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] Andrew Ng gets into AI Engineering",
   "url": "https://www.latent.space/p/ainews-andrew-ng-gets-into-ai-engineering",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-25T02:50:57.000Z",
   "summary": "An industry legend starts covering the inevitable!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "SiFive’s First Server Platform",
   "url": "https://chipsandcheese.com/p/sifives-first-server-platform",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-25T01:48:39.000Z",
   "summary": "Hello you fine Internet folks,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Disrupting a new covert influence campaign from Russia",
   "url": "https://openai.com/index/disrupting-malicious-uses-of-ai-influence-campaign-russia",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "OpenAI banned Russia-origin accounts using AI to promote a fake Israel-based think tank and a “sovereignty” index praising Russia and criticizing the West.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing the Admin plugin for ChatGPT Work and Codex",
   "url": "https://openai.com/index/introducing-admin-plugin",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Use the Admin plugin for ChatGPT Work and Codex to analyze workspace usage, manage members and permissions, adjust limits, and act on admin requests.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "MiniMax M3 and M2.7 are free on AI Gateway",
   "url": "https://vercel.com/changelog/minimax-m3-and-m2-7-are-free-on-ai-gateway",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "MiniMax M3 and M2.7 are free on AI Gateway via GMI Cloud through Sunday, September 6. Use minimax/minimax-m3-free or minimax/minimax-m2.7-free to route requests to GMI Cloud. These model IDs will return an error after the free period ends. To keep requests working after the free period, use the standard model ID without the -free suffix and place GMI Cloud first in the provider order: AI Gateway tries GMI Cloud first and can fall back to another provider if GMI Cloud can't serve the request. This lets the same code continue working after the free period. Outside the free GMI Cloud route, reque",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Wan 3.0 now available on AI Gateway",
   "url": "https://vercel.com/changelog/wan-3-0-now-available-on-ai-gateway",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Wan 3.0 from Alibaba is now available on AI Gateway as alibaba/wan-v3.0-video . Wan 3.0 combines text-to-video, image-to-video, first- and last-frame conditioning, and reference-based generation in one model. References can include images, video, and audio. It generates clips up to 30 seconds at 30 fps in 480p, 720p, or 1080p, with synchronized audio. Previously, Wan 2.7 required separate -t2v and -r2v model IDs and was limited to 15-second clips at 24 fps. Generate a video Wan 3.0 supports asynchronous generation , so no HTTP request needs to remain open for the entire render. Pass a webhook ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AI Gateway now supports asynchronous video generation",
   "url": "https://vercel.com/changelog/ai-gateway-now-supports-asynchronous-video-generation",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Video generation on AI Gateway can now run asynchronously. By default, generateVideo keeps one HTTP request to AI Gateway open until the result is ready. Because video generation can take seconds or minutes, that request can exceed request timeouts. With asynchronous generation, your application can receive a webhook, poll for completion, or start a generation and retrieve the result in a later request. Choose an option based on whether your process can keep running and whether your application can receive webhooks: Existing generateVideo calls continue to work as before. All four options supp",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Vercel Connect is now generally available",
   "url": "https://vercel.com/changelog/vercel-connect-ga",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Vercel Connect is now generally available on all plans and in v0 . Instead of storing long-lived provider secrets, your code requests short-lived, scoped tokens at runtime. Deployments authenticate with their existing Vercel OIDC identity. Each token is scoped to the task, refreshed automatically, and expires on its own. Any service with one command Register a connector once from the CLI . Pass the service name and the CLI pre-populates the brand name, icon, auth type, and MCP or discovery URL, then prompts for any credentials the service requires: Connect ships with 100+ preset connectors for",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Chat SDK now supports Slack Enterprise Grid",
   "url": "https://vercel.com/changelog/chat-sdk-slack-enterprise-grid",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Chat SDK's Slack adapter now supports Slack Enterprise Grid . Bots installed org-wide work across every workspace, with correct token resolution, tenant-scoped caches, and event retry deduplication. The adapter now stores org-wide installations by enterprise ID. This matches how tokens are resolved for incoming events, slash commands, and interactive payloads. SlackInstallation records the new identity fields: Token resolution behaves the same over HTTP webhooks and Socket Mode. Events route by the installation identity in the envelope's authorizations field, which keeps routing correct for Sl",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Vercel Connect now supports Linq",
   "url": "https://vercel.com/changelog/vercel-connect-now-supports-linq",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Vercel Connect now includes a managed connector for Linq , so your apps and agents can send and receive messages over iMessage, RCS, and SMS. As a Vercel Managed Connector , Vercel can create a Linq account and phone number for you, or link an existing account. You never manage credentials yourself. Create a connector from the dashboard or Vercel CLI : Give your eve agent a phone number The connector powers the new Linq channel in eve . Run eve add channel/linq , choose Vercel Connect, and eve wires up the connector, phone numbers, and webhook for you: The channel marks accepted messages as re",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Bring your agent to Notion with Chat SDK",
   "url": "https://vercel.com/changelog/notion-chat-sdk",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Your team already works in Notion. Now your agent can too. With the new Notion adapter for Chat SDK, the same agent you run on Slack, Discord, GitHub, Teams, or WhatsApp can join comment discussions on your Notion pages, no separate codebase required. Each Notion page maps to a channel and each comment thread to a thread, so replies stay threaded automatically. The adapter supports mentions, message editing, conversation history, and up to three file attachments. By default, your bot replies when @-mentioned and where mentions aren't available in a workspace, it can trigger on a keyword or on ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Chat SDK now supports XChat",
   "url": "https://vercel.com/changelog/chat-sdk-now-supports-xchat",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "You can now build bots that hold end-to-end encrypted 1:1 and group conversations on XChat with the new XChat adapter for Chat SDK. The adapter handles all encryption, key management, and signature verification automatically. Bots can also message users first, as long as the user has encrypted chat set up and follows the bot. XChat has no markdown rendering, so the adapter falls back automatically: URLs and @mentions render as tappable links, tables as ASCII code blocks, and cards as text with a link preview. Streaming works through message edits. Read the XChat adapter documentation to get st",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What Is AI Agent Observability? How to Trace, Govern, and Control Agents at Scale",
   "url": "https://www.openhands.dev/blog/ai-agent-observability",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Platform and security teams can't govern what they can't see. A 2026 guide to tracing agent reasoning, tool calls, outcomes, and cost at enterprise scale.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What Is an Incident Commander? Role, Responsibilities, and When to Assign One",
   "url": "https://www.openhands.dev/blog/what-is-an-incident-commander",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "What is an incident commander? Learn the IC's role, responsibilities, and how automated triage speeds up incident response.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Wire It, Run It, Deploy It: AI Workflows in Gradio",
   "url": "https://huggingface.co/blog/gradio-workflow-guide",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Enabling Physical AI Agents with Lemonade",
   "url": "https://rocm.blogs.amd.com/ecosystems-and-partners/rai-lemonade-agents/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "In this blog we will demonstrate how to run local agents enabled by VLMs (Vision-Language Models) hosted by the Lemonade framework in the domain of robot control. These models allowed us to run an interactive robotic arm manipulation simulation entirely locally, showing the effectiveness of Lemonade for Physical AI.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Push-button migration from Confluent to Redpanda with Shadowing",
   "url": "https://www.redpanda.com/blog/migrate-confluent-redpanda-shadowing",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-08-25T00:00:00.000Z",
   "summary": "Migrate off Confluent without the “big cutover weekend.” Redpanda Shadowing carries topic data, schemas, offsets, and ACLs on a single link. Available on Self-Managed , BYOC, and Dedicated.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Your alt text passes automated checks. That doesn’t mean it’s any good.",
   "url": "https://github.blog/engineering/user-experience/your-alt-text-passes-automated-checks-that-doesnt-mean-its-any-good",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-24T20:56:32.000Z",
   "summary": "We built a plugin for the GitHub Accessibility Scanner to make sure your alt text is actually accessible. Here's how it works. The post Your alt text passes automated checks. That doesn’t mean it’s any good. appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Hot Chips 2026: Intel’s Wildcat Lake",
   "url": "https://chipsandcheese.com/p/hot-chips-2026-intels-wildcat-lake",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-24T19:50:30.000Z",
   "summary": "All about making things affordable",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing new Ray capabilities on SageMaker HyperPod",
   "url": "https://aws.amazon.com/blogs/machine-learning/introducing-new-ray-capabilities-on-sagemaker-hyperpod",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-24T19:32:14.000Z",
   "summary": "Amazon SageMaker HyperPod now offers managed Ray support on Amazon EKS. Create and monitor Ray clusters, connect JupyterLab and Code Editor notebooks to live clusters, get out-of-the-box observability, and run resilient distributed training and accelerated inference from SageMaker Studio, all with open-source KubeRay and standard Ray APIs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Cloudflare Blog – Brought to you by EmDash",
   "url": "https://blog.cloudflare.com/cloudflare-blog-uses-emdash",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-24T19:00:00.000Z",
   "summary": "We migrated the Cloudflare Blog to EmDash to prove our stack at massive scale. Here is how we stress-tested performance, safely routed production traffic, and redesigned the frontend experience.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Democratizing institutional knowledge: Building an AI-powered knowledge management system with AWS",
   "url": "https://aws.amazon.com/blogs/machine-learning/democratizing-institutional-knowledge-building-an-ai-powered-knowledge-management-system-with-aws",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-24T18:59:15.000Z",
   "summary": "Learn how to build a customizable, smart-caching knowledge management system on AWS that captures and delivers institutional (tribal) knowledge through a voice-first AI avatar. The accelerator uses Amazon Bedrock Knowledge Bases for retrieval-augmented generation and deploys in hours with AWS CloudFormation.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "MetaRoCE: A New RDMA Transport Built for AI-Scale Ethernet",
   "url": "https://engineering.fb.com/2026/08/24/networking-traffic/metaroce-rdma-transport-ai-ethernet",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-24T18:02:29.000Z",
   "summary": "Training and serving frontier AI models depends on fast, reliable networks that move data between GPUs without wasting compute cycles. To meet this challenge at scale, Meta designed MetaRoCE – a clean-sheet RDMA transport protocol purpose-built for AI workloads on commodity Ethernet. We’re releasing the MetaRoCE specification, a reference software implementation and a compliance test [...] Read More... The post MetaRoCE: A New RDMA Transport Built for AI-Scale Ethernet appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "MTIA 300: Meta’s First Training Chip with Built-in NICs and Communication-Offloading Engines",
   "url": "https://engineering.fb.com/2026/08/24/networking-traffic/mtia-300-meta-training-chip-built-in-nics",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-24T17:45:52.000Z",
   "summary": "MTIA 300 is the first of Meta’s family of in-house training and inference accelerators optimized for training ranking and recommendation models. We’re sharing how MTIA 300’s built-in NIC chiplets allow it to meet the communication needs associated with training recommendation models with superior performance over general-purpose GPUs. By co-designing MTIA’s communication library, HCCL, alongside the [...] Read More... The post MTIA 300: Meta’s First Training Chip with Built-in NICs and Communication-Offloading Engines appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Databricks Uses AI to Accelerate Incident Investigation",
   "url": "https://www.databricks.com/blog/how-databricks-uses-ai-accelerate-incident-investigation",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-24T17:00:00.000Z",
   "summary": "In our previous blog post, we shared how Databricks uses AI to debug thousands of...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Hot Chips 2026: CUDA Targets RISC-V",
   "url": "https://chipsandcheese.com/p/hot-chips-2026-cuda-targets-risc",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-24T16:49:25.000Z",
   "summary": "CUDA is the most important software framework in the GPU compute world, and Nvidia is looking at supporting CUDA on RISC-V. Terms and conditions may apply.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "llm-anthropic 0.27",
   "url": "https://simonwillison.net/2026/Aug/24/llm-anthropic",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-24T16:27:04.000Z",
   "summary": "Release: llm-anthropic 0.27 This release of the Anthropic plugin for LLM mainly provides compatibility with the recently released anthropic v1.0.0 Python library, which switches from httpx to httpx2 . OpenAI made the same change in their v3.0.0 release two weeks ago. Anthropic provide this migration guide for upgrading to 1.0, so I prompted Fable 5 in Claude Code with: Upgrade to anthropic>=1 - read https://raw.githubusercontent.com/anthropics/anthropic-sdk-python/refs/heads/main/MIGRATION.md and get the tests passing Here's the resulting PR . Tags: python , httpx , llm , anthropic , claude",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agentic Resource Discovery (ARD): An open specification for agent discovery",
   "url": "https://aws.amazon.com/blogs/machine-learning/agentic-resource-discovery-ard-an-open-specification-for-agent-discovery",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-24T16:22:03.000Z",
   "summary": "AWS Agent Registry gives your organization a centralized, searchable catalog for agents, tools, and skills. It works with the open Agentic Resource Discovery (ARD) standard to enable cross-environment discovery and governance at scale.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building a restaurant telephony AI host with Amazon Connect",
   "url": "https://aws.amazon.com/blogs/machine-learning/building-a-restaurant-telephony-ai-host-with-amazon-connect",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-24T16:13:09.000Z",
   "summary": "Learn how to build a voice ordering system for restaurants that answers a phone call and takes an order end to end, with no app, no website, and no sign-in. It uses Amazon Connect for telephony, Amazon Connect Agentic Voice for real-time speech, an Amazon Connect AI agent for reasoning, and Amazon Bedrock AgentCore Gateway to reach backend tools through MCP.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Elastic build machines now use Turborepo cache hits to prevent downgrades",
   "url": "https://vercel.com/changelog/elastic-build-machines-now-use-turborepo-cache-hits-to-prevent-downgrades",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-24T16:11:00.000Z",
   "summary": "Elastic build machines now consider Turborepo cache hits when deciding whether to use a smaller build machine. A warm-cache build no longer triggers a downgrade. A warm-cache build can use less CPU and memory than the same build with a cold cache. Downgrading based on that lower usage could leave a later cold-cache build without enough resources to complete successfully. This change applies automatically to all builds using Elastic build machines. No action is required. Learn more in the build documentation . Read more",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AI-powered metadata correction and harmonization",
   "url": "https://aws.amazon.com/blogs/machine-learning/ai-powered-metadata-correction-and-harmonization",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-24T15:53:03.000Z",
   "summary": "Metadata harmonization (standardizing labels, identifiers, and formats so datasets can work together) is still largely manual. This post shows how AI-powered metadata correction works in practice, covering two approaches, human-in-the-loop validation and autonomous agent-driven workflows, plus governance considerations for production deployment.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Giga-Scale AI and the Ethernet Evolution: How Spectrum-X Ethernet Rewrites the Rules",
   "url": "https://developer.nvidia.com/blog/giga-scale-ai-ethernet-evolution-spectrum-x-ethernet-rewrites-rules",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-24T15:08:39.000Z",
   "summary": "The massive growth of generative AI has fundamentally altered data center design. As distributed model training scales to span hundreds of thousands of GPUs,...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "NVIDIA Vera Rubin and Blackwell Set a New Standard for Agentic AI Performance per Watt",
   "url": "https://developer.nvidia.com/blog/nvidia-vera-rubin-and-blackwell-set-a-new-standard-for-agentic-ai-performance-per-watt",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-24T15:00:05.000Z",
   "summary": "AI agents have expanded inference from single-turn interactions into multi-step workflows that reason, invoke tools, coordinate subagents, and carry growing...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "NVIDIA BlueField-4 Powers New Scale-In Network Infrastructure for Agentic AI Factories",
   "url": "https://developer.nvidia.com/blog/nvidia-bluefield-4-powers-new-scale-in-network-infrastructure-for-agentic-ai-factories",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-24T15:00:00.000Z",
   "summary": "Traditional cloud infrastructure was designed for predictable, general-purpose workloads and standard interfaces. Agentic AI factories connect diverse users,...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Solving Agentic AI Fleet Challenges with NVIDIA Vera CPU",
   "url": "https://developer.nvidia.com/blog/solving-agentic-ai-fleet-challenges-with-nvidia-vera-cpu",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-24T15:00:00.000Z",
   "summary": "AI factories are interconnected systems where fleet economics depend on how efficiently the entire stack converts power and capital into completed agent tasks....",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How NVIDIA Groq 3 LPX Unlocks Ultrafast Interactivity at Long Context on NVIDIA Vera Rubin",
   "url": "https://developer.nvidia.com/blog/how-nvidia-groq-3-lpx-unlocks-ultrafast-interactivity-at-long-context-on-nvidia-vera-rubin",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-24T15:00:00.000Z",
   "summary": "NVIDIA Groq 3 LPX is the interactive AI inference accelerator for the NVIDIA Vera Rubin platform. At the core of the platform is NVIDIA Vera Rubin NVL72, the...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Maximizing AI Factory Performance per Watt with NVIDIA DSX MaxLPS",
   "url": "https://developer.nvidia.com/blog/maximizing-ai-factory-performance-per-watt-with-nvidia-dsx-maxlps",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-24T15:00:00.000Z",
   "summary": "AI factories are power-constrained industrial systems. The question is no longer how many GPUs fit in a data center, but how much AI output each available...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The TailscaleUp-date",
   "url": "https://tailscale.com/blog/tailscaleup-2026-product-update",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-08-24T14:30:00.000Z",
   "summary": "A preview of what we’re building—and what we’ll share at TailscaleUp.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Sprig Replaced Postgres, ClickHouse & Redis…with 4-8x Better Latency",
   "url": "https://www.scylladb.com/2026/08/24/sprig-replaced-postgres-clickhouse-redis-4-8x-better-latency",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-08-24T13:30:56.000Z",
   "summary": "With ScyllaDB, a small engineering team could focus on building their product instead of battling their databases.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Advancing price-performance for developers with GPT‑5.6 in Kiro",
   "url": "https://openai.com/index/gpt-5-6-in-kiro",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-24T12:00:00.000Z",
   "summary": "GPT‑5.6 is now available in Kiro, helping developers plan, build, review, and test software with better price-performance.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Your executable is a SQLite database",
   "url": "https://simonwillison.net/2026/Aug/24/your-executable-is-a-sqlite-database",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-24T11:38:15.000Z",
   "summary": "Your executable is a SQLite database Farid Zakaria describes a neat Linux pattern for creating a SQLite database file that can be directly used as an executable binary. The trick sets the SQLite file format's 4-byte application ID (68 bytes into the file) to SELF, standing for Structured Executable & Linkable Format. The various components of the ELF executable format are then arranged into a number of different SQLite tables, using this schema . Their self-exec interpreter ( C code here ) can then extract and execute the necessary pieces. You can additionally use a Linux mechanism called binf",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Vercel Sandbox is now globally available",
   "url": "https://vercel.com/changelog/vercel-sandbox-is-now-globally-available",
   "source": "Vercel",
   "group": "Coding-agent builders",
   "published": "2026-08-24T04:00:00.000Z",
   "summary": "Vercel Sandbox now runs globally, starting with four regions: iad1 (Washington, D.C.), sfo1 (San Francisco), cle1 (Cleveland), and cdg1 (Paris). iad1 remains the default. Support for all Vercel regions is coming soon. Choose a region close to the databases, object storage, and other services your sandboxes access to reduce latency. Region selection is available on all plans. Pro and Enterprise teams can also configure failover regions. If the primary region is unavailable, new sandboxes start in the closest configured failover region. Failover only applies to regions you configure; without it,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Serving 64Mi-Token Contexts on One AMD Instinct™ MI355X Node",
   "url": "https://rocm.blogs.amd.com/artificial-intelligence/long-context-serving/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-24T00:00:00.000Z",
   "summary": "Error parsing meta tag attribute “keywords”: No content.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Ensuring reliable OpenTelemetry ingestion at scale",
   "url": "https://clickhouse.com/blog/reliable-opentelemetry-ingestion-at-scale",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-24T00:00:00.000Z",
   "summary": "How ClickHouse Cloud collects 50m events per second through OpenTelemetry",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Core Banking Modernization with Temenos Core and CockroachDB",
   "url": "https://cockroachlabs.com/blog/core-banking-modernization-temenos-cockroachdb",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-08-24T00:00:00.000Z",
   "summary": "Banking has become an always-on business. Customers expect instant payments, accurate balances, and continuous access to financial services...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Hot Chips 2026: Applying High Bandwidth Flash (HBF)",
   "url": "https://chipsandcheese.com/p/hot-chips-2026-applying-high-bandwidth",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-23T22:51:05.000Z",
   "summary": "Machine learning workloads have an insatiable appetite for DRAM capacity. Flash memory is cheaper per gigabyte of capacity than DRAM. Could it offer a way out?",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Anthropic’s best AI model struggles to attract users as cheaper tools thrive",
   "url": "https://simonwillison.net/2026/Aug/23/anthropics-best-ai-model-struggles-to-attract-users-as-cheaper-t",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-23T20:24:52.000Z",
   "summary": "Anthropic’s best AI model struggles to attract users as cheaper tools thrive A few interesting numbers in this FT story gathered from \"people with knowledge of the matter\": Anthropic's \"annualized revenue\" for July is up to $65bn - it was $47bn in May, and I collected more historic numbers here . Anthropic expect Q3 to be profitable according to the same model they used to declare Q2 profitable. \"It also told investors that it had 6,000 customers that spend $100,000 annually or more.\" As for OpenAI, \"annualised revenue has jumped 35 per cent in the quarter to date and is now over $40bn, with t",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Quoting Drew Breunig",
   "url": "https://simonwillison.net/2026/Aug/23/drew-breunig",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-23T19:55:30.000Z",
   "summary": "Prior to Fable, it felt silly to waste too much time improving your coding harness or context strategies. A new model would arrive at the same price (or cheaper!) and paper over most of your problems. But then Fable landed. It was (and still is!) incredible . But the cost was so high and Opus was good enough (as was 5.6, K3, and even GLM) for most of the code we needed. So we started to think about what work went where. — Drew Breunig , Fable & The End of the Free Lunch Tags: drew-breunig , anthropic , claude , llm-pricing , ai , llms , generative-ai , claude-mythos-fable",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Hot Chips 2026: Samsung and HBM Base Die Opportunities",
   "url": "https://chipsandcheese.com/p/hot-chips-2026-samsung-and-hbm-base",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-23T19:01:23.000Z",
   "summary": "Machine learning applications demand ever more memory capacity and bandwidth. Samsung responded by fabricating their HBM base dies on a logic node, which opens up more opportunities",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Exploring Speculative Decoding in vLLM on AMD GPUs",
   "url": "https://vllm.ai/blog/2026-08-23-speculative-decoding-amd-gpus",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-23T00:00:00.000Z",
   "summary": "A practical guide to speculative decoding in vLLM on AMD GPUs, covering draft-and-verify mechanics, MTP, EAGLE-3, DFlash, DSpark, configuration, tuning, and benchmark results.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Quoting Linus Torvalds",
   "url": "https://simonwillison.net/2026/Aug/22/linus-torvalds",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-22T21:04:26.000Z",
   "summary": "And this was a debug session from hell, enormously helped by an AI doing much of the grunt-work. I'd like to call it my tireless helper, but the AI several times stated flat out that this was impossible and unsolvable and that we should just write a report about it. I suspect those things have been trained by people who may not be quite as stubborn as I am. But while the AI was ready to give up several times, it did keep adding debug code and analyzing it faithfully when I pushed. So credit where credit is due and I let the AI write the commit message above. — Linus Torvalds , drm/xe: Don't ha",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "llm 0.33",
   "url": "https://simonwillison.net/2026/Aug/22/llm",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-22T17:01:16.000Z",
   "summary": "Release: llm 0.33 My highlights from this release: Upgraded to the OpenAI Python library 3.x and switched the HTTP client dependency from httpx to httpx2 . #1608 , #1631 I shipped a quick 0.32.1 fix for this yesterday, but this is the more comprehensive fix. llm embed and llm embed-multi now accept --key . The Python EmbeddingModel.embed() , EmbeddingModel.embed_multi() , Collection.embed() and Collection.embed_multi() methods accept key= too, passing the resolved per-call key to embedding plugins without changing shared model state. Existing plugins that read self.key continue to work through",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "More than just code review",
   "url": "https://simonwillison.net/2026/Aug/22/more-than-just-code-review",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-22T15:56:54.000Z",
   "summary": "The key skill required to make productive use of coding agents is being able to confidently instruct them on how to make changes and then confidently verify that those changes have been applied in the correct way. Sometimes this involves reviewing every line of code they have written, but there are other ways to achieve that goal. Eyeballing every line of code has never been the most effective way to validate a change to a piece of software. Tags: code-review , coding-agents , generative-ai , agentic-engineering , ai , llms",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] 10% worse, 100x cheaper, 10000x faster: Why Simulation is taking over",
   "url": "https://www.latent.space/p/ainews-10-worse-100x-cheaper-10000x",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-22T07:36:00.000Z",
   "summary": "Did you think RSI stopped at model training?",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Evolution of the Agent Harness",
   "url": "https://www.latent.space/p/attention-interface",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-22T07:30:52.000Z",
   "summary": "Models keep absorbing the harness into their weights — soon, it will be a harness for human attention rather than for the model.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Large-Scale Sharded Weight Transfer with Ray Direct Transport (RDT) in vLLM",
   "url": "https://vllm.ai/blog/2026-08-22-rdt-weight-transfer",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-22T00:00:00.000Z",
   "summary": "We implement a native sharded weight transfer engine in vLLM utilizing Ray Direct Transport (RDT), achieving weight transfer for the Kimi K2 model in BF16 on 48 8xH100 nodes in 7.53s",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Simulation: the new Scaling Law — Joon Sung Park, Simile AI",
   "url": "https://www.latent.space/p/simile",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-21T23:37:38.000Z",
   "summary": "Simile’s CEO about his journey from the viral Generative Agents to creating 8 Billion Digital Twins of every living human... and why it’s gone from fun exploration to very serious business.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Say it once: introducing Bot Preference Sync",
   "url": "https://blog.cloudflare.com/bot-preference-sync",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-21T23:19:57.000Z",
   "summary": "Cloudflare's new Bot Preference Sync automatically aligns your robots.txt file with your AI bot policies for Search, Agent, and Training. Easily manage which bots access your content without maintaining static files.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "llm 0.32.1",
   "url": "https://simonwillison.net/2026/Aug/21/llm",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-21T17:16:13.000Z",
   "summary": "Release: llm 0.32.1 Fresh installs of LLM stopped working the other day because the OpenAI Python library dropped its usage of httpx , and it turned out LLM depended on that library but only installed it via a transitive openai dependency. This dot-release fixes that for the moment by pinning to openai<3 , and a soon-to-drop 0.33 release will switch from httpx to httpx2 . Tags: httpx , openai , llm",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agentic Data Operations Platform (ADOP): Data engineering into hours",
   "url": "https://aws.amazon.com/blogs/machine-learning/agentic-data-operations-platform-adop-data-engineering-into-hours",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-21T17:06:17.000Z",
   "summary": "The Agentic Data Operations Platform (ADOP) is a reference architecture on Amazon Bedrock that uses specialized AI agents to automate the full Bronze-to-Silver-to-Gold data pipeline lifecycle, compressing new-source onboarding from weeks to hours while keeping data governance and compliance controls inline.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Govern AI agent tool access with Amazon Bedrock AgentCore Gateway",
   "url": "https://aws.amazon.com/blogs/machine-learning/govern-ai-agent-tool-access-with-amazon-bedrock-agentcore-gateway",
   "source": "AWS Machine Learning",
   "group": "Large-scale production systems",
   "published": "2026-08-21T17:02:35.000Z",
   "summary": "Give your AI agents governed, auditable access to enterprise tools without consolidating infrastructure. This post walks through a four-scope maturity model (Connect, Control, Catalog, and Harden) for building a governed tool gateway with Amazon Bedrock AgentCore, advancing only when real governance pain demands it.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "llm-openrouter 0.7",
   "url": "https://simonwillison.net/2026/Aug/21/llm-openrouter",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-21T16:58:19.000Z",
   "summary": "Release: llm-openrouter 0.7 Now that this plugin is compatible with LLM 0.32 it can display the reasoning traces for LLMs available through OpenRouter. Updated for compatibility with LLM 0.32 . Models now use OpenRouter's implementation of the Responses API . Three new server-side tools: Shell , WebFetch , and WebSearch . Enable these with options like -T WebSearch . Tags: llm , openrouter",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GPU-Accelerated Clustering for Financial Instruments at Scale",
   "url": "https://developer.nvidia.com/blog/gpu-accelerated-clustering-for-financial-instruments-at-scale",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-21T16:21:04.000Z",
   "summary": "Use AdaptGrow, a GPU-accelerated matrix factorization algorithm, to turn rolling correlation and tail-dependence matrices into hard clusters, soft factor...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Stop Making TUIs",
   "url": "https://simonwillison.net/2026/Aug/21/stop-making-tuis",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-21T16:07:32.000Z",
   "summary": "Stop Making TUIs Thomas Ptacek advocates for building real native user interfaces for even the smallest of personal tools, because coding agents have reduced the cost of getting a usable-enough GUI up and running to almost nothing. I wrote about my vibe-coded bandwidth and GPU monitoring macOS task bar apps back in March , and I'm still using both of those on a daily basis. I'm not habitually knocking out real UIs for my other projects yet, but I'm running out of excuses! Thomas: If you haven’t tried your hand at turning one of your 500 throwaway CLIs into a native app, you’re doing yourself a",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A Tale of Two Flink Autoscalers",
   "url": "https://netflixtechblog.com/a-tale-of-two-flink-autoscalers-e9f6a1b1492b",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-08-21T16:01:01.000Z",
   "summary": "Samuel Yeboah , Francesco Di Chiara and Mingliang Liu Today, Netflix runs two Flink autoscalers. That is exactly one more than we want. We built the first one in-house years ago, when there was no mature option suited to our platform. The second came from the Apache Flink community, and it can scale workloads our homegrown system was never designed for. We now run both in production and are steadily converging on the open-source one. Along the way we learned some hard lessons about metrics, cost, and the real price of maintaining infrastructure you could instead adopt, and we hope they are use",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cloud CISO Perspectives: Sticking to security fundamentals in the AI era",
   "url": "https://cloud.google.com/blog/products/identity-security/cloud-ciso-perspectives-sticking-to-security-fundamentals-in-the-ai-era",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-21T16:00:00.000Z",
   "summary": "Welcome to the first Cloud CISO Perspectives for August 2026. Today, Chris Betz explains why the AI era makes it more important than ever to lean into security fundamentals. As with all Cloud CISO Perspectives, the contents of this newsletter are posted to the Google Cloud blog . If you’re reading this on the website and you’d like to receive the email version, you can subscribe here . aside_block <ListValue: [StructValue([('title', 'Get vital board insights with Google Cloud'), ('body', <wagtail.rich_text.RichText object at 0x7f6002362b90>), ('btn_text', 'Visit the hub'), ('href', 'https://cl",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How agents can delegate better",
   "url": "https://cloud.google.com/blog/products/ai-machine-learning/how-agents-can-delegate-better",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-21T16:00:00.000Z",
   "summary": "In any organizational behavior class, students will learn that effective delegation is among the most important skills for a seasoned leader. Getting meaningful work done involves careful coordination, starting with a subdivision of projects into manageable tasks, mapped onto the skills of the team, and assigned to the right people. At Google Cloud, we’re learning a similar lesson when it comes to building and deploying AI agents in enterprise workflows. These workflows are best approached by multi-agent systems that can break apart and execute complex tasks. To do so, AI agents need to become",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Quoting Matt Webb",
   "url": "https://simonwillison.net/2026/Aug/21/matt-webb",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-21T15:06:26.000Z",
   "summary": "After I released version 1.0, I figured I would have to do the rotations myself. So I sat down with ChatGPT and I didn’t get it to write the code, but I got it to educate me. With a patient, interactive tutor, I was able to finally do what I hadn’t by reading books and asking mathematician friends – I learnt how to use quaternions just enough to make the app work. So learning doesn’t stop just because I outsource a bunch of thinking to AI. It pushes me to learn more. I like that as an outcome. — Matt Webb , Galactic Compass 2: now with new augmented reality mode Tags: matt-webb , generative-ai",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What else runs on your Postgres server, and how do we stop it from taking the database down?",
   "url": "https://clickhouse.com/blog/protect-postgres-from-supporting-processes",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-21T14:11:30.000Z",
   "summary": "ClickHouse Managed Postgres uses runtime budgets, cgroup limits, and disk-full session exemptions to keep supporting services from compromising database availability.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "NVIDIA AVO Reaches 100% on ARC-AGI-3, Demonstrating a Frontier-Level General-Purpose Architecture for Long-Horizon Autonomous Agents",
   "url": "https://developer.nvidia.com/blog/nvidia-avo-reaches-100-on-arc-agi-3-demonstrating-a-frontier-level-general-purpose-architecture-for-long-horizon-autonomous-agents",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-21T13:00:00.000Z",
   "summary": "A frontier language model is only one component of an AI agent. The surrounding agent system—often called a harness—determines how the model receives...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Where Security Fits in an AI Agent Stack",
   "url": "https://developer.nvidia.com/blog/where-security-fits-in-an-ai-agent-stack",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-21T13:00:00.000Z",
   "summary": "As AI agents become more capable and operate over longer horizons, building security and trust into the applications they power becomes increasingly important....",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "August 2026 newsletter",
   "url": "https://clickhouse.com/blog/202608-newsletter",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-21T09:12:10.000Z",
   "summary": "Welcome to the August 2026 ClickHouse newsletter, which will round up what’s happened in real-time data warehouses over the last month.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] Poolside gets $12B reverse-execuhire to NVIDIA; founders stay for $1B, employees go for $6B, Infraco scaling to 7GW neocloud",
   "url": "https://www.latent.space/p/ainews-poolside-gets-12b-reverse",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-21T05:45:21.000Z",
   "summary": "Yes, we’re confused too.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "IsoExec: Unified Execution to Eliminate Trainer-Inference Mismatch in SkyRL",
   "url": "https://vllm.ai/blog/2026-08-21-isoexec",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-21T00:00:00.000Z",
   "summary": "IsoExec unifies numerical execution across SkyRL's vLLM and Megatron runtimes, reducing the average rollout-versus-training logprob difference below 1e-6 on Qwen3.5-35B-A3B with 25% overhead.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Fast Engine Recovery: Sub-Second Engine Restart for SGLang via Weight Cache Daemon",
   "url": "https://lmsys.org/blog/2026-08-21-sglang-fast-recovery",
   "source": "SGLang and LMSYS",
   "group": "Serving engines and runtimes",
   "published": "2026-08-21T00:00:00.000Z",
   "summary": "Nowadays, State-of-the-Art (SOTA) models are getting much bigger and reloading the model service after a crash is very expensive. Therefore, we are introducing the Weight Cache Daemon, a persistent GP...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Chasing the Batch-1 Floor: Ling-3.0-flash Speculative Decode on Blackwell",
   "url": "https://lmsys.org/blog/2026-08-21-ling3-flash-spec-decode-blackwell",
   "source": "SGLang and LMSYS",
   "group": "Serving engines and runtimes",
   "published": "2026-08-21T00:00:00.000Z",
   "summary": "Batch-1 decode keeps getting more important. Xiaomi MiMo, for example, announced MiMo-V2.5-Pro UltraSpeed in June, claiming 1,000 tok/s decode on a one-trillion-parameter MoE model. Batch 1 gives an ...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Hugging Face Inference Endpoints, Jobs, and Buckets Power Search on Papers with Code",
   "url": "https://huggingface.co/blog/pwc-search",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-21T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Measuring benchmark optimization in speech recognition",
   "url": "https://huggingface.co/blog/asr-benchmark-optimization",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-21T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GLM-5.3 vs. GPT-5.6 Sol on DeepSWE: Cost, Coding, and Routing",
   "url": "https://www.together.ai/blog/glm-5-3-vs-gpt-5-6-sol-on-deepswe-cost-coding-and-routing",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-08-21T00:00:00.000Z",
   "summary": "We ran 904 DeepSWE rollouts on GLM-5.3 and GPT-5.6 Sol. Sol leads pass@1 by 3.7 points; GLM-5.3 wins pass@4 at half the cost, and a GLM-first cascade hits 85.9%.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GLM-5.3 vs. Claude Fable 5 on DeepSWE: Cost, Coding, and Routing",
   "url": "https://www.together.ai/blog/glm-5-3-vs-claude-fable-5-on-deepswe-cost-coding-and-routing",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-08-21T00:00:00.000Z",
   "summary": "We ran 904 DeepSWE rollouts on GLM-5.3 and Claude Fable 5. A tie on pass@1, but GLM-5.3 wins pass@4 and costs 5.4x less: \\$3.99 per rollout vs. \\$21.63.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "DI Series: Scaling GLM-5.1-FP8 to 64 MI300X GPUs",
   "url": "https://rocm.blogs.amd.com/software-tools-optimization/di-glm-wideep/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-21T00:00:00.000Z",
   "summary": "Serving a frontier Mixture-of-Experts (MoE) model well is a systems problem, and it gets harder the moment one node is not enough. GLM-5.1 is a good example: it is a large, sparse MoE that users want to run at long context, and it ships a new attention family that breaks assumptions older serving stacks quietly relied on. Fitting it on eight GPUs is only the start. The real question is how to keep it correct and fast as you spread it across several nodes.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ChatGPT search now uses the site:operator at scale",
   "url": "https://simonwillison.net/2026/Aug/20/chatgpt-search-now-uses-the-siteoperator-at-scale",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-20T23:57:32.000Z",
   "summary": "ChatGPT search now uses the site:operator at scale Promptwatch is part of the emerging \"GEO\" space, for Generative Engine Optimization - the chatbot version of SEO, where companies offer tools and consulting to help your site increase its presence in replies to prompts inside tools like ChatGPT. The Promptwatch product uses automation to track responses to prompts across end-user chat products like ChatGPT, Claude, and Gemini. They publish aggregate reports on this as part of their own content marketing strategy, which do seem to provide credible hints as to otherwise invisible design changes ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The /wayfinder Skill: Navigating the “Fog of War” of Planning",
   "url": "https://www.latent.space/p/wayfinder-skill",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-20T20:59:09.000Z",
   "summary": "Matt Pocock tells us about his /wayfinder skill, for greenfield projects or for when the way forward is unclear.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The August 17 outage, and the work ahead",
   "url": "https://github.blog/news-insights/company-news/the-august-17-outage-and-the-work-ahead",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-20T18:36:11.000Z",
   "summary": "An update on the August 17 outage and the steps we're taking to improve reliability. The post The August 17 outage, and the work ahead appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Expanding Google Antigravity for enterprise customers",
   "url": "https://cloud.google.com/blog/products/ai-machine-learning/expanding-google-antigravity-for-enterprise-customers",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-20T17:30:00.000Z",
   "summary": "Since announcing Google Antigravity in Gemini Enterprise Agent Platform at I/O in May, we’ve heard helpful feedback from our customers. Your developers want easy access to coding agents across surfaces. Your enterprise governance team wants security controls and license management. And your finance team wants pooled usage so that no prepaid token ever goes unused. Now, everybody finally gets what they want: Antigravity is available now as part of eligible Gemini Enterprise app subscriptions, including out-of-the-box administrative and spend controls. New IDE extensions let developers use Antig",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "From all-or-nothing to task-based OAuth consent",
   "url": "https://blog.cloudflare.com/task-based-oauth-consent",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-20T17:03:03.000Z",
   "summary": "Cloudflare OAuth now supports optional scopes, giving users more control over what an app can access and helping developers build secure consent flows around the task at hand.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Up to 3.2x Faster Inference with LFM2.5-DSpark",
   "url": "https://huggingface.co/blog/LiquidAI/lfm25-dspark",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-20T16:52:57.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "POSETTE Talk Recap - Postgres Isn't Slow. Your Storage Is",
   "url": "https://clickhouse.com/blog/posette-talk-recap-postgres-isnt-slow-your-storage-is",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-20T16:11:08.000Z",
   "summary": "A recap of Sai Srirampur's POSETTE 2026 talk on storage performance in Postgres, with a local NVMe vs. EBS benchmark and the production setup behind it.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Generative Recommenders Are Redefining RecSys at Scale",
   "url": "https://developer.nvidia.com/blog/how-generative-recommenders-are-redefining-recsys-at-scale",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-20T16:00:00.000Z",
   "summary": "Recommender systems (RecSys) are one of the most ubiquitous machine learning problems in the consumer internet industry yet notoriously difficult to train and...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How AlloyDB ScaNN scales vector search to 10 billion vectors",
   "url": "https://cloud.google.com/blog/products/databases/alloydb-scann-index-four-level-tree-improves-vector-search",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-20T16:00:00.000Z",
   "summary": "To satisfy the demands of enterprise-grade agentic AI applications, underlying vector databases often struggle to scale effectively as modern use cases can scale to billions of vectors. As a fully managed PostgreSQL-compatible database service, AlloyDB is engineered to handle demanding enterprise workloads. Combining Google's infrastructure with the reliability of commercial databases, it delivers high availability, scalability, and includes a cutting-edge analytical engine, optimal for agentic AI use cases. A key part of this is its ScaNN index , which now operates efficiently at a scale of 1",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "10 questions every startup should answer before moving to production with their AI prototype",
   "url": "https://cloud.google.com/blog/topics/developers-practitioners/10-questions-for-your-startup-developers",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-20T16:00:00.000Z",
   "summary": "It’s never been easier to start an AI-powered startup on Google Cloud. You grab an API key from Google AI Studio at breakfast, paste it into Antigravity, and by lunch you’ll have a nascent prototype of your product. But it’s not all one straight line to progress. It's common to bump into these three challenges as you build out your stack: A leaked API key racks up a large bill in 48 hours . A \"quick\" migration from AI Studio to Gemini Enterprise Agent Platform stalls the roadmap for weeks because nobody on the team owns Identity and Access Management (IAM). The launch works, until the app star",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A shot-scraper-style JSON API on Bun 1.4's new Bun.WebView",
   "url": "https://simonwillison.net/2026/Aug/20/bun-webview-json-api",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-20T15:37:00.000Z",
   "summary": "Research: A shot-scraper-style JSON API on Bun 1.4's new Bun.WebView Today saw the long awaited release of Bun 1.4 , the first stable version since the infamous Rust rewrite a few months ago . Interestingly, the Rust rewrite was downplayed in the release notes, which introduced a bewildering array of new features and claimed 2,900 additional bug fixes: Bun 1.4 adds +1,517 tests from the Node.js test suite - our biggest jump in Node.js compatibility since Bun 1.0. Bun v1.4 also fixes over 2,900 issues. It reduces idle CPU usage by 5x, reduces memory usage by up to 35%, and starts 50% faster on ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Intelligence Age",
   "url": "https://openai.com/index/introducing-ai-futures",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-20T07:00:00.000Z",
   "summary": "Introducing Intelligence Age, a new OpenAI blog exploring how transformative AI could reshape power, governance, the economy, and individual freedom.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Intelligence Age",
   "url": "https://openai.com/index/introducing-intelligence-age",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-20T07:00:00.000Z",
   "summary": "Introducing Intelligence Age, a new OpenAI blog exploring how transformative AI could reshape power, governance, the economy, and individual freedom.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] Death of Params: Z.ai CEO Jie Tang on GLM 5.3 and the new Post-training Scaling Law",
   "url": "https://www.latent.space/p/ainews-death-of-params-zai-ceo-jie",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-20T05:17:12.000Z",
   "summary": "Every lab CEO is on X now",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Stampli cuts launch hours by 68% using ChatGPT Work",
   "url": "https://openai.com/index/stampli",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-20T00:00:00.000Z",
   "summary": "With a fixed deadline and design resources committed elsewhere, Stampli used Codex and ChatGPT Work to compress weeks of launch production into days.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "VeRL-Omni v0.2.0: Faster Diffusion RL and Stable Omni Training",
   "url": "https://vllm.ai/blog/2026-08-20-verl-omni-v0-2-0",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-20T00:00:00.000Z",
   "summary": "A release focused on higher-throughput diffusion rollout, reusable omni adapters, and broader recipe coverage.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Mooncake for Miles: From Fragmented Rollout Data to Efficient Bulk I/O",
   "url": "https://lmsys.org/blog/2026-08-20-miles-mooncake-rollout-data-transfer",
   "source": "SGLang and LMSYS",
   "group": "Serving engines and runtimes",
   "published": "2026-08-20T00:00:00.000Z",
   "summary": "Reinforcement learning for large language models combines two very different workloads: rollout generation and model training. During rollout, inference workers run the current policy on a set of pro...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Protocol-Aware Deterministic Simulation Testing",
   "url": "https://tigerbeetle.com/blog/2026-08-20-protocol-aware-dst",
   "source": "TigerBeetle",
   "group": "Others",
   "published": "2026-08-20T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Shopify powers observability for global-scale commerce with ClickHouse",
   "url": "https://clickhouse.com/blog/shopify-observability-at-global-scale",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-20T00:00:00.000Z",
   "summary": "Shopify unified global-scale observability on ClickHouse, achieving up to 30x faster queries while ingesting 100 million events per second at peak.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "smolmachines / smolvm as a sandbox for untrusted Python & JavaScript",
   "url": "https://simonwillison.net/2026/Aug/19/smolmachines-untrusted-sandbox",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-19T23:16:00.000Z",
   "summary": "Research: smolmachines / smolvm as a sandbox for untrusted Python & JavaScript I tasked Claude Fable 5 running in Claude Code for web with the following research task: Put https://smolmachines.com through its paces as a fast secure sandbox. Explore what it would take to use this to run untrusted Python and JavaScript code in a way that is limited in what RAM and CPU time it can take up (protection against \"while true\") with no network access and filesystem access only to designated files Goal is to be able to use this to execute user-provided tasks for things like data transformations It quick",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Quoting Jeremy Morrell",
   "url": "https://simonwillison.net/2026/Aug/19/jeremy-morrell",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-19T22:56:31.000Z",
   "summary": "My hypothesis is that there is a new opportunity for Extensible Software on the web . LLMs radically lower the cost of authoring extensions, and modern sandbox primitives lower the deployment cost and provide good security boundaries. We can build our app as a solid, accountable core, and allow users to safely extend it in many directions by having LLMs fill in the missing pieces. We can give our users super powers. — Jeremy Morrell , Extensible Software in the age of LLMs Tags: sandboxing , llms , ai , generative-ai",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Conceptual integrity and counting lines of code",
   "url": "https://simonwillison.net/2026/Aug/19/conceptual-integrity-and-counting-lines-of-code",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-19T22:46:07.000Z",
   "summary": "Last week I recorded an episode of the Talking Postgres podcast with Claire Giordano on the subject of \"How AI is changing software development\". We had a really great conversation. Here are a couple of my highlights from a lightly edited transcript (prompt to Claude: \"very minor edits to remove disfluencies\"). This is the latest version of an argument I've been trying to build about why sometimes it does make sense to talk about lines of code as an indicator of productivity with coding agents, at 35:01 : A lot of people will tell you it makes no sense to measure productivity in lines of code.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Developing NVIDIA Holoscan Applications with CLI, Skills, and AI Coding Agents",
   "url": "https://developer.nvidia.com/blog/developing-nvidia-holoscan-applications-with-cli-skills-and-ai-coding-agents",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-19T22:22:37.000Z",
   "summary": "NVIDIA Holoscan is a platform for building real-time AI applications at the edge, from medical imaging to robotics. HoloHub is its companion repository: a...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Offering Zero Data Retention for frontier models",
   "url": "https://openai.com/index/offering-zero-data-retention-for-frontier-models",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-19T19:00:00.000Z",
   "summary": "OpenAI reaffirms Zero Data Retention for eligible API customers and previews Private Safety Processing for advanced AI safety without compromising data privacy.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building Federated Multimodal AI Workflows with NVIDIA FLARE",
   "url": "https://developer.nvidia.com/blog/building-federated-multimodal-ai-workflows-with-nvidia-flare",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-19T17:50:47.000Z",
   "summary": "Modern vision-language models (VLMs) can support tasks such as visual question answering, captioning, and image-text reasoning. In practice, however, the data...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GitHub Copilot app for Beginners: Managing your work",
   "url": "https://github.blog/ai-and-ml/github-copilot/github-copilot-app-for-beginners-managing-your-work",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-19T17:50:23.000Z",
   "summary": "If you’re juggling multiple Copilot sessions, use the My work pane to track what's in flight, what's done, and what's next. The post GitHub Copilot app for Beginners: Managing your work appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A revisit of remote Spectre attacks on Cloudflare Workers",
   "url": "https://blog.cloudflare.com/revisiting-spectre-attacks-on-workers",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-19T16:00:28.000Z",
   "summary": "In 2024 and 2025, we reassessed remote Spectre attacks on our Workers infrastructure. We share details about the new attack primitives like Spectre gadgets, remote timers, achieving co-location and how new defenses further harden Cloudflare Workers.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Post-Train NVIDIA Cosmos 3 Edge for On-Device Robot Control",
   "url": "https://developer.nvidia.com/blog/post-train-nvidia-cosmos-3-edge-for-on-device-robot-control",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-19T16:00:00.000Z",
   "summary": "Robots need policies that can adapt to their sensors, environments, and tasks while running on onboard computing hardware. World models offer a foundation for...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Evaluating AI Agent Skill Performance with NVIDIA SkillEvaluator",
   "url": "https://developer.nvidia.com/blog/evaluating-ai-agent-skill-performance-with-nvidia-skillevaluator",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-19T16:00:00.000Z",
   "summary": "AI agents are only as effective as the context they receive. Even with capable models and well-documented NVIDIA libraries, agents can spend extra steps finding...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] Memory prices up 500% in 12 months",
   "url": "https://www.latent.space/p/ainews-memory-prices-up-500-in-12",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-19T08:44:52.000Z",
   "summary": "the Memory crunch continues - Moore’s Law reversed to 2007 levels",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Replit expands access to software creation with GPT-5.6 Luna",
   "url": "https://openai.com/index/replit",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-19T07:00:00.000Z",
   "summary": "Replit introduces Free Mode, powered by GPT-5.6 Luna, so anyone can turn ideas into working software without worrying about token costs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "CVE-2025-62593 and the CISA KEV listing: what Ray users need to know",
   "url": "https://anyscale.com/blog/ray-cve-2025-62593-kev-what-you-need-to-know",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-19T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Pushing the Limits of Serving DeepSeek-V4-Pro",
   "url": "https://lmsys.org/blog/2026-08-19-deepseek-v4-pro-engine-optimization-h20",
   "source": "SGLang and LMSYS",
   "group": "Serving engines and runtimes",
   "published": "2026-08-19T00:00:00.000Z",
   "summary": "DeepSeek-V4-Pro is a 1.6-trillion-parameter Mixture-of-Experts (MoE) model released with both FP8 and FP4 weights. Models at this scale naturally benefit from accelerators such as NVIDIA Blackwell GPU...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "4 FAQs about designing agentic systems for production",
   "url": "https://www.redpanda.com/blog/designing-agentic-systems-architect-faqs",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-08-19T00:00:00.000Z",
   "summary": "Learn about the technical challenges many organizations face while deploying agentic systems, and how to ensure AI agents can operate safely and consistently.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A2A Is Now an Open Standard. The Data Layer Underneath It Isn't.",
   "url": "https://cockroachlabs.com/blog/a2a-agent-state-data-layer",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-08-19T00:00:00.000Z",
   "summary": "In April 2025, Google released a protocol for agent-to-agent communication. Within three months, Google had donated it to the Linux Foundation...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ChatGPT Ads expands across Europe",
   "url": "https://openai.com/index/chatgpt-ads-expands-across-europe",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-18T22:00:00.000Z",
   "summary": "ChatGPT Ads is expanding to 31 European markets. Learn how advertisers can reach people as they explore, compare options, and make decisions.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Databricks Document Intelligence: pushing the frontier for complex document extraction",
   "url": "https://www.databricks.com/blog/databricks-document-intelligence-pushing-frontier-complex-document-extraction",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-18T22:00:00.000Z",
   "summary": "Every enterprise has valuable data trapped in messy, unstructured documents. Today,...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Frontier Model Cost and Open-Weights Popularity is Driving Demand for Model Routing",
   "url": "https://www.latent.space/p/glean-model-routing",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-18T21:41:10.000Z",
   "summary": "Glean CEO Arvind Jain explains why model routing helps control AI costs for organizations, and how human feedback loops at scale improve its routing systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Mojo🔥 is now open source",
   "url": "https://simonwillison.net/2026/Aug/18/mojo-is-now-open-source",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-18T21:39:20.000Z",
   "summary": "Mojo🔥 is now open source The Mojo programming language has been promising an open source release since May 2023 . Last week they shipped their 1.0 and today they have followed through on that original promise, releasing the compiler and toolchain under an Apache 2 license. When Mojo first launched the stated goal was to produce a superset of Python, so existing Python code could be used to bootstrap their own ecosystem. That plan changed around August 2025 : Mojo may or may not evolve into a full superset of Python, and it’s okay if it doesn’t. We’re encouraged by how well AI-assisted coding ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Strengthening democratic oversight in national security",
   "url": "https://openai.com/index/strengthening-democratic-oversight-in-national-security",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-18T19:00:00.000Z",
   "summary": "OpenAI launches an initiative to strengthen democratic oversight of AI in national security, supporting government institutions with tools, training, and expertise.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Much Memory Does Your Agent Actually Need?",
   "url": "https://huggingface.co/blog/ibm-research/altk-evolve-hmm",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-18T18:09:38.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How AI Coding Agents Can Unlock Materials Simulation with NVIDIA ALCHEMI Toolkit",
   "url": "https://developer.nvidia.com/blog/how-ai-coding-agents-can-unlock-materials-simulation-with-nvidia-alchemi-toolkit",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-18T18:00:00.000Z",
   "summary": "Atomistic simulation requires three things: knowledge of the science, compute-efficient implementation of simulations, and accessible interfaces to the...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Improving infrastructure efficiency for growing demand in the age of AI",
   "url": "https://dropbox.tech/infrastructure/improving-infrastructure-efficiency-for-growing-demand-in-the-age-of-ai",
   "source": "Dropbox Tech",
   "group": "Large-scale production systems",
   "published": "2026-08-18T17:00:00.000Z",
   "summary": "As demand for AI continues to grow, so does the infrastructure needed to support it.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Run Massive-Scale UMAP in Minutes Using Multiple GPUs—Without Losing Accuracy",
   "url": "https://developer.nvidia.com/blog/run-massive-scale-umap-in-minutes-using-multiple-gpus-without-losing-accuracy",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-18T16:48:08.000Z",
   "summary": "Uniform Manifold Approximation and Projection (UMAP) is a dimensionality reduction technique widely used for visualization and feature extraction. Applications...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Sony LIV uses ClickHouse Cloud to deliver live streaming analytics at billion-row scale",
   "url": "https://clickhouse.com/blog/sony-liv-real-time-analytics",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-18T16:20:22.000Z",
   "summary": "Sony LIV consolidated fragmented batch, Elasticsearch, and BigQuery workloads on ClickHouse Cloud, delivering sub-second analytics across billions of daily streaming events.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Async inference in practice: a video-indexing service on Ray Serve",
   "url": "https://anyscale.com/blog/ray-serve-async-inf-in-practice",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-18T16:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Box is unlocking multimodal enterprise agents with Gemini Embeddings 2",
   "url": "https://cloud.google.com/blog/topics/partners/box-ai-agents-gemini-embeddings-multimodal-enterprise-ai",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-18T16:00:00.000Z",
   "summary": "Enterprise content management is experiencing its biggest architectural shift since the cloud migration era. For years, enterprises have stored trillions of gigabytes of critical data in Box: financial models, clinical trial protocols, M&A due diligence rooms, engineering schematics, and legal compliance playbooks. Up to this point, text-based search and retrieval-augmented generation (RAG) have successfully unlocked the vast narrative knowledge within these repositories, establishing a powerful and highly effective baseline for enterprise AI intelligence. Traditional RAG architectures have ma",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building cost-effective, high-throughput gen AI workflows in Google Dataflow",
   "url": "https://cloud.google.com/blog/products/data-analytics/cost-effective-genai-workflows-in-google-dataflow",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-18T16:00:00.000Z",
   "summary": "Real-time streaming pipelines are the operational backbone of modern enterprises, continuously processing everything from customer support interactions to transaction logs. Traditionally, streaming DAGs are static; once deployed, their processing logic and execution paths are fixed. However, by integrating generative AI agents, we can move beyond static logic to adaptive execution. This allows streaming workflows to dynamically construct plans, query databases, and trigger custom remediation paths at runtime depending on the content of the data. For example, when a customer sends an angry mess",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "BGP Role model: tracking the adoption of RFC 9234",
   "url": "https://blog.cloudflare.com/rfc9234-bgp-role-model",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-18T15:21:32.000Z",
   "summary": "RFC 9234 lets routers reject route leaks on their own, using BGP Roles and the Only to Customer attribute. We measured who has deployed it, and found two Tier 1 networks unexpectedly stripping OTC.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Evaluating AI Agents Live at the Grounded Reasoning Cup",
   "url": "https://www.databricks.com/blog/evaluating-ai-agents-live-grounded-reasoning-cup",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-18T15:00:00.000Z",
   "summary": "This year, Databricks hosted the inaugural Grounded Reasoning Cup, a first-of-its-kind...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What's New with Monitoring in PostgreSQL 19",
   "url": "https://clickhouse.com/blog/postgres-19-monitoring-whats-new",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-18T14:31:36.000Z",
   "summary": "What's New with Monitoring in PostgreSQL 19",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building operational resilience with agentic AI in financial services",
   "url": "https://cloud.google.com/blog/topics/financial-services/building-operational-resilience-with-agentic-ai-in-financial-services",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-18T14:00:00.000Z",
   "summary": "For financial institutions, operational resilience has long been embedded in regulatory and supervisory expectations — to say nothing of the high expectations of consumers. With the implementation of the European Union’s Digital Operational Resiliency Act (DORA), those expectations have become even more stringent, with more explicit, harmonized, and evidence-driven requirements. Firms must now demonstrate that their critical business services and supporting digital infrastructures can withstand disruption, support coordinated response, and recover with control. To meet these conditions, Deutsc",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Partnering with CodeAI to prepare the first AI generation",
   "url": "https://openai.com/index/partnering-with-codeai",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-18T11:00:00.000Z",
   "summary": "OpenAI and CodeAI are partnering to help students build AI literacy, think critically about AI, and develop the skills to use and shape it responsibly.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Pacing model development in an era of cyber-critical capabilities",
   "url": "https://openai.com/index/pacing-model-development-cyber-capabilities",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-18T11:00:00.000Z",
   "summary": "OpenAI is strengthening monitoring, alignment, and security for frontier AI models. See how new safeguards are guiding the pace of model development.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing ChatGPT for Teens: Built for learning, backed by protections",
   "url": "https://openai.com/index/chatgpt-for-teens",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-18T11:00:00.000Z",
   "summary": "ChatGPT for Teens helps teens learn, think critically, and use AI with confidence, with stronger built-in protections, healthy-use features, and additional controls for parents.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Using Ray Direct Transport for Fast and Easy Weight Syncing in Reinforcement Learning (Part 2)",
   "url": "https://anyscale.com/blog/rdt-ray-direct-transport-fast-easy-weight-syncing-for-rl-reinforcement-learning",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-18T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Asana cleared 5 years of engineering work in 2 weeks with Codex",
   "url": "https://openai.com/index/asana",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-18T07:00:00.000Z",
   "summary": "Asana used OpenAI Codex to replace an outdated testing system in two weeks, completing work expected to take five years for about $12K.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How NVIDIA scales expertise with ChatGPT Work",
   "url": "https://openai.com/index/nvidia/chatgpt-work",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-18T00:00:00.000Z",
   "summary": "NVIDIA teams use ChatGPT Work to reduce manual tasks, connect fast-moving signals, and scale successful workflows globally.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Miles v0.1: Production-level Post-training",
   "url": "https://lmsys.org/blog/2026-08-18-miles-v0-1",
   "source": "SGLang and LMSYS",
   "group": "Serving engines and runtimes",
   "published": "2026-08-18T00:00:00.000Z",
   "summary": "We present Miles v0.1, a full-stack production-ready system for frontier post-training, the successor to our first Miles release. Building upon slime's clean design, Miles optimizes every stage in the...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Multi-Vector (Late Interaction) Embedding Models with Sentence Transformers",
   "url": "https://huggingface.co/blog/multi-vector-encoder",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-18T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "DeepSeek V4 Pro 0813 vs GPT-5.6 Sol on DeepSWE: Cost, Coding, and Routing",
   "url": "https://www.together.ai/blog/deepseek-v4-pro-0813-vs-gpt-5-6-sol-on-deepswe-cost-coding-and-routing",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-08-18T00:00:00.000Z",
   "summary": "We ran 904 DeepSWE rollouts on DeepSeek V4 Pro 0813 and GPT-5.6 Sol. Sol leads pass@1 by 10 points at 35x the cost; Pro wins pass@4, and a Pro-first cascade hits 83.0%.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling RL with verl on AMD Instinct MI355X: Async Walkthrough and Sync Benchmark",
   "url": "https://rocm.blogs.amd.com/artificial-intelligence/verl/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-18T00:00:00.000Z",
   "summary": "Reinforcement learning (RL) for large language models (LLMs) alternates between two phases: generation (rollout), where the current policy produces responses, and training, where those responses are used to update the policy. In verl, the key design choices are when these phases run relative to each other (synchronously or with overlap) and where they run (colocated on the same GPUs or on separate GPU pools). This blog first explains the differences between the two modes and when to use each.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Exploring XGBoost: A Deep Dive",
   "url": "https://rocm.blogs.amd.com/software-tools-optimization/xgboost_deep_dive/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-18T00:00:00.000Z",
   "summary": "XGBoost (Extreme Gradient Boosting) is an open-source library that implements gradient-boosted decision trees, an ensemble method that builds an additive sequence of trees where each new tree is fit to the gradient of the loss left by the ones before it. It supports regression, classification, ranking, and survival objectives behind a single training loop, and is implemented as a high-performance C++ core with CPU and CUDA/HIP backends, exposed through Python, R, and JVM bindings. On large tabular datasets it is a standard production choice for both accuracy and training throughput. This blog ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Qwen 3.8 27B scores 52 on the Artificial Analysis Intelligence Index",
   "url": "https://simonwillison.net/2026/Aug/17/qwen-38-27b-scores-52",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-17T23:58:14.000Z",
   "summary": "Qwen 3.8 27B scores 52 on the Artificial Analysis Intelligence Index That's the same score as GPT-5.6 Luna (max), and just one point behind GLM-5.2 (max) and DeepSeek V4 Pro 0813 (max) - that GLM is 753B and that DeepSeek is 1.7T parameters , and Luna is size unknown but presumably a whole lot bigger than 27B. Qwen 3.8 27B is a truly astonishing model . Via Hacker News Tags: ai , generative-ai , llms , qwen , ai-in-china , artificial-analysis",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] Stripe buys OpenRouter for $7B",
   "url": "https://www.latent.space/p/ainews-stripe-buys-openrouter-for",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-17T23:13:41.000Z",
   "summary": "No GPUs, no Agents, just really, really, really good infra and distribution.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Databricks Feature Store serves features with sub-second freshness",
   "url": "https://www.databricks.com/blog/how-databricks-feature-store-serves-features-sub-second-freshness",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-17T23:10:00.000Z",
   "summary": "Machine learning models are only as good as the signals they receive. A fraud detection...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Same Cluster, 33 Points More Utilization: What Changed Was the Order",
   "url": "https://huggingface.co/blog/Dharma-AI/gpu-management-pt2",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-17T19:46:21.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "From single call to agents: five new Claude capabilities now available in Microsoft Foundry",
   "url": "https://devblogs.microsoft.com/foundry/five-new-claude-capabilities-now-available-in-foundry",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-17T19:20:07.000Z",
   "summary": "Structured outputs, web search, web fetch, MCP connector, and tool search are now available for Claude models hosted on Azure in Microsoft Foundry, turning a model endpoint into a production agent platform. The post From single call to agents: five new Claude capabilities now available in Microsoft Foundry appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Developing Nemotron 3.5 Lightning NVFP4 with QAD Using NVIDIA Model Optimizer",
   "url": "https://developer.nvidia.com/blog/developing-nemotron-3-5-lightning-nvfp4-with-qad-using-nvidia-model-optimizer",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-17T18:12:48.000Z",
   "summary": "Teams customize their models to hit their targets for latency, speed, memory, and compute. With the open NVIDIA Nemotron family of models, developers can find...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How canvases make agentic workflows visible, steerable, and cost-efficient",
   "url": "https://github.blog/ai-and-ml/github-copilot/how-canvases-make-agentic-workflows-visible-steerable-and-cost-efficient",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-17T16:00:00.000Z",
   "summary": "Chat is great for intent, but agent work gets lost in the scroll. Here is how I use canvases with my agentic workflows—and why your workflow also deserves a canvas. The post How canvases make agentic workflows visible, steerable, and cost-efficient appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "We Tracked a Shipment of Rare Books. It Ended at an Amazon AI Training Facility",
   "url": "https://simonwillison.net/2026/Aug/17/we-tracked-a-shipment-of-rare-books-it-ended-at-an-amazon-ai-tra",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-17T15:21:29.000Z",
   "summary": "We Tracked a Shipment of Rare Books. It Ended at an Amazon AI Training Facility Excellent piece of reporting from 404 Media. For a while now there have been stories of book dealers receiving orders for large volumes of books from apparently price-insensitive anonymous customers, widely suspected to be companies looking to scan them for AI training (see my previous coverage of Anthropic's book scanning from June 2025.) 404 Media investigated with an AirTag! In July, one bookseller told me they received a very large order of around 1,000 books on Biblio, one of these marketplaces. The seller agr",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Teaching Everyone to Fish for Tokens",
   "url": "https://www.interconnects.ai/p/teaching-everyone-to-fish-for-tokens",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-08-17T15:07:49.000Z",
   "summary": "Nvidia wants you building your own model, not buying from Anthropic/OpenAI.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Sticky Until Saturated: Token-Aware Routing in llm-d",
   "url": "https://llm-d.ai/blog/sticky-until-saturated-token-aware-routing",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-08-17T09:00:00.000Z",
   "summary": "The llm-d router's default configuration is built on token-aware routing: keep each request on the cache-warm endpoint until a calibrated token-load limit is exceeded, then route by load alone, sustaining 2-3x round-robin throughput with the one required threshold derived from a single hardware calibration measurement.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Defender’s Window",
   "url": "https://openai.com/index/the-defenders-window",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-17T05:30:00.000Z",
   "summary": "AI is reshaping cybersecurity for attackers and defenders alike. Learn how OpenAI is strengthening its defenses and what security teams can do now.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OpenAI joins PORTS-Pike project",
   "url": "https://openai.com/index/openai-joins-ports-pike-project",
   "source": "OpenAI",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-17T05:00:00.000Z",
   "summary": "OpenAI joins PORTS-Pike project, expanding community investment and supporting thousands of Southern Ohio jobs",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Distributed Layerwise Offload: Scaling Toward 200B+ DiT Models Efficiently in vLLM-Omni",
   "url": "https://vllm.ai/blog/2026-08-17-distributed-layerwise-offload",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-17T00:00:00.000Z",
   "summary": "Distributed Layerwise Offload shards and streams DiT weights across devices, serving a measured 124 GB Cosmos3 model on 64 GB HBM and estimating a path toward 200B+ models.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "DeepSeek V4 Pro 0813 vs Claude Fable 5 on DeepSWE: Cost, Coding, and Routing",
   "url": "https://www.together.ai/blog/deepseek-v4-pro-0813-vs-claude-fable-5-on-deepswe-cost-coding-and-routing",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-08-17T00:00:00.000Z",
   "summary": "We ran 904 DeepSWE rollouts on DeepSeek V4 Pro 0813 and Claude Fable 5. Fable leads pass@1 at 90x the cost; Pro wins pass@4, and a Pro-first cascade hits 82.7%.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A/B test models in production",
   "url": "https://www.together.ai/blog/a-b-test-models-in-production",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-08-17T00:00:00.000Z",
   "summary": "Shadow traffic proves a candidate is operationally sound. It can't tell you if users like it better. Run the split at the endpoint instead of in your app code.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Memory Instruction Scheduling for Lock-Stepped Kernels on AMD Instinct™ MI300X: Introducing the Series",
   "url": "https://rocm.blogs.amd.com/software-tools-optimization/scheduling_memory_ops_gfx942/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-17T00:00:00.000Z",
   "summary": "This post introduces a multi-part study of how instruction scheduling can influence the behavior of memory operations within loop iterations in GPU kernels where multiple waves execute in a steady-state lock-stepping manner. In this post, we establish the motivating tiled GEMM kernel, shared vocabulary, and ATT-based methodology; upcoming posts in the series will analyze specific VMEM and LDS scheduling bottlenecks in detail. GPUs can deliver very high memory throughput in general, but that potential is sometimes hard to realize in practice. Lock-stepping behavior, commonly seen in AI kernels ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Bring Claude Code On‑Prem with AMD Instinct GPUs",
   "url": "https://rocm.blogs.amd.com/software-tools-optimization/claude-code-onprem/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-17T00:00:00.000Z",
   "summary": "Agentic coding has become an indispensable part of modern software development. Tools like Claude Code don’t just autocomplete lines — they read entire codebases, plan and execute multi-file refactors, run tests, interpret failures, and iterate autonomously until a task is done. Developers who adopt these workflows report dramatic reductions in time spent on boilerplate, debugging, and context-switching. For engineering teams, agentic coding is fast becoming a competitive necessity rather than a convenience.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Markdown SVG upgrades",
   "url": "https://simonwillison.net/2026/Aug/16/markdown-svg-upgrades",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-16T23:59:37.000Z",
   "summary": "I started building my markdown-svg-renderer tool in May , but I've since added enough features to it that it's worth talking about here again. It's evolved into my ideal tool for sharing Markdown transcripts that include SVG documents. Given my proclivity for drawing pelicans riding bicycles this is a problem that I needed to solve! The tool is very simple. Navigate to markdown-svg-renderer in your browser and paste in some Markdown to see it rendered... or save that Markdown to a CORS-friendly URL or a GitHub Gist and paste in a URL to that document. The URL option will give you a bookmarkabl",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Qwen 3.8 27B is excellent, but it defaults to wildly overthinking things",
   "url": "https://simonwillison.net/2026/Aug/16/qwen-38-27b",
   "source": "Simon Willison",
   "group": "Independent sources",
   "published": "2026-08-16T22:00:39.000Z",
   "summary": "Friday's big release was Qwen 3.8 27B , an Apache 2 licensed 27B parameter vision-capable LLM from Alibaba's Qwen research lab. I've been looking forward to this one: 27B is an excellent size for running a model on a reasonably specced laptop, and its predecessor Qwen 3.6 27B was impressive. Qwen's self-reported benchmarks for this model are eye-opening. They show a boost from both Qwen 3.6 27B and the closed-weight Qwen 3.7-Plus, which was one of Qwen's strongest models of any size as recently as May this year . It will be interesting to hear what independent benchmarks have to say about the ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "React for Agents: Astro Creator Brings Hooks to his Meta-Harness, Flue",
   "url": "https://www.latent.space/p/flue-2",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-15T15:46:22.000Z",
   "summary": "Flue 2 takes its inspiration from React. Creator Fred Schott, of Astro fame, tells Latent Space why he added hooks and why agents are defined by their harnesses.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Pull, Don't Recompute: Peer-to-Peer KV Cache Sharing in llm-d",
   "url": "https://llm-d.ai/blog/p2p-kv-cache-sharing-llm-d",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-08-15T09:00:00.000Z",
   "summary": "When load balancing or serving topology separates a request from its cached prefix, llm-d can move the KV from a peer instead of recomputing it.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GLM-5.3: How Chinese labs keep stride with the frontier",
   "url": "https://www.interconnects.ai/p/glm-53-how-chinese-labs-keep-stride",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-08-14T21:23:35.000Z",
   "summary": "Hint: It’s really not a distillation story.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing support for ClickStack in the ClickHouse Terraform provider",
   "url": "https://clickhouse.com/blog/clickstack-terraform-provider",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-14T16:03:36.000Z",
   "summary": "The ClickHouse Terraform provider now manages ClickStack dashboards, alerts, sources, and webhooks, putting observability config in version control.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to bring your software delivery workflow into GitHub with agent apps",
   "url": "https://github.blog/ai-and-ml/github-copilot/how-to-bring-your-software-delivery-workflow-into-github-with-agent-apps",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-14T16:00:00.000Z",
   "summary": "See how four GitHub agent apps can help you scope, secure, roll out, and ship a feature across the SDLC–all without leaving GitHub. The post How to bring your software delivery workflow into GitHub with agent apps appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Cloudflare detects MCP traffic and helps secure it",
   "url": "https://blog.cloudflare.com/mcp-security-updates",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-14T13:12:12.000Z",
   "summary": "Cloudflare Gateway identifies MCP requests using protocol-level heuristics. Security teams can use that signal to find shadow MCP traffic, enforce Portal-only access for approved servers, and block direct connections on managed network paths.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Secure all your internal vibe-coded applications — in one click",
   "url": "https://blog.cloudflare.com/workers-protected-by-access",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-14T13:00:00.000Z",
   "summary": "Introducing Cloudflare Access for Workers. Attach an Access policy directly to a Worker and it applies everywhere that Worker runs — routes, custom domains, workers.dev, and previews — automatically.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] Cursor's $60B acquisition by SpaceXai closes",
   "url": "https://www.latent.space/p/ainews-cursors-60b-acquisition-by",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-14T06:16:00.000Z",
   "summary": "Congrats to the team!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] Gemini 3.7 Flash brings GDM back to the forefront",
   "url": "https://www.latent.space/p/ainews-gemini-37-flash-brings-gdm",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-14T05:30:39.000Z",
   "summary": "Down, but not out!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Adaptive Verification in vLLM: DSpark confidence-scheduled verification",
   "url": "https://vllm.ai/blog/2026-08-14-dspark-adaptive-verification",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-14T00:00:00.000Z",
   "summary": "Sizing the DSpark draft-verification budget from per-request confidence instead of verifying every drafted token, so one configuration holds the throughput/latency frontier from batch size 1 to 256.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "State of Open Models: Summer 2026 Observations",
   "url": "https://huggingface.co/blog/state-of-open-models-summer-2026",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-14T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Total eclipse of the Internet: traffic impacts in Iceland, Spain, and Portugal",
   "url": "https://blog.cloudflare.com/total-eclipse-internet-traffic-iceland-spain-portugal",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-13T19:58:01.000Z",
   "summary": "Cloudflare's data shows a clear impact on Internet traffic from Iceland to Spain and Portugal, following the path of totality of the total solar eclipse that occurred on August 12, 2026.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Your guide to GitHub Universe 2026 is here: The schedule just launched!",
   "url": "https://github.blog/news-insights/company-news/your-guide-to-github-universe-2026-is-here-the-schedule-just-launched",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-13T19:00:00.000Z",
   "summary": "The GitHub Universe session catalog is live. Explore interactive workshops, community talks, demos, and panels. Plus, register before August 19 to save $300. The post Your guide to GitHub Universe 2026 is here: The schedule just launched! appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Record, train, and deploy from one place with Strands Agents, LeRobot, and Hugging Face Storage Buckets",
   "url": "https://huggingface.co/blog/amazon/strands-lerobot-streaming-data-loop",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-13T17:16:04.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Using BigQuery Graphs with measures for trusted agentic workloads",
   "url": "https://cloud.google.com/blog/products/data-analytics/bigquery-graphs-with-measures-for-trusted-agentic-workloads",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-13T17:00:00.000Z",
   "summary": "When enterprises transition from using simple chat assistants to autonomous, agentic workloads, they quickly run into a hard truth: Agents are prone to inaccurate insights when working with directly raw tables. BigQuery Graph helps organizations move beyond flat, static tables to represent enterprises exactly how they exist in the physical world: as interconnected business entities with real-world dependencies. With the support of measures in BigQuery Graph (preview), we are unifying governed metrics with relationship mapping. This allows your agents to reason across complex dependencies captu",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What’s new in the ClickHouse Grafana plugin",
   "url": "https://clickhouse.com/blog/clickhouse-grafana-plugin-4-20",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-13T16:14:40.000Z",
   "summary": "ClickHouse Grafana plugin 4.20 brings compact query mode, click-to-filter log investigation, guided variable and annotation editors, and OpenTelemetry dashboards",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What 50 open source projects taught us about security in the AI era",
   "url": "https://github.blog/open-source/maintainers/what-50-open-source-projects-taught-us-about-security-in-the-ai-era",
   "source": "GitHub Blog",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-13T16:00:00.000Z",
   "summary": "See how the open source projects in Session 4 of the GitHub Secure Open Source Fund combined AI-assisted workflows, maintainer expertise, GitHub security tools, expert guidance, and funding to improve project security. The post What 50 open source projects taught us about security in the AI era appeared first on The GitHub Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Certificate Transparency Monitoring is now generally available",
   "url": "https://blog.cloudflare.com/certificate-transparency-monitoring-ga",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-13T13:00:00.000Z",
   "summary": "Cloudflare's Certificate Transparency Monitoring is now generally available. The biggest change: we no longer email you about certificates Cloudflare issued for your domain, so when an alert lands in your inbox, it's worth a look.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ClickStack and Hud bring runtime intelligence to AI-powered development",
   "url": "https://clickhouse.com/blog/clickstack-hud-runtime-intelligence",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-13T12:53:04.000Z",
   "summary": "ClickStack and Hud now share trace IDs, pairing service-level observability with function-level runtime forensics so coding agents can assess risky changes before they ship, catch regressions right after deploy, and fix them with real production context.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Maximizing the Power of NVIDIA GB300 NVL72: NVLink Domain-Aware Placement Groups in Ray",
   "url": "https://anyscale.com/blog/nvidia-gb300-nvlink-domain-aware-placement-groups-ray",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-08-13T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] SpaceXAI Grok 4.6 and Grok @Bot",
   "url": "https://www.latent.space/p/ainews-spacexai-grok-46-and-grok",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-13T01:53:47.000Z",
   "summary": "AI teammate category just had its most significant new entrant yet",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What We Learned by Reproducing 2,200 papers from ICML",
   "url": "https://huggingface.co/blog/icml-2026-open-reproductions",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-13T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Production-Ready MXFP4 Online Rotation with Fused Kernels on AMD Instinct™ MI355X",
   "url": "https://rocm.blogs.amd.com/software-tools-optimization/mxfp4-fused-rotation/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-13T00:00:00.000Z",
   "summary": "Serving large language models affordably increasingly depends on low-bit quantization, and MXFP4 is one of the most aggressive options — but the smaller models that need it most rely on online rotation to stay accurate, and that rotation has historically carried a steep latency tax. In this post you will learn how a single fused Gluon (Triton) kernel on AMD Instinct™ MI355X (CDNA4) removes that tax: we walk through the kernel-fusion design, the RS=64 optimization, the GEAK + Hyperloom tuning workflow, and end-to-end measurements showing online-rotation overhead falling from a prohibitive +5–10",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kubernetes on Oxide: How Customer Needs Shaped Our Integrations",
   "url": "https://oxide.computer/blog/kubernetes-on-oxide",
   "source": "Oxide Computer",
   "group": "Others",
   "published": "2026-08-13T00:00:00.000Z",
   "summary": "How customer needs shaped Oxide's Kubernetes integrations.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Databricks Network Configuration delivery to Tens of Millions of Serverless VMs",
   "url": "https://www.databricks.com/blog/databricks-network-configuration-delivery-tens-millions-serverless-vms",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-12T20:00:00.000Z",
   "summary": "SummaryDatabricks' serverless platform launches tens of millions of VMs daily, and...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "From Sync APIs to support for the GPT-5 model series and agentic workflows: What’s new in Azure Content Understanding – August 2026",
   "url": "https://devblogs.microsoft.com/foundry/azure-content-understanding-updates-august-2026",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-12T18:43:01.000Z",
   "summary": "Enterprise content is no longer just something people read. AI apps and agents are only as useful as the information they can understand, yet much of the world’s enterprise knowledge is locked in documents, forms, tables, images, audio, and video. The latest Azure Content Understanding updates help developers turn that content into structured, grounded data […] The post From Sync APIs to support for the GPT-5 model series and agentic workflows: What’s new in Azure Content Understanding – August 2026 appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Azure Content Understanding GPT-5 Series Guide: Model Selection, Grounding Improvements, and Confidence Enhancements",
   "url": "https://devblogs.microsoft.com/foundry/azure-content-understanding-gpt-5-series-guide-model-selection-grounding-improvements-and-confidence-enhancements",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-12T18:42:25.000Z",
   "summary": "Enterprise content is no longer just something people consume. As organizations increasingly rely on AI to extract and act on information from documents, images, audio, and video, Azure Content Understanding is expanding support for the GPT-5 series and improving grounding and confidence to deliver greater flexibility, efficiency, and quality. This expanded model catalog enables organizations […] The post Azure Content Understanding GPT-5 Series Guide: Model Selection, Grounding Improvements, and Confidence Enhancements appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Serve Qwen3.8-2.4T-A95B, a 2.4T-Parameter Model, with Configurable Reasoning on NVIDIA GB300 NVL72",
   "url": "https://developer.nvidia.com/blog/serve-qwen3-8-2-4t-a95b-a-2-4t-parameter-model-with-configurable-reasoning-on-nvidia-gb300-nvl72",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-12T18:23:13.000Z",
   "summary": "Alibaba released the open weights for Qwen3.8-2.4T-A95B (Qwen3.8-Max), its largest open-weight model, bringing near-frontier capabilities to the open...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing OlmoEarth embeddings: Custom embedding exports from OlmoEarth Studio for downstream analysis",
   "url": "https://huggingface.co/blog/allenai/olmoearth-embeddings",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-12T16:14:36.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to Choose Full-Stack Observability for NVIDIA AI Factories",
   "url": "https://developer.nvidia.com/blog/how-to-choose-full-stack-observability-for-nvidia-ai-factories",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-12T16:13:47.000Z",
   "summary": "AI infrastructure spans multiple layers, from compute and networking to storage, orchestration, and applications. When performance degrades, identifying the...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How iFood built its agentic security platform on ClickHouse Cloud",
   "url": "https://clickhouse.com/blog/ifood-agentic-security-platform",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-12T15:29:05.000Z",
   "summary": "iFood rebuilt its in-house security platform on ClickHouse Cloud, getting 9-16x faster queries at 40-50% of the cost and unlocking agentic threat hunts that cut a week of analyst work down to two hours.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we tracked down a 16-year-old SQLite bug",
   "url": "https://tailscale.com/blog/sqlite-wal-reset-bug",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-08-12T14:00:00.000Z",
   "summary": "SQLite corruption? Sorted.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "I wrote an AI textbook — how long until AI can do it better?",
   "url": "https://www.interconnects.ai/p/i-wrote-an-ai-textbook-how-long-until",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-08-12T13:01:16.000Z",
   "summary": "Reflections on AI's writing ability and how AI models get more capable.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How We’re Building Scam Alert on WhatsApp With End-to-End Encryption and Verifiability Guarantees",
   "url": "https://engineering.fb.com/2026/08/12/security/how-were-building-scam-alert-whatsapp",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-12T13:00:28.000Z",
   "summary": "WhatsApp is committed to helping people stay safe while protecting the privacy of their messages. As scam tactics evolve — from impersonation to social engineering to AI-generated lures — we’re always evolving as well, so that our protections stay ahead of scammers while protecting people’s personal messages with end-to-end encryption. Today, we’re sharing an early [...] Read More... The post How We’re Building Scam Alert on WhatsApp With End-to-End Encryption and Verifiability Guarantees appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "[AINews] How to steal a Reasoning Trace",
   "url": "https://www.latent.space/p/ainews-how-to-steal-a-reasoning-trace",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-12T07:11:08.000Z",
   "summary": "Speculative Decoding by any other name would distil as sweet",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Day 0 Support for Qwen3.8-2.4T-A95B on vLLM",
   "url": "https://vllm.ai/blog/2026-08-12-qwen3.8",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-12T00:00:00.000Z",
   "summary": "Day-0 vLLM support for Qwen3.8-2.4T-A95B: a 2.4-trillion-parameter hybrid MoE model served out of the box, with FP8/BF16 checkpoints plus NVFP4 and MXFP4 quantized weights, and co-developed kernels on NVIDIA and AMD hardware.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Qwen3.8-2.4T-A95B now available on Modal",
   "url": "https://modal.com/blog/qwen3-8-2-4t-a95b-now-available-on-modal",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-08-12T00:00:00.000Z",
   "summary": "Qwen3.8-2.4T-A95B by Alibaba, with a 1M token context window, is now available via Modal Auto Endpoints.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Using ODC to Accelerate AMD SFT Training",
   "url": "https://rocm.blogs.amd.com/software-tools-optimization/odc-accelerate-training/README.html",
   "source": "AMD ROCm",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-12T00:00:00.000Z",
   "summary": "Large-scale training spends a surprising share of its wall-clock time waiting instead of computing. Under Fully Sharded Data Parallel (FSDP), every layer ends in a collective all-gather or reduce-scatter, and every collective is a barrier that the whole data-parallel group has to reach together. Feed that machinery variable-length supervised fine-tuning (SFT) data and the picture gets worse: some ranks draw long documents while others draw short ones, so the fast ranks sit idle waiting for the slow ones.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "🔬The BioAI Phase Shift - Matthew McPartlon & Neil Patil, Chai Discovery",
   "url": "https://www.latent.space/p/chai-discovery",
   "source": "Latent Space",
   "group": "Independent sources",
   "published": "2026-08-11T21:03:50.000Z",
   "summary": "Pharma is suddenly paying for Bio × AI tools, and Chai is leading the pack with four deals closed this summer. Cofounder Matt McPartlon and Product leader Neil Patil explain why.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "NVIDIA JetPack 7.2.1 Adds Agentic Video Skills and T3000 Emulation",
   "url": "https://developer.nvidia.com/blog/nvidia-jetpack-7-2-1-adds-agentic-video-skills-and-t3000-emulation",
   "source": "NVIDIA Technical Blog",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-11T19:00:00.000Z",
   "summary": "Video is a core data path across NVIDIA Jetson applications, from robotics and intelligent video analytics to industrial automation, healthcare, media...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What's new in pg_clickhouse v0.10.0: Subqueries, TPC-H Speedups, C Driver, and Aggregates",
   "url": "https://clickhouse.com/blog/pg_clickhouse-whats-new-july-2026",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-11T17:40:40.000Z",
   "summary": "Discover what's new in pg_clickhouse v0.10.0: 16 of 22 TPC-H queries now push down in full, plus a rebuilt C driver and broader aggregate support.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Talking with Synopsys about the Physics of Chip Design at DAC 2026",
   "url": "https://chipsandcheese.com/p/talking-with-synopsys-about-the-physics",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-11T16:52:16.000Z",
   "summary": "Hello you fine Internet folks,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Medium Powers Real-Time Recommendations at 1M OPS",
   "url": "https://www.scylladb.com/2026/08/11/medium-real-time-recommendations",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-08-11T16:40:45.000Z",
   "summary": "Inside Medium’s move from relational features to list features in its ScyllaDB-based feature store “Keep readers reading” is the not-so-simple goal of Medium’s recommendations system. To predict what’s most likely to appeal to a particular reader at any given time, Medium continuously processes user activity signals (stories read, recommendations shown, follows, likes, etc.). It then immediately correlates that with the steady stream of new articles, which is estimated at millions per month. Smart models and good inference logic are required, but that’s not enough. The data must be stored and ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Looker’s semantic layer governs Gemini Enterprise data for user trust",
   "url": "https://cloud.google.com/blog/products/business-intelligence/integrating-looker-and-gemini-enterprise",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-11T16:00:00.000Z",
   "summary": "For organizations deploying AI agents at scale, there’s often a critical divide between structured and unstructured data. While large language models (LLMs) excel at parsing text documents, emails, and PDFs, they can struggle when presented with raw enterprise databases. Meanwhile, standard natural-language-to-SQL (NL2SQL) models often guess how database schemas fit together, which can lead to unpredictable queries, inconsistent metrics, and AI hallucinations that erode user trust. Gemini Enterprise brings the best of Google AI to every employee through an intuitive chat interface that acts as",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Thinking of ACE? We Can Do It with Fewer Tokens",
   "url": "https://huggingface.co/blog/ibm-research/altk-evolve-sldd",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-11T13:37:10.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cloudflare DDoS Threat Report H1 2026: 1 Tbps attacks soar as DNS floods and geopolitical tensions drive a new wave",
   "url": "https://blog.cloudflare.com/ddos-threat-report-2026-h1",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-11T13:00:00.000Z",
   "summary": "In the first half of 2026, Cloudflare detected a 519% surge in hyper-volumetric DDos attacks across its network. These attacks were driven heavily by DNS and CLDAP reflection vectors. This report breaks down how major geopolitical conflicts reshaped the global cyber threat landscape.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "bitdrift brings ClickHouse-powered mobile observability to ClickStack",
   "url": "https://clickhouse.com/blog/bitdrift-clickhouse-mobile-observability-for-clickstack",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-11T00:00:00.000Z",
   "summary": "bitdrift joins ClickHouse’s House Mates program with a mobile observability integration, bringing mobile-native telemetry, tracing, and debugging to ClickStack.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Multi-region high availability for Kafka workloads with a single Stretch Cluster",
   "url": "https://www.redpanda.com/blog/multi-region-high-availability-kafka-stretch-clusters",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-08-11T00:00:00.000Z",
   "summary": "Redpanda Operator 26.2 brings GA Stretch Clusters for multi-region replication, Redpanda Connect pipelines as K8s resources, Gateway API support, and safer rolling restarts.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Multi-Tenant AI Agents: Why Data Isolation Starts at the Database",
   "url": "https://cockroachlabs.com/blog/multi-tenant-ai-agents-why-data-isolation-starts-at-the-database",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-08-11T00:00:00.000Z",
   "summary": "Most SaaS teams shipping agentic features focus on prompt safety and API-layer filtering. But effective AI agent security also depends on...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Everything we launched during Agents Week",
   "url": "https://blog.cloudflare.com/agents-week-review-august-2026",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-10T18:34:10.000Z",
   "summary": "Our latest Agents Week has come to a close. Here’s a recap of all the announcements we made from Wallets to Radar.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Whats new in ClickStack - June + July",
   "url": "https://clickhouse.com/blog/whats-new-in-clickstack-june-2026",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-10T17:53:42.000Z",
   "summary": "Explore ClickStack’s latest upgrades, from richer trace navigation and Prometheus connectivity to smarter dashboards, quieter alerts, faster filtering and support for exponential histogram metrics.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Build Low-Latency Multilingual Voice Agents: Open Weights & Full Deployment Control with NVIDIA Magpie TTS",
   "url": "https://huggingface.co/blog/nvidia/magpie-tts-multilingual-voice-agents",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-10T16:25:36.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How WPP operationalizes platform and data engineering for AI marketing",
   "url": "https://cloud.google.com/blog/products/media-entertainment/how-wpp-operationalizes-platform-and-data-engineering-for-ai-marketing",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-10T16:00:00.000Z",
   "summary": "Between chaotic levels of market fragmentation and economic volatility, marketing and communications agencies can no longer rely on the human intuition they’ve traditionally used to win clients and optimize their ad spend. WPP is replacing that guesswork with an AI-powered view of shifting market dynamics, giving brands predictive certainty that lets them invest with confidence while moving at the speed of the market. That’s the value of WPP Open , its agentic marketing system. But before it could begin applying sophisticated AI models to power those insights, WPP had to overcome a critical en",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Malachyte solves retail’s cold-start problem with managed real-time AI",
   "url": "https://cloud.google.com/blog/products/data-analytics/solving-retails-cold-start-problem-malachytes-recommendation-reinvention",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-10T16:00:00.000Z",
   "summary": "What’s the best way to recommend products to little-known users? We’ve spent our careers trying to solve this problem for major companies like Spotify and Priceline, and it’s why Sidd founded Malachyte , an AI-powered ecommerce recommendation platform. These days, consumers have come to expect content that feels personalized and relevant, and online services competing for their attention have no choice but to do this exceptionally well. Malachyte was inspired by some unique insights into how advanced AI models, and large language models in particular, could be applied in new ways to old challe",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Google named a Leader in The Forrester Wave™: AI Platforms, Q3 2026",
   "url": "https://cloud.google.com/blog/products/ai-machine-learning/google-named-a-leader-in-the-forrester-wave-ai-platforms",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-10T16:00:00.000Z",
   "summary": "At Google Cloud, we help organizations of all sizes build and operationalize complex agentic workflows with total confidence. By combining world-class AI research with an open, fully integrated AI platform, we give customers the flexibility to innovate and the foundation to deliver measurable business value. At the center of it all is Gemini Enterprise, a unified platform designed to power the agentic enterprise , meet builders where they are , and deliver enterprise trust by default . We believe this integrated approach is why Google has been named a Leader in The Forrester Wave ™: AI Platfor",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ScyllaDB Customer Experience Spotlight: Susie Solis",
   "url": "https://www.scylladb.com/2026/08/10/cx-spotlight-susie-solis",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-08-10T14:28:35.000Z",
   "summary": "Meet Susie Solis, a Technical Support Engineer on the Customer Experience team here at ScyllaDB.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "5 useful things you'll learn in my new post-training textbook (shipping now!)",
   "url": "https://www.interconnects.ai/p/5-useful-things-youll-learn-in-my",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-08-10T13:02:42.000Z",
   "summary": "After a few long years of finding time to document my lessons from training open models, my post-training book is done!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Serving the most critical missions: Cloudflare for Government achieves FedRAMP Class D (High) Certified status",
   "url": "https://blog.cloudflare.com/fedramp-class-d-certification",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-10T13:00:00.000Z",
   "summary": "Cloudflare for Government achieves FedRAMP Class D (High) Certified status. We also announce our commitment to pursue DoD IL4 authorization. Cloudflare brings world-class security, performance, and developer products to the public sector.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Making Knowledge Distillation Cheap Enough to Run at Scale",
   "url": "https://huggingface.co/blog/MultiverseComputingCAI/efficient-knowledge-distillation",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-10T10:05:36.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing Day-0 Support for NVIDIA Nemotron 3.5 Lightning on vLLM",
   "url": "https://vllm.ai/blog/2026-08-10-nemotron-3-5-lightning-vllm",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-10T00:00:00.000Z",
   "summary": "How vLLM serves NVIDIA Nemotron 3.5 Lightning with OpenAI-compatible APIs, speculative decoding, and BF16/NVFP4 checkpoints across NVIDIA GPUs and edge systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "OpenHands Joins the Open Secure AI Alliance to Help Build AI Security in the Open",
   "url": "https://www.openhands.dev/blog/open-secure-ai-alliance",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-10T00:00:00.000Z",
   "summary": "OpenHands joins the NVIDIA-led Open Secure AI Alliance to advance open, inspectable infrastructure for building and running secure AI agents.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The 8 Best Automated Vulnerability Remediation Tools in 2026",
   "url": "https://www.openhands.dev/blog/automated-vulnerability-remediation-tools",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-10T00:00:00.000Z",
   "summary": "Compare 8 automated vulnerability remediation tools for 2026 on fixes, prioritization, integrations, and deployment control.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Claude Code vs Cursor: Which AI Coding Tool to Use in 2026",
   "url": "https://www.openhands.dev/blog/claude-code-vs-cursor",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-08-10T00:00:00.000Z",
   "summary": "Claude Code vs Cursor compared on interface, autonomy, models, context, execution, and cost, plus where OpenHands fits.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Meta is back with Muse Glimmer: local, agentic, multimodal, and open source",
   "url": "https://huggingface.co/blog/muse-glimmer",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-10T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Lessons from the hacks",
   "url": "https://www.interconnects.ai/p/lessons-from-the-hacks",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-08-09T14:57:11.000Z",
   "summary": "Musings on model alignment, what determines safety, and where we go from here.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Musinsa scaled its audience engine with ClickHouse Cloud and reduced TCO by 71.4%",
   "url": "https://clickhouse.com/blog/musinsa-customer-data-platform",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-09T00:00:00.000Z",
   "summary": "Musinsa migrated its audience engine from self-hosted ClickHouse to ClickHouse Cloud, cutting storage costs by 86.5% and total cost of ownership by up to 71.4% while simplifying real-time ingestion with ClickPipes.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How and Why Netflix Built a Real-Time Distributed Graph: Part 3 — Querying the graph with gRPC…",
   "url": "https://netflixtechblog.com/how-and-why-netflix-built-a-real-time-distributed-graph-part-3-querying-the-graph-with-grpc-0f3468349607",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-08-07T16:01:02.000Z",
   "summary": "How and Why Netflix Built a Real-Time Distributed Graph: Part 3 — Querying the graph with gRPC execution API Authors: Nilesh Mishra and Ajit Koti This is the third entry of a multi-part blog series describing how we built a Real-Time Distributed Graph (RDG). In Part 1 , we discussed the motivation for creating the RDG and the architecture of the data processing pipeline that populates it. In Part 2 , we discussed how we designed the storage layer to handle billions of nodes and edges while maintaining single-digit-millisecond latency. In Part 3, we will explore how we designed a fast, flexible",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Unveiling good and bad behaviors on the Agentic Internet",
   "url": "https://blog.cloudflare.com/good-and-bad-agentic-behaviors",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-07T13:01:00.000Z",
   "summary": "Cloudflare is shifting bot mitigation from point-in-time Risk assessment to continuous Trust evaluation. Learn how new good and bad behaviors from bots and agents are assessed by our systems, including BotBase and Precursor — and try out our Precursor Trace simulation to see how your own cursor movements would be assessed as human or bot.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Radar Researcher: An AI tool for exploring Internet data in plain language",
   "url": "https://blog.cloudflare.com/introducing-radar-researcher",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-07T13:00:00.000Z",
   "summary": "Cloudflare Radar Researcher is a new AI-powered tool that lets you explore global Internet trends and traffic data using plain language. Built entirely on Cloudflare's Developer Platform, it turns natural language queries into real, interactive charts.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing Cloudflare Ambassadors, Community Engineers, and another $1M in open-source funding",
   "url": "https://blog.cloudflare.com/community-program-refresh",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-07T13:00:00.000Z",
   "summary": "We are launching updated community programs, including Cloudflare Ambassadors and Community Engineers, backed by $1M in open-source funding. Learn how we are supporting maintainers and scaling our developer community.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Unifying Workers AI and AI Gateway into a single AI control plane",
   "url": "https://blog.cloudflare.com/workers-ai-gateway-unification",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-07T13:00:00.000Z",
   "summary": "Cloudflare is unifying AI Gateway and Workers AI into a single control plane, giving developers observability, billing, and dynamic routing across both managed GPUs and external providers. Learn how unified bindings and model-first routing simplify building resilient AI applications.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Efficient Decode Context Parallelism with vLLM for Long Context Workloads",
   "url": "https://vllm.ai/blog/2026-08-07-decode-context-parallelism",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-07T00:00:00.000Z",
   "summary": "Decode Context Parallelism (DCP) in vLLM shards KV cache across GPUs by sequence dimension, enabling 3× higher throughput on long-context agentic workloads compared to standard tensor parallelism.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "DORA Compliance for AI Agents: Database Requirements Before Deployment",
   "url": "https://cockroachlabs.com/blog/dora-database-requirements-ai-agents",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-08-07T00:00:00.000Z",
   "summary": "Financial AI agents operating under DORA, the EU AI Act, and GDPR need infrastructure that supports operational resilience, traceability, and reliable transaction processing.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What's new in ClickHouse Managed Postgres: Customer notifications, better observability, faster backups, extensions, and more",
   "url": "https://clickhouse.com/blog/managed-postgres-notifications-observability-backups",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-06T17:32:00.000Z",
   "summary": "Explore the latest ClickHouse Managed Postgres updates, including proactive notifications, richer observability, faster backups, and expanded extension support.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing OfficeQA Pro V2: A New Benchmark for Enterprise Grounded-Reasoning",
   "url": "https://www.databricks.com/blog/introducing-officeqa-pro-v2-new-benchmark-enterprise-grounded-reasoning",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-06T16:00:00.000Z",
   "summary": "Today, we are releasing OfficeQA Pro V2, a new benchmark designed to evaluate whether...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Your agentic summer: No-cost lessons from Google experts to build and scale agents",
   "url": "https://cloud.google.com/blog/topics/training-certifications/free-gemini-enterrprise-training",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-06T16:00:00.000Z",
   "summary": "I’ve talked to developers, IT leaders, and builders who all ask the same question: How do we actually get agents into production? The answer isn't theoretical — it's hands-on. Whether it’s designing a system that allows your agents to interact with external data sources while maintaining strict security guardrails or creating self-optimizing supply chain workflows or whatever you can think up, we’ve got you covered. That’s why we’ve designed a path to help you take your AI ideas from a rough sketch to fully autonomous agents running in production. This summer, you can harness the same framewor",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Tailscale mitigates the lethal trifecta",
   "url": "https://tailscale.com/blog/aperture-lethal-trifecta",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-08-06T14:00:00.000Z",
   "summary": "Keep AI agents useful without combining their riskiest capabilities.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cloudflare AI Search: give your agents a search engine for your data",
   "url": "https://blog.cloudflare.com/ai-search-easier",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-06T13:00:00.000Z",
   "summary": "AI Search makes search easier than ever, with no Cloudflare primitives to stitch together. Point it at your data to create a search for your own files and websites. We're also sharing a preview of our new pricing model.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The next generation of MCP",
   "url": "https://blog.cloudflare.com/mcp-v2",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-06T13:00:00.000Z",
   "summary": "The next version of MCP has a rewritten, stateless core that just works on Workers. We cover upgrades to the protocol, the new feature lifecycle and SDK migration path, and hear from early adopters already running it in production.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "From ranking to recommended: get your site ready to thrive in the age of AI agents",
   "url": "https://blog.cloudflare.com/aeo",
   "source": "Cloudflare",
   "group": "Large-scale production systems",
   "published": "2026-08-06T13:00:00.000Z",
   "summary": "More than half of requests now come from machines, not people. Agent Readiness shows how well agents can discover and read your site, while Answer Engine Optimization tracks how often AI assistants recommend you.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Mirendil taps AI Hypercomputer TPUs and GPUs for pre- and post-training applications",
   "url": "https://cloud.google.com/blog/topics/startups/mirendil-selects-ai-hypercomputer",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-06T13:00:00.000Z",
   "summary": "Nearly every major AI lab uses Google Cloud infrastructure, including for training of models, inference for agents, and new frontier research. Google Cloud also continues to be the platform of choice for new, high-growth AI startups who are driving much of the industry’s research and innovation. Today, we’re announcing that Mirendil , an exciting frontier AI lab focused on accelerating AI development, will also utilize Google Cloud’s AI Hypercomputer . This includes using a mix of Google’s TPU AI accelerators and full-stack NVIDIA AI infrastructure running on Google Cloud; this purpose-built A",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "vLLM Reaches 25K Total TPS/GPU on Qwen3.5",
   "url": "https://vllm.ai/blog/2026-08-06-qwen35-25k-tps",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-08-06T00:00:00.000Z",
   "summary": "How vLLM reaches 25K total TPS/GPU on Qwen3.5-397B-A17B-NVFP4 with GB200 NVL72 disaggregated serving, Blackwell GDN kernels, HMA cache transfer, async scheduling fixes, and srt-slurm recipes.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Baseten on Hugging Face Inference Providers 🔥",
   "url": "https://huggingface.co/blog/baseten",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-08-06T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "DeepSeek-V4 Flash 0731 vs GPT-5.6 Luna on DeepSWE: Cost and Coding",
   "url": "https://www.together.ai/blog/deepseek-v4-flash-0731-vs-gpt-5-6-luna-on-deepswe-cost-and-coding",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-08-06T00:00:00.000Z",
   "summary": "We ran 900 DeepSWE rollouts on DeepSeek-V4 Flash and GPT-5.6 Luna. Luna leads pass@1 by 14 points; DeepSeek delivers 4.8x the solves per dollar.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ClickHouse Release 26.7",
   "url": "https://clickhouse.com/blog/clickhouse-release-26-07",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-06T00:00:00.000Z",
   "summary": "This release brings speedups for GROUP BY ... ORDER BY ... LIMIT, three JOIN improvements, four vector search improvements, position-aware phrase search, EXPLAIN ANALYZE, unified URL access, and more!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Physical Intelligence unified its robotics data stack with Postgres managed by ClickHouse",
   "url": "https://clickhouse.com/blog/physical-intelligence-rds-to-clickhouse-managed-postgres",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-06T00:00:00.000Z",
   "summary": "Physical Intelligence runs both its OLAP and OLTP workloads on ClickHouse managed Postgres and ClickHouse Cloud",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Database at 550 Kilometers: What Orbital Computing Means for Distributed Databases",
   "url": "https://cockroachlabs.com/blog/orbital-computing-distributed-databases",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-08-06T00:00:00.000Z",
   "summary": "In Ashburn, Virginia, a row of servers draws 40 megawatts from the grid and exhales it as heat.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "From User Sequences to Scaling Laws: A Multi-Stage Architecture for Meta’s Ads Ranking",
   "url": "https://engineering.fb.com/2026/08/05/ml-applications/from-user-sequences-to-scaling-laws-a-multi-stage-architecture-for-metas-ads-ranking",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-05T19:20:20.000Z",
   "summary": "Every day, Meta’s recommendation platforms handle billions of user interactions, generating rich temporal signals that capture individual preferences and intent across products, ads, and content. In our 2024 post on sequence learning for ads recommendations, we showed how modeling the order and timing of user actions (rather than relying on static, manually engineered sparse features) [...] Read More... The post From User Sequences to Scaling Laws: A Multi-Stage Architecture for Meta’s Ads Ranking appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "NVIDIA’s Vera Whitepaper Has a Thread Loose",
   "url": "https://chipsandcheese.com/p/nvidias-vera-whitepaper-has-a-thread",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-05T19:19:22.000Z",
   "summary": "Hello you fine Internet folks,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What is WAL backpressure, and why does ClickHouse Managed Postgres need it?",
   "url": "https://clickhouse.com/blog/wal-backpressure-clickhouse-managed-postgres",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-05T15:58:59.000Z",
   "summary": "How ClickHouse Managed Postgres uses WAL-aware backpressure to slow client writes, protect disk space, and let the archiver recover.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling Vision-Heavy Kimi-VL with Heterogeneous E/PD on llm-d and SGLang",
   "url": "https://llm-d.ai/blog/scaling-vision-heavy-kimi-vl-with-heterogeneous-epd",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-08-05T09:00:00.000Z",
   "summary": "Benchmarking disaggregated VLM serving of Kimi-VL-A3B-Instruct with 4 Intel Arc Pro B60 vision encoder and 1 NVIDIA H200 language model worker, delivering 2.4x-2.8x higher throughput and 69%-80% lower TTFT than collocated serving.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Your AI Performance Stack is Fireworks Models with Voyage AI embeddings",
   "url": "https://fireworks.ai/blog/voyage-ai-models-now-on-fireworks",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-08-05T00:00:00.000Z",
   "summary": "Fireworks is the first and only dedicated inference platform Voyage AI by MongoDB has partnered with. The full Voyage lineup now runs natively on Fireworks: the Voyage 4 family, voyage-multimodal-3.5, and rerank-2.5.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Database-Aware Kubernetes Operations Are Here",
   "url": "https://cockroachlabs.com/blog/cockroachdb-kubernetes-operator",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-08-05T00:00:00.000Z",
   "summary": "The production-proven K8s operator is now available with full lifecycle automation, zero-downtime migration, as well as the flexibility and scalability to fit how your team actually works.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Mercado Libre rebuilt its observability platform on ClickHouse Cloud with 50x faster trace queries",
   "url": "https://clickhouse.com/blog/mercado-libre-observability-on-clickhouse-cloud",
   "source": "ClickHouse",
   "group": "Others",
   "published": "2026-08-04T19:10:00.000Z",
   "summary": "How Mercado Libre rebuilt its observability platform on ClickHouse Cloud, cutting trace query times from over five minutes to about four seconds (a 50x speedup) with up to 89% compression while ingesting 400 million spans per minute.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Target is enhancing retail discovery and cutting database maintenance by 50% with Spanner Graph",
   "url": "https://cloud.google.com/blog/topics/retail/how-target-rebuilt-retail-discovery-with-spanner-graph",
   "source": "Google Cloud AI Infrastructure",
   "group": "Hardware and accelerator stack",
   "published": "2026-08-04T16:00:00.000Z",
   "summary": "In today’s retail environment, shoppers expect highly personalized product discovery experiences and conversational assistance that feels genuine, natural, and genuinely helpful. Today, successful product discovery is about understanding semantic meaning and the rich, connected relationships between products, categories, and guest intent. It is no longer just about keywords and basic browsing. At Target, this work is handled by our Guest Product Confidence platform team. They are responsible for building the features that establish trust and guide purchasing decisions, such as ratings, reviews",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "DDIA 2nd Edition Excerpt: On Scalability",
   "url": "https://www.scylladb.com/2026/08/04/ddia-2nd-edition-excerpt-on-scalability",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-08-04T12:59:50.000Z",
   "summary": "Martin Kleppmann and Chris Riccomini's scalability considerations for designing data-intensive applications -- from the second edition of the Designing Data-Intensive Applications book",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Bringing serverless functions closer to the speed of wire",
   "url": "https://modal.com/blog/bringing-serverless-functions-closer-to-the-speed-of-wire",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-08-04T00:00:00.000Z",
   "summary": "Modal’s Function Call data path is now >50ms faster. Our new routing layer is geographically distributed, so you can further reduce your network overhead.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GEM Training: How Meta Doubled the Efficiency of Its LLM-Scale Ads Foundation Model",
   "url": "https://engineering.fb.com/2026/08/03/ml-applications/training-gem-at-llm-scale-meta-ads-recommendation-foundation-model",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-08-03T18:00:17.000Z",
   "summary": "Meta’s Generative Ads Recommendation Model (GEM), the foundation model behind ads recommendations across Instagram and Facebook, now trains at LLM scale on several thousand of the latest-generation GPUs. This post goes into the details on how we achieved: doubling end-to-end (E2E) training efficiency to 20–25% Model FLOPs Utilization (MFU) while scaling training FLOPs 4x in [...] Read More... The post GEM Training: How Meta Doubled the Efficiency of Its LLM-Scale Ads Foundation Model appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing our Artifacts Hub and Adoption Dashboard",
   "url": "https://www.interconnects.ai/p/introducing-our-artifacts-hub-and",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-08-03T14:03:08.000Z",
   "summary": "Scaling our curation and measurement of the open ecosystem.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Out-of-band Policy Engine: governance AI agents can't ignore",
   "url": "https://www.redpanda.com/blog/agentic-ai-needs-out-of-band-governance",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-08-03T00:00:00.000Z",
   "summary": "Somewhere in your company, right now, someone is building an agent. Here’s how the latest release of Redpanda’s Agentic Data Plane makes it safe to run them.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agentic kill switch is a database problem. So we built Redpanda SQL",
   "url": "https://www.redpanda.com/blog/query-real-time-analytics-google-cloud",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-08-03T00:00:00.000Z",
   "summary": "Agentic governance needs a new kind of database, so we built Redpanda SQL. Now available on both AWS and Google Cloud.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Arm’s Cortex A55",
   "url": "https://chipsandcheese.com/p/arms-cortex-a55",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-08-02T22:02:44.000Z",
   "summary": "Arm’s 5-series cores are meant for tasks where performance barely matters, but power and area efficiency are top priorities.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Latest open artifacts (#23): Laguna S2.1, Inkling, & Kimi K3 show the utility of open models on the Pareto frontier",
   "url": "https://www.interconnects.ai/p/latest-open-artifacts-23-laguna-s21",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-08-02T13:01:40.000Z",
   "summary": "Capacity to train strong models is proliferating.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K3: the complete developer guide",
   "url": "https://www.together.ai/blog/kimi-k3-guide",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-08-01T00:00:00.000Z",
   "summary": "Kimi K3 is the first open 3T-class model. See how it benchmarks, what it costs, and how to call it on the Together AI API, with copy-paste code examples.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Tailscale didn’t stop the Hugging Face intrusion",
   "url": "https://tailscale.com/blog/hugging-face-intrusion",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-07-31T18:30:00.000Z",
   "summary": "Tailscale wasn’t exploited. We still should have stopped the intrusion.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Modeling Device Capabilities for Analytics",
   "url": "https://netflixtechblog.com/modeling-device-capabilities-for-analytics-e7607acebde8",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-07-31T16:01:02.000Z",
   "summary": "by Aarti Laddha , Richard Diaz-Cool , Rishika Idnani , Venkatesh Selveraj Netflix supports a vast and evolving set of features and content types, ranging from 4K streaming and immersive audio to live streaming and cloud gaming, across a diverse ecosystem of devices. However, not all devices are created equal. Hardware limitations such as available RAM, CPU cores, display capabilities, or platform support mean that some features cannot be supported on certain device models. To ensure the best possible user experience, we rely on a deep understanding of device capabilities. We have invested in b",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "RosaicLabs, Atom RTL, and 32-Tile AMX: Trying to Piece Together a x86 Puzzle",
   "url": "https://chipsandcheese.com/p/rosaiclabs-atom-rtl-and-32-tile-amx",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-07-31T01:39:23.000Z",
   "summary": "Hello you fine Internet folks,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Autoscaling endpoints for LLM inference",
   "url": "https://www.together.ai/blog/autoscaling-endpoints-for-llm-inference",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-31T00:00:00.000Z",
   "summary": "GPU utilization can read healthy while your queue backs up, and a new replica takes minutes to warm. Here's how to pick autoscaling metrics, tune scale-up/down windows, and budget for cold starts on dedicated inference.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GenRec: Towards LLM-Native Recommendation at Netflix",
   "url": "https://netflixtechblog.com/genrec-towards-llm-native-recommendation-at-netflix-f20be6f643e3",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-07-30T20:10:15.000Z",
   "summary": "Authors: Ying Li , Arjun Rao , Shradha Sehgal Introduction Recommendations sit at the heart of the Netflix experience. Our current production models rely on thousands of hand‑crafted features over users, items, and interactions, along with specialized architectures for sequence modeling, feature interactions, and multi‑task objectives. This stack has evolved over many years to support diverse content types (movies, series, games, live, podcasts) and product surfaces, but its complexity makes it costly to onboard new use cases: adding a content type or surface can require significant feature en",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agentic media buying cannot scale without the right foundation. See how buyers and sellers get there on Databricks.",
   "url": "https://www.databricks.com/blog/agentic-media-buying-cannot-scale-without-right-foundation-see-how-buyers-and-sellers-get",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-30T19:10:00.000Z",
   "summary": "The bottleneck in media buying today isn't talent, it's coordinationEvery day, billions...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GPU Management: Why Idle GPUs Are the New Grounded Aircraft",
   "url": "https://huggingface.co/blog/Dharma-AI/gpu-management",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-07-30T15:09:09.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Anyscale signs definitive agreement to join Nscale",
   "url": "https://anyscale.com/blog/anyscale-signs-definitive-agreement-to-join-nscale",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-07-30T05:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Optimizing vLLM on Arm CPUs",
   "url": "https://vllm.ai/blog/2026-07-29-optimizing-vllm-on-arm-cpus",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-29T00:00:00.000Z",
   "summary": "An overview of Arm CPU enablement and inference performance optimizations in vLLM.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A note on the Hugging Face agent incident",
   "url": "https://modal.com/blog/a-note-on-the-hugging-face-agent-incident",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-07-29T00:00:00.000Z",
   "summary": "Hugging Face published a technical timeline of a recent agent intrusion. Modal's platform and isolation were not compromised in this incident.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Together AI announces strategic partnership with Moonshot AI to natively serve Kimi models",
   "url": "https://www.together.ai/blog/together-ai-announces-strategic-partnership-with-moonshot-ai-to-natively-serve-kimi-models",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-29T00:00:00.000Z",
   "summary": "Together AI partners with Moonshot AI to natively serve Kimi models, starting with the 2.8T parameter Kimi K3, with day zero access and post-training.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Configuring Dedicated Model Inference",
   "url": "https://www.together.ai/blog/configuring-dedicated-model-inference",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-29T00:00:00.000Z",
   "summary": "The three-part resource model behind Together AI Dedicated Model Inference—endpoints, deployments, configs—and how capacity-aware routing ties them together.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ThunderAgent: 2x Faster Agentic Inference for Synthetic Data Generation at Scale",
   "url": "https://www.together.ai/blog/thunderagent",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-29T00:00:00.000Z",
   "summary": "ThunderAgent is a program-aware scheduler for agentic inference. By treating each agent workflow as a schedulable program, it eliminates KV cache thrashing to deliver more than 2x single-node throughput and near-linear multi-node scaling.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Fine-Tune Your Own Embedding Model from an LLM — for the Price of a Coffee",
   "url": "https://fireworks.ai/blog/fine-tuning-your-own-embeddings-model",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-07-29T00:00:00.000Z",
   "summary": "Use state-of-the-art, open-source LLMs and image models at blazing fast speed, or fine-tune and deploy your own at no additional cost with Fireworks AI!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "10X more data, same 4 seconds: single-query scaling in Redpanda SQL on 1TB",
   "url": "https://www.redpanda.com/blog/single-query-scaling-redpanda-sql",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-07-29T00:00:00.000Z",
   "summary": "A benchmark on how a single analytical query behaves in Redpanda SQL as the dataset and the cluster grow.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LTM Partners with Cognition To Reduce Cyber Risk in Financial Services",
   "url": "https://cognition.com/blog/ltm-cognition-partnership",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-28T17:00:00.000Z",
   "summary": "LTM has partnered with Cognition to deploy Devin, the AI software engineer, across its global client base and cybersecurity practice serving over 260…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Oxide joins Anthropic's Project Glasswing",
   "url": "https://oxide.computer/blog/oxide-anthropic-project-glasswing",
   "source": "Oxide Computer",
   "group": "Others",
   "published": "2026-07-28T17:00:00.000Z",
   "summary": "Oxide joins Anthropic's Project Glasswing, applying Claude Mythos 5 to find and patch vulnerabilities across its open-source stack, firmware to network.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Lessons Learned from Real-World NoSQL Database Migrations",
   "url": "https://www.scylladb.com/2026/07/28/lessons-learned-from-real-world-nosql-database-migrations",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-07-28T13:49:40.000Z",
   "summary": "Discover the strategies, challenges, and trade-offs teams faced in a few real-world migrations to ScyllaDB",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Parallel All the Way Down: Beyond Single-Token Generation with Speculative Decoding",
   "url": "https://vllm.ai/blog/2026-07-28-speculators-parallel-drafting",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-28T00:00:00.000Z",
   "summary": "Speculators and vLLM now support P-EAGLE, DFlash, and DSpark — three parallel drafting algorithms that move beyond sequential token generation to deliver faster, simpler, and more scalable speculative decoding for LLM serving.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Coding Agents and Technical Debt",
   "url": "https://www.openhands.dev/blog/coding-agents-and-technical-debt",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-07-28T00:00:00.000Z",
   "summary": "Say you forked the OpenHands app twelve months ago and never merged from upstream. You would now be 2,600 merged PRs behind, including 866 bug fixes you do not have.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to Consolidate Your Database Stack for Production AI",
   "url": "https://cockroachlabs.com/blog/database-consolidation-production-ai",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-28T00:00:00.000Z",
   "summary": "When builders ship AI-powered applications, the data layer quietly becomes the hardest part of the stack.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The EU Digital Product Passport: a traceability deadline",
   "url": "https://www.databricks.com/blog/eu-digital-product-passport-traceability-deadline",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-27T23:00:00.000Z",
   "summary": "Informational only, not legal advice. Confirm all regulatory details against the...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AMD Advancing AI 2026: Talking CDNA5 with AMD’s Alan Smith",
   "url": "https://chipsandcheese.com/p/amd-advancing-ai-2026-talking-cdna5",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-07-27T19:43:53.000Z",
   "summary": "Hello you fine Internet folks,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "NVIDIA Cosmos-H-Dreams: Bringing Real-Time Generative Simulation to Surgical Robotics",
   "url": "https://huggingface.co/blog/nvidia/cosmos-h-dreams",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-07-27T09:32:20.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K3 Is Here: Efficient Day-0 Support on vLLM",
   "url": "https://vllm.ai/blog/2026-07-27-k3",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-27T00:00:00.000Z",
   "summary": "vLLM delivers day-0 Kimi K3 serving with hybrid KDA prefix caching, DSpark speculative decoding, production-scale disaggregation, and optimized kernels across NVIDIA and AMD GPUs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Anatomy of a Frontier Lab Agent Intrusion: A Technical Timeline of the July 2026 Incident",
   "url": "https://huggingface.co/blog/agent-intrusion-technical-timeline",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-07-27T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K3 by Moonshot now available on Modal",
   "url": "https://modal.com/blog/kimi-k3-by-moonshot-now-available-on-modal",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-07-27T00:00:00.000Z",
   "summary": "Kimi K3, a 2.8 trillion parameter multimodal model by Moonshot, along with a custom-trained DFlash speculator, is now available on Modal.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K3 on Fireworks: Frontier Intelligence You Can Own",
   "url": "https://fireworks.ai/blog/kimik3-on-fireworks",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-07-27T00:00:00.000Z",
   "summary": "Kimi K3 on Fireworks brings frontier-level intelligence to an open-weight 2.8-trillion parameter model, delivering matching performance to closed alternatives like Opus 5 at up to 5x lower cost per task. Hosted with US-based serverless endpoints and Zero Data Retention (ZDR), Fireworks gives developers full roadmap ownership, day-0 API access, and pay-per-token serverless fine-tuning. With #1 global rankings in front-end code, top-tier legal analysis, and specialized long-horizon agentic capabilities, Kimi K3 on Fireworks establishes a new standard for high-performance, enterprise-grade AI inf",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K3 vs GPT-5.6 Sol on DeepSWE: Cost, Coding, and Routing",
   "url": "https://www.together.ai/blog/kimi-k3-vs-gpt-5-6-sol-on-deepswe-cost-coding-and-routing",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-26T00:00:00.000Z",
   "summary": "We ran 904 DeepSWE rollouts on Kimi K3 and GPT-5.6 Sol. Sol leads pass@1; Kimi K3 wins pass@4 at 2.8x the solves per dollar, and routing between them reaches ~85.6%.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Trilogy’s Playbook for Open-Weight Cybersecurity with Kimi K3",
   "url": "https://fireworks.ai/blog/trilogy-coe-playbook-for-cybersecurity",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-07-26T00:00:00.000Z",
   "summary": "Trilogy’s new AI Cybersecurity Playbook centers on Kimi K3, leveraging open weights to ensure defenders have reliable, high-volume inference options. By moving beyond closed models, organizations gain the deployment flexibility—from managed endpoints to dedicated capacity—needed to scale security workflows like repository auditing and alert triage. This approach keeps infrastructure under engineering control while integrating powerful model reasoning directly into the defensive stack.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Open Models Are Closing the Coding Gap",
   "url": "https://www.openhands.dev/blog/open-models-are-closing-the-coding-gap",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-07-24T00:00:00.000Z",
   "summary": "Kimi K3 and Thinking Machines' Inkling shipped open weights this week. What self-hosting a frontier-class coding model buys you on your own infrastructure, and what it costs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Open Models Are Closing the Coding Gap",
   "url": "https://www.openhands.dev/blog/open-models-are-closing-the-coding-gap-2",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-07-24T00:00:00.000Z",
   "summary": "Kimi K3 and Thinking Machines' Inkling shipped open weights this week. What self-hosting a frontier-class coding model buys you on your own infrastructure, and what it costs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K3 vs Claude Fable 5 on DeepSWE: Cost and Coding",
   "url": "https://www.together.ai/blog/kimi-k3-vs-claude-fable-5-on-deepswe-cost-and-coding",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-24T00:00:00.000Z",
   "summary": "We ran 452 DeepSWE rollouts on Kimi K3 and Claude Fable 5. Fable leads pass@1 by 1.4 points; Kimi K3 wins pass@4 and delivers 2.8x the solves per dollar.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Turn And Face The Strange",
   "url": "https://fly.io/blog/kurt-scott-money-sprites",
   "source": "Fly.io",
   "group": "Others",
   "published": "2026-07-24T00:00:00.000Z",
   "summary": "We’re Fly.io, a public cloud platform that is both our favorite way to put an app on the Internet and our favorite way to safely let a frontier agent coding harness cook. This is a post about our company, the future, and Sprites, which are computers for agents that you can check out right now. This is a complicated post. So I need you to promise me something: if you read past this introduction, you’ll read the whole rest of the way through. It’s an honor thing. A couple months back, Theo Browne ran a video rating the “best place to host a new application in 2026” . Theo tends to say nice thing",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AMD’s Instinct MI455X: Aiming for the Sun",
   "url": "https://chipsandcheese.com/p/amds-instinct-mi455x-aiming-for-the",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-07-23T17:37:10.000Z",
   "summary": "Editor’s Note (7/25/2026): The article has been edited with more information about the L2 behavior along with the bandwidth of the die to die interface.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Welcoming The Interaction Company",
   "url": "https://cognition.com/blog/interaction",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-23T17:00:00.000Z",
   "summary": "Today, we’re welcoming The Interaction Company of California, the makers of Poke, to Cognition.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing the Anyscale Physical AI Skill",
   "url": "https://anyscale.com/blog/introducing-the-anyscale-physical-ai-skill",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-07-23T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "From Day 0 to Production SLAs: Serving GLM-5.2 on 24 NVIDIA B300 GPUs with vLLM",
   "url": "https://vllm.ai/blog/2026-07-23-glm-5.2-nvfp4-b300-pd",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-23T00:00:00.000Z",
   "summary": "How we took GLM-5.2-NVFP4 from 40 ms to 17 ms mean TPOT on 24 B300 GPUs with vLLM: P/D disaggregation, MTP speculative decoding, Model Runner V2, and the SLA-first trade-offs behind the final configuration.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing vLLM AFD Plugin: Disaggregating Attention and FFN for Flexible MoE Serving",
   "url": "https://vllm.ai/blog/2026-07-23-vllm-afd-plugin",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-23T00:00:00.000Z",
   "summary": "What vLLM AFD Plugin adds to the vLLM ecosystem: Attention–FFN disaggregation for MoE serving, GPU and Ascend NPU backends, connector-based execution, and graph and ubatching support.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Bringing Nunchaku 4-bit Diffusion Inference to Diffusers",
   "url": "https://huggingface.co/blog/nunchaku-diffusers",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-07-23T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The production platform for open-weight AI inference",
   "url": "https://www.together.ai/blog/the-production-platform-for-open-weight-ai-inference",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-23T00:00:00.000Z",
   "summary": "Run open models in production with full control over performance, cost, and quality. Deploy in minutes, roll out safely, and scale to your SLOs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Drag, drop, done: a visual composer for Redpanda Connect",
   "url": "https://www.redpanda.com/blog/redpanda-connect-pipeline-builder",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-07-23T00:00:00.000Z",
   "summary": "A first look at the new visual Pipeline Builder for Redpanda Connect (preview), plus a roundup of recent CDC and connector updates.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Simplify AI agent orchestration with Lakebase Postgres",
   "url": "https://www.databricks.com/blog/simplify-ai-agent-orchestration-lakebase-postgres",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-22T23:00:00.000Z",
   "summary": "IntroductionTraditionally, auditing is a tedious process that often requires detailed...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Offloading I/O to Dedicated Cores: An Asymmetric io_uring Backend for Seastar and ScyllaDB",
   "url": "https://www.scylladb.com/2026/07/22/asymmetric-io_uring-backend-seastar",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-07-22T18:03:10.000Z",
   "summary": "We moved low-level I/O execution off application cores to dedicated networking cores using Seastar’s new asymmetric_io_uring backend. Explore the architecture design, trade-offs, and benchmark results.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cognition Signs MOU with U.S. Department of Energy to Join The Genesis Mission",
   "url": "https://cognition.com/blog/cognition-doe-genesis-mission",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-22T17:00:00.000Z",
   "summary": "Cognition has signed a memorandum of understanding with the U.S. Department of Energy to join the Genesis Mission, a national initiative described as…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building Agents that Act on Your Behalf with Toolboxes in Foundry",
   "url": "https://devblogs.microsoft.com/foundry/building-agents-that-act-on-your-behalf-with-toolboxes-in-foundry",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-07-22T15:00:29.000Z",
   "summary": "How Toolboxes in Foundry simplify user delegation At some point, many agents move from answering questions to taking action. And when it does, the question becomes: whose identity is it acting with? Imagine you are building an internal employee agent. It needs to call a private, Entra-protected MCP server for orders, and it also needs to use Microsoft’s managed Work IQ MCP server to reason over the employee’s Microsoft 365 […] The post Building Agents that Act on Your Behalf with Toolboxes in Foundry appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Open models recap: more on Kimi K3, Qwen 3.8, Xi's WAIC speech, distillation, the open-closed gap, and what's next",
   "url": "https://www.interconnects.ai/p/open-models-recap-more-on-kimi-k3",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-07-22T14:09:04.000Z",
   "summary": "A podcast with Florian Brand.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Serving GLM-5.2 for Agentic Workloads on llm-d",
   "url": "https://llm-d.ai/blog/serving-glm-5-2-agentic-workloads-on-llm-d",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-07-22T09:00:00.000Z",
   "summary": "Cost-efficient, responsive GLM-5.2 agentic inference on H200 with llm-d, using KV-cache-aware routing, tiered prefix-cache offloading, and multi-token prediction.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A Preview of Production-Scale Kimi K3 Support on vLLM",
   "url": "https://vllm.ai/blog/2026-07-22-kimi-k3-preview",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-22T00:00:00.000Z",
   "summary": "A preview of production-scale Kimi K3 support in vLLM, including KDA-aware prefix caching, fused kernels, optimized MXFP4 MoE, multimodal integration, and initial NVIDIA and AMD paths.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agent Sandboxing: What OpenAI got wrong with the HuggingFace hack",
   "url": "https://www.openhands.dev/blog/agent-sandboxing-what-openai-got-wrong-with-the-huggingface-hack",
   "source": "OpenHands",
   "group": "Agent frameworks and runtimes",
   "published": "2026-07-22T00:00:00.000Z",
   "summary": "OpenAI accidentally hacked HuggingFace during a security exercise. Learn what they got wrong, and how you can avoid the same mistakes.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Why R&D Data Belongs in the Lakehouse - and Why Agents Need It There",
   "url": "https://www.databricks.com/blog/why-rd-data-belongs-lakehouse-and-why-agents-need-it-there",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-21T19:45:00.000Z",
   "summary": "The setupAt cellcentric, a joint venture of Daimler Truck and Volvo Group, we develop...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What Matters Most for NoSQL Migrations",
   "url": "https://www.scylladb.com/2026/07/21/what-matters-most-for-nosql-migrations",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-07-21T13:44:46.000Z",
   "summary": "How to prioritize the things that matter most for planning, executing and de-risking your NoSQL database migration",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Engineering quality compounds",
   "url": "https://tailscale.com/blog/welcome-mike-shaver",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-07-21T13:00:00.000Z",
   "summary": "Mike Shaver joins Tailscale to scale engineering quality.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Beyond a Single Model: Building Mixture-of-Models Systems with vLLM Semantic Router",
   "url": "https://vllm.ai/blog/2026-07-21-vllm-sr-new-chapter-mom",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-21T00:00:00.000Z",
   "summary": "vLLM Semantic Router is expanding from intelligent routing into a system for building, evaluating, and running Mixture-of-Models.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Grabette: an open system to record robot-manipulation data",
   "url": "https://huggingface.co/blog/grabette",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-07-21T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Devin Outposts on Modal",
   "url": "https://modal.com/blog/devin-outposts-run-devin-in-modal-sandoxes",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-07-21T00:00:00.000Z",
   "summary": "Devin, built by Cognition, is an AI software engineer: it plans, writes, tests, and ships code semi-autonomously. With Outposts, Devin can now run its work in Modal sandboxes.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K3 vs Fable 5: Benchmarks, Cost, and When to Route",
   "url": "https://fireworks.ai/blog/kimik3-fable",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-07-21T00:00:00.000Z",
   "summary": "Compare Kimi K3 to Fable 5 on coding benchmarks, costs, and more. See results across 1,000+ agentic tasks, pricing breakdowns, and when to route to each.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Deploy agents you can trust with centralized AI governance",
   "url": "https://www.redpanda.com/blog/deploy-agents-you-can-trust-with-centralized-ai-governance",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-07-21T00:00:00.000Z",
   "summary": "Discover why organizations are struggling to deploy and scale agentic systems, and how a centralized AI governance platform can help you trust and scale agents.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling document classification to 100k+ labels",
   "url": "https://www.databricks.com/blog/scaling-document-classification-100k-labels",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-20T18:15:00.000Z",
   "summary": "Across Databricks, thousands of customers build production workloads that map freeform...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Welcoming TierZero to Cognition",
   "url": "https://cognition.com/blog/welcoming-tierzero",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-20T17:00:00.000Z",
   "summary": "Anhang and Yun have gone deep on everything that keeps software running once it ships, and we are excited to bring their work on automations into Devin.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K3: The open-weights escalation",
   "url": "https://www.interconnects.ai/p/kimi-k3-the-open-weights-escalation",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-07-20T15:48:28.000Z",
   "summary": "The global implications on the AI ecosystem.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How our universal content processing platform Riviera evolved for AI and beyond",
   "url": "https://dropbox.tech/infrastructure/how-our-universal-content-processing-platform-riviera-evolved-for-ai-and-beyond",
   "source": "Dropbox Tech",
   "group": "Large-scale production systems",
   "published": "2026-07-20T15:00:00.000Z",
   "summary": "Riviera is the Dropbox content processing platform that’s been iteratively improving content transformation in our products for roughly a decade.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "LLVM Divination of GFX1251’s Differences",
   "url": "https://chipsandcheese.com/p/llvm-divination-of-gfx1251s-differences",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-07-20T09:23:59.000Z",
   "summary": "Hello you fine Internet folks, this article is a sequel to the Scrying the AMD GFX1250 LLVM Tea Leaves article where we are going to look at the differences between GFX1250 and GFX1251.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Together AI and Y Combinator partner to launch the first dedicated GPU cluster for the YC community",
   "url": "https://www.together.ai/blog/together-yc-gpu-cluster",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-20T00:00:00.000Z",
   "summary": "No more two-year compute contracts. Together AI and YC just gave YC startups a faster way to get GPUs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scrying the AMD GFX1250 LLVM Tea Leaves",
   "url": "https://chipsandcheese.com/p/scrying-the-amd-gfx1250-llvm-tea",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-07-19T03:21:03.000Z",
   "summary": "In just a few short days, AMD will be showing off their brand new MI400 series of Datacenter Accelerators at their Advancing AI event but before that event comes, we thought it would be fun to attempt to scry the tea leaves that are LLVM commits to see what we can ascertain about this next generation of AMD accelerator.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "In-House LLM Serving at Netflix",
   "url": "https://netflixtechblog.com/in-house-llm-serving-at-netflix-a5a8e799ea2c",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-07-17T21:32:39.000Z",
   "summary": "By AI Platform’s Model Runtime team and Inference team Introduction Most organizations consume LLMs through hosted APIs. Netflix went further — we run the full stack ourselves, from model deployment through inference, inside our existing production environment rather than a separate ML silo. Some of those decisions weren’t obvious, and a few revealed their trade-offs only under production load. This post focuses on the choices where alternatives were seriously considered: engine selection, model packaging, API surface design, deployment strategy, and output constraints enforcement. The goal is",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building a soccer coaching app on Databricks",
   "url": "https://www.databricks.com/blog/building-soccer-coaching-app-databricks",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-17T17:00:00.000Z",
   "summary": "Tracking data is now the richest signal in sport, but the real gap is turning the...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What a $20 coding subscription actually buys",
   "url": "https://tailscale.com/blog/aperture-ai-passthrough-subscription-costs",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-07-17T16:00:00.000Z",
   "summary": "Someone's subsidizing your coding agent. Aperture shows whether it's you.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Why Your AI Agent Has an Identity Problem",
   "url": "https://cockroachlabs.com/blog/ai-agent-identity-security",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-17T00:00:00.000Z",
   "summary": "The agent your team just shipped to external users has the ability to read customer records, execute transactions, and call external APIs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Remotely access Home Assistant via Tailscale for free",
   "url": "https://tailscale.com/blog/remotely-access-home-assistant",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-07-16T20:00:00.000Z",
   "summary": "Take Home Assistant beyond your home network with Tailscale.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What happens in the milliseconds after you tap pay",
   "url": "https://www.databricks.com/blog/what-happens-milliseconds-after-you-tap-pay",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-16T18:00:00.000Z",
   "summary": "You're standing at the register. You tap your card. A tiny spinner appears for maybe...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Newer Models, Same Advantage",
   "url": "https://huggingface.co/blog/Dharma-AI/newer-models-same-advantages",
   "source": "Hugging Face",
   "group": "Inference companies",
   "published": "2026-07-16T11:49:48.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Apache Spark 4.2",
   "url": "https://www.databricks.com/blog/introducing-apache-spark-42",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-16T07:00:01.000Z",
   "summary": "IntroductionApache Spark 4.2 moves more of the modern data and AI stack into the...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Keeping vLLM Production Quality: A Look Inside CI, Benchmarking, and the Release Process",
   "url": "https://vllm.ai/blog/2026-07-16-keeping-vllm-production-quality",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-16T00:00:00.000Z",
   "summary": "How vLLM maintains production quality with extensive CI across diverse accelerators, nightly performance benchmark and accuracy evaluation, and a two-week release process.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling to 1 million concurrent sandboxes in seconds",
   "url": "https://modal.com/blog/scaling-to-1-million-concurrent-sandboxes-in-seconds",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-07-16T00:00:00.000Z",
   "summary": "How (and why) we built a scheduling system that can scale to 1 million concurrent sandboxes (per workspace) in seconds.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What does 99.9% uptime mean for inference?",
   "url": "https://www.together.ai/blog/99-9-uptime-for-inference",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-16T00:00:00.000Z",
   "summary": "Reliability numbers are easy to publish. We break down what 99%, 99.9%, and 99.99% uptime actually require, the failure domains each tier has to survive, and the questions to ask any inference provider before you commit.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Exploring Hierarchical Interest Representation For Meta Ads Deep Funnel Optimization",
   "url": "https://engineering.fb.com/2026/07/15/ai-research/exploring-hierarchical-interest-representation-for-meta-ads-deep-funnel-optimization",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-15T17:00:52.000Z",
   "summary": "Hierarchical Interest Representation is a research area for Meta Ads. We’re exploring an upstream representation layer over the universe of Ads entities – users, advertisers, products, services – learning unified embeddings that connect users’ inferred interests with the breadth of what advertisers offer in their deep funnel ads. The innovations in Hierarchical Interest Representation are [...] Read More... The post Exploring Hierarchical Interest Representation For Meta Ads Deep Funnel Optimization appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "TML Inkling on vLLM: Day-0 Support with Optimized Performance",
   "url": "https://vllm.ai/blog/2026-07-15-inkling",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-15T00:00:00.000Z",
   "summary": "vLLM brings day-0 support to TML Inkling, a 1T-parameter multimodal model, with MTP, long-context serving, parallelism, and up to 380 tokens per second per user on NVIDIA GB200 GPUs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Inkling by Thinking Machines now available on Modal",
   "url": "https://modal.com/blog/inkling-by-thinking-machines-labs-now-available-on-modal",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-07-15T00:00:00.000Z",
   "summary": "Inkling, a general-purpose multimodal model by Thinking Machines, along with a custom trained DFlash speculator, is now available on Modal.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Together AI brings Thinking Machines Lab’s new model Inkling on day 0",
   "url": "https://www.together.ai/blog/together-ai-brings-thinking-machines-labs-new-model-inkling-on-day-0",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-15T00:00:00.000Z",
   "summary": "Together AI offers day zero access to Inkling, Thinking Machines Lab's multimodal mixture-of-experts model for text, image, and audio reasoning.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "New in Together GPU Clusters: Reliability and control for production GPU clusters",
   "url": "https://www.together.ai/blog/new-in-together-gpu-clusters-reliability-and-control-for-production-gpu-clusters",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-15T00:00:00.000Z",
   "summary": "See how Together AI is improving production GPU clusters with passive health checks, node repair, stronger Slurm reliability, OIDC, and startup scripts.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Fireworks Secures $1.5 Billion in Series D Funding",
   "url": "https://fireworks.ai/blog/series-d-announcement",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-07-15T00:00:00.000Z",
   "summary": "Use state-of-the-art, open-source LLMs and image models at blazing fast speed, or fine-tune and deploy your own at no additional cost with Fireworks AI!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How BetterTracker Replaced Its Vector Store with CockroachDB",
   "url": "https://cockroachlabs.com/blog/bettertracker-replaced-vector-store-cockroachdb",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-15T00:00:00.000Z",
   "summary": "BetterTracker eliminated a standalone vector database by running OLTP and vector search together on CockroachDB—cutting costs, complexity, and compliance risk in one move.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "One Year of Building Together",
   "url": "https://cognition.com/blog/one-year-of-building-together",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-14T16:00:00.000Z",
   "summary": "One year ago, Cognition and Windsurf came together over the craziest 72 hours of our lives. Here’s what we’ve built together since.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Build Durable Chat Memory for RAG Using ScyllaDB and LangChain",
   "url": "https://www.scylladb.com/2026/07/14/durable-chat-memory-for-rag-scylladb-and-langchain",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-07-14T12:54:49.000Z",
   "summary": "How to replace LangChain's in-memory chat history with ScyllaDB — so your RAG chatbot retains context across restarts and scales across replicas",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Enhancing Ray Cluster Stability With Resource Isolation",
   "url": "https://anyscale.com/blog/enhancing-ray-cluster-stability-with-resource-isolation",
   "source": "Anyscale",
   "group": "Serving engines and runtimes",
   "published": "2026-07-14T09:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Is x86 ready to ACE it?",
   "url": "https://chipsandcheese.com/p/is-x86-ready-to-ace-it",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-07-14T01:23:37.000Z",
   "summary": "CPU designs must evolve to keep up with changing workloads.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "vLLM x TileRT: Specialized Decode for Latency-Critical Serving",
   "url": "https://vllm.ai/blog/2026-07-14-vllm-tilert-pd",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-14T00:00:00.000Z",
   "summary": "vLLM prefill paired with TileRT decode through vLLM V1's connector interface: a specialized, latency-optimized decode engine that coexists with native vLLM decode behind one shared serving layer, with zero changes to vLLM.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Full CDC semantics land in the Iceberg output for Redpanda Connect",
   "url": "https://www.redpanda.com/blog/cdc-semantics-iceberg-redpanda-connect",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-07-14T00:00:00.000Z",
   "summary": "Full CDC semantics in Iceberg mean the lakehouse reflects what the source database looks like right now, not what it looked like during last night’s batch window. Get to know the latest addition to Redpanda Connect.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing the CIS Benchmark for CockroachDB v25.x",
   "url": "https://cockroachlabs.com/blog/cis-benchmark-cockroachdb-security",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-14T00:00:00.000Z",
   "summary": "We're proud to announce that the Center for Internet Security (CIS) has published the CIS CockroachDB v25.x Benchmark.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building Service Topology at Scale: Architecture, Challenges, and Lessons Learned",
   "url": "https://netflixtechblog.com/building-service-topology-at-scale-architecture-challenges-and-lessons-learned-f4b792f3f0d8",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-07-13T22:44:11.000Z",
   "summary": "By Parth Jain , Rakesh Sukumar , Yingwu Zhao , Renzo Sanchez-Silva & Nathan Fisher A deep dive into the engineering challenges of building a real-time service dependency map at Netflix scale: from streaming architectures and distributed aggregation pipelines to time-travel queries and the methodology that made it work. Introduction In our first post , we introduced the problem: engineers at Netflix needed a unified, real-time view of service dependencies to troubleshoot faster, understand blast radius, and navigate our distributed architecture. We described our multi-source approach, combining",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Ultra-Fast Anomaly Detection using Apache Spark Real-Time Mode",
   "url": "https://www.databricks.com/blog/ultra-fast-anomaly-detection-using-apache-spark-real-time-mode",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-13T17:25:00.000Z",
   "summary": "This post establishes a reusable pattern for operational workloads that genuinely move the needle: fraud detection...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Making Fable Cheaper Than Opus",
   "url": "https://cognition.com/blog/making-fable-cheaper-than-opus",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-13T17:00:00.000Z",
   "summary": "Fable 5 costs twice as much per token as Opus 4.8. But when we ran both models on FrontierCode 1.1 using our new Fusion architecture, Fable cost less —…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Modernizing the Meta Ads Service With an Open-Source Kernel Scheduler",
   "url": "https://engineering.fb.com/2026/07/13/ml-applications/modernizing-the-meta-ads-service-with-an-open-source-kernel-scheduler",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-13T16:00:50.000Z",
   "summary": "TL; DR At Meta’s scale, a few milliseconds of latency degradation can have a significant negative impact on ads performance. When a Linux kernel upgrade risked regressing latency across Meta’s ad serving fleet, we turned to sched_ext — the upstream, BPF-based extensible scheduling framework — to build a scheduling policy customized to the Ads delivery [...] Read More... The post Modernizing the Meta Ads Service With an Open-Source Kernel Scheduler appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Devin is Now FedRAMP High In-Process, Unlocking Autonomous AI Engineering for Federal Agencies",
   "url": "https://cognition.com/blog/devin-fedramp-high-in-process",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-13T11:00:00.000Z",
   "summary": "Cognition’s entire platform is now FedRAMP Class D (High) In-Process and listed on the FedRAMP Marketplace, giving engineering teams working in federal…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "EAGLE3 Speculative Decoding on AMD Instinct GPUs: Training and Serving with vLLM and AMD Quark",
   "url": "https://vllm.ai/blog/2026-07-13-eagle-3-amd-instinct",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-13T00:00:00.000Z",
   "summary": "How AMD Quark trains, quantizes, and serves EAGLE3 speculative-decoding drafts with vLLM on AMD Instinct GPUs, delivering up to 2.00x throughput gains for Kimi-K2.5 and 1.79x for MiniMax-M2.5.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to Automate CockroachDB Operations with AI Using Cockroach University",
   "url": "https://cockroachlabs.com/blog/ai-cockroachdb-training-courses",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-13T00:00:00.000Z",
   "summary": "AI agents aren't just changing how applications interact with databases, they're changing how teams operate them.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "6 months to live for open models",
   "url": "https://www.interconnects.ai/p/6-months-to-live-for-open-models",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-07-12T16:47:42.000Z",
   "summary": "The most serious test to date of open source AI’s viability is happening right now.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "vime + ROCm: End-to-End RL Post-Training on AMD Instinct™ GPUs",
   "url": "https://vllm.ai/blog/2026-07-10-vime-rocm",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-10T00:00:00.000Z",
   "summary": "Announcing ROCm support for vime, now running end-to-end on AMD Instinct MI355X GPUs with prebuilt container.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Optimizing MiniMax M3 Sparse Attention on NVIDIA Blackwell",
   "url": "https://fireworks.ai/blog/kernel-optimization-for-minimax-m3-on-nvidia-blackwell",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-07-10T00:00:00.000Z",
   "summary": "Fireworks built a KV-stationary sparse-attention kernel for MiniMax M3 on NVIDIA Blackwell (SM100), reaching ~980 TFLOP/s: 1.9–2.4× a query-stationary baseline and ~1.6× open-source MSA. The post walks through the Q-outer vs KV-outer design space, an I/O roofline with the reuse crossover (nsb/N < 2.85), and the kernel optimizations: contiguous partial-O stores with a gathered combine, a 3-warp cp.async query gather, a shortened softmax→store pipeline, load-balanced split-Q scheduling, D2H elimination, and a C++ AOT dispatch backend, plus full-module benchmarks against FlashInfer and MiniMax's ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Modal is a computer",
   "url": "https://modal.com/blog/modal-is-a-computer",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-07-09T00:00:00.000Z",
   "summary": "What is Modal? A machine that runs programs of arithmetic & logic operations on information.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Gumloop Scaled Open-Weight Model Usage 7x in 3 Weeks with Fireworks",
   "url": "https://fireworks.ai/blog/gumloop",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-07-09T00:00:00.000Z",
   "summary": "Gumloop scaled its open-weight model usage 7x in three weeks by partnering with Fireworks. By optimizing its agent harness and switching to models like GLM-5.2, Gumloop achieved up to 72% cost savings while maintaining production-level quality and reliability, proving that open-weight models are now viable for complex, real-world AI workloads.1",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What is an Agentic Data Plane?",
   "url": "https://www.redpanda.com/blog/what-is-an-agentic-data-plane",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-07-09T00:00:00.000Z",
   "summary": "The Agentic Data Plane is the governance and runtime layer that connects your AI agents to everything they act on. Learn what it does, why existing tools can't replace it, and what to look for in an enterprise-grade one.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Navigating a Synapse Migration to Databricks",
   "url": "https://www.databricks.com/blog/navigating-synapse-migration-databricks",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-08T19:00:00.000Z",
   "summary": "Azure Synapse has served as a reliable foundation for SQL analytics at scale, and...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "SWE-1.7: Frontier Intelligence at a Fraction of the Cost",
   "url": "https://cognition.com/blog/swe-1-7",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-08T17:02:00.000Z",
   "summary": "Today, we’re launching SWE-1.7, the most capable model we’ve trained so far. It reaches frontier-level intelligence at a much lower cost, advancing the…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Measuring the Trustworthiness of Open-Source-Derived Models",
   "url": "https://cognition.com/blog/measuring-open-source-model-trustworthiness",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-08T17:01:00.000Z",
   "summary": "We built an evaluation suite to assess model trustworthiness. Our results indicate that models developed from open-source models can be trusted, provided…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ScyllaDB Is Now Supported in MCP Toolbox for Databases",
   "url": "https://www.scylladb.com/2026/07/08/scylladb-is-now-supported-in-mcp-toolbox-for-databases",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-07-08T12:01:04.000Z",
   "summary": "Connect your AI agents to ScyllaDB using the new ScyllaDB integration in MCP Toolbox for Databases",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Open, convenient and predictable: Introducing Provisioned Throughput",
   "url": "https://www.together.ai/blog/provisioned-throughput",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-08T00:00:00.000Z",
   "summary": "Provisioned Throughput gives you reserved inference capacity for frontier open models like MiniMax M3 and GLM-5.2. Token-based pricing, a 99% uptime SLA, and up to 90% lower cost than proprietary APIs. No GPU-hour math, no infrastructure to manage.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Best Open Source LLMs in 2026: We Reviewed 7 Models",
   "url": "https://fireworks.ai/blog/best-open-source-llms",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-07-08T00:00:00.000Z",
   "summary": "Compare Kimi K3 with GLM 5.2 and other leading LLMs using current benchmark results and deployment options, with weight and license status called out.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Durable Execution with DBOS and CockroachDB",
   "url": "https://cockroachlabs.com/blog/embedded-durable-execution-dbos-cockroachdb",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-08T00:00:00.000Z",
   "summary": "Modern AI applications are no longer single-shot inference calls. They are long-running agents that plan, act, observe, and retry across time.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agent Memory at Monster Scale with Mem0 and ScyllaDB Cloud",
   "url": "https://www.scylladb.com/2026/07/07/agent-memory-at-monster-scale-with-mem0-and-scylladb-cloud",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-07-07T13:08:54.000Z",
   "summary": "Combine Mem0’s memory management with ScyllaDB’s persistence features to deploy large-scale AI agents",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What’s New in Microsoft Foundry | June 2026",
   "url": "https://devblogs.microsoft.com/foundry/whats-new-in-microsoft-foundry-june-2026",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-07-07T07:10:25.000Z",
   "summary": "Claude is now generally available in Microsoft Foundry. Here's everything else that shipped between Build 2026 and the end of June — autopilot agents, expanded Toolboxes and Routines, Agent Optimizer's private preview, and more. The post What’s New in Microsoft Foundry | June 2026 appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Stream Oracle changes to ClickHouse in real time",
   "url": "https://www.redpanda.com/blog/stream-oracle-changes-to-clickhouse-real-time-cdc",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-07-07T00:00:00.000Z",
   "summary": "A hands-on walkthrough of the new Oracle input in Redpanda Connect. No Debezium, no Kafka Connect runtime, no JVM.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How CockroachDB and IBM LinuxONE Rockhopper 5 Power Resilient AI Infrastructure",
   "url": "https://cockroachlabs.com/blog/ibm-linuxone-rockhopper-5-ai-infrastructure",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-07T00:00:00.000Z",
   "summary": "Why does AI require a new approach to infrastructure?",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ScyllaDB vs Aerospike, Wide-Column vs. Key/Value",
   "url": "https://www.scylladb.com/2026/07/06/scylladb-vs-aerospike-wide-column-vs-key-value",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-07-06T13:13:23.000Z",
   "summary": "Wide-column flexibility doesn’t have to come at the expense of performance -- see where the two models differ, where each one wins, and why you no longer have to choose",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to price serverless GPUs",
   "url": "https://modal.com/blog/how-to-price-serverless",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-07-06T12:00:00.000Z",
   "summary": "To compare rates for serverless and reserved GPUs, look at your application's peak-to-average ratio.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "vLLM × HPC-Ops: High-Performance Attention and MoE Backends from Tencent Hunyuan",
   "url": "https://vllm.ai/blog/2026-07-06-vllm-hpc-ops",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-06T00:00:00.000Z",
   "summary": "How HPC-Ops integrates Hopper-optimized attention and FP8 MoE backends into vLLM for Tencent Hunyuan Hy3, improving mixed-length decode, MoE latency, TTFT, and TPOT on NVIDIA H20.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing the Devin Security Vulnerability Remediation Program",
   "url": "https://cognition.com/blog/devin-security-vulnerability-remediation-program",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-02T17:00:00.000Z",
   "summary": "The Devin Security Vulnerability Remediation Program helps organizations clear their vulnerability backlog and set up continuous remediation.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Real-time Change Data Capture with Redpanda Connect and MySQL",
   "url": "https://www.redpanda.com/blog/real-time-cdc-my-sql",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-07-02T00:00:00.000Z",
   "summary": "A step-by-step tutorial on how to stream every insert, update, and delete from your database using MySQL and a faster, simpler alternative to Kafka Connect.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "RoachFest London 2026 Recap: An Astronaut, AI Agents, Amazingness",
   "url": "https://cockroachlabs.com/blog/roachfest-london-2026-recap",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-02T00:00:00.000Z",
   "summary": "RoachFest London 2026 has been and gone, and I'm still buzzing.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we keep GPUs reliable across Databricks AI",
   "url": "https://www.databricks.com/blog/how-we-keep-gpus-reliable-across-databricks-ai",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-01T23:00:00.000Z",
   "summary": "Distributed GPU training has become routine across the industry. Teams now train...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Devin Security Swarm",
   "url": "https://cognition.com/blog/introducing-devin-security-swarm",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-07-01T17:00:00.000Z",
   "summary": "Devin Security Swarm finds vulnerabilities across the codebase, validates exploitability at runtime, and ships remediation PRs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Meta’s AI Storage Blueprint at Scale",
   "url": "https://engineering.fb.com/2026/07/01/data-infrastructure/metas-ai-storage-blueprint-at-scale",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-07-01T16:00:36.000Z",
   "summary": "Over the past several years, model capabilities and training dataset sizes have experienced exponential growth. During the past year or so, the time between new-frontier-model releases has gone down from months to weeks. Reliable and fast access to storage is important to both the speed and computational cost of this AI innovation. If AI is [...] Read More... The post Meta’s AI Storage Blueprint at Scale appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cutting P99 Latency 1000X During Connection Storms by Hardening ScyllaDB Admission Control",
   "url": "https://www.scylladb.com/2026/07/01/cutting-p99-during-connection-storms",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-07-01T13:35:06.000Z",
   "summary": "ScyllaDB successfully mitigated performance-degrading connection storms by optimizing caching, throttling, and password hashing to achieve a 1000x reduction in tail latency",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Experience and Lessons Learned from Serving Multi-Stage Qwen3-Omni in vLLM-Omni",
   "url": "https://vllm.ai/blog/2026-07-01-qwen3-omni-optimization",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-07-01T00:00:00.000Z",
   "summary": "How vLLM-Omni serves and optimizes Qwen3-Omni with staged Thinker-Talker-Code2Wav execution, batching, CUDA Graphs, async chunk, async output, replicas, hot-path cleanup, and perf validation.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Multi-token Residual Prediction",
   "url": "https://modal.com/blog/multi-token-residual-prediction",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-07-01T00:00:00.000Z",
   "summary": "One tiny module, two ways to win on Diffusion LMs",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing our $800M Series C to accelerate the shift to open-source AI",
   "url": "https://www.together.ai/blog/announcing-our-series-c",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-07-01T00:00:00.000Z",
   "summary": "We raised $800M to accelerate the shift to open-source AI. Here's why the economics of closed models don't scale, and what we're building next.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Why Agent Loops Fail in Production (and the Database Patterns That Fix Them)",
   "url": "https://cockroachlabs.com/blog/agent-loops-production-database-patterns",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-07-01T00:00:00.000Z",
   "summary": "Agent loops fail in production for reasons that have little to do with the model, and everything to do with what happens to their state between iterations.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "From monolith to Lakebase to LTAP: rethinking the database from storage up",
   "url": "https://www.databricks.com/blog/lakebase-ltap-rethinking-database-storage",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-06-30T18:05:48.000Z",
   "summary": "When I started my PhD at UC Berkeley 16 years ago, my advisor told me: \"OLTP databases...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "10 Years of Meta’s Commitment to Python",
   "url": "https://engineering.fb.com/2026/06/30/open-source/10-years-of-metas-commitment-to-python",
   "source": "Meta Engineering",
   "group": "Large-scale production systems",
   "published": "2026-06-30T16:00:46.000Z",
   "summary": "This year marks Meta’s 10th consecutive year as a sponsor of the Python Software Foundation (PSF), the charitable organization dedicated to advancing, supporting, and protecting the open-source Python programming language and the community that sustains it. Python is one of the world’s most influential programming languages, and we use it across our engineering stack, from [...] Read More... The post 10 Years of Meta’s Commitment to Python appeared first on Engineering at Meta .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to audit what your AI agents are accessing",
   "url": "https://tailscale.com/blog/aperture-audit-AI-agents",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-06-30T14:00:00.000Z",
   "summary": "AI agents do a lot. Aperture keeps the audit trail.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How ScyllaDB’s Trie-Based Index Delivers Up to 3X More Throughput",
   "url": "https://www.scylladb.com/2026/06/30/trie-index-3x-more-throughput",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-06-30T13:03:08.000Z",
   "summary": "By transitioning from separate summary and index files to a prefix tree, we optimized cache efficiency, reduced disk I/O, and reduced memory overhead",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Anthropic integration with Modal brings scalable compute to Claude Science",
   "url": "https://modal.com/blog/modal-integration-brings-scalable-compute-to-claude-science",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-30T00:00:00.000Z",
   "summary": "Announcing our integration with Claude Science, bringing Modal's elastic compute to researchers when they need it.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Together AI at ICML 2026: frontier research across the full stack",
   "url": "https://www.together.ai/blog/icml-2026",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-06-30T00:00:00.000Z",
   "summary": "Nine papers at ICML 2026 across the full stack. The research that becomes the Together platform. Find us at booth B714 in Seoul.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GLM 5.2 Fast is live on Fireworks",
   "url": "https://fireworks.ai/blog/glm-5p2-fast",
   "source": "Fireworks AI",
   "group": "Inference companies",
   "published": "2026-06-30T00:00:00.000Z",
   "summary": "Use state-of-the-art, open-source LLMs and image models at blazing fast speed, or fine-tune and deploy your own at no additional cost with Fireworks AI!",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Redpanda Cloud Topics rethinks Kafka compaction",
   "url": "https://www.redpanda.com/blog/how-redpanda-cloud-topics-rethinks-kafka-compaction",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-30T00:00:00.000Z",
   "summary": "Compaction can overwhelm poorly sized Kafka clusters, leading to full disks and maxed-out CPUs. Learn how Redpanda's Cloud Topics architecture redesigns compaction to cut redundant work, reduce cloud storage costs, and preserve the Kafka semantics you rely on.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ScyllaDB 2026.2: DynamoDB Streams and Vector Search, Trie Indexes, and Strongly Consistent Tables",
   "url": "https://www.scylladb.com/2026/06/29/scylladb-2026-2",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-06-29T19:18:20.000Z",
   "summary": "ScyllaDB 2026.2 brings a combination of GA new features, exciting experimental features, and multiple stability and external use case improvements.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Devin Fusion",
   "url": "https://cognition.com/blog/devin-fusion",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-06-29T18:00:00.000Z",
   "summary": "Introducing Devin Fusion: a hybrid-model harness that keeps frontier-level coding intelligence while cutting costs with sidekick agents and dynamic…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GenPage: Towards End-to-End Generative Homepage Construction at Netflix",
   "url": "https://netflixtechblog.com/genpage-towards-end-to-end-generative-homepage-construction-at-netflix-77146fba8a08",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-06-29T13:01:02.000Z",
   "summary": "Authors: Lequn Wang , J iangwei Pan , and Linas Baltrunas Figure 1. Autoregressive homepage generation. GenPage builds a Netflix homepage one row or entity at a time, each one conditioned on what’s already on the page and the user’s context. Introduction The Netflix homepage is the first thing users see when they open the app and the primary way they discover content to enjoy. Almost every part of it is personalized, including which rows appear, which entities show up within those rows, and how everything is arranged on the page. Constructing that homepage is a genuinely hard problem. It is no",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Micro-Agent: Beat Frontier Models with Collaboration inside Model API",
   "url": "https://vllm.ai/blog/2026-06-29-micro-agent-frontier-models",
   "source": "vLLM",
   "group": "Serving engines and runtimes",
   "published": "2026-06-29T00:00:00.000Z",
   "summary": "How vLLM Semantic Router turns vllm-sr/auto into a bounded micro-agent runtime for Confidence, Ratings, ReMoM, Fusion, Workflows, and benchmark-shaped collaboration.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "PostgreSQL-Compatible Databases for AI at Scale: What to Evaluate from Day One",
   "url": "https://cockroachlabs.com/blog/postgresql-compatible-databases-ai-scale",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-06-29T00:00:00.000Z",
   "summary": "The database you choose at the start of an AI project is the one you'll be living with, or paying to escape, for years.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Latest open artifacts (#22): Zyphra, Cohere, and Poolside are expanding the breadth of the ecosystem",
   "url": "https://www.interconnects.ai/p/artifacts-22-zyphra-cohere-and-poolside",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-28T17:03:07.000Z",
   "summary": "An assessment of the open ecosystem and the motivations behind releasing models",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "RL Post-Training: Co-Operative Time-Slicing with llm-d",
   "url": "https://llm-d.ai/blog/rl-post-training-co-operative-time-slicing",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-06-28T16:41:00.000Z",
   "summary": "Introducing Co-operative Time-Slicing to eliminate idle accelerators in distributed RL post-training loops.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Databricks is turning video into searchable, actionable intelligence",
   "url": "https://www.databricks.com/blog/how-databricks-turning-video-searchable-actionable-intelligence",
   "source": "Databricks Engineering",
   "group": "Large-scale production systems",
   "published": "2026-06-26T20:30:00.000Z",
   "summary": "A utility company deploys drones to inspect hundreds of miles of power lines. A police...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "A no-nonsense explainer to Agentic AI",
   "url": "https://tailscale.com/blog/agents-are-coming",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-06-26T17:00:00.000Z",
   "summary": "Cut through the buzzwords with a clear explanation of the agentic AI stack.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "TOP500 at ISC’26: We have a New Number 1",
   "url": "https://chipsandcheese.com/p/top500-at-isc26-we-have-a-new-number",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-06-25T21:15:33.000Z",
   "summary": "Hello you fine Internet folks,",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we used DSPy to turn AI evaluations into better responses in Dash chat",
   "url": "https://dropbox.tech/machine-learning/how-we-turned-ai-evaluations-into-better-responses-in-dash-chat",
   "source": "Dropbox Tech",
   "group": "Large-scale production systems",
   "published": "2026-06-25T16:30:00.000Z",
   "summary": "We used DSPy to improve LLM judges and optimize our chat experience, creating an evaluation-driven feedback loop that produced better outputs.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Stop sharing access secrets—try Border0 + Tailscale for free",
   "url": "https://tailscale.com/blog/border0-free-trial",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-06-25T14:00:00.000Z",
   "summary": "Border0 ties every connection to a real person, securing databases, Kubernetes, SSH, and more.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Modal's serverless Servers",
   "url": "https://modal.com/blog/serverless-servers",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-25T00:00:00.000Z",
   "summary": "A deep dive inside our new ultra-low-latency primitive.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kafka's log compaction corrupts data. Here's how we fixed it",
   "url": "https://www.redpanda.com/blog/kafka-log-compaction-bug-fix-streaming",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-25T00:00:00.000Z",
   "summary": "There's a problem with Apache Kafka's log compaction. Here's what we found, how to reproduce it, and how we solved it in Redpanda.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Riding the Raft to Strong Consistency in ScyllaDB",
   "url": "https://www.scylladb.com/2026/06/24/raft-strong-consistency",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-06-24T16:51:27.000Z",
   "summary": "How ScyllaDB is using per-tablet Raft groups to bring strong consistency to data, without sacrificing the parallelism that makes it fast",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "llm-d v0.8: From Platform to Control Plane",
   "url": "https://llm-d.ai/blog/llm-d-v0.8-from-platform-to-control-plane",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-06-24T09:00:00.000Z",
   "summary": "llm-d v0.8 graduates Flow Control, Batch Gateway, and multi-modal serving to production, extends beyond Kubernetes with RL/Slurm support, and aligns with upstream vLLM — sharpening the project's identity as an inference control plane for any environment.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Achieve state-of-the-art inference latencies with speculative decoding",
   "url": "https://modal.com/blog/achieve-sota-specdec",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-24T00:00:00.000Z",
   "summary": "How Modal and Decagon worked together to cut inference latency - and you can too.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Networking for Distributed Inference in llm-d",
   "url": "https://llm-d.ai/blog/networking-for-distributed-inference-llm-d",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-06-23T09:00:00.000Z",
   "summary": "How llm-d transfers the KV Cache between prefill and decode workers — NIXL's pluggable backend architecture, the new UCCL backend, head-to-head benchmarks of UCCL/UCX/Mooncake over RDMA and TCP, and preflight tooling for catching networking misconfigurations before serving traffic.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Toward More Controllable AI Video Editing: An Early Research Exploration at Netflix",
   "url": "https://netflixtechblog.com/toward-more-controllable-ai-video-editing-an-early-research-exploration-at-netflix-eb8160ed60a2",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-06-23T00:31:01.000Z",
   "summary": "By Zhuoning Yuan , Ta-Ying Cheng , Benjamin Klein , Bahareh Azarnoush Introduction At Netflix, we build technology to help storytellers bring their creative visions to life and to help members discover the stories they love. To connect stories with diverse audiences around the world, we produce promotional assets, including trailers, teasers, and social short‑form videos, that build on and elevate the original footage. Through close collaboration with the teams crafting these assets, we identified a recurring gap in current tools. Transforming raw footage into a polished final asset often requ",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Modal Auto Endpoints: Optimized inference you actually own",
   "url": "https://modal.com/blog/introducing-auto-endpoints",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-23T00:00:00.000Z",
   "summary": "LLM inference at SotA speeds and Modal quality, now available to everyone.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ParallelKernelBench: Frontier LLMs can't write fast multi-GPU kernels (yet)",
   "url": "https://www.together.ai/blog/parallelkernelbench",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-06-23T00:00:00.000Z",
   "summary": "ParallelKernelBench tests whether LLMs can write fast multi-GPU CUDA kernels across 87 real workloads. The best model solves under a third, but a few generated kernels beat any public implementation.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Bridge Queries in Redpanda SQL",
   "url": "https://www.redpanda.com/blog/bridge-queries-in-redpanda-sql",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-23T00:00:00.000Z",
   "summary": "Stop choosing between fresh data and robust Parquet files. Redpanda SQL bridge queries let you query live streaming topics and historical Iceberg tables together, without the compaction overhead.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Netflix Simplified Batch Compute with Kueue",
   "url": "https://netflixtechblog.com/how-netflix-simplified-batch-compute-with-kueue-87860682629c",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-06-22T21:35:01.000Z",
   "summary": "By Alvin Bao , Alex Petrov , Jennifer Lai , Aidan Sherr , and Samartha Chandrashekar As a part of the journey to transition Netflix’s compute infrastructure to be more Kubernetes-native, we have leaned into incorporating components from the Kubernetes ecosystem into our container platform Titus . One example of this is our use of Kueue , a cloud-native job queueing system for batch workloads, which has largely replaced the custom queuing and scheduling logic in our homegrown managed batch solution Compute Managed Batch (CMB). In this post, we’ll give an overview of what motivated the migration",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "GLM-5.2 is the step change for open agents",
   "url": "https://www.interconnects.ai/p/glm-52-is-the-step-change-for-open",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-22T14:52:45.000Z",
   "summary": "A capability threshold I've been carefully monitoring.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Unpacking sandbox startup latency: why started ≠ ready",
   "url": "https://modal.com/blog/unpacking-sandbox-startup-latency",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-22T00:00:00.000Z",
   "summary": "Building performant sandbox systems goes way beyond the initial container boot. Here, we unpack what that means, and discuss some tools to help you manage the entire lifecycle.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Data Canary: How Netflix Validates Catalog Metadata",
   "url": "https://netflixtechblog.com/the-data-canary-how-netflix-validates-catalog-metadata-18b699d58e36",
   "source": "Netflix TechBlog",
   "group": "Large-scale production systems",
   "published": "2026-06-19T23:54:17.000Z",
   "summary": "By Celina Amados At Netflix, our catalog metadata is crucial to our member experience, and a single corrupted data state can impact millions of viewers immediately. To protect streaming reliability, we built an automated data canary system that validates data transformations using production traffic. This canary detects issues in under 10 minutes, and blocks bad data from reaching our members. Intro Catalog metadata is what makes Netflix functional. It defines what titles exist, where they’re available, whether they can be played, and more. This data gets transformed and distributed across our",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Banning Open Source AI Would Be A Mistake",
   "url": "https://www.interconnects.ai/p/banning-open-source-ai-would-be-a",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-19T13:02:47.000Z",
   "summary": "This post was originally an op-ed co-authored with Kevin Xu of Interconnected for a general, non-technical audience.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Speculation Is All You Need",
   "url": "https://modal.com/blog/spec-is-all-u-need",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-19T00:00:00.000Z",
   "summary": "Why we're all-in on speculative decoding.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Thundering Herd Problem in Agentic AI: Why Traditional Fixes Fall Short",
   "url": "https://cockroachlabs.com/blog/agentic-ai-thundering-herd-problem",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-06-19T00:00:00.000Z",
   "summary": "The thundering herd of the past was externally triggered.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Outcome-driven learning systems: Enterprise RL with OpenEnv and Foundry",
   "url": "https://devblogs.microsoft.com/foundry/outcome-driven-learning-systems-enterprise-rl-with-openenv-and-foundry",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-06-18T20:37:18.000Z",
   "summary": "We shipped a lot at Build 2026: hosted agents, Toolboxes, Foundry IQ, Memory, Managed Compute, fine‑tuning, Frontier Tuning, and a new evaluation and optimization stack. Read as a feature list, it is a lot to hold in your head. So here is a simpler way to see it: these are the parts you need to […] The post Outcome-driven learning systems: Enterprise RL with OpenEnv and Foundry appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Performance Has Layers",
   "url": "https://oxide.computer/blog/performance-has-layers",
   "source": "Oxide Computer",
   "group": "Others",
   "published": "2026-06-18T16:30:00.000Z",
   "summary": "Tuning network performance across the layers of a stack you build end to end",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Adaptive write request scheduling in Redpanda's Cloud Topics",
   "url": "https://www.redpanda.com/blog/adaptive-write-request-scheduling-in-redpandas-cloud-topics",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-18T00:00:00.000Z",
   "summary": "How we turned to the buddy allocator algorithm for Redpanda’s Cloud Topics to balance batching efficiency against latency and cost.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Send Tailscale logs to Azure Blob Storage",
   "url": "https://tailscale.com/blog/azure-blob-storage-streaming",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-06-17T17:00:00.000Z",
   "summary": "Keep Tailscale logs with the rest of your security data.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "State of the blog, mid-2026",
   "url": "https://www.interconnects.ai/p/state-of-the-blog-mid-2026",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-17T14:29:06.000Z",
   "summary": "About 3 years since I started writing weekly.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Kimi K2.7 Code vs Claude Fable 5: Landing pages that cost 94% less",
   "url": "https://www.together.ai/blog/kimi-k2-7-code-vs-claude-fable-5",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-06-17T00:00:00.000Z",
   "summary": "We generated 12 landing pages with Kimi K2.7 Code and Claude Fable 5. Kimi cost 94% less and scored within a few points on every page. Here's what actually moved the needle.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Brazos: Bringing liquid cooling to air-cooled data centers",
   "url": "https://cloud.google.com/blog/topics/systems/brazos-liquid-cooling-system-for-air-cooled-data-centers",
   "source": "Google Cloud Infrastructure",
   "group": "Large-scale production systems",
   "published": "2026-06-16T16:00:00.000Z",
   "summary": "Next-generation artificial intelligence (AI) and high-performance computing (HPC) chips routinely exceed 1000 W Thermal Design Power (TDP). Simply put, standard air cooling cannot manage these extreme heat loads. The alternative — retrofitting entire data center facilities with chilled water loops — requires extensive amounts of capital and time. To solve this problem, Google developed Brazos, a rack-mounted, closed-loop liquid-to-air cooling system that lets you deploy high-density, liquid-cooled equipment inside existing air-cooled environments. Brazos is generally available, and our manufac",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Frontier post-training recipe review with Finbarr Timbers",
   "url": "https://www.interconnects.ai/p/frontier-post-training-recipe-review",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-16T13:29:30.000Z",
   "summary": "\"Interview\" #18",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Aperture: Accelerate AI adoption without the lock-in",
   "url": "https://tailscale.com/blog/ai-without-lock-in",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-06-16T13:00:00.000Z",
   "summary": "Avoid AI lock-in with Aperture’s flexible, identity-aware AI stack",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Automate ScyllaDB X Cloud Clusters with Terraform",
   "url": "https://www.scylladb.com/2026/06/16/automate-scylladb-x-cloud-clusters-with-terraform",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-06-16T12:27:30.000Z",
   "summary": "The ScyllaDB Cloud Terraform provider gives you infrastructure-as-code control over your clusters.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What's new in the Redpanda Agentic Data Plane",
   "url": "https://www.redpanda.com/blog/governing-ai-agents-in-production-agentic-data-plane",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-16T00:00:00.000Z",
   "summary": "Redpanda Agentic Data Plane is now generally available on AWS. Deploy governed, enterprise-scale, agentic AI across all your data, safely.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Product updates: VM Sandboxes, Lower latency routing, RBAC, and more",
   "url": "https://modal.com/blog/product-updates-vm-sandboxes-domain",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-15T00:00:00.000Z",
   "summary": "Recent product updates and news from around the community.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Welcome to the AGI era of AI governance",
   "url": "https://www.interconnects.ai/p/welcome-to-the-agi-era-of-ai-governance",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-14T17:43:42.000Z",
   "summary": "It's a one-way door and we weren't ready for it.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Serving Hybrid Models at Scale in llm-d",
   "url": "https://llm-d.ai/blog/serving-hybrid-models-at-scale-in-llm-d",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-06-13T09:00:00.000Z",
   "summary": "llm-d extends vLLM's Hybrid Memory Allocator across KV offloading to CPU and storage and KV-aware routing, making the offload connector HMA-aware - for 1.8–1.9x faster KV loads and about 115% higher throughput at high request rates with stable latency.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Dropbox uses MCP and Dash to close the design-to-code security gap",
   "url": "https://dropbox.tech/security/dropbox-mcp-dash-design-code-security",
   "source": "Dropbox Tech",
   "group": "Large-scale production systems",
   "published": "2026-06-12T18:00:00.000Z",
   "summary": "Using an agentic AI system to surface threat models during code review and spot gaps between security requirements and implementation.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "More Tailscale tricks for your jailbroken Kindle",
   "url": "https://tailscale.com/blog/jailbroken-kindle-proxy-tun-modes",
   "source": "Tailscale",
   "group": "Others",
   "published": "2026-06-12T15:45:00.000Z",
   "summary": "Tailscale on jailbroken Kindles now supports proxies and SSH. And a plugin supports Kobos and Pocket Readers, too.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "An Interview with Intel's Kira Boyko: Xeon 6+'s Product Director",
   "url": "https://chipsandcheese.com/p/an-interview-with-intels-kira-boyko",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-06-12T05:13:37.000Z",
   "summary": "Hello you fine Internet folks, today we have an interview with Kira Boyko, the Product Director of Intel Xeon 6+.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ScyllaDB Customer Experience Spotlight: Faisal Saeed",
   "url": "https://www.scylladb.com/2026/06/11/cx-spotlight-faisal-saeed",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-06-11T12:01:56.000Z",
   "summary": "Meet Faisal Saeed, Principal Customer Engineer on the Customer Experience team here at ScyllaDB.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Making FlashAttention-4 faster for inference",
   "url": "https://modal.com/blog/flash-attention-4-faster",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-11T12:00:00.000Z",
   "summary": "What part of \"dtype = 'fp8', num_splits = 0, pack_gqa = True, q_stage = 1, page_size = 1\" do you not understand?",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cloud Topics: the Metastore",
   "url": "https://www.redpanda.com/blog/cloud-topics-metastore",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-11T00:00:00.000Z",
   "summary": "Learn how Redpanda's metastore powers Cloud Topics, from offset lookups and whole cluster restore to cross-region read replicas, and why it's built to be a foundational primitive for the future.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Agentic AI Architecture: How CockroachDB Supports Memory, Context, and Control",
   "url": "https://cockroachlabs.com/blog/agentic-ai-architecture-memory-control",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-06-11T00:00:00.000Z",
   "summary": "What happens when you connect a fleet of autonomous AI agents to your enterprise data stack? You quickly discover...",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "ScyllaDB Operator 1.21 Release — with Oracle Kubernetes Engine (OKE) Support",
   "url": "https://www.scylladb.com/2026/06/10/scylladb-operator-1-21-release-with-oracle-kubernetes-engine-oke-support",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-06-10T16:03:55.000Z",
   "summary": "Introducing Oracle Kubernetes Engine support, stronger TLS, and a lighter dependency footprint",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building trust in enterprise AI: Together AI earns ISO 27001:2022 certification",
   "url": "https://www.together.ai/blog/iso-27001-2022-certification",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-06-10T00:00:00.000Z",
   "summary": "Together AI has earned ISO 27001:2022 certification, validating our commitment to enterprise-grade security for production AI workloads.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The Bill Arrives: How to Manage Agentic AI Costs at Scale",
   "url": "https://cockroachlabs.com/blog/agentic-ai-costs-at-scale",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-06-10T00:00:00.000Z",
   "summary": "What do the Uber budget blowout, a 24x token multiplier, and context teach us about building a real business case for AI Agents in production?",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Claude Fable 5 and new AI safety fables",
   "url": "https://www.interconnects.ai/p/claude-fable-5-and-new-ai-safety",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-09T22:59:35.000Z",
   "summary": "One step further into the power politics of frontier AI systems.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Heterogeneous inference serving across three GPU vendors with llm-d",
   "url": "https://llm-d.ai/blog/heterogeneous-inference-3-vendor-sovereign-cluster",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-06-09T09:00:00.000Z",
   "summary": "Benchmarking llm-d's prefix-cache-aware routing across anonymized GPU pools on the NxtGen sovereign cloud, showing how one routing layer improves throughput and TTFT across single-vendor and heterogeneous fleets.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "The four pillars for AI agent governance at scale",
   "url": "https://www.redpanda.com/blog/ai-agent-governance-at-scale-four-pillars-every-enterprise-needs",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-09T00:00:00.000Z",
   "summary": "AI agents need governance infrastructure, not just “better models”. Here are the four pillars every enterprise needs to deploy agents safely at scale: identity, authorization, observability, and accountability.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "RoachFest London 2026 Preview: AI Agents and an Astronaut Walk Into a Database Conference",
   "url": "https://cockroachlabs.com/blog/roachfest-london-2026-preview",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-06-09T00:00:00.000Z",
   "summary": "Cockroach Labs has been hosting our annual database conference since 2022, and I'm honored to be MCing RoachFest London for the third year running.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing FrontierCode",
   "url": "https://cognition.com/blog/frontier-code",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-06-08T18:00:00.000Z",
   "summary": "Today’s coding benchmarks have established that models can write correct code, but the question we should really be asking is: can models actually write…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Building Agents that Don't Break Themselves",
   "url": "https://fly.io/blog/building-agents-that-dont-break-themselves",
   "source": "Fly.io",
   "group": "Others",
   "published": "2026-06-08T00:00:00.000Z",
   "summary": "Building agents is fun. Rebuilding agents that break themselves… less so. A lot of Fly people are building agents with less of a penchant for self-destruction by teaching their agents to do anything risky in a Sprite. You get an agent that stays alive long enough to actually use its snazzy self-improvement features, and you can allow your agent to try things that would otherwise be battleship-scale footguns. Here’s how to do it. Brains vs Hands Your agent would be pretty useless without a shell, because this is where it does agent things. Run the test suite, apply the migration, install the de",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "BLIS: Evolving llm-d at Simulation Speed",
   "url": "https://llm-d.ai/blog/blis-evolving-llm-d-at-simulation-speed",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-06-05T09:00:00.000Z",
   "summary": "BLIS is a calibrated discrete-event simulator for llm-d control-plane behavior. It helps developers evaluate routing, admission, KV cache, batching, prefill/decode placement, and capacity choices before spending time on cluster validation.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Accelerate Edge AI Development with Foundry Local",
   "url": "https://devblogs.microsoft.com/foundry/accelerate-edge-ai-development-with-foundry-local",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-06-04T13:00:24.000Z",
   "summary": "Why edge AI development is still hard AI is no longer confined to cloud experiments. Developers are increasingly expected to deliver AI inside apps, devices, and edge systems where responsiveness, privacy, resilience, and local control are essential. But building those experiences for production is still difficult. Teams often have to solve model packaging, runtime fragmentation, hardware differences, […] The post Accelerate Edge AI Development with Foundry Local appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Estimating the Productivity of an Autonomous AI Software Engineer",
   "url": "https://cognition.com/blog/ai-productivity",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-06-04T00:00:00.000Z",
   "summary": "Engineering leaders want to know how much value AI is actually providing. We built a system that measures the number of human-equivalent hours of Devin's…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AI should earn its keep: Introducing the AI Productivity Guarantee",
   "url": "https://cognition.com/blog/ai-guarantee",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-06-04T00:00:00.000Z",
   "summary": "Introducing the AI Productivity Guarantee for enterprise customers. If Devin delivers less engineering value than you’re paying for, Cognition will fund…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "What Breaks When Agentic AI Reaches Production?",
   "url": "https://cockroachlabs.com/blog/agentic-ai-production-infrastructure",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-06-04T00:00:00.000Z",
   "summary": "Most enterprise AI teams have built an agent that was impressive; far fewer have shipped one without a production incident that made someone question the whole program.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Build 2026: From observability to ROI for AI agents on any framework",
   "url": "https://devblogs.microsoft.com/foundry/build-2026-from-observability-to-roi-for-ai-agents-on-any-framework",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-06-03T22:44:01.000Z",
   "summary": "9 min read · June 3, 2026 · Sebastian Kohlmeier Shipping an AI agent is the easy part. Keeping it accurate, safe, and accountable in production is where teams get stuck. Agents are non-deterministic. Their behavior shifts as models update, tools change, and traffic patterns evolve and most of that drift happens silently, long after the demo. End-to-end observability covering […] The post Build 2026: From observability to ROI for AI agents on any framework appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Expanding the Reach of Document Translation – New Capabilities Announced at Microsoft Build",
   "url": "https://devblogs.microsoft.com/foundry/document-translation-build-2026",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-06-03T17:00:00.000Z",
   "summary": "Learn how new Document Translation capabilities in Azure Translator, available in Foundry Tools, help developers translate images, PDFs, Office files, DITA, XLIFF, and future LLM-powered document workflows. The post Expanding the Reach of Document Translation – New Capabilities Announced at Microsoft Build appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Announcing Foundry Managed Compute: Run open models in Microsoft Foundry",
   "url": "https://devblogs.microsoft.com/foundry/announcing-foundry-managed-compute",
   "source": "Microsoft Foundry",
   "group": "Agent frameworks and runtimes",
   "published": "2026-06-03T16:00:54.000Z",
   "summary": "Microsoft Foundry Managed Compute is a new GPU platform-as-a-service for hosting open-source and custom AI models behind the same endpoint, SDKs, and bill as frontier models. The post Announcing Foundry Managed Compute: Run open models in Microsoft Foundry appeared first on Microsoft Foundry Blog .",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "iddqd, or the hardest kind of unsafe Rust",
   "url": "https://oxide.computer/blog/iddqd-unsafe",
   "source": "Oxide Computer",
   "group": "Others",
   "published": "2026-06-02T16:00:00.000Z",
   "summary": "How our Rust collections library defends against adversarial trait implementations.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Using Salting to Lower Latency for Large Blobs in ScyllaDB",
   "url": "https://www.scylladb.com/2026/06/02/using-salting-to-lower-latency-for-large-blobs-in-scylladb",
   "source": "ScyllaDB",
   "group": "Others",
   "published": "2026-06-02T15:00:41.000Z",
   "summary": "A modified salting technique that cuts P99 write latency 22x for large blobs",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Farewell Ai2",
   "url": "https://www.interconnects.ai/p/farewell-ai2",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-02T14:15:59.000Z",
   "summary": "This was my last week at the Allen Institute for AI (Ai2), where I got the great privilege to work on the Olmo models, to grow, to learn, and to have broad lasting impacts.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Devin Desktop",
   "url": "https://cognition.com/blog/introducing-devin-desktop",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-06-02T00:00:00.000Z",
   "summary": "The next generation of Windsurf, built around Devin Cloud, the Agent Command Center, and a full IDE for when you need to jump into the code.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Serving MiniMax-M3 for efficient inference: Unlocking 1M-Token Context and Multimodality Without Regrets",
   "url": "https://www.together.ai/blog/serving-minimax-m3-for-efficient-inference-unlocking-1m-token-context-and-multimodality-without-regrets",
   "source": "Together AI",
   "group": "Inference companies",
   "published": "2026-06-02T00:00:00.000Z",
   "summary": "How Together served MiniMax-M3 efficiently with KV-block-major sparse attention, paged MSA decode, optimized index scoring, and a Rust-based multimodal gateway.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How OmniNode uses Redpanda to scale AI agent workflows",
   "url": "https://www.redpanda.com/blog/omninode-scale-ai-agent-workflows",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-02T00:00:00.000Z",
   "summary": "OmniNode’s founder shares his journey building the AI agent workflows that became OmniNode, and how Redpanda keeps topic names from drifting with contracts.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Open and closed models are on different exponentials",
   "url": "https://www.interconnects.ai/p/open-and-closed-models-are-on-different",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-06-01T13:03:48.000Z",
   "summary": "Where marginally higher intelligence drives value, and where it doesn't.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Reinforcement learning is an infrastructure problem",
   "url": "https://modal.com/blog/reinforcement-learning-infrastructure-problem",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-06-01T00:00:00.000Z",
   "summary": "What we've seen helping teams run Reinforcement Learning at scale on Modal. Plus an open-source library to skip the scaffolding.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Real-time streaming for the agentic era with NVIDIA",
   "url": "https://www.redpanda.com/blog/nvidia-ai-ecosystem",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-06-01T00:00:00.000Z",
   "summary": "NVIDIA Vera launches today with Redpanda as part of the ecosystem, delivering 5.5x lower latencies for agents running in mission-critical environments.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Verifying Agentic Development at Scale",
   "url": "https://cognition.com/blog/testing-development",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-05-29T00:00:00.000Z",
   "summary": "What we’ve learned building end-to-end testing capabilities in Devin’s virtual machine",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Beyond code generation: rethinking engineering productivity in the age of AI agents",
   "url": "https://dropbox.tech/culture/beyond-code-generation-rethinking-engineering-productivity-in-the-age-of-ai-agents",
   "source": "Dropbox Tech",
   "group": "Large-scale production systems",
   "published": "2026-05-28T18:00:00.000Z",
   "summary": "How Dropbox is moving from AI tools that assist engineers to agentic systems that can execute scoped tasks, and how we’re building platforms to support those workflows.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "AI Agents Need Context to Reason, Not Just Data",
   "url": "https://cockroachlabs.com/blog/ai-agent-context-management",
   "source": "Cockroach Labs",
   "group": "Others",
   "published": "2026-05-28T00:00:00.000Z",
   "summary": "When your AI agent makes a bad decision in production, what do you blame?",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "More Devins in More Places",
   "url": "https://cognition.com/blog/series-d",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-05-27T00:00:00.000Z",
   "summary": "Cognition has raised over $1B at a $26B valuation, led by Lux Capital, General Catalyst, and 8VC.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Role-Based Access Control for humans and agents",
   "url": "https://modal.com/blog/role-based-access-control-for-humans-and-agents",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-05-27T00:00:00.000Z",
   "summary": "Introducing Role-Based Access Control for humans and agents, now available for all users on Teams and Enterprise plans.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Redpanda SQL is GA: the query engine that skips the pipeline",
   "url": "https://www.redpanda.com/blog/redpanda-sql-ga",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-05-27T00:00:00.000Z",
   "summary": "Your warehouse can't query data that hasn't been ingested yet. Redpanda SQL can. Ad hoc SQL against live topics and Iceberg history, no ETL pipeline required.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we evolved Google’s global and data center networks for the AI era",
   "url": "https://cloud.google.com/blog/products/networking/data-center-and-global-networks-built-for-ai-era",
   "source": "Google Cloud Infrastructure",
   "group": "Large-scale production systems",
   "published": "2026-05-26T16:00:00.000Z",
   "summary": "Over the last 25 years of building Google’s global network, we’ve navigated major architectural eras — from the Internet, to streaming, and the cloud. Today, we are squarely in the midst of a fourth: the AI era. The applications in the AI era are fundamentally different from the consumer and enterprise applications of the previous eras and impose a set of novel and demanding requirements — on compute resources, of course, but also on the network. Consider the fundamental physical challenge, which is that it is far more difficult to move electrons (electrical power) than it is to move photons (",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Some ideas for what comes next, May 2026",
   "url": "https://www.interconnects.ai/p/some-ideas-for-what-comes-next-may",
   "source": "Interconnects",
   "group": "Independent sources",
   "published": "2026-05-26T15:39:02.000Z",
   "summary": "Gemini Flash 3.5, Mythos, open-closed balance, America's open-source surge, emerging power struggles and more.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "No Kubernetes? No Problem: llm-d Now Runs Anywhere",
   "url": "https://llm-d.ai/blog/running-llm-d-without-kubernetes",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-05-26T09:00:00.000Z",
   "summary": "llm-d's routing intelligence was entangled with Kubernetes. A new endpoint-discovery abstraction separates the two, so KV-cache-aware scheduling, prefix affinity, and P/D run on Slurm, Ray, bare metal, or a laptop.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Evaluating SPEC CPU2026",
   "url": "https://chipsandcheese.com/p/evaluating-spec-cpu2026",
   "source": "Chips and Cheese",
   "group": "Independent sources",
   "published": "2026-05-23T08:40:38.000Z",
   "summary": "SPEC’s CPU benchmark suite has been a long established industry standard, and is almost impossible to miss when reading through various publications.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Nova, our internal platform for coding agents",
   "url": "https://dropbox.tech/machine-learning/introducing-nova-our-internal-platform-for-coding-agents",
   "source": "Dropbox Tech",
   "group": "Large-scale production systems",
   "published": "2026-05-21T16:00:00.000Z",
   "summary": "Nova lets engineers run multiple coding sessions in parallel and lets internal systems use AI agents as part of automated workflows.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Devin is Getting a Windows PC",
   "url": "https://cognition.com/blog/devin-is-getting-a-windows-pc",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-05-21T00:00:00.000Z",
   "summary": "Devin now builds, runs, and tests natively in Windows VMs, bringing the full power of autonomous AI engineering to the world's most mature developer…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Modal's Series C: Raising $355M at a $4.65B valuation",
   "url": "https://modal.com/blog/modal-series-c",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-05-21T00:00:00.000Z",
   "summary": "We've raised $355M at a $4.65B valuation to continue building the production cloud for AI.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Scaling reinforcement learning at Applied Compute",
   "url": "https://modal.com/blog/applied-compute-reinforcement-learning",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-05-20T00:00:00.000Z",
   "summary": "How Applied Compute trains custom agents with Reinforcement Learning for enterprises like DoorDash, Cognition, and Mercor on Modal.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Claude Managed Agents with Modal Sandboxes",
   "url": "https://modal.com/blog/introducing-claude-managed-agents-with-modal-sandboxes",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-05-19T00:00:00.000Z",
   "summary": "",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Cloud Topics: Level Zero garbage collection",
   "url": "https://www.redpanda.com/blog/cloud-topics-level-zero-garbage-collection",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-05-19T00:00:00.000Z",
   "summary": "How Redpanda Cloud Topics tracks the lifecycle of temporary L0 objects and determines when they're safe to delete, without risking data loss or runaway storage costs. Read more.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How Google Does It: Fleet-wide, large-scale A/B experimentation",
   "url": "https://cloud.google.com/blog/topics/systems/how-google-does-it-fleet-wide-large-scale-ab-experimentation",
   "source": "Google Cloud Infrastructure",
   "group": "Large-scale production systems",
   "published": "2026-05-18T16:00:00.000Z",
   "summary": "When most people think of A/B experimentation, they think of button colors, landing page layouts, or checkout flows. At Google, many fundamental infrastructure improvements also need the rigor of A/B experimentation. Optimizing a memory allocator or a kernel scheduler can unlock massive savings in compute resources and slash latency for millions of users. But experimenting with such critical changes is inherently risky; a buggy kernel update doesn't just result in an unhappy user, it can take down large swaths of machines. To innovate safely and at scale, you must perform A/B experimentation o",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Introducing Auto-Triage",
   "url": "https://cognition.com/blog/auto-triage",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-05-18T00:00:00.000Z",
   "summary": "Devin can monitor for bugs, alerts, and incidents. When something breaks, Devin responds immediately, investigates with your tools, connects related…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Devin now supports Android emulators",
   "url": "https://cognition.com/blog/android-emulator",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-05-13T00:00:00.000Z",
   "summary": "Devin can now spin up an Android Virtual Device (AVD), enabling autonomous development for Android applications.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How we achieved truly serverless GPUs",
   "url": "https://modal.com/blog/truly-serverless-gpus",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-05-12T12:00:00.000Z",
   "summary": "A deep dive on Modal's deep tech for fast boots.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "llm-d v0.7: From Feature Introduction to Production Hardening",
   "url": "https://llm-d.ai/blog/llm-d-v0.7-from-feature-introduction-to-production-hardening",
   "source": "llm-d",
   "group": "Serving engines and runtimes",
   "published": "2026-05-12T00:00:00.000Z",
   "summary": "llm-d v0.7 shifts focus from proving capabilities to making them deployable, with changes across deployment tooling, hardware support, documentation, and continuous integration.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "5 predictions about agentic AI and analytics in 2026",
   "url": "https://www.redpanda.com/blog/5-predictions-about-agentic-ai-and-analytics-in-2026",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-05-12T00:00:00.000Z",
   "summary": "Learn about what’s top of mind among enterprises planning for agentic systems, and top AI predictions for 2026 and beyond.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "How to Automate Failure Triages and 10x Test Generation: What We've Learned Deploying AI Across HIL/SIL Workflows",
   "url": "https://cognition.com/blog/how-to-automate-failure-triages-and-10x-test-generation-what-weve-learned-deploying-ai-across-hilsil-workflows",
   "source": "Cognition",
   "group": "Coding-agent builders",
   "published": "2026-05-11T00:00:00.000Z",
   "summary": "Requirements and ticket volumes keep growing while engineering capacity hasn't kept up. Through deployments with RV Tech and Mercedes, we share how AI is…",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Engineering Den: Claude Code skills and hooks for frontend teams",
   "url": "https://www.redpanda.com/blog/claude-code-skills-hooks-frontend",
   "source": "Redpanda",
   "group": "Others",
   "published": "2026-05-07T00:00:00.000Z",
   "summary": "Ship cleaner frontend code faster with new shared Claude skills (40+) and hooks (90+). Includes powerful workflows like /work, /go, /tdd, and built-in quality gate checks.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  },
  {
   "title": "Boosting multimodal inference performance by >10% with a single Python dictionary",
   "url": "https://modal.com/blog/boosting-multimodal-inference-performance-by-greater-than-10-with-a-single-python-dictionary",
   "source": "Modal",
   "group": "Inference companies",
   "published": "2026-05-04T00:00:00.000Z",
   "summary": "If we've said it once, we've said it once per millisecond: never block the GPU.",
   "firstSeen": "2026-08-27T23:08:15.157Z"
  }
 ]
}