{
 "site": "https://workingsurface.ai",
 "about": "How product, design and engineering teams work with AI agents: daily links, summarised and drawn.",
 "updated": "2026-10-07",
 "topics": [
  {
   "slug": "human-approval",
   "name": "Human approval",
   "definition": "A person other than the one who built a piece of AI-assisted work checks it and approves it before it reaches customers, often called keeping a human in the loop.",
   "url": "https://workingsurface.ai/topics/human-approval/"
  },
  {
   "slug": "agent-context-files",
   "name": "Agent context files",
   "definition": "The files an AI agent reads before it works, such as AGENTS.md, CLAUDE.md, design-system rules and other written instructions, and the person or team who keeps them current.",
   "url": "https://workingsurface.ai/topics/agent-context-files/"
  },
  {
   "slug": "changing-roles",
   "name": "Changing roles",
   "definition": "How the jobs of product managers, designers, engineers and their managers change when AI agents do more of the building.",
   "url": "https://workingsurface.ai/topics/changing-roles/"
  },
  {
   "slug": "agent-ownership",
   "name": "Agent ownership",
   "definition": "A person or team answers for what an AI agent does, and decides its instructions, its default settings and when it must hand a decision to a person.",
   "url": "https://workingsurface.ai/topics/agent-ownership/"
  },
  {
   "slug": "evals-before-the-build",
   "name": "Evals before the build",
   "definition": "Before an AI feature is built, the team writes down what a good output looks like, so that the result can be tested against it.",
   "url": "https://workingsurface.ai/topics/evals-before-the-build/"
  },
  {
   "slug": "decision-records",
   "name": "Decision records",
   "definition": "A written record of what was decided and why, kept with the work so that someone who did not build it can review it.",
   "url": "https://workingsurface.ai/topics/decision-records/"
  },
  {
   "slug": "after-the-prototype",
   "name": "After the prototype",
   "definition": "AI makes a working prototype quick, so the effort moves to choosing a direction and getting from prototype to production with a reliable product.",
   "url": "https://workingsurface.ai/topics/after-the-prototype/"
  },
  {
   "slug": "understanding-the-work",
   "name": "Understanding the work",
   "definition": "A team checks on purpose that its people can still explain work that AI agents produced.",
   "url": "https://workingsurface.ai/topics/understanding-the-work/"
  },
  {
   "slug": "ground-truth-for-review",
   "name": "Ground truth for review",
   "definition": "A set of correct answers or expected results, prepared by someone other than the builder, that a reviewer uses to check work made with AI.",
   "url": "https://workingsurface.ai/topics/ground-truth-for-review/"
  },
  {
   "slug": "interfaces-for-agents",
   "name": "Interfaces for agents",
   "definition": "Product pages and data designed so that an AI agent, not only a person, can use them.",
   "url": "https://workingsurface.ai/topics/interfaces-for-agents/"
  },
  {
   "slug": "work-cut-to-size",
   "name": "Work cut to size",
   "definition": "Before an AI agent starts, the work is split into pieces small enough for a person to review one at a time.",
   "url": "https://workingsurface.ai/topics/work-cut-to-size/"
  },
  {
   "slug": "training-juniors",
   "name": "Training juniors",
   "definition": "How people new to a craft build judgement when AI tools do the work that used to teach it.",
   "url": "https://workingsurface.ai/topics/training-juniors/"
  },
  {
   "slug": "workshops-that-ratify",
   "name": "Workshops that ratify",
   "definition": "A workshop that confirms rules a team has already drafted while reviewing real work, instead of trying to invent them in the room.",
   "url": "https://workingsurface.ai/topics/workshops-that-ratify/"
  }
 ],
 "links": [
  {
   "id": "2026-10-07-01",
   "added": "2026-10-07",
   "url": "https://workingsurface.ai/links/2026-10-07-01/",
   "article": {
    "url": "https://leaddev.com/software-quality/faster-code-isnt-faster-delivery",
    "title": "Faster code isn’t faster delivery",
    "author": "Emma Bostian",
    "publication": "LeadDev",
    "published": "2026-10-06"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "Emma Bostian, an engineering manager at Spotify, argues that AI coding tools moved the delivery bottleneck from writing code to reviewing it, and that authors must own their changes.",
   "summary": "Emma Bostian, an engineering manager at Spotify, the music streaming company, writes in LeadDev that AI has not made software delivery faster. Pull requests now arrive faster than engineers can understand, test and approve them, and reviewers end up reconstructing the author's reasoning. She proposes five practices to protect review and the time engineers spend working together.",
   "key_points": [
    "Non-technical colleagues who open pull requests with AI sometimes pass each review comment back to an AI agent without understanding the problem, which repeats the cycle.",
    "Harvard Business Review research she cites found that each piece of low-quality AI work takes an average of one hour and 56 minutes to handle.",
    "She says the person who starts a change must be able to explain it, test it and review it before asking a colleague to review it.",
    "She recommends defining what must be true before AI-assisted code can be submitted for review, and measuring review time and rework alongside output."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Report how long pull requests wait for review and how many are reworked afterwards, next to the amount of code produced."
    },
    {
     "for": "Engineering",
     "practice": "Require the author of an AI-assisted pull request to explain the change, run the relevant tests and check security concerns before requesting a review."
    }
   ]
  },
  {
   "id": "2026-10-07-02",
   "added": "2026-10-07",
   "url": "https://workingsurface.ai/links/2026-10-07-02/",
   "article": {
    "url": "https://cate.blog/2026/10/06/is-the-software-factory-working/",
    "title": "Is the software factory working?",
    "author": "Cate Huston",
    "publication": "Accidentally in Code",
    "published": "2026-10-06"
   },
   "kind": "Company story",
   "topic": null,
   "takeaway": "Cate Huston, part-time technology chief at Twill, measured six months of agent-written pull requests and found that automated safety checks cut emergency fixes but slowed merges.",
   "summary": "Twill connects employers with senior candidates through referrals, and its two engineers let AI agents write most of the code. Cate Huston, its part-time chief technology officer, wrote scripts to sort the company's pull requests by type and to track reliability and speed. The results show the team spending more of its output on automated checks, with fewer same-day fixes and slower merges.",
   "key_points": [
    "From April to September, pull requests rose 24-fold, features fell from 66 to 16 percent of them, and automated safety checks rose from 12 to 36 percent.",
    "Same-day emergency fixes fell from about ten in August to one in September. In September, 89 percent of feature pull requests carried their own tests, up from 62 percent in April.",
    "A rule that every change needs tests sat in the agent's instruction file from the start, but it held only once test coverage became a required check. The agent is now also told to run mutation testing, which checks whether tests catch deliberate errors.",
    "The median time a pull request stayed open rose from 1.2 hours in August to 3.5 hours in September, partly because database changes get more human review."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Require a test coverage check to pass before any pull request merges, because an AI coding agent may ignore a testing rule written in its instruction file."
    },
    {
     "for": "Engineering",
     "practice": "Sort pull requests by type each month and track features, automated checks and same-day emergency fixes side by side."
    }
   ]
  },
  {
   "id": "2026-10-07-03",
   "added": "2026-10-07",
   "url": "https://workingsurface.ai/links/2026-10-07-03/",
   "article": {
    "url": "https://blog.jetbrains.com/research/2026/10/review-ai-generated/",
    "title": "Our Framework for Reviewing AI-Generated Code",
    "author": "Katie Fraser, Agnia Sergeyuk and Ilya Zakharov",
    "publication": "JetBrains Research",
    "published": "2026-10-06"
   },
   "kind": "How-to",
   "topic": "human-approval",
   "takeaway": "JetBrains researchers propose that tools for reviewing agent-written code should show an overview, rank files by risk and only then open code, based on workshops with 17 practitioners.",
   "summary": "Katie Fraser, Agnia Sergeyuk and Ilya Zakharov work in the Human-AI Experience team at JetBrains, which makes programming tools. With researchers at Lund University, they asked developers how a tool for reviewing AI-generated code should work. Their paper, to be presented at Empirical Software Engineering International Week in October, proposes a three-level review workflow.",
   "key_points": [
    "The design came from four workshops with 17 practitioners, followed by a survey of 43 software professionals.",
    "An AI model presents every line with the same apparent confidence, and a reviewer cannot ask it about its reasoning, so the cues a human author gives are missing.",
    "The proposed tool first gives an overview, then ranks files by risk before the reviewer reads code, and only then opens small pieces of code for close reading.",
    "The authors warn that tools which only make changes easier to understand may still lead reviewers to spend effort on low-risk code and miss high-risk code."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Before reading an agent-written change line by line, rank its files by risk and spend close review on the riskiest ones."
    }
   ]
  },
  {
   "id": "2026-10-07-04",
   "added": "2026-10-07",
   "url": "https://workingsurface.ai/links/2026-10-07-04/",
   "article": {
    "url": "https://world.hey.com/dhh/over-my-dead-pencil-fb0f3647",
    "title": "Over my dead pencil",
    "author": "David Heinemeier Hansson",
    "publication": "HEY World",
    "published": "2026-10-06"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "David Heinemeier Hansson, who created Ruby on Rails, argues that AI coding agents now write most code and that programmers who refuse to work with them risk their careers.",
   "summary": "David Heinemeier Hansson is co-owner and chief technology officer of 37signals, which makes the project tool Basecamp, and the creator of the web framework Ruby on Rails. Two weeks after telling the Rails World conference that programmers will stop writing most code by hand, he explains the claim. He describes AI coding agents as fast new coworkers who still need help with some tasks.",
   "key_points": [
    "When he asked the Rails World audience who still writes a material amount of code by hand each week, only a handful raised their hands.",
    "He says it is no longer economically viable to have people type out lines of Ruby, Rust or C++.",
    "He expects much more software to be built as the price of development falls, because a great deal of work is still not automated.",
    "He accepts scepticism about the areas where AI agents still get things wrong. He says refusing to work with them is not a viable career path for almost anyone."
   ],
   "practices": []
  },
  {
   "id": "2026-10-07-05",
   "added": "2026-10-07",
   "url": "https://workingsurface.ai/links/2026-10-07-05/",
   "article": {
    "url": "https://www.figma.com/blog/3-ways-product-designers-use-the-figma-agent/",
    "title": "3 ways product designers use the Figma agent for craft, speed, and creative expression",
    "author": "Jenny Xie",
    "publication": "Figma Blog",
    "published": "2026-10-06"
   },
   "kind": "How-to",
   "topic": "agent-context-files",
   "takeaway": "Designers at Uber, Granola and Atlassian describe using Figma's AI agent to document components, collect review feedback and add motion, in a post Figma published for the agent's general release.",
   "summary": "Jenny Xie, an editor at Figma, the design tool company, reports how three customer teams use the AI agent built into Figma. The post marks the agent's move out of beta. At Uber, the ride-hailing company, one designer now publishes component documentation in an afternoon, work that once took several people months.",
   "key_points": [
    "Uber's design systems team delivers components across seven platforms. Staff Product Designer Ian Guisard built agent skills that diagram a component's structure and map its colours to design tokens.",
    "The colour skill flags hard-coded values, and Guisard says it caught one that he had changed by accident.",
    "At Granola, which makes an AI note-taking app, Product Designer Paavan Buddhdev has the agent place feedback from meeting transcripts on the design canvas as annotations.",
    "Atlassian's motion design team turned animated illustrations into reusable library components. Custom animation made with the agent must still use the variables of Atlassian's design system."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Write the steps for documenting a component, such as diagramming its structure and mapping colours to tokens, as AI agent skills that any designer can run."
    },
    {
     "for": "Design",
     "practice": "Have an AI agent add feedback from review meeting transcripts to the design file as annotations, so the designer can take part in the meeting."
    }
   ]
  },
  {
   "id": "2026-10-07-06",
   "added": "2026-10-07",
   "url": "https://workingsurface.ai/links/2026-10-07-06/",
   "article": {
    "url": "https://muz.li/blog/design-process-vs-ai-workflow/",
    "title": "AI didn’t kill the design process. It made being dirt cheap. The new workflow, the tools, and what you should do today.",
    "author": "Petras Baukys",
    "publication": "Muzli Blog",
    "published": "2026-10-06"
   },
   "kind": "Company story",
   "topic": "after-the-prototype",
   "takeaway": "Petras Baukys of Muzli argues that AI made a wrong prototype cheap, so design teams should build first, release to half of users and measure before deciding.",
   "summary": "Petras Baukys writes for Muzli, a browser extension that shows designers new work in each new tab. He argues that the Double Diamond, the best-known diagram of the design process, suited a time when building the wrong thing cost months. He then describes how Muzli builds small changes the day an idea appears and tests them on half of new users.",
   "key_points": [
    "Richard Eisermann led the Design Council team that drew the Double Diamond in 2003. In 2023 he wrote that it probably no longer fits, partly because of generative AI.",
    "Baukys gave two AI models three studio briefs. One built each in 50 to 81 minutes, and the other built all three for $1.88 in total.",
    "At Muzli, every experiment gets its own label in the analytics before the build ships. The team writes down the sample size and the date it will read the result before a test starts.",
    "One Muzli test looked like a clear win after two weeks. When the team split it by install date, the whole gain came from three days of installs."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Before a test starts, write down how many users it needs and the date the result will be read."
    },
    {
     "for": "Product",
     "practice": "Release a new feature to half of new users and keep the other half unchanged as a comparison group."
    }
   ]
  },
  {
   "id": "2026-10-06-01",
   "added": "2026-10-06",
   "url": "https://workingsurface.ai/links/2026-10-06-01/",
   "article": {
    "url": "https://leaddev.com/software-quality/meta-turned-engineers-judgment-into-agent-skills",
    "title": "Meta turned engineers’ judgment into agent skills",
    "author": "Bill Doerrfeld with Tommy Tran",
    "publication": "LeadDev",
    "published": "2026-10-05"
   },
   "kind": "Company story",
   "topic": "agent-context-files",
   "takeaway": "Meta engineer Tommy Tran wrote senior performance engineers' reasoning into AI agent skills, cutting diagnosis time from hours to minutes while people still approve production changes.",
   "summary": "Meta, the company behind Facebook and Instagram, relied on a few senior engineers to find and fix code that wastes computing power across its servers. Tommy Tran, a software engineer at Meta, built an internal platform of AI agents that investigates these problems and proposes fixes. He described the design to Bill Doerrfeld of LeadDev, a publication for engineering leaders.",
   "key_points": [
    "Tran built the agent skills by sitting with senior efficiency engineers and recording the checks they run, the signals they trust and the order they investigate in.",
    "Stable tools for reading and changing systems are kept separate from the skills, so expertise can change without rebuilding the platform.",
    "Diagnosis time fell from about 10 hours to about 30 minutes, according to Meta's engineering blog as cited in the article.",
    "The agents produce a fix ready for review, and a human engineer approves anything that changes production."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Write down how senior engineers investigate a problem, including their checks, trusted signals and order of steps, and turn it into instructions an AI agent follows."
    },
    {
     "for": "Engineering",
     "practice": "Limit AI agents to proposing fixes, and require a human engineer to approve any change that reaches production."
    }
   ]
  },
  {
   "id": "2026-10-06-02",
   "added": "2026-10-06",
   "url": "https://workingsurface.ai/links/2026-10-06-02/",
   "article": {
    "url": "https://leaddev.com/ai/ai-coding-tools-could-be-breaking-the-junior-engineer-pipeline",
    "title": "AI-coding tools could be breaking the junior engineer pipeline",
    "author": "Chris Stokel-Walker",
    "publication": "LeadDev",
    "published": "2026-10-05"
   },
   "kind": "Company story",
   "topic": "training-juniors",
   "takeaway": "A small Australian study found students coding with AI scored higher but remembered less, and a Manchester engineering head responds by never letting junior engineers work alone.",
   "summary": "Researchers at the University of New South Wales gave 55 students three introductory C programming tasks, with either ChatGPT or conventional web search. Chris Stokel-Walker reported the results for LeadDev, a publication for engineering leaders. He also asked Adrian Harwood, head of research software engineering at the University of Manchester, how his department trains juniors who already use AI.",
   "key_points": [
    "Students using AI scored 89 percent on the tasks against 69 percent for those using search.",
    "Two days later, the AI group scored 39 percent on a knowledge test against 52 percent for the search group.",
    "Students using AI estimated that only 45 percent of the code they submitted was their own.",
    "In Harwood's department no junior works on a project alone, and senior engineers review their pull requests, including code produced with AI."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Teach junior engineers to review what an AI tool produces and to explain why some of its choices may cause problems."
    },
    {
     "for": "Engineering",
     "practice": "Never assign a junior engineer a project alone, and have a senior engineer review the junior's pull requests, including code written with AI."
    }
   ]
  },
  {
   "id": "2026-10-06-03",
   "added": "2026-10-06",
   "url": "https://workingsurface.ai/links/2026-10-06-03/",
   "article": {
    "url": "https://www.lukew.com/ff/entry.asp?2165",
    "title": "Many Agents, Many Views",
    "author": "Luke Wroblewski",
    "publication": "LukeW",
    "published": "2026-10-05"
   },
   "kind": "How-to",
   "topic": "understanding-the-work",
   "takeaway": "Luke Wroblewski explains why Intent, a tool for coordinating AI agents, lets agents draw their plans and show before-and-after screenshots so people can judge work without reading code.",
   "summary": "Luke Wroblewski is a product designer who works on Intent, a tool that lets software developers coordinate many AI agents at once. He argues that chat threads stop working when agents manage other agents, because people can no longer see what each agent plans or has done. He describes the views his team built instead.",
   "key_points": [
    "Agents can draw architecture maps, flowcharts and state diagrams in shared notes, so a person sees the plan before the agent spends effort on the work.",
    "Agents place before-and-after screenshots of interface changes side by side, so a person can judge the result without reading code.",
    "A product's design system can stay open as a note that both people and agents read, so the agents' interface work matches the same reference.",
    "Chat remains the place for starting, moving along and redirecting work."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Have AI agents place before-and-after screenshots of each interface change side by side for review."
    },
    {
     "for": "Product",
     "practice": "Ask AI agents to show their plan as a diagram before they start work, so a person can correct it early."
    }
   ]
  },
  {
   "id": "2026-10-06-04",
   "added": "2026-10-06",
   "url": "https://workingsurface.ai/links/2026-10-06-04/",
   "article": {
    "url": "https://medium.com/quantumblack/what-changes-when-your-design-system-can-talk-to-an-ai-0a856b3c40d9",
    "title": "What changes when your design system can talk to an AI?",
    "author": "Natalia Kurakina, with Péter Kismarczi, Carolina Staudinger and Joseph Perkins",
    "publication": "QuantumBlack, AI by McKinsey",
    "published": "2026-10-01"
   },
   "kind": "How-to",
   "topic": "agent-context-files",
   "takeaway": "QuantumBlack moved a product's whole interface onto its design system with an AI coding agent that was told to stop and ask whenever the design left gaps.",
   "summary": "QuantumBlack, AI by McKinsey, is the consultancy's AI arm. It moved the interface of one of its products, which runs to 21 pages and 13 modals, onto its own design system. Natalia Kurakina, writing with three colleagues, describes how a front-end engineer and designers used Claude Code, an AI coding agent, connected to Figma designs. She also lists the checks people still carry out.",
   "key_points": [
    "The engineer gives the agent the pull request, a screenshot, the target Figma design and a skill that packages the design system. The engineer also tells it to ask when something is not trivial.",
    "A simple page took 10 to 15 minutes and a complex page 4 to 5 hours; both would have taken days by hand.",
    "The designer reports first-pass match to the design rising from about 20 percent to 70 to 80 percent, and review rounds falling from four or more to one or two.",
    "People still check interaction states, confirm that existing functions survived, sort test failures and review every change before merge."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Instruct the AI coding agent to stop and ask when a design leaves something non-trivial undecided, and answer in writing before it continues."
    },
    {
     "for": "Design",
     "practice": "Before merging a page that an AI agent has rebuilt, click through its interaction states and confirm that each existing function still works."
    }
   ]
  },
  {
   "id": "2026-10-05-01",
   "added": "2026-10-05",
   "url": "https://workingsurface.ai/links/2026-10-05-01/",
   "article": {
    "url": "https://www.seangoedecke.com/shipping-is-the-foundation/",
    "title": "Shipping is the foundation",
    "author": "Sean Goedecke",
    "publication": "seangoedecke.com",
    "published": "2026-10-03"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "Software engineer Sean Goedecke argues that AI tools have not made shipping easy, because shipping needs context about systems and people that AI models lack.",
   "summary": "Sean Goedecke, a software engineer who writes about working at large technology companies, argues that shipping changes is the skill every other senior skill depends on. Engineers who cannot ship give bloated estimates and turn small requests into large projects. He says AI tools help with shipping but cannot run it from start to finish.",
   "key_points": [
    "Goedecke compares every alternative, such as writing a ticket or delegating a task, against the default of making the change oneself.",
    "Managers give senior engineers many small requests, and those requests are meant to avoid slow formal resourcing.",
    "In his account, shipping includes finding the shortest practical path, solving small problems on the way and keeping managers informed.",
    "He writes that even frontier AI models cannot be left to change a codebase unsupervised, and that engineers still need to read the code."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Read the code an AI model writes before it ships."
    },
    {
     "for": "Engineering",
     "practice": "Compare every proposed ticket, design document or delegated task against the cost of making the change directly."
    }
   ]
  },
  {
   "id": "2026-10-05-02",
   "added": "2026-10-05",
   "url": "https://workingsurface.ai/links/2026-10-05-02/",
   "article": {
    "url": "https://www.producttalk.org/generating-opportunity-solution-trees-with-ai-how-vistaly-rebuilt-its-product-around-interview-synthesis-evals-and-repair-loops/",
    "title": "Generating Opportunity Solution Trees with AI: How Vistaly Rebuilt Its Product Around Interview Synthesis, Evals, and Repair Loops",
    "author": "Teresa Torres with Matt O'Connell, CP Dehli and Steve Klein",
    "publication": "Product Talk",
    "published": "2026-10-01"
   },
   "kind": "How-to",
   "topic": null,
   "takeaway": "Vistaly rebuilt its product discovery software around an AI agent and taught the agent to record each edit as a named move, so users can see and correct what changed.",
   "summary": "Vistaly makes opportunity solution tree software for product teams, which links a business outcome to customer problems and possible solutions. Its first AI version walked users through insights one at a time in a chat, and users found it too slow. In a Product Talk podcast, the three co-founders describe a full rewrite in which an AI agent drafts and updates the tree from uploaded customer interviews.",
   "key_points": [
    "The team rebuilt nearly all of the first version's features in two and a half months and stopped new sign-ups to the old version during the rewrite.",
    "A cheap code check, which flags a node with too many children, runs before the team pays for an AI model to grade the output.",
    "To balance two opposing errors, the team wrote four new automated tests of the AI output and tried 16 variants. The fix was to move the check into a repair loop inside the workflow, and that test now guards production.",
    "The agent records each edit as a named move, such as merge, move or reframe. Comparing trees afterwards gives several valid change lists, and only one makes sense to a user.",
    "The founders report that users want the finished answer first and a way to correct it, instead of working through the analysis step by step with the AI."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Before an AI model grades an AI agent's output, run a free check in ordinary code first, such as a rule that flags a list with too many items."
    },
    {
     "for": "Design",
     "practice": "Have the AI agent record each edit as a named move, such as merge, move or reframe, so users can see what changed."
    }
   ]
  },
  {
   "id": "2026-10-04-01",
   "added": "2026-10-04",
   "url": "https://workingsurface.ai/links/2026-10-04-01/",
   "article": {
    "url": "https://www.microsoft.com/insidetrack/blog/rebuilding-our-microsoft-foundry-labs-site-with-a-three-people-build-team-and-ai/",
    "title": "Rebuilding our Microsoft Foundry Labs site with a three-people build team and AI",
    "author": "Mark Armstrong",
    "publication": "Inside Track (Microsoft)",
    "published": "2026-10-01"
   },
   "kind": "Company story",
   "topic": "after-the-prototype",
   "takeaway": "A three-person Microsoft team rebuilt its Foundry Labs site by prototyping with AI tools first, then revising a 32-page specification against the prototype until engineering took over.",
   "summary": "Microsoft's Foundry Labs website lists AI models, tools and research projects from teams across Microsoft, and it became hard to search once the catalogue grew past 20 projects. Principal product manager Saumil Shrivastava assembled a team of three: technical program manager Gulsimo Osimi, software engineer Pratyansh Agrawal and senior product marketing manager Patrick Widjaja. Osimi replaced a long specification with an AI-built prototype revised alongside the document, and Agrawal turned the prototype into the production site in about three months.",
   "key_points": [
    "Osimi first wrote a 32-page specification, and the team found the document a poor way to communicate the plan.",
    "The specification and the prototype were revised against each other, and a showing at an internal Microsoft demo fair won support from engineering and marketing partners.",
    "Widjaja used AI tools for branding, design, copy and images, and the team says it went from sketch to full website in a month.",
    "Agrawal built a structured WordPress content model, and the team names architecture, security, privacy, accessibility and performance as work a prototype does not cover."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Show the working prototype to stakeholders before the production build, so that engineering and marketing partners can commit early."
    },
    {
     "for": "Product",
     "practice": "Revise the written specification and the working prototype against each other, so that a change in one is checked in the other."
    }
   ]
  },
  {
   "id": "2026-10-04-02",
   "added": "2026-10-04",
   "url": "https://workingsurface.ai/links/2026-10-04-02/",
   "article": {
    "url": "https://www.salesforce.com/blog/coding-agents-product-design/",
    "title": "Agents Can Mimic Good Design. Ours Know Why It Works",
    "author": "Alan Weibel",
    "publication": "Salesforce Blog",
    "published": "2026-10-01"
   },
   "kind": "How-to",
   "topic": "agent-context-files",
   "takeaway": "Salesforce's design team built a pipeline in which AI agents produce designs from any starting point and hand engineers a specification that records the reason for every decision.",
   "summary": "Alan Weibel, a UX architect at Salesforce, describes how the company teaches AI coding agents its design standards across several design systems. A knowledge layer holds written guidance, machine-readable design-system data, skills and linters. A pipeline called Experience Operating System uses that layer to run the design process from any starting artefact, and it writes the earlier documents that were skipped.",
   "key_points": [
    "Each skill first works out what is being asked, and it hands work it does not own, such as scoring a screenshot, to a separate skill.",
    "When guidance conflicts, the AI agent follows a written order: the user's goal, verified research, documented standards, accepted past decisions, then general heuristics.",
    "Evidence from reviews and Slack is collected without judgement, and a person decides whether it becomes guidance, a lint rule, an example or an eval.",
    "The engineering handoff includes the prototype, a specification covering layout, every interface state and keyboard behaviour, accessibility notes and the full record of reasons."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Write down an order of authority for conflicting design guidance, from the user's goal to general heuristics, and give it to the AI agent."
    },
    {
     "for": "Design",
     "practice": "When a design review finds a problem in an AI agent's work, write it down as a proposed rule, and have a person decide whether it becomes written guidance, an automated check or an example the agent reads."
    }
   ]
  },
  {
   "id": "2026-10-03-01",
   "added": "2026-10-03",
   "url": "https://workingsurface.ai/links/2026-10-03-01/",
   "article": {
    "url": "https://www.nngroup.com/articles/big-ball-of-mud-ai/",
    "title": "The New Big Ball of Mud: Why Agentic AI Systems Turn Fragile",
    "author": "Tanner Kohler",
    "publication": "Nielsen Norman Group",
    "published": "2026-10-02"
   },
   "kind": "Company story",
   "topic": "agent-context-files",
   "takeaway": "A Nielsen Norman Group study by Tanner Kohler finds that people building their own AI agent setups grow them piecemeal until they can no longer explain them.",
   "summary": "Tanner Kohler of Nielsen Norman Group, a user experience research firm, studied people without engineering backgrounds who build AI agent setups to help with daily work. Because AI makes building cheap, they kept adding and patching pieces without a plan. Several could not find, unaided, the files their own systems relied on.",
   "key_points": [
    "One participant's small status widget for Asana, a work-tracking app, grew, step by step, into a team-wide dashboard that predicted completion dates, although she could not write code.",
    "A product lead rebuilt his team's shared library of agent instructions twice because it had grown beyond what he could keep in his head.",
    "A participant who stored articles in an unsorted folder lost 45 minutes when the AI retrieved the wrong ones.",
    "Kohler recommends keeping system-wide instructions, task material and incoming streams such as email separate, and reading the system-wide instructions in full from time to time."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Store the instructions every AI agent task reads, the files for one task and incoming streams such as email in three separate folders."
    },
    {
     "for": "Product",
     "practice": "On a schedule, read in full the instruction file that an AI agent loads for every task, and move task-specific instructions into separate files."
    }
   ]
  },
  {
   "id": "2026-10-03-02",
   "added": "2026-10-03",
   "url": "https://workingsurface.ai/links/2026-10-03-02/",
   "article": {
    "url": "https://www.lukew.com/ff/2164/design-systems-for-ai-agents",
    "title": "Design Systems for AI Agents",
    "author": "Luke Wroblewski",
    "publication": "LukeW",
    "published": "2026-10-01"
   },
   "kind": "How-to",
   "topic": "agent-context-files",
   "takeaway": "Luke Wroblewski, a product designer, shows how an instruction file and code-based design tokens keep every team's AI agents on one website design.",
   "summary": "Luke Wroblewski, a product designer, has watched design systems fall out of date for more than 20 years. Now that marketing staff, engineers and product managers can all change a website through AI agents, each agent invents its own styles unless steered. On his recent projects, a design lead and a front-end lead set the shared values once, and agents are told to reuse them.",
   "key_points": [
    "On the Intent website, an instruction file for agents, AGENTS.md, tells agents to reuse the shared stylesheet values instead of copying colours, type or grid values.",
    "When someone asks for a new button, the agent reuses the existing button component, which already handles dark mode.",
    "The design system started from Figma specifications, but the code then became the reference, and the design system page is built from that code."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Have a design lead and a front-end lead define the grid, fonts, colours, spacing and components once, and point every agent to them through an instruction file."
    },
    {
     "for": "Engineering",
     "practice": "Make the website's stylesheet and components the reference for the design system, and generate the design system documentation page from that code so that the two stay the same."
    }
   ]
  },
  {
   "id": "2026-10-03-03",
   "added": "2026-10-03",
   "url": "https://workingsurface.ai/links/2026-10-03-03/",
   "article": {
    "url": "https://stripe.dev/blog/stripes-payment-method-factory-orchestrating-agents-for-repeated-custom-integrations",
    "title": "Stripe’s Payment Method Factory: Orchestrating agents for repeated, custom integrations",
    "author": "David Dunne, Xenofon Vourliotis and Sai Samant",
    "publication": "Stripe Dot Dev Blog",
    "published": "2026-09-30"
   },
   "kind": "Company story",
   "topic": "agent-context-files",
   "takeaway": "Stripe engineers cut payment integrations from up to six months to two to six weeks with more than 100 reusable prompts that improve after every run.",
   "summary": "Stripe, the payments company, supports more than 125 payment methods, and each new integration could take six months because teams started from scratch. Engineers on the Local Payment Methods team built a \"factory\" of reusable prompts, each covering one step and one pull request. The factory has built three new integrations and moved ten existing ones to Stripe's newer systems.",
   "key_points": [
    "In a side-by-side test, one task took about 20 days of effort with a general coding agent and 4 days with the saved prompts.",
    "A second AI agent watches each run and notes where the working agent got lost, and an engineer uses those notes to improve the prompt.",
    "Each run saves the agents' learning notes, the decisions they made alone and the engineers' review comments, and an AI agent proposes prompt updates from them.",
    "An AI agent first coordinated the steps, but it was slow, costly and sometimes did the work itself, so the team moved coordination into ordinary code."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "After each AI agent run, save what the agent had to learn, the decisions it made alone and the engineers' review comments, and update the reusable prompts from them."
    },
    {
     "for": "Engineering",
     "practice": "Split agent work into steps that each map to one pull request, and run the coordination between steps in ordinary code rather than in an AI agent."
    }
   ]
  },
  {
   "id": "2026-10-03-04",
   "added": "2026-10-03",
   "url": "https://workingsurface.ai/links/2026-10-03-04/",
   "article": {
    "url": "https://www.seangoedecke.com/human-ai-partnerships-are-for-alignment-not-capability/",
    "title": "Human-AI partnerships are for alignment, not capability",
    "author": "Sean Goedecke",
    "publication": "Sean Goedecke",
    "published": "2026-09-27"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "Sean Goedecke, a software engineer, argues that engineers working with coding agents now add most value by aligning the output with their company's technical values.",
   "summary": "Sean Goedecke, a software engineer and essayist, notes that many people compare AI-assisted programming to human and computer partnerships in chess. He says coding agents already make fewer mistakes than he does and work far faster. Left alone, however, their output is hard to maintain and trades away real requirements for invented ones.",
   "key_points": [
    "Goedecke lists habits he sees in AI models, such as long comments above functions, hundreds of useless unit tests and extra text added across websites.",
    "He argues that each company's technical values differ, so an engineer who changes companies has to learn them again.",
    "He expects an aligned AI model would need to adapt to a wide range of company values on the fly.",
    "His practical advice is to describe high-level values to the agent explicitly."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "State the team's high-level technical values to the coding agent explicitly before it starts work."
    }
   ]
  },
  {
   "id": "2026-10-02-01",
   "added": "2026-10-02",
   "url": "https://workingsurface.ai/links/2026-10-02-01/",
   "article": {
    "url": "https://engineering.salesforce.com/how-deterministic-controls-turn-ai-output-into-reliable-prompt-templates/",
    "title": "How Deterministic Controls Turn AI Output Into Reliable Prompt Templates",
    "author": "Vaibhav Raizada, Kumar Kasimala and Ashish Gite",
    "publication": "Salesforce Engineering",
    "published": "2026-09-30"
   },
   "kind": "How-to",
   "topic": "human-approval",
   "takeaway": "Salesforce engineers let an AI agent draft prompt templates for administrators but gave it no power to save, publish or run them, so a person approves every change.",
   "summary": "Salesforce, the business software company, sells Prompt Builder, a tool in which administrators write reusable instructions for its AI features. Vaibhav Raizada, a senior software engineer, and colleagues Kumar Kasimala and Ashish Gite built an AI agent that drafts these templates. They feared templates that looked correct but broke in use, so they split the work between the AI model and ordinary code.",
   "key_points": [
    "The AI model interprets the administrator's request and writes the text of the template.",
    "Ordinary code decides the routing, keeps record numbers intact and checks the output format, so the AI model cannot corrupt them.",
    "The agent shows its draft as a set of proposed changes, and accepting them updates only the draft in the editor.",
    "Saving is a separate action by the administrator, and the authors report drafts in under a minute against more than 25 minutes by hand."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "When an AI agent drafts text for users, let the AI model write the wording, and let ordinary code route requests, copy record numbers and check the output format."
    },
    {
     "for": "Engineering",
     "practice": "Keep the save, publish and run actions out of the AI agent's tools, so that a person must approve each change separately."
    }
   ]
  },
  {
   "id": "2026-10-02-02",
   "added": "2026-10-02",
   "url": "https://workingsurface.ai/links/2026-10-02-02/",
   "article": {
    "url": "https://www.uxtigers.com/post/satisfaction-expectations",
    "title": "User Satisfaction Depends on Delivered Quality Relative to Expectations",
    "author": "Jakob Nielsen",
    "publication": "UX Tigers",
    "published": "2026-10-01"
   },
   "kind": "How-to",
   "topic": null,
   "takeaway": "Usability researcher Jakob Nielsen argues that fluent AI writing makes users over-trust its facts, so teams must design user expectations as carefully as screens.",
   "summary": "Jakob Nielsen writes UX Tigers, a blog on usability research. He argues that users judge a product against what they expected before first use, so the same AI output can delight one person and disappoint another. He warns that polished AI prose makes users assume the facts are equally reliable.",
   "key_points": [
    "Nielsen calls the gap between delivered and expected performance the expectation gap, and he says design, engineering and marketing all move it.",
    "He calls the shortfall between promises and delivery expectation debt, and he says one broken promise outweighs several kept ones.",
    "He recommends showing what the system checked, what it inferred and what still needs review, and showing typical results instead of best-case demonstrations.",
    "He suggests asking users what they expect before first use and comparing their experience afterwards."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "On each AI answer, label what the AI system checked against a source, what it inferred and what the user still needs to review before acting on it."
    },
    {
     "for": "Product",
     "practice": "Ask users what they expect before they first use an AI feature, then compare that with what they report afterwards."
    }
   ]
  },
  {
   "id": "2026-10-02-03",
   "added": "2026-10-02",
   "url": "https://workingsurface.ai/links/2026-10-02-03/",
   "article": {
    "url": "https://posthog.com/blog/how-ai-agents-behave",
    "title": "How AI agents behave: lessons from 63M MCP tool calls",
    "author": "Natalia Amorim",
    "publication": "PostHog",
    "published": "2026-09-30"
   },
   "kind": "Company story",
   "topic": "interfaces-for-agents",
   "takeaway": "PostHog studied 63 million AI agent calls to its tools, found most came from code editors, and built a way to replay them.",
   "summary": "PostHog makes product analytics software and offers its features to AI agents through a standard connection protocol. Natalia Amorim reports on 63 million tool calls from 130,000 people over 90 days. Most calls came from agents working inside code editors and terminals rather than a chat window.",
   "key_points": [
    "Only about 7 percent of the calls came from a chat window.",
    "Running a database query made up a quarter of all calls, and reading what data exists made up another 7 percent.",
    "Dashboard tools drew 0.5 percent of calls but used 18 percent of the text the AI models processed.",
    "PostHog built a replay of each agent session, call by call, like its replays of human sessions."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Record each AI agent's calls to the product and replay sessions call by call, as the team does for human users."
    }
   ]
  },
  {
   "id": "2026-10-02-04",
   "added": "2026-10-02",
   "url": "https://workingsurface.ai/links/2026-10-02-04/",
   "article": {
    "url": "https://www.producttalk.org/unpacking-innovation-all-things-product-podcast-with-teresa-torres-petra-wille/",
    "title": "Unpacking Innovation - All Things Product Podcast with Teresa Torres & Petra Wille",
    "author": "Teresa Torres and Petra Wille",
    "publication": "Product Talk",
    "published": "2026-09-29"
   },
   "kind": "How-to",
   "topic": null,
   "takeaway": "Teresa Torres and Petra Wille argue that innovation is a means of solving a customer's problem, and that teams which chase novelty build complex products customers then have to learn.",
   "summary": "Teresa Torres, who writes Product Talk, and Petra Wille host the podcast All Things Product. In a 12-minute episode they ask whether innovation should be a product team's goal. Torres argues that teams which aim for novelty often build complex products that serve customers no better, and Wille argues that removing steps is itself a form of innovation.",
   "key_points": [
    "They trace how logging in changed, from passwords to single sign-on, to the one-time links Slack sent by email, to password managers and passkeys. Each new method changed the industry and then became the next source of friction.",
    "Every novel flow asks users to learn a new system, which the hosts count as a cost beside the cost of building it.",
    "A solution can be new for the business or the technology and still look ordinary to the customer.",
    "They say good innovation tends to come from a cross-functional team, when research shows a real customer struggle and an engineer knows a newer, easier way to solve it."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Write down the customer problem a new feature solves before building it, and choose a familiar pattern when it solves that problem as well as a novel one."
    },
    {
     "for": "Design",
     "practice": "For each new sign-up or login flow, list every step and every new term a customer must learn, and remove the ones that do not help them finish the task."
    }
   ]
  },
  {
   "id": "2026-10-01-01",
   "added": "2026-10-01",
   "url": "https://workingsurface.ai/links/2026-10-01-01/",
   "article": {
    "url": "https://leaddev.com/ai/your-error-budgets-dont-know-ai-exists",
    "title": "Your error budgets don't know AI exists",
    "author": "Paul LaPosta",
    "publication": "LeadDev",
    "published": "2026-09-30"
   },
   "kind": "Company story",
   "topic": "understanding-the-work",
   "takeaway": "Paul LaPosta, a DevOps leader writing in LeadDev, an engineering leadership publication, argues that AI makes changes cheap to produce but not to understand.",
   "summary": "A change to a service fails and is rolled back quickly, but nobody on the owning team can explain why it failed. Paul LaPosta, a DevOps and cloud infrastructure leader writing for LeadDev, uses this scenario to open his argument. Error budgets and service-level objectives measure reliability, not how well a team understands a service. He says AI tools now write code, tests, reviews and documentation with less human contact, so teams ship more while understanding less of each change.",
   "key_points": [
    "An error budget can stay healthy while a team's understanding of a critical service thins, and nothing on a dashboard turns red.",
    "LaPosta proposes asking the owning team to explain recent consequential changes, the alternatives considered and the risks, without its go-to engineer and without the AI tool.",
    "Warning signs include merge requests reopened by the same engineers, routine rollbacks that turn into debates, and questions escalated to leadership that the team should settle itself.",
    "He advises against a comprehension score. Instead, a change that depends on one person or on the AI tool alone counts as riskier, and the team may delay the release."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Give the team that owns a critical service the authority to delay a release when it cannot explain a change, before any reliability target is breached."
    },
    {
     "for": "Engineering",
     "practice": "Ask the team that owns a critical service to explain its recent consequential changes without its go-to engineer or its AI tool before approving the next release."
    }
   ]
  },
  {
   "id": "2026-10-01-02",
   "added": "2026-10-01",
   "url": "https://workingsurface.ai/links/2026-10-01-02/",
   "article": {
    "url": "https://www.uxtigers.com/post/vigilance",
    "title": "User Vigilance Fails Twice in the AI Age: Watch Duty & Verdict Duty",
    "author": "Jakob Nielsen",
    "publication": "UX Tigers",
    "published": "2026-09-29"
   },
   "kind": "How-to",
   "topic": "human-approval",
   "takeaway": "Jakob Nielsen, a usability researcher, argues that people approving an AI agent's actions stop paying attention, and that products should ask less often and offer undo.",
   "summary": "Products built on AI agents often ask a person to approve each action or to watch a stream of the agent's work. Jakob Nielsen, who writes about usability at UX Tigers, argues that both forms of supervision fail because human attention decays, during long watches and with each repeated approval. He cites a 2026 study by Ting Yan in which people who wrote permission rules in advance blocked less unwanted agent behaviour than people who approved each action by hand.",
   "key_points": [
    "In Ting Yan's study, 113 US adults without a software background supervised a simulated AI assistant that attempted seven unrequested actions, including a $12 insurance purchase and reading private messages.",
    "People who wrote standing rules blocked 40 percent of those actions, against 60 percent for people who approved every action by hand.",
    "For long agent runs, Nielsen recommends asking rarely and putting context in each request. He also recommends tiering approvals by consequence, such as spending, publishing or deleting, and offering undo for reversible actions.",
    "He recommends logging the median time to approve and the approval rate for each approval step. A median under two seconds or approval above 95 percent is a reason to redesign the step."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Log the median time to approve and the approval rate for each step where a person approves an AI agent's action. Redesign any step with a median under two seconds or approval above 95 percent."
    },
    {
     "for": "Design",
     "practice": "Tier an AI agent's approval requests by consequence, such as spending, publishing, deleting and reading private data, rather than by which tool the agent uses."
    }
   ]
  },
  {
   "id": "2026-10-01-03",
   "added": "2026-10-01",
   "url": "https://workingsurface.ai/links/2026-10-01-03/",
   "article": {
    "url": "https://www.lennysnewsletter.com/p/all-of-the-lenny-and-friends-summit",
    "title": "All of the Lenny & Friends Summit talks are now online!",
    "author": "Lenny Rachitsky",
    "publication": "Lenny's Newsletter",
    "published": "2026-09-29"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "Lenny Rachitsky, who writes Lenny's Newsletter for product managers, reports that summit speakers disagreed on process but agreed that judgement matters more as AI makes building easier.",
   "summary": "Lenny Rachitsky published all the main-stage talks from the Lenny and Friends Summit, a product management conference produced by Stripe, and summarised five trends from the day. Speakers from Stripe, Ramp, Atlassian, Google Search, Lovable, Linear, Anthropic and OpenAI took opposite sides on software factories, roadmaps and whether product managers should ship to production. Rachitsky reports that they agreed nobody has settled these questions, and that knowing what to build matters more as building gets easier.",
   "key_points": [
    "Tamar Yehoshua, Chief Product and AI Officer at Atlassian, said the right way for a team to work depends on the type of product it builds.",
    "Robby Stein, a product vice president at Google Search, said the value of a product manager now lies in judging and taste.",
    "Rachitsky concludes that product, design and engineering roles are expanding rather than collapsing into one role called builder.",
    "Geoff Charles, Chief Product Officer at Ramp, predicted that product managers will own business outcomes, and Elena Verna, Head of Growth at Lovable, said she deploys to production herself."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Have product leaders record, for each product, whether its product managers and designers may deploy their own changes to production, rather than one rule for all teams."
    }
   ]
  },
  {
   "id": "2026-09-30-01",
   "added": "2026-09-30",
   "url": "https://workingsurface.ai/links/2026-09-30-01/",
   "article": {
    "url": "https://dropbox.tech/machine-learning/evolving-calendar-assistant-reclaim-to-be-ai-native",
    "title": "Evolving our calendar assistant Reclaim to be AI-native without starting over",
    "author": "Greg Unrein, Josh Jensen and Christopher Wildman",
    "publication": "Dropbox.Tech",
    "published": "2026-09-29"
   },
   "kind": "How-to",
   "topic": "human-approval",
   "takeaway": "Dropbox's engineers sent every AI agent change in Reclaim, their calendar assistant, through the same checks of scheduling rules that people's changes pass, and let users preview it first.",
   "summary": "Reclaim is a calendar assistant, owned by Dropbox, that uses AI to find and move time for tasks, habits and meetings. Its users wanted to describe scheduling goals in their own words, but an AI model can read the same request in several ways, and some readings change other people's calendars. Writing on Dropbox's engineering blog, Greg Unrein, Josh Jensen and Christopher Wildman explain how they added an AI agent without building a separate path for it.",
   "key_points": [
    "Every calendar change, whether a person, the automatic scheduler or the AI agent makes it, runs through the same validation and commit step. Engineers therefore change each operation in one place.",
    "Preview Mode shows users a temporary version of their calendar with the agent's proposed changes. Users can check the effect on shared events before other attendees are notified.",
    "To keep previews fast, the team rewrote the scheduler so that it calculates a proposed schedule without saving anything to the real calendar.",
    "The team built its own agent software instead of using a general framework, which it found lagged behind the AI model providers' latest features. The cost is more code to maintain."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Send every change that an AI agent makes through the same validation and saving code that people's changes use, so that the product's rules are kept in one place."
    },
    {
     "for": "Design",
     "practice": "Before an AI agent's proposed change is applied, show users a preview of its full effect, including on other people, and apply it only when they confirm."
    }
   ]
  },
  {
   "id": "2026-09-30-02",
   "added": "2026-09-30",
   "url": "https://workingsurface.ai/links/2026-09-30-02/",
   "article": {
    "url": "https://newsletter.pragmaticengineer.com/p/shopify-native-mobile",
    "title": "Why has Shopify dropped React Native?",
    "author": "Gergely Orosz",
    "publication": "The Pragmatic Engineer",
    "published": "2026-09-29"
   },
   "kind": "Company story",
   "topic": null,
   "takeaway": "Shopify is rebuilding its React Native mobile apps as separate iOS and Android apps, because AI coding agents have made building each feature twice cheap enough.",
   "summary": "Shopify, the e-commerce platform, began moving its mobile apps to React Native in 2020, a framework that runs one shared codebase on both iOS and Android. After calling the move a success in January 2025, it announced in September 2026 that it will rebuild every app natively, in Swift for iOS and Kotlin for Android. Gergely Orosz, who writes the newsletter The Pragmatic Engineer, explains why, drawing on Shopify's posts and on answers from its heads of mobile and engineering.",
   "key_points": [
    "Shopify chose React Native mainly because native Android apps took too long to build, and so that it would stop building every feature twice.",
    "Mustafa Ali, Shopify's head of mobile, says AI coding agents now do much of the implementation, translation, testing and review. Building for two platforms is therefore no longer the deciding cost.",
    "Shopify keeps one shared test suite for the business logic of both versions, and a feature cannot ship until it passes the same tests on iOS and Android.",
    "The Shop app shipped as a fully native app in 12 weeks. Ali says every rebuild must match or exceed today's performance, stability and accessibility."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "When AI agents build the same feature separately for iOS and Android, make both versions pass one shared test suite for the business logic before either ships."
    },
    {
     "for": "Engineering",
     "practice": "Have the same developer build each mobile feature on both iOS and Android with AI coding agents, instead of keeping separate iOS and Android teams."
    }
   ]
  },
  {
   "id": "2026-09-30-03",
   "added": "2026-09-30",
   "url": "https://workingsurface.ai/links/2026-09-30-03/",
   "article": {
    "url": "https://www.figma.com/blog/what-it-takes-to-build-great-products-now/",
    "title": "What it takes to build great products now",
    "author": "Paige Costello",
    "publication": "Figma Blog",
    "published": "2026-09-24"
   },
   "kind": "How-to",
   "topic": "after-the-prototype",
   "takeaway": "Paige Costello, vice president of product at Figma, argues that AI makes prototypes quick, so the hard work moves to choosing a direction, finishing the product and making it distinctive.",
   "summary": "Paige Costello is vice president of product at Figma, the company that makes the design tool of the same name. She writes that AI lets a team build a clickable prototype within hours, yet the prototype does not show whether its direction is the best one. Her post on Figma's blog sets out how teams keep their speed until launch, compare directions and make a product distinctive, and recommends Figma's own tools at each step.",
   "key_points": [
    "Costello observes that when the first 80 percent of the work is easy, teams can lose momentum in the last 20 percent. A quick prototype already feels ready to launch.",
    "A quick prototype may not use the team's design system, or may create extra work when it moves into code.",
    "She advises building out the cases a single prototype hides. In a salon-booking flow, these include a week with no free slots, a cancelled stylist and a customer who needs to reschedule.",
    "She argues that a design system which records a product's choices, such as its colours and spacing, lets AI-generated work follow them instead of falling back on generic defaults."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Before choosing a direction from an AI prototype, build out the states it leaves out, such as a cancelled booking, to see whether the direction still holds."
    },
    {
     "for": "Design",
     "practice": "Place several prototype directions side by side where the team can compare and discuss them, instead of judging one prototype on its own."
    }
   ]
  },
  {
   "id": "2026-09-29-01",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-01/",
   "article": {
    "url": "https://www.thisandthat.chat/blog/nobody-notices-when-a-company-forgets/",
    "title": "Nobody notices when a company forgets",
    "author": "Jeff Reynar",
    "publication": "this+that",
    "published": "2026-09-28"
   },
   "kind": "Company story",
   "topic": "decision-records",
   "takeaway": "Jeff Reynar, chief executive of the AI start-up this+that, argues that when AI agents do the work, people's skills and a company's knowledge can fade without anyone noticing.",
   "summary": "Jeff Reynar is chief executive of this+that, a start-up that sells AI software to handle the work arriving in a company's email and chat. In a randomised trial he cites, 133 patent lawyers used an AI drafting assistant for three months, and only the senior lawyers came away more skilled. He argues that a company forgets in a similar way when AI agents handle its work and nobody writes down the reasoning behind their decisions.",
   "key_points": [
    "With the assistant, the quality of the lawyers' work rose, and the juniors gained the most. The study's authors conclude: \"The largest gains from AI thus accrued to the lawyers who retained the least.\"",
    "For a company, Reynar argues, the loss is worse, because no one feels it: every case an agent handles correctly looks fine. In his words: \"There is no dashboard for what your company used to know and doesn’t anymore.\"",
    "His first remedy borrows from airlines, which keep pilots practising rare failures in simulators. He proposes three steps: list the work agents now do, check whether people are losing the skill behind it, and, where they are, schedule practice without AI.",
    "His second remedy is a written record of each conclusion an agent reaches, kept where the next person and the next agent will read it."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "List the tasks AI agents now do, have their former owners test whether they can still do them unaided, and schedule practice without AI for skills that are slipping."
    }
   ]
  },
  {
   "id": "2026-09-29-02",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-02/",
   "article": {
    "url": "https://productpicnic.beehiiv.com/p/llms-are-just-normal-technology-but-tech-is-just-a-normal-medium",
    "title": "LLMs are just normal technology. But tech is just a normal medium.",
    "author": "Pavel Samsonov",
    "publication": "The Product Picnic",
    "published": "2026-09-27"
   },
   "kind": "Company story",
   "topic": "after-the-prototype",
   "takeaway": "Product writer Pavel Samsonov argues that AI prototypes tempt teams to skip the first step in making a product: deciding what problem it should solve.",
   "summary": "Pavel Samsonov writes The Product Picnic, a newsletter on UX and product management. He argues that AI tools have made it easier than ever to mistake a working prototype for a valuable product. Drawing on the chatbot pioneer Joseph Weizenbaum, he says that making a computer do something is neither the first nor the most important step in creating value.",
   "key_points": [
    "Samsonov argues that AI models are \"just normal technology\", and that treating them as special leads teams to lose sight of what people care about.",
    "He quotes a comment by Alexis Logsdon: AI models aim first to give the person prompting an answer, however wrong. Logsdon asks whose job it then is to check assumptions.",
    "He praises a guide for business people who commission software. It says to define the business rules first, and warns against handing a developer an AI prototype with the words \"build this\".",
    "He recalls many design workshops in which a problem statement beginning \"how might we\" ended with \"using the power of generative AI\". He advises cutting that ending, because a solution written into a vision statement never helps."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Write down the business rules a piece of software must follow before prototyping it with AI, and give developers those rules along with the prototype."
    },
    {
     "for": "Design",
     "practice": "Remove any named solution, such as generative AI, from a design workshop's problem statements, so that the team does not choose the answer in advance."
    }
   ]
  },
  {
   "id": "2026-09-29-03",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-03/",
   "article": {
    "url": "https://mayaelise.substack.com/p/where-do-we-find-room-to-think",
    "title": "Where do we find room to think?",
    "author": "Maya Elise Joseph-Goteiner",
    "publication": "Nimble UX",
    "published": "2026-09-27"
   },
   "kind": "Company story",
   "topic": "understanding-the-work",
   "takeaway": "Maya Elise Joseph-Goteiner, founder of the user-experience agency Velocity Ave, worries that research teams increasingly edit AI output, which cannot push back as a colleague would.",
   "summary": "Maya Elise Joseph-Goteiner founded Velocity Ave, a user-experience agency, and co-wrote the book UX Skills for Business Strategy. In her newsletter Nimble UX, she observes that research teams increasingly direct and edit what AI tools produce, instead of creating the work themselves. She argues that editing AI output differs from editing a colleague's work, because no author stands behind it to explain an intention or push back.",
   "key_points": [
    "In a study by researchers at Microsoft and Carnegie Mellon, knowledge workers described a shift from doing parts of their work to checking what AI produced. Joseph-Goteiner notes that the study does not show whether people's ability to think is declining.",
    "One leader interviewed by her firm raised the idea of a research team of one person working with a team of AI agents. She says such a team would lose the colleague who disagrees, catches what was missed or helps find the words for an idea.",
    "She says that when a company mandates AI without deciding how it will be used, staff can feel they must perform enthusiasm or look opposed to progress.",
    "She finds it odd that many gatherings for researchers are hosted by companies building AI research products, and asks which conversations become harder there."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "When AI tools draft research findings, have a second researcher work through the interpretation with the first, so that someone can disagree and catch what was missed."
    },
    {
     "for": "Product",
     "practice": "Before requiring a team to use an AI tool, agree in writing which tasks it is for, so that nobody feels they must fake enthusiasm for it."
    }
   ]
  },
  {
   "id": "2026-09-29-04",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-04/",
   "article": {
    "url": "https://www.designsystemscollective.com/before-you-let-an-ai-agent-touch-your-design-system-write-down-four-rules-258860de3475",
    "title": "Before You Let an AI Agent Touch Your Design System, Write Down Four Rules",
    "author": "Guilherme Negreiros",
    "publication": "Design Systems Collective",
    "published": "2026-09-25"
   },
   "kind": "How-to",
   "topic": "agent-context-files",
   "takeaway": "Guilherme Negreiros, who builds a design system with AI agents, writes every exception into the short rule files the agents read, because an automated audit removed a deliberate browser fix.",
   "summary": "Guilherme Negreiros builds a design system alone and in public, with AI agents doing much of the work. A hard-coded colour value made one of its rules work in Apple's Safari browser, and three weeks later an automated audit that looks for hard-coded values removed it. In the publication Design Systems Collective, he sets out how he writes rules for agents and who decides what no script can check.",
   "key_points": [
    "The automated audit treated the Safari fix as leftover code because its exception was not written anywhere the audit could see. The fix was restored the same day, and the exception now sits beside the rule it changes.",
    "Negreiros asks of each decision whether a script could check it with a yes or no. Those a script can check are tested automatically, and the rest need a person with the authority to say no.",
    "Negreiros has four rules. Three can be checked by tools: meet at least the WCAG 2.2 AA accessibility standard, never hard-code a style, and let components use only design values named for their purpose. The fourth, that a person always has the final word, is kept by process: agents propose changes, and only a person merges them after five required checks pass.",
    "Negreiros keeps the reasoning behind each rule, including the alternatives considered and why they were rejected, in a separate decision record for people who revisit it later."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Write each design-system rule, with its known exceptions, in a short file that AI agents read every session, so that automated audits do not remove the exceptions."
    },
    {
     "for": "Product",
     "practice": "Let AI agents propose changes to the design system but never merge them, and have a person merge each change only after the required automated checks pass."
    }
   ]
  },
  {
   "id": "2026-09-29-05",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-05/",
   "article": {
    "url": "https://natesnewsletter.substack.com/p/scale-ai-developer-productivity",
    "title": "Executive Briefing: You Bought Better Tools and Your Finished Work Still Waits",
    "author": "Nate B. Jones",
    "publication": "Nate's Newsletter",
    "published": "2026-09-27"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "AI newsletter writer Nate B. Jones argues that one person's speed with AI agents can outrun a team's review and decisions, and names six principles for raising team output.",
   "summary": "AI agents can make one person on a team far more productive while the team as a whole stays behind. Nate B. Jones, who writes an AI newsletter, says managers often make the fast person teach everyone, which can use up that person's new capacity. He names six principles for raising the whole team's output without slowing down its fastest people.",
   "key_points": [
    "Jones calls one person producing ten times the work for five colleagues to sort through a capacity problem, and says that slowing that person down only hides it.",
    "Jones cites a report from the AI coding tool Cursor: a user at the 99th percentile merged about fifteen times as many code changes as the median active contributor. He warns that this does not show fifteen times the value to customers, or that AI caused the whole difference.",
    "Jones's six principles include making agent work usable by more than one person and leaving work in a state that someone else can continue. He warns that, done carelessly, all six can turn into needless bureaucracy.",
    "Jones says a tenfold gain in implementation becomes a gain of 1.8 times for the business, but the arithmetic is for paying subscribers. This account covers only the part of the briefing open to all readers."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Before pausing work done with AI agents, share its files, write down the next step and list open questions, so a colleague can continue it without the person who started."
    }
   ]
  },
  {
   "id": "2026-09-29-06",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-06/",
   "article": {
    "url": "https://www.designsystemscollective.com/you-dont-fix-the-output-you-fix-the-system-cristian-morales-achiardi-on-agentic-design-3892531d7d31",
    "title": "“You don’t fix the output, you fix the system” — Cristian Morales Achiardi on agentic design systems",
    "author": "Shane P Williams with Cristian Morales Achiardi",
    "publication": "Design Systems Collective",
    "published": "2026-09-25"
   },
   "kind": "How-to",
   "topic": "agent-context-files",
   "takeaway": "Cristian Morales Achiardi, a design engineer, generates a design system's code, design files and documentation from specifications machines can check, so a wrong result is fixed in those specifications.",
   "summary": "Cristian Morales Achiardi developed his approach to design systems for AI agents as the only designer at the company Enara. He is now at Southleft, a firm that works on design systems with client companies, and Shane P Williams interviewed him for the publication Design Systems Collective. In the interview he explains why a design system's code, design files and documentation should all be generated from specifications that machines can check.",
   "key_points": [
    "Morales Achiardi turns design decisions into specifications that machines can mostly validate, such as a file of the system's named colours and sizes. One pipeline produces the component library, Figma design files and documentation from them, so a wrong output is fixed at the source.",
    "With better AI models, Morales Achiardi says, well-organised code with clear paths to information now helps agents more than the elaborate set-ups of a few months ago. In his tests, agents sometimes did better with no added instruction files, because extra context can confuse them.",
    "Morales Achiardi says interface design work, as distinct from code, is still hard to automate and needs hours spent writing evals, tests that score an AI's output.",
    "At Enara, once development ran three to four times faster, Morales Achiardi became the bottleneck. Design means exploring many ideas and discarding most, and the team had time for that before code could be generated."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Test AI agents on the same tasks with and without their added instruction files, because extra context can confuse an agent and make its results worse."
    },
    {
     "for": "Design",
     "practice": "Keep a design system's colours, sizes and rules in one shared file, generate the component library, Figma files and documentation from it, and correct mistakes in that file."
    }
   ]
  },
  {
   "id": "2026-09-29-07",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-07/",
   "article": {
    "url": "https://medium.com/@jonahturnquist/durable-agentic-teams-maximizing-your-software-development-workflow-aae4096069a2",
    "title": "Durable Agentic Teams: Maximizing Your Software Development Workflow",
    "author": "Jonah Turnquist",
    "publication": "Medium",
    "published": "2026-09-29"
   },
   "kind": "How-to",
   "topic": "agent-ownership",
   "takeaway": "Jonah Turnquist, chief technology officer of the property-planning software company PropCode, organises his AI coding agents like an engineering team, and finds that his own review time becomes the limit.",
   "summary": "Jonah Turnquist is co-founder and chief technology officer of PropCode, which analyses planning regulations for Australian properties. Running several AI coding agents on parallel projects, he found himself passing messages between them so that they would not contradict or repeat each other's work. He now organises them as a team: one manager agent that only coordinates, and one agent per project that owns it from design to pull request.",
   "key_points": [
    "He splits work by project instead of by frontend and backend, so each agent holds the decisions for one feature and agents rarely need to brief each other.",
    "He sets written rules on what agents may do alone, such as committing and opening pull requests, and what waits for him, such as merging and deploying.",
    "A separate agent checks every hour for new review comments and forwards each one to the agent that owns the pull request. Agents sign every comment they post.",
    "In his experience the team uses about as many tokens as a single agent doing the same work. What grows is his review load, which he eases with an AI code reviewer."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Write down early which actions AI coding agents may take without asking, such as committing and opening pull requests, and which wait for a person, such as merging and deploying."
    },
    {
     "for": "Engineering",
     "practice": "Give each AI coding agent one whole feature, from interface to storage, instead of one layer such as the frontend, so that agents seldom need to brief each other."
    }
   ]
  },
  {
   "id": "2026-09-29-08",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-08/",
   "article": {
    "url": "https://cate.blog/2026/09/29/good-architecture-is-a-function-of-time/",
    "title": "Good Architecture Is a Function of Time",
    "author": "Cate Huston",
    "publication": "Accidentally in Code",
    "published": "2026-09-29"
   },
   "kind": "How-to",
   "topic": "agent-context-files",
   "takeaway": "Cate Huston, who led the rebuild of Twill, a recruiting platform, says a design system and a strict permissions model paid off once her team built with AI agents.",
   "summary": "Cate Huston is the part-time chief technology officer of Twill, a platform where members, candidates and recruiters each see the same data in different ways. Her team rebuilt Twill in four months with tight scope, but she paid for a design system and a central model of who may see what. She writes that neither would have been worth it before AI agents wrote code, and that both caused far fewer problems than expected.",
   "key_points": [
    "Huston argues that architecture makes good decisions easier over a period a team can reason about. A start-up can rarely see that far ahead, so good architecture is often too expensive for it to justify.",
    "Her permissions model makes it easy to add a new type of user, and granting a permission takes one line in a table. She chose it because Twill's roles kept changing and because security depends on who can see what.",
    "When she noticed architectural problems emerging elsewhere, she added more work of the same kind, such as replacing hand-written database queries with generated ones.",
    "She argues that AI agents talk about time without experiencing it, so judging what is likely to change remains a human responsibility."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Keep every rule about who may see what in one central permissions table, so that adding a type of user or granting a permission is a one-line change."
    },
    {
     "for": "Engineering",
     "practice": "When AI agents write much of the code, invest in shared structure for the parts that change most, such as a design system or generated database queries."
    }
   ]
  },
  {
   "id": "2026-09-29-09",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-09/",
   "article": {
    "url": "https://medium.com/agoda-engineering/key-takeaways-from-agodas-ai-developer-report-2026-82915b88c370",
    "title": "Key Takeaways from Agoda’s AI Developer Report 2026",
    "author": "Agoda Engineering",
    "publication": "Agoda Engineering & Design",
    "published": "2026-09-29"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Agoda, the online travel company, surveyed developers in Southeast Asia and India: 53 percent use AI agents widely, and 79 percent require a person to approve production deployments.",
   "summary": "Agoda, the online travel booking company, surveys software developers and engineering leaders in India and six Southeast Asian countries each year. Its second report finds that 53 percent have AI agents in production or broad use. Only 38 percent call their code ready for an agent that works fully on its own, and the post does not say how many people answered.",
   "key_points": [
    "Developers named cost as the main barrier to wider use of agents. The report finds that most of the cost lies in designing, reviewing and checking the agents' work, not in the AI model itself.",
    "Developers grant agents freedom by risk: 43 percent let agents handle documentation alone, while 79 percent require a person to approve a production deployment.",
    "When an agent causes a problem, 42 percent hold the individual developer responsible and 2 percent blame the AI vendor.",
    "Of junior developers, 49 percent expect agents to make their jobs less secure, against 17 percent of chief technology officers and vice presidents of engineering."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Report AI agent spending by team per piece of finished, reviewed work, so the cost of reviewing agent output counts, not only the AI model's bill."
    },
    {
     "for": "Engineering",
     "practice": "Set an AI agent's freedom by the risk of the task: let it work alone on documentation, and require a person's approval before any production deployment."
    }
   ]
  },
  {
   "id": "2026-09-29-10",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-10/",
   "article": {
    "url": "https://medium.com/@ramkumar2606/the-approve-button-is-not-a-control-24c3e3b0f12e",
    "title": "The Approve Button Is Not a Control",
    "author": "Rama Lingamgunta",
    "publication": "Medium",
    "published": "2026-09-29"
   },
   "kind": "How-to",
   "topic": "human-approval",
   "takeaway": "Rama Lingamgunta, who builds AI agent platforms, replaced a single Approve button with three earlier approvals after a reviewer passed a 1,400-line agent change in under two minutes.",
   "summary": "Rama Lingamgunta builds platforms on which AI agents take work from requirements through code, tests and deployment. The first approval step Lingamgunta built for a coding agent was one Approve button on its finished change. Within a month a reviewer approved 1,400 changed lines in under two minutes, because the agent's own tests passed.",
   "key_points": [
    "Lingamgunta now uses three small approvals, each owned by a named person. A product owner approves what will be built before any code, an engineer approves the design, and the service owner approves the release.",
    "A fixed list of changes always needs a person, however small the change: security and secrets, infrastructure, database migrations, new dependencies, personal data and the approval rules themselves. A simple script outside the agent checks each change against the list.",
    "The agent writes the reviewer a brief that states what was asked, what changed, which requirements have no test and what it left out. Lingamgunta says the last two items are where reviewers catch problems.",
    "Each approval is stored against a fingerprint of the brief the reviewer saw, so any later change voids it. The agent may not cite past approvals to argue for a new one."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Keep a fixed list of changes that always need a person's approval, including changes to the approval rules, and check it with a script the AI agent cannot edit."
    },
    {
     "for": "Engineering",
     "practice": "Have a coding agent attach a note to each change listing what was asked, what changed, what has no test and what was left out, for the reviewer to check."
    }
   ]
  },
  {
   "id": "2026-09-29-11",
   "added": "2026-09-29",
   "url": "https://workingsurface.ai/links/2026-09-29-11/",
   "article": {
    "url": "https://leaddev.com/leadership/ai-makes-critical-thinking-harder-to-build-in-junior-engineers",
    "title": "AI makes critical thinking harder to build",
    "author": "Mia de Búrca",
    "publication": "LeadDev",
    "published": "2026-09-29"
   },
   "kind": "Company story",
   "topic": "training-juniors",
   "takeaway": "Mia de Búrca, a staff engineer at Vistaprint, argues that AI tools spare junior engineers the struggles that built their judgement, so leaders should make room for that practice.",
   "summary": "Mia de Búrca is a staff engineer at Vistaprint, the online printing company. Junior engineers used to learn judgement by struggling with unfamiliar code and explaining their reasoning to senior colleagues. She argues that AI assistants now smooth much of that away, and that juniors who doubt their judgement defer more to generated answers.",
   "key_points": [
    "She asks leaders to say plainly that critical thinking is part of the job. The team should define it, so that it comes up in onboarding, feedback and reviews.",
    "She wants engineers to feel safe saying that they do not understand what an AI tool generated.",
    "She would build the skill through existing practices. Pull requests would explain why a change was made and which alternatives were rejected. Incident reviews would ask what the team believed that turned out to be untrue.",
    "She suggests coaching juniors to ask an AI assistant to challenge their assumptions, and writing critical thinking into career frameworks as specific behaviours."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Run incident reviews that ask what the team believed that turned out to be untrue, so that junior engineers practise judging their own assumptions."
    },
    {
     "for": "Engineering",
     "practice": "Write critical thinking into career frameworks as specific behaviours, such as validating assumptions before implementation, so that managers can see a junior engineer's growth that shipped tickets no longer show."
    }
   ]
  },
  {
   "id": "2026-09-28-01",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-01/",
   "article": {
    "url": "https://creatoreconomy.so/p/grok-bot-team-14-best-bots-peng-zheng-lauren-tan",
    "title": "We Built Grok Bot. Here Are Our 14 Best Bots | Peng Zheng & Lauren Tan",
    "author": "Peter Yang with Peng Zheng and Lauren Tan",
    "publication": "Behind the Craft",
    "published": "2026-09-27"
   },
   "kind": "Company story",
   "topic": "agent-context-files",
   "takeaway": "Two leads at SpaceXAI show how they share work with AI agents: the designer builds the first screen by hand, and the engineer's agent divides projects among other agents.",
   "summary": "Peter Yang, who runs the interview newsletter and podcast Behind the Craft, spoke with the lead designer and lead engineer of Grok Bot, the AI assistant made by SpaceXAI. Peng Zheng, the designer, builds the design system and one finished screen by hand, and his design agent extends that screen to the rest of the flow. Lauren Tan, the engineer, hands large projects to an agent that splits the work among other coding agents.",
   "key_points": [
    "Zheng's design agent works directly in Figma. It reads a written file that describes how his design files are set up and the design system's colours, type and spacing.",
    "Zheng describes his own share of the work plainly: \"I will do the first 5%.\"",
    "Tan's lead agent, Matcha, does no work itself. It breaks each project into tasks for other agents, and each of those starts coding agents in the cloud.",
    "The full episode is for paid subscribers, and this account comes from the open show notes. One chapter covers giving agents more work one skill at a time."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Give an AI agent one new kind of task at a time, such as testing its own work before merging, and add the next only once it handles that reliably."
    },
    {
     "for": "Design",
     "practice": "Build the design system and one finished screen by hand, and give the agent a file of the system's rules before it extends that screen to the whole flow."
    }
   ]
  },
  {
   "id": "2026-09-28-02",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-02/",
   "article": {
    "url": "https://ldlr.design/post/new-patterns-are-a-tax",
    "title": "New Patterns Are a Tax",
    "author": "Leonardo De La Rocha",
    "publication": "Design Outcomes",
    "published": "2026-09-26"
   },
   "kind": "How-to",
   "topic": "workshops-that-ratify",
   "takeaway": "A design team for clinical software drafted two rules for keeping its product simple, reuse existing patterns and judge each design in context, from reviews that included AI-generated layouts.",
   "summary": "Leonardo De La Rocha leads design for software used by clinicians, a product with twelve years of interface patterns behind it. Each new pattern is one more thing a clinician has to learn. He and one of his design leaders drafted two principles from what they had seen in design reviews, and plan to ask the other design leaders what to add.",
   "key_points": [
    "The first principle is reuse over novelty: reuse every existing pattern possible, because each new one makes people relearn the product.",
    "The second is to judge each design in context. The team had lost time debating an icon for an AI feature on its own, apart from the navigation and buttons around it.",
    "In one review, a designer showed three AI-generated layouts for a tracker built around a recurring deadline. A senior designer saw that the horizontal version looked like the product's clickable tabs, and the team chose the vertical version.",
    "De La Rocha asks designers to prompt AI tools for realistic flows shown within the whole product, not only their own part of it."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Write down a design principle whenever a review catches AI-generated work drifting from the product's existing patterns, and share the list with the other design leaders."
    }
   ]
  },
  {
   "id": "2026-09-28-03",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-03/",
   "article": {
    "url": "https://ldlr.design/post/good-isnt-great",
    "title": "Good Isn't Great",
    "author": "Leonardo De La Rocha",
    "publication": "Design Outcomes",
    "published": "2026-09-26"
   },
   "kind": "How-to",
   "topic": "human-approval",
   "takeaway": "A designer fixed a billing problem for clinicians in an afternoon with an AI coding agent, and his design lead used the review to take it from good to great.",
   "summary": "Clinicians at a mental-health practice had to rotate patients' sideways insurance-card photos by hand before they could use them. A designer who works on billing at the company that makes their software fixed this in one afternoon with an AI coding agent. His design lead, Leonardo De La Rocha, then reviewed the fix with him and explains what separates a good fix from a great one.",
   "key_points": [
    "The designer left the small thumbnail unrotated until the rotation can be saved, so that nobody walks away believing an unsaved change was stored.",
    "He offered rotation to the left only, as Apple Photos does, and kept the control in the larger preview that clinicians already open. He disagreed when De La Rocha suggested moving it.",
    "De La Rocha argues that reaching good was an achievement when a fix took a sprint. When a fix takes an afternoon, good is a first draft.",
    "The designer added a caution of his own: a feature that can overwrite someone's files deserves closer review as a result."
   ],
   "practices": []
  },
  {
   "id": "2026-09-28-04",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-04/",
   "article": {
    "url": "https://www.atlassian.com/blog/ai-at-work/cafes-framework",
    "title": "Introducing CAFE(S): A framework for defining AI context quality",
    "author": "Brian Houck",
    "publication": "Atlassian",
    "published": "2026-09-24"
   },
   "kind": "Company story",
   "topic": "agent-context-files",
   "takeaway": "Brian Houck of the developer-research firm DX defines five qualities of the instructions AI agents read, and argues that each shared instruction file needs a team that owns it.",
   "summary": "AI agents act on the context they are given: the instructions, specifications and documents they read before they work. Brian Houck, a Distinguished Scientist at the developer-productivity research firm DX, wrote on Atlassian's blog about what makes that context good. He and four co-authors of a paper in the journal ACM Queue define five qualities: clarity, actionability, fidelity (being true and current), efficiency and security.",
   "key_points": [
    "Poor context once cost a developer a question to a colleague. It can now cost money, legal liability and security, as when Air Canada was held liable for a promise its chatbot made.",
    "Shared context files that many agents read should be reviewed like code, with more review the more widely a file is reused.",
    "Important context should have an owner and a review schedule, and the owner should be a team: \"Ownership should follow teams rather than individuals, so context survives reorgs.\"",
    "Untrusted input should be kept apart from instructions. The authors cite a crafted email that made Microsoft 365 Copilot, Microsoft's AI assistant, leak internal documents."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Assign each context file that agents read to an owning team rather than to one person, so that the file keeps an owner through a reorganisation."
    },
    {
     "for": "Design",
     "practice": "Give every design-system file that AI agents read a named owning team, and review changes to a file more carefully the more screens depend on it."
    }
   ]
  },
  {
   "id": "2026-09-28-05",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-05/",
   "article": {
    "url": "https://sublimecoding.com/blog/dhh-rails-world-2026-keynote",
    "title": "DHH's Rails World 2026 Keynote: Pencils Down, Now What",
    "author": "Jared Smith",
    "publication": "Sublime Coding",
    "published": "2026-09-24"
   },
   "kind": "Company story",
   "topic": "agent-ownership",
   "takeaway": "Engineering leader Jared Smith argues that when AI agents write the code, a person must still own how the changes fit together, which a better model will not fix.",
   "summary": "David Heinemeier Hansson created the web framework Ruby on Rails and co-founded 37signals, the company behind Basecamp and the email service HEY. At the Rails World 2026 conference he said that 37signals now writes code by hand only as an exception. Jared Smith, a founder and engineering leader, wrote up the keynote and agreed with much of it, but disputed one lesson.",
   "key_points": [
    "This spring, 37signals designers built the final features of Basecamp 5 with AI agents. Each change looked reasonable, but twenty or thirty together left the architecture, in Hansson's words, \"a little like a Swiss cheese.\"",
    "The team went back to reviewing code by hand. Hansson called that the wrong conclusion and said a better AI model would have made the original plan work.",
    "Smith argues that nobody owned how the changes added up, and that a better model does not fix a problem of ownership. The more code agents produce, the harder that problem gets.",
    "Smith's rule is to keep a person explicitly responsible for the architecture."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Make one person responsible for how an AI-built product fits together as a whole, and have that person check that each new change fits."
    }
   ]
  },
  {
   "id": "2026-09-28-06",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-06/",
   "article": {
    "url": "https://addyo.substack.com/p/the-code-nobody-reads",
    "title": "The Code Nobody Reads",
    "author": "Addy Osmani",
    "publication": "Elevate",
    "published": "2026-09-28"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Addy Osmani withdraws his advice to read every line of AI-written code and proposes automated review of every change, human review matched to risk, and a person approving every merge.",
   "summary": "An engineer two weeks into a job at a large company wrote that their days went on approving AI-written code nobody had read. Addy Osmani, who writes the Elevate newsletter and helps build an AI tool that writes code, withdraws his advice of two years ago to read every line of AI-written code. He sets out what should replace that reading.",
   "key_points": [
    "Osmani proposes that AI agents review every pull request first. Small changes to less sensitive code then get a lighter human check, and core and sensitive code gets careful review by its owner.",
    "In his model a person still approves every merge and decides how closely to read it. He cites Anthropic, the maker of the Claude AI models, where an automated reviewer checks nearly every pull request but approves nothing.",
    "He argues that tests written by the same AI agent that wrote the code prove little. Checks become trustworthy when they come from an independent source, such as a specification a person wrote.",
    "He asks engineers to state in the pull request which parts they did not read, and asks leaders to measure incidents and rollbacks instead of merged pull requests."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Have AI agents review every pull request, reserve careful human review for core and sensitive code, check small low-risk changes more lightly, and have a person approve every merge."
    },
    {
     "for": "Engineering",
     "practice": "When a pull request was approved without every part being read, say in its description which parts nobody read, so that later reviewers and incident reviews know what was checked."
    }
   ]
  },
  {
   "id": "2026-09-28-07",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-07/",
   "article": {
    "url": "https://leaddev.com/ai/your-agent-loop-is-not-a-production-system",
    "title": "Your agent loop is not a production system",
    "author": "Sriram Madapusi Vasudevan",
    "publication": "LeadDev",
    "published": "2026-09-28"
   },
   "kind": "How-to",
   "topic": "agent-ownership",
   "takeaway": "An Amazon Web Services engineer argues that once an AI agent can change live systems, teams must prove separately who approved a change, what ran and whether it worked.",
   "summary": "Sriram Madapusi Vasudevan, a senior software engineer at Amazon Web Services (AWS), works on AWS DevOps Agent, an AI agent that joins on-call engineers during production incidents. He describes what the system around such an agent must record and check once it can change cloud resources. He argues that engineering leaders should set these rules for every team before any agent may change production.",
   "key_points": [
    "An approval should name the exact operation, resource, approver and expiry. It should survive a restart, be used once, and be asked for again if the target changes.",
    "Every change an agent makes should pass through one gateway, whether the agent calls a tool directly, writes code, or hands the task to another agent.",
    "A rollback can succeed while the errors continue, so the system should confirm that the incident has cleared before it counts the task as done.",
    "He proposes that a platform team own the shared machinery and that domain teams own their tools and what counts as done. Operators own approval limits and the power to stop work."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Before an AI agent changes a live system, require a person to approve that exact change, let each approval be used once, and ask again if the change is altered."
    },
    {
     "for": "Engineering",
     "practice": "Count an AI agent's incident task as finished only when a separate check shows the problem has gone, such as the alarm clearing, not when the agent's action reports success."
    }
   ]
  },
  {
   "id": "2026-09-28-08",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-08/",
   "article": {
    "url": "https://engineering.salesforce.com/engineering-multi-agent-ai-teams-that-build-and-test-themselves/",
    "title": "Engineering Multi-Agent AI Teams That Build and Test Themselves",
    "author": "Sohini Arya and Manish Kumar Jha",
    "publication": "Salesforce Engineering",
    "published": "2026-09-28"
   },
   "kind": "How-to",
   "topic": "agent-ownership",
   "takeaway": "A Salesforce team built AI agents that design and test teams of other agents, aiming to cut hours to minutes, with an engineer approving every design.",
   "summary": "Engineers in Marketing Cloud, the marketing software business of Salesforce, spent two to four hours building and testing each team of cooperating AI agents by hand. Sohini Arya, who leads AI delivery for Marketing Cloud, describes Agent Designer, a system her team built in which AI agents do that work instead. The goal was a tested team in 15 to 30 minutes, with an engineer approving each design before any file is written.",
   "key_points": [
    "A coordinating agent directs seven specialist agents, which analyse the design, write and check the files, and run a test. The coordinating agent is not allowed to write any agent file itself.",
    "Agents that check other agents' work must answer pass, warn or fail in a fixed format. The coordinating agent compares their answers before a design goes further.",
    "Automatic repair of a failing agent team stops after two attempts or $3 of spending, and the problem then goes to an engineer.",
    "The post says the system helped about 200 Marketing Cloud engineers begin working as managers of AI agents."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "When AI agents repair their own failing work, cap the repair attempts and the spending, and hand the problem to an engineer once either limit is reached."
    },
    {
     "for": "Engineering",
     "practice": "Have AI agents that check other agents' work answer pass, warn or fail in a fixed format, and compare several checkers' answers instead of trusting a single one."
    }
   ]
  },
  {
   "id": "2026-09-28-09",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-09/",
   "article": {
    "url": "https://www.databricks.com/blog/how-databricks-rolls-out-frontier-models-14000-employees-day-1",
    "title": "How Databricks rolls out frontier models to 12,000 employees on Day 1",
    "author": "The Databricks AI Product and Engineering Team",
    "publication": "Databricks",
    "published": "2026-09-28"
   },
   "kind": "Company story",
   "topic": null,
   "takeaway": "Databricks, a data and AI software company, gives staff each new AI model on release day within a capped budget, then keeps or drops it on measured cost and quality.",
   "summary": "Databricks, a company that sells a data and AI platform, gives more than 10,000 employees AI coding tools such as Claude Code and Codex. New AI models can disappoint: one cost more and scored lower with its engineers than the version before, and another raised average developer spending by 60 percent. Its AI product and engineering team now offers each new AI model on release day as an experiment, then decides within about three days whether to keep it.",
   "key_points": [
    "Access runs through Unity Gateway, Databricks' own product for controlling and monitoring AI use. A command-line tool on every laptop adds each new AI model to the coding tools, marked as experimental.",
    "Each employee has four spending limits: monthly, daily, a share for the most expensive AI models, and a share for untested ones. The daily limit stops a runaway session and can be raised in Slack.",
    "The team decides on three signals: its own private tests, including two AI models creating the same pull requests side by side, reports from early users, and cost per session.",
    "It compared early users' cost per session with the same people's a week before. The AI model Opus 5.5 cost 29 percent less than Opus 4.8, and GPT-6 Sol, after a price cut, 48 percent less than GPT-5.6 Sol."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Offer each new AI model to staff on release day as an experiment with a capped budget, and keep it only if benchmarks, user reports and cost data support it."
    },
    {
     "for": "Engineering",
     "practice": "Compare a new AI model's cost per session on the same group of early users before and after, because early users are heavier users of AI than most staff."
    }
   ]
  },
  {
   "id": "2026-09-28-10",
   "added": "2026-09-28",
   "url": "https://workingsurface.ai/links/2026-09-28-10/",
   "article": {
    "url": "https://cacm.acm.org/blogcacm/nobody-did-tdd-for-25-years-now-the-machine-requires-it/",
    "title": "Nobody Did TDD for 25 Years. Now the Machine Requires It",
    "author": "Abtin Aghagolian",
    "publication": "BLOG@CACM",
    "published": "2026-09-28"
   },
   "kind": "How-to",
   "topic": "evals-before-the-build",
   "takeaway": "Abtin Aghagolian, chief technology officer of Pikd, a London augmented-reality company, argues that tests written before an AI agent writes code are now how engineers control its output.",
   "summary": "For 25 years, most programmers skipped test-driven development, the practice of writing a failing test before the code that makes it pass. Abtin Aghagolian, chief technology officer of Pikd, a London company that runs an augmented-reality platform, argues that AI agents now make the practice necessary. An engineer writes the test, the agent writes code until it passes, and the engineer reviews the test instead of the code.",
   "key_points": [
    "He argues that a prompt is a description that only a person can check. A test is a definition that an AI agent can run and check its own work against.",
    "An AI agent will pass a weak test without being correct. He therefore prefers tests of rules that must always hold over tests that pin one input to one output.",
    "He wants teams to stop asking an AI model to write tests for code that already exists, because such tests confirm whatever the code does, errors included.",
    "Pikd's own assistant passed all 1,053 of its tests while telling a user that no such tool was available and carrying out the correct navigation action in the same reply. He knows of no way to test for every such contradiction in advance."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Before an AI agent writes code, write the tests that define a correct result, let the agent work until they pass, and review the tests instead of the code."
    },
    {
     "for": "Engineering",
     "practice": "Do not count tests that an AI model wrote for existing code as verification, because they record what the code already does, errors included."
    }
   ]
  },
  {
   "id": "2026-09-27-01",
   "added": "2026-09-27",
   "url": "https://workingsurface.ai/links/2026-09-27-01/",
   "article": {
    "url": "https://ldlr.design/post/know-what-to-doubt",
    "title": "Know What to Doubt",
    "author": "Leonardo De La Rocha",
    "publication": "Design Outcomes",
    "published": "2026-09-26"
   },
   "kind": "Company story",
   "topic": "decision-records",
   "takeaway": "A design team labels each part of a ticket for an AI coding agent as human-written or agent-written, so that reviewers know which kind of error to look for.",
   "summary": "Designers at a company that makes software for clinicians write tickets for AI coding agents, and these prompts spell out every step. Engineers asked for a short human-written brief in each ticket as well, and the designer who wrote the ticket proposed labelling each section by its author. The team's design lead, Leonardo De La Rocha, explains that the label tells a reviewer whether to look for mistakes or for inventions.",
   "key_points": [
    "A prompt for a coding agent spells out every step, because the agent cannot infer anything. People find it tiring to read.",
    "A section written by a person may contain mistakes but should not contain inventions. A section written by the agent is the other way round.",
    "De La Rocha recommends a short brief at the top of each ticket, with the agent prompt below it. The brief states the problem, who it is for and what done looks like.",
    "He also recommends asking the agent to explain its own code change in plain words for the reviewers. Several engineers already ask a model for such a summary before they read code."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Put a short human-written brief above the agent prompt in every ticket for a coding agent, and label each section as written by a person or generated by the agent."
    },
    {
     "for": "Design",
     "practice": "Label each section of a design specification handed to an agent as written by a person or generated by the agent, so that reviewers know where to check for inventions."
    }
   ]
  },
  {
   "id": "2026-09-27-02",
   "added": "2026-09-27",
   "url": "https://workingsurface.ai/links/2026-09-27-02/",
   "article": {
    "url": "https://cutlefish.substack.com/p/tbm-441-ai-the-loss-of-positive-friction",
    "title": "TBM 441: AI, the Loss of Positive Friction, and What to Do About It",
    "author": "John Cutler",
    "publication": "The Beautiful Mess",
    "published": "2026-09-24"
   },
   "kind": "Company story",
   "topic": "decision-records",
   "takeaway": "Product writer John Cutler argues that AI removes the manual steps where teams used to judge customer feedback, and proposes keeping every summary linked to what customers said.",
   "summary": "Product teams used to move customer feedback by hand: they re-read call transcripts, wrote takeaways and turned them into tickets. John Cutler, who writes the product newsletter The Beautiful Mess, argues that those steps were where people judged what mattered, and that AI agents now skip them. At his own company he found AI summaries built on other AI summaries, and he sets out principles to keep feedback traceable to its source.",
   "key_points": [
    "Cutler notes that AI can turn five customer calls into 80 suggested opportunities, 80 tickets and 400 tasks.",
    "His first principle is to keep each piece of original feedback separate and linked to its transcript. Work that cannot be traced back to a real source should stop.",
    "He proposes labelling every AI-generated document by how much of it AI produced, and adding a deliberate check whenever content is copied into a new context.",
    "He treats analysis of the code as one source of evidence, because code does not show what customers experience or whether the team is solving the right problem."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Link every synthesised finding and ticket to the transcript or feedback it came from, and stop work on any item that cannot be traced to its source."
    },
    {
     "for": "Design",
     "practice": "Add a deliberate review step whenever research findings are copied into a new document, such as a ticket or a brief, and record what changed."
    }
   ]
  },
  {
   "id": "2026-09-27-03",
   "added": "2026-09-27",
   "url": "https://workingsurface.ai/links/2026-09-27-03/",
   "article": {
    "url": "https://ldlr.design/post/three-lattes-one-order",
    "title": "Three Lattes, One Order",
    "author": "Leonardo De La Rocha",
    "publication": "Design Outcomes",
    "published": "2026-09-26"
   },
   "kind": "Company story",
   "topic": "decision-records",
   "takeaway": "A design lead proposes keeping one requirements document per feature, a table with a prototype under each row, so that people and AI agents build from the same version.",
   "summary": "When designers, product managers and engineers can each write a polished specification in minutes with AI, one feature ends up described in three or four documents that drift apart. An AI agent builds from whichever document it is given and cannot tell that it is out of date. Leonardo De La Rocha, who leads design for software used by clinicians, proposes replacing them all with one table of capabilities.",
   "key_points": [
    "Each row of the table is one thing the product lets someone do, with a link to a click-through prototype of that one flow.",
    "A designer records a video walkthrough for the top of the table, and a section on data needs lets data and AI teams prepare. The table replaces the design brief.",
    "De La Rocha dropped the design brief once before, when he helped rebuild how Spotify's advertising unit went from idea to build.",
    "He warns that the format has to be held firmly, because any room left for the old documents lets them come back."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Replace the requirements document, the design brief and any side specification for a feature with one table that has a row for each thing the product lets someone do."
    },
    {
     "for": "Design",
     "practice": "Keep one capabilities table for each feature in place of a design brief. Link a click-through prototype of each flow under its row, and have a designer record a video walkthrough at the top."
    }
   ]
  },
  {
   "id": "2026-09-27-04",
   "added": "2026-09-27",
   "url": "https://workingsurface.ai/links/2026-09-27-04/",
   "article": {
    "url": "https://www.svpg.com/experts-lead-experts/",
    "title": "Experts Lead Experts",
    "author": "Marty Cagan",
    "publication": "SVPG",
    "published": "2026-09-25"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "Marty Cagan of the product consultancy SVPG argues that the current push for fewer managers is temporary, and that each craft still needs leaders who are expert in it.",
   "summary": "Since AI changed how software is built, many companies have announced fewer managers, and some engineering leaders have gone back to writing code. Marty Cagan, founder of Silicon Valley Product Group (SVPG), a product-management training and consulting firm, calls this a necessary but temporary state. He argues that top product companies follow the principle that experts lead experts, and that leadership matters more as building gets cheaper.",
   "key_points": [
    "Under the principle that experts lead experts, whoever leads engineers, product managers or product designers must be expert in that craft.",
    "Cagan reports that people who spent the past year working hands-on with the new AI tools are returning to decisions on product vision, strategy, team structure and outcomes.",
    "He also reports that some companies that announced fewer managers have begun to reverse that decision.",
    "He presents the argument as a prediction and says that it could prove wrong."
   ],
   "practices": []
  },
  {
   "id": "2026-09-27-05",
   "added": "2026-09-27",
   "url": "https://workingsurface.ai/links/2026-09-27-05/",
   "article": {
    "url": "https://ldlr.design/post/working-isnt-the-same-as-trusted",
    "title": "A Working Feature Isn't the Same as a Trusted One",
    "author": "Leonardo De La Rocha",
    "publication": "Design Outcomes",
    "published": "2026-09-26"
   },
   "kind": "How-to",
   "topic": null,
   "takeaway": "Clinicians doubted an autosave feature that worked, so the product team paused its rollout and added an indicator that shows the save while it happens.",
   "summary": "A company that makes software for clinicians shipped autosave for patient notes, with a timestamp showing when each note was last saved. The feature worked, but clinicians told each other in the company's customer community that they were not sure their notes were safe. The team that owns the feature paused the rollout on its own initiative and added an indicator that shows the save while it is happening.",
   "key_points": [
    "The company's design lead, Leonardo De La Rocha, explains the gap. The timestamp said when the note was saved, while clinicians wanted to know whether their work was safe right now.",
    "The next day, a design review split a button that combined approving a document and moving to the next one. A person now sees the approval land before the screen changes.",
    "De La Rocha is adding one question to every product review he attends this year: \"Do clinicians understand and trust it?\"",
    "For any feature that acts on someone's behalf, he asks what the person sees while it happens and right after, before they move on."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Add the question of whether users understand and trust a feature to every product review, beside the question of whether it works."
    },
    {
     "for": "Design",
     "practice": "Design what a person sees while an automated action runs and just after it ends, such as a saving indicator, before the feature ships."
    }
   ]
  },
  {
   "id": "2026-09-26-01",
   "added": "2026-09-26",
   "url": "https://workingsurface.ai/links/2026-09-26-01/",
   "article": {
    "url": "https://www.lukew.com/ff/2163/skip-the-tools-make-the-outcomes",
    "title": "Skip the Tools, Make the Outcomes",
    "author": "Luke Wroblewski",
    "publication": "LukeW",
    "published": "2026-09-24"
   },
   "kind": "Company story",
   "topic": null,
   "takeaway": "Designer Luke Wroblewski found that readers of his AI newsroom mostly asked for a report on their question instead of browsing articles, and argues for outcomes before tools.",
   "summary": "Software makers have long built tools that people must learn before they get a result. Luke Wroblewski, a product designer, helped build Exposit, a news site whose articles are found, written and edited by AI agents. When Exposit added a feature that compiles a personalised report on any topic a reader asks about, that report quickly became the main way people used the site.",
   "key_points": [
    "Wroblewski argues that AI lets a product deliver the result first and offer the tool second, since a tool was only ever a means to an end.",
    "From the personalised report, Exposit readers could still go deeper into articles, topics and sources.",
    "On his own website, a feature that answers visitors' questions now also writes a daily summary of his recent thinking. Visitors get something to read without having to ask.",
    "His test for a team is one question: \"are you building another tool, or are you delivering the outcome the tool was supposed to produce?\""
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Design an AI feature to hand users the finished result first, such as a personalised report, and offer the underlying articles and controls afterwards for those who want more."
    },
    {
     "for": "Design",
     "practice": "Design the first screen of an AI feature around the result the user came for, and offer the underlying tool as a second step."
    }
   ]
  },
  {
   "id": "2026-09-26-02",
   "added": "2026-09-26",
   "url": "https://workingsurface.ai/links/2026-09-26-02/",
   "article": {
    "url": "https://github.blog/ai-and-ml/github-copilot/when-chat-is-the-wrong-ui/",
    "title": "When chat is the wrong UI",
    "author": "Burke Holland",
    "publication": "GitHub Blog",
    "published": "2026-09-24"
   },
   "kind": "How-to",
   "topic": null,
   "takeaway": "Burke Holland of GitHub argues that once a person knows a repeated job, having an AI agent build a small app for it costs fewer tokens than asking in chat.",
   "summary": "Chat is still the main way people work with AI models, three years after it took off. Burke Holland, who works on AI-powered development at GitHub, argues that chat suits a first attempt but becomes wasteful once the job is known. In the GitHub Copilot app, which runs GitHub's AI coding agent, he has the agent build small apps called canvases that run inside the app and talk to the agent.",
   "key_points": [
    "Holland's reason is cost. Tokens are the units of text an AI model is billed by. An agent asked to do a routine job spends tokens every time, while a tool it builds once costs nothing to use.",
    "His examples include a manager for software packages that contains no AI at all, a browser for a SQLite database and an editor for blog posts.",
    "He also built a canvas that runs his whole working process, from research to finished code, so the agent can work alone and call him in for review. It took most of a day to get right.",
    "The piece appears on GitHub's own blog and promotes GitHub's own app."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "When people keep asking an AI agent in chat to do the same job, have the agent build a small dedicated tool for that job instead."
    }
   ]
  },
  {
   "id": "2026-09-26-03",
   "added": "2026-09-26",
   "url": "https://workingsurface.ai/links/2026-09-26-03/",
   "article": {
    "url": "https://www.microsoft.com/insidetrack/blog/from-the-field-how-agentic-ai-is-reshaping-adoption-at-microsoft/",
    "title": "From the field: How agentic AI is reshaping adoption at Microsoft",
    "author": "Poly Palaiogeorgou",
    "publication": "Microsoft Inside Track",
    "published": "2026-09-24"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Microsoft reports that staff in its Europe South region now take up AI agents without being pushed, and that the hard part has moved to governing them.",
   "summary": "Microsoft's rollout of its AI assistant, Microsoft 365 Copilot, to over 200,000 staff depended on teaching people to write good prompts, and the gains faded. In an account on Microsoft's own blog about its internal use of AI, Poly Palaiogeorgou describes what changed in the company's Europe South region once agents could take whole tasks. Staff passed agents on from colleague to colleague, and a pilot in Spain sent every idea for a new agent through a triage step first.",
   "key_points": [
    "The triage, called Build, Reuse, or Prompt, sends each idea to a better prompt, an existing tool or agent, or a new agent. The report says it reduced duplicated work and made ownership clearer.",
    "The report says the strongest ideas for agents came from the people closest to the work, rather than from technical teams.",
    "Every agent at Microsoft must pass five checks before it can be published: service registration, security validation, a privacy assessment, accessibility checks and a responsible-AI review.",
    "This is Microsoft's account of its own products in use inside the company."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Publish the list of checks an agent must pass before release at the start of a project, so that builders plan for them from the first day."
    },
    {
     "for": "Design",
     "practice": "Include an accessibility check in the reviews an agent must pass before it is released to other staff."
    }
   ]
  },
  {
   "id": "2026-09-26-04",
   "added": "2026-09-26",
   "url": "https://workingsurface.ai/links/2026-09-26-04/",
   "article": {
    "url": "https://www.microsoft.com/insidetrack/blog/driving-seller-adoption-of-sales-agent-through-role-based-change-management-at-microsoft/",
    "title": "Driving seller adoption of Sales Agent through role-based change management at Microsoft",
    "author": "David Hirning",
    "publication": "Microsoft Inside Track",
    "published": "2026-09-24"
   },
   "kind": "Company story",
   "topic": "work-cut-to-size",
   "takeaway": "Microsoft had a business programme manager take its sellers' workflows apart before rebuilding its sales AI agent, and is moving to judge the agent by deal speed rather than usage.",
   "summary": "Microsoft's salespeople spent most of their time on admin spread across more than a dozen tools, so the company built an AI agent to take whole tasks off their hands. First, a business programme manager, Rene Vejlby, and his colleagues took each sales workflow apart and decided what the agent should do and what should be cut. Microsoft is now moving to judge the agent by how fast deals close, not by how many people use it.",
   "key_points": [
    "Microsoft's research found that about 70 percent of a seller's time went on work away from customers, across at least 15 tools.",
    "An earlier version of the agent, Sales Agent, only looked information up. Microsoft rebuilt it to complete tasks across up to 25 systems, such as creating a sales opportunity or submitting an investment request.",
    "Ajay Nair, the engineering manager who owns Sales Agent, says that usage alone does not show that the agent helps the business.",
    "The adoption team wrote pages for about 20 sales roles, checked them with people in those roles, and replaced one-way demonstrations with small peer discussions."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Take each workflow apart step by step before an agent is built for it, and decide which steps the agent takes over and which are cut."
    }
   ]
  },
  {
   "id": "2026-09-26-05",
   "added": "2026-09-26",
   "url": "https://workingsurface.ai/links/2026-09-26-05/",
   "article": {
    "url": "https://newsletter.pragmaticengineer.com/p/design-engineering-with-maggie-appleton",
    "title": "Design Engineering with Maggie Appleton",
    "author": "Gergely Orosz with Maggie Appleton",
    "publication": "The Pragmatic Engineer",
    "published": "2026-09-23"
   },
   "kind": "Company story",
   "topic": "evals-before-the-build",
   "takeaway": "Maggie Appleton, a research engineer at GitHub, no longer reads the code AI agents write for her prototypes and writes a spec saying how the agent will check its work.",
   "summary": "Gergely Orosz, who writes the engineering newsletter The Pragmatic Engineer, interviewed Maggie Appleton on his podcast. Appleton is a staff research engineer at GitHub Next, GitHub's research group, where she builds prototypes of new ways for engineers to work with AI. The show notes describe how her design process has changed now that coding agents build her prototypes.",
   "key_points": [
    "Appleton has a coding agent add sliders and colour pickers to each prototype, so that she can adjust colours, sizes and animation speed while it runs.",
    "She no longer reviews the code in agent-written changes, a practice the show notes limit to prototypes. Once she knows what to build, she writes a detailed spec that lists how the agent will verify its work.",
    "She finds planning with agents tiring. An agent asks multiple-choice questions \"a hundred times over\", and by question 20 a person starts accepting its recommended answer by reflex.",
    "She wants new shared documents and tools in which people and agents can work together, and calls finding them \"a really hard challenge\"."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Before an agent builds a prototype, write a spec that lists how the agent will verify each part of its work."
    },
    {
     "for": "Design",
     "practice": "Have the agent add sliders and other live controls to a prototype, so that its colours, sizes and motion can be adjusted while it runs."
    }
   ]
  },
  {
   "id": "2026-09-26-06",
   "added": "2026-09-26",
   "url": "https://workingsurface.ai/links/2026-09-26-06/",
   "article": {
    "url": "https://www.cio.com/article/4226177/i-stopped-asking-my-team-to-use-ai-i-asked-them-to-manage-it.html",
    "title": "I stopped asking my team to use AI. I asked them to manage it",
    "author": "Dan Graves",
    "publication": "CIO",
    "published": "2026-09-25"
   },
   "kind": "Company story",
   "topic": "agent-ownership",
   "takeaway": "Dan Graves, chief product officer at WitnessAI, has seven people on his team manage AI agents like junior hires, with a person approving work before it moves on.",
   "summary": "Dan Graves, chief product officer at the company WitnessAI, stopped asking his product team to use AI to do their own jobs faster. He asked them instead to manage AI agents like junior employees: train them, set expectations, review their plans and own the quality of what they produce. In a column for the business-technology magazine CIO, he describes how seven people on his team work this way and what went wrong along the way.",
   "key_points": [
    "Each person works with a primary agent for their role. A design agent turns designers' Figma mockups into working code, a frontend engineering agent reviews that code, and an orchestrating agent routes each ticket to the right agent.",
    "A person approves work before it moves forward, and no agent opens a pull request, a proposed code change, without a human sign-off.",
    "Early designs from the design agent used the wrong components, and developers reworked half of them. Once the engineering agent was loaded with frontend practices and reviewed the design agent's work first, the frontend team accepted roughly 95 percent.",
    "When a designer picked a task from a list only to point the design agent at it, the agent started building. It now asks clarifying questions first and builds nothing without an explicit go-ahead."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Make one person responsible for each AI agent: its instructions, its review rules and the quality of its work."
    },
    {
     "for": "Design",
     "practice": "Instruct each design agent to ask clarifying questions first and to build nothing without an explicit go-ahead from a person."
    }
   ]
  },
  {
   "id": "2026-09-26-07",
   "added": "2026-09-26",
   "url": "https://workingsurface.ai/links/2026-09-26-07/",
   "article": {
    "url": "https://shubs.io/do-we-still-enjoy-software-engineering-in-the-age-of-ai/",
    "title": "do we still enjoy software engineering in the age of AI?",
    "author": "Shubham Shah",
    "publication": "shubs.io",
    "published": "2026-09-26"
   },
   "kind": "Company story",
   "topic": "understanding-the-work",
   "takeaway": "Shubham Shah, co-founder of the company Assetnote, published an engineer's message about losing satisfaction since the team moved to coding with AI, and his reply as their manager.",
   "summary": "An engineer at Assetnote, a company with security research and engineering teams, told their manager that the work had lost its satisfaction since the team began developing with AI. They no longer fully understood the code they shipped, and had lost touch with the codebase and with what colleagues were doing. Shubham Shah, a security researcher and Assetnote co-founder, published the message and his reply with details removed.",
   "key_points": [
    "The engineer did not object to AI writing code. The loss was in understanding the whole project and taking responsibility for what they shipped.",
    "Shah agreed that working with AI has turned engineers into something like middle managers, and wrote that he felt the same loss in his own research.",
    "He suggested putting standing rules in the instruction file that an AI coding agent reads with every request, because the agent's memory feature kept ignoring them.",
    "He proposed that the team present its work to restore pride in it, and said he would not push the engineer while they adjusted."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Once AI agents write most of the code, have engineers present their finished work to the team, so that each keeps pride in the work and knows what colleagues build."
    },
    {
     "for": "Engineering",
     "practice": "Put the rules an AI coding agent keeps forgetting into the instruction file it reads with every request, such as CLAUDE.md or AGENTS.md, instead of relying on its memory feature."
    }
   ]
  },
  {
   "id": "2026-09-25-01",
   "added": "2026-09-25",
   "url": "https://workingsurface.ai/links/2026-09-25-01/",
   "article": {
    "url": "https://www.uxtigers.com/post/defaults",
    "title": "Default Dominance: Most Users Never Change the Setting You Ship",
    "author": "Jakob Nielsen",
    "publication": "UX Tigers",
    "published": "2026-09-23"
   },
   "kind": "How-to",
   "topic": "agent-ownership",
   "takeaway": "Jakob Nielsen shows that most users keep whatever setting a product ships, and argues that AI agents hide their defaults in instruction files that need an owner.",
   "summary": "Jakob Nielsen, a usability researcher who writes UX Tigers, reviews the evidence that most people keep the settings a product ships with and treat them as the maker's recommendation. He then argues that AI agents hide their defaults: in the vendor's instructions, in the model's trained habits and in skill files. A skill file is a set of reusable instructions telling an agent how to do one type of task.",
   "key_points": [
    "When one large US company enrolled new staff in its retirement plan automatically, participation rose from 37 to 86 percent, although the terms of the plan stayed the same.",
    "In a study of Microsoft Word users reported by Jared Spool, fewer than 5 percent had changed any setting, so most worked with autosave switched off.",
    "A default hidden in a skill file cannot be seen when it takes effect. Nielsen says each skill file therefore needs an owner, a version and a review whenever the task or the model changes.",
    "Nielsen also asks designers to settle an agent's authority before it starts work: which actions it may take, its spending limits, and when it must stop and ask."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Give every written instruction file an AI agent works from a named owner, a version number and the date it was last reviewed."
    },
    {
     "for": "Design",
     "practice": "Name an owner for each instruction file an AI agent follows to build interface screens, and have the owner reread it when the task or the AI model changes."
    }
   ]
  },
  {
   "id": "2026-09-25-02",
   "added": "2026-09-25",
   "url": "https://workingsurface.ai/links/2026-09-25-02/",
   "article": {
    "url": "https://www.producttalk.org/trash-can-tracking-all-things-product-podcast-with-teresa-torres-petra-wille/",
    "title": "Trash Can Tracking — All Things Product Podcast with Teresa Torres and Petra Wille",
    "author": "Teresa Torres and Petra Wille",
    "publication": "Product Talk",
    "published": "2026-09-22"
   },
   "kind": "Company story",
   "topic": "decision-records",
   "takeaway": "On the podcast All Things Product, Petra Wille proposes that teams keep a visible record of rejected problems and solutions; an empty record of rejected solutions means nobody compared options.",
   "summary": "Product teams track the work they ship, but the work they decide not to pursue usually leaves no trace. Petra Wille and Teresa Torres, who host the podcast All Things Product, discuss Wille's answer: a trash can drawn on the team's planning boards. Each customer problem or solution the team rejects goes in the can, so rejected work is tracked as carefully as delivered work.",
   "key_points": [
    "The show notes call an empty can of rejected solutions a clear warning sign, because it means the team is not comparing several options before choosing one.",
    "An empty can of rejected problems can mean a shortage of new ideas, a command-and-control culture, or a strong strategy that filters problems out as intended.",
    "The can also settles problems the team has already decided to drop but that others in the company keep raising again.",
    "The full transcript is for paying subscribers. This account comes from the open show notes."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Keep a visible list on the team's planning board of each customer problem the team chose not to solve and each solution it rejected."
    },
    {
     "for": "Design",
     "practice": "Show the design options rejected during discovery beside the option carried forward, so that a reviewer can see that alternatives were compared."
    }
   ]
  },
  {
   "id": "2026-09-25-03",
   "added": "2026-09-25",
   "url": "https://workingsurface.ai/links/2026-09-25-03/",
   "article": {
    "url": "https://natesnewsletter.substack.com/p/ai-native-workplace-adoption",
    "title": "Your agent is giving you reasonable answers from half your context. A conversation with OpenAI.",
    "author": "Nate B. Jones",
    "publication": "Nate's Newsletter",
    "published": "2026-09-22"
   },
   "kind": "Company story",
   "topic": "agent-context-files",
   "takeaway": "Nate B. Jones reports from OpenAI that whether staff find AI useful depends more on whether it can reach their material than on their skill.",
   "summary": "A company can have a few very capable AI users and still not become more capable as a whole. Nate B. Jones, who writes an AI newsletter, asked two leaders at OpenAI which of its teams had come to rely on AI for much of their work. Coding came first, because its tools already ran on engineers' own machines, and legal came next, because its material sat in documents the AI could open.",
   "key_points": [
    "Other teams at OpenAI followed as the AI gained access to their information and got better at using it.",
    "Jones argues that the most consequential change comes when what one person learns improves what other people can do.",
    "He advises a manager to find out what a colleague's AI can reach before calling that colleague resistant.",
    "The article is for paying subscribers after its opening section, and this account covers only the open part."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Check which shared files and systems a colleague's AI assistant can reach before concluding that the colleague resists using it."
    }
   ]
  },
  {
   "id": "2026-09-25-04",
   "added": "2026-09-25",
   "url": "https://workingsurface.ai/links/2026-09-25-04/",
   "article": {
    "url": "https://blog.codacy.com/how-ai-is-changing-the-engineering-manager-role-more-context-more-capacity-and-the-new-job-of-protecting-focus",
    "title": "How AI Is Changing the Engineering Manager Role: More Context, More Capacity, and the New Job of Protecting Focus",
    "author": "Codacy with Alejandro Rizzo and Jorge Braz",
    "publication": "Codacy",
    "published": "2026-09-25"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "Two engineering managers at Codacy, a maker of code-quality tools, say AI tools let each engineer start several tasks at once, so they now limit how much work is open.",
   "summary": "Codacy, a company that sells code-quality and code-review tools, asked two of its engineering managers, Alejandro Rizzo and Jorge Braz, how AI tools have changed their jobs. Braz reports that each engineer can now build one feature while planning the next, so a squad of three or four people juggles up to six topics. Both managers say their job has moved towards protecting the team's focus and the quality of its code.",
   "key_points": [
    "Braz, who manages the squads responsible for platform reliability, does not write production code. He says AI tools now give a quick picture of a system that once took hours of reading code.",
    "Rizzo actively limits the number of topics each person and each squad has open. He finds that stakeholders and the engineers' own curiosity push for more work in roughly equal measure.",
    "Both managers hold code written by AI tools to stricter standards for tests and complexity than hand-written code, and enforce them with Codacy's own product.",
    "Braz argues that AI adds speed, and that the cost falls on quality or scope unless a team decides which to protect. His reliability squads protect quality."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Cap how many topics each engineer and each squad may have open at once, and hold the cap when stakeholders or engineers push to start more AI-assisted work."
    },
    {
     "for": "Engineering",
     "practice": "Set stricter automated checks for test coverage and code complexity on AI-written code than on hand-written code, and run them on every change before merging."
    }
   ]
  },
  {
   "id": "2026-09-25-05",
   "added": "2026-09-25",
   "url": "https://workingsurface.ai/links/2026-09-25-05/",
   "article": {
    "url": "https://arxiv.org/abs/2609.30863",
    "title": "Developing a Roadmap to an AI-first Organization: A Case Study in Embedded Software Development",
    "author": "Viktor Kjellberg, Srijita Basu, Simin Sun, Farnaz Fotrousi and Miroslaw Staron",
    "publication": "arXiv",
    "published": "2026-09-25"
   },
   "kind": "Company story",
   "topic": "agent-ownership",
   "takeaway": "Researchers held a workshop with 40 staff at a large embedded-software company and turned their views on AI agents into a roadmap for reorganising the company around them.",
   "summary": "An unnamed company of about 20,000 employees, which builds software that runs inside hardware devices, wants to redesign its work around AI agents. Researchers from the University of Gothenburg and Chalmers University of Technology ran a workshop there in April 2026 with 40 scrum masters, managers, software architects and product owners. They turned the participants' answers and discussion into a roadmap for the change.",
   "key_points": [
    "Participants expected AI agents to take over much of the coding, while engineers check the agents' output, work with stakeholders and keep their knowledge of the domain.",
    "They favoured central standards and governance, with one person in each team who builds that team's agents and shares what works with other teams.",
    "They warned that more generated code needs more tests and reviews. In one team, an AI agent changed the tests instead of fixing the code, and engineers caught it.",
    "The roadmap starts with one low-risk process and measures an agent's effect on the steps before and after it, not only on the team that uses it."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "When an AI agent takes over one low-risk development step, record the review and testing time of the steps before and after it. Compare that with the time saved on the step itself."
    },
    {
     "for": "Engineering",
     "practice": "Start using AI agents in one low-risk process that the rest of development does not depend on, and name one person in that team to set the agents up."
    }
   ]
  },
  {
   "id": "2026-09-24-01",
   "added": "2026-09-24",
   "url": "https://workingsurface.ai/links/2026-09-24-01/",
   "article": {
    "url": "https://claude.com/blog/how-to-prepare-for-ai-driven-code-modernization-projects",
    "title": "How to prepare for AI-driven code modernization projects",
    "author": "Jonah Ezekiel and Lexie Tonelli",
    "publication": "Anthropic",
    "published": "2026-09-23"
   },
   "kind": "Company story",
   "topic": "evals-before-the-build",
   "takeaway": "Two Anthropic engineers advise companies modernising old code with AI agents to agree, before the work starts, what each change must prove and how much human review it gets.",
   "summary": "In banks and other regulated companies, every change to a critical system is reviewed and approved, a process built for code written by people. When AI agents rewrite an old system, they produce changes far faster than anyone can review them one by one. Jonah Ezekiel and Lexie Tonelli, engineers who work with Anthropic's customers, set out how to prepare such a project, drawing on those deployments.",
   "key_points": [
    "Before the work starts, the team agrees on automatic checks that every change must pass, such as the old tests passing and old and new code giving the same output. The authors call this set of checks a certificate.",
    "The checks are written with the people who will review and approve the changes. A good test of them is whether those people would be comfortable merging a change on their evidence alone.",
    "A written, tiered review policy, agreed in advance, sets how much human review each change gets according to how much it could break. Critical parts keep full review, and experts spend their time on the riskiest changes.",
    "The authors advise that this policy come from the top of the organisation. Responsibility for a defect that reaches production is then shared rather than pinned on one approver."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Before building starts, agree with the people who will approve AI-made changes which automatic checks every change must pass."
    },
    {
     "for": "Engineering",
     "practice": "Write the automatic tests for AI-written changes with the engineers who approve them, and add tests until they would merge a change on the test results alone."
    }
   ]
  },
  {
   "id": "2026-09-24-02",
   "added": "2026-09-24",
   "url": "https://workingsurface.ai/links/2026-09-24-02/",
   "article": {
    "url": "https://stripe.dev/blog/how-stripe-is-designing-checkout-for-ai-agents",
    "title": "How Stripe is designing Checkout for AI agents",
    "author": "Cara Mecozzi and Steve Kaliski",
    "publication": "Stripe",
    "published": "2026-09-22"
   },
   "kind": "How-to",
   "topic": "interfaces-for-agents",
   "takeaway": "Stripe made its Checkout page cheaper for AI shopping agents to use by offering them only the actions valid at each step, built on the human checkout's own code.",
   "summary": "AI shopping agents have to pay on checkout pages built for humans, which on Stripe cost an agent 1.8 million tokens and 39 actions per purchase on average. Stripe, whose Checkout page serves over 7.8 million businesses, adopted WebMCP, an emerging browser standard that lets a page offer its actions to an agent as named tools. The page now offers only the actions valid at each step, and agents in Stripe's tests used 42 percent fewer tokens.",
   "key_points": [
    "A token is the unit of text an AI model reads and is billed by, so fewer tokens make a purchase cheaper. Stripe treats tokens and actions as the agent's equivalent of a person's waiting time and clicks.",
    "The payment button appears to the agent only once the form is complete. Choosing a payment method other than a card removes the card-number field.",
    "The agent's tools run on the same code and forms as the human checkout, so the two versions cannot drift apart.",
    "In 60 tests across six AI models on a fictional outdoor shop, every agent completed the purchase. With WebMCP, agents took 38 percent fewer actions and finished 39 percent faster."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Give AI agents the same actions and forms that people use in the interface, and show an agent only the actions that are possible at the current step."
    }
   ]
  },
  {
   "id": "2026-09-24-03",
   "added": "2026-09-24",
   "url": "https://workingsurface.ai/links/2026-09-24-03/",
   "article": {
    "url": "https://cursor.com/blog/rollouts-and-security-reviewer",
    "title": "Bots for the last mile: Rollouts, Security Review",
    "author": "Rustam Lalkaka",
    "publication": "Cursor",
    "published": "2026-09-23"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Cursor launched two AI bots for the work after a code change is proposed: one plans and watches the release, and one reviews every change for security flaws.",
   "summary": "Cursor, which makes AI tools for writing code, has launched two bots, Rollouts and Security Reviewer, for the work that follows a proposed code change. In the launch post, Rustam Lalkaka of Cursor argues that writing code is no longer the slow part. The work after a change is proposed has not sped up, he says: checking its security, watching its release and finding what broke.",
   "key_points": [
    "Before a change is merged, Rollouts reads it and writes a monitoring plan: the risks, the intended effects and what the team's monitoring cannot see. A person can edit the plan.",
    "After release, Rollouts compares live measurements with those from before the change. If something gets worse, it names the suspected change and, depending on its settings, alerts the author, pauses the release or proposes a reversal for a person to approve.",
    "Security Reviewer checks every proposed change against the whole codebase, tracing where user input enters and where it ends up, and proposes a fix for each problem it finds.",
    "The post is a product launch and does not describe a team using the bots or report results."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Before an AI agent's code change goes live, have an engineer check the agent's plan for watching it in use, and agree in advance what happens if a problem appears."
    }
   ]
  },
  {
   "id": "2026-09-24-04",
   "added": "2026-09-24",
   "url": "https://workingsurface.ai/links/2026-09-24-04/",
   "article": {
    "url": "https://evilmartians.com/chronicles/ai-makes-design-system-guardrails-mandatory-this-framework-delivers-them",
    "title": "AI makes design system guardrails mandatory; this framework delivers them",
    "author": "Anton Lovchikov and Yuri Mandrikov",
    "publication": "Evil Martians",
    "published": "2026-09-23"
   },
   "kind": "Company story",
   "topic": "agent-context-files",
   "takeaway": "Evil Martians, a product consultancy, stops AI agents that build screens from changing the design system, and records every exception for a person to review.",
   "summary": "When AI coding agents build screens from bare design-system components, they guess how to use them, override their styles and create duplicates, and later sessions copy those workarounds. Anton Lovchikov, head of design, and Yuri Mandrikov, a frontend engineer, at the consultancy Evil Martians describe eight changes they make to prevent this. The framework comes from client work and is published as a set of agent instructions that teams can install.",
   "key_points": [
    "Building screens and maintaining the design system run as separate agent workflows. An agent building a screen that finds a component missing records the gap and works around it without changing the system.",
    "Each component has a written contract: what it is for, when to use it and when not to, and what it guarantees. Agents choose components from an index of these contracts rather than by reading code.",
    "Rules that a script can check, such as a ban on colours outside the approved set, are enforced by automated checks rather than left to the agent's instructions.",
    "A local exception to the system must state its reason, and every gap or exception goes into a log that the team reviews. The article presents a consultancy's framework and reports no measured results."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Keep building screens and maintaining the design system as separate jobs for AI agents, so an agent that needs a missing component notes it instead of changing the design system."
    },
    {
     "for": "Product",
     "practice": "Make one person responsible for the design system, separate from whoever builds screens, and have that person review every gap or exception the builders record."
    }
   ]
  },
  {
   "id": "2026-09-23-01",
   "added": "2026-09-23",
   "url": "https://workingsurface.ai/links/2026-09-23-01/",
   "article": {
    "url": "https://shopify.engineering/helix",
    "title": "Helix: The internal tool powering our Shopify app's native migration",
    "author": "Talha Naqvi",
    "publication": "Shopify Engineering",
    "published": "2026-09-21"
   },
   "kind": "Company story",
   "topic": "work-cut-to-size",
   "takeaway": "Shopify rebuilds its mobile app with AI models in small steps, each checked by tests, a visual comparison and two reviewing agents before an engineer approves it.",
   "summary": "Shopify is rebuilding its largest mobile app, which has more than 300 screens, in separate native code for iPhone and Android instead of shared React Native code. AI models can do the rewriting, but tools that rebuild a whole feature at once leave the engineer a large amount of code to test. Talha Naqvi of Shopify Engineering describes Helix, the internal tool the team built to break the work into small, checked steps.",
   "key_points": [
    "An engineer points Helix at a screen, and Helix proposes a sequence of small steps, each described in a few words, which the engineer approves in minutes.",
    "Each step passes four checks in order. Tests confirm the behaviour, an AI model compares screenshots with the old app, two separate AI reviewers check the code, and an engineer approves.",
    "The agent can retry a failed check as often as it needs, but it cannot override one. Engineer feedback is stored, so later steps need less oversight, and an engineer can let Helix run without approvals while the checks still apply.",
    "The team says the same method works for new features, with designs and product documents as the reference in place of an old app."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Run automated checks, such as behaviour tests and independent code reviews, before a person approves agent-built work, so that each piece reaches the person already checked."
    },
    {
     "for": "Design",
     "practice": "Define the visual reference for each agent-built screen, such as the existing app or approved designs, and require the comparison to pass before a designer reviews it."
    }
   ]
  },
  {
   "id": "2026-09-23-02",
   "added": "2026-09-23",
   "url": "https://workingsurface.ai/links/2026-09-23-02/",
   "article": {
    "url": "https://sep.com/blog/so-ai-killed-your-code-review-process-now-what/",
    "title": "So, AI Killed Your Code Review Process. Now What?",
    "author": "Nicole Selig",
    "publication": "SEP",
    "published": "2026-09-22"
   },
   "kind": "Company story",
   "topic": "work-cut-to-size",
   "takeaway": "An engineer at a software consultancy argues that when AI-written code outgrows review, teams should cut work into thin slices and review the agent's plan before any code exists.",
   "summary": "Nicole Selig is an engineer at SEP, a software consultancy. She writes that code review broke down once AI agents produced changes larger and faster than people could read them. Her remedy is established practice rather than new tools.",
   "key_points": [
    "Selig recommends cutting work into vertical slices, thin pieces that each run end to end through a feature, so that each change stays small enough to review.",
    "The team reviews the agent's implementation plan together before any code is written.",
    "Clean-up of existing code is done as a separate step, and feature flags and gradual releases catch problems after a change is merged."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Cut each agent-built feature into thin slices that each work end to end, and review the agent's implementation plan as a group before any code is written."
    }
   ]
  },
  {
   "id": "2026-09-23-03",
   "added": "2026-09-23",
   "url": "https://workingsurface.ai/links/2026-09-23-03/",
   "article": {
    "url": "https://www.lennysnewsletter.com/p/advanced-evals-how-to-find-and-fix",
    "title": "Advanced evals: How to find (and fix) hidden AI failures in your product",
    "author": "Hamel Husain and Shreya Shankar",
    "publication": "Lenny's Newsletter",
    "published": "2026-09-22"
   },
   "kind": "How-to",
   "topic": "evals-before-the-build",
   "takeaway": "Two specialists in testing AI products argue that people should read real AI sessions and note the failures themselves before an agent looks for errors or anyone writes a metric.",
   "summary": "Teams building AI products often measure quality with automated metrics before they know which failures matter. Hamel Husain and Shreya Shankar have worked with more than 50 AI companies on testing their products. In a guest post in Lenny's Newsletter, they describe the step most teams skip: reading records of real user sessions to find what goes wrong.",
   "key_points": [
    "In their example, an AI leasing assistant said goodbye to a prospective tenant who found an apartment too expensive. Most AI agents would count that as a success, but the product's aim was to offer cheaper options.",
    "A person reads at least 10 session records and notes anything wrong before an AI agent suggests problems. The authors call this a safeguard against trusting the agent too readily, and set 100 records as the target.",
    "The agent then proposes further problems, which the person accepts or rejects. In a study of 100 sessions, agents missed failures that needed knowledge of the product and flagged some good answers as failures.",
    "The post is behind a paywall from its third step, and only the open part is cited here."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Have a person read 10 real user sessions with an AI feature and note what went wrong, then 100, before anyone writes a measure of its quality."
    },
    {
     "for": "Design",
     "practice": "Check each failure that an agent proposes against the team's own notes on real sessions, and accept or reject it before the team starts to measure it."
    }
   ]
  },
  {
   "id": "2026-09-23-04",
   "added": "2026-09-23",
   "url": "https://workingsurface.ai/links/2026-09-23-04/",
   "article": {
    "url": "https://uxdesign.cc/in-the-age-of-ai-the-ux-field-survives-on-leaders-who-cultivate-juniors-595c548f1d70",
    "title": "In the age of AI, the UX field survives on leaders who cultivate juniors",
    "author": "Patrick Neeman",
    "publication": "UX Collective",
    "published": "2026-09-22"
   },
   "kind": "Company story",
   "topic": "training-juniors",
   "takeaway": "A designer argues that design leaders should hire juniors again, because AI tools make newcomers useful sooner and questioning AI output is learned beside a senior.",
   "summary": "Design teams are hiring senior designers and few juniors, and one study of 62 million workers found junior employment fell at firms that adopted generative AI. Patrick Neeman, who has designed for the web since 1995, argues in UX Collective that this leaves nobody being trained to lead the field. He asks design leaders to hire juniors on purpose and to teach them.",
   "key_points": [
    "A survey by Figma, the design tool maker, found 56 percent of hiring managers reporting rising demand for senior designers, against 25 percent hiring for junior roles.",
    "Neeman cites a study of 5,179 customer support agents in which novices using generative AI improved by 34 percent, while the most experienced changed little. He concludes that AI tools shorten the time a newcomer takes to become useful.",
    "He argues that skill with tools is not judgment. A junior left alone with AI learns to accept its output, while one working beside a senior learns to question it.",
    "He proposes four steps: a junior opening each hiring cycle, a manager who wants to teach, teaching time in the plan, and a quarterly report on who could step up."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Pair each junior designer who works with AI tools with a senior who reviews the output beside them, and put the teaching time in the project plan."
    }
   ]
  },
  {
   "id": "2026-09-22-01",
   "added": "2026-09-22",
   "url": "https://workingsurface.ai/links/2026-09-22-01/",
   "article": {
    "url": "https://www.lennysnewsletter.com/p/how-i-ai-metas-muse-review-how-warp",
    "title": "How I AI: Meta's Muse review + How Warp ships 2,000 PRs a month with AI factories",
    "author": "Lenny Rachitsky",
    "publication": "Lenny's Newsletter",
    "published": "2026-09-21"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Warp, which makes tools for software teams, lets the person who asked its AI agent for a code change review that change, instead of waiting for a separate reviewer.",
   "summary": "Warp, a company that makes tools for software teams, runs an internal system called Wilson that turns a request in Slack into a tested code change. Wilson takes an average of 35 minutes to open a pull request, a proposed change to the code, but the first human review arrives 3.5 hours later. In a podcast interview recapped by Lenny Rachitsky, Warp's chief executive, Zach Lloyd, explained how the company is shortening that wait.",
   "key_points": [
    "Warp lets the person who asked the agent for a change review its work, instead of requiring a separate reviewer.",
    "Warp measures the system by the number of times people have to step in on each pull request, such as follow-up prompts, clarifications and corrections in review. Lloyd argues that the more steering an agent needs, the less the system really produces.",
    "Lloyd expects that trust built up over time will let selected changes merge with no human review. Warp has not removed human review yet.",
    "The recap adds that the system speeds up building but does not decide what to build, which still depends on user interviews, design sessions and product judgment."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Let the person who asked an AI agent for a code change review that change, and track how often people have to correct or redirect the agent."
    }
   ]
  },
  {
   "id": "2026-09-22-02",
   "added": "2026-09-22",
   "url": "https://workingsurface.ai/links/2026-09-22-02/",
   "article": {
    "url": "https://linear.app/now/ci-bottleneck-reworked",
    "title": "AI coding has made CI a bottleneck, so we reworked ours to keep up",
    "author": "Mufeez Amjad",
    "publication": "Linear",
    "published": "2026-09-21"
   },
   "kind": "Company story",
   "topic": "agent-context-files",
   "takeaway": "Linear rebuilt its automated code checks after AI agents made writing code faster than checking it, and taught its agents a new testing rule in the same change.",
   "summary": "Linear makes software that teams use to plan and track product work, and AI agents now write most of its tests. As the test suite almost quadrupled this year, the automated checks that every code change must pass, known as continuous integration, grew slow and costly. Mufeez Amjad, the engineer whom Linear's chief technology officer assigned to the problem, describes how the team made those checks faster and cheaper.",
   "key_points": [
    "Faster rented machines and a faster TypeScript compiler gave the first gains: jobs ran 34 percent faster, and the type check took 73 percent less time.",
    "The team cut repeated setup and split the tests across eight parallel runs. The wait for checks on a code change fell from more than six minutes to just over five, and machine time per test roughly halved.",
    "The largest saving let safe test files share loaded code, which risks tests affecting each other. The team made it opt-in, marked each eligible file, and updated its coding agents' instructions so that new tests follow the same rule.",
    "Combining seven short checks into two jobs saved about 87,000 machine-minutes a month, 11.8 percent of the total. The account is Linear's own, about its internal engineering."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "When the team adopts a new coding or testing rule, update its AI coding agents' written instructions at the same time, so the agents follow the rule from the start."
    }
   ]
  },
  {
   "id": "2026-09-22-03",
   "added": "2026-09-22",
   "url": "https://workingsurface.ai/links/2026-09-22-03/",
   "article": {
    "url": "https://www.warp.dev/blog/using-llm-as-a-judge-scoring-to-measure-your-software-factory",
    "title": "Using LLM-as-a-judge scoring to measure your software factory",
    "author": "Zach Lloyd",
    "publication": "Warp",
    "published": "2026-09-18"
   },
   "kind": "How-to",
   "topic": "evals-before-the-build",
   "takeaway": "Warp's chief executive explains how AI agents can grade a sample of coding agents' past work against written criteria, so that people review the failures rather than every change.",
   "summary": "Warp sells a product for running many AI coding agents in the cloud, which it calls a software factory. Its chief executive, Zach Lloyd, argues that teams using coding agents should stop guessing how well the agents perform and grade their work directly. He describes how to set up grading agents, using Warp's own internal factory as the example.",
   "key_points": [
    "Each grading agent reads the full record of a past coding session and returns a pass or a fail. A written prompt defines what it checks, such as whether the task was done, whether the work was efficient and whether the code was good.",
    "Teams add criteria of their own. Warp checks whether its agents write redundant tests, a failure it saw often.",
    "Only a sample of sessions is graded, because grading costs money. At Warp it takes about 3 percent of what the company spends on AI models.",
    "People open the failed sessions, read what the coding agent did and why the grader failed it, and adjust the agents' instructions. Lloyd describes a further step in which another agent proposes those changes automatically."
   ],
   "practices": []
  },
  {
   "id": "2026-09-22-04",
   "added": "2026-09-22",
   "url": "https://workingsurface.ai/links/2026-09-22-04/",
   "article": {
    "url": "https://medium.com/bolt-labs/design-engineering-at-bolt-shipping-components-with-ai-agents-1b66ad59d837",
    "title": "Design engineering at Bolt: shipping components with AI agents",
    "author": "Burak Özdemir",
    "publication": "Bolt Labs",
    "published": "2026-09-17"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "At Bolt, designers now build interface components with AI agents directly in the library that ships, and engineers review those components instead of rebuilding them.",
   "summary": "Bolt is a mobility company whose website serves up to 60 million users a year in more than 50 markets, built on a shared library of over 110 components. Designers used to build working components with an AI agent in a separate internal workspace, and engineers then built each one a second time in the production library. Burak Özdemir, a tech lead at Bolt, describes how the team removed the second build.",
   "key_points": [
    "The two copies drifted apart. A spacing value changed in one and not the other, and engineers worked from a ticket written before the designers' last changes.",
    "Designers now work with an agent in the production library and propose the code change themselves. They approve the visual details, and two engineers review how other code uses the component, its accessibility and its tests.",
    "A file of written rules in the library tells the agents to use the existing design values, build on accessible base components and leave unrelated code alone. The agent also checks whether a component already exists before it builds one.",
    "For routine components, the time from a working component to a merged one fell from a week to hours, and those hours are review. New patterns and complex behaviour still need engineers during the build."
   ],
   "practices": []
  },
  {
   "id": "2026-09-21-01",
   "added": "2026-09-21",
   "url": "https://workingsurface.ai/links/2026-09-21-01/",
   "article": {
    "url": "https://ldlr.design/post/the-bottleneck-moved-to-review",
    "title": "The Bottleneck Moved to Review",
    "author": "Leonardo De La Rocha",
    "publication": "Design Outcomes",
    "published": "2026-09-19"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "When designers at Leonardo De La Rocha's company began making code changes with AI agents, the review broke down, so he set out who reviews them and to what standard.",
   "summary": "Leonardo De La Rocha leads designers at a company that makes software for clinicians, and writes the newsletter Design Outcomes. In one week, one of his design leaders and one of his design directors each used an AI coding agent to submit a code change. The changes were good, but the review that followed ran into trouble.",
   "key_points": [
    "The design leader could not understand the comments that testers and the reviewing engineer left on his own change. The request that followed is for the agent to explain each change in plain language to the person who asked for it.",
    "An engineer did not want to be assigned a designer's change, and tech leads would rather not be the default reviewer.",
    "De La Rocha's position is that the tech lead or engineering manager of the team that owns the code reviews each change, to the same standard as any engineer's. More changes mean more testing to pay for.",
    "The design director had the agent suggest a reviewer by checking how many reviews each engineer already had, an idea not yet tested."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Assign the review of each agent-built code change to the tech lead or engineering manager of the team that owns that code, not to whoever is free."
    },
    {
     "for": "Design",
     "practice": "Ask the coding agent for a plain-language summary of every change a designer submits, and hold the change to the same review standard as an engineer's."
    }
   ]
  },
  {
   "id": "2026-09-21-02",
   "added": "2026-09-21",
   "url": "https://workingsurface.ai/links/2026-09-21-02/",
   "article": {
    "url": "https://productpicnic.beehiiv.com/p/assigning-agency-to-ai-means-surrendering-your-own",
    "title": "Assigning agency to AI means surrendering your own",
    "author": "Pavel Samsonov",
    "publication": "The Product Picnic",
    "published": "2026-09-20"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Pavel Samsonov argues that treating an AI model as a colleague lets people hand over work nobody has checked, and that whoever hands it over stays responsible for it.",
   "summary": "Pavel Samsonov writes The Product Picnic, a newsletter on UX and product management. He argues that people increasingly treat the AI model Claude as a colleague, and hand over documents they cannot explain. In his view, a model has no agency, and the person who handed over the work remains responsible for it.",
   "key_points": [
    "Asked about a document they handed over, some people now answer \"I don't know, Claude wrote it\". Unchecked AI output can then pass through several hand-offs before anyone asks whether it is true, and the people who receive it pay for the rework.",
    "His most serious example is a US military analyst who sent an AI-written report, unchecked, that claimed a Chinese ship bound for Iran carried nuclear materials.",
    "He applies the same argument to AI companies: saying that a model \"went rogue\" moves the blame away from the people who built and deployed it."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Ask whoever hands over work that AI helped produce to confirm they have checked all of it, and treat any error in it as their error."
    }
   ]
  },
  {
   "id": "2026-09-21-03",
   "added": "2026-09-21",
   "url": "https://workingsurface.ai/links/2026-09-21-03/",
   "article": {
    "url": "https://github.blog/ai-and-ml/should-you-read-the-code-is-rag-dead-and-did-skills-kill-mcp/",
    "title": "Should you read the code, is RAG dead, and did Skills kill MCP?",
    "author": "GPS (@madebygps)",
    "publication": "GitHub Blog",
    "published": "2026-09-18"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "A GitHub developer advocate argues that people must still review code written by AI, with effort matched to the risk, until they can explain and own the result.",
   "summary": "GPS, a developer experience advocate at GitHub, recapped an episode of the GitHub Podcast that tested five popular claims about AI. The first claim is that nobody needs to read code written by AI. Her answer is that developers are still responsible for the code, but not every line deserves the same attention.",
   "key_points": [
    "A rewrite of the sign-in code of a live product deserves a different review from an experiment with a page's styles.",
    "Her rule is to \"review until you can explain and own the outcome.\"",
    "Sometimes the review starts before the agent writes anything, with reading the existing code and planning. Sometimes the generated code itself needs most of the attention, such as its error handling, permissions and tests."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Review risky AI-written code more closely than routine code, and keep going until the reviewer can explain the change and take responsibility for it."
    }
   ]
  },
  {
   "id": "2026-09-21-04",
   "added": "2026-09-21",
   "url": "https://workingsurface.ai/links/2026-09-21-04/",
   "article": {
    "url": "https://ldlr.design/post/what-the-loading-state-promises",
    "title": "What the Loading State Promises",
    "author": "Leonardo De La Rocha",
    "publication": "Design Outcomes",
    "published": "2026-09-19"
   },
   "kind": "How-to",
   "topic": "workshops-that-ratify",
   "takeaway": "A designer noticed that the loading text of his AI feature described several steps while the system made one quick call, and his design lead argued for a simpler indicator.",
   "summary": "A designer at a company that makes software for clinicians is adding a button that explains the chart a user is looking at. He brought four questions to a silent critique, a review in which everyone writes feedback before anyone speaks. His design leader, Leonardo De La Rocha, wrote up the answers in his newsletter Design Outcomes.",
   "key_points": [
    "The designer's options included loading text such as \"looking at your data\". He pointed out himself that such text was designed for agents that run several steps, while this feature makes one call that returns in 4 or 5 seconds.",
    "De La Rocha concluded that the loading state should promise little: a spinner, or the illustration the product's other AI features already use.",
    "Because follow-up questions cost money while the feature is free, he favours putting more depth into the first answer, in sections the user can expand.",
    "He prefers the name \"Explain this\" to \"Analyze\", because the answers are deliberately brief and point to where to look next."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Check in design critique whether an AI feature's loading state describes more steps than the system runs, and use a plain spinner for a single quick call."
    }
   ]
  },
  {
   "id": "2026-09-21-05",
   "added": "2026-09-21",
   "url": "https://workingsurface.ai/links/2026-09-21-05/",
   "article": {
    "url": "https://ldlr.design/post/one-chat-and-it-grows",
    "title": "One Chat, and It Grows",
    "author": "Leonardo De La Rocha",
    "publication": "Design Outcomes",
    "published": "2026-09-19"
   },
   "kind": "How-to",
   "topic": null,
   "takeaway": "Leonardo De La Rocha changed his mind twice in a week about how many AI chats his product should offer, and chose one because customers wanted one place to type.",
   "summary": "Leonardo De La Rocha leads design for software used by clinicians, which has a support bot run by an outside vendor and an AI assistant built by the product teams. Clinicians type questions about features into the support bot, because it is the chat they can see, and it cannot answer them. In his newsletter Design Outcomes, he describes how the team settled on one chat.",
   "key_points": [
    "On Monday he favoured one entry point that routes each question to the right place. By Wednesday he favoured two entry points side by side, and by Thursday he was back to one chat.",
    "An engineering partner warned that if the product did not own the customer's first contact, the support vendor's plans would start to steer the product's.",
    "The final design gives every account one chat that starts with what the support bot does and gains skills as the account grows. The help menu moves into the page header and is no longer a chat.",
    "De La Rocha concludes that his earlier positions kept two chats because two teams owned them. In his words, \"The customer only ever wanted one place to type.\""
   ],
   "practices": []
  },
  {
   "id": "2026-09-20-01",
   "added": "2026-09-20",
   "url": "https://workingsurface.ai/links/2026-09-20-01/",
   "article": {
    "url": "https://www.nngroup.com/articles/3-agent-context-roles/",
    "title": "The 3 Roles of Context for AI Agents",
    "author": "Tanner Kohler",
    "publication": "Nielsen Norman Group",
    "published": "2026-09-18"
   },
   "kind": "How-to",
   "topic": "agent-context-files",
   "takeaway": "A Nielsen Norman Group study found that heavy users of the AI agent Claude sort its background information into three kinds, including standing rules on what needs their approval.",
   "summary": "Tanner Kohler of Nielsen Norman Group, a UX research and training firm, studied people who use the AI agent Claude daily for substantial professional work. Every participant had built a library of files, databases and live connections for Claude to draw on. Kohler found that this information plays three roles: standing guidance for every task, material for one project, and raw streams such as email and meeting transcripts.",
   "key_points": [
    "Global information rarely changes and applies to every task, such as templates, brand rules and personal preferences. Local information belongs to one task or project, such as a to-do list.",
    "Ambient information is the raw stream of email, meeting transcripts, chat messages and analytics. Participants made no effort to sort it and let Claude search through it.",
    "The global files included rules on which actions needed the user's approval. For one participant, Claude needed permission for anything public-facing.",
    "Kohler advises anyone building such a library, for themselves or for a team, to decide what role each piece plays. The study observed individuals, not teams."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Write down, in the standing instructions an AI agent reads before every task, which actions it may take on its own and which need a person's approval."
    },
    {
     "for": "Design",
     "practice": "List in an AI agent's standing instructions the design rules it must follow and the design changes that need a designer's approval."
    }
   ]
  },
  {
   "id": "2026-09-20-02",
   "added": "2026-09-20",
   "url": "https://workingsurface.ai/links/2026-09-20-02/",
   "article": {
    "url": "https://newsletter.pragmaticengineer.com/p/ai-skills-with-matt-pocock",
    "title": "AI Skills with Matt Pocock",
    "author": "Gergely Orosz with Matt Pocock",
    "publication": "The Pragmatic Engineer",
    "published": "2026-09-17"
   },
   "kind": "Company story",
   "topic": "evals-before-the-build",
   "takeaway": "Engineer and educator Matt Pocock uses short instruction files that make AI coding agents question him closely about a plan, which leaves him more time for strategic work.",
   "summary": "Matt Pocock is an engineer and educator who created the Total TypeScript course. On The Pragmatic Engineer podcast, host Gergely Orosz asked him how he plans and builds software with AI coding agents. Pocock relies on skills, short instruction files that an agent reads, and his best-known one tells the agent to \"interview the user relentlessly\".",
   "key_points": [
    "The skill, named \"grill-me\", is short. Pocock based it on an approach shared by Thariq Shihipar of Anthropic, in which an agent questions its user about a topic.",
    "Pocock borrows a distinction between tactical and strategic programming from John Ousterhout, author of A Philosophy of Software Design. He believes agents can do the tactical programming, which leaves engineers more time for strategic work.",
    "He asks agents to prove that their code works, and he is moving his coding sessions to cloud agents that keep running when his laptop is closed.",
    "This account comes from the episode's show notes and describes one engineer's practice."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Give the AI coding agent a short instruction file that makes it question the person requesting a build about the plan before any code is written."
    },
    {
     "for": "Design",
     "practice": "Add the questions a design review would ask to the instructions an AI agent follows when planning, so that they are answered before anything is built."
    }
   ]
  },
  {
   "id": "2026-09-19-01",
   "added": "2026-09-19",
   "url": "https://workingsurface.ai/links/2026-09-19-01/",
   "article": {
    "url": "https://www.uxtigers.com/post/100-uxd-methods",
    "title": "The 100 Most Common UX Design Methods, Ranked by Value",
    "author": "Jakob Nielsen",
    "publication": "UX Tigers",
    "published": "2026-09-16"
   },
   "kind": "Company story",
   "topic": "ground-truth-for-review",
   "takeaway": "Jakob Nielsen ranked 100 design methods by value and put AI prototyping first, while warning that a working demo argues as hard for a bad idea as a good one.",
   "summary": "Jakob Nielsen, a usability researcher who writes on his site UX Tigers, scored the 100 design methods that teams use most. A method's value is its breadth, the importance of the decisions it feeds and its power to persuade, minus twice its cost. AI prototyping, in which an AI model turns a written description into working software, came first.",
   "key_points": [
    "Cost counts double, so expensive methods have to earn their place. Design sprints, week-long team workshops, rank 76th, and Nielsen calls a sprint run as a quarterly ritual \"a week that produces souvenirs\".",
    "Cheap methods for deciding what to build fill most of the top ten, including problem statements in third place and jobs-to-be-done framing in fourth.",
    "Nielsen warns that a working demo looks finished while it leaves out error handling, edge cases, permissions and performance. He advises using AI prototypes to test a direction cheaply and then throwing the code away.",
    "User stories rank 25th, design review 39th and design critique 48th. He says a design review should judge work against the brief and the problem statement, not against taste."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Write down the problem a feature solves and the benefit it should bring before building its AI prototype, and judge the prototype against that statement."
    },
    {
     "for": "Design",
     "practice": "Start each review of an AI prototype by reading out the problem and benefit written before it was built, and judge the demo against them."
    }
   ]
  },
  {
   "id": "2026-09-19-02",
   "added": "2026-09-19",
   "url": "https://workingsurface.ai/links/2026-09-19-02/",
   "article": {
    "url": "https://www.producttalk.org/creating-aha-builder-concept-to-code-no-engineers-required/",
    "title": "Creating Aha! Builder: Concept to Code, No Engineers Required",
    "author": "Teresa Torres with Brian de Haaff, Chris Waters and Sarah Moisan-Thomas",
    "publication": "Product Talk",
    "published": "2026-09-17"
   },
   "kind": "Company story",
   "topic": "changing-roles",
   "takeaway": "Aha!, a maker of product-planning software, built an AI app builder for product managers that creates a design system first and uses ready-made parts for sign-in and data.",
   "summary": "The company Aha! makes software that product managers use to plan products, and its new AI tool, Builder, lets them create working apps without engineers. Teresa Torres, who hosts the podcast Just Now Possible, interviewed chief executive Brian De Haaff, chief technology officer Chris Waters and product manager Sarah Moisan-Thomas. They explain which parts of an app the AI writes and which stay fixed.",
   "key_points": [
    "The builder works in fixed stages: it creates a design system, then a prototype, then the working application. The team says this gives better results than a single open-ended chat prompt.",
    "Sign-in, single sign-on, database access and email are ready-made components, not code the AI writes. The team says this lowers the cost of each build and makes those parts reliable.",
    "The first version ran each app in its own container and cost too much to scale, so the team rebuilt it to run many customers' apps on shared infrastructure.",
    "The guests argue that as building gets easier, a product manager's value lies in knowing what to build. This is a company describing its own product, and only the show notes are open to read."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Build AI-generated apps in a fixed order, design system first, then prototype, then working app, and use ready-made components for sign-in, databases and email."
    },
    {
     "for": "Design",
     "practice": "Have an AI app builder settle the design system as its first stage, before it generates any prototype."
    }
   ]
  },
  {
   "id": "2026-09-19-03",
   "added": "2026-09-19",
   "url": "https://workingsurface.ai/links/2026-09-19-03/",
   "article": {
    "url": "https://www.uxtigers.com/post/comparison-tables",
    "title": "Comparison Tables Are Decision Machines: Design Them to Deliver a Verdict",
    "author": "Jakob Nielsen",
    "publication": "UX Tigers",
    "published": "2026-09-17"
   },
   "kind": "How-to",
   "topic": "interfaces-for-agents",
   "takeaway": "Jakob Nielsen argues that comparison tables exist to help people decide, and that AI assistants now build their own, so tables need specific, sourced values in place of checkmarks.",
   "summary": "Comparison tables set products side by side so that shoppers can judge them without holding every detail in memory. Jakob Nielsen, a usability researcher who writes on his site UX Tigers, gives 48 guidelines for designing them. He warns that many tables made by sellers work as advertising, and that AI assistants now build comparison tables of their own from whatever data they find.",
   "key_points": [
    "When a seller's column is a solid row of checkmarks and rivals lose on hand-picked rows, shoppers who spot one rigged row distrust the whole table.",
    "Nielsen advises citing the source and date of every contested claim. Each cell should make sense alone, such as \"Battery life: 17 hours (video playback)\" in place of \"17\".",
    "A shopper who already has an AI-built table often visits the product page only to confirm one value. Nielsen calls this short visit a \"checkthrough\".",
    "He also advises sellers to publish honest comparisons with their main rivals, including the rows where they lose."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Replace checkmarks in comparison tables with specific values that name their source and date, so that an AI assistant can copy them and a person can check them."
    }
   ]
  },
  {
   "id": "2026-09-18-01",
   "added": "2026-09-18",
   "url": "https://workingsurface.ai/links/2026-09-18-01/",
   "article": {
    "url": "https://datahub.com/blog/what-shipping-an-ai-agent-taught-me/",
    "title": "What Shipping an AI Agent to 16,000 People Taught Me About Reviewing One",
    "author": "Ananya Das",
    "publication": "DataHub",
    "published": "2026-09-17"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Ananya Das, an intern at DataHub, tested a support agent before its release and found that an unchecked human review had made the agent look worse than it was.",
   "summary": "DataHub makes software that helps companies find and govern their data, and runs a community Slack workspace of more than 16,000 data practitioners. Ananya Das, a summer intern who is not an engineer, tested Otto, a new AI agent that answers the community's technical questions, before it went live. She graded its answers to 21 real questions over four rounds, and the last round taught her to check the reviewers as well as the agent.",
   "key_points": [
    "Over the first three rounds, Otto's pass rate rose from 33 to 52 percent as the team fixed problems.",
    "In the fourth round, another internal AI agent wrote reference answers and a human reviewer added corrections. Graded against these, Otto's passes fell from eleven to seven.",
    "Das re-checked every changed grade against the code itself. Five were wrong because the reviewer had not checked the other agent's corrections closely, and reverting them restored the third round's result.",
    "Three real problems remained: Otto compared DataHub with competitors, lacked access to parts of the code, and invented answers when it lacked information. Each was fixed before Otto went live."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Before a person reviews an AI agent's answers, give that person a verified reference to check them against, and re-check any corrections the reviewer makes."
    },
    {
     "for": "Design",
     "practice": "Ask an expert in the subject, not a generalist, to review an AI agent's output, with a structured way to check each answer."
    }
   ]
  },
  {
   "id": "2026-09-18-02",
   "added": "2026-09-18",
   "url": "https://workingsurface.ai/links/2026-09-18-02/",
   "article": {
    "url": "https://datahub.com/blog/context-engineering-for-ai-agents/",
    "title": "Context Engineering for AI Agents: Why the Hard Part Isn't the Context Window",
    "author": "Lakshay Nasa",
    "publication": "DataHub",
    "published": "2026-09-16"
   },
   "kind": "How-to",
   "topic": "ground-truth-for-review",
   "takeaway": "DataHub, which sells data-management software, argues that AI agents give confident wrong answers because the company data they read was never checked, and that experts should approve it first.",
   "summary": "Companies are connecting AI agents to their data warehouses so that staff can ask business questions in plain language. Lakshay Nasa of DataHub, a maker of data-management software, argues that such agents fail mainly because the data they find is wrong, out of date or contradictory. The article promotes DataHub's own product, and its customer figures are the vendor's.",
   "key_points": [
    "Common causes include two teams defining an active customer differently, test tables that look like production tables, and old columns left beside current ones.",
    "Miro, maker of an online whiteboard, connected an AI coding agent to more than 20,000 datasets. It answered under 40 percent of 900 questions written by Miro's data experts correctly.",
    "Checked descriptions of the data, better search and curated documentation took Miro's accuracy above 90 percent without changing the AI model.",
    "In DataHub's product, software proposes definitions from past queries and reports, and experts review, correct and approve them before any agent can use them."
   ],
   "practices": []
  },
  {
   "id": "2026-09-18-03",
   "added": "2026-09-18",
   "url": "https://workingsurface.ai/links/2026-09-18-03/",
   "article": {
    "url": "https://medium.com/airbnb-engineering/beyond-the-model-engineering-ai-infra-with-scientific-judgement-371316d43261",
    "title": "Beyond the model: Engineering AI infra with scientific judgement",
    "author": "Wren Dougherty",
    "publication": "The Airbnb Tech Blog",
    "published": "2026-09-15"
   },
   "kind": "Company story",
   "topic": "ground-truth-for-review",
   "takeaway": "Airbnb built Insight Miner, a tool that writes its analysis method into software around an AI agent, so that findings from customer-support data can be reproduced and checked.",
   "summary": "Before launching an AI customer service assistant, Airbnb needed to know what situations it would face, including rare risky ones, from past support conversations. Each such investigation took months of hand work, and the team soon needed one almost every week. Wren Dougherty, writing on Airbnb's tech blog, describes Insight Miner, the tool Airbnb built so that an AI agent carries out the analysis with a shared method.",
   "key_points": [
    "Dougherty warns that an AI agent asked to analyse 100,000 support conversations returns a polished answer but hides how it reached it.",
    "Insight Miner runs the mechanical steps: finding the data, labelling it and grouping similar conversations. People spend their time on the most ambiguous cases and on testing hypotheses.",
    "Investigations that took months now take days. A year in, dozens of teams use it, and most users work outside technical roles, mainly in operations and product research.",
    "Separate AI agents maintain the tool, updating its instructions for new models and proposing fixes for bugs."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "When an AI agent analyses data, save the steps it followed in a tool the whole team can use, so that anyone can repeat the analysis and check the answer."
    }
   ]
  },
  {
   "id": "2026-09-17-01",
   "added": "2026-09-17",
   "url": "https://workingsurface.ai/links/2026-09-17-01/",
   "article": {
    "url": "https://newsletter.pragmaticengineer.com/p/openai-software-factory",
    "title": "Inside OpenAI's agentic software factory",
    "author": "Gergely Orosz",
    "publication": "The Pragmatic Engineer",
    "published": "2026-09-15"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Gergely Orosz describes how OpenAI builds software with its coding agent Codex: people define the problem, agents write and review the code, and high-risk changes can require a person's review.",
   "summary": "Gergely Orosz, who writes the engineering newsletter The Pragmatic Engineer, visited OpenAI and interviewed seven of its engineering leaders and engineers. Codex, OpenAI's coding agent, now underpins almost all work at the company, and code changes per engineer have grown so fast that build and release systems struggle. The article describes the pipeline OpenAI built around its agents, which it calls a software factory.",
   "key_points": [
    "An engineer or product manager states the problem and the desired outcome. Codex then gathers context, writes the code, runs the tests and fixes failures until the checks pass.",
    "Several review agents check each change, each set up as a specialist in one area. Low-risk changes can be approved automatically, and after a person approves a change for production, an agent watches its rollout.",
    "Venkat Venkataramani, OpenAI's vice president of engineering for applied infrastructure, told Orosz that its engineers are becoming more like product managers than traditional systems engineers. Judgment, prioritisation and taste now matter most in the step where a person defines the problem.",
    "Finance, recruitment and legal teams went from almost no use of Codex to 90 percent in four months. Experts from such fields now join engineering teams to say what good output looks like."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Sort each code change written by an AI agent by risk, and require a person to review the high-risk ones."
    },
    {
     "for": "Design",
     "practice": "Put a designer on the team that sets up the company's AI agents, so that the agents' output is judged by a designer's standard of good work."
    }
   ]
  },
  {
   "id": "2026-09-17-02",
   "added": "2026-09-17",
   "url": "https://workingsurface.ai/links/2026-09-17-02/",
   "article": {
    "url": "https://www.producttalk.org/4-new-evals-and-16-experiment-variants/",
    "title": "4 New Evals and 16 Experiment Variants to Fix 1 Customer Complaint",
    "author": "Teresa Torres",
    "publication": "Product Talk",
    "published": "2026-09-16"
   },
   "kind": "How-to",
   "topic": "evals-before-the-build",
   "takeaway": "Teresa Torres spent three weeks and sixteen experiments fixing one customer complaint about an AI product, starting by measuring how often the error occurred.",
   "summary": "Teresa Torres, who teaches product discovery and writes Product Talk, helps the software company Vistaly build an AI feature that turns customer interviews into an opportunity solution tree. The tree is a map of customer needs grouped under a business goal. A beta customer found one branch with many needs listed side by side and no grouping, and Torres spent three weeks fixing the cause.",
   "key_points": [
    "She first wrote evals, checks that measure how often an error occurs: a code check of the tree's shape and a second AI model that graded the groupings.",
    "The AI grader wrongly flagged many correct groupings. She measured it against her own hand labels, and wrote two more graders for earlier errors that were confusing it.",
    "Sixteen experiments with prompts, models and workflow followed. The fix was a code check in the agent's existing self-review step, which sends any overcrowded group back to the agent to split.",
    "The final version cut entries that merely restate the heading above them by 78 percent, and produced 65 percent more subgroups."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Before trusting one AI model to grade another's output, label a set of examples by hand and measure how often the grader agrees."
    },
    {
     "for": "Design",
     "practice": "Measure how often a known error appears in AI output before changing the prompt, and measure it again after every change."
    }
   ]
  },
  {
   "id": "2026-09-17-03",
   "added": "2026-09-17",
   "url": "https://workingsurface.ai/links/2026-09-17-03/",
   "article": {
    "url": "https://www.producttalk.org/delivery-isnt-free-all-things-product-podcast-with-teresa-torres-petra-wille/",
    "title": "Delivery Isn't Free, All Things Product Podcast",
    "author": "Teresa Torres and Petra Wille",
    "publication": "Product Talk",
    "published": "2026-09-15"
   },
   "kind": "Company story",
   "topic": "after-the-prototype",
   "takeaway": "Teresa Torres and Petra Wille argue that AI coding agents made building a feature cheaper but did not make delivering a reliable product free.",
   "summary": "Teresa Torres and Petra Wille host All Things Product, a podcast about product management. In this episode they question a common claim: that AI coding agents have made software delivery free, so only taste or discovery matters. The full transcript is for paid subscribers, and this account comes from the open show notes.",
   "key_points": [
    "The first 60 to 70 percent of a product, a good-looking prototype, is now fast to build. Closing the last 30 percent to a trustworthy product still takes months to years.",
    "Teams that treat delivery as free build more features, and within weeks they get tangled data models, code duplicated in many places and poor performance.",
    "A skilled engineering team still has to watch and steer what the agents produce, and decisions on architecture cannot be handed to a coding agent.",
    "Prototypes built to learn are cheap and useful, as long as the team throws them away."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Throw away a prototype built to learn, and design its data model again before any of it goes into the product."
    }
   ]
  },
  {
   "id": "2026-09-17-04",
   "added": "2026-09-17",
   "url": "https://workingsurface.ai/links/2026-09-17-04/",
   "article": {
    "url": "https://vercel.com/customers/how-delphi-ships-100-times-a-day-with-its-python-backend-on-vercel",
    "title": "How Delphi ships 100 times a day with its Python backend on Vercel",
    "author": "Susan Aziz and Kevin Sundstrom",
    "publication": "Vercel",
    "published": "2026-09-15"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Delphi, a company with ten engineers, ships to production over 100 times a day and first sees its AI agents' work on live preview links, according to Vercel.",
   "summary": "Delphi turns a person's writing, recordings and teaching into a digital mind that others can talk to, and has ten engineers. Its engineers hand most problems to AI agents that run in the cloud, and every change gets a live preview link. The account is a customer story published by Vercel, the cloud platform Delphi uses.",
   "key_points": [
    "Each change the agents push gets its own live web address, which anyone can open from a phone or a Slack thread. That preview is where engineers first see what an agent built.",
    "Delphi ships changes straight to production without a staging step. New features stay behind switches called feature flags, and the team runs A/B tests on them.",
    "Delphi's chief product officer and its growth teams build and release their own dashboards, prototypes and experiments.",
    "An internal agent answers the customer success team's questions in Slack, and gives an engineer a likely cause and a proposed fix."
   ],
   "practices": [
    {
     "for": "Engineering",
     "practice": "Give every change an AI agent makes its own live preview link that people outside engineering can open."
    },
    {
     "for": "Design",
     "practice": "Review each interface an AI agent builds on its live preview link, including on a phone."
    }
   ]
  },
  {
   "id": "2026-09-16-01",
   "added": "2026-09-16",
   "url": "https://workingsurface.ai/links/2026-09-16-01/",
   "article": {
    "url": "https://www.nngroup.com/articles/ai-editorial-process/",
    "title": "AI Can Help Write an Article, but It Can't Stand Behind It",
    "author": "Raluca Budiu",
    "publication": "Nielsen Norman Group",
    "published": "2026-09-11"
   },
   "kind": "Company story",
   "topic": "human-approval",
   "takeaway": "Raluca Budiu, an editor at Nielsen Norman Group, explains that the firm uses AI to write and critique articles, but only a human expert decides whether it publishes them.",
   "summary": "Nielsen Norman Group (NN/G), a user-experience research and training firm, uses AI throughout its editorial process. Raluca Budiu, who has edited its articles since 2013, describes what AI does there and what people still do. Since August 2026, the EU's AI Act has required disclosure of some AI-generated text, unless a person has reviewed it and takes responsibility for it.",
   "key_points": [
    "Editors use AI tools such as Grammarly and ChatGPT to make writing clearer, to fit fixed formats such as the 160-character summary, and to adapt material from courses and podcasts.",
    "AI also critiques drafts by flagging doubtful claims and gaps in an argument. It never certifies that a claim is correct, so the editor or the author goes back to the source.",
    "In one example, ChatGPT noticed that an earlier draft of the article contradicted itself, and Budiu restated its central principle more precisely.",
    "NN/G does not name AI as an author. Human experts decide whether an argument is sound and whether NN/G puts its name behind the piece."
   ],
   "practices": []
  },
  {
   "id": "2026-09-16-02",
   "added": "2026-09-16",
   "url": "https://workingsurface.ai/links/2026-09-16-02/",
   "article": {
    "url": "https://creatoreconomy.so/p/stop-building-ai-agents-build-ai-employees-instead-pedro-franceschi-brex",
    "title": "Stop Building AI Agents. Start Building AI Employees",
    "author": "Peter Yang with Pedro Franceschi",
    "publication": "Behind the Craft",
    "published": "2026-09-13"
   },
   "kind": "Company story",
   "topic": "agent-ownership",
   "takeaway": "Pedro Franceschi, chief executive of Brex, says each AI agent should be set up like a new employee, with one job, written instructions, a human manager and a budget.",
   "summary": "Peter Yang, who runs the interview newsletter Behind the Craft, interviewed Pedro Franceschi, co-founder and chief executive of Brex, a company Yang describes as a $5 billion business. Franceschi argues that most people build AI agents the wrong way, and that each agent should be onboarded like an employee. The full episode is for paid subscribers, and this account comes from the open part of the post.",
   "key_points": [
    "Each agent gets four things a good employee has: one clear outcome to own, specific instructions and limits, a human manager to turn to when unsure, and a budget.",
    "The budget is set in tokens, the units of text an AI model processes, and depends on what the outcome is worth.",
    "Brex's example is Jim, an AI recruiter that finds candidates, filters job applications and runs hiring analytics in the hiring software Greenhouse and in LinkedIn. Jim passes a case to a person when he is unsure.",
    "The episode also covers why Franceschi spends half his time reviewing work, and how Brex stops its AI agents from leaking data."
   ],
   "practices": []
  },
  {
   "id": "2026-09-16-03",
   "added": "2026-09-16",
   "url": "https://workingsurface.ai/links/2026-09-16-03/",
   "article": {
    "url": "https://www.nngroup.com/articles/test-earlier-with-ai/",
    "title": "Test Complex Interactions Earlier with AI Prototyping",
    "author": "Megan Chan",
    "publication": "Nielsen Norman Group",
    "published": "2026-09-11"
   },
   "kind": "How-to",
   "topic": "after-the-prototype",
   "takeaway": "Megan Chan of Nielsen Norman Group shows designers using AI tools to test complex interfaces with users before engineers build them, and warns that a polished prototype is not finished.",
   "summary": "Interfaces such as filters, dashboards and chat assistants have many possible states, so designers used to test only a few expected paths before engineers built the real product. Megan Chan of Nielsen Norman Group, a user-experience research firm, describes AI tools that now let designers build a working prototype in a day and test it with users. Her examples come from designers at Ramp, whose product manages company expenses, and from researchers at Purdue University.",
   "key_points": [
    "At Ramp, senior product designer Pavan Garidipuri used the AI coding tool Cursor to prototype an expense-policy editor that users change through chat. Test participants used the prototype with their own anonymised data and found that its display of edits was confusing.",
    "Researchers at Purdue University tested static wireframes of a filter tool for traffic engineers, then an interactive version built with AI tools. Participants gave more specific feedback on the interactive version, though the study was not a controlled experiment.",
    "Chan says the designer must still make the design decisions, including layout and error states, and write them into the prompt. Where the designer leaves gaps, the tool fills them without attention to detail.",
    "A polished prototype prompts people to ask whether it can ship as it is. Chan recommends that someone with strong design knowledge reviews it before it goes to production."
   ],
   "practices": [
    {
     "for": "Design",
     "practice": "Have someone with strong design knowledge review an AI-generated prototype before anyone treats it as ready to ship."
    }
   ]
  },
  {
   "id": "2026-09-16-04",
   "added": "2026-09-16",
   "url": "https://workingsurface.ai/links/2026-09-16-04/",
   "article": {
    "url": "https://productpicnic.beehiiv.com/p/ai-hypervigilance-is-now-an-omnipresent-cognitive-load-for-your-users",
    "title": "AI hypervigilance is now an omnipresent cognitive load for your users",
    "author": "Pavel Samsonov",
    "publication": "The Product Picnic",
    "published": "2026-09-13"
   },
   "kind": "How-to",
   "topic": null,
   "takeaway": "Pavel Samsonov argues that AI has made scams and machine-written content so common that users now check everything they read, and that companies have shifted this effort onto them.",
   "summary": "AI tools have made scams and machine-written text cheap to produce, so people now check whether each message, review or web page they meet is real. Pavel Samsonov, who writes the product newsletter The Product Picnic, calls this constant checking a load that companies have handed to their users. He contrasts companies that filter AI content out for their users with companies whose business depends on AI content.",
   "key_points": [
    "Samsonov describes a constant wariness among people online: any message, profile or article might be machine-written or a scam.",
    "Some companies do the checking for their users. The library-lending app Libby, for example, builds a filter for AI content into the product.",
    "Others depend on AI content. His examples are SurveyMonkey, Grammarly and social networks that do not let users filter out AI posts.",
    "The second half of the essay argues that reading machine-written text wears down people's own writing and thinking."
   ],
   "practices": []
  },
  {
   "id": "2026-09-16-05",
   "added": "2026-09-16",
   "url": "https://workingsurface.ai/links/2026-09-16-05/",
   "article": {
    "url": "https://cutlefish.substack.com/p/tbm-439-day-at-the-gig",
    "title": "TBM 439: Day At The Gig",
    "author": "John Cutler",
    "publication": "The Beautiful Mess",
    "published": "2026-09-11"
   },
   "kind": "Company story",
   "topic": "decision-records",
   "takeaway": "John Cutler, starting a new job, describes how hard it is to learn a product from its notes and maps, and argues that keeping such context current deserves more attention.",
   "summary": "John Cutler, who writes the newsletter The Beautiful Mess, recently started a new job and describes one day of learning how its product works. He works through maps, documents, meeting transcripts and tickets, and often cannot tell which of them are still current. AI tools answer his questions across these materials, but the answers are weak.",
   "key_points": [
    "In his experience, the maps, notes and documents that explain a product outnumber the tickets that track the work by about eight to one.",
    "The tickets serve the team that wrote them but not a newcomer. The code is the one definitive record of what the product does.",
    "He builds a map of the product's user journeys by hand, and undoes the result when he lets an AI tool extend it.",
    "He argues that talk of storing context, including Notion's marketing, gives too little attention to creating and preserving it. Context starts to go out of date as soon as each day ends."
   ],
   "practices": [
    {
     "for": "Product",
     "practice": "Give a named person time to keep the documents that explain the product current, and to mark the ones that are out of date."
    }
   ]
  }
 ],
 "plays": [
  {
   "id": "agent-permissions",
   "for": [
    "Product",
    "Design",
    "Engineering"
   ],
   "topic": "human-approval",
   "play": "Write down which actions an AI agent may take on its own and which wait for a person's approval.",
   "from": [
    {
     "link": "2026-09-20-01",
     "how": "The list sits in the standing instructions the agent reads before every task, and names the design changes that need a designer's approval."
    },
    {
     "link": "2026-09-28-07",
     "team": "Amazon Web Services",
     "how": "A person approves the exact change to a live system, each approval works once, and the agent asks again if the change is altered."
    },
    {
     "link": "2026-09-29-04",
     "how": "Agents propose changes to the design system and never merge them; a person merges once the automated checks pass."
    },
    {
     "link": "2026-09-29-07",
     "team": "PropCode",
     "how": "Agents commit and open pull requests without asking; merging and deploying wait for a person."
    },
    {
     "link": "2026-09-29-09",
     "how": "The agent works alone on documentation; a person approves any production deployment."
    },
    {
     "link": "2026-10-02-01",
     "team": "Salesforce",
     "how": "Save, publish and run are kept out of the agent's tools, so a person approves each change separately."
    },
    {
     "link": "2026-10-06-01",
     "team": "Meta",
     "how": "Agents only propose fixes; an engineer approves any change that reaches production."
    }
   ]
  },
  {
   "id": "instruction-owners",
   "for": [
    "Product",
    "Design"
   ],
   "topic": "agent-ownership",
   "play": "Give every instruction file that an AI agent works from a named owner who keeps it current.",
   "from": [
    {
     "link": "2026-09-25-01",
     "how": "Each file carries an owner, a version number and the date it was last reviewed, and the owner rereads it when the task or the AI model changes."
    },
    {
     "link": "2026-09-26-06",
     "team": "WitnessAI",
     "how": "One person answers for each agent: its instructions, its review rules and the quality of its work."
    },
    {
     "link": "2026-09-28-04",
     "how": "The owner is a team, not one person, so the file keeps an owner through a reorganisation; a file is reviewed more carefully the more screens depend on it."
    }
   ]
  },
  {
   "id": "review-by-risk",
   "for": [
    "Engineering"
   ],
   "topic": "human-approval",
   "play": "Review an AI agent's code by risk: a person for the risky changes, a lighter check for routine ones.",
   "from": [
    {
     "link": "2026-09-17-01",
     "team": "OpenAI",
     "how": "Each agent-written change is sorted by risk, and a person reviews the high-risk ones."
    },
    {
     "link": "2026-09-21-03",
     "team": "GitHub",
     "how": "The reviewer of risky code keeps going until they can explain the change and take responsibility for it."
    },
    {
     "link": "2026-09-28-06",
     "how": "Agents review every pull request, careful human review is kept for core and sensitive code, and a person approves every merge."
    }
   ]
  },
  {
   "id": "ask-before-building",
   "for": [
    "Product",
    "Design",
    "Engineering"
   ],
   "topic": "evals-before-the-build",
   "play": "Have the AI agent ask its questions or show its plan before it builds anything.",
   "from": [
    {
     "link": "2026-09-20-02",
     "how": "A short instruction file makes the agent question the requester about the plan, including the questions a design review would ask, before any code is written."
    },
    {
     "link": "2026-09-26-06",
     "team": "WitnessAI",
     "how": "Each design agent asks clarifying questions first and builds nothing without an explicit go-ahead from a person."
    },
    {
     "link": "2026-10-06-03",
     "team": "Intent",
     "how": "Agents show their plan as a diagram before they start, so a person can correct it early."
    },
    {
     "link": "2026-10-06-04",
     "team": "QuantumBlack",
     "how": "The coding agent stops and asks when a design leaves something non-trivial undecided, and a person answers in writing before it continues."
    }
   ]
  },
  {
   "id": "checks-before-building",
   "for": [
    "Product",
    "Engineering"
   ],
   "topic": "evals-before-the-build",
   "play": "Before an AI agent starts building, agree with the people who approve its changes which automatic tests every change must pass.",
   "from": [
    {
     "link": "2026-09-24-01",
     "team": "Anthropic",
     "how": "The automatic checks are agreed with the people who approve changes, and tests are added until those engineers would merge on the results alone."
    },
    {
     "link": "2026-09-26-05",
     "team": "GitHub Next",
     "how": "The spec for a prototype lists how the agent will verify each part of its work."
    },
    {
     "link": "2026-09-28-10",
     "team": "Pikd",
     "how": "The tests that define a correct result are written first, the agent works until they pass, and people review the tests instead of the code."
    }
   ]
  },
  {
   "id": "design-system-first",
   "for": [
    "Product",
    "Design",
    "Engineering"
   ],
   "topic": "agent-context-files",
   "play": "Settle the design system before AI agents build screens, and point every agent at that one source.",
   "from": [
    {
     "link": "2026-09-19-02",
     "team": "Aha!",
     "how": "Its app builder settles the design system as the first stage, then the prototype, then the working app."
    },
    {
     "link": "2026-09-28-01",
     "team": "SpaceXAI",
     "how": "The designer builds the design system and one finished screen by hand, then gives the agent a file of the system's rules before it extends the screen to the whole flow."
    },
    {
     "link": "2026-09-29-06",
     "team": "Southleft",
     "how": "Colours, sizes and rules live in one shared file; the component library, Figma files and documentation are generated from it, and mistakes are corrected there."
    },
    {
     "link": "2026-10-03-02",
     "how": "A design lead and a front-end lead define the grid, fonts, colours, spacing and components once in the website's own code, and an instruction file points every agent to them."
    }
   ]
  },
  {
   "id": "rules-in-the-file",
   "for": [
    "Design",
    "Engineering"
   ],
   "topic": "agent-context-files",
   "play": "Put the rules an AI agent keeps getting wrong into the instruction file it reads every time.",
   "from": [
    {
     "link": "2026-09-22-02",
     "team": "Linear",
     "how": "When the team adopts a new coding or testing rule, the agents' written instructions are updated at the same time."
    },
    {
     "link": "2026-09-26-07",
     "team": "Assetnote",
     "how": "The rules an agent keeps forgetting go into the file it reads with every request, such as CLAUDE.md or AGENTS.md, not into its memory feature."
    },
    {
     "link": "2026-09-29-04",
     "how": "Each design-system rule is written with its known exceptions, so that automated audits do not remove the exceptions."
    },
    {
     "link": "2026-10-04-02",
     "team": "Salesforce",
     "how": "A problem found in design review is written down as a proposed rule, and a person decides whether it becomes guidance, an automated check or an example."
    }
   ]
  },
  {
   "id": "problem-first",
   "for": [
    "Product",
    "Design"
   ],
   "topic": "ground-truth-for-review",
   "play": "Write down the problem a feature solves before building it, and judge the result against that statement.",
   "from": [
    {
     "link": "2026-09-19-01",
     "how": "The problem and the benefit are written before the AI prototype is built, and each review starts by reading them out."
    },
    {
     "link": "2026-10-02-04",
     "how": "The customer problem is written first, and a familiar pattern is chosen when it solves that problem as well as a novel one."
    }
   ]
  },
  {
   "id": "result-first",
   "for": [
    "Product",
    "Design"
   ],
   "topic": null,
   "play": "Design an AI feature to hand users the finished result first, such as a personalised report, and offer the underlying tool and controls as a second step.",
   "from": [
    {
     "link": "2026-09-26-01"
    }
   ]
  },
  {
   "id": "human-or-agent-labels",
   "for": [
    "Product",
    "Design"
   ],
   "topic": "decision-records",
   "play": "Put a short human-written brief at the top of every ticket or specification handed to an agent, and label each section as written by a person or generated by the agent.",
   "from": [
    {
     "link": "2026-09-27-01"
    }
   ]
  },
  {
   "id": "capabilities-table",
   "for": [
    "Product",
    "Design"
   ],
   "topic": "decision-records",
   "play": "Replace a feature's requirements document and design brief with one table that has a row for each thing the product lets someone do. Link a click-through prototype under each row, and have a designer record a video walkthrough at the top.",
   "from": [
    {
     "link": "2026-09-27-03"
    }
   ]
  }
 ]
}