{
  "schemaVersion": "1.0",
  "id": "ai-engineer-sydney-2026",
  "name": "AI Engineer Sydney 2026",
  "description": "AI Engineer Sydney is a two-day conference for people building and leading production AI systems, with an additional workshop day and an online streaming option.",
  "url": "https://webdirections.org/ai-engineer/",
  "canonicalPath": "/ai-engineer/",
  "image": "https://webdirections.org/ai-engineer/images/aie-syd-social-card.png",
  "dates": {
    "start": "2026-12-07",
    "end": "2026-12-08",
    "timezone": "Australia/Sydney"
  },
  "attendanceMode": "mixed",
  "status": "scheduled",
  "venue": {
    "name": "Hilton Sydney",
    "sourceLabel": "Hilton Hotel, Sydney, Australia",
    "streetAddress": "488 George Street",
    "locality": "Sydney",
    "region": "NSW",
    "postalCode": "2000",
    "country": "AU"
  },
  "organiser": {
    "name": "Web Directions",
    "url": "https://webdirections.org/",
    "email": "info@webdirections.org"
  },
  "audience": [
    "AI engineers and machine learning engineers",
    "Software engineers using AI-assisted development tools",
    "CTOs, VPs of AI, architects and senior technical leaders",
    "Product and infrastructure teams shipping AI-native products"
  ],
  "program": {
    "status": "speakers-published-schedule-pending",
    "note": "The speaker lineup and talk descriptions are public. Session days, times, rooms and the full timetable have not yet been published.",
    "speakerCount": 57,
    "sessionCount": 55,
    "speakers": [
      {
        "id": "04ac8e57-1b16-4a16-9639-d8d901a8a4c1",
        "slug": "adeline-yaw",
        "name": "Adeline Yaw",
        "pronouns": "She/Her",
        "jobTitle": "AI Integration Specialist",
        "employer": "Automattic",
        "bio": "Adeline Yaw spent eight years answering questions from WordPress.com users via live chat and email. She now leads the AI Squad at Automattic and looks after the AI assistants that users talk to first. She judges them the way a support person would: did the user get the right help, and did they reach a person when they needed one?",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/adeline-yaw.webp?v=9e492a1b",
        "url": "https://webdirections.org/ai-engineer/speakers/adeline-yaw/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/adelineyaw/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://adelineyaw.com"
        }
      },
      {
        "id": "2d40fe39-54e0-4faa-b123-ea65d94337da",
        "slug": "adesh-gairola-2",
        "name": "Adesh Gairola",
        "pronouns": "He/Him",
        "jobTitle": "Founder and CEO",
        "employer": "raxIT Labs",
        "bio": "Adesh Gairola is a security engineer and founder of raxIT Labs. For close to two decades, he has worked across governance, risk and compliance, from building controls to auditing them and deciding what they are for. At AWS, he moved regulated systems into audited cloud production, built the AI security practice for Australia and New Zealand, and ran hundreds of assessments for banks, governments and telcos. He helped ship the guardrails behind the first generative AI deployments at two of Australia’s four major banks and co-authored AWS’s reference threat model for AI agents, combining STRIDE with the Cloud Security Alliance’s MAESTRO framework. Earlier, he led a 30-engineer VPN and firewall team at Cisco and earned a CCIE in Security. He founded raxIT Labs in Sydney in 2025, presented “Kill the God Agent” at AI Engineer Melbourne 2026, runs AI Security Circle Sydney, and contributes to the OWASP LLM Top 10.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/adesh-gairola-2.webp?v=bd4befc9",
        "url": "https://webdirections.org/ai-engineer/speakers/adesh-gairola-2/",
        "social": {
          "social_media": "https://x.com/adeshgairola",
          "linkedin": "https://www.linkedin.com/in/adeshgairola/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://adeshgairola.com/"
        }
      },
      {
        "id": "e3cfdc31-2952-4db1-8124-21bf9d6bbba8",
        "slug": "aj-fisher",
        "name": "AJ Fisher",
        "pronouns": "he / him",
        "jobTitle": "VP Digital Science",
        "employer": "Tetratherix",
        "bio": "AJ Fisher is a technologist, researcher and writer, working at the intersection of AI, web and digital innovation. A regular speaker at Web Directions conferences, AJ brings a pragmatic, builder-first perspective to how emerging technologies reshape software engineering practice. He writes at ajfisher.me, where he explores everything from agentic coding workflows and local LLM setups to the strategic implications of AI adoption in the enterprise. At Tetratherix, he focusses on using digital and AI tools to help drive innovation and scale in scientific research and advanced manufacturing.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/aj-fisher.webp?v=b0a11e16",
        "url": "https://webdirections.org/ai-engineer/speakers/aj-fisher/",
        "social": {
          "social_media": "x: @ajfisher bs: @ajfisher.social LI: https://www.linkedin.com/in/andrewfisher/ w: https://ajfisher.me   gh: github.com/ajfisher",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": "https://ajfisher.me"
        }
      },
      {
        "id": "c3534f0b-dc48-4b9e-b4bc-b4ea9c1518a6",
        "slug": "anannya-roy-chowdhury",
        "name": "Anannya Roy Chowdhury",
        "pronouns": "she/her",
        "jobTitle": "GenAI Developer Advocate",
        "employer": "AWS",
        "bio": "Anannya Roy is an AI Engineer and Architect specializing in agentic systems, production-grade GenAI, and Responsible AI design. As a GenAI Developer Advocate at AWS, she works at the intersection of multi-agent architectures, observability, and real-world AI deployment helping developers move from demos to scalable, trustworthy systems.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/anannya-roy-chowdhury.webp?v=e6c2c693",
        "url": "https://webdirections.org/ai-engineer/speakers/anannya-roy-chowdhury/",
        "social": {
          "social_media": "https://www.linkedin.com/in/royanannya/",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": "https://github.com/royanannya"
        }
      },
      {
        "id": "1b33891f-3253-4a6c-9a96-4334beedf25a",
        "slug": "andrew-murphy",
        "name": "Andrew Murphy",
        "pronouns": null,
        "jobTitle": "CEO",
        "employer": "Prudixa",
        "bio": "Im Andrew, Co-founder & CEO of Prudixa and CTO of PatientNotes.app. Ive spent 20+ years in tech, from writing software to leading engineering teams. Alongside building businesses, I speak and teach about the skills developers need beyond coding. \r\n\r\n🚀 Co-founder & CEO at Prudixa.\r\nWe're building AI wallets that talent teams and event organisers can fund and manage themselves, making it easier to provide AI access for hiring assessments, workshops and events. \r\n\r\nIt started with a conversation about candidates being asked to pay for AI tools to complete a hiring assessment. I believe that if you ask someone to use a tool to show what they can do, you should provide it. \r\n\r\n🩺 CTO at PatientNotes.app Alongside Prudixa, Im CTO at PatientNotes.app, an AI healthcare company. That work gives me hands-on experience of building with AI and leading teams as the technology changes.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/andrew-murphy.webp?v=fe7335b3",
        "url": "https://webdirections.org/ai-engineer/speakers/andrew-murphy/",
        "social": {
          "social_media": "https://www.linkedin.com/in/andrewamurphy/",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": "https://andrewmurphy.io"
        }
      },
      {
        "id": "95984cba-c326-49cb-a9ff-ab0a1f4df719",
        "slug": "artem-yakimenko",
        "name": "Artem Yakimenko",
        "pronouns": "he/him",
        "jobTitle": "Engineering Director, Site Reliability",
        "employer": "Culture Amp",
        "bio": "I am an Engineering Director, SRE at Culture Amp, where I lead platform and reliability across a polyglot estate and support wider AI efforts across the business. I'm a long-time open source maintainer and my recent projects - Butter, an AI proxy gateway, and Denkeeper, a cost-budgeted personal AI agent - sit squarely in AI infrastructure. Ex-Google SRE/CRE.",
        "photo": "https://speakers.webdirections.org/photos/photo/95984cba-c326-49cb-a9ff-ab0a1f4df719?v=1791539865259",
        "url": "https://webdirections.org/ai-engineer/speakers/artem-yakimenko/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/temikus",
          "bluesky": null,
          "mastodon": null,
          "website": "https://startscaling.substack.com/"
        }
      },
      {
        "id": "25d01937-f1ce-4199-bae3-15119f9a87f1",
        "slug": "burin-choomnuan",
        "name": "Burin Choomnuan",
        "pronouns": "He/Him",
        "jobTitle": "Principal Engineer/Team Lead/AI Engineer",
        "employer": "NewsCorp Australia",
        "bio": "Burin Choomnuan is a Senior Principal Engineer at News Corp Australia, where two decades of shipping software has lately turned into shipping software with agents. Over the past year a two-person team has used ClojureDart to put more than a dozen production apps on iOS, Android and macOS, most of that code written through a structured agent workflow where implementer agents do the typing, reviewer agents grade the result, and the human stays architect.\r\n\r\nOutside work he ported a C game library to three young Lisp runtimes, 422 examples across Chez Scheme, C++/LLVM and the JVM, nearly all of it agent-written in languages with close to zero public training data. That work is open source. He has also rebuilt seven LLM agent harnesses in Clojure, porting from TypeScript, Rust and Python originals.\r\n\r\nHe speaks at Clojure/Conj and at Sydney meetups including FP-SYD and AI SYD. He recently restarted the Sydney Clojure meetup as Clojure Australia, a national group running monthly hybrid sessions, and co-hosts Gen AI Enthusiasts, a 2,000-member community meeting fortnightly on practical LLM use. He writes at b12n.net.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/burin-choomnuan.webp?v=ea067b2d",
        "url": "https://webdirections.org/ai-engineer/speakers/burin-choomnuan/",
        "social": {
          "social_media": "@agilecreativity",
          "linkedin": "https://linkedin.com/in/burinc",
          "bluesky": null,
          "mastodon": null,
          "website": "https://b12n.net"
        }
      },
      {
        "id": "429b809f-62da-4eea-85b0-f61b0030e4f7",
        "slug": "charli-posner",
        "name": "Charli Posner",
        "pronouns": "she/her",
        "jobTitle": "Builder",
        "employer": "Stile Education",
        "bio": "Charli Posner is an AI engineer exploring the limits of modern AI models in real-world systems.\r\n\r\nAt Stile Education, she has built AI pipelines that digitise handwritten student answers, image editing workflows that generate new facial expressions for hand-drawn characters, and infrastructure that lets coding agents test and debug changes in a running application. Her work also includes tools for monitoring agent behaviour and evaluating LLM outputs.\r\n\r\nHer background includes vector database R&D and two years researching human pose estimation at Toshiba and the University of Bristol. She focuses on the practical challenges of deploying AI systems: handling noisy inputs, unpredictable outputs, and making models reliable in real-world applications.\r\n\r\nOutside work, she builds creative tech projects spanning interactive pose-tracking, projection mapping and audiovisuals, and blogs about her experiments with emerging AI tools.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/charli-posner.webp?v=7f895de0",
        "url": "https://webdirections.org/ai-engineer/speakers/charli-posner/",
        "social": {
          "social_media": "https://www.linkedin.com/in/charli-posner-41195b20a/",
          "linkedin": "https://www.linkedin.com/in/charli-posner-41195b20a/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://charliposner.com/"
        }
      },
      {
        "id": "c0571a6f-24d4-4f2e-b3ab-ad9492054086",
        "slug": "chris-lienert",
        "name": "Chris Lienert",
        "pronouns": "he/his",
        "jobTitle": "Senior Principal Engineer",
        "employer": "Smart AI Connect",
        "bio": "Chris started out as a web developer when Netscape ruled the world and works as a Senior Principal Engineer at Smart AI Connect. Aside from musical distractions and accumulating frequent flyer points, Chris and his wife Sarah can be found in the company of their not-so-small human.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/chris-lienert.webp?v=590de553",
        "url": "https://webdirections.org/ai-engineer/speakers/chris-lienert/",
        "social": {
          "social_media": null,
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "dbb3f2e9-1c47-4553-b239-aafed793cfbc",
        "slug": "daniel-nadasi",
        "name": "Daniel Nadasi",
        "pronouns": "He/Him",
        "jobTitle": "Principal Engineer",
        "employer": "Google",
        "bio": "Daniel serves as a Principal Engineer for Google's developer infrastructure, focusing on enabling the company to scale from 100k human developers to millions of machine-speed agents, and serves as global co-chair of Google's Software Engineering Steering, which defines the role and craft of Saftware Engineering across Google. At Google prior to this Daniel has led cross-functional teams across the software stack including Google’s geographic data infrastructure, Google My Business Locations, Google Photos, and Google Tasks among others and was a founding lead for Google's office of Cross-Google Engineering. His experience traverses the technical spectrum and includes infrastructure, machine learning, mobile and web.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/daniel-nadasi.webp?v=831f8df8",
        "url": "https://webdirections.org/ai-engineer/speakers/daniel-nadasi/",
        "social": {
          "social_media": "LinkedIn: nadasi",
          "linkedin": "linkedin.com/in/nadasi",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "842323fe-f341-4947-9fa0-fec7ddc1893f",
        "slug": "dave-currie",
        "name": "Dave Currie",
        "pronouns": null,
        "jobTitle": "AI Engineer",
        "employer": "Square Peg",
        "bio": "David Currie is the AI Janitor at Square Peg. Over 15 years in software, including roles at Atlassian, Xero and Relevance AI, he has worked across support, product engineering, architecture and developer tooling. He now builds and operates an AI-first platform, working out how coding agents can move quickly without quietly wrecking the architecture.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/dave-currie.webp?v=ecd0bbb1",
        "url": "https://webdirections.org/ai-engineer/speakers/dave-currie/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/david-currie-52398617/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://www.linkedin.com/in/david-currie-52398617/"
        }
      },
      {
        "id": "757ebe2d-5644-4d2a-863c-b4dab4cae7fd",
        "slug": "dean-soste",
        "name": "Dean Soste",
        "pronouns": null,
        "jobTitle": "Senior Machine Learning Engineer",
        "employer": "Canva",
        "bio": "Dean Soste is a Senior Machine Learning Engineer at Canva, specialising in AI evaluation and the infrastructure needed to improve production AI systems. He has integrated production telemetry into Canva’s core help-domain AI products: the Help Assistant and Omni-agent, its support-ticket responder. His work supports a help ecosystem serving Canva’s 250 million monthly active users and handling roughly 180,000 support tickets each month.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/dean-soste.webp?v=ab7e9ecb",
        "url": "https://webdirections.org/ai-engineer/speakers/dean-soste/",
        "social": {
          "social_media": "https://au.linkedin.com/in/dean-soste",
          "linkedin": "https://au.linkedin.com/in/dean-soste",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "953c92d6-6dbb-4bc8-a701-04824d4d03b8",
        "slug": "donna-zhou",
        "name": "Donna Zhou",
        "pronouns": "She/Her",
        "jobTitle": "Software engineer",
        "employer": "Lorikeet",
        "bio": "Building AI customer support concierges for regulated industries, where the cost of a wrong conversation is high. Making non-deterministic systems reliable. Before I discovered coding, I was a stand-up comedian.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/donna-zhou.webp?v=d2f9a648",
        "url": "https://webdirections.org/ai-engineer/speakers/donna-zhou/",
        "social": {
          "social_media": null,
          "linkedin": "https://linkedin.com/in/donnazhou",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "5fa70ed5-de42-4062-808d-ede98a1e6c02",
        "slug": "ez-herrmann",
        "name": "Ez Herrmann",
        "pronouns": "he/him",
        "jobTitle": "Technical Solution Architect",
        "employer": "Sitback Solutions",
        "bio": "I'm AI Tech Lead and Enterprise Architect at Sitback in Sydney.\r\n\r\nMy path started with a robot. In 2013, I worked on PANTHER, the University of Bristol's first functioning robot, writing concurrent motor control across RS232 and a BeagleBoard with object recognition on top. From there I went deep into broadcast at Quicklink: satellite communications, media-over-IP streaming and hardware programming, including QuickLink TX, the Microsoft-partnered Skype and Teams broadcast transceiver used by live productions worldwide.\r\n\r\nI hold an MEng in Computing from Swansea University. My thesis examined critical systems reliability in medical device automation, analysing real-world failure scenarios in commercial syringe pumps.\r\n\r\nI've been coding since 2009, starting with C++ and PHP, and now work mostly across .NET and the cloud-native Azure ecosystem. I'm the technical half of the pair driving Sitback's AI rollout, covering client proofs of concept and MVPs on Azure AI Foundry, Algolia neural search, custom governance MCPs that load in every developer's IDE, the Umbraco Accelerator MCP, and Windmill orchestration.\r\n\r\nOutside work, I run a fairly serious homelab, build my own tooling end-to-end, and spend a lot of time on local LLM infrastructure and agentic coding workflows, which keeps the AI strategy work honest.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/ez-herrmann.webp?v=0e460f0e",
        "url": "https://webdirections.org/ai-engineer/speakers/ez-herrmann/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/ezherrmann",
          "bluesky": null,
          "mastodon": null,
          "website": "https://ezherrmann.dev/"
        }
      },
      {
        "id": "f29357fd-efe5-4394-a4cf-7f515de3ad87",
        "slug": "fawaz-ahmad",
        "name": "Fawaz Ahmad",
        "pronouns": "he/him/his",
        "jobTitle": "Senior Engineering Director",
        "employer": "Canva",
        "bio": "Fawaz Ahmad has led the Infrastructure org at Canva since 2020, our Core Infra Platform, Developer Experience, and Data Platform, of over 250 engineers across 27 teams that runs one of the largest cloud footprints in the region and builds the tooling to empower 3,000 engineers. \r\n\r\nHe also drives Canva Technology's AI First Engineering goal, owning how the company measures AI value, governs token spend across multiple model vendors and coding agents, and tells its AI-enabled engineering story. His org drove Canva's Kubernetes and Aurora re-platforms, same-day dev-to-production releases and tens of millions in annual infrastructure savings. \r\n\r\nFawaz spoke at the AWS Summit Sydney Executive Forum in 2026 and writes a regular \"Top of Mind\" column for Canva's technology org on operating an infrastructure organisation in an agent-first world.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/fawaz-ahmad.webp?v=7db6238f",
        "url": "https://webdirections.org/ai-engineer/speakers/fawaz-ahmad/",
        "social": {
          "social_media": "https://au.linkedin.com/in/thefwz",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "65ae247b-696e-4928-9a9e-bb1ce2e9c86e",
        "slug": "fiona-chan",
        "name": "Fiona Chan",
        "pronouns": "she/her",
        "jobTitle": "Partner & Technical Recruiter",
        "employer": "Lookahead",
        "bio": "Fiona is a partner & technical recruiter at Lookahead, where she's helped Australian product companies hire across engineering, product and design. Before moving into recruitment, she worked as a senior frontend engineer and was the co-founder of the meetup SydCSS.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/fiona-chan.webp?v=0f36e1de",
        "url": "https://webdirections.org/ai-engineer/speakers/fiona-chan/",
        "social": {
          "social_media": null,
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": "https://www.linkedin.com/in/fionakychan/"
        }
      },
      {
        "id": "1727370a-ee5c-4f17-9b6f-e0b563453afe",
        "slug": "gareth-williams",
        "name": "Gareth Williams",
        "pronouns": null,
        "jobTitle": "Principal Engineer",
        "employer": "Wesfarmers",
        "bio": "[Gareth Williams](https://www.linkedin.com/in/gareth-williams-ai/) is a Principal AI Engineer at Wesfarmers AI Accelerator. He has spent fifteen years building software, most of it in regulated enterprise - banking (Westpac, Lloyds), funds (Colonial First State), energy (Shell Energy) and government - and the last two working out how to get agentic systems doing useful work without quietly making large mistakes.\r\n\r\nBefore OneDigital he was a Principal Solutions Architect at Versent, where he helped run the AI Guild and started the Tech Radar. He writes on [Medium](https://gazzwi86.medium.com) and [Substack](https://prompttoprod.substack.com), mostly on [ontologies](https://gazzwi86.medium.com/ontologies-helping-ai-agents-understand-how-your-business-operates-f943354e68f9): giving agents a working model of how the business actually operates. He also writes on [comprehension debt](https://gazzwi86.medium.com/comprehension-debt-the-silent-risk-of-ai-accelerated-delivery-b8aa54455797) and on agentic engineering - briefing, constraining and reviewing agents like the keen but amnesiac graduates they are.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/gareth-williams.webp?v=9816c13d",
        "url": "https://webdirections.org/ai-engineer/speakers/gareth-williams/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/gareth-williams-solutions-consultant/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://prompttoprod.substack.com"
        }
      },
      {
        "id": "8df67c68-18d6-4cc2-a60c-5b5eb44ed4a6",
        "slug": "hamish-songsmith",
        "name": "Hamish Songsmith",
        "pronouns": "he",
        "jobTitle": "Head of Applied AI",
        "employer": "Silicon Quantum Computing",
        "bio": "Head of Applied AI at Australia's leading Quantum Computing lab. Ex Applied AI lead at Optiver, deep experience in AI risk, governance, scalable deployments of AI agents.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/hamish-songsmith.webp?v=0fc452d6",
        "url": "https://webdirections.org/ai-engineer/speakers/hamish-songsmith/",
        "social": {
          "social_media": "www.linkedin.com/in/hsongsmith",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": "https://hsongsmith.com"
        }
      },
      {
        "id": "d9989619-2b45-47da-9dd8-56c7f90e5ba7",
        "slug": "harlan-wilton",
        "name": "Harlan Wilton",
        "pronouns": "He / Him",
        "jobTitle": "Open-source developer",
        "employer": "Self Employed",
        "bio": "Independently funded open-source developer working on the Nuxt core team, amongst other projects.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/harlan-wilton.webp?v=ede3d77b",
        "url": "https://webdirections.org/ai-engineer/speakers/harlan-wilton/",
        "social": {
          "social_media": "https://x.com/harlan_zw",
          "linkedin": "n/a",
          "bluesky": "https://bsky.app/profile/harlanzw.com",
          "mastodon": null,
          "website": "https://harlanzw.com"
        }
      },
      {
        "id": "26daf7a5-1b4a-4bba-8bfa-e286f2b1b762",
        "slug": "harry-nguyen",
        "name": "Harry Nguyen",
        "pronouns": "he/him",
        "jobTitle": "AI engineer",
        "employer": "insightfactory.ai",
        "bio": "I'm an AI Engineer at insightfactory.ai, where I build multi-agent systems at massive scale. I studied Computer Science at the University of Adelaide, and previously worked as a Machine Learning Engineer at the Australian Institute of Machine Learning, where my research focused on adversarial attacks and backdoor attacks against state-of-the-art object detectors, published at top-tier security conferences.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/harry-nguyen.webp?v=0e87579f",
        "url": "https://webdirections.org/ai-engineer/speakers/harry-nguyen/",
        "social": {
          "social_media": "https://www.linkedin.com/in/quangdangnguyen/",
          "linkedin": "https://www.linkedin.com/in/quangdangnguyen/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://harryng.dev/"
        }
      },
      {
        "id": "1b09233a-caa7-49c6-96e1-46107d84731b",
        "slug": "hugo-oconnor",
        "name": "Hugo O'Connor",
        "pronouns": "he/him",
        "jobTitle": "R&D Engineer",
        "employer": "Anuna Research",
        "bio": "I am an engineer, founder and researcher building tools that help groups of people coordinate. I co-founded Australia’s first cryptocurrency exchange (acquired by Kraken), am a co-inventor on two international patents in digital identity, and have built formally verified languages and deterministic reasoning systems. My experience spans digital finance, supply chain, biosecurity, e-government and real estate. I am a founding member of Anuna Research, an R&D cooperative with a mission to build life-ennobling technology that nourishes people, communities and the planet.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/hugo-oconnor.webp?v=641db4e9",
        "url": "https://webdirections.org/ai-engineer/speakers/hugo-oconnor/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/hugo-o-connor-43512a3a/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://anuna.io"
        }
      },
      {
        "id": "dcd0092c-f257-49c5-ba66-1d34c8d3715d",
        "slug": "inga-pflaumer",
        "name": "Inga Pflaumer",
        "pronouns": "she/her",
        "jobTitle": "Consulting CTO",
        "employer": null,
        "bio": "With 15+ years in tech and a background in AI and web3, Inga Pflaumer has led teams across startups and scaleups. Inga is also a mentor, diversity advocate, and part-time game developer who believes great engineering is equal parts logic, curiosity, and emotional intelligence\r\n\r\nInga was formerly Head of Engineering at Relevance AI, where she built agentic workflows that automate the painful parts of work - including her own.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/inga-pflaumer.webp?v=30163fd8",
        "url": "https://webdirections.org/ai-engineer/speakers/inga-pflaumer/",
        "social": {
          "social_media": null,
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "98b328a7-bf1a-4d51-a153-af092a33e723",
        "slug": "jack-bear",
        "name": "Jack Bear",
        "pronouns": null,
        "jobTitle": "CEO Founder",
        "employer": "Norg AI",
        "bio": "Jack is a 24 year old founder with a deep, self-taught command of artificial intelligence. Since 2021, he has been studying and working hands on with emerging AI systems, model behaviour, search evolution and large scale content engineering. He offers rich insight into how AI models ingest, interpret and surface information, and has a clear understanding of how AI technology is evolving.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/jack-bear.webp?v=c7ab0540",
        "url": "https://webdirections.org/ai-engineer/speakers/jack-bear/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/jack-bear-726578226/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://norg.ai/"
        }
      },
      {
        "id": "b8c00501-6f53-4e78-9335-811244b0f3bd",
        "slug": "jack-rudenko",
        "name": "Jack Rudenko",
        "pronouns": "He/Him",
        "jobTitle": "Chief AI Officer (CAIO)",
        "employer": "10X Labs",
        "bio": "The CTO at MadAppGang and founder of 10x Labs, Sydney, Jack works with real Australian businesses running AI on real workflows. Not pilots. Not demos. Actual stuff they depend on. \r\n\r\nWe're deep in the experimental phase - learning more from what breaks than what works.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/jack-rudenko.webp?v=e6c97434",
        "url": "https://webdirections.org/ai-engineer/speakers/jack-rudenko/",
        "social": {
          "social_media": "https://www.linkedin.com/in/erudenko/https://www.threads.com/@jackrudenkohttps://github.com/erudenkohttps://x.com/jackrudenko",
          "linkedin": "https://www.linkedin.com/feed/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://10xlabs.com.au"
        }
      },
      {
        "id": "f01a5cf9-b41e-4f3a-939d-b0d097795af9",
        "slug": "kexuan-xin",
        "name": "Jade Xin",
        "pronouns": "she",
        "jobTitle": "Principal Research Scientist",
        "employer": "TensorPRO",
        "bio": "Dr Kexuan (Jade) Xin is a Principal Research Scientist at TensorPRO in Sydney, working on the EDN Memory Engine, which is a provenance-governed external memory layer for LLM systems. She leads EDN's benchmark evaluation, running LongMemEval and LoCoMo end to end with real production metrics, and its correction-driven personalization engine. Her recent work includes the first real-data validation of EDN's AI-generated-content exclusion invariant, and a study of where provenance-governed retrieval fits and where it does not. She holds a PhD in Computer Science from the University of Queensland, and has worked on NLP, knowledge graphs, and reliable AI at the University of Illinois Urbana-Champaign and Macquarie University. She is particularly interested in the engineering boundary between what an AI system can generate and what it should be allowed to remember as fact.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/kexuan-xin.webp?v=53e99d54",
        "url": "https://webdirections.org/ai-engineer/speakers/kexuan-xin/",
        "social": {
          "social_media": null,
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": "https://scholar.google.com/citations?user=I5jkzbEAAAAJ&hl=en"
        }
      },
      {
        "id": "a98c3b0f-5daf-432e-acff-be5956f64e17",
        "slug": "james-peter",
        "name": "James Peter",
        "pronouns": "he/him",
        "jobTitle": "Co-Founder",
        "employer": "JustEvery",
        "bio": "James is an entrepreneur and software engineer building AI products across design, coding and automation. He is the co-founder of JustEvery, a group of AI companies run by a small team and thousands of agents. They include 12ui, which turns visual references into editable, production-ready interfaces, and Every Code, an popular open-source fork of OpenAI's Codex CLI with over 4k GitHub stars. He is now building RunEditRun, an open-source alternative to many SaaS products. Trained in physics and NLP, James is most interested in the gap between what models can do in a demo and what it takes for agents to work reliably when nobody is watching.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/james-peter.webp?v=c3376609",
        "url": "https://webdirections.org/ai-engineer/speakers/james-peter/",
        "social": {
          "social_media": "@zemaj",
          "linkedin": "https://www.linkedin.com/in/jamespeter/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://zemaj.com"
        }
      },
      {
        "id": "49a97552-09a6-4153-a064-a81022fdaf7b",
        "slug": "jan-peer-stocklmair",
        "name": "Jan Peer Stöcklmair",
        "pronouns": null,
        "jobTitle": "Senior Software Engineer",
        "employer": "Sentry",
        "bio": "Jan Peer Stöcklmair is a Senior Software Engineer at Sentry. When he's not deep in code, you'll probably find him kitesurfing, road biking, swimming or hanging off a rock wall. Whether he's debugging or chasing wind, he's always looking for that perfect flow.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/jan-peer-stocklmair.webp?v=282300de",
        "url": "https://webdirections.org/ai-engineer/speakers/jan-peer-stocklmair/",
        "social": {
          "social_media": null,
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": "https://linkedin.com/in/jpeer"
        }
      },
      {
        "id": "17836589-1ad9-4722-a55e-787c5dca9e73",
        "slug": "jeffrey-aven",
        "name": "Jeffrey Aven",
        "pronouns": "He/Him",
        "jobTitle": "Maintainer",
        "employer": "StackQL Studios",
        "bio": "https://www.linkedin.com/in/jeffreyaven/",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/jeffrey-aven.webp?v=e35ce9bf",
        "url": "https://webdirections.org/ai-engineer/speakers/jeffrey-aven/",
        "social": {
          "social_media": "https://x.com/stackql, https://www.linkedin.com/company/stackql/",
          "linkedin": "https://www.linkedin.com/in/jeffreyaven/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://github.com/stackql/stackql"
        }
      },
      {
        "id": "dd6d963a-4992-4da0-937d-0ce47c3e0f50",
        "slug": "jiggy-kakkad",
        "name": "Jiggy Kakkad",
        "pronouns": "he/him",
        "jobTitle": "Staff AI Engineer",
        "employer": "Quantium",
        "bio": "Jiggy Kakkad is a Staff AI Engineer at Quantium, working on Checkout AI’s conversational analytics\r\nagent. He came to AI engineering from a software engineering background, and spends his time on\r\nharness engineering and automation: the layers around the model that decide what it can do and what\r\nreaches production.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/jiggy-kakkad.webp?v=1934a313",
        "url": "https://webdirections.org/ai-engineer/speakers/jiggy-kakkad/",
        "social": {
          "social_media": "jig2nesh",
          "linkedin": "https://www.linkedin.com/in/jiggy/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://jiggykakkad.com"
        }
      },
      {
        "id": "f0ce70c7-a741-47eb-8b17-d848e84143ad",
        "slug": "jon-shen",
        "name": "Jon Shen",
        "pronouns": null,
        "jobTitle": "AI Practice Executive - Strategy and Governance",
        "employer": "Suncorp",
        "bio": "Jon is a data science actuary and AI Practice Executive at Suncorp, building AI applications to make insurance easier for customers in their time of need. That means better, faster claims decisions so they can get on with their lives. In his role, Jon ensures deployed AI systems can be trusted by staff and customers alike, by establishing best practices around AI risks, security, ethics and safety. This requires a pragmatic balance between commercial value, risk management, and change adoption. \r\n\r\nJon recently published the dialogue paper \"Building Tomorrow: Preparing Australia for the Age of AI\" to spark conversations about Australia's approach to AI. He also serves on the Board of the Actuaries Institute.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/jon-shen.webp?v=b6e59953",
        "url": "https://webdirections.org/ai-engineer/speakers/jon-shen/",
        "social": {
          "social_media": "https://www.linkedin.com/in/jon-shen/",
          "linkedin": "https://www.linkedin.com/in/jon-shen/",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "5cd0ac5a-618e-4d7c-a894-cd93bbbf0c14",
        "slug": "karla-burnett",
        "name": "Karla Burnett",
        "pronouns": "she/her",
        "jobTitle": "Security Engineer",
        "employer": "Lorikeet",
        "bio": "I'm a security- and infrastructure-focused software engineer at Lorikeet, an AI customer support automation platform for regulated industries. At Lorikeet,  I've built production AI systems including our RAG stack, migrated our ticket processing pipeline to Temporal, built the public-facing CTF I'll be speaking about, and cut our infrastructure costs by 70% in the space of 2.5 months.\r\n\r\nBefore Lorikeet, I spent eight years at Stripe across security and product engineering. There, I rewrote Stripe's web authentication stack, helped build the first version of Stripe Sigma, cut account-takeover losses in half, built systems for securely executing arbitrary code against production, and, at various points, phished the entire company.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/karla-burnett.webp?v=d7487040",
        "url": "https://webdirections.org/ai-engineer/speakers/karla-burnett/",
        "social": {
          "social_media": "https://www.linkedin.com/in/karla-burnett-02997762/",
          "linkedin": "https://www.linkedin.com/in/karla-burnett-02997762/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://karla.io"
        }
      },
      {
        "id": "7a3ef210-0b37-498c-8285-ea01a6427080",
        "slug": "kevin-nguyen",
        "name": "Kevin Nguyen",
        "pronouns": "He/Him",
        "jobTitle": "Senior Product Engineer",
        "employer": "Skip",
        "bio": "I love Jesus and building things, currently working on automating credit decisioning at Skip (Home loans).\r\nPreviously, I was a Senior Engineer at Sauce (Feedback engine) working in SWE/AI/ML.\r\nBefore that, I was a Senior Frontend Engineer at Mirvac (Construction) working on the web and mobile apps.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/kevin-nguyen.webp?v=a2d59626",
        "url": "https://webdirections.org/ai-engineer/speakers/kevin-nguyen/",
        "social": {
          "social_media": "https://x.com/kndwindev",
          "linkedin": "https://www.linkedin.com/in/kndwindev/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://kndwin.dev"
        }
      },
      {
        "id": "40dfbfd0-f98b-4e91-add0-c24b04172fd4",
        "slug": "khali-kalpa-young",
        "name": "Khali Kalpa-Young",
        "pronouns": "He",
        "jobTitle": "Founder",
        "employer": "Alchymie",
        "bio": "Khali Kalpa-Young is the founder of Alchymie, an Australian consultancy that builds and operates AI agents inside small and mid-sized businesses. He began as a software engineer at ThoughtWorks in Melbourne and San Francisco, then spent a decade leading digital transformation at scale, including a Vodafone program of roughly 100 people across six to eight teams that doubled online sales and turned employee engagement and NPS around, and enterprise coaching work at Optus and ELMO Software.\r\n\r\nBetween those two phases he spent several years as a leadership coach and facilitator, running trainings and courses for corporate teams and for Triibe, the practice he co-founded. That work is the reason his agents get built the way they do: the hardest part of putting an agent into someone's business is not the model, it is extracting the tacit standard the owner has never written down.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/khali-kalpa-young.webp?v=635a2461",
        "url": "https://webdirections.org/ai-engineer/speakers/khali-kalpa-young/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/khaliyoung/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://alchymie.ai"
        }
      },
      {
        "id": "23d5163e-d113-416c-8c77-812b54fa154a",
        "slug": "khang-nguyen-hoang",
        "name": "Khang Nguyen Hoang",
        "pronouns": "He",
        "jobTitle": "Data Scientist",
        "employer": "Hello Clever",
        "bio": "**Khang Nguyen** is a Data Scientist at Hello Clever, a global payments company supporting checkout in more than 25 currencies, where he owns the data and AI side of the business: the agent systems behind its AI products, the pipelines underneath them, and the evaluation and instrumentation that keep them honest in production. He has spent four years in data and AI, mostly on the unglamorous half of the work: measuring systems, finding out why they misbehave under real traffic, and rebuilding them when the architecture turns out to be wrong. Khang holds an Honours degree in Software Engineering and a Master of AI from RMIT, finishing in the top 2% of the university for both.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/khang-nguyen-hoang.webp?v=80f85ad1",
        "url": "https://webdirections.org/ai-engineer/speakers/khang-nguyen-hoang/",
        "social": {
          "social_media": "https://www.linkedin.com/in/hoangkhangn/",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "b03abb6d-c070-4ab8-824f-f49ec8179fa8",
        "slug": "lee-simpson",
        "name": "Lee Simpson",
        "pronouns": null,
        "jobTitle": "Engineering Partner",
        "employer": "Deloitte Australia",
        "bio": "Lee is a Partner in our Engineering, AI and Data team in Queensland and is responsible for driving innovation and delivering cutting-edge digital solutions across the region. Lee's primary focus is delivering custom software, and complex enterprise integration for large digital transformations, so that our customers remain competitive in a dynamic market.\r\n \r\nLee brings extensive experience in leveraging modern engineering techniques to deliver new platforms with seamless system integration, reduce operational costs, and create exceptional digital experiences. \r\n\r\nLee is the Head of AI Accelerated Engineering is transforming how we deliver software across Deloitte’s people, processes and tools.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/lee-simpson.webp?v=c1ccc13c",
        "url": "https://webdirections.org/ai-engineer/speakers/lee-simpson/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/lee-simpson/",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "cc875df4-a947-46aa-9cb0-63c72de3595a",
        "slug": "leigh-whiting",
        "name": "Leigh Whiting",
        "pronouns": null,
        "jobTitle": "Principal Engineer",
        "employer": "Atlassian",
        "bio": "Leigh is a Principal Engineer at Atlassian, where he works in Jira on bringing to life our agentic product strategy. His current focus is enabling AI agents to seamlessly operate at scale across Jira while preserving the trust, performance and permission guarantees customers depend on.\r\n\r\nBefore Atlassian, Leigh built software across a broad range of domains including institutional banking platforms, programmatic advertising infrastructure, enterprise search systems, autonomous robotics, and safety-critical medical device software. This background — spanning regulated finance, real-time bidding, information retrieval, and systems where failure isn't an option — informs his approach to building reliable, scalable platforms that balance innovation with operational rigour at scale.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/leigh-whiting.webp?v=6cb27afa",
        "url": "https://webdirections.org/ai-engineer/speakers/leigh-whiting/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/leigh-whiting-47177ba7/",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "df3e8324-5c0f-4726-bc3c-6f8908e4d2f3",
        "slug": "leo-borges",
        "name": "Leo Borges",
        "pronouns": null,
        "jobTitle": "Chief Engineer - GenAI",
        "employer": "Commbank",
        "bio": "Leonardo Borges is Chief Engineer for Generative AI at CommBank, where he leads engineering strategy for AI adoption across one of the world's largest financial institutions. Previously Senior Director of AI Engineering at Optus, he has spent over a decade building production distributed systems and now focuses on the architecture, security, and evaluation challenges of deploying AI agents at enterprise scale.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/leo-borges.webp?v=50f71101",
        "url": "https://webdirections.org/ai-engineer/speakers/leo-borges/",
        "social": {
          "social_media": "@theleoborges",
          "linkedin": "https://www.linkedin.com/in/theleoborges/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://steelthread.cc/"
        }
      },
      {
        "id": "d6194323-1564-4f45-bb71-6d8e9f5dea01",
        "slug": "mark-johnson",
        "name": "Mark Johnson",
        "pronouns": null,
        "jobTitle": "CEO",
        "employer": "Cardiobase",
        "bio": "I'm the CEO of Cardiobase. We build clinical information systems — Cardiobase (cardiac), Rezibase (respiratory & sleep) and Clinibase (clinical trial management) — for public hospitals in Australia, New Zealand, the UK and Ireland. I'm a Chartered Systems Engineer with a background in building and optimising critical infrastructure. Over the past year I've applied that discipline to harness engineering: a V-model \"software factory\" that takes every Jira ticket from vague request to verified PR, with agents doing the work, deterministic gates enforcing the process, adversarial verifiers checking it, and humans approving at exactly two points. It runs largely unattended and has cut our ticket cycle time from 45 days to 6. I write about AI leverage at thelearning.ceo.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/mark-johnson.webp?v=f3c66c3a",
        "url": "https://webdirections.org/ai-engineer/speakers/mark-johnson/",
        "social": {
          "social_media": "https://www.linkedin.com/in/thelearningceo/",
          "linkedin": "https://linkedin.com/in/thelearningceo",
          "bluesky": null,
          "mastodon": null,
          "website": "https://thelearning.ceo"
        }
      },
      {
        "id": "8d0ce144-235a-4f85-b514-690ad8b4c62a",
        "slug": "mark-pesce",
        "name": "Mark Pesce",
        "pronouns": "he/him",
        "jobTitle": "Co-Founder",
        "employer": "Noops",
        "bio": "Co-inventor VRML; author 12 books, including \"Getting Started with ChatGPT and AI Chatbots\"; founded postgraduate programs at USC and AFTRS; honorary associate at the University of Sydney. Award-winning columnist for The Register; Award-winning podcaster with \"The Next Billion Seconds\". Futurist, public speaker, and frequent keynoter of Web Directions. Co-founder of NOOPS, a research advisory focused on the entire AI stack, \"from sand to software\".",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/mark-pesce.webp?v=de9d5b78",
        "url": "https://webdirections.org/ai-engineer/speakers/mark-pesce/",
        "social": {
          "social_media": "https://www.linkedin.com/in/markpesce/",
          "linkedin": "https://www.linkedin.com/in/markpesce/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://markpesce.com"
        }
      },
      {
        "id": "1af19cf3-d7d1-4046-b745-d9118603e501",
        "slug": "mukesh-singh",
        "name": "Mukesh Singh",
        "pronouns": null,
        "jobTitle": "Principal Detection and Response Engineer",
        "employer": "Atlassian",
        "bio": "Mukesh Singh is a Principal Security Detection & Response Engineer at Atlassian, leading the development of AI agents designed for real-time threat detection and response. His work spans the complete engineering lifecycle from high-throughput stream processing and OCSF normalisation pipelines to building dbt data products on Databricks and maintaining the detection-as-code rule corpus that powers autonomous agent operations. In addition to his core detection engineering work, Mukesh leads Atlassian’s ML platform for threat detection.\r\n\r\nDeeply passionate about applied AI engineering, Mukesh focuses on evaluation design, token economics, and the production edge cases that standard benchmarks miss. He specializes in bridging the gap between theoretical AI models and real-world security operations focusing on failure mode analysis and determining the exact unit economics where agentic systems transition from cost center to measurable force multipliers.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/mukesh-singh.webp?v=68aceda9",
        "url": "https://webdirections.org/ai-engineer/speakers/mukesh-singh/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/singh-mukesh2028/",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "104fdda7-943c-49a1-a9bd-90736afd6e12",
        "slug": "nadia-makarevich",
        "name": "Nadia Makarevich",
        "pronouns": "she/her",
        "jobTitle": "Principal Engineer",
        "employer": "Heatseeker",
        "bio": "Nadia has spent 20 years engineering frontend systems, including a few years in the Jira Frontend Platform team (1M+ LOC, 300+ engineers, Atlassian), and is now Principal Engineer at Heatseeker, leading context engineering and AI coding infrastructure.\r\n\r\nAuthor of _Advanced React_ and _Web Performance Fundamentals_. Spends too much time writing long-form investigations and deep dives at developerway.com (read by ~500k engineers a year), mostly because she insists on backing up claims with numbers and reproducible code examples.\r\n\r\nCurrently investigating: AI memory and Context Engineering. Currently surviving: the AI hype cycle, by shipping production systems instead of demos.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/nadia-makarevich.webp?v=53963c25",
        "url": "https://webdirections.org/ai-engineer/speakers/nadia-makarevich/",
        "social": {
          "social_media": "https://x.com/adevnadia",
          "linkedin": "https://www.linkedin.com/in/adevnadia/",
          "bluesky": "https://bsky.app/profile/adevnadia.bsky.social",
          "mastodon": null,
          "website": "https://developerway.com/"
        }
      },
      {
        "id": "e0d72b91-4095-4495-8541-518f14d99ab1",
        "slug": "rahul-trikha",
        "name": "Rahul Trikha",
        "pronouns": null,
        "jobTitle": "Principal AI Engineer",
        "employer": "Zendesk",
        "bio": "**Rahul Trikha** is a **Principal AI Engineer at [Zendesk](https://www.zendesk.com/)**, where he builds the platforms and evaluation systems that help AI agents operate reliably in production. His work spans agent observability, trace-based evaluation, durable execution, governed tools, and safe execution at enterprise scale. He is interested in the practical engineering seams that turn promising agent demos into systems teams can trust.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/rahul-trikha.webp?v=586d0e0d",
        "url": "https://webdirections.org/ai-engineer/speakers/rahul-trikha/",
        "social": {
          "social_media": "https://www.linkedin.com/in/rahult/",
          "linkedin": "https://www.linkedin.com/in/rahult/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://rahultrikha.com"
        }
      },
      {
        "id": "1f224d74-f60a-4969-a431-36097829dc9b",
        "slug": "rupert-manfredi",
        "name": "Rupert Manfredi",
        "pronouns": "he/him",
        "jobTitle": "Head of Design",
        "employer": "Telepath",
        "bio": "Rupert Manfredi is co-founder of Telepath, where he is building Television, a GUI for personal agents. He began working with language models at Google Creative Lab in 2018, fine-tuning models to create early generative interfaces for writers and musicians. He subsequently worked as a design technologist at Mozilla and at Adept.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/rupert-manfredi.webp?v=7385d2ef",
        "url": "https://webdirections.org/ai-engineer/speakers/rupert-manfredi/",
        "social": {
          "social_media": "@ruperts.world (Bluesky) @rupertmanfredi (X)",
          "linkedin": "https://linkedin.com/in/rupertmanfredi",
          "bluesky": "@ruperts.world",
          "mastodon": null,
          "website": "https://ruperts.world"
        }
      },
      {
        "id": "69073d24-65c8-4d33-80fd-89b1a554f90a",
        "slug": "sandra-arato",
        "name": "Sandra Arato",
        "pronouns": "she / her",
        "jobTitle": "Senior Software Engineer",
        "employer": "Canva",
        "bio": "Sandra Arato is a Senior Software Engineer on Canva's Leonardo.AI team, where she builds AI-agent experiences and the evaluation and observability systems used to improve them. Her work spans frontend architecture, serverless services, tracing, datasets and model evaluation, with a focus on turning specialist workflows into shared capabilities for engineering, design and research. Sandra regularly speaks at internal engineering and AI forums about measuring non-deterministic systems and previously presented at Next.js Sydney. She is based on the Sunshine Coast, Australia.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/sandra-arato.webp?v=e8ef6881",
        "url": "https://webdirections.org/ai-engineer/speakers/sandra-arato/",
        "social": {
          "social_media": null,
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "fa084c3b-8e0a-46c0-a28c-297f734e3173",
        "slug": "sharat-madanapalli",
        "name": "Sharat Madanapalli",
        "pronouns": "he/him",
        "jobTitle": "Founder",
        "employer": "InTune AI",
        "bio": "Dr Sharat Madanapalli is the founder of InTune AI, where he helps businesses adopt AI through purpose-built systems and executive AI workshops. His career spans AI research, software engineering and startup leadership. He earned his PhD in AI from UNSW Sydney before leading data and AI teams at technology startups. Sharat is an active member of Sydney’s AI community and regularly speaks at industry and community events.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/sharat-madanapalli.webp?v=9177cb8d",
        "url": "https://webdirections.org/ai-engineer/speakers/sharat-madanapalli/",
        "social": {
          "social_media": "@sharat910",
          "linkedin": "https://linkedin.com/in/sharat910",
          "bluesky": null,
          "mastodon": null,
          "website": "https://www.intuneai.com.au"
        }
      },
      {
        "id": "4ed433ee-59b3-482d-b66c-edee77c4fc93",
        "slug": "shivay-lamba",
        "name": "Shivay Lamba",
        "pronouns": "He/Him",
        "jobTitle": "Lead Machine Learning | Developer Experience Engineer",
        "employer": "Qualcomm",
        "bio": "Shivay Lamba is a software engineer and open source contributor passionate about AI and edge computing. With experience across startups and enterprise tech, he focuses on simplifying complex technologies for developers. Shivay actively speaks at global conferences, organizes community events, and contributes to projects in cloud-native ecosystems, WebAssembly, and machine learning. He has also built and mentored educational programs to bridge gaps in emerging tech adoption. When not coding, he enjoys writing, traveling, and mentoring young technologists.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/shivay-lamba.webp?v=ca60d5f2",
        "url": "https://webdirections.org/ai-engineer/speakers/shivay-lamba/",
        "social": {
          "social_media": "https://x.com/HowDevelop",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": "https://shivaylamba.me/"
        }
      },
      {
        "id": "14d4a7be-0922-4b9e-92da-b817fca99f04",
        "slug": "shrey-somaiya",
        "name": "Shrey Somaiya",
        "pronouns": "they/them",
        "jobTitle": "Principal Engineer - Jira Frontend Platform",
        "employer": "Atlassian",
        "bio": "Shrey is a Principal Engineer at Atlassian and an engineering leader in Jira Frontend Platform. They work across frontend infrastructure, build tooling, and systems programming to improve the developer experience for more than 1500 engineers in a >18-million-line codebase. They like fast feedback, tools that get out of the way, and the colour purple.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/shrey-somaiya.webp?v=1d4157de",
        "url": "https://webdirections.org/ai-engineer/speakers/shrey-somaiya/",
        "social": {
          "social_media": "@spanishpear.bsky.social‬",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "f4255d66-a440-4a28-9eb0-649df4db680c",
        "slug": "simon-harloff",
        "name": "Simon Harloff",
        "pronouns": "He / Him",
        "jobTitle": "CTO",
        "employer": "Dam Secure",
        "bio": "Simon has spent the last decade building platform infrastructure and developer tooling at some of Australia's leading software companies.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/simon-harloff.webp?v=f373e3ff",
        "url": "https://webdirections.org/ai-engineer/speakers/simon-harloff/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/simonharloff/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://damsecure.ai/"
        }
      },
      {
        "id": "94d3fb30-ac78-493e-8529-163ecf6e63f9",
        "slug": "stephen-sennett",
        "name": "Stephen Sennett",
        "pronouns": "he/him/his",
        "jobTitle": "Principal Forward-Deployed Engineer - Applied AI",
        "employer": "V2 AI",
        "bio": "Stephen Sennett is a cloud technology leader, content creator, educator, and speaker. He worked in the industry for over a decade across in a variety of roles, currently as a Principal Forward-Deployed Engineer specialising in Applied AI and Cloud with V2 AI. He holds high-level certifications across multiple technologies, has been recognised as an AWS Community Hero, spoken at events around the world, and authors technical content with A Cloud Guru (a Pluralsight company).\r\n\r\nOutside work, he is a dedicated volunteer across numerous organisations, primarily in the Emergency Management sector, serving around the country during several major national disasters.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/stephen-sennett.webp?v=22dda5f7",
        "url": "https://webdirections.org/ai-engineer/speakers/stephen-sennett/",
        "social": {
          "social_media": "https://linkedin.com/in/ssennettau/",
          "linkedin": "https://www.linkedin.com/in/ssennettau/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://ssennett.net/"
        }
      },
      {
        "id": "2358d0c6-848e-494d-a0d2-36b1c52e462e",
        "slug": "sudharsanam-narasimhan",
        "name": "Sudharsanam Narasimhan",
        "pronouns": "he/him",
        "jobTitle": "Senior engineering manager, Design Systems & AI Developer Platforms",
        "employer": "Atlassian",
        "bio": "Sudharsanam Narasimhan is a Senior Engineering Manager at Atlassian, where he leads teams across Atlassian design systems, AI-enabled productivity, accessibility, and internationalisation. His work focuses on turning deep domain knowledge into scalable AI platforms that improve software quality, developer productivity, and operational efficiency across large organisations.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/sudharsanam-narasimhan.webp?v=39d0aa58",
        "url": "https://webdirections.org/ai-engineer/speakers/sudharsanam-narasimhan/",
        "social": {
          "social_media": "https://www.linkedin.com/in/sudharsanam-narasimhan-bba08631/",
          "linkedin": "https://www.linkedin.com/in/sudharsanam-narasimhan-bba08631/",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "45acc7e8-f224-4635-855e-924390d1604f",
        "slug": "tanya-dixit",
        "name": "Tanya Dixit",
        "pronouns": null,
        "jobTitle": "Forward Deployed Engineer",
        "employer": "Google",
        "bio": "Tanya Dixit is a Forward Deployed Engineer at Google, partnering with enterprise customers across APAC to ship production AI systems. Her work spans agentic AI, voice AI, and multimodal architectures, with deep focus on banking, financial services, and healthcare. She also supports Google's university partnerships program in healthcare AI. Based in Sydney, Tanya writes and speaks regularly on moving voice and agent systems from demo to production reliability.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/tanya-dixit.webp?v=bfa280f4",
        "url": "https://webdirections.org/ai-engineer/speakers/tanya-dixit/",
        "social": {
          "social_media": "https://www.linkedin.com/in/tanya-dixit-computer-vision/",
          "linkedin": null,
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "9490204f-a75e-4b8d-a7d9-4455c2f3aed9",
        "slug": "theodoros-galanos",
        "name": "Theodoros Galanos",
        "pronouns": "he/him",
        "jobTitle": "FDE @ APAC",
        "employer": "Nomic",
        "bio": "Theodoros Galanos has spent more than a decade at the intersection of AI, design and engineering. He currently leads the Forward Deployed Engineering in APAC region for Nomic.ai. Previously he was the Generative AI leader at Aurecon, a tier 1 engineering firm. Through The Harness, he explores the environments and evaluations that turn model capability into reliable engineering work.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/theodoros-galanos.webp?v=d8d231f7",
        "url": "https://webdirections.org/ai-engineer/speakers/theodoros-galanos/",
        "social": {
          "social_media": "@TheodoreGalanos",
          "linkedin": "https://www.linkedin.com/in/theodorosgalanos/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://theharness.blog/"
        }
      },
      {
        "id": "e79e5ac2-a80f-4e5f-85af-a32327058a98",
        "slug": "tinus-willemse",
        "name": "Tinus Willemse",
        "pronouns": "he/him",
        "jobTitle": "Executive Manager, AI & Data Science",
        "employer": "Quantium",
        "bio": "Tinus Willemse is an Executive Manager, AI & Data Science at Quantium, building Checkout AI, a conversational retail analytics product. He works on the agent architecture at its core — planning, running and narrating the analysis, the evals that keep it honest — and on making a non-deterministic system a reliable one.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/tinus-willemse.webp?v=554bb86e",
        "url": "https://webdirections.org/ai-engineer/speakers/tinus-willemse/",
        "social": {
          "social_media": null,
          "linkedin": "Great to see this is up. Could I ask that you please add my LInkedIn link to speaker page too? https://www.linkedin.com/in/marthinuswillemse/",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "d5a94f28-6080-42d8-a08d-37a0b9f0b383",
        "slug": "vighnesh-deshpande",
        "name": "Vighnesh Deshpande",
        "pronouns": null,
        "jobTitle": "AI Engineer",
        "employer": "Vivanti Consulting",
        "bio": "Viggy is a Consultant at Vivanti, an Australian Data & AI consulting firm, where he leads delivery of production AI systems for regulated enterprises across government, financial services and consumer health. He leads Vivanti's AI governance cadence and works across the delivery, governance and commercial sides of getting AI into production. He contributes to the Vivanti Academy and is managing the LangChain partnership within the company.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/vighnesh-deshpande.webp?v=7f297cf0",
        "url": "https://webdirections.org/ai-engineer/speakers/vighnesh-deshpande/",
        "social": {
          "social_media": null,
          "linkedin": "https://www.linkedin.com/in/vighnesh-deshpande-834734171/",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      },
      {
        "id": "93b19f77-cea1-44af-9a96-a6dc467a1c6a",
        "slug": "vivek-katial",
        "name": "Vivek Katial",
        "pronouns": "He/him",
        "jobTitle": "Engineering Lead, Applied AI",
        "employer": "Heidi Health",
        "bio": "Co-founder and Executive Director at Good Data Institute, connecting charities to socially-minded data professionals.\r\n\r\nEngineering Lead at Heidi Health, created and leading the Applied AI function. Also built the internal agents function to scale.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/vivek-katial.webp?v=4e7cad4d",
        "url": "https://webdirections.org/ai-engineer/speakers/vivek-katial/",
        "social": {
          "social_media": "https://www.linkedin.com/in/vivekkatial/",
          "linkedin": "https://www.linkedin.com/in/vivekkatial",
          "bluesky": null,
          "mastodon": null,
          "website": "https://www.vivekkatial.com"
        }
      },
      {
        "id": "415c32e8-33a4-4249-889f-3122353ce4c4",
        "slug": "vlad-gavrilov",
        "name": "Vlad Gavrilov",
        "pronouns": "he/him",
        "jobTitle": "Senior AI Engineer",
        "employer": "Heidi",
        "bio": "I’m a Senior AI Engineer with nearly a decade of experience across data science and machine learning. After starting in consulting and leading a government data science team, I returned to a hands-on role building production AI systems at Heidi Health. My work focuses on evaluation, harness optimisation, and model training, with experience across speech recognition, personalisation and agentic systems. I’m particularly interested in rigorous evals and building reliable AI that adapts safely to changing user behaviour.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/vlad-gavrilov.webp?v=e19d7c32",
        "url": "https://webdirections.org/ai-engineer/speakers/vlad-gavrilov/",
        "social": {
          "social_media": "X: @vlad__gav",
          "linkedin": "https://www.linkedin.com/in/vladislav-gavrilov-7aa9aab5/",
          "bluesky": null,
          "mastodon": null,
          "website": "https://www.linkedin.com/in/vladislav-gavrilov-7aa9aab5/"
        }
      },
      {
        "id": "790d28af-b37f-4d0f-9f06-75c093762f3a",
        "slug": "yulia-kuchina",
        "name": "Yulia Kuchina",
        "pronouns": "she",
        "jobTitle": "Staff AI Engineer",
        "employer": "Software at Scale",
        "bio": "Yulia Kuchina is a Staff AI Engineer at Software@Scale, currently working with Commonwealth Bank on rebuilding enterprise banking software with AI coding agents and developing ways to evaluate the code they generate.\r\n\r\nShe specialises in making LLM systems measurable and safe to change, with hands-on experience across evaluation, provenance, observability, citation grounding, confidence routing and production failure analysis. Previously, at LEAP legal software, she owned a production AI document pipeline and built evaluation and regression controls for assessing changes before release.\r\n\r\nHer background spans 12 years across frontend, full-stack and AI engineering in banking, legal technology, government and e-commerce. She is particularly interested in replacing impressive demos and intuitive prompt changes with evidence: production-derived test cases, versioned baselines and metrics that reveal whether a system has actually improved.",
        "photo": "https://webdirections.org/ai-engineer/images/speakers/yulia-kuchina.webp?v=c8b8bd91",
        "url": "https://webdirections.org/ai-engineer/speakers/yulia-kuchina/",
        "social": {
          "social_media": "https://www.linkedin.com/in/yulia-kuchina/",
          "linkedin": "https://www.linkedin.com/in/yulia-kuchina/",
          "bluesky": null,
          "mastodon": null,
          "website": null
        }
      }
    ],
    "sessions": [
      {
        "id": "d0aec6ab-fa84-40bc-ab43-ae1257707f4e",
        "title": "Half our handoffs are the bot's idea",
        "description": "About one in four conversations with WordPress.com's AI support bot ends with a person. In about half of those, the bot suggested it.\r\n\r\nThese are two different problems. In one, the user asks for a person. In the other, the bot decides on its own to stop. On paid plans, when a user asks for a person, we hand the conversation to our support team with as few extra questions as we can. The two cases fail in different ways and need different fixes, but the second one gets much less attention.\r\n\r\nThis talk is about the second one, when the bot decides to stop. Our bot checks whether a conversation needs a person before it answers, not only after an answer has failed. I'll explain why it works this way and what that costs us. Then I'll cover what the bot says when it stops, and what it passes on.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "04ac8e57-1b16-4a16-9639-d8d901a8a4c1",
            "name": "Adeline Yaw",
            "url": "https://webdirections.org/ai-engineer/speakers/adeline-yaw/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/adeline-yaw/"
      },
      {
        "id": "4e957eb1-903b-4536-bede-495379691e38",
        "title": "Autonomous agents need autonomous protection",
        "description": "Autonomous agents need autonomous protection. An agent that acts alone at 3am cannot wait for a quarterly pen test, a consultant's report, or a security engineer the team does not have. We started with the data. We built and published an open threat intelligence map, 6,330 documented AI security incidents connected to 217 attack techniques and the control that stops each one. Then we asked what protection looks like when it reads that map on its own and improves every time a new incident lands. The loop is simple to describe. Each incident, such as the Lovable apps deployed without row level security or the Hugging Face agent swarm, is classified against a technique. Each technique becomes a threat signature and each signature becomes a CI test, a static rule and a runtime check. The system reads what each agent can do, which tools it calls and which data it reaches, selects only the protections that apply and opens a pull request that a developer reviews like any other. Every incident added to the map adds protection for every agent that follows.\r\n\r\nThis talk covers the building blocks and the five rules we learned by building them. Performance comes first, because developers bypass slow checks. Security has to arrive inside the workflow developers already use. The first run has to fix something real, or nobody runs it a second time. Enforcement has to be deterministic. We would never accept a firewall that is probably right, so the model reasons and a policy engine decides. And the check needs memory. Reading customer data is fine. Sending an email is fine. Doing both in one session, to a domain nobody recognises, is not. We will demo the full loop in under four minutes, from a new incident on the map to a merged pull request. Three pieces are open for you to reuse: the incident map, the signature format and the demo repo. The pattern is yours to adopt.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "2d40fe39-54e0-4faa-b123-ea65d94337da",
            "name": "Adesh Gairola",
            "url": "https://webdirections.org/ai-engineer/speakers/adesh-gairola-2/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/adesh-gairola-2/"
      },
      {
        "id": "492b6ae7-300e-4c90-8b72-9ecb84c0bba4",
        "title": "Your spaghetti code has an invoice now: what complexity does to coding agents",
        "description": "We've told engineers for decades that high complexity makes code harder for humans to reason about. It turns out it makes code measurably more expensive for agents too. Unlike human frustration, this shows up directly on your API bill. \r\n\r\nI show and discuss my research based around a series of controlled benchmarks on the same codebase with different levels of cyclomatic complexity, which shows that Claude Code uses up to 50% more tokens on the identical, simple feature requests, in a repeatable and controlled environment (i.e. 2 functionally identical branches, same codebase, caching disabled, same change sizes, etc.)\r\n\r\nI'll walk through the benchmark design (and its limitations), the numbers, and what they imply for teams adopting agentic coding at scale: refactoring now has a directly quantifiable ROI, complexity linting belongs in your agent-readiness checklist alongside CLAUDE.md files and MCP servers, and \"tech debt\" arguments you've been losing with vibes can now be won with direct financial impact.\r\n\r\nI expect people to leave the talk with a replicable methodology for benchmarking your own codebase, tools to measure complexity across a polyglot estate, and a budget-shaped argument for the refactor they've been putting off.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "95984cba-c326-49cb-a9ff-ab0a1f4df719",
            "name": "Artem Yakimenko",
            "url": "https://webdirections.org/ai-engineer/speakers/artem-yakimenko/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/artem-yakimenko/"
      },
      {
        "id": "30bf1873-c62f-4de6-a8a9-6d0a79f87b41",
        "title": "209 ports in six days, in a language the model barely knew",
        "description": "Jolt and jank are two young Clojure implementations, one running on Chez Scheme and one compiling to native code through C++/LLVM. Between them they have almost no public code for a model to have learned from. Ask Claude for jank and it confidently writes JVM Clojure, and none of that compiles. Over six days in July I ported 209 of raylib's official C examples to jank anyway, across 194 commits. Then again in Jolt, and once more in JVM Clojure over Panama. That's 422 working examples across three runtimes, every one of them driving the same C game library.\r\n\r\nThis talk is about the harness, because the model was the same model everyone in the room already uses. Everything around it is what changed. Three pieces do most of the work. The compiler becomes the eval, so \"Mismatched 'if' branch types 'Color' and 'nil'\" is a grading signal, and every error the model hits once turns into a rule it never hits twice. The trap catalog is context engineering with receipts: a 240-line AGENTS.md where each rule cites the committed example that proves it. Verification has to be cheap and unattended, so a graphics program that opens a window gets an alarm, a screenshot and an exit code an agent can read on its own.\r\n\r\nThe failures get airtime too. The model never derived the AArch64 rule that lets a 24-byte struct pass as a pointer, and no prompt got it there. You'll leave with all four artifacts, and a way to size up any codebase your agent has never seen.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "25d01937-f1ce-4199-bae3-15119f9a87f1",
            "name": "Burin Choomnuan",
            "url": "https://webdirections.org/ai-engineer/speakers/burin-choomnuan/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/burin-choomnuan/"
      },
      {
        "id": "e76575d9-d8fb-4dac-abaa-b09afe38ee86",
        "title": "Lessons from Economics for the Human / Agent Software Workforce",
        "description": "AI is reshaping software engineering, creating uncertainty about how engineering roles will change as increasingly capable systems take on work that was once performed by people. While no one can predict the future with certainty, history offers examples of how technical innovation has transformed work in other industries and provides useful ways to think about the current transition.\r\n\r\nThis session explores what economics can teach us about the future of software engineering and agent-assisted development. We will examine how similar periods of technological change have affected work, productivity, and workforce composition, and discuss which lessons are applicable to software engineering today. We'll also look at how engineering organizations can use insights from their own codebases to better understand the impact of AI on engineering work, rather than relying on assumptions or conclusions drawn from broader industry discussions.\r\n\r\nNo background in economics is required. We will focus on minimal theory and practical concepts that help engineers and leaders interpret ongoing changes and make informed decisions.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "dbb3f2e9-1c47-4553-b239-aafed793cfbc",
            "name": "Daniel Nadasi",
            "url": "https://webdirections.org/ai-engineer/speakers/daniel-nadasi/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/daniel-nadasi/"
      },
      {
        "id": "01d0a8c8-8dec-419e-b800-5d8e2d02b680",
        "title": "AI Janitor: Making Architecture Review Executable for Coding Agents",
        "description": "AI coding agents accelerate code generation, but they also accelerate architectural drift, regressions and false confidence. While building a multi-service production platform with only two of us, I found that a normal human review loop could not keep up with the volume of AI-generated change. I needed some of the boundaries and feedback a platform team would normally provide, so I began turning repeated review comments, architectural decisions and past failures into checks that agents could act on themselves.\r\n\r\nThis is a practical case study focused on two mechanisms. First, custom ESLint rules whose messages tell an agent what went wrong, why it matters and how to fix it, giving the agent enough information to correct its own change before human review. Second, fail-closed verification: what I learned from a guardrail that stayed green while silently doing nothing, and how I changed the system to prove that its checks actually ran. \r\n\r\nI’ll show where these controls work, where they become lint theatre and which decisions still need a human. The point is not to encode every architectural decision or replace review. It is to make the repetitive, objective parts executable and reserve human attention for judgement. Attendees will leave with a method for identifying repetitive review work and turning the right parts into fast, mechanical feedback—without building a huge internal platform or trusting AI output by default.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "842323fe-f341-4947-9fa0-fec7ddc1893f",
            "name": "Dave Currie",
            "url": "https://webdirections.org/ai-engineer/speakers/dave-currie/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/dave-currie/"
      },
      {
        "id": "3ea45df7-829b-4a3e-8612-7b26ee91daa6",
        "title": "The Elephant and the Goldfish: Architecture Patterns for Cutting 70% of Agent Token Costs in Production",
        "description": "LLM providers sell you a 2-million-token context window like an elephant that never forgets. If you actually build production agents that way, your latency explodes, your retrieval drifts, and your CFO will shut you down in 90 days. In production, the best agents think like elephants, but operate like goldfish.\n\nDetails: While frontier models offer million-token context windows, treating autonomous agents like an \"elephant\" that carries entire conversational histories and raw tool outputs into every turn leads to compounding inference costs, bloated time-to-first-token (TTFT), and severe retrieval distraction. This talk introduces the **Elephant and Goldfish** architecture—an open engineering pattern that decouples persistent system state (external session stores and vector indexes) from ephemeral inference context, strictly bounding working memory to lean, task-scoped execution frames. Grounded in reproducible benchmarks across multi-turn agent tasks, we explore four public, cost-cutting optimization layers: enforcing **Subagent Boundary Isolation** using strictly-typed JSON schemas (<500 tokens) to kill quadratic transcript inheritance; applying **upstream semantic context pruning** (e.g., cross-encoder sentence masking and extractive chunk filtering) to drop 40%–70% of retrieved RAG bulk prior to tokenization; implementing **multimodal triage** that prioritizes structured accessibility trees and keyframe diffs over brute-force video/pixel streaming; and shifting **evaluations and judging** from monolithic frontier models to decomposed, pointwise rubrics on lightweight flash models. Attendees will walk away with an experimental framework and concrete, open-source-friendly code patterns to audit token leakage, benchmark context compression, and slash agent operating costs without degrading task success rates.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "45acc7e8-f224-4635-855e-924390d1604f",
            "name": "Tanya Dixit",
            "url": "https://webdirections.org/ai-engineer/speakers/tanya-dixit/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/tanya-dixit/"
      },
      {
        "id": "b5c982d7-8c5c-4fb3-a069-43590a0c7500",
        "title": "Building AI for students who can't tell you when it's wrong",
        "description": "Teachers in Australian specialist schools spend two hours or more preparing a single lesson, then adapt it again for every ability group in the room. Many are already pasting student details into consumer chatbots to keep up. That shadow AI is ungoverned, inconsistent and potentially harmful.\r\n\r\nWe built IRIS with the Alannah & Madeline Foundation, a Trusted eSafety Provider endorsed by the eSafety Commissioner, to change that. It adapts Australian curriculum-aligned lessons for learners with complex needs, running on Azure OpenAI behind a governance layer designed for a child-facing product. Six educators across five schools co-designed it. It has been through UAT and is built for national rollout across every Specialist School and Special Development School in Australia: 23,000 educators, supporting over a million students who require educational adjustment due to disability.\r\n\r\nOne constraint shaped every technical decision. The recipient of our output is often a non-verbal student working several year levels below their age. They cannot look up from a worksheet and say it's wrong. Take away that safety net and the usual AI engineering defaults stop being safe.\r\n\r\nFive we had to re-evaluate, and the trade-offs behind each:\r\n\r\nWe stopped trusting the model for anything checkable. Six deterministic post-processors run after every generation and every chat turn, unit tested without ever calling an LLM. Section durations must sum exactly to the lesson length, because the school bell will always ring.\r\nWe show educators precisely which sections the AI touched, and make them acknowledge it before export.\r\nOur privacy guard has three states, because a check that failed silently would look identical to a clean pass.\r\nWe deleted health terms from PII detection. A tool that adapts lessons for autistic students should not warn teachers against writing \"autism\".\r\n\"Lesson\" is tagged as a surname in the NLP lexicon we started with, so the warning fired on nearly every prompt in a lesson planning tool. Student names belong in a classroom. We had to warn without getting in the way.\r\n\r\nIncludes a live demo, and reflections on the things an LLM still can't quite get right.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "5fa70ed5-de42-4062-808d-ede98a1e6c02",
            "name": "Ez Herrmann",
            "url": "https://webdirections.org/ai-engineer/speakers/ez-herrmann/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/ez-herrmann/"
      },
      {
        "id": "5b91ef58-e286-43bf-b20d-7f721ab6aa03",
        "title": "Dependency hell is back. This time it's your agent's config.",
        "description": "Everyone has an agentic harness now. Ours is not special. What nobody has solved is running one across a whole team without every engineer drifting into a private setup.\r\n\r\nWe run Claude Code across 50 engineers and dozens of client codebases at 10xlabs. Within a few months, the drift was everywhere. One engineer's agent was excellent. The next one ran a config from three months ago. Skills have no version field, so \"the frontend skill\" meant something different on every laptop. A plugin from a marketplace nobody had vetted turned up inside a client project. Then a Claude Code update changed the settings schema, and unpinned machines quietly fell back to default behavior. \r\n\r\nWe solved this exact problem 15 years ago. It was called dependency hell, and the fix was lockfiles, semver, and registries. So we built the same thing for agent configs. Famulus is a dependency manager for harness components. Its manifest is called a grimoire. It pins skills, plugins, hooks and MCP servers to exact versions, detects drift across machines, and gives the team lead a say in what a project is allowed to load.\r\n\r\nThis talk is the war story with numbers. How far have the configs drifted? What broke when we pinned them? Why marketplaces, Microsoft's APM, and skills.sh solve distribution but not team identity or drift. I'll demo Famulus live on a real project and release it as open source on stage.\r\n\r\nThen the honest ending. Pinning a skill to a version doesn't tell you whether the agent ever reads it. That is a different problem, and it's still open.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "b8c00501-6f53-4e78-9335-811244b0f3bd",
            "name": "Jack Rudenko",
            "url": "https://webdirections.org/ai-engineer/speakers/jack-rudenko/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/jack-rudenko/"
      },
      {
        "id": "617e1b26-44ee-4a1c-b4ae-3980e570e425",
        "title": "From Prompt Rules to Structural Guarantees: The Harness Behind a Production Analytics Agent",
        "description": "Checkout AI answers open-ended questions about retail sales data in natural language. It plans,\r\ncalls analytics tools over MCP, executes Python in a sandbox, and returns a written analysis with\r\ncharts. In a system like this, failure is rarely a crash: the chart renders, the prose is fluent, and\r\nan incorrect figure reaches a decision-maker unchallenged.\r\n\r\nPrompts define the agent’s behaviour but cannot enforce it. Enforcement is engineered into\r\nthe layers around the model, at three points in the lifecycle: a bad answer is caught before it\r\nreaches the user, a bad change before it reaches the codebase, and a bad build before it reaches\r\nproduction.\r\n\r\nBefore the user. The agent does not author its own charts. It calls typed tools whose outputs\r\nare validated against data contracts shared with the renderer, so a malformed visualisation cannot\r\nbe constructed, let alone displayed. Deterministic validation sits between analysis and synthesis,\r\nand every figure in the narrative is checked against the data the user can actually see.\r\n\r\nBefore the merge. Fourteen evaluation metrics gate development: golden-case metrics run\r\non every prompt and plan change, while metrics that need no expected answer score live produc-\r\ntion traffic. Production failures are replayed in a local development harness that reproduces the\r\nfull agent stack, and are captured as golden cases before a fix is written.\r\n\r\nBefore production. Immutable, commit-tagged builds, a single source of deployment truth,\r\nand supply-chain scanning ahead of every merge.\r\n\r\nNone of these layers required a better model. We close with the case for treating the harness,\r\nnot the model, as the primary engineering surface of a production LLM agent.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "dd6d963a-4992-4da0-937d-0ce47c3e0f50",
            "name": "Jiggy Kakkad",
            "url": "https://webdirections.org/ai-engineer/speakers/jiggy-kakkad/"
          },
          {
            "id": "e79e5ac2-a80f-4e5f-85af-a32327058a98",
            "name": "Tinus Willemse",
            "url": "https://webdirections.org/ai-engineer/speakers/tinus-willemse/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/jiggy-kakkad/"
      },
      {
        "id": "a5764637-5374-4a78-933d-db1a88c06255",
        "title": "Oops I Hacked It Again: Six Months of Red Teaming Our AI Agents",
        "description": "In March I onboarded our first team member dedicated to AI Red Teaming. That was a month before the announcement of Claude Mythos Preview, and the open letters that followed from APRA in April and ASIC in May (Australia’s prudential and conduct regulators) warning organisations to act by shoring up their AI risk management and security practices. We moved before these events because we could see the impending and growing risks, and that specialised skills are needed to keep our customers safe.  \r\n\r\nSix months and eight deployments later we have broken in repeatedly. We have great data scientists and AI engineers, but the reality is that the skills and mindset needed to build robust AI systems are different from what's necessary to defend them. The surprising thing was the multitude of failure modes we observed, across prompts and architecture, demonstrating the many gaps developers must intentionally think about.\r\n\r\nThis talk covers what we tested and how, including the process failures we didn't expect, and what we've built to prepare for the future. Leaders will leave with practical steps to start building their own AI red team and defending their AI systems.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "f0ce70c7-a741-47eb-8b17-d848e84143ad",
            "name": "Jon Shen",
            "url": "https://webdirections.org/ai-engineer/speakers/jon-shen/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/jon-shen/"
      },
      {
        "id": "91156333-4d72-4dc1-864f-caa493e455c8",
        "title": "Deliberate Holes Only",
        "description": "Making an AI agent vulnerable to exactly one attack is harder than making it completely secure, let alone doing it six different times.\r\n\r\nIn this talk, I'll describe the process of building an agentic AI capture the flag challenge (https://owngoal.lorikeetcx.ai), in which each level has a more sophisticated agent than the last, always with one hole left for a player to find. Hundreds of people played the game, but only around 30 were able to complete it successfully.\r\n\r\nI'll cover:\r\n* choosing each level's vulnerability, from trivial prompt injection through to subtle social engineering, and using evals to prove nothing else got through;\r\n* where the built-in defences from frontier labs helped, and where they got in the way;\r\n* rewriting agent responses on the fly to hint solutions to players, and the security tradeoff that creates;\r\n* shipping in a real environment: last minute model deprecations and a full retheme from marketing, along with the evals that made both survivable.\r\n\r\nAttendees will leave knowing which types of vulnerabilities to watch out for in their own public-facing agents, and how to build evals to make sure there are no unexpected holes in them.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "5cd0ac5a-618e-4d7c-a894-cd93bbbf0c14",
            "name": "Karla Burnett",
            "url": "https://webdirections.org/ai-engineer/speakers/karla-burnett/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/karla-burnett/"
      },
      {
        "id": "e2da93b1-7d7e-4a25-a8d1-6d55d3595ef4",
        "title": "How we introduced LLMs into credit decisioning",
        "description": "Over the past nine months we've built and deployed an AI-assisted credit assessment system that's now used in production to help credit assessors process home loan applications. More importantly, it was signed off by our founders and risk underwriters.\r\n\r\nStarting from a simple assumption, the same application should receive the same outcome regardless of who assesses it, we treated the problem as an engineering challenge. Which parts of the assessment could be made deterministic? Where could an LLM safely assist? And how do you design the system so it doesn't depend on the next model release to become safe?",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "7a3ef210-0b37-498c-8285-ea01a6427080",
            "name": "Kevin Nguyen",
            "url": "https://webdirections.org/ai-engineer/speakers/kevin-nguyen/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/kevin-nguyen/"
      },
      {
        "id": "672b2bfb-1045-4f51-9a27-88693840dff9",
        "title": "The guess never becomes a fact - provenance-governed, model-free memory for LLMs",
        "description": "Long context solves within-session coherence, not cross-session persistence. Once an assistant starts storing memories, a quieter failure appears: its own inferences can be written back, retrieved in later sessions, and presented as user facts. Hallucinations then compound over time. We encountered this while building EDN, TensorPRO's external memory middleware. Our response was intentionally less \"smart\": make provenance structural, and keep generative models out of the retrieval path. Every record carries one of seven provenance classes, including user-stated, document-verified, and AI-generated. A single write chokepoint enforces those labels. At session start, EDN assembles a provenance-gated Session Initialization Package (SIP) using cosine-similarity retrieval. AI-generated records cannot enter factual sections; if one crosses that boundary, the context builder fails loudly. On LongMemEval-S (sealed evaluation, three runs, GPT-4o judge), this design increased answer accuracy from 45.5% with full-history context to 85.4%, while using 9.7× fewer input tokens. SIP assembly p95 was 30–35 ms. Ablations were equally instructive: storing complete turn pairs (round-level records) was the single largest gain, at +22.5 points, while diversity re-ranking (MMR) added no significant gain, so we removed it from the read path. Those results changed what we optimised: not ever more sophisticated retrieval, but governance over what is allowed to become memory.\r\n\r\nThis talk walks through the failure mode, architecture, evaluation, and trade-offs, including what we would change next. Attendees will leave with a practical framework for deciding when provenance over prediction and determinism over \"smart\" retrieval are worth the rigidity, and when a probabilistic pipeline is the better engineering choice.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "f01a5cf9-b41e-4f3a-939d-b0d097795af9",
            "name": "Jade Xin",
            "url": "https://webdirections.org/ai-engineer/speakers/kexuan-xin/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/kexuan-xin/"
      },
      {
        "id": "64bb7f18-2058-4521-a4ae-256ecc8976e3",
        "title": "Sovereignty Is an Inference Problem",
        "description": "**AI sovereignty isn’t about where the GPUs sit. It’s about who can switch the intelligence off.**\r\n\r\nAustralia is spending billions on sovereign AI infrastructure. But onshore compute running models subject to another country’s export controls is still a dependency. Owning the building isn’t owning the model.\r\n\r\nEnterprises have the same blind spot. Running one closed model through two providers looks like redundancy — until both depend on the same weights and export regime. That’s two doors to one room.\r\n\r\nThis talk connects national AI strategy with production architecture: how to build an open-weight fallback, why model provenance matters as much as infrastructure, and how enterprise demand can reshape what hyperscalers build.\r\n\r\n**The takeaway:** sovereignty isn’t a grant we wait for. It’s a demand signal.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "df3e8324-5c0f-4726-bc3c-6f8908e4d2f3",
            "name": "Leo Borges",
            "url": "https://webdirections.org/ai-engineer/speakers/leo-borges/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/leo-borges/"
      },
      {
        "id": "cb7df392-f655-419d-97be-f9be558f2560",
        "title": "Office-Brain and the Night Shift: What Happens When You Turn Off the Meter",
        "description": "\"Good enough\" AI now runs on a reasonably beefy laptop. Qwen3.8-27B benchmarks at an Intelligence Index of 52 - roughly the best money could buy back in February - and inferences happily on consumer kit. When tokens are minted on your own machines, the meter goes away, and a different way of working becomes economic: batch processing returns. I call it the night shift - hand an agent a long-horizon task at bedtime, let it grind through the wee hours at 13 tokens a second, then have it write the RESUME.md it needs to pick up the thread the next evening.\r\n\r\nSince Qwen3.8-27B dropped, I've run night shifts across a five-machine home fleet - from a 24GB MacBook Air to a decade-old PC - under a three-file trail architecture (TASK.md, AGENTS.md, RESUME.md) that lets smaller-context minds recover from almost anything, including a hard power cycle. This talk is the field report: the auth hang that lost a night, the context overflow, the API that returned 200 OK while writing 127 hollow documents (rule: read back every write), the trade of a smaller quant for crash-resistance, and the first clean eight-hour hold. The always-on agent becomes an \"office-brain,\" doing consolidation and reconciliation while the humans sleep.\r\n\r\nThe takeaway: as inferencing arrives on most of our devices, hold time is the key metric, and it's about to matter everywhere. Live demo: a laptop on stage, mid-task, logs streaming.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "8d0ce144-235a-4f85-b514-690ad8b4c62a",
            "name": "Mark Pesce",
            "url": "https://webdirections.org/ai-engineer/speakers/mark-pesce/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/mark-pesce/"
      },
      {
        "id": "b4169c04-2f5d-4992-9921-8e21cd5b1888",
        "title": "Building Agentic Memory for an AI SOC: Why Our Semantic Cache Was the Wrong Answer",
        "description": "We run a fleet of AI agents against production security detections at Atlassian. One tunes noisy detection rules, another reviews new ones, and more are coming for alert triage and hunting. Every one of them started blind and stateless, re-deriving the same context from Jira tickets, a Databricks lake, a rule repo and ATT&CK, or drowning in a raw dump of all of it. Nothing they learned survived the session, so they repeated each other's mistakes and contradicted each other.\r\n\r\nOur first design was a semantic cache. It was wrong, and working out why reshaped everything. A semantic cache is keyed on input embedding and returns a cached conclusion, which in a domain with consequences means serving a stale security verdict to a look-alike question. What we built instead is an entity-keyed, facts-only store, each agent reads a bounded, per-action slice before it acts, then writes facts back. Conclusions are never cached. They are re-derived every run, deliberately.\r\n\r\nI will walk through the schema, the four rules we enforce (facts only, provenance on every row, derived and never authoritative, bounded slices instead of whole-history blobs), and the point where native prompt caching stops paying and a store starts.\r\n\r\nThen the measurements, on real tickets with human ground-truth labels that analysts produce as a byproduct of triage, input tokens per run, eliminating a 900-second lake query, false positives removed with zero true positives lost, and how consistent two agents stay when they share one memory instead of guessing separately.\r\n\r\nThen what broke: entity resolution, write contention, memory poisoning, and the crossover where memory costs more than it saves.\r\n\r\nYou will leave with the schema, the eval design and the four rules, all of which transfer to any agent fleet that needs to remember something.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "1af19cf3-d7d1-4046-b745-d9118603e501",
            "name": "Mukesh Singh",
            "url": "https://webdirections.org/ai-engineer/speakers/mukesh-singh/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/mukesh-singh/"
      },
      {
        "id": "1dbab403-8760-4dc6-a251-9cf9df96a415",
        "title": "From Documents to Defensible Evidence: Rethinking RAG for Audits",
        "description": "Audits impose strict requirements for evidence, traceability and expert judgement. Each entity being audited supplies a new body of evidence, made up of manuals, records, tables, diagrams, scans and multilingual documents. Assessment against an audit standard follows defined guidelines and requires precise citations. A plausible answer has little value if an auditor cannot verify its source, understand why the system selected it or correct the result before making a compliance decision.\r\n\r\nI will walk through the approaches we tried, what failed, what improved and how those lessons led to a workflow that balances AI assistance with expert judgement. The key shift was to separate evidence selection, compliance assessment and expert review, then measure each stage independently. From that work, I will share a practical method for designing document-based AI systems: define the outcome, measure the factors that matter, understand the task and document constraints, then choose the architecture. I will cover ingestion cost, latency, evidence quality, citation quality and where human review adds the most value. We tested the system against completed audits. It reached up to 94% accuracy on compliance assessments, and auditors approved more than 90% of the evidence mappings they reviewed. The system now runs in production for a firm that conducts more than 70 audits each year. The same method applies to other regulated document workflows where people must verify the evidence before acting.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "fa084c3b-8e0a-46c0-a28c-297f734e3173",
            "name": "Sharat Madanapalli",
            "url": "https://webdirections.org/ai-engineer/speakers/sharat-madanapalli/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/sharat-madanapalli/"
      },
      {
        "id": "cba7530b-b2b2-473e-b563-4b7394222b68",
        "title": "Your Coding Agent Is Fast. Your Codebase Is the Bottleneck.",
        "description": "AI agents can write code faster than teams can land it. Jira’s frontend codebase contains over 15 million lines of code, supports thousands of contributors, and doubles in size roughly every two years.\r\n\r\nAt this scale, verification, slow CI, defense against legacy patterns had become the bottleneck. AI is great at generation of code - but design? Architechture? Data flow?\r\n\r\nAny small change could affect a large part of the dependency graph. A small change could build, check (lint/typecheck), and test hundreds of thousands of files - even if completely unrelated to the change. Increasing the number of agent-generated pull requests increases queueing, bugs, slow feedback loops and rework without improving delivery speed. \r\n\r\nWe addressed the underlying problem: codebase coupling. We invested in a incremental static-analysis platform that parses changed files once and stores facts about their dependencies, structure, and behaviour in a shared cache. We overhauled CI tools to query this data to identify the code affected by each change. Their cost now scales with the change rather than the repository.\r\n\r\nThis platform powers affected-test selection, dependency analysis, caching, automated migrations, and architectural checks. Through an initiative called Thunderstone, we used these capabilities to restructure hundreds of thousands of files to coupling while teams continued shipping features. \r\n\r\nThis talk covers the platform’s architecture, its adoption costs, and the decisions we made, and would change! It also examines where coding agents helped with large migrations, where they failed to reason reliably about the whole repository, and why deterministic program analysis remains an essential partner in an agentic development workflow :D Thanks :)",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "14d4a7be-0922-4b9e-92da-b817fca99f04",
            "name": "Shrey Somaiya",
            "url": "https://webdirections.org/ai-engineer/speakers/shrey-somaiya/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/shrey-somaiya/"
      },
      {
        "id": "55fb0cbc-f032-4c0a-840a-f53cc92b6372",
        "title": "Building a Security Agent: Model choice, Harnesses and Evals",
        "description": "In this presentation, I’ll show how we set out to give developers useful security feedback on every pull request in under three minutes. The benchmark results and methodology are published here: https://docs.damsecure.ai/blog/pr-review-security-benchmark-update/. It has become our most-cited research to date. We initially expected to compare models using established security benchmarks, but discovered that those tests made agents look more capable than they were. In our testing, frontier models recognised canonical vulnerable repositories such as OWASP Juice Shop, while markers intended for traditional scanners gave them clues that real pull requests would not contain. We needed private fixtures with known ground truth, realistic code and no familiar answers. We also needed to evaluate the complete review system, including the model, prompts, tools, context and stopping rules.\r\n\r\nWe first built synthetic replicas of open-source applications. They gave us controlled tests, but producing realistic codebases was too slow. We switched to reversing real CVE patches and replaying the vulnerabilities into matching open-source commit histories, then built a model- and harness-independent eval suite that records findings, files opened, searches, tool calls, runtime and cost. I’ll demonstrate the approach by running a synthetic pull request containing an IDOR through an open-source harness, revealing the hidden ground truth and inspecting the agent’s trace. Our benchmark covered ten planted access-control bugs, with five runs per model. Recall ranged from 30 to 100 percent and cost ranged from about $0.04 to $4.25 per pull request. I’ll close with what we would change: start with replayable workloads, capture traces from the first run, and choose metrics based on the failures developers care about.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "f4255d66-a440-4a28-9eb0-649df4db680c",
            "name": "Simon Harloff",
            "url": "https://webdirections.org/ai-engineer/speakers/simon-harloff/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/simon-harloff/"
      },
      {
        "id": "673eb9f5-5f4c-4ce5-b7ae-26a4861c10c1",
        "title": "The Model Wasn't the MOAT: How 10 Design System Engineers Turned Platform Knowledge into Enterprise AI",
        "description": "Most enterprise AI strategies begin with models, central AI teams, and a list of potential use cases. We started somewhere else: with a platform organisation that already shaped how more than 2,000 frontend engineers built software.\r\n\r\nA team of 10 engineers working across design systems, internationalisation, and accessibility had something a general-purpose model did not: deep organisational context, trusted workflows, enforceable quality standards, and a distribution path into nearly every product team.\r\n\r\nWe first transformed our design system into an AI-native platform. Structured content, AI skills, and Model Context Protocol capabilities gave developers and coding agents access to the same product knowledge. This raised the quality floor for AI-generated UI by helping generated code align with our components, styling conventions, accessibility expectations, and platform requirements. \r\n\r\nIn controlled frontend development tasks, measured implementation accuracy improved by 52% and task completion time fell by 34%. We then introduced a new operating model that replaced slow manual review gates with distributed ownership and automated quality enforcement. Cross-product component development that previously took months could move in days.\r\n\r\nWe applied the same strategy to localisation, a $3 million annual cost base with volume growing 272% year over year. A 45-day evaluation compared five AI pipelines with our existing vendor across 22 languages. The findings changed our build-versus-buy decision and created a pathway to up to $2 million in annual savings. But AI translation only works if the source code is clean. AI translating messy source just produces the wrong translation faster. So we ran two bets that reinforced each other: AI drafting every translation with human review, and AI-powered linting that fixed the source before it ever reached translation. The strategic advantage was not simply lower cost. It was owning the feedback loop: every human correction improved prompts, glossaries, style guidance, and future translation quality.\r\n\r\nInternationalisation exposed another platform opportunity. We resolved more than 20,000 issues across the codebase by combining codified domain rules, automated remediation, and platform-level enforcement. What had previously looked like an unbounded manual cleanup effort became a repeatable engineering system.\r\n\r\nAccessibility presented a similar scaling challenge. With more than 10,000 identified issues, no central team could remediate the backlog manually. We began building an AI accessibility remediation factory using pattern-based batching, orchestrated agents, automated validation, and feature-gated pull requests. The goal was not only to clear a backlog, but to move accessibility intelligence earlier into the development lifecycle so developers and agents could prevent issues while software was being created.\r\n\r\nThis talk presents a practical leadership and architecture framework for identifying where platform teams create genuine AI leverage, deciding what to centralise, choosing when to build or buy, and turning experiments into durable organisational capability.\r\n\r\nThe model was not the moat. The moat was the organisation's ability to turn domain knowledge, feedback loops, quality standards, and distribution into systems that AI could reliably use.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "2358d0c6-848e-494d-a0d2-36b1c52e462e",
            "name": "Sudharsanam Narasimhan",
            "url": "https://webdirections.org/ai-engineer/speakers/sudharsanam-narasimhan/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/sudharsanam-narasimhan/"
      },
      {
        "id": "b1d25aeb-2ffc-4740-ad61-e72177c8e0ed",
        "title": "Classifiers are dead. Long live classifiers!",
        "description": "Training a classifier used to mean collecting labelled data, choosing an architecture, training a model, evaluating it, and deploying it. Today, for a surprising number of problems, you can replace most of that with a prompt.\r\n\r\nAt Stile, we’ve been doing exactly that: using multimodal generative models to classify handwritten marks on scanned worksheets. In this talk, I’ll explore where generative models make surprisingly good classifiers, and where the abstraction starts to break down. We’ll look at why asking a model to make a binary decision performed badly on messy student handwriting, while asking it to describe what it saw and applying deterministic rules afterwards worked far better.\r\n\r\nThe problem is that some inputs are genuinely ambiguous. A cross might mean “selected” when every other box is empty, but lose to a tick elsewhere. A crossed-out tick means “not selected”... unless the student then circles the box to select it again. Sometimes the right answer is simply to ask the teacher who knows that student’s handwriting. To safely automate the obvious cases, we therefore need to identify the ones near the decision boundary. And that’s where generative classifiers get awkward.\r\n\r\nTraditional classifiers give us scores over a fixed set of classes that we can evaluate and calibrate. Generative models give us text, and sometimes token log probabilities (if the provider exposes them at all). Are those meaningful classification probabilities? What happens with structured outputs, multi-token labels and constrained decoding? And if they aren’t, how do we decide when to trust the model and when to ask a human?\r\n\r\nThe goal isn’t to build an LLM classifier that never gets things wrong. It’s to build one that knows when its answer is uncertain enough to hand back to a human.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "429b809f-62da-4eea-85b0-f61b0030e4f7",
            "name": "Charli Posner",
            "url": "https://webdirections.org/ai-engineer/speakers/charli-posner/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/charli-posner/"
      },
      {
        "id": "63a27d12-476c-46d9-b8ae-69fb922a2f59",
        "title": "Tools Before Autonomy: What 200,000 Tool Calls Taught Us About Agentic Evals",
        "description": "Production AI problems are rarely observable through deterministic CI checks. They surface as user feedback, bad LLM-as-a-Judge scores, alerts, or even a vague sense that something is wrong. The evidence that explains them is scattered across traces, prompts, content, documentation, support tickets and logs.\r\n\r\nAt Canva, we treated that evidence as a graph and built an internal MCP layer that lets agents traverse it on demand, with every claim retaining a path back to its source.\r\n\r\nOur teams use these tools to find production bugs; iterate on prompts and content; correlate text across Java code, Langfuse prompts and Contentful entries; run agentic end-to-end tests; and find examples of specific behaviours buried in production traces. 70 people have made over 200,000 tool calls, and one $8.87 session uncovered three reproducible bugs in production.\r\n\r\nRather than building a graph database or eval agent, we focused on tools that mirror the process of a good *human* investigator: treating the initial signal as a lead rather than a concrete fact, searching across systems and providers, retrieving detail and domain knowledge only when needed, tracing findings to their sources, exposing gaps, and identifying where to look next.\r\n\r\nTwo constraints shared by LLM systems become critical in longer-running agents: every problem has a context floor, but every token fights for attention. Too little context leaves an agent guessing; too much dilutes attention, raises cost and accumulates across the task. The balance is what we call Minimally Sufficient Context: the smallest set of information that still contains everything an agent needs to reach a justified conclusion and a human needs to verify it.\r\n\r\nUsage has also exposed clear limits: agents miss inaccessible evidence, conclude too early and struggle to judge real-world impact. This talk turns those lessons into a practical build order for agentic evals: make evidence navigable, test tools with people on real problems, observe how they are used, and automate only what proves useful.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "757ebe2d-5644-4d2a-863c-b4dab4cae7fd",
            "name": "Dean Soste",
            "url": "https://webdirections.org/ai-engineer/speakers/dean-soste/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/dean-soste/"
      },
      {
        "id": "c4f49d68-b118-4813-a69e-3725799394b0",
        "title": "Query before mutation: guardrail patterns for agents that touch production infrastructure",
        "description": "Covered in the session:\n\n- Why plan-and-apply was always a human safety mechanism: a person reads the plan, notices something is off, and stops the apply - and why this quietly disappears the moment an agent is the operator\n- Query before mutation as the foundational pattern: requiring agents to establish current state from the live environment before any write, and why reasoning from stale context is the root cause of most agent-inflicted damage\n- Idempotent assertions over imperative commands: expressing changes as desired state the agent can safely retry, rather than one-shot operations that compound when repeated\n- Policy gates that work without a human: encoding the judgement a reviewer would have applied as machine-checkable rules at the interface, not in a document\n- Blast radius controls: scoping agent credentials and mutation surfaces so the worst case is bounded, including read-only by default with explicit mutation grants\n- What went wrong on the way here: real examples of agent behaviour against live cloud environments that motivated each pattern, including failures the patterns would not have caught\n- A demo of an agent operating against real infrastructure with these guardrails active, including a blocked mutation and the recovery path\n\nAttendees will leave with a practical guardrail checklist for any agent that can modify infrastructure, ordered by which controls give the most protection for the least effort.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "17836589-1ad9-4722-a55e-787c5dca9e73",
            "name": "Jeffrey Aven",
            "url": "https://webdirections.org/ai-engineer/speakers/jeffrey-aven/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/jeffrey-aven/"
      },
      {
        "id": "6bedd5cf-abc1-4868-ba04-c25c92580bd2",
        "title": "Give Every Agent a Flight Recorder",
        "description": "Agent teams should not need to file a ticket with a central evaluation team just to learn whether a new prompt, model, or tool made their agent better. At Zendesk, we developed and deployed a trace-first evaluation platform that gives every agent a flight recorder: a versioned, safe trail from execution to release decision.\r\n\r\nThe stakes are real. Our agents generate millions of executions each month. To illustrate the scaling challenge, even a modest sample with several graders can create hundreds of thousands of evaluation jobs—before retries, comparisons, or human review. The hard problem is no longer writing a score; it is preserving trust under load.\r\n\r\nIn 18 minutes, we’ll show the platform in action. We’ll take a deliberately broken tool trajectory, inspect the recorded spans that expose the failure, and watch the CI gate reject the change without exposing raw customer payloads. Then we’ll unpack the design that lets teams self-serve without turning quality into chaos: immutable trace datasets and snapshots, versioned evaluators, metadata-first evidence, scoped access, governed content review, quotas, back-pressure, and comparable baselines.\r\n\r\nThe result is a quality loop teams can own, with guardrails the platform can enforce. The takeaway: empower teams to learn from every agent execution—but make “cannot verify” a visible outcome, never a green check.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "e0d72b91-4095-4495-8541-518f14d99ab1",
            "name": "Rahul Trikha",
            "url": "https://webdirections.org/ai-engineer/speakers/rahul-trikha/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/rahul-trikha/"
      },
      {
        "id": "2e791cc0-f381-41bb-b389-a55e1e5f0e21",
        "title": "How to Change an LLM System Without Guessing",
        "description": "We were running a production LLM pipeline that classified legal documents, and every change was a guess. Swap a prompt, change a model — better or worse? Nobody could say. The outputs looked plausible either way, and “plausible” is exactly how LLM systems hide their regressions.\r\n\r\nThis talk is how we went from operating on faith to changing the system on evidence — a systems story, not a model story. The layer, built in order: prompt fingerprinting so every output traces to the exact prompt and model; provenance on every document record; failure visibility so classification failures surface as data, not silence; and an evaluation harness wired into CI, scoring every change against a labelled baseline — report-only today, becoming a blocking gate once the corpus supports a threshold we trust. The corpus is the honest bottleneck — a starving eval set can’t gate anything — so we’re building a pipeline that mints eval fixtures from real production failures without PII ever entering git: labels in version control, documents in an erasable store inside the production boundary.\r\n\r\nThe trade-offs, honestly: why we grounded extraction in document citations the model has to resolve — fabricated evidence fails to resolve instead of producing a false highlight — why a lawyer’s override is becoming first-class state the system can re-check, and the decisions I’d make differently. I’ll show a regression the harness surfaced before it shipped, and a defect that sailed past every automated check: a legally-wrong output that was mechanically correct, in a dimension none of our instruments measured — where human judgment takes over.\r\n\r\nThe takeaway: the reliability of an AI feature lives in the harness around the model, not the model. Prompts, parsers, evals, telemetry, and release gates are one system — once you can measure it, you stop guessing.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "790d28af-b37f-4d0f-9f06-75c093762f3a",
            "name": "Yulia Kuchina",
            "url": "https://webdirections.org/ai-engineer/speakers/yulia-kuchina/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/yulia-kuchina/"
      },
      {
        "id": "05004661-6647-491c-83ca-2bbfdd8931ee",
        "title": "Don't Fight Hallucinations. Make Them Impossible",
        "description": "The Heatseeker AI chat answers data questions for marketers who make decisions with million-dollar budgets. Wrong answers or hallucinated numbers are not an option here, as you can imagine ;)\r\n\r\nThe fight against them (hallucinations, not marketers) was long and painful. We started with a \"naive\" approach, which we all tried at some point, I imagine: \"Hey, AI, here's all of our data, extract what we need, make no mistakes.\" It kinda worked but also kinda didn't - the output, especially on more complicated questions, could vary significantly.\r\n\r\nThen we tried various techniques of validating the answer before sending the result to the user. This greatly improved the correctness of the data, but made the code fragile (this policing involved a lot of regular expressions), the chat very slow because of too much back-and-forth, and worst of all - it still couldn't guarantee that the answer was actually correct. Yes, all the numbers were citable, but were they answering the question the user actually asked? No post-validation can tell you that.\r\n\r\nIt was time to look at the problem more holistically. Can we beat AI hallucinations and make them structurally and architecturally impossible? In this talk, I’ll share our journey and how we flipped everything upside down to do that ;)",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "104fdda7-943c-49a1-a9bd-90736afd6e12",
            "name": "Nadia Makarevich",
            "url": "https://webdirections.org/ai-engineer/speakers/nadia-makarevich/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/nadia-makarevich/"
      },
      {
        "id": "76af4d11-dbb2-496e-a2ba-78558bcd99c0",
        "title": "Why software factories can't be trusted (and how the Systems Engineering V-model helps)",
        "description": "Coding agents will do anything to get to a PR. Ours marked tasks complete that didn't exist, wrote \"next steps: verification\" under unreviewed pull requests, and quietly skipped every boring stage between the ticket and the diff. In our world, that isn't a quirk — it's disqualifying. We build cardiology, respiratory and clinical-trials systems for public hospitals: large brownfield codebases with decades of accumulated behaviour, where you don't get to make mistakes in production. When mistakes reach production, people die.\r\n\r\nSo we stopped prompting harder and applied Systems Engineering rigour. Our harness is built from four high-level blocks: the Systems Engineering V-model, deterministic verifiable gates, adversarial verifiers, and loops that carry every piece of work from Jira ticket to plan to PR.\r\n\r\nThe decisions that made it work: the gates are control flow, not instructions — prompts get rationalised away under pressure; workflow code doesn't. The verifiers are adversarial — their job is to refute the work, not confirm it — and anything red loops back until it's green. And humans remain a really important part of the process; where they sit, and why, is half the story.\r\n\r\nThe factory now runs unattended on a VM, dispatching work across model providers. Ticket cycle time has dropped from 45 days to 6, and production code issues are down roughly 40%. I'll walk through the thinking behind each block, the limitations we've hit honestly, and leave you with a skill you can use to build your own V-model harness.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "d6194323-1564-4f45-bb71-6d8e9f5dea01",
            "name": "Mark Johnson",
            "url": "https://webdirections.org/ai-engineer/speakers/mark-johnson/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/mark-johnson/"
      },
      {
        "id": "f68351a4-f32e-49d0-b672-ac016ea7b866",
        "title": "Running 100K+ Concurrent Agents: It's Never the Prompt",
        "description": "Your Agents Aren't Failing at Prompting\r\n\r\nEvery agent demo works. Then you run ten thousand at once, and the failures have nothing to do with the model: dead tuples piling up, a sync HTTP client freezing the event loop, requests hanging forever with no total timeout.\r\n\r\nThis talk is a look inside how we build and run agents at massive scale in production. It covers what broke, what fixed it, and the lessons that stuck, from why prefetch is really a distributed semaphore, to why layered timeouts beat one big number, to how Postgres as a queue can drive autoscaling at massive scale. It also shows how to get the most out of Postgres and its cache before bringing in Redis.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "26daf7a5-1b4a-4bba-8bfa-e286f2b1b762",
            "name": "Harry Nguyen",
            "url": "https://webdirections.org/ai-engineer/speakers/harry-nguyen/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/harry-nguyen/"
      },
      {
        "id": "88aeb7c0-147a-4e6a-874c-6eff9efeeff2",
        "title": "Regulated Doesn't Mean On Rails: what compliance actually asks of your engineers",
        "description": "We spent three months rebuilding a regulated enterprise's development lifecycle around agents, then worked out how much of the governance we had added was doing real compliance work. Less than we assumed.\r\n\r\nAcross APRA's prudential standards, the ISM, SOC 2 and ISO 27001, the same four obligations recur: traceability, change management, testing proportionate to change, and independence between whoever makes a change and whoever approves it. Not one says how an engineer should get to a change.\r\n\r\nA specification records what you intended, not what happened. The trace already exists — the session, the tool calls, the commit that carries them — and unlike a document, an auditor can re-run it.\r\n\r\nI will show what we kept, what we tore out, and the thing nobody has solved: every separation-of-duties rule I can find governs \"persons\". An agent that implements and approves its own change breaches it today.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "94d3fb30-ac78-493e-8529-163ecf6e63f9",
            "name": "Stephen Sennett",
            "url": "https://webdirections.org/ai-engineer/speakers/stephen-sennett/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/stephen-sennett/"
      },
      {
        "id": "0e864faa-d427-4ff6-b094-e392707246b0",
        "title": "Adaptive Learning: Encoding Clinician Edits as Memory",
        "description": "Clinicians edit their AI-generated notes for many reasons: adding or removing content, fixing spelling, and preferences for structure, wording, or terminology. Some of these edits are contextual, applying in some situations but not others. The data is noisy, but inside it is useful signal that can be used to predict and make these edits before a clinician ever sees the note, cutting the time spent editing and giving it back to the patient.\r\n\r\nTo do this we built a custom agentic workflow that currently processes hundreds of thousands of notes and edits in production, and encodes them into reusable memories per user that correct their future notes. Each memory carries a sense of when it should and shouldn't apply, which is what keeps edits precise. In this talk I'll walk through the harness design (tools, memory, evals, self-healing), as well as some key lessons from the failure modes and experimental results that helped shape the build.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "415c32e8-33a4-4249-889f-3122353ce4c4",
            "name": "Vlad Gavrilov",
            "url": "https://webdirections.org/ai-engineer/speakers/vlad-gavrilov/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/vlad-gavrilov/"
      },
      {
        "id": "02419951-c6be-49e5-8a04-599d18e6efd0",
        "title": "Ontologies: AI’s Operating Manual For Your Business",
        "description": "Tools churn. Factories commoditise. When everyone has the same models, the advantage goes to whoever gives their agents the clearest blueprint of how the business works - what exists, how it connects and the rules it runs on. That blueprint is an ontology.\r\n\r\nIn data.world's benchmark, an LLM querying enterprise SQL directly answered 16% of questions correctly. Over a knowledge graph with an ontology, 54%. With the ontology catching the model's bad queries for repair, 72%. Same questions. Only the modelling changed.\r\n\r\nI'll untangle ontologies from knowledge graphs, semantic layers and taxonomies, then demo an open-source ontology I built for Monsters, Inc. (OWL, SHACL, SPARQL, PROV-O). Watch an agent refuse to skip a compliance step, a rule stop even the CEO exporting a child's data, and a validator catch three planted violations unaided.\r\n\r\nYou'll leave knowing how to start, and why AI has finally made one cheap to maintain.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "1727370a-ee5c-4f17-9b6f-e0b563453afe",
            "name": "Gareth Williams",
            "url": "https://webdirections.org/ai-engineer/speakers/gareth-williams/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/gareth-williams/"
      },
      {
        "id": "dec59d08-c89a-425d-90ec-ea7e3df50841",
        "title": "Should You Build a Software Factory? Tales from an Open-Source Maintainer",
        "description": "Coding agents helped me write code faster, but I kept running into problems elsewhere in my engineering workflow. Tests passed without proving much. Parallel tasks changed the same files. Reviews piled up. This talk follows the changes I made while building a SaaS and maintaining my open-source projects. I'll use real code and PRs to show what each change solved, and what still needed my attention.\r\n\r\nYou'll learn how to turn repeated instructions into shared skills, then enforce the important rules with checks. We'll look at tests that exercise behaviour and worktree setups that let agents edit and check changes independently. You'll see why those isolated changes still need to be checked together before merging. I'll show how separate Review and Repair agents handle findings, with a fresh review after every repair. We'll also cover keeping the queue within your review capacity and bringing production failures back as new work. Before leaving agents unattended, we'll look at retry limits and recovery when work gets interrupted. Those improvements eventually became my self-hosted, open-source factory around GitHub issues. We'll open its live board at the end and follow work submitted by the audience. You can take these improvements back to your own workflow, even if you only run one agent.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "d9989619-2b45-47da-9dd8-56c7f90e5ba7",
            "name": "Harlan Wilton",
            "url": "https://webdirections.org/ai-engineer/speakers/harlan-wilton/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/harlan-wilton/"
      },
      {
        "id": "6c224363-6430-407b-bfaf-4af9566a7333",
        "title": "Trust is engineered, not granted: why we focus on verifying before background coding agents",
        "description": "Heidi is an AI scribe used by 130K clinicians a week. Our 150 engineers ship 100+ PRs daily into prod, and AI made writing code so cheap that review became the bottleneck: our P75 review wait was 14 hours, almost all of it queue time. A 14-hour queue is a reliability problem — it batches changes, delays fixes, and pushes people toward the \"just approve it\" path. So we built Hubert, a suite of review agents that runs on every PR and can approve the safe ones straight to production.\r\n\r\nThe hard part wasn't the agent. It was earning the authority to approve. As such we built the eval set out of our own incident history: every PR implicated in a past postmortem, replayed against the reviewer, scored on whether it would have blocked the change. That corpus is the thing that made approval authority defensible to a skeptical engineering org, not hype-d benchmarks on \"SWEBench\", but \"here is the change that took us down in April, and here is the gate that stops it now.\" Every new incident adds a case; the agent is only ever as good as the failures we've taught it.\r\n\r\nI'll cover how we built and scored the replay corpus, the shadow-mode rollout that ran Hubert alongside human reviewers until the disagreement rate was boring, the per-gate telemetry that tells us which reviewers pull their weight and which are theatre, and where the blast radius stops when it's wrong. Since May: 4,000+ auto-approved PRs, about a quarter of everything we merge, zero production incidents traced back to an auto-approval.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "93b19f77-cea1-44af-9a96-a6dc467a1c6a",
            "name": "Vivek Katial",
            "url": "https://webdirections.org/ai-engineer/speakers/vivek-katial/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/vivek-katial/"
      },
      {
        "id": "d9ead44a-5a69-4c3e-b14a-5ffad843352f",
        "title": "It's time for a new kind of software",
        "description": "Applications package a predetermined interface, data model and set of capabilities around somebody else’s idea of a task. Generative models can now produce software on demand, but generative UI today either sits atop the existing application stack or replaces the richness of the app with an interface disconnected from the data and capabilities needed to do useful work.\r\n\r\nThis talk asks what happens when we replace the app all the way down.\r\n\r\nIn this live, demo-led talk, I’ll use Television—a system we are building for full-stack generative software—to make the architecture concrete. I’ll walk through the post-app stack, the problems each layer solves, the alternatives we tried and why we arrived at the current design. The result is not an app with AI in it, or a chat box that launches apps, but software that can take a different shape each time while remaining connected to dependable state and real-world actions. I’ll cover the central trade-offs between generated and durable software, flexibility and consistency, and agent autonomy and legible control. Attendees will leave with a practical architectural model for building agentic software beyond the app for production systems—and an understanding of why generating pixels is the easy part.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "1f224d74-f60a-4969-a431-36097829dc9b",
            "name": "Rupert Manfredi",
            "url": "https://webdirections.org/ai-engineer/speakers/rupert-manfredi/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/rupert-manfredi/"
      },
      {
        "id": "eec6878e-6beb-45ef-a170-761eb41260cf",
        "title": "Don't stop, won't stop. Automated verification of long-running agentic loops",
        "description": "Long-running agentic loops create different engineering problems than code generation. Once agents work unattended for many hours, iterate through 20 or more review cycles, and build stacked PRs towards a larger goal, the challenge changes to keeping that activity aligned, verifiable and useful. This talk looks at the verification systems built around behavioural claims, acceptance evidence, system invariants, decision records and independent review so agents can keep correcting themselves as they work.\r\n\r\nUsing real production examples, I’ll show how these loops converge, how automated review feeds back into implementation, and how each change set identifies the assumptions, risks and decisions that still need human attention. The aim is to increase throughput without creating cognitive debt, allowing larger and more autonomous changes while keeping the system aligned to an architect’s intent and focusing human review on the decisions that actually matter and need judgement.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "e3cfdc31-2952-4db1-8124-21bf9d6bbba8",
            "name": "AJ Fisher",
            "url": "https://webdirections.org/ai-engineer/speakers/aj-fisher/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/aj-fisher/"
      },
      {
        "id": "b4f4960d-8afa-4667-ad52-a11e7c932850",
        "title": "Token usage is the new lines of code (and it's just as useless)",
        "description": "Canva hit mid-2026 with every AI metric a vendor could hand us and no answer to the only question the business actually cared about: is this working? Adoption flattened at around 85% of ~3,000 engineers, and the number stopped telling us anything. AI touched 43% of PRs, PR volume was up 44% year over year (and growing week-over-week), but none of us could say what any of that had actually bought the business. Token spend ran eight figures annually, and one engineer alone racked up a $75K bill. Ask five leaders what it meant and you'd get five different answers.\r\n\r\nThis is the story of what we built to fix that, and what broke along the way.\r\n\r\nWe ended up with a three-layer value architecture:\r\n1) The bottom layer is spend and adoption telemetry. A necessary baseline, but worthless on its own. \r\n\r\n2) The middle layer is team-level ROI: we join AI spend to DORA metrics, grade teams within cohorts of similar peers, use median-plus-20% as a talking point rather than a hard cap, and deliberately report a fuzzy signal (\"bottom third of your cohort\") instead of raw numbers, because raw numbers get gamed the same way PR counts did. \r\n\r\n3) The top layer is an exec-facing layer any LLM can query we wrote our measurement methodology as a skill over the warehouse, so a CxO can ask any AI tool and get the same answer we'd give them directly.\r\n\r\nAcross it all, using analytics first principles to measure and experiment (A/B test) our developer platform’s AI capabilities to concretely show value, as well as something I don’t think anyone else is running at scale: measuring attentive human time per agent conversation, then having a post-merge agent estimate time saved against tokens spent. So ROI gets measured at the point where its value is created. \r\n\r\n\r\nWhat you'll walk away with is the architecture, the failures behind it, and the one metric we're betting on: engineering capacity created",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "f29357fd-efe5-4394-a4cf-7f515de3ad87",
            "name": "Fawaz Ahmad",
            "url": "https://webdirections.org/ai-engineer/speakers/fawaz-ahmad/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/fawaz-ahmad/"
      },
      {
        "id": "756e8601-0a9c-4df0-89fd-11b844647777",
        "title": "The CLI is dead, long live the CLI",
        "description": "Coding agents didn't kill the CLI. They became its most demanding users: they can't see a spinner, stall on prompts they can't answer, and treat error messages as instructions.\r\n\r\nSo we rewrote Sentry's CLI from scratch for humans and agents. This talk is a practical guide to building CLIs for both: output that adapts to its reader, prompts that never block, errors that tell an agent what to do next, and agent skills shipped inside the binary.\r\n\r\nAnd no, MCP isn't dead either. We still run our MCP server, and I'll show when it beats a CLI, when it doesn't, and how we trace what agents do with both in production.\r\n\r\nThe human-only CLI is dead. Long live the CLI.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "49a97552-09a6-4153-a064-a81022fdaf7b",
            "name": "Jan Peer Stöcklmair",
            "url": "https://webdirections.org/ai-engineer/speakers/jan-peer-stocklmair/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/jan-peer-stocklmair/"
      },
      {
        "id": "8f0c8a0c-2a8b-4446-a126-e34170de3318",
        "title": "Assert Chaos",
        "description": "Once a product reaches the wilds of production, all sorts of things can go wrong. Carefully logging and monitoring are one thing, but what if we could do better?\r\n\r\nBring new levels of code stability by recreating the ‘cat on a keyboard’ randomness of production with a chaos test agent. This talk goes through what it takes to implement chaos test agents on a budget and what they can do when unleashed in a production environment.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "c0571a6f-24d4-4f2e-b3ab-ad9492054086",
            "name": "Chris Lienert",
            "url": "https://webdirections.org/ai-engineer/speakers/chris-lienert/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/chris-lienert/"
      },
      {
        "id": "78f99404-213d-43cf-a126-12506ef62101",
        "title": "The agents went rogue at 2%",
        "description": "Giving an agent a skill sounds straightforward: it loads a Markdown file, follows the instructions and completes the task.\r\n\r\nOurs launched a long-running, paid process and told the agent to wait for the result. But agents repeatedly started the process, became sidetracked and rebuilt the output themselves. In one test, only 1 of 16 runs produced a fully correct result, while 12 of 16 paid for exports they never used.\r\n\r\nThe agents had gone rogue, and we thought we knew the fix. We rewrote the instructions. Then we rewrote them louder, again and again. The agents still got sidetracked.\r\n\r\nThe real problem turned out to be one line, and it wasn't in the prompt.\r\n\r\nThis talk retraces that investigation from the agent's side of the screen: why an agent's strangest decision can be its most reasonable one, and what agents need from the systems they depend on.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "a98c3b0f-5daf-432e-acff-be5956f64e17",
            "name": "James Peter",
            "url": "https://webdirections.org/ai-engineer/speakers/james-peter/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/james-peter/"
      },
      {
        "id": "c3b36efa-9a99-44c6-ae31-b8c9fe242dd2",
        "title": "Where Should the Dice Roll? Placing Non-Determinism Deliberately in Enterprise AI",
        "description": "Most production AI guidance assumes you want consistency: pin the prompt, lower the temperature, eval for drift. But a whole class of enterprise use cases - idea generation, recommendation, exploration, synthesis - is worthless if the output is predictable. The engineering problem flips: how do you make variance useful, keep it safe, and convince a governance function that a system with no single right answer is still under control?\r\n\r\nThis talk is the build story of a production AI concept-generation pipeline. I'll cover the concrete decisions: constraining the shape of outputs with strict schemas while leaving the content free; grounding generation in competitive and regulatory data so novelty stays feasible; designing a three-stage human funnel (expert weighted scoring, swipe-based crowd screening, review loop) that turns volume into a shortlist.\r\n\r\nThe evaluation approach when there's no ground truth - measuring diversity, feasibility, and survival-through-funnel instead of correctness.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "d5a94f28-6080-42d8-a08d-37a0b9f0b383",
            "name": "Vighnesh Deshpande",
            "url": "https://webdirections.org/ai-engineer/speakers/vighnesh-deshpande/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/vighnesh-deshpande/"
      },
      {
        "id": "5b215820-4e4d-46e3-b714-7d6fc03945ad",
        "title": "The Benchmark Ends. The World Doesn’t: Building Persistent Engineering Environments for Continual Learning",
        "description": "Large language models can complete sequences of tasks—but continual learning is not just doing more tasks. Most agent environments reset after every episode: state disappears, required follow-up vanishes, delayed effects are cut off, and the next task arrives as though the previous one never happened.\r\n\r\nReal engineering projects do not reset. They deal in atoms, not bits: decisions end up in steel and concrete, and a missed comment cannot be patched after the pour. Evidence arrives late, decisions expire, interventions require verification, and a new agent may inherit the consequences of decisions it did not make. A project is one continuing, long-horizon world, and agents must carry work forward across tasks, revisions, and handovers.\r\n\r\nThis talk presents Persistent Task Worlds: executable environments for evaluating and training agents across continuing engineering project histories. They separate world state, observable evidence, institutional records, conversation state, and learner state, because a record is not the world: a stored session lists which file an agent read, not what it saw, so learning from history means replaying it. Time advances independently of agent turns; documents and requests arrive mid-stream; actions create requirements that persist; fresh agents inherit the same world, including the project memory earlier agents wrote; and evaluation includes effects beyond the visible scoring window, such as whether later revisions preserve earlier work.\r\n\r\nI'll present results and early evidence from two studies, both starting from single-episode evaluation (one instruction in, one answer out) as the baseline. First, fixed models and agent harnesses traverse matched environments with different ways of carrying the past: continuous context, raw history, structured state, project memory written by the agent and harness, or only the current snapshot. We measure decision quality, missed follow-up requirements, revision behaviour, how memory is written and reused, downstream outcomes, and interaction cost.\r\n\r\nSecond, we compare frozen models with models post-trained on real agent trajectories, contrasting trajectories flattened into single episodes with continuing histories. We measure forward transfer, sample efficiency, and performance on untouched histories. External records remain available as a control, letting us separate improvement in the learner from improvement in record-keeping.\r\n\r\nThe same problem appears in coding agents, scientific workflows, infrastructure, and any system where today's actions change tomorrow's work. Persistent memory is not a persistent world—and a persistent world is not proof of continual learning.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "9490204f-a75e-4b8d-a7d9-4455c2f3aed9",
            "name": "Theodoros Galanos",
            "url": "https://webdirections.org/ai-engineer/speakers/theodoros-galanos/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/theodoros-galanos/"
      },
      {
        "id": "2f5c66c3-bd08-4e56-9e86-39de07b57e93",
        "title": "The five stages of losing our craft",
        "description": "Your best engineer won't use AI tools. Your tech lead is using them but won't tell anyone. Your new grads don't understand what the fuss is about. And you, the leader, are supposed to have answers for all of them.\r\n\r\nI've had some version of this coaching conversation every week for the past year. What I've found is that teams aren't just \"resistant to change.\" They're grieving. And that grief follows a predictable pattern that, once you can name it, you can actually do something about.\r\n\r\nThis talk maps the five stages I'm watching play out across engineering teams right now: denial, anger, bargaining, depression, and an acceptance that looks nothing like the LinkedIn version.\r\n\r\nFor each stage, I'll share how to recognise it in yourself and your people, what to say (and what to absolutely not say), and how to help someone move forward without invalidating what they're feeling.\r\n\r\nYou'll walk away with a practical framework for the conversations your team needs you to have about AI, craft, and what it means to be an engineer now.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "1b33891f-3253-4a6c-9a96-4334beedf25a",
            "name": "Andrew Murphy",
            "url": "https://webdirections.org/ai-engineer/speakers/andrew-murphy/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/andrew-murphy/"
      },
      {
        "id": "ed81e3d6-18fe-44ac-87f1-871821de4096",
        "title": "Management Is Dead, Leadership Is Not",
        "description": "Engineering teams are getting smaller. A team of four with strong AI tooling now ships what took twelve people two years ago, and the org chart has not caught up. The first layer to feel it is middle management - the coordination tier that existed largely to move information between people who were too numerous to talk to each other directly. When the team fits in one room, that layer is overhead.\r\n\r\nThe conclusion many organisations draw is that leadership itself is what got automated away. That is the wrong read. Smaller teams make more consequential decisions per person, move faster than any review process can keep up with, and have fewer places to hide a bad call. What they need is not someone tracking status. They need direction, judgement about what not to build, and someone accountable for the quality of what the agents produce.\r\n\r\nIn this talk I will draw on leading engineering teams through this shift to cover what genuinely disappears when teams shrink, what quietly becomes more important, and how to tell the difference before restructuring around the wrong one.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "dcd0092c-f257-49c5-ba66-1d34c8d3715d",
            "name": "Inga Pflaumer",
            "url": "https://webdirections.org/ai-engineer/speakers/inga-pflaumer/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/inga-pflaumer/"
      },
      {
        "id": "943e458b-876f-42db-951a-3d77b06312b0",
        "title": "From REST to Agentic: Trusted APIs in the age of AI",
        "description": "Enterprise platforms don't get to start from scratch. When AI agents need to act on behalf of users, they inherit the API surfaces those users already depend on — surfaces evolved over many years for human-driven workflows, not autonomous reasoning. This talk explores the practical journey of adapting established, trusted API layers to serve agentic workloads while preserving the performance, predictability, and permission guarantees that earned customer trust in the first place.\r\n\r\nIn this talk we’ll explore two contrasting cases: Jira — a 25-year-old platform with myriad REST endpoints evolved over time and deep ecosystem integrations — and Teamwork Graph, a newer data layer, purpose-built for cross-product context retrieval. Together they illustrate the spectrum from built “for-human” to “machine-native\" and the path to creating a complete agentic offering without sacrificing the trust customers have built over decades.\r\n\r\nKey Takeaways:\r\n\r\nTrust is inherited — Agent-facing APIs must honour the same permission, performance, and data-sovereignty contracts as the human-facing surfaces they extend.\r\nAPI Evolution demands strategy — Moving from human-driven REST to agent-ready tooling isn't a refactor, it's a product decision that demands intentional design and governance from day one.\r\nMachine-native is not the same as starting over - purpose-built agentic APIs don't require abandoning your existing ecosystem, they require a deliberate context layer on top of it",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "cc875df4-a947-46aa-9cb0-63c72de3595a",
            "name": "Leigh Whiting",
            "url": "https://webdirections.org/ai-engineer/speakers/leigh-whiting/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/leigh-whiting/"
      },
      {
        "id": "2009ffb6-aede-4ca0-8956-38bc2e835a4e",
        "title": "How to Count to One Hundred",
        "description": "Agents don't work well together in chat rooms: put a few in a channel and they talk past each other. Natural language and unrestricted JSON feed an open-ended pipeline of reasoning and tool calls with full Turing power. Deciding whether to accept a message based on what arbitrary computation will do is undecidable in general. Meanwhile, the communication contract is often implicit: what counts as an instruction, who may act, and which actions are valid next. Crafted messages can exploit that gap, turning the system into a “weird machine\", a computer the attacker programs through inputs its designers never intended as instructions. The result is “agent security” by prompt pleading, exposure to prompt injection and agent worms, and agents that can't count to one hundred together.\r\n\r\nThis is a field report on giving humans and AI agents a shared language they can extend and check. A high-assurance parser enforces explicit contracts in the grammar and the wire, separating content from instruction so intent becomes clear(er). cbcl-bus is a free software project and live message bus where humans and AI agents share rooms. Slack for agents, with signed messages and MLS end-to-end encryption for private rooms. Its language is CBCL, named after McCarthy's 1982 proposal. Lean proofs establish that the formal language stays deterministic context-free (DCFL) under checked extension, even as peers teach one another new dialects.\r\n\r\nThe payoff is shared objects with typed actions, causal protocols and rules for deriving state, presented through replaceable hypermedia interfaces. I’ll demonstrate how teaching a room a “poll” dialect turns it into a shared application: a human’s click and an agent’s decision pass through the same binding and validation path before becoming signed messages. Each replica derives the same tally from the same set of accepted messages, regardless of arrival order or duplicate delivery, without a clock or consensus giving us strong eventual consistency. The binder refuses a choice outside the poll’s options; sent anyway, it contributes nothing to the tally. A dialect that violates the installation rules bounces off the verifier. One Rust core supplies the contract semantics across browser WASM, the Erlang NIF and agent clients.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "1b09233a-caa7-49c6-96e1-46107d84731b",
            "name": "Hugo O'Connor",
            "url": "https://webdirections.org/ai-engineer/speakers/hugo-oconnor/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/hugo-oconnor/"
      },
      {
        "id": "3d87474a-6760-4fe7-ada5-1d09ae1c7017",
        "title": "Extracting Tacit Knowledge for Production Agents",
        "description": "An eval is a definition of \"good\" written down well enough for a machine to grade against. In a real business, \"good\" isn't written down anywhere. It lives in someone's reaction, in what past work actually did, and in how a person talks. So the eval comes second. First you get \"good\" out of wherever it lives, and each place needs its own method.\r\n\r\nThe first place is the person's reaction. Two domain experts corrected a coaching agent in a Telegram group, and each correction became a commit to the agent's own instruction file, routed by whether the agent could apply it alone, had to ask me first, or should stay a note. Then a panel of critics I built to grade runs lied to me, rating a shallow run \"demo-ready and close to trial-ready.\" The fix was not a better judge prompt. It was turning that failure into a permanent human-labelled case and refusing to trust any critic that couldn't reproduce my verdict.\r\n\r\nThe second place is the system's residue. For a second agent, the owner couldn't describe his own process, so I reconstructed it from past campaigns and tags on old lists and proposed it back for correction. The third place is his conversation history. I mined it for every correction and acceptance and built a simulated version of him, so new versions could run against him without him present.\r\n\r\nA business can only hand an agent the work it can check, so the eval is the edge of what you can delegate. I'll cover the infrastructure: an agent that commits to its own instructions under authority rules, hot reload into live sessions, simulated users built from transcripts, a gate-first rubric, and critics that must pass a human-labelled test before they're trusted.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "40dfbfd0-f98b-4e91-add0-c24b04172fd4",
            "name": "Khali Kalpa-Young",
            "url": "https://webdirections.org/ai-engineer/speakers/khali-kalpa-young/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/khali-kalpa-young/"
      },
      {
        "id": "57a9afcd-0fc0-4976-9506-2b6b780577b8",
        "title": "My Agents broke APIs - Fixing Multi-Agent Systems with MCP",
        "description": "Modern AI agents struggle not because of reasoning limits, but because of interaction with tools on interfaces designed for humans. In agentic systems, this mismatch leads to incorrect tool selection, redundant calls, increased latency, & weak workflows that fail under real-world conditions. As MCP emerges as a standard for how models connect with tools & applications, it provides a path to move from heuristic interactions to structured, contract-driven systems.\r\nTaking a real-world use case, this talk explores how the MCP redefines agent-tool interaction through schema-based contracts, enabling deterministic execution & reducing ambiguity. We’ll dive into MCP architectures & demonstrate how standardized tool definitions improve reliability & efficiency in agent workflows. We’ll compare a traditional API-driven approach with an MCP-based design, highlighting measurable improvements in latency, cost, & system behavior.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "c3534f0b-dc48-4b9e-b4bc-b4ea9c1518a6",
            "name": "Anannya Roy Chowdhury",
            "url": "https://webdirections.org/ai-engineer/speakers/anannya-roy-chowdhury/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/anannya-roy-chowdhury/"
      },
      {
        "id": "34cbdd7a-0d15-4c99-b1fe-190b17e5f9a5",
        "title": "We Deleted Most of Our Agents. Everything Got Faster",
        "description": "The default advice is to add agents: specialised roles, an orchestrator, handoffs between them. We built that. It was a genuinely useful way to explore the problem space, and under real production traffic it was slow and expensive.\r\n\r\nWhen an agent system is slow, the instinct is to reach for a faster model. That was the wrong lever. Our latency wasn't dominated by inference — it was dominated by round trips. Every handoff re-serialised context, every planner turn burned tokens producing nothing a user would ever see, and the overhead compounded per hop. It's the N+1 query problem wearing a different hat: hundreds of small calls where one would do.\r\n\r\nSo we started deleting. Collapsing to a single agent holding multiple toolkits cut cost per resolved task by roughly two-thirds and latency by about half — same model, no infrastructure change. A caching layer over the repeated-question tail took a further ~30% off the already-reduced figures, on the same principle: the cheapest model call is the one you never make.\r\n\r\nI'll show what we measured at each hop to identify round trips as the real cost driver, what broke when we consolidated — tool-selection accuracy, eval attribution, the loss of per-subtask model tiering — and how we handled each. The patterns for managing large toolkits are already known: retrieval over tool definitions, hierarchical grouping, progressive disclosure. What isn't documented is which ones survive production, and where each quietly stops working. Figures are directional and normalised. Includes a side-by-side trace replay of both architectures on an identical task.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "23d5163e-d113-416c-8c77-812b54fa154a",
            "name": "Khang Nguyen Hoang",
            "url": "https://webdirections.org/ai-engineer/speakers/khang-nguyen-hoang/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/khang-nguyen-hoang/"
      },
      {
        "id": "07c114f0-a670-4295-ac61-710d0e496b7d",
        "title": "Your website is a terrible API: serving agents a different page at the edge",
        "description": "Every enterprise site we work on was built for a human with a browser. When an AI agent fetches the same URL it receives navigation, cookie banners, client-rendered components and marketing prose, then has to guess at the facts underneath. Retrieval quality suffers and the citation goes to whoever structured their data better.\r\n\r\nOur first attempt put the structured data on a separate directory subdomain. It worked, and then it did not. Crawl budget fragmented, authority did not transfer, and the subdomain was treated as a secondary source rather than the primary one. This talk is the rebuild. We now map structured data onto the client's existing sitemap and use a Cloudflare worker to detect verified agent requests and serve an agent-optimised representation of the same page in place of the human HTML.\r\n\r\nI will walk the architecture end to end: verifying agent identity rather than trusting user agent strings, deciding what belongs in HTML versus JSON-LD versus Markdown versus an MCP tool call, keeping both representations semantically identical, and the canonicalisation and cloaking questions every reviewer raises within thirty seconds of seeing this.\r\n\r\nIncludes a live side-by-side demonstration of a production enterprise page as a human sees it and as an agent sees it, plus before and after crawler and citation data from the real world. I will also cover what still does not work and where I think this approach may be wrong.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "98b328a7-bf1a-4d51-a153-af092a33e723",
            "name": "Jack Bear",
            "url": "https://webdirections.org/ai-engineer/speakers/jack-bear/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/jack-bear/"
      },
      {
        "id": "363ab615-2804-40a1-859c-5f9bbe26649b",
        "title": "From vibes to a systematic eval flywheel: evals for high stakes AI agents",
        "description": "If you are sceptical about how evals drive real results, or looking for more depth than introductory tutorial videos, this talk is for you.\r\n\r\nWe'll show you how Lorikeet, an AI customer service startup, built an eval system to rapidly raise the quality of a complex AI agent. Coach is our AI assistant that helps customers configure their AI concierges that automate millions of tickets every month, with a huge surface area of use cases. We'll share our journey from receiving vague vibes-based feedback about low quality answers, to how we systematically built a flywheel to rapidly improve the quality.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "953c92d6-6dbb-4bc8-a701-04824d4d03b8",
            "name": "Donna Zhou",
            "url": "https://webdirections.org/ai-engineer/speakers/donna-zhou/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/donna-zhou/"
      },
      {
        "id": "8ddcd6b0-06cc-4fc7-bb0d-b3aa21aca145",
        "title": "AI Sandboxes: Running Coding Agents Safely in Production-Grade Environments",
        "description": "The number of cyber attacks and security risks related to Coding Agents has sky rocketed. AI coding agents like Claude Code, Codex CLI, and Gemini CLI don’t behave like your typical developer tools. They install system packages, modify configurations, delete files, run services, and even spin up Docker containers, often requiring constant permission prompts or risky access to your host machine.\r\n\r\nThis talk explores Docker Sandboxes as a new execution model for autonomous coding agents. Built on microVM-based isolation, Docker Sandboxes provide disposable, agent-safe environments where coding agents can run unattended while remaining fully isolated from the host system.\r\n\r\nWe’ll walk through why traditional approaches like OS sandboxing, containers, and full virtual machines, break down for agent workflows, and how Docker Sandboxes combine the developer experience of containers with the hard security boundaries of VMs. Using live examples, we’ll show how agents can safely run Docker-in-Docker, install dependencies, access only the project workspace, and be reset instantly.\r\n\r\nBy the end of this session, you’ll have a clear mental model for when and how to use Docker Sandboxes to unlock higher levels of agent autonomy without compromising safety, security, or developer experience.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "4ed433ee-59b3-482d-b66c-edee77c4fc93",
            "name": "Shivay Lamba",
            "url": "https://webdirections.org/ai-engineer/speakers/shivay-lamba/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/shivay-lamba/"
      },
      {
        "id": "5c1518c8-7a13-4c9e-9446-1178d168c955",
        "title": "Shipping together in an AI native team",
        "description": "While building an AI agent for marketing workflows at Leonardo.Ai, we discovered that the traditional design-to-engineering handoff was failing. A static mock could describe a happy path, but not how an agent would group assets, select tools, recover from errors, or change behaviour when its underlying model changed. Engineering could implement the interface correctly while still shipping the wrong behaviour.\r\n\r\nWe began replacing handoffs with shared, executable evidence. Designers built working prototypes inside the product repository and used preview deployments for review. Engineering constrained generative UI through typed tool contracts and a closed registry of design-system components, rather than allowing an LLM to emit arbitrary JSX. We added full-path tracing, tiered evaluators and repeatable datasets so design, research and engineering could inspect the same behaviour. Finally, we packaged that evaluation knowledge as repository-local AI skills, designed to let contributors test ideas without depending on an evaluation specialist.\r\n\r\nThe results included image and video generation, shipping eight days after we aligned on a reusable prototype path. One model comparison found tool-error rates ranging from 2.0% to 8.1%, including 45 retries caused by a single runaway trace - evidence that a cheaper model substitution was also a product-quality decision. The wider team ultimately reduced agent cost per user turn from 10 cents to 1 cent.\r\n\r\nThis talk explains why broader authorship needs sharper contracts, observability and accountability - not fewer boundaries.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "69073d24-65c8-4d33-80fd-89b1a554f90a",
            "name": "Sandra Arato",
            "url": "https://webdirections.org/ai-engineer/speakers/sandra-arato/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/sandra-arato/"
      },
      {
        "id": "6b390ce4-98aa-4ad2-b19e-39cc8bf60cbc",
        "title": "The AI year: how we work, hire and grow now",
        "description": "For a lot of teams, this was the year AI stopped being an experiment and started being part of the job. Shipping got faster, but so did the pace, the pressure and the pile of code waiting for review. \r\n\r\nDrawing on Lookahead's 2026 research with people across Australian tech community and interviews with over 20 tech leaders, Fiona will share what's really changing: how the work feels, which skills are rising and fading, how roles are blending across engineering, product and design, and what it all means for hiring and careers.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "65ae247b-696e-4928-9a9e-bb1ce2e9c86e",
            "name": "Fiona Chan",
            "url": "https://webdirections.org/ai-engineer/speakers/fiona-chan/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/fiona-chan/"
      },
      {
        "id": "2f8bc3e3-9f2e-4204-876b-b81aca169945",
        "title": "Does This Agent Make My Context Look Big? Right-Sizing AI Architectures for Production",
        "description": "All-in-one personal agent harnesses showcase the incredible potential of capability-rich AI assistants. But deploying a monolithic \"do-it-all\" agent into production often leaves teams struggling with context dilution, fragile tool calls, un-evaluable execution paths, and massive security blast radiuses.\r\n\r\nHowever, swinging to the opposite extreme—decomposing every task into a complex micro-agent graph—swaps prompt engineering problems for software engineering complexity. Inter-agent handoffs are inherently lossy, graph state deadlocks happen, and you lose the \"free capability boost\" of swapping in new foundation models overnight.\r\n\r\nIn this session, we will unpack the art of Right-Sizing AI Architectures. We will examine how to use soft-deterministic micro-agents where they yield the highest ROI (like context-heavy routing and non-LLM ground-truth evals), while avoiding the unnecessary overhead of over-engineered agent networks. Attendees will leave with a practical decision framework for matching agent boundaries to actual production requirements.",
        "type": "talk",
        "track": null,
        "date": null,
        "startTime": null,
        "endTime": null,
        "durationMinutes": null,
        "location": null,
        "speakers": [
          {
            "id": "8df67c68-18d6-4cc2-a60c-5b5eb44ed4a6",
            "name": "Hamish Songsmith",
            "url": "https://webdirections.org/ai-engineer/speakers/hamish-songsmith/"
          }
        ],
        "url": "https://webdirections.org/ai-engineer/speakers/hamish-songsmith/"
      }
    ]
  },
  "offers": [
    {
      "id": "conference",
      "name": "Conference",
      "attendanceMode": "in-person",
      "price": 1195,
      "priceCurrency": "AUD",
      "validThrough": "2026-10-09",
      "availability": "in-stock",
      "includes": [
        "Keynotes",
        "Two dedicated tracks",
        "Networking Reception",
        "Videos post-conference, on-demand",
        "Fully catered"
      ],
      "url": "https://register.webdirections.org/ai-engineer-sydney-2026?tickets=conference:1"
    },
    {
      "id": "conference-workshop",
      "name": "Conference + Workshop Day",
      "attendanceMode": "in-person",
      "price": 1695,
      "priceCurrency": "AUD",
      "validThrough": "2026-10-09",
      "availability": "in-stock",
      "includes": [
        "Keynotes",
        "Two dedicated tracks",
        "Networking Reception",
        "Videos post-conference, on-demand",
        "Fully catered",
        "Plus the Workshop Day — 9 December"
      ],
      "url": "https://register.webdirections.org/ai-engineer-sydney-2026?tickets=conference-workshop:1"
    },
    {
      "id": "leadership",
      "name": "Leadership",
      "attendanceMode": "in-person",
      "price": 1795,
      "priceCurrency": "AUD",
      "validThrough": "2026-10-09",
      "availability": "in-stock",
      "includes": [
        "Exclusive Leadership track",
        "VIP Keynote seating",
        "Access to all other tracks",
        "Speaker Dinner",
        "Networking Reception",
        "Videos post-conference, on-demand",
        "Fully catered"
      ],
      "url": "https://register.webdirections.org/ai-engineer-sydney-2026?tickets=leadership:1"
    },
    {
      "id": "leadership-workshop",
      "name": "Leadership + Workshop Day",
      "attendanceMode": "in-person",
      "price": 2195,
      "priceCurrency": "AUD",
      "validThrough": "2026-10-09",
      "availability": "in-stock",
      "includes": [
        "Exclusive Leadership track",
        "VIP Keynote seating",
        "Access to all other tracks",
        "Speaker Dinner",
        "Networking Reception",
        "Videos post-conference, on-demand",
        "Fully catered",
        "Plus the Workshop Day — 9 December"
      ],
      "url": "https://register.webdirections.org/ai-engineer-sydney-2026?tickets=leadership-workshop:1"
    },
    {
      "id": "workshop",
      "name": "Workshop Day only",
      "attendanceMode": "in-person",
      "price": 795,
      "priceCurrency": "AUD",
      "validThrough": "2026-10-09",
      "availability": "in-stock",
      "includes": [
        "Workshop Day — 9 December",
        "A full day of hands-on, practical workshops",
        "Fully catered"
      ],
      "url": "https://register.webdirections.org/ai-engineer-sydney-2026?tickets=workshop:1"
    },
    {
      "id": "streaming",
      "name": "Streaming",
      "attendanceMode": "online",
      "price": 495,
      "priceCurrency": "AUD",
      "validThrough": "2026-10-09",
      "availability": "in-stock",
      "includes": [
        "Live stream — all sessions",
        "Live chat and Q&A",
        "Videos post-conference, on-demand"
      ],
      "url": "https://register.webdirections.org/ai-engineer-sydney-2026?tickets=streaming:1"
    }
  ],
  "registration": {
    "event": "ai-engineer-sydney-2026",
    "source": "https://register.webdirections.org/agent/events/ai-engineer-sydney-2026",
    "venue": "Hilton Hotel, Sydney, Australia",
    "teamOffer": {
      "summary": "5+ tickets in any mix of Conference and Conference + Workshop Day, at team prices, with one free upgrade to Leadership for every 5 tickets.",
      "minimum": 5,
      "upgrade_every": 5,
      "passes": [
        {
          "pass": "conference",
          "name": "Team — Conference",
          "tier": "Early Bird",
          "price": 995,
          "currency": "AUD",
          "gst_included": true,
          "price_ends": "2026-10-09",
          "next_price": {
            "tier": "Standard",
            "price": 1195,
            "from": "2026-10-10"
          },
          "available": true,
          "includes": [
            "Keynotes + two dedicated tracks",
            "Networking Reception",
            "Videos post-conference, on-demand",
            "Fully catered",
            "Team rate — $200 off every ticket"
          ]
        },
        {
          "pass": "conference-workshop",
          "name": "Team — Conference + Workshop Day",
          "tier": "Early Bird",
          "price": 1495,
          "currency": "AUD",
          "gst_included": true,
          "price_ends": "2026-10-09",
          "next_price": {
            "tier": "Standard",
            "price": 1695,
            "from": "2026-10-10"
          },
          "available": true,
          "includes": [
            "Everything in Conference",
            "Plus the Workshop Day — 9 December",
            "Team rate — $200 off every ticket"
          ]
        }
      ],
      "codes_accepted": false,
      "book": "https://register.webdirections.org/ai-engineer-sydney-2026/team"
    }
  },
  "links": {
    "website": "https://webdirections.org/ai-engineer/",
    "speakers": "https://webdirections.org/ai-engineer/#speakers",
    "schedule": null,
    "registration": "https://webdirections.org/ai-engineer/#register",
    "registrationLlms": "https://register.webdirections.org/llms.txt",
    "registrationLlmsMirror": "https://webdirections.org/llms.txt",
    "registrationApi": "https://register.webdirections.org/agent/events/ai-engineer-sydney-2026",
    "agents": "https://webdirections.org/ai-engineer/for-agents/",
    "llms": "https://webdirections.org/ai-engineer/llms.txt",
    "markdown": "https://webdirections.org/ai-engineer/index.md",
    "json": "https://webdirections.org/ai-engineer/conference.json",
    "sessionsJson": "https://data.webdirections.org/ai-engineer-sydney/sessions.json",
    "speakersJson": "https://data.webdirections.org/ai-engineer-sydney/speakers.json",
    "calendar": null,
    "mcp": "https://data.webdirections.org/ai-engineer-sydney/mcp"
  },
  "updatedAt": "2026-10-09T09:57:50.346Z"
}
