{"$schema": "https://c3voc.de/schedule/schema.json", "generator": {"name": "pretalx", "version": "2026.3.0.dev0", "url": "https://pretalx.com"}, "schedule": {"url": "https://pretalx.com/scipy-2026/schedule/", "version": "0.51", "base_url": "https://pretalx.com", "conference": {"acronym": "scipy-2026", "title": "SciPy 2026", "start": "2026-07-13", "end": "2026-07-19", "daysCount": 7, "timeslot_duration": "00:05", "time_zone_name": "US/Central", "colors": {"primary": "#002d5d"}, "rooms": [{"name": "Intro", "slug": "5835-intro", "guid": "2f29e03a-2f54-5ec6-96ac-9d0f86fc4356", "description": null, "capacity": null}, {"name": "Viz", "slug": "5834-viz", "guid": "36e5ee07-f760-5cfd-bacb-315638b76894", "description": null, "capacity": null}, {"name": "AI/ML", "slug": "5836-aiml", "guid": "05cd4401-7011-5ff3-8594-435e398e3500", "description": null, "capacity": null}, {"name": "Accelerated Computing", "slug": "5837-accelerated-computing", "guid": "bfa9288e-3927-57bc-b25f-87fd081098fa", "description": null, "capacity": null}, {"name": "Other", "slug": "5838-other", "guid": "f76ef0fb-9d4f-5b9d-a2da-dff64db2f835", "description": null, "capacity": null}, {"name": "Memorial Hall", "slug": "5839-memorial-hall", "guid": "551415a3-62b8-5495-bd07-a1eae3920fb6", "description": null, "capacity": null}, {"name": "Johnson Great Room", "slug": "5840-johnson-great-room", "guid": "7a5b1cea-0fba-5130-a43b-083922c50e02", "description": null, "capacity": null}, {"name": "Thomas Swain Room", "slug": "5841-thomas-swain-room", "guid": "2af548af-3c62-5aa4-ae05-2c2f94e4a7df", "description": null, "capacity": null}, {"name": "University Hall", "slug": "5842-university-hall", "guid": "5edbe1c7-2cdc-5019-8a11-b20de55c4811", "description": null, "capacity": null}, {"name": "Virtual Sessions", "slug": "6081-virtual-sessions", "guid": "19361295-a671-5028-989f-9fb8a0159534", "description": null, "capacity": null}], "tracks": [{"name": "Keynotes", "slug": "6430-keynotes", "color": "#00acc1"}, {"name": "General", "slug": "6425-general", "color": "#54524c"}, {"name": "Tutorials", "slug": "6426-tutorials", "color": "#000000"}, {"name": "Birds of a Feather (BoFs)", "slug": "6427-birds-of-a-feather-bofs", "color": "#ff69b4"}, {"name": "Lunch and Learn", "slug": "6429-lunch-and-learn", "color": "#2f4c79"}, {"name": "Spirit of SciPy", "slug": "6421-spirit-of-scipy", "color": "#a82d8f"}, {"name": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "slug": "6418-data-driven-discovery-machine-learning-and-artificial-intelligence", "color": "#e86252"}, {"name": "Physics and Astronomy", "slug": "6424-physics-and-astronomy", "color": "#9b5de5"}, {"name": "Environmental, Earth, and Climate Sciences", "slug": "6422-environmental-earth-and-climate-sciences", "color": "#f69240"}, {"name": "Scientific Computing in Education", "slug": "6423-scientific-computing-in-education", "color": "#c24e75"}, {"name": "Biological and Medical Sciences", "slug": "6420-biological-and-medical-sciences", "color": "#e9c46a"}, {"name": "Maintainers and Community", "slug": "6419-maintainers-and-community", "color": "#5aa9a3"}, {"name": "Lightning Talks", "slug": "6428-lightning-talks", "color": "#2e7d32"}, {"name": "Poster Session", "slug": "7301-poster-session", "color": "#ff69b4"}, {"name": "SciPy Tools", "slug": "7302-scipy-tools", "color": "#2f4c79"}, {"name": "Social Event", "slug": "7568-social-event", "color": "#f8bbd0"}], "days": [{"index": 1, "date": "2026-07-13", "day_start": "2026-07-13T04:00:00-05:00", "day_end": "2026-07-14T03:59:00-05:00", "rooms": {"Intro": [{"guid": "36c03e77-4651-5f90-83e6-0e75bbefb642", "code": "M8AZH8", "id": 93259, "logo": null, "date": "2026-07-13T08:00:00-05:00", "start": "08:00", "end": "2026-07-13T12:00:00-05:00", "duration": "04:00", "room": "Intro", "slug": "scipy-2026-93259-introduction-to-python-and-programming-room-hsec-3-110", "url": "https://pretalx.com/scipy-2026/talk/M8AZH8/", "title": "Introduction to Python and Programming (Room HSEC 3-110)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Enjoy a gentle introduction to Python for folks who are completely new to it and may not have much experience programming. Learn how to write Python while practicing loops, if\u2019s, functions, and usage of Python\u2019s built-in features in a series of fun, interactive exercises inside Jupyter Notebooks. By the end you\u2019ll be ready to write your own basic Python -- but most importantly, I want you to learn the form and vocabulary of Python so that you can understand Python documentation, interpret code written by others, and get the most out of other SciPy tutorials.\n\nInstallation Instructions: https://github.com/jiffyclub/scipy-2026-intro-to-python#setup-instructions", "description": "To make the most of SciPy it helps to have some basic familiarity with the Python language itself. This beginner level tutorial is designed for folks who are brand-new to Python and may not even have much programming experience. I\u2019ll help you get a working Python installation in which you can launch Jupyter Notebooks, a common tool used in scientific research with Python and in SciPy tutorials.\n\nAttendees will learn to work with Python variables, the object interface, loops, conditional statements, function definitions, and the use of basic Python data structures through hands-on exercises inside of Jupyter. Students will use the ipythonblocks library to manipulate an image-like grid of colors for immediate, interactive feedback that makes it easy to tell whether code had the intended effect.\n\nMy goal is for you to leave the tutorial with a basic familiarity with Python (and a working Python installation) that helps you focus on the scientific libraries you\u2019ll learn about in the other tutorials and throughout SciPy. Familiarity with the usage and features of Jupyter will also help you dive headfirst into other tutorials.", "recording_license": "", "do_not_record": false, "persons": [{"code": "7Q9H8Q", "name": "Matt Davis", "avatar": null, "biography": "Matt has been using Python to work with data in science and at startups since 2008, after getting degrees in Astronomy and Aerospace Engineering. He maintains some moderately popular open-source Python libraries, including SnakeViz and Palettable. Today Matt is the lead software engineer at Populus, a startup helping city governments manage various aspects of transportation.", "public_name": "Matt Davis", "guid": "8067f177-3e3b-58e2-b33e-cc87d71b95bd", "url": "https://pretalx.com/scipy-2026/speaker/7Q9H8Q/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/M8AZH8/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/M8AZH8/", "attachments": []}], "Viz": [{"guid": "9c3132f5-98d7-5c2d-8450-d3500c938e29", "code": "TBVN9E", "id": 93214, "logo": null, "date": "2026-07-13T08:00:00-05:00", "start": "08:00", "end": "2026-07-13T12:00:00-05:00", "duration": "04:00", "room": "Viz", "slug": "scipy-2026-93214-interactive-computing-with-marimo-and-anywidget-room-hsec-2-138", "url": "https://pretalx.com/scipy-2026/talk/TBVN9E/", "title": "Interactive computing with marimo and anywidget (Room HSEC 2-138)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "This tutorial is for anyone who works with data in Python notebooks. [marimo](https://marimo.io) is a reactive notebook that can serve as a personal data environment. Cells run in a deterministic order based on their dependencies, interactivity is built in, and notebooks are self-contained Python scripts you can share, version, and deploy. We start with a tour of marimo: its execution model, interactive elements, SQL, plotting, and sharing. From there, we get practical, composing off-the-shelf UI elements to build interactive tools for your data, then creating your own custom widgets with [anywidget](https://anywidget.dev) when you need something tailored to your workflow.\n\nInstallation Instructions: https://github.com/manzt/scipy-2026-anywidget", "description": "marimo is a reactive Python notebook. Cells declare dependencies through variable references, and the runtime executes them in a deterministic order. Notebooks are stored as pure Python scripts with inline dependency metadata (PEP 723), so they carry everything needed to reproduce themselves. Prior web experience is not required for most of the tutorial, but familiarity with HTML, CSS, and JavaScript will help in the custom widget sections.\n\nThe tutorial is split into four sections.\n\n**A tour of marimo.** Covers the reactive execution model, cell types (Python, SQL, markdown), plotting (matplotlib, Altair), and built-in UI elements (sliders, dropdowns, tables, interactive charts, dataframe explorers). Shows how elements compose across cells: a dropdown drives a chart, a selection filters a dataframe, a table displays the result. Also covers grouping patterns (arrays, dictionaries, batch, form), layout (tabs, accordion, sidebar, grid), and editor features like the dependency graph, package management, and app view.\n\n**Building custom widgets with anywidget.** Sometimes you need a specialized view of your data, a custom visualization to explore a relationship, or an interface tailored to a specific analysis. anywidget lets you build these: you define an ESM module and a Python class, and it handles the communication between Python and the browser. Covers one-way and two-way data bindings, importing third-party JavaScript libraries (D3, Leaflet), and modern JavaScript fundamentals. Each exercise builds on the last. Custom widgets participate in marimo's dataflow graph like any built-in element.\n\n**Beyond the notebook.** Covers how to get your work out of the editor and in front of others. marimo notebooks are stored as Python files and can be executed as standalone scripts (e.g., `uv run my_notebook.py`), but also viewed and shared in a variety of ways depending on audience (e.g., interactive web app, exported as a PDF, turned into slides). They can be versioned on GitHub, published as gists, exported as WASM pages, or shared on molab (https://molab.marimo.io), a cloud-hosted workspace for marimo notebooks.\n\n**Open exploration.** Q/A, advanced topics (packaging widgets to PyPI, the widget ecosystem), or one-on-one help. Contact us in advance with your project so we can plan accordingly.\n\n### Learning Goals\n\nAfter this tutorial, attendees will be able to:\n- Work in marimo's reactive execution model\n- Compose built-in interactive elements across cells\n- Build custom widgets with anywidget\n- Share and deploy marimo notebooks in multiple formats\n\n### Prerequisites\n\nAttendees should have basic understanding of:\n- Python: imports, if statements, for loops, function definitions, class definitions, return statements\n- Python environments: ability to create a new environment for the tutorial\n- Notebooks: launch a notebook, code in cells, execute code\n- Python data science libraries: basic knowledge of NumPy arrays and Pandas DataFrames\n- Web fundamentals (custom widget sections only): basic JavaScript (functions, arrow functions, async) and basic DOM manipulation. No need to be an expert; we cover what you need.\n\n### Outline\n\n**Part 1: A tour of marimo (~60 min)**\n- Reactivity: cells, variables, and the dataflow graph\n- Cell types: Python, SQL, markdown\n- Plotting: matplotlib, Altair\n- UI elements: sliders, dropdowns, number inputs, tables, interactive charts, dataframe explorer\n- Composing elements across cells: a dropdown drives a chart, a selection filters a dataframe\n- Grouping patterns: `mo.ui.array`, `mo.ui.dictionary`, `mo.ui.batch`, `mo.ui.form`\n- Layout: `mo.hstack`, `mo.vstack`, tabs, accordion, sidebar, grid\n- Editor features: dependency graph, package management, app view\n- Hands-on: build an interactive data explorer\n\n**Part 2: Building custom widgets with anywidget (~60 min)**\n- Motivation: specialized views, custom visualizations, tailored interfaces\n- Modern JavaScript fundamentals: ESM, web platform APIs, the DOM\n- How Python and the browser communicate (widget protocol)\n- \"Hello world\" anywidget (one-way data binding)\n- Counter widget (two-way data binding)\n- Importing third-party JavaScript libraries (e.g., D3, Leaflet)\n- Accessing selections, serializing dataframes, sending binary data\n- Composing custom widgets with built-in elements in the dataflow graph\n- Hands-on: build a custom widget for a specific data task\n\n**Part 3: Beyond the notebook (~30 min)**\n- Running notebooks headless as scripts\n- Inline dependencies (PEP 723) and sandboxed execution with `uv`\n- App mode: serving notebooks as interactive apps with `marimo run`\n- Exporting: PDFs with rich outputs, slides\n- Sharing: molab, GitHub gists, WASM standalone pages\n- Hands-on: export and share a notebook in multiple formats\n\n**Part 4: Open exploration (~30 min)**\n- Packaging and publishing widgets to PyPI\n- Tour of the anywidget/widget ecosystem\n- Q/A and one-on-one help with personal projects", "recording_license": "", "do_not_record": false, "persons": [{"code": "7D3LYU", "name": "Trevor Manz", "avatar": "https://pretalx.com/media/avatars/XZRHLS_DpLy6QB.webp", "biography": "Trevor is a researcher and software engineer working on interactive computing tools in Python. He completed his PhD in the HIDIVE Lab at Harvard Medical School, where he developed interactive visualization and analysis tools for biological and AI applications. He created anywidget and contributes to open-source data tooling. Trevor now works at marimo, building a reactive notebook environment for Python. He lives in Brooklyn, NY with his two cats, Laird and Minnow, whom he\u2019s very fond of.", "public_name": "Trevor Manz", "guid": "0ee2c78d-97cc-5d77-8aa4-0013a2c86f26", "url": "https://pretalx.com/scipy-2026/speaker/7D3LYU/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/TBVN9E/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/TBVN9E/", "attachments": []}, {"guid": "0928ebe1-63b2-5c2c-a4e4-1056b6c6a5d4", "code": "FVRNKP", "id": 92514, "logo": null, "date": "2026-07-13T13:30:00-05:00", "start": "13:30", "end": "2026-07-13T17:30:00-05:00", "duration": "04:00", "room": "Viz", "slug": "scipy-2026-92514-create-custom-image-visualization-and-analysis-tools-with-napari-room-hsec-2-110", "url": "https://pretalx.com/scipy-2026/talk/FVRNKP/", "title": "Create custom image visualization and analysis tools with napari (Room HSEC 2-110)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "With everything from microscopes to telescopes to satellites, scientists produce image data in countless formats, shapes, sizes, and dimensions. Python provides a rich ecosystem of libraries to make sense of them. napari is a Python library for multidimensional image visualization, but it does double duty as a standalone application that can be easily extended with GUI tools for analysis, visualization, and annotation. In this tutorial, we'll start with the basics of image visualization and analysis in napari, then show how to extend the napari user interface to make analysis workflows as easy as pushing a button, and finally show how to share these extensions as *plugins*, which can be easily installed by users and collaborators. If you work with images (particularly multidimensional images), and especially if you work with scientists who may not be comfortable with Python, this tutorial might be for you!\n\nInstallation Instructions: https://napari.org/workshops/extend/setup/", "description": "Just like we take more pictures of food than we will ever look at, scientists are using powerful microscopes, telescopes, satellites, MRI machines and myriad other sensors to produce more images than they can ever look at. These images come in different file formats, they might be 3D, contain a time-lapse component, many different channels, or other features that increase the complexity of loading them for visualization. Even when specialized viewers provide ways to load these images and look at them, analyzing, interacting with, and visualizing the results of these analyses can still be a challenge.\n\nThis tutorial is aimed at folks who have some experience in scientific computing with Python. To get the most out of it, you should be familiar with NumPy arrays, Jupyter notebooks, and Python scripts. Ideally, you should have some idea of how images can be represented as arrays of numbers, and the types of analyses that might be performed on these arrays e.g. filtering and segmentation. You don\u2019t necessarily need to be familiar with how these tools and methods work - it\u2019s enough to know that they are out there!\n\nThe tutorial will be split into three main parts, each around an hour to 75 minutes long. Each part will cover a different aspect of how napari can be used to simplify your analysis workflows, and the workflows of your colleagues and coworkers. \n\n**Part 1: Using Python and napari to view and analyze imaging data**\nIn this section we will look at opening and viewing 2D, 3D and even 4D images in napari. We will see how different layer types can help you display your analysis results, how Jupyter notebooks can streamline your image processing, and how napari\u2019s plugins can help you access different analyses through the napari viewer.\n\n**Part 2: Customizing your analysis workflow by extending napari\u2019s functionality**\nWe will teach you how to customize your analysis workflow by adding new keybindings and mouse bindings to napari, and adding event handlers that can listen for different layer and viewer events. Finally, we will show you how easy it can be to add your own GUI widgets with minimal code.\n\n**Part 3: Distributing your customized functionality with plugins**\nOnce you\u2019re happy with your customized analysis tools, you may want to distribute them to other colleagues and coworkers, or to napari users at large! This section will cover how to package your custom bits of code into pip-installable napari plugins.", "recording_license": "", "do_not_record": false, "persons": [{"code": "TXRYUQ", "name": "Tim Monko", "avatar": "https://pretalx.com/media/avatars/ANMQRF_B4QUTlT.webp", "biography": "I am a full-time maintainer and community manager of napari, an interactive multi-dimensional Python image and data viewer, and its plugin ecosystem. I work to extend the plugin ecosystem and help scientists achieve their goals with image analysis.", "public_name": "Tim Monko", "guid": "6e7a85e7-db43-5cee-a74a-aa6865f5e5ea", "url": "https://pretalx.com/scipy-2026/speaker/TXRYUQ/"}, {"code": "3JTZ8R", "name": "Ashley Anderson", "avatar": null, "biography": "I am one of the less-active napari core devs, with most of my contributions these days coming through vispy or on the web infrastructure (npe2api and napari hub). I first started working on napari as part of the CZI Imaging Tech team, but I now participate primarily in my spare time.\n\nI got into Python and open source while studying Medical Physics at UW-Madison. After a few years in academia at Barrow Neurological Institute (Phoenix, AZ), I made the switch to industry. I first worked as an on-site clinical MRI scientist for Philips (Mayo Clinic, Rochester, MN), then joined a low-field MRI startup called Hyperfine (Guilford, CT). I'm now a remote worker for Biohub and still reside in Guilford, CT; but I'm in the middle of a move to Dobbs Ferry, NY.", "public_name": "Ashley Anderson", "guid": "1e8d7755-86a3-5adc-9ff7-662c8976d11f", "url": "https://pretalx.com/scipy-2026/speaker/3JTZ8R/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/FVRNKP/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/FVRNKP/", "attachments": []}], "AI/ML": [{"guid": "efcce81b-9186-54e8-9c1a-532883bb5b81", "code": "BMPMUR", "id": 92510, "logo": null, "date": "2026-07-13T08:00:00-05:00", "start": "08:00", "end": "2026-07-13T12:00:00-05:00", "duration": "04:00", "room": "AI/ML", "slug": "scipy-2026-92510-building-a-deep-research-agent-room-hsec-3-150", "url": "https://pretalx.com/scipy-2026/talk/BMPMUR/", "title": "Building A Deep Research Agent (Room HSEC 3-150)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Through the construction of a Deep Research Agent, tutorial participants will learn the fundamental building blocks of LLM-driven applications. Starting with in-context learning and prompt design, we will progress through memory management, tool integration via the Model Context Protocol (MCP), and planning workflows. Participants will build a working agent that can query a Zotero citation library, synthesize literature summaries, and engage in multi-turn research conversations. We will also discuss failure modes, limitations, and the role of such agents in an age of coding assistants.\n\nInstallation Instructions: https://github.com/ericmjl/build-deep-research-agent/", "description": "In this tutorial, we will walk you through the practical construction of a Deep Research Agent - an LLM-powered system that can search, summarize, and synthesize scientific literature from a Zotero library. While building agents can seem daunting, breaking it down into core components makes it approachable.\n\nWe will start with what we think is the most intuitive way to understand agents - seeing them as LLM-backed systems with memory, tools, and planning capabilities. From there, we will show you how to build each component: crafting effective prompts for research tasks, managing conversation state, connecting to external tools via MCP, and implementing both deterministic and ReAct-style planning workflows.\n\nBased on our experience building research agents, we've designed a progression that builds a fully functional single agent. We will also demonstrate how specialized agents can collaborate on literature review tasks. Tutorial participants will leave with a working agent and the knowledge to customize it for their own research workflows.\n\nThis tutorial is structured based on what we wished we knew when we first started building LLM agents, and is ordered for maximum productivity in learning. By the end of the tutorial, participants should be able to build and customize their own research agents!", "recording_license": "", "do_not_record": false, "persons": [{"code": "DFNKWR", "name": "Benjamin Batorsky", "avatar": "https://pretalx.com/media/avatars/G3VLZT_q95id3u.webp", "biography": "Ben is the Responsible AI Lead at Philips, working on building AI governance tools and policies.  Previously, he led data science teams in academia, industry and government.  He has focused primarily on language technology and has delivered tutorials and designed classes on LLMs and agent development.  He contributes to the larger data science community through independent projects and leading the PyData Boston chapter.", "public_name": "Benjamin Batorsky", "guid": "86bf0a66-8680-535f-9de5-5e12eb6b7080", "url": "https://pretalx.com/scipy-2026/speaker/DFNKWR/"}, {"code": "9NRRJH", "name": "Eric Ma", "avatar": "https://pretalx.com/media/avatars/EP39HL_u5E0WzK.webp", "biography": "As Senior Principal Data Scientist at Moderna Eric leads the Data Science and Artificial Intelligence (Research) team to accelerate science to the speed of thought. Prior to Moderna, he was at the Novartis Institutes for Biomedical Research conducting biomedical data science research with a focus on using Bayesian statistical methods in the service of discovering medicines for patients. Prior to Novartis, he was an Insight Health Data Fellow in the summer of 2017 and defended his doctoral thesis in the Department of Biological Engineering at MIT in the spring of 2017.\n\nEric is also an open-source software developer and has led the development of pyjanitor, a clean API for cleaning data in Python, and nxviz, a visualization package for NetworkX. He is also on the core developer team of NetworkX and PyMC. In addition, he gives back to the community through code contributions, blogging, teaching, and writing.\n\nHis personal life motto is found in the Gospel of Luke 12:48.", "public_name": "Eric Ma", "guid": "0c3ba6a2-c8b9-5f73-9828-d3217a7a15e6", "url": "https://pretalx.com/scipy-2026/speaker/9NRRJH/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/BMPMUR/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/BMPMUR/", "attachments": []}, {"guid": "1ec1ecad-c636-59b5-8fcf-bcc3da48dc5a", "code": "HFWAYG", "id": 92304, "logo": null, "date": "2026-07-13T13:30:00-05:00", "start": "13:30", "end": "2026-07-13T17:30:00-05:00", "duration": "04:00", "room": "AI/ML", "slug": "scipy-2026-92304-intro-to-safe-reliable-and-maintainable-ai-apps-in-python-room-hsec-3-150", "url": "https://pretalx.com/scipy-2026/talk/HFWAYG/", "title": "Intro to Safe, Reliable, and Maintainable AI Apps in Python (Room HSEC 3-150)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Large Language Models (LLMs) are transforming how we build applications, but the path from \"cool demo\" to \"production-ready tool\" is littered with challenges: hallucinations, verifiability, and more. This tutorial covers the basics of how to build AI apps that avoid these challenges, yet are still effective and simple to build.\n\nWe'll start with [**querychat**](https://pypi.org/project/querychat/), an open-source package that lets users explore data through natural language. querychat demonstrates a powerful pattern: rather than letting an LLM access raw data directly (where it can hallucinate calculations), it constrains the LLM to generate SQL queries that are displayed and executed by a proper database engine. This \"tool-based\" architecture ensures reliability through transparency and precision -- users see exactly what query was executed along with it's exact results.\n\nFrom there, we'll peel back the layers to reveal [**chatlas**](https://pypi.org/project/chatlas/), the foundation powering querychat. chatlas provides a unified, provider-agnostic interface to 19+ LLM providers (OpenAI, Anthropic, Google, local models via Ollama, and more). You'll learn how chatlas makes it trivial to:\n\n- Build multi-turn conversations with history management\n- Stream responses in real-time for responsive UIs\n- Switch between providers with minimal code changes\n- Define custom tools that let LLMs interact with external systems (safely)\n- Extract structured data using Pydantic models\n\nBy the end of this tutorial, we'll have built two complete apps: a data exploration chatbot (using querychat with your own data) and a custom AI assistant with tools you define. You'll leave with practical patterns for constraining LLM behavior, validating outputs, and building apps that are genuinely useful, maintainable and production ready.\n\nInstallation Instructions: Go to [dev.workshop.posit.team](dev.workshop.posit.team) and sign in prior to the workshop. This will ensure you can access the provided computing environment for the tutorial. Once logged in, click \"New Session\", then \"Launch\". You may see a blank page for a minute before being directed to a hosted [Positron session](https://positron.posit.co/) (https://positron.posit.co/). If you run into issues, or have any questions, please email carson@posit.co.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "ZDSTNC", "name": "Carson Sievert", "avatar": null, "biography": "Carson is currently a Principal Software Engineer at [Posit Software, PBC](https://posit.co/). He's an original author and maintainer of projects such as [shiny](https://shiny.posit.co/py/), [shinywidgets](https://shiny.posit.co/py/docs/jupyter-widgets.html), [shinylive](https://shinylive.io/py/examples/), and [chatlas](https://github.com/posit-dev/chatlas/). Prior to joining Posit, Carson was an engineer at [Plotly](https://plotly.com/) for numerous years, won the [ASA's Chambers Award](https://community.amstat.org/jointscsg-section/awards/john-m-chambers), and received his PhD in 2017.", "public_name": "Carson Sievert", "guid": "547196f7-b1af-54f2-814a-cb598828c1c5", "url": "https://pretalx.com/scipy-2026/speaker/ZDSTNC/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/HFWAYG/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/HFWAYG/", "attachments": []}], "Accelerated Computing": [{"guid": "d726d70e-032e-5d4b-ad3e-fd90ca0302a7", "code": "SPXK7T", "id": 90434, "logo": "https://pretalx.com/media/scipy-2026/submissions/SPXK7T/image_8cmmJQF.webp", "date": "2026-07-13T08:00:00-05:00", "start": "08:00", "end": "2026-07-13T12:00:00-05:00", "duration": "04:00", "room": "Accelerated Computing", "slug": "scipy-2026-90434-accelerated-python-math-libraries-room-hsec-2-110", "url": "https://pretalx.com/scipy-2026/talk/SPXK7T/", "title": "Accelerated Python Math Libraries (Room HSEC 2-110)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "GPU-powered math libraries are the core of accelerated scientific computing.  The nvmath-python package aims to provide intuitive pythonic APIs giving users full access to all features offered by NVIDIA's libraries in a variety of execution spaces.  It is your one-stop shop for Pythonic math libraries on the GPU.\n\nInstallation Instructions: We will provide Nvidia Brev cloud instances.  Attendees will only need their laptops and an Internet connection.", "description": "In this hands-on tutorial, we will explore the nvmath-python library, bringing the power of the CUDA-X math libraries to Python.  You will learn:\n- The landscape of CUDA Python libraries\n- nvmath-python host, device, and distributed APIs\n- How nvmath-python interoperates with existing array/tensor libraries", "recording_license": "", "do_not_record": false, "persons": [{"code": "Z9ENP8", "name": "Katrina Riehl", "avatar": "https://pretalx.com/media/avatars/Z9ENP8_ml6Zw2v.webp", "biography": "Dr. Katrina Riehl is a Principal Technical Product Manager at NVIDIA leading the CUDA Education program. For over two decades, Katrina has worked extensively in the fields of scientific computing, machine learning, data science, and visualization. Most notably, she has helped lead data initiatives at the University of Texas Austin Applied Research Laboratory, Anaconda, Apple, Expedia Group, Cloudflare, and Snowflake. She is an active volunteer in the Python open-source scientific software community and currently serves on the Advisory Council for NumFOCUS.", "public_name": "Katrina Riehl", "guid": "885c34b1-3992-5e82-988f-01bce678c58b", "url": "https://pretalx.com/scipy-2026/speaker/Z9ENP8/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/SPXK7T/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/SPXK7T/", "attachments": []}, {"guid": "bf148705-add1-53b2-9d07-04394781ff79", "code": "9FQMMN", "id": 88569, "logo": null, "date": "2026-07-13T13:30:00-05:00", "start": "13:30", "end": "2026-07-13T17:30:00-05:00", "duration": "04:00", "room": "Accelerated Computing", "slug": "scipy-2026-88569-reproducible-cuda-accelerated-workflows-for-scientists-with-pixi-room-hsec-2-138", "url": "https://pretalx.com/scipy-2026/talk/9FQMMN/", "title": "Reproducible CUDA Accelerated Workflows for Scientists with Pixi (Room HSEC 2-138)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Scientific researchers need reproducible software environments for complex applications that can run across heterogeneous computing platforms. Modern open source tools, like [Pixi](https://pixi.sh/), provide automatic reproducibility solutions for all dependencies while providing a high level interface well suited for researchers.\n\nThis tutorial will provide a practical introduction to using Pixi to easily create scientific and AI/ML environments that benefit from hardware acceleration, across multiple machines and platforms. The focus will be on CUDA applications, such as machine learning frameworks and use of CUDA Tile, as well as using pixi-build to construct bespoke CUDA enabled conda packages.\n\nInstallation Instructions: https://matthewfeickert-talks.github.io/reproducible-cuda-workflows-with-pixi-scipy-2026/setup/", "description": "As artificial intelligence (AI) and machine learning (ML) becomes a modern part of the scientific toolkit, the need to have robustly reproducible scientific computing environments that support hardware acceleration, e.g. with CUDA, becomes more important. However, historically just installing a working CUDA environment on a single machine, let alone on multiple platforms with different requirements, could be a difficult task for non-experts. This led to many scientific machine learning workflows being reliably runnable on only particular machines, and, even worse, with environments that were not reproducible across time.\n\nWith significant recent advancements by the NVIDIA open source team and the conda-forge open source community, the entire CUDA stack &mdash; from compilers to runtime libraries &mdash; is now distributed on conda-forge. This significantly reduces the overhead to _install_ CUDA dependencies, but packaging and distribution of binaries alone does not solve the problem of reproducibility. With automatic multi-platform hash-level lock file support for all dependencies that are available on package indexes (like PyPI and conda-forge), highly efficient solving strategies, and high level user interfaces, Pixi provides a missing piece to the scientific researcher toolkit. With Pixi, researchers are able to easily specify the hardware acceleration requirements they have, multiple different computational environments needed for their experiments, and the required software dependencies, and then quickly solve for a multi-platform lock file of all the dependencies required, down to the compiler level. This makes it possible to have multiple hardware accelerated environments defined that are able to run hardware accelerated workflows across heterogeneous machines with different GPU types and CUDA compatibility.\n\nThis tutorial will be targeted to scientific researchers who use Python for scientific computing and use hardware accelerated workflows in their research, with a particular focus on AI/ML. No prior expertise with hardware accelerator systems is assumed. The tutorial structure will begin with an introduction to Pixi as a computational environment manager, and explore how it provides features beyond other more common package managers that might be used for Python dependencies. It will then extend to adding CUDA requirements to Pixi environments, and provide participants with exercises for solving environments and running simple AI/ML workflows using the PyTorch machine learning library and the [cuTile Python library](https://docs.nvidia.com/cuda/cutile-python/). The tutorial will then move towards more complex environment requirements in later exercises. The tutorial will conclude with examples and exercises on building bespoke CUDA enabled conda packages with pixi-build.\n\nTutorial participants will code all examples themselves. Participants will also be given time to explore solutions to their own hardware accelerated Python workflows. To make the tutorial more practical and interactive, NVIDIA has agreed to donate cloud GPU resources on the [NVIDIA Brev](https://developer.nvidia.com/brev) platform, which will allow for participants to have CUDA enabled GPU resources to run their own examples on.", "recording_license": "", "do_not_record": false, "persons": [{"code": "H8ZFYG", "name": "Matthew Feickert", "avatar": "https://pretalx.com/media/avatars/H8ZFYG_0B3tdYV.webp", "biography": "Matthew is a research scientist in experimental high energy physics and data science at the University of Wisconsin-Madison Data Science Institute (a \u201cdata physicist\u201d). He works as a member of the ATLAS collaboration on searches for physics beyond the standard model with experiments performed at CERN's Large Hadron Collider (LHC) in Geneva, Switzerland. He also serves on the executive board of the Institute for Research and Innovation in Software for High Energy Physics (IRIS-HEP) where he is a researcher and the Analysis Systems Area lead. He is also a topical editor for physics and data science for the Journal of Open Source Software. He previously did his Ph.D. (2019) research at Southern Methodist University, also on the ATLAS experiment, and was a postdoc at the University of Illinois at Urbana-Champaign, and the University of Wisconsin-Madison.", "public_name": "Matthew Feickert", "guid": "24bbffd4-66b8-5c0f-8c5f-c38640f4d575", "url": "https://pretalx.com/scipy-2026/speaker/H8ZFYG/"}, {"code": "EKHRSA", "name": "Ruben Arts", "avatar": "https://pretalx.com/media/avatars/EKHRSA_R8Mr2L8.webp", "biography": "Ruben is part of the Prefix.dev core team, builing Pixi and other tools in the package management space. Originally he's a Robotics engineer working on industrial robots, but quickly figuring out that solving development and deployment problems were one of the bigger issues that robotics developers had to deal with. Joining Prefix.dev allowed him to focus on improving the UX/DX of a large group of software engineers. Over the years he's been doing multiple talks and workshops on how to properly manage software and development workflows.", "public_name": "Ruben Arts", "guid": "116518af-d223-54bc-b652-fa3ad14a6897", "url": "https://pretalx.com/scipy-2026/speaker/EKHRSA/"}, {"code": "Z9ENP8", "name": "Katrina Riehl", "avatar": "https://pretalx.com/media/avatars/Z9ENP8_ml6Zw2v.webp", "biography": "Dr. Katrina Riehl is a Principal Technical Product Manager at NVIDIA leading the CUDA Education program. For over two decades, Katrina has worked extensively in the fields of scientific computing, machine learning, data science, and visualization. Most notably, she has helped lead data initiatives at the University of Texas Austin Applied Research Laboratory, Anaconda, Apple, Expedia Group, Cloudflare, and Snowflake. She is an active volunteer in the Python open-source scientific software community and currently serves on the Advisory Council for NumFOCUS.", "public_name": "Katrina Riehl", "guid": "885c34b1-3992-5e82-988f-01bce678c58b", "url": "https://pretalx.com/scipy-2026/speaker/Z9ENP8/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/9FQMMN/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/9FQMMN/", "attachments": []}], "Other": [{"guid": "30d36ebd-2ce9-54f1-b069-0978a651339c", "code": "CLBN3Z", "id": 92109, "logo": "https://pretalx.com/media/scipy-2026/submissions/CLBN3Z/image_N1um6yT.webp", "date": "2026-07-13T08:00:00-05:00", "start": "08:00", "end": "2026-07-13T12:00:00-05:00", "duration": "04:00", "room": "Other", "slug": "scipy-2026-92109-one-language-to-rule-them-all-developing-reactive-scientific-web-apps-in-pure-python-with-tethys-platform-room-hsec-4-103-5", "url": "https://pretalx.com/scipy-2026/talk/CLBN3Z/", "title": "One Language to Rule Them All: Developing Reactive, Scientific Web Apps in Pure Python with Tethys Platform (Room HSEC 4-103/5)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Bridging the gap between new scientific findings and an accessible decision-support tool often requires researchers to either hire a web developer or self-navigate a likely unfamiliar and fragmented landscape of JavaScript frameworks, HTML templating, and CSS. Tethys Platform, a free and open source Python software package, helps bridge that gap by providing a Python-heavy development stack designed specifically for geoscientific and environmental web applications.\n\nThis tutorial introduces the latest evolution of Tethys Platform: Tethys Component Apps. By integrating ReactPy, Tethys Platform builds on the shoulders of giants to facilitate the development of rich, robust, and reactive user interfaces entirely in Python\u2014eliminating the need for separate scripts and frontend languages. If you have enough Python prowess to write code for your scientific workflows, you can harness it to develop web applications that leverage and showcase these existing workflows.\n\nIn this hands-on session, participants will learn basic concepts of Reactive, Pythonic component web app development while building a basic, scientific app, step-by-step. These basic concepts include:\n  - Reusable Web Components and UI Design\n  - User Interactions and Event Handling\n  - Application State Management\n  - User Experience (UX)\n  - Integrating 3rd Party Web Component Libraries\n\n## Prerequisites\nAn intermediate knowledge of Python is recommended. Familiarity with basic web development concepts is helpful but not required.\n\nInstall and deploy a local Tethys Portal using Conda or Pip (see [Tethys Quickstart](https://docs.tethysplatform.org/en/stable/#quick-start)). Time estimate: <10 minutes.\nClone and install the Component Playground application ([GitHub clone link]([url](https://github.com/shawncrawley/tethysapp-component_playground.git))) into your local Tethys Portal (see [Development Installation]([url](https://docs.tethysplatform.org/en/stable/recipes/scaffold_an_app_via_command_line.html#development-installation))). Time estimate: <10 minutes.\n\nInstallation Instructions: https://gist.github.com/mwcraig/1baeeee30055e6deb8b5addc4846b702", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "CP9YCG", "name": "Shawn Crawley", "avatar": "https://pretalx.com/media/avatars/KDCRPU_u2kQiAX.webp", "biography": "Shawn Crawley is an Associate Software Engineer at Lynker where he primarily supports the Geospatial Intelligence Division of the National Water Center under NOAA's Office of Water Prediction. His expertise includes automating and optimizing GIS-data-driven workflows based in SQL, Python and JavaScript. He received his Master's Degree in Civil and Environmental Engineering from Brigham Young University with a focus on GIS and Hydroinformatics.", "public_name": "Shawn Crawley", "guid": "14bed788-b39e-52de-85a3-553a8ca71d8f", "url": "https://pretalx.com/scipy-2026/speaker/CP9YCG/"}], "links": [{"title": "Tethys Platform QuickStart Installation", "url": "https://docs.tethysplatform.org/en/stable/#quick-start", "type": "related"}, {"title": "Component Playground App GitHub Repository", "url": "https://github.com/shawncrawley/tethysapp-component_playground", "type": "related"}, {"title": "Tethys Platform GitHub Repository", "url": "https://github.com/tethysplatform/tethys/", "type": "related"}, {"title": "Tethys Platform Official Website", "url": "https://www.tethysplatform.org/", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/CLBN3Z/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/CLBN3Z/", "attachments": []}, {"guid": "3e328ff1-fc39-5d48-8062-1643a8489b13", "code": "GB3N9K", "id": 87948, "logo": "https://pretalx.com/media/scipy-2026/submissions/GB3N9K/causal_dag_KjBC6yt_LDzMK_0H0EjOK.webp", "date": "2026-07-13T13:30:00-05:00", "start": "13:30", "end": "2026-07-13T17:30:00-05:00", "duration": "04:00", "room": "Other", "slug": "scipy-2026-87948-introduction-to-causal-inference-room-hsec-3-110", "url": "https://pretalx.com/scipy-2026/talk/GB3N9K/", "title": "Introduction to Causal Inference (Room HSEC 3-110)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "This tutorial session is intended to give attendees a gentle introduction to applying causal thinking and inference using python. Causal data analysis is very common in many academic domains (e.g. in social psychology, epidemiology, macroeconomics, public policy research, sociology, and more) as well as in industry (all of the largest Silicon Valley tech companies employ teams of scientists who answer business questions purely with causal inference methods).\n\nThe tutorial will involve a combination of presentations with open Q&A and hands-on exercises contained in Marimo notebooks. This session will cover the difference between correlation and causation, the pitfalls of conducting an analysis using observational data, how causal inference can help get around these pitfalls, and examples of common, modern modeling approaches using the latest python causal inference frameworks (e.g. DoWhy). After the tutorial, the attendees should have a good foundational understanding of causality and the ability to confidently explore the topic on their own. Causal inference can be a very theory-heavy topic, making it impenetrable to novices. In this tutorial, we'll aim to take a more practical perspective on causal inference, while still occasionally touching on the theory.\n\nTutorial participants are not expected to be familiar with causal inference before attending, but we hope they have an earnest curiosity to learn about it! To get the most out of the session, the participants ought to have experience working with the common python data stack: matplotlib, numpy, pandas, and scikit-learn. Attendees should have some experience conducting classic machine learning modeling using the scikit-learn API, although having advanced machine learning expertise is absolutely not a prerequisite. A very basic understanding of statistics would be helpful (e.g. understanding what a mean is, what confidence intervals represent).\n\nMaterials and installation instructions can be found here: https://github.com/ronikobrosly/scipy_2026_causal_inference_tutorial", "description": "* Course Introduction (5 minutes)\n    * Introduce myself and course format\n    * Poll: Poll learners\u2019 comfort level with topic so I can fine-tune pacing and descriptions to match.\n    * Outline goals of training session\n\n* Two initial examples of causal inference problems (10 minutes):\n    * Hotel bookings and prices\n    * Customers quitting a subscription and receiving a special deal\n\n* Introducing causal thinking and causal graphs (60 min)\n    * Counterfactuals\n    * Thinking of counterfactuals as a missing data problem.\n    * Experiments and their limitations\n    * The hierarchy of statistical associations, causal inference, and experiments\n    * Causal inference vs typical ML questions\n    * Causal graphs\n        * Explaining the basics\n        * GROUP EXERCISE: Audience helps me build a causal graph by shouting out answers (car insurance example) \n    * The 3 primary types of causal relationships:\n        * Confounding\n        * Colliding\n        * Mediation\n\n* Notebook 1 exercises: Exploring causal graphs and relationships (20 minutes)\n\n* Break (20 min)\n\n* Causal thinking continued (20 minutes):\n    * A suggested workflow\n    * Assumptions of causal inference (30 min)\n    * GROUP EXERCISE: I talk through 4 bad examples of causal inference work, and audience shouts out the violated assumptions\n    \n* Causal inference analyses (30 minutes):\n    * Metrics:\n        * A reminder about counterfactuals\n        * Walk through all of the flavors of average treatment effect (ATE)\n    * Interrupted Time Series\n    * Difference in differences\n    * Bayesian structural time series\n    * Propensity Score Matching (PSM)\n        * Talk through how PSM looks when using a dataset\n    *  Metalearners (S-learner / G-computation)\n        * Talk through an example\n\n* Notebook 2 exercises: Metalearning exercise (20 minutes)\n\n* Break (15 min)\n\n* Overview of `DoWhy` framework (20 minutes)\n    * Core workflow: Model, Identify, Estimate, Refute \n    * Metalearning and causal root cause analysis\n\n* Notebook 3 exercises: Exploring DoWhy (20 minutes)\n\n* Explain bonus exercise notebook 4: Bayesian structural time series\n\n* Closing remarks (15 minutes)\n    * How to troubleshoot common issues in causal inference analyses. \n    * Returning to the basic causal inference assumptions we discussed before, with a final warning about them. \n    * General Q&A and Wrap Up", "recording_license": "", "do_not_record": false, "persons": [{"code": "PVWVHW", "name": "Roni Kobrosly", "avatar": "https://pretalx.com/media/avatars/PVWVHW_tSJYW1K.webp", "biography": "I am a former epidemiology researcher who has spent approximately a decade employing causal modeling and inference. The bulk of my academic career was spent conducting data analyses to estimate the population-level effects of harmful environment exposures, when traditional randomized experiments were infeasible or unethical.\n\nSince leaving the academic world, I've been loving my second life in the tech industry as a data scientist, AI/ML engineer, and more recently as a Director of Data Science Observability at Capital One. I love mentoring junior data folks and explaining the magic of data analysis and modeling to non-technical audience. I am also a proud member of the open-source community!\n\nRelevant links:\n\n* https://ronikobrosly.github.io/\n* https://www.linkedin.com/in/ronikobrosly/", "public_name": "Roni Kobrosly", "guid": "566e0f05-6423-51a2-8b39-e343f08c11b4", "url": "https://pretalx.com/scipy-2026/speaker/PVWVHW/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/GB3N9K/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/GB3N9K/", "attachments": []}]}}, {"index": 2, "date": "2026-07-14", "day_start": "2026-07-14T04:00:00-05:00", "day_end": "2026-07-15T03:59:00-05:00", "rooms": {"Intro": [{"guid": "88b58f3c-73f9-5873-8797-7a643b7f85c9", "code": "GRTY3K", "id": 91917, "logo": null, "date": "2026-07-14T08:00:00-05:00", "start": "08:00", "end": "2026-07-14T12:00:00-05:00", "duration": "04:00", "room": "Intro", "slug": "scipy-2026-91917-thinking-in-arrays-room-hsec-2-132", "url": "https://pretalx.com/scipy-2026/talk/GRTY3K/", "title": "Thinking in Arrays (Room HSEC 2-132)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Installation Instructions: https://github.com/ikrommyd/2026-07-14-scipy2026-tutorial-thinking-in-arrays\nPlease do the setup before the tutorial.\n\nPython has become the dominant language in scientific computing, even in domains that demand high performance. This is largely due to the power of array-oriented programming, which separates complex problems into two parts: lightweight bookkeeping and heavy numerical computation. The latter is handled efficiently by vectorized operations that rely on fast, precompiled libraries.\n\nThis tutorial introduces array-oriented programming as a distinct mindset that encourages new ways of structuring problems. Rather than focusing on any one library, we\u2019ll cover general techniques that apply to any array library with a particular focus on NumPy and JAX. You'll work in groups on some short puzzles and three class projects: Conway's Game of Life using arrays just-in-time (JIT) compilation for the Mandelbrot set, and exploring data in ragged arrays. This tutorial focuses on the thought process: all of the problems are to be solved in an imperative way (for loops) and an array-oriented way.", "description": "Installation Instructions: https://github.com/ikrommyd/2026-07-14-scipy2026-tutorial-thinking-in-arrays\nPlease do the setup before the tutorial.\n\nThe tutorial will alternate between short lectures and short exercises for the audience followed by a guided tour through solutions, alternatives, and\ntrade-offs. For exact time slots for each lecture and project, consult the table below.\n\nPart 1: Array-Oriented Programming Fundamentals\nLecture 1: Introduce array-oriented programming as a paradigm. Compare imperative, functional, and array-oriented styles using simple and complex examples\n(3-body problem). Demonstrate speed/memory advantages. Work through all 5 NumPy puzzles from the lecture notebook: attendees\nsolve each on their own, then the solution is shown.\nProject 1: Attendees implement Conway's Game of Life using arrays. Given imperative solution, attendees create a NumPy version that's significantly faster.\nStretch goal: discover convolution-based solution.\nSolutions: Present manual solution, boundary condition handling, and elegant convolution approach with performance comparisons.\n\nPart 2: Limitations of Array-Oriented Programming\nLecture 2: Discuss disadvantages: (1) intermediate arrays problem (quadratic formula example with timing), (2) \"iterate until converged\" problem (Newton's\nmethod, connection to ML epochs). Lecture only this time \u2014 no live project.\n(Optional homework, not covered live) Project 2: Attendees perform tree-traversal in an array-oriented way, walking all input points down a Scikit-Learn\ndecision tree simultaneously. Solutions present immutable and in-place approaches, comparing performance across Python, NumPy, Numba, and JAX.\n\nPart 3: JIT Compilation\nLecture 3: Introduce JIT compilation as a solution. Demonstrate Numba (requires imperative code) and JAX (array-oriented but limited by dynamic branching) on\nthe quadratic formula.\nProject 3: Students accelerate Mandelbrot set computation using Numba and JAX. Compare performance of imperative Python, NumPy, and JIT-compiled versions.\nSolutions: Show optimized implementations, \"Mandelbrot on all accelerators,\" discuss GPU programming advantages.\n\nPart 4: Ragged and Nested Arrays\nLecture 4: Present ragged, nested, missing, and heterogeneous data examples.\nProject 4: Students compute path lengths from Chicago taxi trip data in Parquet format with ragged coordinate arrays.\nSolutions: Present efficient solution, discuss practical handling of ragged arrays, mention additional resources.\n\nHere is a general outline:\n\n- 0:00\u20120:40 (40 min) Lecture 1: Array-oriented programming and its benefits, including all 5 NumPy puzzles\n- 0:40\u20120:45 (5 min) Break\n- 0:45\u20121:05 (20 min) Project 1: Conway's Game of Life using arrays\n- 1:05\u20121:15 (10 min) Break\n- 1:15\u20121:30 (15 min) Solutions to project 1\n- 1:30\u20121:50 (20 min) Lecture 2: Disadvantages of array-oriented programming\n- 1:50\u20122:00 (10 min) Break\n- 2:00\u20122:15 (15 min) Lecture 3: JIT-compilation with Numba and JAX\n- 2:15\u20122:35 (20 min) Project 3: JIT-compilation of the Mandelbrot set\n- 2:35\u20122:45 (10 min) Break\n- 2:45\u20123:00 (15 min) Solutions to project 3\n- 3:00\u20123:15 (15 min) Lecture 4: Ragged and deeply nested arrays\n- 3:15\u20123:35 (20 min) Project 4: Exploring data in ragged arrays\n- 3:35\u20123:45 (10 min) Break\n- 3:45\u20124:00 (15 min) Solutions to project 4", "recording_license": "", "do_not_record": false, "persons": [{"code": "3E3SVE", "name": "Iason Krommydas", "avatar": "https://pretalx.com/media/avatars/3E3SVE_gPPzXZE.webp", "biography": "I'm a PhD student in the Department of Physics and Astronomy at Rice University, conducting research in high-energy physics as a member of the CMS experiment at the Large Hadron Collider at CERN. My work focuses on studying Higgs boson decays into two photons, analyzing data collected by the CMS detector, and contributing to software development for large-scale scientific analyses. I'm passionate about scientific computing and open-source tools that enable reproducible and efficient research. I\u2019m maintainer of Awkward Array, an array library for nested, variable-sized data, using NumPy-like idioms, and an author and maintainer of Coffea, a toolkit designed to simplify data analysis in particle physics. With deep experience in the scientific Python ecosystem, I enjoy building tools that drive insight and accelerate scientific discovery.", "public_name": "Iason Krommydas", "guid": "acba2ad9-cd31-50a0-a998-aea6be3632e7", "url": "https://pretalx.com/scipy-2026/speaker/3E3SVE/"}, {"code": "BGX8FE", "name": "Jim Pivarski", "avatar": "https://pretalx.com/media/avatars/BGX8FE_lkJQBU6.webp", "biography": "Jim was trained as a particle physicist with a Ph.D. from Cornell and helped commission the CMS experiment at the Large Hadron Collider (LHC). He has worked as a data scientist (at Open Data Group) and a software developer (at Princeton), and was the founder of the Awkward Array project. Jim is now at the University of Chicago's Data Science Institute, where he solves data analysis problems for nonprofit organizations.", "public_name": "Jim Pivarski", "guid": "4831e653-73cc-5b87-8fd9-3ad6177051a3", "url": "https://pretalx.com/scipy-2026/speaker/BGX8FE/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/GRTY3K/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/GRTY3K/", "attachments": []}, {"guid": "157ec176-cdc3-5aa3-a55a-aaf26cd4f9cd", "code": "XHCZTH", "id": 93238, "logo": null, "date": "2026-07-14T13:30:00-05:00", "start": "13:30", "end": "2026-07-14T17:30:00-05:00", "duration": "04:00", "room": "Intro", "slug": "scipy-2026-93238-everything-is-an-xarray-dataset-room-hsec-2-138", "url": "https://pretalx.com/scipy-2026/talk/XHCZTH/", "title": "Everything is an Xarray Dataset (Room HSEC 2-138)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Xarray provides data structures for multi-dimensional labeled arrays and a toolkit for scalable data analysis on large, complex datasets. Many real-world datasets fit this structure. However, a common roadblock for users is knowing how to load the data in Xarray and then how to best use Xarray\u2019s tools to represent the structure of the data. In this hands-on tutorial we will showcase how to work with Xarray, various ways to get real-world data into Xarray (with examples from geosciences and biology) and finally how to easily make complex selections on data using community developed custom indexes.\n\nInstallation Instructions: https://tutorial.xarray.dev/workshops/scipy2026/index.html", "description": "In this hands-on tutorial, users will work with example data from multiple fields of science (including biology and geosciences) to achieve these learning objectives:\n### Understand xarray\u2019s core data structures\n\n   * Named arrays and coordinates (`Variable`)\n   * Groups of arrays with coordinates (`DataArray` and `Dataset`)\n   * Hierarchical trees of related groups (`DataTree`)\n\n### Understand how to load data from different formats as an Xarray object with different access patterns:\n   * VirtualiZarr\n   * Intake\n   * Backend engines\n      - Rioxarray (rasterio)\n      - pyDAP\n      - Zarr\n   * Icechunk\n\n### How to use Xarray [flexible indexes](https://xarray-indexes.readthedocs.io/) to make queries on the data once it is loaded\n\n   * Forecasts\n   * Tree based indexing\n   * Lazy Out of Memory\n \n## Familiarity\n\nThis hands-on tutorial assumes participants have some familiarity with Jupyter Notebooks, NumPy, Pandas, and Xarray, and focuses on intermediate workflows using  real-world datasets. All material will be presented in curated Jupyter Notebooks with exercises to solidify understanding of key concepts. Tutorial material is available [online](https://tutorial.xarray.dev/) with instructions for running examples on free hosted infrastructure or on a local computer. No specific scientific domain expertise is required to participate effectively in this tutorial. Example datasets will either be small enough to download locally or available as in public cloud buckets.\n\nWe encourage participants to review last year\u2019s [tutorial](https://tutorial.xarray.dev/workshops/scipy2025/index.html) prior to attending and bring your questions and enthusiasm to make our 4-hour session as interactive as possible!", "recording_license": "", "do_not_record": false, "persons": [{"code": "Y3HT8K", "name": "Ian Hunt-Isaak", "avatar": null, "biography": "I am working as an Xarray community developer at Earthmover. In this role I am focused on improving Xarray\u2019s support and documentation for the biology/biomedical community. Prior to this I completed my PhD in which I extensively used Xarray, zarr and the Pydata stack to implement custom microscope control software and analyze multimodal timelapse single cell microscopy data. I loved the open source scientific software so much that now I get to work full time improving it and sharing it with others.", "public_name": "Ian Hunt-Isaak", "guid": "71e2c4db-d4ce-5032-8d7c-b813c88572fd", "url": "https://pretalx.com/scipy-2026/speaker/Y3HT8K/"}, {"code": "DTWZ3J", "name": "Nick Hodgskin", "avatar": "https://pretalx.com/media/avatars/DTWZ3J_wWCXMIg.webp", "biography": "Nick Hodgskin is a Research Software Engineer and Xarray maintainer working at Utrecht University, primarily on Parcels - a Lagrangian simulation framework used in physical oceanography. Here he has been leading a rewrite of Parcels to use Xarray as a core data structure, along the way improving Parcels interoperability with the Pangeo ecosystem of packages. A self proclaimed \"Pangeo evangelist\", Nick loves communicating the power of the Scientific Python stack with Xarray for geospatial analysis - which he does by organising fortnightly talks at his institute, as well as by giving tutorials. When he isn't coding, you can find Nick playing Ultimate frisbee, hiking in nature, reading, or learning languages.\n\n- [GitHub](https://github.com/VeckoTheGecko)", "public_name": "Nick Hodgskin", "guid": "8d65d57f-fa36-56f1-9d0e-4c4292e30949", "url": "https://pretalx.com/scipy-2026/speaker/DTWZ3J/"}, {"code": "RW7ZG9", "name": "Eniola Awowale", "avatar": null, "biography": "Eni is a scientific software developer at NASA Goddard\u2019s Earth Science and Information Services Center (GESDISC) and Xarray core developer. At GESDISC she uses open-source tools to create services in support of NASA\u2019s vast earth science catalog and contributes to several enterprise NASA ESDIS tools. She is interested in using computational science to improve our understanding of the natural world.", "public_name": "Eniola Awowale", "guid": "519d8533-a399-5973-8131-f88c07df0ce7", "url": "https://pretalx.com/scipy-2026/speaker/RW7ZG9/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/XHCZTH/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/XHCZTH/", "attachments": []}, {"guid": "3c83a9f7-9871-5a59-b2cf-b774938aa317", "code": "7GVREJ", "id": 103272, "logo": null, "date": "2026-07-14T19:00:00-05:00", "start": "19:00", "end": "2026-07-14T23:00:00-05:00", "duration": "04:00", "room": "Intro", "slug": "scipy-2026-103272-evening-social-taco-tuesday-at-the-market-at-malcolm-yards", "url": "https://pretalx.com/scipy-2026/talk/7GVREJ/", "title": "Evening Social: Taco Tuesday at the Market at Malcolm Yards", "subtitle": "", "track": "Social Event", "type": "Tutorial", "language": "en", "abstract": "Join fellow SciPy attendees for a casual, self-organized dinner at **The Market at Malcolm Yards** (501 30th Ave SE), a lively food hall with tacos, Asian cuisine, pizza, vegan options, desserts, and beverages, including local craft beer and non-alcoholic options.\n\nThis is an informal, community-organized meetup. Attendees are responsible for purchasing their own food and drinks.\n\n**Schedule**\n\n - 6:30 PM: Walk departs from the Graduate Hotel (led by Ed Rogers)\n - 6:45 PM: Passing the Days Hotel for anyone staying nearby\n - 7:00 PM: Arrival at Malcolm Yards\n\nFood Hall:\nhttps://malcolmyards.market/food/\n\nDirections:\nhttps://maps.app.goo.gl/z33DMx8QJfkuDKKG6", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/7GVREJ/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/7GVREJ/", "attachments": []}], "Viz": [{"guid": "68ea72b7-2d08-5e84-8398-7b556d807c27", "code": "XZLPB3", "id": 92493, "logo": null, "date": "2026-07-14T08:00:00-05:00", "start": "08:00", "end": "2026-07-14T12:00:00-05:00", "duration": "04:00", "room": "Viz", "slug": "scipy-2026-92493-shiny-for-python-building-production-ready-dashboards-in-python-pwb-3-152", "url": "https://pretalx.com/scipy-2026/talk/XZLPB3/", "title": "Shiny for Python: Building Production-Ready Dashboards in Python (PWB 3-152)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Shiny is a framework for building web applications and data dashboards in Python.\nIn this workshop,\nyou will see how the basic building blocks of shiny can be extended to create\nyour own scalable production-ready python applications.\n\nIn particular, this workshop covers:\n\n- Overview of the basic building blocks of a Shiny for Python application\n- How to refactor applications into shiny modules\n- How to write tests for your shiny application\n- Deploy and share your application\n\nAt the end of this course you will be able to:\n\n- Build a Shiny app in Python\n- Refactor your reactive logic into Shiny Modules\n- Identify when to write Shiny modules\n- Write unit tests and end-to-end tests for your shiny application\n- Deploy and share your application (for free!)\n\nInstallation Instructions: https://chendaniely.github.io/scipy-2026-shiny/setup.html", "description": "Shiny is a framework for building web applications and data dashboards in Python.\nIn this one-day workshop,\nyou will see how the basic building blocks of shiny can be extended to create\nyour own scalable production-ready python applications.\n\nIn particular, this workshop covers:\n\n- 0-50: Overview of the basic building blocks of a Shiny for Python application\n- How to refactor applications into shiny modules\n- How to write tests for your shiny application\n- Deploy and share your application\n\nAt the end of this course you will be able to:\n\n- Build a Shiny app in Python\n- Refactor your reactive logic into Shiny Modules\n- Identify when to write Shiny modules\n- Write unit tests and end-to-end tests for your shiny application\n- Deploy and share your application (for free!)\n\nThe workshop will have both a lecture component and hands-on live coding practical component.\nWe will work together to build and understand one of our Shiny for Python's Dashboard Templates:\n<https://shiny.posit.co/py/templates/>\n\n### Workshop Breakdown:\n\nFirst Hour: Introduction\n\n- :00-:20  Overview of the basic building blocks of a Shiny for Python application\n- :20-:35  Input components\n- :35-:50  Output components\n- :50-1:00 break\n\nSecond Hour: Build a more complex app\n\n- 1:00 1:35 A more complex application with multiple input and output components\n- 1:35-1:50  Introduction to Shiny's reactivity programming model.\n- 1:50-2:00 Break\n\n3rd Hour: Refactoring your application and Shiny Models\n\n- 2:00-2:15 Introduction to shiny modules\n- 2:15-2:30 Refactor current app into modules\n- 2:30-2:50 Import your Shiny Modules into the new application\n- 2:50-3:00 Break\n\n4th Hour: Testing and deployment\n\n- 3:00-3:30 Testing your shiny apps with playwright\n- 3:30-4:00 Deploying your application to the web (for free!)\n\n\n### Workshop preparation:\n\nWe will be using Positron in the workshop with the VSCode Shiny extension.\nYou can also use VSCode with the Shiny extension as well.\n\n- Positron: <https://positron.posit.co/>\n- VSCode: <https://code.visualstudio.com/>\n- Shiny Extension: <https://marketplace.visualstudio.com/items?itemName=Posit.shiny>\n\nYou will need the following python packages installed. An example `requirements.txt`:\n\n```\nfaicons\nshiny\nshinywidgets\nplotly\npandas\nridgeplot\nipykernel\n```\n\n\n### FAQ\n\n1. What if I'm a complete beginner?\n\n- You should have a basic understanding of Python and be able to install packages with pip, do basic data manipulation, and draw plots.\n\n2. What if I've never built a Shiny app before?\n\nThis workshops doesn\u2019t require any Shiny or web application experience.\nWe'll focus more on practical examples in the course.\nWe do have additional resources for you to dive more into more Shiny details,\nbut we will cover the basics needed to build larger and scalable applications.\n\n3. Why should I learn Shiny if I already know Streamlit or Dash?\n\nWe believe that Shiny is the best framework for building data applications in Python.\nIt\u2019s reactive execution model means that you can build performant applications without\nexplicitly caching data or managing application state.\nSee\n[this blog post](https://posit.co/blog/why-shiny-for-python/)\nfor more on why we think that Shiny is worth learning.\n\n4. I already know Shiny for R, is this workshop for me?\n\nThe R and Python Shiny packages are quite similar,\nso some of the content in this workshop may be familiar to you.\nThat said it\u2019s a great opportunity to fill in missing pieces and ask question about Python best practices.\nWe will also talk about Shiny modules and testing in this workshop,\nwhich will also be a precursor for you to learn more or incorporate Python Packaging\nto your Shiny applications.", "recording_license": "", "do_not_record": false, "persons": [{"code": "8PRQQA", "name": "Daniel Chen", "avatar": "https://pretalx.com/media/avatars/7BJQSY_sLwbtih.webp", "biography": "Daniel Chen is a data science educator working in developer relations at Posit, PBC, and a lecturer at the University of British Columbia. He specializes in teaching and advocating for modern data science tools and workflows.", "public_name": "Daniel Chen", "guid": "b23672c9-e04c-5fb7-b553-3e05dd32e83b", "url": "https://pretalx.com/scipy-2026/speaker/8PRQQA/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/XZLPB3/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/XZLPB3/", "attachments": []}, {"guid": "081b5d55-f133-5006-aa71-3019a302dd37", "code": "VW3PAF", "id": 92413, "logo": "https://pretalx.com/media/scipy-2026/submissions/VW3PAF/image_wXfs4Qo.webp", "date": "2026-07-14T13:30:00-05:00", "start": "13:30", "end": "2026-07-14T17:30:00-05:00", "duration": "04:00", "room": "Viz", "slug": "scipy-2026-92413-hvplot-and-panel-powerful-data-visualization-exploration-and-apps-room-hsec-4-103-5", "url": "https://pretalx.com/scipy-2026/talk/VW3PAF/", "title": "hvPlot and Panel: Powerful data visualization, exploration, and apps (Room HSEC 4-103/5)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "This tutorial will show you how to use the Pandas, Dask, or Xarray APIs you already know to interactively explore and visualize your data, even if the data is gigabyte or petabyte sized or is in non-columnar scientific formats such as multidimensional arrays, networks, or unstructured grids. As soon as you have something you like, you can then share a live app as HTML+WASM or backed by a live Python server, by simply replacing your expression arguments with widgets so that users can explore it on their own. These tools let you focus on your data rather than the API, and let you build linked, interactive drill-down exploratory apps without having to run a web-technology software development project, which you can then share without becoming an operations specialist.\n\nInstallation instructions: https://holoviz.org/tutorial/Setup.html", "description": "Python offers many powerful visualization tools (listed on pyviz.org), each with their own strengths and advantages. Few people have the time and interest to learn all the different APIs required to use these different tools, but a de-facto standard API for data plotting has emerged in the Pandas .plot() API, now supported by many different plotting packages.\n\nIn this tutorial, you will learn how to use hvPlot, a high-level interactive plotting library that exposes the power of Bokeh, Matplotlib, Plotly, Datashader, HoloViews, GeoViews, and Cartopy using the same .plot API you may already know from using Pandas, Dask, or Xarray's plotting interface. We'll also show you how to turn nearly any expression you can write with that API into a web app with plots and tables by simply substituting widgets for any parameters you want users to be able to change, easily creating reactive expression pipelines. Thanks to the HoloViz tools on which hvPlot is built, the resulting apps can easily handle big data (up to billions of rows on an ordinary laptop or petabytes on a distributed cluster), remote data (either in Jupyter or in standalone apps), streaming data, geographical data (building on the geoscience software stack), and multidimensional data (using Xarray).\n\nhvPlot's high-level interface should be sufficient for nearly all of the common data-exploration and data-analysis tasks you want to do with Pandas, Dask, or Xarray, but in keeping with the HoloViz philosophy of \"shortcuts rather than dead ends\", we'll also show you how and when to drop down to lower-level APIs when you need to, such as when building more complex apps using Panel, doing complex graphical data calculations using Datashader, or integrating plotting and interactivity into your own libraries using Param and HoloViews.\n\nWe'll also provide guidance on how to use AI effectively with the HoloViz ecosystem, including a brief preview of our new Lumen.HoloViz.org tool for natural-languge data exploration along with advice for AI code generation.\n\nWith the techniques you learn in the hands-on exercises in this tutorial, you'll get the tools and know-how to effectively explore, analyze and visualize simple or complex, small or large, and static or dynamic data easily, concisely, and reproducibly. The resulting visualizations and apps can be shared as static images, simple HTML documents with limited interactivity, HTML+WASM documents with full Python-backed interactivity, or as Python apps deployed on a remote server. We expect participants to have previously used some sort of plotting tool and to be comfortable with Python and at least one array-based Python library (Numpy, Pandas, Xarray, CuPy, cuDF, Dask, etc.).", "recording_license": "", "do_not_record": false, "persons": [{"code": "PHKMS3", "name": "James A. Bednar", "avatar": "https://pretalx.com/media/avatars/YLVRB8_0sXqxyG.webp", "biography": "Jim Bednar is the Senior Director of Professional Services at Anaconda, Inc. Dr. Bednar holds a Ph.D. in Computer Science from the University of Texas, along with degrees in Electrical Engineering and Philosophy. He has published more than 50 papers and books about the visual system, software development, and reproducible science. Dr. Bednar founded the HoloViz project, a collection of open-source Python tools that includes Panel, hvPlot, Datashader, HoloViews, GeoViews, Param, Lumen, and Colorcet. Dr. Bednar was a Lecturer and Reader in Computational Neuroscience at the University of Edinburgh from 2004-2015, and previously worked in hardware engineering and data acquisition at National Instruments.", "public_name": "James A. Bednar", "guid": "339947d6-df49-5eb4-b350-18cbba7c652e", "url": "https://pretalx.com/scipy-2026/speaker/PHKMS3/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/VW3PAF/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/VW3PAF/", "attachments": []}], "AI/ML": [{"guid": "cd4bea75-c5fd-5c7d-8a26-4f74a8daa908", "code": "KHQ3EK", "id": 92198, "logo": "https://pretalx.com/media/scipy-2026/submissions/KHQ3EK/image_CuMK08X.webp", "date": "2026-07-14T08:00:00-05:00", "start": "08:00", "end": "2026-07-14T12:00:00-05:00", "duration": "04:00", "room": "AI/ML", "slug": "scipy-2026-92198-build-a-scipy-coding-assistant-with-rag-room-hsec-3-110", "url": "https://pretalx.com/scipy-2026/talk/KHQ3EK/", "title": "Build a SciPy Coding Assistant with RAG (Room HSEC 3-110)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Have you ever been frustrated when an LLM generates outdated or deprecated code? It's more common than you'd think. LLMs are trained up to a certain point, but software keeps moving forward. Functions get deprecated, new versions drop, APIs change, old patterns get replaced, and your model has no idea any of it happened.\n\nRAG, or Retrieval-Augmented Generation, is the fix. Instead of relying solely on what a model learned during training, RAG lets you supply it with current, curated information at the moment it generates a response.\n\nIn this 4-hour, hands-on workshop, you'll build a RAG-powered SciPy coding assistant from the ground up. Here's what that looks like in practice:\n\n- **RAG Fundamentals**: You'll start by getting familiar with the core ideas behind RAG: what embeddings are (numerical representations of text that capture meaning), how vector similarity works, and how ChromaDB (a lightweight vector database) stores and retrieves that information.\n\n- **Building the Knowledge Base**: From there, you'll build the SciPy knowledge base itself. That means scraping SciPy's documentation, chunking it into digestible pieces, and processing it in a way that's aware of code structure, not just plain text.\n\n- **Wiring Up the Pipeline**: Once the knowledge base is ready, you'll wire up the full pipeline: querying it intelligently, engineering prompts that produce reliable code, and integrating with both OpenAI and Ollama (a tool for running models locally) so you're not locked into one provider.\n\n- **Evaluation and Deployment**: Finally, you'll wrap everything up by evaluating your system using real retrieval and generation metrics, and deploying a Gradio web app, a simple tool for building interactive UIs in Python, so your assistant is actually usable by people who aren't staring at a Jupyter notebook.\n\nBy the end, you'll have a working SciPy assistant and, more importantly, a solid understanding of every moving part inside it.\n\nInstallation Instructions: https://github.com/cynthiiaa/scipy-RAG#quick-start", "description": "Have you ever been frustrated when an LLM generates outdated or deprecated code? It's more common than you'd think. LLMs are trained up to a certain point, but software keeps moving forward. Functions get deprecated, new versions drop, APIs change, old patterns get replaced, and your model has no idea any of it happened.\n\nSo how do you generate reliable code that reflects current patterns and practices? That's where RAG comes in. RAG, or Retrieval-Augmented Generation, is a technique that updates what your LLM \"knows\" at query time by pulling in fresh, relevant context from a knowledge base you control.\n\nIn this workshop, you'll build a RAG-powered SciPy coding assistant from the ground up. That means scraping and processing SciPy documentation, embedding it into a vector database with ChromaDB, and wiring up a generation pipeline that pulls the right context before producing code. By the end, you'll have a working Gradio web app and a solid understanding of every moving part inside it.", "recording_license": "", "do_not_record": false, "persons": [{"code": "DAGKKN", "name": "Cynthia Ukawu", "avatar": "https://pretalx.com/media/avatars/VQQNZH_7wezeXe.webp", "biography": "Cynthia is a Machine Learning Engineer, where she serves as head architect and developer for a DARPA-funded ML analysis platform. The platform analyzes LLM embeddings using topology as the underlying mathematical framework, enabling researchers to study language model behavior at scale.\n\nBeyond the core platform, she builds production APIs for vision-language models and text-to-speech systems, bringing cutting-edge AI research into operational deployment.\n\nWith an MS in Applied & Computational Mathematics from Johns Hopkins and prior experience at General Atomics-CCRi and Arity (geospatial ML at scale), she combines deep mathematical foundations with practical engineering expertise in PyTorch, AWS, and distributed systems.\n\nShe has mentored over 30 students through programs at The Coding School, Masterschool, and Springboard, helping them transition into data and ML roles. She also writes about ML systems on her blog: cynscode.com", "public_name": "Cynthia Ukawu", "guid": "0567263f-df61-51e5-8fe9-a94e99885bac", "url": "https://pretalx.com/scipy-2026/speaker/DAGKKN/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/KHQ3EK/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/KHQ3EK/", "attachments": []}, {"guid": "5e0cc062-b8cd-50ed-a65c-3a9f5ad5da38", "code": "NWPPHA", "id": 92299, "logo": null, "date": "2026-07-14T13:30:00-05:00", "start": "13:30", "end": "2026-07-14T17:30:00-05:00", "duration": "04:00", "room": "AI/ML", "slug": "scipy-2026-92299-engineering-better-retrieval-for-rag-room-hsec-3-110", "url": "https://pretalx.com/scipy-2026/talk/NWPPHA/", "title": "Engineering Better Retrieval for RAG (Room HSEC 3-110)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "The quality of the retrieval component is what drives Retrieval-Augmented Generation (RAG) systems. Therefore, a well-structured, measurable,  and robust retrieval pipeline is critical to building effective large language model (LLM) applications.\n\nWorking through guided code examples and hands-on experimentation, attendees will collectively develop, optimize,  and enhance the performance of a complete RAG pipeline by improving retrieval in three stages: _Pre-Retrieval_, _Mid-Retrieval_, and _Post-Retrieval_. We will also cover structured and multimodal document parsing with _Docling_, systematic evaluation with _RAGAS_, and a capstone _Agentic RAG_ demo using _LangGraph_. The toolkit integrates _Qdrant_ for vector search and the _LangChain_ ecosystem for orchestration and experimentation.\n\nDuring the hands-on session, attendees will use Jupyter notebooks to learn about, experiment with,  and benchmark techniques that produce significant improvements to retrieval quality using production-ready open-source libraries. At the end of the session, each participant will be equipped with a reusable _\u201cRetrieval Playground\u201d_ framework that can be leveraged to design, evaluate,   and continuously improve RAG systems across various application domains.\n\nInstallation Instructions: https://github.com/mahimaarora/retrieval-playground/tree/main/setup-guides", "description": "Retrieval is the foundation of modern LLM applications in science, engineering, and industry. However, most RAG implementations rely on naive chunking and basic vector similarity search, leading to brittle systems, hallucinations, and poor performance on structured and multimodal data.\n\nThis tutorial provides a **structured, engineering-focused approach to optimizing retrieval pipelines** using a practical framework, a modular Python toolkit for experimentation, benchmarking, and evaluation.\n\nParticipants will iteratively build a RAG pipeline and improve it across three stages:\n\n1. **Pre-Retrieval Optimization** - Preparing data and queries correctly  \n2. **Mid-Retrieval Optimization** - Improving search quality and diversity  \n3. **Post-Retrieval Optimization** - Filtering, refining, compressing, and assembling context before generation  \n\nWe will also cover **structured and multimodal parsing for RAG** with Docling, including:\n- Typed text, table, and image chunks from PDFs\n- Hybrid Docling chunking alongside baseline, recursive, parent-child, and contextual strategies\n- Multimodal-aware metadata for richer retrieval (without separate SQL or ad-hoc query pipelines)\n\nThe tutorial is designed for active coding, experimentation, and measurable benchmarking. More than 70% of the session is hands-on coding in Jupyter notebooks. Attendees will implement techniques step-by-step and evaluate performance improvements live.\n\n#### Detailed Outline (4 Hours Total)\n\n##### Part 1: Foundations - Lecture + Guided Setup (40 minutes)\n- Introduction to Retrieval in RAG Systems\n- Why retrieval fails in real-world systems\n- The three-stage optimization framework \n- Overview of the Retrieval Playground toolkit and notebook flow (1A \u2192 5)\n- Evaluation overview: retrieval, generation, and tool/agent metrics (RAGAS + custom)\n- Dataset introduction\n\n##### Part 2: Pre-Retrieval Optimization - Hands-On Notebook (50 minutes)\n\n1. Document Chunking\n- Recursive chunking\n- Contextual\n- Parent-child\n- Docling-based structured and multimodal chunking (text, tables, images)\n\n2. Query Enhancement\n- Query expansion\n - Multi-query / RAG Fusion\n - Query decomposition\n - Query rewriting\n - Step-back prompting\n - Complexity classification and auto-orchestration \n\n3. Semantic routing\n\n##### Part 3: Mid-Retrieval Optimization - Hands-On Notebook (60 minutes)\n\n- Dense Search \n- Hybrid Search\n- Reranking\n- Parent-Child Retrieval\n- Multi-Query Hybrid\n- Route-Driven Retrieval\n- Adaptive Retrieval\n\nInterval: 15 minutes \n\n##### Part 4: Post-Retrieval, Evaluation & Agentic RAG - Hands-On Notebook (60 minutes)\n\n1. Post-Retrieval Context Preparation\n- Retrieval grading (relevant / irrelevant / ambiguous)\n- Knowledge refinement (sentence- or passage-level tightening)\n- Context compression (extractive and abstractive)\n- Document assembly (stuff chain for final generation)\n\n2. Systematic Evaluation\n- Classical retrieval checks (hit rate@k, MRR, keyword overlap)\n- RAGAS context precision/recall, faithfulness and answer accuracy\n- Tool-selection metrics from routing and agent traces\n- Baseline vs post-retrieval A/B comparison and pipeline scorecard\n\n3. Agentic RAG Capstone (intro + demo)\n- LangGraph ReAct agent with a retrieval tool backed by the workshop RAG stack\n- Prompt-based routing (direct answers vs retrieval)\n- Lightweight tool-selection evaluation\n\n##### Final 15 Minutes: Wrap-Up, Future Directions and Q&A\n- Best practices and limitations\n- Production considerations and scaling strategies\n- Open discussion and troubleshooting\n\n_Expected Level: Beginner to intermediate._\n\n_Target Audience: ML engineers, data scientists, developers working with LLMs in production, and anyone looking to learn how to build robust AI workflows using open source tools._", "recording_license": "", "do_not_record": false, "persons": [{"code": "VPFWJA", "name": "Mahima Arora", "avatar": "https://pretalx.com/media/avatars/KEHTJF_rdJZiaX.webp", "biography": "Mahima Arora is a Senior Data Scientist on the Data & AI team at Red Hat, specializing in Generative AI applications. She develops AI-powered solutions that enhance efficiency and effectiveness, leading initiatives to optimize AI systems for greater impact. Passionate about open source, Mahima actively explores emerging tools and technologies to drive innovation and knowledge sharing, and has presented her work at PyData Amsterdam 2025 and PyCon India 2025.", "public_name": "Mahima Arora", "guid": "588f5b0e-aa66-5979-90f4-3445a2b57399", "url": "https://pretalx.com/scipy-2026/speaker/VPFWJA/"}, {"code": "9TQTHL", "name": "Aarti Jha", "avatar": "https://pretalx.com/media/avatars/SFVFTF_uQLwnBl.webp", "biography": "Aarti Jha is a Principal Data Scientist at Red Hat, where she leads the development of AI-driven solutions that streamline internal operations and reduce costs. She has more than seven years of experience designing and deploying machine learning and generative AI solutions across multiple industries. A frequent speaker at developer and data science conferences, she shares practical insights on applied AI, LLMs, and building AI systems that deliver measurable business value.", "public_name": "Aarti Jha", "guid": "d936acba-6849-52e7-a592-9f3023ff8a58", "url": "https://pretalx.com/scipy-2026/speaker/9TQTHL/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/NWPPHA/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/NWPPHA/", "attachments": []}], "Accelerated Computing": [{"guid": "a0f64a33-20c7-54e1-8133-268735c3d112", "code": "XSWVVE", "id": 92294, "logo": null, "date": "2026-07-14T08:00:00-05:00", "start": "08:00", "end": "2026-07-14T12:00:00-05:00", "duration": "04:00", "room": "Accelerated Computing", "slug": "scipy-2026-92294-deploying-and-debugging-gpu-accelerated-python-workloads-room-hsec-2-110", "url": "https://pretalx.com/scipy-2026/talk/XSWVVE/", "title": "Deploying and debugging GPU accelerated Python workloads (Room HSEC 2-110)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "As GPU acceleration becomes essential for scaling Python workloads, many developers face new challenges: understanding installation, managing dependencies, and deploying GPU-enabled environments. Even experienced Python users can struggle to integrate GPUs effectively or troubleshoot performance issues.\n\nThis tutorial addresses those barriers by walking participants step-by-step through the process of getting started with GPUs. Using NVIDIA\u2019s RAPIDS ecosystem and familiar python tools, we\u2019ll demonstrate how to set up, monitor, optimize and debug GPU-powered workflows\u2014turning what often feels like complex infrastructure work into an approachable, reproducible process.\n\nInstallation Instructions: https://developer.nvidia.com/nsight-systems/get-started", "description": "Leveraging GPU acceleration is now a common necessity for scaling Python projects. NVIDIA GPUs offer unmatched speed and efficiency for data processing and model training, significantly reducing the time and cost associated with these tasks. GPU acceleration is already baked into many projects, or available via plugins. You can use PyData libraries including pandas, polars and networkx without needing to rewrite your code to get the benefits of GPU acceleration. \n\nHowever, integrating GPUs into our workflow can be a new challenge where we need to learn about installation, dependency management, and deployment in the Python ecosystem. When writing code, we also need to monitor performance, leverage hardware effectively, and debug when things go wrong.\n\nThis is where RAPIDS and its tooling ecosystem comes to the rescue. RAPIDS, is a collection of open source software libraries to execute end-to-end data pipelines on NVIDIA GPUs using familiar PyData APIs. RAPIDS libraries give users access to GPU acceleration, reducing execution time and cost, but without needing to learn a whole new set of tools.\n\nIn this tutorial we will cover:\n\n- A high level overview of popular Python libraries that have GPU acceleration\n- Answers to questions like: \u201cWhere do I get a GPU?\u201d, \u201cHow do I run a container on a VM with a GPU?\u201d, \u201cHow do I install GPU packages into an existing environment?\u201d, \u201cWhat if I use uv pip?\u201d, \u201cWhat about conda? \u201das well as follow along examples to get a GPU up and running.\n- The GPU software stack from driver to Python and everything in between\n- Troubleshooting and monitoring:  Examples of performance analysis, diagnostics, and debugging. Showcasing of diagnostic tools like nvdashboard, nvtop, nsys, pynvml, etc.  \n\n#### Audience\nThis is a hands-on tutorial, participants should ideally have some experience using Python, pandas and sci-kit learn. We'll use cloud-based VMs, so familiarity with the cloud and resource creation is helpful but not required. No prior GPU knowledge is needed.\n\nTo maximize the tutorial's relevance, we will provide participants with the opportunity to submit their specific environment configurations ahead of time. Submissions received with adequate notice (between tutorial acceptance and conference date) will be integrated into the tutorial examples, allowing participants to see their real-world use cases addressed.\n\n**Key takeaways for participants will be:**\n- An understanding of the GPU Python software stack from driver through core libraries to high-level Python libraries\n- How they can use their preferring tooling and package managers to install all the components they need\n- How to monitor their GPUs and understand how well they are using their hardware\n- How to attach debuggers to their GPU code or record traces and profiles for debugging later", "recording_license": "", "do_not_record": false, "persons": [{"code": "VXQXZP", "name": "Naty Clementi", "avatar": null, "biography": "Naty Clementi is a senior software engineer at [NVIDIA](https://www.nvidia.com/). She is a former academic with a Masters in Physics and PhD in Mechanical and Aerospace Engineering to her name. Her work involves contributing to [RAPIDS](https://rapids.ai/), and in the past she has also contributed and maintained other open source projects such as [Ibis](https://ibis-project.org/) and [Dask](https://www.dask.org/). She is an active member of [PyLadies](https://pyladies.com/) and an active volunteer and organizer of [Women and Gender Expansive Coders DC meetups](https://www.meetup.com/women-and-gender-expansive-coders-dc-wgxc-dc/).", "public_name": "Naty Clementi", "guid": "039b72d7-581e-5888-8cb9-a36d8a2ca95a", "url": "https://pretalx.com/scipy-2026/speaker/VXQXZP/"}, {"code": "AURRUC", "name": "Jacob Tomlinson", "avatar": null, "biography": null, "public_name": "Jacob Tomlinson", "guid": "7d5794a8-e43e-58a6-9a19-8751d101fde1", "url": "https://pretalx.com/scipy-2026/speaker/AURRUC/"}, {"code": "WHWU9R", "name": "Jaya Venkatesh", "avatar": "https://pretalx.com/media/avatars/RTDCAL_IR6nvLd.webp", "biography": "Jaya Venkatesh is a Software Engineer at NVIDIA, working on the RAPIDS ecosystem to streamline the deployment of GPU-accelerated data science workflows across cloud and distributed systems. Previously, he was a Machine Learning Engineer at Pixxel Space, where he developed large-scale, real-time data processing and inference pipelines for Earth observation using GPU-accelerated Python libraries.", "public_name": "Jaya Venkatesh", "guid": "33becb3b-74ef-592e-af56-01c14d57b9be", "url": "https://pretalx.com/scipy-2026/speaker/WHWU9R/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/XSWVVE/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/XSWVVE/", "attachments": []}, {"guid": "ca3ae37e-eb02-52d9-9235-7127a58a6231", "code": "RQQDGA", "id": 92408, "logo": null, "date": "2026-07-14T13:30:00-05:00", "start": "13:30", "end": "2026-07-14T17:30:00-05:00", "duration": "04:00", "room": "Accelerated Computing", "slug": "scipy-2026-92408-computational-methods-for-simulation-using-jax-and-numpy-room-hsec-2-110", "url": "https://pretalx.com/scipy-2026/talk/RQQDGA/", "title": "Computational Methods for Simulation using JAX and NumPy (Room HSEC 2-110)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "This tutorial demonstrates how to accelerate agent-based simulations using modern Python tools. Using Thomas Schelling's classic segregation model as a running example, participants will learn to transform readable but slow Python code into high-performance implementations using NumPy and JAX. The tutorial explores how mild individual preferences can lead to extreme aggregate outcomes through simulation, while teaching practical techniques for leveraging modern hardware (including GPUs) to make realistic large-scale simulations computationally feasible. Participants will gain hands-on experience with performance optimization strategies applicable to economic modeling, urban planning, epidemiology, and other domains requiring large-scale agent-based simulations.\n\nInstallation Instructions: https://github.com/QuantEcon/scipy_tutorial_2026", "description": "Simulation is a critical methodology for policy analysis across economics, public health, urban planning, and environmental science. Examples include DSGE models for monetary policy, pension reform analysis, climate policy evaluation, and agent-based urban models. However, realistic simulations often require tracking thousands or millions of agents over many time periods, making computational efficiency essential.\n\nThis tutorial addresses the computational challenges of simulation through a concrete, historically significant example: Thomas Schelling's 1969 segregation model, which earned him the 2005 Nobel Prize in Economic Sciences. The model demonstrates a surprising result: extreme residential segregation can emerge even when individuals have only mild preferences for same-type neighbors. This finding has profound implications for understanding persistent urban segregation patterns observed in American cities.\n\nWe begin with an intuitive object-oriented Python implementation that prioritizes readability, then systematically optimize performance through:\n1. Array-based computing with NumPy\n2. Just-in-time compilation and GPU acceleration with JAX\n3. Parallelization strategies for modern hardware\n\nThrough live coding demonstrations and hands-on exercises, participants will transform a slow baseline implementation (taking minutes) into a highly optimized version (running in seconds) capable of simulating realistic urban scenarios with tens of thousands of agents. The tutorial emphasizes transferable skills, and the optimization patterns learned apply broadly to agent-based models in computational science.\n\nThe tutorial also explores the substantive implications of the model, connecting computational results to real-world segregation patterns and policy questions. Participants will see how computational tools enable researchers to test hypotheses about social dynamics that would be impossible to study analytically.", "recording_license": "", "do_not_record": false, "persons": [{"code": "UNJCPG", "name": "Smit Lunagariya", "avatar": "https://pretalx.com/media/avatars/YZFEHQ_3QUZStf.webp", "biography": "Smit Lunagariya is a Machine Learning Engineer at Google and an active contributor to the scientific Python and open-source ecosystems. He holds an Integrated Dual Degree (Bachelor's and Master's) in Mathematics and Computing Engineering from the Indian Institute of Technology (BHU), Varanasi.\n\nHis involvement in open source began with Google Summer of Code in 2020 and has since expanded to include contributions to several scientific computing projects, including SciPy, SymPy, LPython, LFortran, QuantEcon, and Aesara.", "public_name": "Smit Lunagariya", "guid": "323a1765-bcb4-575a-9f92-dc31cd13b78c", "url": "https://pretalx.com/scipy-2026/speaker/UNJCPG/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/RQQDGA/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/RQQDGA/", "attachments": []}], "Other": [{"guid": "2fa3ad01-a656-5ac9-93bd-9ea4245d0839", "code": "VNQPKP", "id": 93111, "logo": null, "date": "2026-07-14T08:00:00-05:00", "start": "08:00", "end": "2026-07-14T12:00:00-05:00", "duration": "04:00", "room": "Other", "slug": "scipy-2026-93111-network-analysis-made-simple-hsec-4-103-5", "url": "https://pretalx.com/scipy-2026/talk/VNQPKP/", "title": "Network Analysis Made Simple (HSEC 4-103/5)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Through the use of NetworkX's API, tutorial participants will learn about the basics of graph theory and its use in applied network science. Starting with a computationally-oriented definition of a graph and its associated methods, we will progress through the following concepts: path and structure finding, visualization, and graph storage on disk. We will also offer tutorial participants the option of one advanced topic overview, including the use of graphs alongside LLMs for knowledge retrieval, scalable alternatives to NetworkX including cuGraph, and the use of linear algebraic translation of graph problems to speed up computations.\n\nInstallation Instructions: https://github.com/ericmjl/Network-Analysis-Made-Simple/", "description": "In this tutorial, we will walk you through what we consider the most practical aspects of graph theory using NetworkX. While graph theory can seem abstract at first, having a computational framework like NetworkX makes it much more approachable.\n\nWe will start with what we think is the most intuitive way to understand graphs - seeing them as computational objects we can manipulate with code. From there, we will show you how we approach common tasks like finding paths between nodes, analyzing graph structure, and creating visualizations that actually make sense. We will also cover how to store and read graphs to/from disk.\n\nBased on our experience working with graphs, we've selected three cutting-edge topics that we think are worth exploring: using graphs with LLMs for knowledge retrieval, scaling up to larger datasets with cuGraph and linear algebra, or an introduction to the use of graphs in deep learning. Tutorial participants will get to choose one of these topics live.\n\nThis tutorial is structured based on what we wished we knew when we first started working with graphs, and is structured in the order that we believe to be most productive for learning. By the end of the tutorial, participants should be able to productively prototype with graphs immediately!", "recording_license": "", "do_not_record": false, "persons": [{"code": "9NRRJH", "name": "Eric Ma", "avatar": "https://pretalx.com/media/avatars/EP39HL_u5E0WzK.webp", "biography": "As Senior Principal Data Scientist at Moderna Eric leads the Data Science and Artificial Intelligence (Research) team to accelerate science to the speed of thought. Prior to Moderna, he was at the Novartis Institutes for Biomedical Research conducting biomedical data science research with a focus on using Bayesian statistical methods in the service of discovering medicines for patients. Prior to Novartis, he was an Insight Health Data Fellow in the summer of 2017 and defended his doctoral thesis in the Department of Biological Engineering at MIT in the spring of 2017.\n\nEric is also an open-source software developer and has led the development of pyjanitor, a clean API for cleaning data in Python, and nxviz, a visualization package for NetworkX. He is also on the core developer team of NetworkX and PyMC. In addition, he gives back to the community through code contributions, blogging, teaching, and writing.\n\nHis personal life motto is found in the Gospel of Luke 12:48.", "public_name": "Eric Ma", "guid": "0c3ba6a2-c8b9-5f73-9828-d3217a7a15e6", "url": "https://pretalx.com/scipy-2026/speaker/9NRRJH/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/VNQPKP/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/VNQPKP/", "attachments": []}, {"guid": "968ca6b3-214b-58d5-b3d0-5c68e5bc1773", "code": "B8S8PH", "id": 91953, "logo": null, "date": "2026-07-14T13:30:00-05:00", "start": "13:30", "end": "2026-07-14T17:30:00-05:00", "duration": "04:00", "room": "Other", "slug": "scipy-2026-91953-microwave-image-processing-exploring-realms-of-earth-through-spaceborne-radars-using-python-room-pwb-3-152", "url": "https://pretalx.com/scipy-2026/talk/B8S8PH/", "title": "Microwave Image Processing: Exploring realms of Earth through spaceborne Radars using Python (Room PWB 3-152)", "subtitle": "", "track": "Tutorials", "type": "Tutorial", "language": "en", "abstract": "Remote Sensing has proved to be an important tool in monitoring our earth's ecosystem. Satellite imaging is a vital part of Remote Sensing. Predominantly, Satellite Imaging of the earth has been done in the optical domain and optical Images serve the majority of purpose for earth monitoring. But, these satellites do not have all-weather acquisition capability and this lacuna is filled by the satellite sensors working in the Microwave domain of the Electromagnetic spectrum. Synthetic Aperture Radar(SAR) is an Imaging Radar that acquires images of a particular area on Earth in the microwave region of electro-magnetic spectrum. This workshop deals with the processing of SAR Images and how these images can be beneficial in a variety of geographical applications.\n\nInstallation Instructions: Participants should use a computer or cloud compute with at least 16GB of memory. Downloads of approximately 1-2GB will be needed during the tutorial.", "description": "Intended Audience : The workshop will be aimed at the audience belonging to any level of education. It will introduce them to the wonderful class of SAR images and how are these images useful from the perspective of various applications.\n\nExpected Outcomes (after the workshop, the audience will be) :\n\ni) Able to understand the acquisition of SAR imagery.\n\nii) Able to understand the types of datasets utilized in remote sensing\n\niii) Able to use the GDAL library to perform operations on images\n\niv) Able to efficiently process SAR imagery using Python\n\nv) Able to draw a roadmap in order to utilize SAR imagery for various geographic applications\n\nOutline\n\nThe workshop will be divided into the following sub-sessions :\n\nSub-Session-1: Introduction to Microwave Remote Sensing (1.5 hrs) - This part will discuss the foundations of Microwave Remote Sensing. Theoretical aspects regarding the acquisition of images, the formation of images encompassing the generation of complex images and ground range detected images will be discussed. This session will also cover key topics such as basic utilization of GDAL, Numpy and Matplotlib Libraries for opening and visualizing Images which will cover developing basic codes for plotting, visualizing  and understanding the imagery data.\n\nSub-Session-2: Pythonic Way to SAR Image Processing (2.5 hrs): This part will focus on achieving the following Key points:\n\n1) Codes will be developed separately for calibration for each SAR sensor(esp. Sentinel-1, Radarsat-2) from scratch.(1.5 hrs)\n\n2) Utilization of the codes developed in (1) for various applications such as Oceanography, Forestry, etc.(1 hr)\n\nDatasets: Free Imagery data sets of Sentinel-1 SAR will be utilized. Also, free sample datasets available for different SAR earth observation sensors will be utilised. In addition, sample datasets of Radarsat-2, RISAT- 1 which are freely downloadable will be utilized. The sample datasets will be provided. For better understanding of the datasets, the participants may download and utilize the Sentinel-1 SAR free Image dataset initially .Sentinel-1 Free SAR Imagery (https://www.copernicus.eu/en)\n\nConduct of the workshop : The workshop will be conducted through the means of Jupyter Notebooks. Along with the sessions, the audience will be provided with the exercises to clear their concepts of SAR Imagery.\n\nTotal Duration\n\nThe duration of the workshop will be 4 hours.", "recording_license": "", "do_not_record": false, "persons": [{"code": "NPQLXC", "name": "Shubham Sharma", "avatar": "https://pretalx.com/media/avatars/NPQLXC_KrZ9qST.webp", "biography": "Shubham Sharma is a Senior Data Scientist with more than ten years of experience at the convergence of Remote Sensing, Image Processing, Computer Vision, and Deep Learning. He has been an active contributor to the open-source scientific computing ecosystem, engaging with communities through conferences such as SciPy and as a past speaker at PyCon. His work includes significant experience in Synthetic Aperture Radar (SAR) image processing, and he is deeply committed to advancing knowledge of satellite image analysis using Python within the open source community.", "public_name": "Shubham Sharma", "guid": "5b394c10-5ba2-56fc-ad7f-43a5be4daa62", "url": "https://pretalx.com/scipy-2026/speaker/NPQLXC/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/B8S8PH/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/B8S8PH/", "attachments": []}]}}, {"index": 3, "date": "2026-07-15", "day_start": "2026-07-15T04:00:00-05:00", "day_end": "2026-07-16T03:59:00-05:00", "rooms": {"Memorial Hall": [{"guid": "d1a3c8bc-e922-59a0-9f46-e1dd71039928", "code": "ZFNUEG", "id": 97802, "logo": null, "date": "2026-07-15T09:15:00-05:00", "start": "09:15", "end": "2026-07-15T10:00:00-05:00", "duration": "00:45", "room": "Memorial Hall", "slug": "scipy-2026-97802-opening-keynote-thomas-caswell-stories-in-code", "url": "https://pretalx.com/scipy-2026/talk/ZFNUEG/", "title": "Opening Keynote: Thomas Caswell, \"Stories in Code\"", "subtitle": "", "track": "Keynotes", "type": "Keynote", "language": "en", "abstract": "Matplotlib Project Lead and Computational Scientist at Brookhaven National Laboratory", "description": "Stories are a core to the human experience and core to our understanding of complex technical systems.  This talk will discuss the role that stories play in software in general and open source specifically.  The stories we collectively write and share, in the form of code, are the concrete artifacts we create. How we go about organizing the development of these stories and the relationships between people are the spirit of SciPy.", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/ZFNUEG/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/ZFNUEG/", "attachments": []}, {"guid": "5847e2d6-a779-5f03-9ff3-e845437cc4fc", "code": "YZU7X8", "id": 97804, "logo": null, "date": "2026-07-15T10:00:00-05:00", "start": "10:00", "end": "2026-07-15T10:25:00-05:00", "duration": "00:25", "room": "Memorial Hall", "slug": "scipy-2026-97804-scipy-tools-plenary", "url": "https://pretalx.com/scipy-2026/talk/YZU7X8/", "title": "SciPy Tools Plenary", "subtitle": "", "track": "SciPy Tools", "type": "Tools Plenary", "language": "en", "abstract": "A session featuring updates and roadmaps from maintainers of core Scientific Python libraries and tools.", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/YZU7X8/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/YZU7X8/", "attachments": []}, {"guid": "5987fdae-1082-5c1e-b876-dc6ebfca2813", "code": "YUVEYH", "id": 93254, "logo": null, "date": "2026-07-15T10:45:00-05:00", "start": "10:45", "end": "2026-07-15T11:15:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-93254-one-problem-many-projects-how-scientific-needs-built-an-ecosystem", "url": "https://pretalx.com/scipy-2026/talk/YUVEYH/", "title": "One Problem, Many Projects: How Scientific Needs Built an Ecosystem", "subtitle": "", "track": "Spirit of SciPy", "type": "Talk", "language": "en", "abstract": "In 2004, Matthew Brett asked me a provocative question born of frustration with existing fMRI tools: \"Why don't we rewrite them in Python?\" That question led to a 2005 meeting that brought together a small group of core Scientific Python tool builders from astronomy, neuroscience, physics, and statistics, and then to a series of follow-up meetings alternating between Berkeley, Enthought's offices, and other locations. This talk traces how that ground-up, cross-disciplinary collaboration helped turn SciPy from a workshop curiosity into the backbone of today's ecosystem, and how the same people and patterns later shaped the Scientific Python project.", "description": "This talk tells the story of Scientific Python's growth through the lens of one institution and one fateful question. It follows how a small, domain-driven collaboration at UC Berkeley helped catalyze tools, practices, and organizations that now define the Scientific Python ecosystem, and how SciPy and its conferences became a shared planning space that later informed cross-project efforts like the Scientific Python project.\n\nI begin in 2000--2004, when I joined UC Berkeley's Brain Imaging Center at a time when Python was only starting to be used seriously for numerical work. SciPy 0.1 had just been released, the first SciPy workshop at Caltech in 2002 drew only a few dozen scientists, and our neuroimaging work was dominated by large, opaque lab-owned research software. My colleague Matthew Brett and I wanted to build something better for fMRI analysis in Python, a goal that quickly pulled us into broader discussions about the future of Numeric, numarray, and SciPy's architecture. Those conversations ultimately matured into the Neuroimaging in Python (NIPY) project and a series of tools and publications that showed what it meant in practice to build domain-specific software on top of a young ecosystem---where we could lean on NumPy, SciPy, and matplotlib as they were, and where we had to contribute upstream to make the work possible.\n\nThe core of the talk focuses on the 2005--2007 period. A 2005 meeting at Berkeley brought together John Hunter (matplotlib), Fernando Perez (IPython), Travis Oliphant (then developing what became NumPy), Perry Greenfield (numarray/STScI), and others to sketch out concrete plans to unify on a single array core, refactor SciPy around that core, and treat SciPy as the base of a larger ecosystem rather than a monolithic library. Out of those conversations, and the broader discussions they sparked in the early developer community, came the decision to converge on NumPy, to split SciPy's functionality into a \"core\" plus separately maintained domain packages, and to prioritize packaging and installation so that scientists could actually adopt these tools. I will describe how this initial gathering turned into a series of small follow-up meetings---alternating between Berkeley, Enthought's offices in Austin, and other locations---that refined these ideas and effectively set the development roadmap for NumPy, SciPy, and the emerging Scientific Python ecosystem.\n\nThe third act zooms out to the conference and community layer. Beginning in 2007, I served as release manager for NumPy and SciPy and later chaired the SciPy conference (2008--2011) and edited its proceedings (2008--2013) as it evolved from a small workshop into an international venue with peer-reviewed papers. During that time, we also built the pre-Curvenote proceedings machinery, an early example of shared documentation and publishing infrastructure that supported reproducible research across projects. I will connect those roles to the founding of NumFOCUS in 2012, formalizing community infrastructure that had grown out of the same set of collaborations.\n\nFinally, I bring the story to the recent past. At Berkeley's Institute for Data Science we helped launch the Scientific Python project, including SPECs, cross-project tooling, and the Scientific Python developer summits, explicitly aiming to recreate the collaborative atmosphere of the early SciPy workshops in a modern, multi-project setting. The Berkeley Open Source Program Office now helps sustain this kind of cross-lab, cross-institution collaboration as part of the university's regular activity rather than a one-off effort.\n\nThroughout, the intended audience is broadly the SciPy community: developers, researchers, and practitioners who use the ecosystem daily. Attendees will learn:\n\n- How one domain-specific frustration (fMRI analysis software) helped catalyze cross-project collaboration at a critical moment for scientific Python.\n- How small, in-person meetings and local institutional support can have long-term ecosystem impact, from the first SciPy workshop through Enthought and INRIA to the Scientific Python developer summits.\n- How Berkeley's roles---as an early scientific user, as SciPy conference chair and editor, and now as a home for the Scientific Python project---fit into the larger history of Scientific Python.", "recording_license": "", "do_not_record": false, "persons": [{"code": "EKWFU8", "name": "Jarrod Millman", "avatar": null, "biography": "Jarrod Millman is the Executive Director for Berkeley's Open Source Program Office (OSPO). With a background in computer science, mathematics, and statistics, and degrees from Cornell and Berkeley, Millman is a founding member of the scientific Python ecosystem. His primary focus is on developing and sustaining open-source, community-owned scientific software tools. Millman serves on the steering council of NetworkX, is a core developer of scikit-image, and was an early contributor to NumPy, SciPy, and scikit-learn. He has co-founded several influential initiatives to advance open and reproducible research, including the Scientific Python project, the nonprofit NumFOCUS, and the Neuroimaging in Python project.", "public_name": "Jarrod Millman", "guid": "24565714-bc69-52cc-91e3-cbf205d3c36e", "url": "https://pretalx.com/scipy-2026/speaker/EKWFU8/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/YUVEYH/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/YUVEYH/", "attachments": []}, {"guid": "a5f02ae1-f5e2-5fe4-8d25-de73dfe0ed25", "code": "HBZ9RC", "id": 90436, "logo": null, "date": "2026-07-15T11:25:00-05:00", "start": "11:25", "end": "2026-07-15T11:55:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-90436-tying-up-loose-threads-making-your-project-no-gil-ready", "url": "https://pretalx.com/scipy-2026/talk/HBZ9RC/", "title": "Tying Up Loose Threads: Making your Project No-GIL Ready", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "If you messed around with Python's command line options or read the official documentation, you might wonder what the -Xgil option or the PYTHON_GIL environment variable did to your scripts, and whether setting either affects performance. The hubbub on popular wheels such as pyo3, python-zstandard, numpy, uv, cffi, and cython supporting the free-threaded interpreter is no passing fad either. For Pythonistas that don't read PEPs in their spare time or contribute to the cpython project itself, an adventure that delves into a less known, yet jaw-dropping aspect of Python awaits!\n\nPython's Global Interpreter Lock, which determines which single thread can execute native Python code and call C API functions, simplifies writing multithreaded code. However, sticking with this execution model leaves out extra performance afforded by modern multicore CPUs with hyperthreading, as automatic locking and unlocking of the GIL does not scale well with thread counts, especially in performance-sensitive workloads.\n\nThe newfangled free-threaded interpreter promises salvation when running either pure Python code or with compiled extensions. General multithreading rules apply (prefer thread-local variables, using locks to prevent simultaneous access of shared data), but when dealing with projects containing compiled extensions that directly or indirectly interface with Python's C API, more porting rules also apply.\n\nKey porting tips, including projects using the Limited API, include: port native code away from C API functions that avoid borrowed references because they aren't thread-safe; modify unit tests to catch concurrency bugs arising from assuming the presence of the GIL; and extend CI coverage of Python interpreters both for testing and to build free-threaded compatible wheels.\n\nOutline:\n* Introduction (2-3 min.)\n* What is the -Xgil option?\n* What is the GIL?\n* What is the free-threaded interpreter? (6-8 min.)\n* Global Interpreter Lock: downsides of automatic serialization of parallel workloads\n* How to try out the free-threaded interpreter\n* Increased parallelism with the no-GIL interpreter with multi-core CPUs\n* Porting tips (15-18 min)\n* Adding a trove classifier in pyproject.toml\n* Marking your extension module as supporting no-GIL\n* Limited API (and PEP 803)\n* Bumping key dependencies, including FFI wheels\n* Using locks, mutexes, and atomics in native code to prevent concurrency bugs\n* Including pytest-run-parallel to catch threading bugs\n* Closing Remarks (2 min.)\n* Q&A (2 min.)", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "CF8GNQ", "name": "Charlie Lin", "avatar": "https://pretalx.com/media/avatars/CF8GNQ_bi6rPLz.webp", "biography": "", "public_name": "Charlie Lin", "guid": "5a966d22-7d50-557a-9f59-b22f900098b2", "url": "https://pretalx.com/scipy-2026/speaker/CF8GNQ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/HBZ9RC/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/HBZ9RC/", "attachments": []}, {"guid": "f1783e0f-1a10-5483-b665-c359d205f15c", "code": "8QMU8G", "id": 88933, "logo": null, "date": "2026-07-15T13:15:00-05:00", "start": "13:15", "end": "2026-07-15T13:45:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-88933-ai-powered-field-inspection-voice-capture-data-extraction-and-intelligent-multi-source-routing", "url": "https://pretalx.com/scipy-2026/talk/8QMU8G/", "title": "AI-Powered Field Inspection: Voice Capture, Data Extraction, and Intelligent Multi-Source Routing", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "Field inspections in agriculture and science face a common problem: hands are full, data needs structure, and decisions require both inspection history and domain expertise. I built HiveGuide, an open-source field inspection system with three main components: (1) voice transcription for hands-free data entry, (2) AI extraction to structured data and action items, and (3) an AI assistant that provides intelligent advice by routing between personal inspection history and authoritative domain literature. For the assistant, I tested 7 routing strategies on 500+ queries to solve a dual-source problem: when to query your data versus domain references. The LLM classifier approach balanced accuracy and speed without requiring training data. The architecture is transferable to any inspection domain where you need minimal device interaction and intelligent advising.", "description": "## Background\nField work across agriculture, environmental monitoring, and scientific research shares common constraints: practitioners need structured data capture while their hands are occupied, dirty, or gloved. Current solutions\u2014voice memos, note apps, or digitized forms\u2014produce unstructured data that's difficult to query or analyze. More critically, field decisions require combining two distinct knowledge sources: personal inspection history (\"what's normal for my sites?\") and authoritative domain knowledge (\"what do experts recommend?\"). Generic AI assistants can't access your data; domain-specific apps don't leverage expert knowledge.\n\nI developed HiveGuide to solve this for beekeeping inspections, where you're holding frames with thousands of stinging insects while wearing propolis-covered gloves. The architecture proved generalizable to any inspection workflow requiring minimal interaction, structured capture, and intelligent advising.\n\n## Methods\nThe system is built on a Python FastAPI backend with PostgreSQL database, using LangChain for the RAG architecture and OpenAI APIs for transcription and language models. The frontend is React Native (iOS) and React Native Web, with the Python backend handling all AI/ML processing.\nThe system has three components:\n1. _Voice Transcription:_ Real-time streaming transcription (2-second delay) via iOS native app. Audio sent to server in chunks every few seconds, minimizing data loss risk compared to batch processing. Platform choice (native vs web) drove capability\u2014web apps can't achieve this latency or reliability.\n2. _AI Extraction:_ LLM converts voice notes to structured fields. Example: \"It's in the 60s and cloudy. Fresh eggs in good pattern, didn't spot the queen\" extracts temperature, queen_visible: False, eggs_visible: True, laying_pattern: \"solid\". Structured data enables querying and generates automated action items based on inspection findings.\n3. _AI Assistant with Intelligent Routing_: This solved the core technical problem. Field inspection questions require either personal data (\"Is my hive at normal weight?\"), domain knowledge (\"What causes bee dysentery?\"), or both (\"Is my hive's October weight normal for Wisconsin?\"). I implemented 7 routing approaches:\n    - LLM classifier (pre-classifies query intent)\n    - Heuristic rules (keyword matching)\n    - Embedding similarity (query vector vs source vectors)\n    - Supervised classifier (trained on labeled queries)\n    - Agent-based (agent decides tool usage)\n    - Hybrid combinations\n    - Always-both baseline\n\n    Each routed to SQL database (inspection history) and/or vector search with pgvector (RAG over authoritative sources). A LangChain agent synthesized retrieved context. Validation layer caught generic responses and forced retry.\n\n## Results\nTesting on 500+ queries:\n- Supervised classifier: 97.8% accuracy (highest), requires labeled training data\n- LLM classifier: 95.2% accuracy, ~1s overhead, no training needed\u2014selected for deployment\n- Agent-based: good retrieval, higher error rates from increased complexity\n\nThe LLM classifier balanced performance with practical deployment constraints. In use across multiple hives over several months, the system successfully generated structured inspection data, automated task lists, and provided contextualized advice combining personal history with domain references.\nTo mitigate hallucination risk, responses link directly to source materials with specific page citations.\n\n## Generalizability\nThis pattern applies wherever you need:\n- Minimal device interaction (hands busy/dirty)\n- Structured data for later analysis\n- Decisions based on inspection history + domain expertise\n\n_Examples: equipment maintenance, scientific field inspections, beekeeping, etc._\n\n## Conclusion\n\nThe dual-source routing problem appears across scientific and agricultural field work but lacks established solutions. Systematic testing of routing strategies showed LLM classifiers provide practical performance without training overhead. The architecture is open-sourced (Creative Commons NC) for adaptation to other inspection domains.\n\n## Links\n\n- GitHub: [github.com/CarolynOlsen/hiveguide_public](github.com/CarolynOlsen/hiveguide_public)\n- Medium writeup: [https://medium.com/@carolyn.olsen/ai-powered-field-inspection-app-design-for-agriculture-and-science-a4507b85e30e](https://medium.com/@carolyn.olsen/ai-powered-field-inspection-app-design-for-agriculture-and-science-a4507b85e30e)\n- arXiv pre-print on routing strategies comparison is upcoming", "recording_license": "", "do_not_record": false, "persons": [{"code": "CR3AXW", "name": "Carolyn Olsen", "avatar": "https://pretalx.com/media/avatars/CR3AXW_A0vqshB.webp", "biography": "Carolyn Olsen developed HiveGuide as an independent open-source project to solve practical field inspection challenges in beekeeping and generalized for broader scientific applications. In her day role, she is Director of Data Science at The Hartford, where she leads an AI Accelerator for two business areas, helping them leverage generative AI. Previously VP of Data Science at Clearcover, she has extensive experience developing production AI systems including LLM-powered tools, supervised ML models, and reinforcement learning models. She holds a Master of Science in Applied Economics from Marquette University and served 8 years in the U.S. Coast Guard Reserve.", "public_name": "Carolyn Olsen", "guid": "1e9ad473-78a4-5463-8579-b1e732464767", "url": "https://pretalx.com/scipy-2026/speaker/CR3AXW/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/8QMU8G/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/8QMU8G/", "attachments": []}, {"guid": "a08a0a11-e249-531d-9647-052bb6581d44", "code": "93ERQP", "id": 91127, "logo": null, "date": "2026-07-15T13:55:00-05:00", "start": "13:55", "end": "2026-07-15T14:25:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-91127-reno-simplifying-application-of-bayesian-inference-to-system-dynamics", "url": "https://pretalx.com/scipy-2026/talk/93ERQP/", "title": "Reno: Simplifying Application of Bayesian Inference to System Dynamics", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "Modeling and simulation enable iterative hypothesis testing and encoding subject matter expertise into reusable tools. While the Python community has a variety of libraries for modeling, few exist for system dynamics, a paradigm for top-down analysis of material and information flows over time. Reno is an open-source package combining creation, visualization, and analysis of system dynamics models with techniques for Bayesian inference through integration with PyMC, supporting probability distributions in system variables and MCMC sampling to produce posterior distributions based on observed values. This approach enables simulation and refinement of time series models where variables, policies, or knowledge are uncertain, and data/observations are sparse.", "description": "# Introduction\n\nSystem dynamics models provide a means for exploring complex systems and effects that can arise from concepts such as feedback loops and time delays. This type of modeling has applications in a wide variety of fields including biology, economics, operations management, and social sciences. Industry standard tools for implementing this modeling process include Vensim and AnyLogic, but require a budget and lack the ability to construct programmatically from within Python. Existing Python-based libraries such as PySD provide the means to run models created in other tools but not to build them directly. \n\nBayesian inference is a statistical tool for modeling with uncertainty and updating probability distributions based on potentially limited amounts of data. PyMC is an established library in the Python ecosystem that provides algorithms for Bayesian inference, but it can be challenging to use for implementing complex system dynamics models. The goal of this project is to provide a Python-based means for building system dynamics models with a straightforward API, and support refinement of unknown or highly uncertain variables through PyMC without requiring the developer to write extensive PyMC specific code. \n\n# System Dynamics Implementation \n\nWe present Reno, a new open-source library with an API that centers around symbolically constructing equations that are used to define and reference stock, flow, and variable components, collectively constituting a system dynamics model. Conceptually similar to libraries like PyTensor and PyTorch, these equations create a compute graph that can be populated and evaluated to produce simulation data. Reno models, once defined, are called like a normal Python function to run a simulation, optionally passing in parameters to configure specific system variables. These model calls can efficiently run many simulations in parallel, allowing exploration of parameter space with parameter sweeps or input distributions, with results returned as XArray datasets. \n\nThis section will discuss an example from a system dynamics textbook and show the process of implementing it in Reno along with possible visualizations and analyses of the system once created. \n\n# Incorporating Bayesian Inference \n\nBy default, a Reno equation evaluates by running corresponding NumPy operations on the data passing through the compute graph. Given the similar API of PyTensor, the mathematics library underlying PyMC, everything within Reno compute graphs can also directly translate into a set of PyTensor/PyMC operations. A Reno model is thus converted into a PyMC model by compiling the component equations that evaluate for a single timestep, then wrapping with the necessary boilerplate to initialize the model and run the timestep function for a full time series simulation. Reno encapsulates this conversion process with a single function call, requiring no additional PyMC code from the model developer. Any observed data or measurements that are included in the function call are set within likelihood distributions and subsequently used in PyMC's MCMC sampling algorithms to approximate posterior distributions. \n\nThis section will expand on the previous example, showing how an uncertain input variable can be provided a prior probability distribution to indicate incomplete or imperfect knowledge. Further demonstration will show how the distribution tightens/converges around the ground truth value as additional observed data points are supplied to the PyMC model calls. \n\n# Links \n\nProject repository: https://github.com/ornl/reno \nExample of a previous SciPy talk: https://youtu.be/uyfIQEoZPOo", "recording_license": "", "do_not_record": false, "persons": [{"code": "XY3B8Y", "name": "Nathan Martindale", "avatar": "https://pretalx.com/media/avatars/XY3B8Y_oc80UAN.webp", "biography": "Nathan Martindale is a data scientist in the Nuclear Nonproliferation Division at Oak Ridge National Laboratory. Nathan completed both his B.S. (2018) and M.S. (2020) degree in computer science at Tennessee Tech University, with his graduate studies focusing on machine learning. His recent research interests have included visual analytics, system dynamics, and knowledge management, and he has released several open source libraries supporting research experiment management, system dynamics model creation and analysis, and interactive machine learning.", "public_name": "Nathan Martindale", "guid": "84624616-b99e-5b78-b156-85939b7f7fb7", "url": "https://pretalx.com/scipy-2026/speaker/XY3B8Y/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/93ERQP/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/93ERQP/", "attachments": []}, {"guid": "4f073143-992f-598d-9cb1-7f04f7724098", "code": "MCEMT9", "id": 92333, "logo": null, "date": "2026-07-15T14:35:00-05:00", "start": "14:35", "end": "2026-07-15T15:05:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-92333-from-hello-world-to-hello-llm-a-python-developer-s-survival-guide", "url": "https://pretalx.com/scipy-2026/talk/MCEMT9/", "title": "From Hello World to Hello LLM: A Python Developer\u2019s Survival Guide", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "AI tooling is moving fast, but many Python developers are unsure where to start or how today\u2019s AI patterns fit into systems they already know how to build. This talk is a practical, hands-on overview of modern AI development patterns in Python, focused on what you need to know  to go from zero to hero. \nWe\u2019ll walk through a real-world coding example broken into parts that illustrate the core building blocks of modern AI applications, and explain when each pattern makes sense. This example is designed in a way that doesn\u2019t require any prior machine learning experience, and attendees will leave with an understanding of how AI systems work, what problems they\u2019re good at solving, and how to maintain and observe what has been built.\nTopics we\u2019ll cover:\nThe modern AI stack in Python: LLM APIs, embeddings, tools, and agents\n\n\nCommon Python AI patterns: prompts, function calling, RAG, and simple agents\n\n\nWhen to use a script vs an agent vs a service (and when not to)\n\n\nHow to get something working quickly without sacrificing reliability or safety\n\n\nPractical guardrails: handling errors, controlling outputs, and protecting data\nHow to generally stand up common AI workflows, such as LLM-powered scripts to  lightweight AI agents / MCP-style services. \n\n\nAttendees will leave with a clear map of the AI landscape, working Python patterns they can reuse immediately, and the confidence to start building AI features without needing a machine learning background.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "L7TDWM", "name": "Audrey Webb", "avatar": null, "biography": "I\u2019m a Machine Learning Engineer with experience spanning data science, data engineering, and AI platform development across research, enterprise, and product driven environments. I began my career in public health and infectious disease research, working with academic and government partners on large scale statistical modeling, record linkage, and population-level analysis. I later transitioned into industry, where I\u2019ve built and deployed production-grade data and machine learning systems that power real world products.\n\nMy work has covered the full ML lifecycle, including data pipelines, feature stores, model training, explainability, observability, and large scale deployment. I\u2019ve contributed to recommendation systems, ranking and propensity models, generative AI applications, and MLOps platforms that enable teams to ship and maintain models reliably in production.\n\nBeyond my core roles, I\u2019ve worked on applied AI and consulting projects, including an AI powered analytics platform for higher education institutions, and advisory work on computer vision and automation for e-commerce workflows. I\u2019m especially interested in scalable and interpretable AI, responsible deployment, and making advanced AI systems practical for real world teams. Outside of work, I enjoy mentoring, volunteering, traveling, and engaging in conversations around AI education, ethics, and regulation.", "public_name": "Audrey Webb", "guid": "c060b719-2c77-5890-bf76-3e9985601161", "url": "https://pretalx.com/scipy-2026/speaker/L7TDWM/"}, {"code": "BPCLT8", "name": "Jasmine Omeke", "avatar": null, "biography": "Jasmine Omeke is a senior software engineer at Airbnb, where she focuses on data infrastructure. Before joining Airbnb, she worked as a software engineer at Netflix and PayPal, specializing in distributed systems and large-scale data processing, and building backend services that handled petabytes of data. She has experience creating eLearning content and authored a Python testing course on LinkedIn Learning. Jasmine also mentors aspiring computer science students through CodePath\u2019s technical interview preparation program. A former Gates Millennium Scholar, she is excited to give back to the community that supported her early academic journey. Outside of work, Jasmine enjoys swimming and sewing. She holds a Bachelor of Arts from Harvard University and a Master of Computer Science from DePaul University.", "public_name": "Jasmine Omeke", "guid": "47109a13-7a6d-59b3-800d-9b37c657db2d", "url": "https://pretalx.com/scipy-2026/speaker/BPCLT8/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/MCEMT9/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/MCEMT9/", "attachments": []}, {"guid": "c5f38149-1088-5c9f-8d2b-8df98f6e9504", "code": "SEYACQ", "id": 92356, "logo": null, "date": "2026-07-15T15:25:00-05:00", "start": "15:25", "end": "2026-07-15T15:55:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-92356-docling-for-multimodal-retrieval", "url": "https://pretalx.com/scipy-2026/talk/SEYACQ/", "title": "Docling for Multimodal Retrieval", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "Scientific breakthroughs don\u2019t happen in plain text, they live inside multi-column research papers, dense data tables, and intricate simulation diagrams. Yet the moment standard AI and Retrieval-Augmented Generation (RAG) pipelines encounter these layouts, they fail. Tables are flattened into meaningless strings. Figures are ignored. The structural signals that drive scientific reasoning disappear.\n\nIn this talk, we show how to rescue scientific knowledge from the \u201ctext-flattening\u201d trap using _Docling_, an open-source document understanding library designed to preserve layout, hierarchy, and element boundaries. Instead of reducing everything to text, we treat tables, figures, and sections as first-class data structures. Attendees will experience a live demo of a realistic scientific R&D workflow: uploading multiple dense technical PDFs, executing cross-document natural language queries, and successfully retrieving synthesized insights from text, structured tables, and scientific images", "description": "When analyzing simulation reports, experimental summaries, and technical journals, researchers must extract evidence, compare results, and validate claims across multiple sources simultaneously. To enable this rigorous level of analysis, we will walk through the technical implementation of a structure-aware ingestion pipeline.\n\nUsing Python, we will demonstrate an architecture that decomposes document layouts into distinct semantic elements - sections, tables, and figures. This approach preserves experimental results as queryable data structures and converts diagrams into searchable semantic signals, all while maintaining the strict document hierarchy required for context-aware retrieval. Building on this foundation, we detail the construction of a hybrid retrieval system that actively supports:\n- Cross-document comparison\n- Numeric reasoning over extracted tables\n- Linking textual claims to supporting figures\n- Combining text, structured data, and visual insights in a single grounded response\n\n#### Outline\n\n- The Scientific Workflow Challenge \n- Structure-Aware Ingestion\n- Preserving and Querying Tables\n- Visual Representation and Linking\n- Live Demo & Multimodal Retrieval\n- Q&A\n\nParticipants will gain a practical design pattern for building multimodal, structure-preserving retrieval systems that strengthen scientific reasoning and data-driven analysis.\n\n## Resources\n\n- \ud83d\udcd1 **Slides:** [Docling for Multimodal Retrieval](https://docs.google.com/presentation/d/1HrMgopkjV8sU8sT63W0DTnUhr-qQU2bN6FJ_xrxW900/edit?usp=sharing)\n- \ud83d\udcbb **Repository:** [multimodal-parser](https://github.com/mahimaarora/multimodal-parser)\n- \u270d\ufe0f **Blog:** [Multimodal Parsing for RAG: Seeing Diagrams and Reading Tables with Docling](https://medium.com/@mahimaarora025/multimodal-parsing-for-rag-seeing-diagrams-and-reading-tables-with-docling-6079668361fb)", "recording_license": "", "do_not_record": false, "persons": [{"code": "VPFWJA", "name": "Mahima Arora", "avatar": "https://pretalx.com/media/avatars/KEHTJF_rdJZiaX.webp", "biography": "Mahima Arora is a Senior Data Scientist on the Data & AI team at Red Hat, specializing in Generative AI applications. She develops AI-powered solutions that enhance efficiency and effectiveness, leading initiatives to optimize AI systems for greater impact. Passionate about open source, Mahima actively explores emerging tools and technologies to drive innovation and knowledge sharing, and has presented her work at PyData Amsterdam 2025 and PyCon India 2025.", "public_name": "Mahima Arora", "guid": "588f5b0e-aa66-5979-90f4-3445a2b57399", "url": "https://pretalx.com/scipy-2026/speaker/VPFWJA/"}, {"code": "9TQTHL", "name": "Aarti Jha", "avatar": "https://pretalx.com/media/avatars/SFVFTF_uQLwnBl.webp", "biography": "Aarti Jha is a Principal Data Scientist at Red Hat, where she leads the development of AI-driven solutions that streamline internal operations and reduce costs. She has more than seven years of experience designing and deploying machine learning and generative AI solutions across multiple industries. A frequent speaker at developer and data science conferences, she shares practical insights on applied AI, LLMs, and building AI systems that deliver measurable business value.", "public_name": "Aarti Jha", "guid": "d936acba-6849-52e7-a592-9f3023ff8a58", "url": "https://pretalx.com/scipy-2026/speaker/9TQTHL/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/SEYACQ/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/SEYACQ/", "attachments": []}, {"guid": "0a76389b-15be-5d2e-81cd-785ff81344a7", "code": "3TBXB8", "id": 92490, "logo": null, "date": "2026-07-15T16:05:00-05:00", "start": "16:05", "end": "2026-07-15T16:35:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-92490-the-future-of-ocr-structured-text-extraction-with-llms", "url": "https://pretalx.com/scipy-2026/talk/3TBXB8/", "title": "The future of OCR? Structured text extraction with LLMs", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "Optical character recognition (OCR) has been a long standing method of extracting text data from images. Traditional OCR models rely on pattern recognition and feature extraction using computer vision techniques and specialized Python libraries. Recently, large language models (LLMs) and generic AI assistants have provided an alternative method of text extraction. This talk explores the efficacy of using LLMs and VLMs for information extraction in production data pipelines and a data-driven approach for evaluating them against traditional OCR methods in terms of accuracy, reliability, latency, and cost.", "description": "Structured data extraction is a classic problem that has applications to many domains, such as document digitization, information extraction, and accessibility. There is great potential for LLMs to enhance automation for routine document processing tasks, but there are notable engineering risks associated with integrating these models into production data pipelines. LLMs can produce inconsistent outputs, produce hallucinations and confabulations, and are vulnerable to prompt injection. When evaluating the efficacy of OCR solutions, it's important to define metrics that capture not only accuracy but also latency, cost, and energy expenses.\n\nThis talk explores the benefits and challenges of applying LLMs to extracting text from scanned images by contrasting three approaches. First, I will explore object detection approaches using the open source docling and RF-DETR Python libraries which directly identify characters and words from images. I will also discuss the docTR library which applies deep learning models to text recognition.\n\nNext, I will explore how state-of-the-art LLMs and AI assistants such as Gemini, Claude, and Qwen can be applied to targeted text extraction tasks. This includes a data-driven evaluation strategy that utilizes both automated and human feedback to compare LLM-based approaches to traditional OCR.\n\nFinally, I will discuss a hybrid approach that combines traditional OCR methods with LLMs. This is a two-stage process that uses an OCR model to extract text from the image, then passes the unstructured text data to an LLM to produce a structured output.\n\nThis talk is for data scientists and machine learning engineers who are interested in prototyping and evaluating text extraction solutions in Python. I will walk through several Python code examples for structured extraction using open source libraries such as docling and docTR and demonstrate how to experimentally validate those methods against modern machine learning approaches that utilize LLMs and VLMs.\n\n### Outline\n\n1. Traditional OCR techniques (5 minutes)\n    a. Object detection approaches with docling and RF-DETR\n    b. Deep learning with open source models and the docTR library\n2. Text extraction with LLMs (5 minutes)\n    a. Extracting structured outputs with pydantic\n    b. Prompt engineering\n    c. Self-hosted vs. managed service models\n3. Hybrid approach (5 minutes)\n    a. Combining traditional OCR with LLMs\n    b. Profiling performance metrics\n4. Evaluating text extraction approaches (10 minutes)\n    a. Automated vs. human evaluation\n    b. Cost metrics (latency, compute and API expenses, energy)\n    c. Creating an evaluation framework", "recording_license": "", "do_not_record": false, "persons": [{"code": "NM9HKJ", "name": "Patrick Deziel", "avatar": "https://pretalx.com/media/avatars/JJ9VUS_HDqayzq.webp", "biography": "Patrick Deziel is a machine learning engineer and Python and Go programmer. Patrick has extensive experience building machine learning powered applications and contributing to open source projects such as Yellowbrick, an ML visualization library written for Python. He currently works at Rotational Labs where he builds software to support prototyping and evaluation of AI/ML powered solutions.", "public_name": "Patrick Deziel", "guid": "35bac99d-dfd5-5582-bb28-d6f360d314df", "url": "https://pretalx.com/scipy-2026/speaker/NM9HKJ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/3TBXB8/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/3TBXB8/", "attachments": []}, {"guid": "fd676bb2-cc1d-5d20-85b6-361fc0251cea", "code": "HZVFJS", "id": 97786, "logo": null, "date": "2026-07-15T17:00:00-05:00", "start": "17:00", "end": "2026-07-15T18:00:00-05:00", "duration": "01:00", "room": "Memorial Hall", "slug": "scipy-2026-97786-lightning-talks", "url": "https://pretalx.com/scipy-2026/talk/HZVFJS/", "title": "Lightning Talks", "subtitle": "", "track": "Lightning Talks", "type": "Lightning Talk", "language": "en", "abstract": "Lightning talks are 5-minute talks on any topic of interest for the SciPy community. We encourage spontaneous and prepared talks from everyone, but we can\u2019t guarantee spots. Sign ups are at the NumFOCUS booth during the conference.", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/HZVFJS/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/HZVFJS/", "attachments": []}], "Johnson Great Room": [{"guid": "458d4f28-a751-5692-a23c-082a9bc5ae62", "code": "CWCSEB", "id": 103283, "logo": null, "date": "2026-07-15T10:45:00-05:00", "start": "10:45", "end": "2026-07-15T11:15:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-103283-first-timer-orientation", "url": "https://pretalx.com/scipy-2026/talk/CWCSEB/", "title": "First-Timer Orientation", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "Are you a first time SciPy attendee? Confused about what the conference elements are or how to get the most out of your experience? Come join us for a casual chat, orientation, and suggestions for maximum fun and learning!", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "LEGKJZ", "name": "Julie Hollek", "avatar": "https://pretalx.com/media/avatars/MZVXN3_4f4q4tM.webp", "biography": "", "public_name": "Julie Hollek", "guid": "ce2160f5-5369-5841-911d-65999c58b85f", "url": "https://pretalx.com/scipy-2026/speaker/LEGKJZ/"}, {"code": "JKPGTQ", "name": "Ed Rogers", "avatar": "https://pretalx.com/media/avatars/SGXLB7_WcBY5Ik.webp", "biography": "", "public_name": "Ed Rogers", "guid": "ee353d38-8822-5359-be5b-7ca8dc3b7e07", "url": "https://pretalx.com/scipy-2026/speaker/JKPGTQ/"}, {"code": "GDZSZV", "name": "Ariana Mendible", "avatar": "https://pretalx.com/media/avatars/3K3RH3_bbG0y1W.webp", "biography": "", "public_name": "Ariana Mendible", "guid": "3e3a20b7-24ae-5e2a-bc60-6e793e7966c5", "url": "https://pretalx.com/scipy-2026/speaker/GDZSZV/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/CWCSEB/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/CWCSEB/", "attachments": []}, {"guid": "b7189b1d-7208-593a-bfe0-21bcb74f4c8f", "code": "9UQN9C", "id": 92471, "logo": null, "date": "2026-07-15T11:25:00-05:00", "start": "11:25", "end": "2026-07-15T11:55:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-92471-automated-data-enrichment-for-police-accountability-where-agentic-judgment-earns-its-place", "url": "https://pretalx.com/scipy-2026/talk/9UQN9C/", "title": "Automated Data Enrichment for Police Accountability: Where Agentic Judgment Earns Its Place", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "Automated data enrichment, filling missing fields in structured records from unstructured sources, is the canonical case for pointing an autonomous agent at a database and letting it fill every blank. In high-stakes data that instinct is dangerous. A confidently wrong value is worse than a blank, and retrieval-grounded extraction reduces but does not remove the tendency to assert what the source never stated. An LLM can extract these fields; this paper asks where agentic judgment earns its place and where it becomes a liability.\n\nWe study this on the Texas Justice Initiative's police shooting databases, where nearly two thousand records are missing the weapon, the subject's race, or the outcome, whose fields volunteers typically recover by hand, fifteen to thirty minutes each. Our LangGraph pipeline, deterministic in its control flow, searches, validates, extracts, and escalates hard cases to a human. It completes 92% of officer and 70% of civilian records and invents zero facts across twenty fabricated incidents. The recovery itself came from a deterministic prompt fix without any agent. An autonomous agent pointed at the same fabricated incidents, with more freedom, commits a wrong-article fabrication the pipeline escalates.\n\nIf the deterministic core does the recovering, the agentic layer earns its place by making those recovered values trustworthy. Agency lives only in this thin judgment layer above extraction, and its components act in one of two ways. One acts on the pipeline's control flow: a relevance judge reads the retrieved articles and, when none actually report this incident, routes the record to a human instead of completing it. The other two pass judgment on what extraction produced: one deletes a value the source never states, and the other explains to the reviewer why the sources disagree on a value. Extraction calls an LLM too, but because it only proposes values for these judges to rule on, we do not count it as agentic. Every judge had to clear a reward-hacking-resistant evaluation gate before it shipped. The main contribution of this paper is a discipline, an \"earn-it\" protocol, for drawing the line between what a high-stakes pipeline should settle deterministically and where it is worth granting agentic judgment.", "description": "## The Data Quality Problem in Public Accountability\n\nIn Texas, the Attorney General is required by law to collect a report on every officer-involved shooting, but the published summaries are high-level. The Texas Justice Initiative (TJI), a nonprofit, re-publishes the underlying incident records with the granularity that makes independent analysis possible. Yet those records are incomplete: across 1,956 incidents (2014\u20132024), 57% of civilian records are missing the weapon, 22.5% are missing the subject's name, and 39% of officer records are missing the officer's name. Volunteers recover these gaps by hand from news archives, fifteen to thirty minutes per record, because journalists routinely name people and describe circumstances that mandatory government filings omit. The author has volunteered with TJI since 2019 and co-authored its published report on officer-involved shootings.\n\n## An Agentic Enrichment Pipeline\n\nThe system is a seven-node LangGraph pipeline that searches the web, extracts fields, and escalates hard cases to a human. But it is mostly a *deterministic workflow*, not an autonomous agent, and that restraint is the point. After a **Load** node pulls a record from PostgreSQL, a deterministic **Coordinator** hub routes every transition through **Search** (Tavily), **Validate** (rule-based date/location/name checks), and **Synthesize** (LLM extraction), ending at **Complete** or **Escalate**. The single largest recovery gain came from a deterministic, dataset-aware extraction prompt, with no agent involved; a reasoning-and-acting loop tested on the same failures recovered nothing more. What ships as genuinely agentic is a thin *faithfulness* layer of three bounded LLM judges with graduated authority: a **relevance judge** that can *block* a record whose articles describe a different shooting, a **race verifier** that *nulls* a race the source never explicitly states, and a **conflict annotator** that only *advises* a human reviewer. The pipeline never writes back to the source database; escalation is a first-class outcome, and every committed value carries a confidence label and its source URLs.\n\n## Evaluation\n\nEvaluation is treated as first-class engineering, not an afterthought. On held-out samples (100 records per dataset, with ground-truth fields hidden from the pipeline and compared only afterward), it completes 92% of officer records and 70% of civilian records, with exact-match field accuracy of 71\u201377% (86\u201389% under fuzzy match); officers complete more often because their shootings draw denser coverage. On a deliberately adversarial probe of 20 fabricated incidents (invented names placed in real Texas cities on real dates, six engineered as traps so that real articles about the *wrong* person would pass date and location checks), the pipeline invented zero facts and escalated all twenty. Run head-to-head on those same twenty incidents, a fully autonomous agent with free-text search, open web access, and none of the pipeline's guards declined most but completed one fabricated record the pipeline escalates, reproducibly and with no signal to warn a reviewer, at several times the cost. A generic instruction to cite sources is not the same as a mechanism with the authority to act on it. The safeguard behind every change is a reward-hacking-resistant, multi-objective evaluation gate: a pure function over two saved reports that scores completion, enforces a hard zero-hallucination veto, and checks field-level correctness on a *stable cohort*, so a completion gain can never launder a correctness loss. End to end, a record costs roughly $0.20 (about $400 for the full archive), set against the hundreds of volunteer-hours the manual workflow would take. Per-race completion is reported as a non-gating diagnostic, kept visible to a human reviewer rather than acted on automatically at these group sizes.\n\n## Design Principles and Broader Applicability\n\nFour principles, each a stance on a tradeoff, shaped the system: do everything deterministic first, because every place a model may choose is a place it can choose wrong; calibrate each component's authority to how sharply it can decide (block, null, or advise); prefer faithfulness over coverage, because in an accountability database a wrong value is worse than a blank; and distrust the headline metric, because \"complete more records\" is trivially gamed by accepting weak extractions. No agentic component shipped on intuition; each had to clear an offline \"earn-it\" gate, and four of the seven ideas tried were gated out, failed, deferred, or declined, which is as much the contribution as the three that shipped. The discipline rests on three domain-general preconditions rather than on TJI specifics: a held-out signal to score against, a hard-veto safety metric that no other gain may override, and decisions that can be ranked by stakes. Where those hold (public-health surveillance, environmental incident tracking, historical archives, and other domains where structured databases have gaps that scattered public sources could fill), the pattern should carry.\n\n## What Attendees Will Learn\n\n- The workflow-versus-agent design axis: where an LLM earns its place, and where a deterministic rule or prompt quietly beats one, shown by a head-to-head in which an unconstrained autonomous agent fabricates a record the bounded pipeline escalates\n- How to build a reward-hacking-resistant, multi-objective evaluation gate that cannot be satisfied by trading correctness for completion\n- Calibrating component authority to stakes (block / null / advise), and designing human-in-the-loop escalation as a first-class outcome rather than a failure\n- An \"earn-it\" protocol for admitting agentic components only after they clear an offline gate, including the null results that kept components out\n- Testing and mocking patterns for pipelines that depend on external web-search APIs and LLMs\n- Using LangGraph deliberately narrowly: typed state, deterministic routing, and a clean seam to inject or mock every model call\n\n### Target audience\n\nData scientists, ML/AI engineers, and scientific Python users interested in applied agentic AI, LLM evaluation, data quality, or civic tech.\n\n### Source code\n\n[github.com/hongsupshin/police-data-intelligence](https://github.com/hongsupshin/police-data-intelligence) (open source, MIT, with tests and CI via GitHub Actions)\n\n### Related publication\n\n[Officer-Involved Shootings in Texas: 2016-2019](https://texasjusticeinitiative.org/publications/officer-involved-shootings-in-texas)\n\n### Speaking experience\n\nThe author has presented at academic conferences and industry events on data science and machine learning topics. Video recordings are available from a [Texas Justice Initiative presentation](https://drive.google.com/file/d/1aXvxJ8E4pP9uE4WU0dW3A4ncglxvZhCH/view) and an [Austin Python Meetup community meetup talk](https://www.youtube.com/watch?v=gfqKaplRTsk).", "recording_license": "", "do_not_record": false, "persons": [{"code": "UUKACJ", "name": "Hongsup Shin", "avatar": "https://pretalx.com/media/avatars/UPLLND_kVopd6R.webp", "biography": "Hongsup Shin is a Senior AI & LLM Engineer at NVIDIA's Silicon Co-design Group, building agentic systems and automation for silicon engineering workflows. His work spans production RAG infrastructure, multi-agent systems, learning-to-rank for hardware verification, and human-in-the-loop system design. He has published at IEEE SOCC (2022, 2024) and previously presented at SciPy 2019 on ML applications for hardware failure detection. He currently serves as SciPy Conference Proceedings Chair, founded the Austin ML Journal Club, and has volunteered with the Texas Justice Initiative since 2019, where he authored TJI's first data analytics report on officer-involved shootings in Texas. He holds a Ph.D. in Neuroscience from Baylor College of Medicine.", "public_name": "Hongsup Shin", "guid": "93177d74-2de2-56f8-81c9-8ae61cf28477", "url": "https://pretalx.com/scipy-2026/speaker/UUKACJ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/9UQN9C/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/9UQN9C/", "attachments": []}, {"guid": "8c4a2588-7c48-54bd-b177-33cb761d75f8", "code": "UFF7UR", "id": 87810, "logo": null, "date": "2026-07-15T13:15:00-05:00", "start": "13:15", "end": "2026-07-15T13:45:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-87810-ship-it-or-skip-it-when-how-to-upgrade-your-open-source-dependencies", "url": "https://pretalx.com/scipy-2026/talk/UFF7UR/", "title": "Ship It or Skip It? When & How to Upgrade Your Open Source Dependencies", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "Upgrading your organization\u2019s dependencies on open source libraries can be daunting. Major version releases promise bug fixes, new features, and security improvements, but these upgrades often require so much more work than just bumping a few numbers and letting your package manager sort out the rest.\n\nFrom planning to deployment, this talk is a step-by-step guide to upgrading your dependencies on open source libraries. We will offer practical strategies for scoping, coordinating, debugging, testing, releasing, and communicating major version upgrades -- all with as little pain for developers and users as possible.\n\nWhether you're maintaining internal extensions, forking core packages, or just trying to stay current, you'll learn real-world strategies to make major upgrades less painful, and maybe even routine.", "description": "Major dependency upgrades are daunting, and the tail of downstream projects and end-user organizations putting them off is always long. \n\nDevelopers often find good reasons to delay major upgrades of dependencies.  Resources are scarce, managers may see more value in introducing new features or fixing active bugs, and upgrading comes with the risk of regressions.\n\nHowever, delaying major upgrades hurts everyone:\n\n- Users must wait for security fixes and new features\n- Developers who put off upgrades end up facing a mountain of changes to simultaneously research, implement, test, and release\n- Upstream maintainers must choose between dropping support for still-popular versions or devoting resources to trying to maintain every version under the sun by backporting fixes to old branches\n\nFaster, more timely upgrade adoption means more platform stability for everyone.\n\nIn this talk, we\u2019ll share lessons learned from years of major upgrades to our platform\u2019s dependencies on Jupyter ecosystem projects (Lab, Widgets, Server, Voila). We aim to provide attendees the strategies -- and confidence -- they need to tackle the next big upgrade long before the typical end-of-maintenance scramble.", "recording_license": "", "do_not_record": false, "persons": [{"code": "BABXQK", "name": "Rebecca Ely", "avatar": "https://pretalx.com/media/avatars/BABXQK_FcuSOZq.webp", "biography": "Rebecca Ely has been involved with Bloomberg\u2019s adoption of open source technologies since 2016. In 2022, Ely joined both the Jupyter Frontends (formerly JupyterLab) Council and the Jupyter Accessibility Council. Ely\u2019s 2015 career transition into tech was preceded by roles as a federal acquisition consultant, math and science teacher, and service monkey trainer. Ely holds a bachelor's degree in peace and justice studies from Wellesley College.", "public_name": "Rebecca Ely", "guid": "61c5b12d-0035-5188-ac06-0c95e6b1161c", "url": "https://pretalx.com/scipy-2026/speaker/BABXQK/"}, {"code": "XETPPQ", "name": "Balaji Sundaram", "avatar": null, "biography": "Balaji Sundaram has been a software engineer at Bloomeberg since 2018, where he works on a team that builds and maintains JupyterLab extensions for data analysis in BQuant. Prior to joining Bloomberg, Balaji worked on building greenfield products for a consulting firm. Balaji holds a bachelor's degree in computer science from North Carolina State University.", "public_name": "Balaji Sundaram", "guid": "9680c00c-8b1b-5e82-b2e3-1ff96756232d", "url": "https://pretalx.com/scipy-2026/speaker/XETPPQ/"}, {"code": "HVJ8QU", "name": "Shruti Sapre", "avatar": "https://pretalx.com/media/avatars/ZZC3X3_13317pH.webp", "biography": "Shruti Sapre has been a software engineer at Bloomberg since 2022, where she works on a team that builds and maintains JupyterLab extensions for data analysis in BQuant. Before joining Bloomberg, she worked on static analysis tools for MATLAB. She holds a master's degree in computer science from the University of Southern California, and a bachelor\u2019s degree in computer engineering from the University of Pune.", "public_name": "Shruti Sapre", "guid": "cfd434c0-1bb5-5458-8f3b-5feff0ae9733", "url": "https://pretalx.com/scipy-2026/speaker/HVJ8QU/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/UFF7UR/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/UFF7UR/", "attachments": []}, {"guid": "5d5b0aad-ea42-59fd-b2e4-041efd8810d8", "code": "CV8VEH", "id": 91050, "logo": "https://pretalx.com/media/scipy-2026/submissions/CV8VEH/image_EaCAME3.webp", "date": "2026-07-15T13:55:00-05:00", "start": "13:55", "end": "2026-07-15T14:25:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-91050-assessing-the-entrepreneurship-option-in-uncertain-times", "url": "https://pretalx.com/scipy-2026/talk/CV8VEH/", "title": "Assessing the entrepreneurship option in uncertain times", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "As funding levels have fallen in both the public sector and startup investing, many in the SciPy community are facing uncertain career futures or even job loss. Others may simply dream of being their own boss. This session will guide participants through a personalized analysis of the feasibility of securing paid work outside of formal employment, including solo consulting, building a product or service business with a team, or joining an existing tech startup. The session will also touch on tips for starting with limited funding, reducing unnecessary risk, leveraging open-source community resources, and pursuing next steps.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "SUS9RF", "name": "Jocelyn Graf", "avatar": "https://pretalx.com/media/avatars/SUS9RF_xrIUq5i.webp", "biography": "Jocelyn Graf is an Entrepreneur in Residence at the University of California Riverside and advises small businesses through the Los Angeles Small Business Development Centers (SBDCs). Her expertise includes helping technical and non-technical people communicate and collaborate. She also helps academics and entrepreneurs apply for federal Small Business Innovation Research (SBIR) grants to commercialize their discoveries. In South Korea, she worked at Samsung and then founded, grew, and sold a biomedical & engineering research editing & translation company. In LA, she has also been a college STEM Director and founded a workforce training company that helped early-career engineers get fun product development experience by designing and building hardware for escape rooms and gaming conventions. Jocelyn works remote from Seattle, is an avid cyclist, and volunteers helping community organizations migrate from big tech platforms to open source alternatives. She is currently working on learning to code in a more \u201cpythonic\u201d style.", "public_name": "Jocelyn Graf", "guid": "c09a7719-1a11-5608-a532-a507b86ada04", "url": "https://pretalx.com/scipy-2026/speaker/SUS9RF/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/CV8VEH/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/CV8VEH/", "attachments": []}, {"guid": "448fecc5-c910-504c-8156-5bc127abd311", "code": "77PGCB", "id": 91679, "logo": null, "date": "2026-07-15T14:35:00-05:00", "start": "14:35", "end": "2026-07-15T15:05:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-91679-profiling-python-gpu-code", "url": "https://pretalx.com/scipy-2026/talk/77PGCB/", "title": "Profiling Python GPU Code", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "Your GPU is fast, so why does your Python code still feel slow? This talk shows a practical, Python-first profiling workflow with Nsight Systems, Nsight Compute, and NVTX for CuPy, Numba, PyTorch, JAX, and CUDA extensions. We will use timelines to find launch overhead, hidden synchronizations, and host-device copies, then drill into kernel bottlenecks like memory throughput and occupancy. You will leave with a repeatable loop for turning profiles into measurable speedups.", "description": "Your GPU is fast, so why does your Python code still feel slow?\n\nWhen you accelerate Python with CuPy, Numba, PyTorch, JAX, or custom CUDA extensions, performance problems rarely look like a single slow kernel. They look like death by a thousand cuts: tiny launches, hidden synchronizations, accidental host-device copies, stream serialization, and kernels that are \"fine\" until you look at memory traffic. The good news is that NVIDIA's developer tools can make these issues obvious, if you know what to capture and how to read it.\n\nIn this talk, I'll show a practical, Python-first profiling workflow using Nsight Systems, Nsight Compute, and NVTX. We'll start at the top with system-level timelines to answer \"where did the time go?\" then drill down into kernel-level analysis to answer \"why is this kernel slow?\" Along the way, you'll learn how to annotate Python code with NVTX so your traces are readable, how to profile from notebooks and CI, and how to turn profiler output into a short, repeatable optimization loop.\n\nKey takeaways:\n- How to use NVTX ranges and markers from Python to make timelines explain themselves.\n- How to capture the right Nsight Systems trace to spot launch overhead, sync points, copies, and stream issues.\n- How to pivot from a timeline hotspot to Nsight Compute and choose metrics that actually answer your question.\n- How to interpret common kernel bottlenecks (memory throughput, occupancy limits, instruction mix) without drowning in counters.\n- A checklist for avoiding profiling traps (implicit sync, warmup, clock variability, sampling noise, and \"profiling changed my code\").\n- A repeatable workflow you can apply to real Python GPU stacks, from single kernels to end-to-end pipelines.\n\nBy the end, you'll be able to profile Python GPU code with intent, isolate the bottleneck you actually have, and make changes you can measure and defend.", "recording_license": "", "do_not_record": false, "persons": [{"code": "VKG8RE", "name": "Bryce Adelstein Lelbach", "avatar": "https://pretalx.com/media/avatars/VKG8RE_qfi0eBh.webp", "biography": "Bryce Adelstein Lelbach has spent over a decade developing programming languages, compilers, and libraries. He is passionate about parallel programming and strives to make it more accessible for everyone.\n\nBryce is a Principal Architect at NVIDIA, where he founded the Core C++ Compute Libraries team and now leads the Vanguard Programming group that drives NVIDIA's roadmap for programming languages, compilers, and core libraries.\n\nHe is a leader of the systems programming language community, having served as chair of the C++ Library Evolution and the US programming language standards committee. He has been an organizer and program chair for many conferences over the years. On the C++ committee, he has worked on concurrency primitives, parallel algorithms, senders, and multidimensional arrays.\n\nHe previously worked at Lawrence Berkeley National Laboratory and Louisiana State University. He is one of the founding developers of the HPX parallel runtime system. \n\nOutside of work, Bryce is passionate about airplanes and watches. He lives in Midtown Manhattan with his girlfriend and dog.", "public_name": "Bryce Adelstein Lelbach", "guid": "2d0de9a8-9373-5b37-85ce-a11d6e0bef3d", "url": "https://pretalx.com/scipy-2026/speaker/VKG8RE/"}, {"code": "GB9NMF", "name": "Bradley Dice", "avatar": "https://pretalx.com/media/avatars/P7A8CE_sT4iFvO.webp", "biography": "Bradley Dice is a Senior Software Engineer in GPU-Accelerated Data Analytics at NVIDIA, designing high-performance open-source libraries for data analytics (cuDF) with modern CUDA, C++, and Python.", "public_name": "Bradley Dice", "guid": "abe9f695-96fc-5ba7-956d-6e87a4851247", "url": "https://pretalx.com/scipy-2026/speaker/GB9NMF/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/77PGCB/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/77PGCB/", "attachments": []}, {"guid": "9061ec92-19b3-5637-8f63-5037e6aa9906", "code": "SUPRRW", "id": 91830, "logo": "https://pretalx.com/media/scipy-2026/submissions/SUPRRW/image_I4OrfPd.webp", "date": "2026-07-15T15:25:00-05:00", "start": "15:25", "end": "2026-07-15T15:55:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-91830-electrifying-aviation-with-python-an-end-to-end-data-pipeline-from-test-stand-to-analytics", "url": "https://pretalx.com/scipy-2026/talk/SUPRRW/", "title": "Electrifying Aviation with Python: An End-to-End Data Pipeline from Test Stand to Analytics", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "This talk showcases a complete Python-based data pipeline for capturing and analyzing test data from electric motors powering BETA Technologies' fully electric CTOL (Conventional Takeoff and Landing) and VTOL (Vertical Takeoff and Landing) aircraft. We demonstrate how Python's open-source ecosystem enables seamless integration from edge to analytics: custom loggers decode machine data; a home-built data service batches and stores raw data in Apache Iceberg on AWS; dbt defines transformations that load into Redshift for analytics; Trino supports querying and joining to data from other sources; and Grafana serves visualizations, all provisioned via AWS CDK in Python. By leveraging modern data infrastructure and cloud solutions, we built an accessible, maintainable solution that handles terabyte-scale test data.", "description": "### Background and Motivation\nElectric aviation represents a critical frontier in sustainable transportation, and Vermont-based BETA Technologies is pioneering this transformation. BETA is taking a methodical \"crawl, walk, run\" approach to FAA certification: first our H500A electric motor, then our ALIA CTOL (Conventional Takeoff and Landing) aircraft, and finally our ALIA VTOL (Vertical Takeoff and Landing) aircraft.\nFast design iteration is fundamental to our development process, making quickly accessible, accurate data integral to the business. Our testing program generates terabytes of data as we validate motor performance and safety. Engineers need access to both near real-time and historical data, supporting everything from millisecond-level debugging to fleet-wide trend analysis. For FAA certification, we must preserve raw data indefinitely, and the pipeline architecture itself must be simple and explainable to regulators.\n\nWe needed a solution our cross-functional team could own, understand, and modify. A Python-based pipeline built on open-source tools aligned perfectly with BETA's collaborative culture, allowing one language to unify our entire data stack from edge to cloud.\n\n### Methods\nWe built an end-to-end pipeline entirely in Python, integrating established open-source tools with modern cloud data infrastructure. Our entire AWS infrastructure is provisioned and managed using CDK, ensuring our infrastructure is as maintainable and version controlled as our application code.\n\n**Data Ingestion:**\nCustom Python decoders parse CAN (Controller Area Network) log formats from test stands. Decode specifications vary frequently as we iterate on motor designs, so we built CLI tools that allow test conductors to push new decode files whenever needed (sometimes 25 times per week!), ensuring the pipeline adapts to evolving requirements without data engineering intervention.\n\n**Raw Storage:** \nWe leverage Apache Iceberg as our lakehouse format on AWS S3, using PySpark for writes. Iceberg provides schema evolution, time travel, and hidden partitioning, which is essential for managing growing datasets while maintaining data quality. This layer preserves tall-format, full-fidelity data indefinitely. Engineers can access any signal on demand, easily add or remove instrumentation as testing needs evolve, and the transparent storage architecture is easily explainable to the FAA for certification purposes.\n\n**Transformation:**\nDbt defines SQL transformations that aggregate raw time series data and event metadata into well-defined, consistent data mart schemas and custom views. Its testing framework ensures data quality, the open-source Python library sqlfluff handles SQL linting, and its documentation features maintain living documentation directly in the code.\n\n**Analytics Layer:**\nTransformed data lands in Amazon Redshift, optimized for fast queries and ready for analysis. This data mart serves multiple downstream uses: engineers perform ad-hoc analysis in Python; automated Python scripts generate derived insights that are written back to our data platform and stored alongside observed data; and a dedicated time tracking database stores aggregated test hours by component and operating condition, critical for FAA testing requirements.\n\nWe use Grafana, an open-source visualization tool that connects to any data source, seamlessly unifying our many databases into a single visualization layer. We've built a library of reusable Grafana dashboards that make stored data immediately accessible, serving analytics across the organization and transforming data into actionable insights for day-to-day decision making.\n\n**Orchestration:**\nApache Airflow DAGs coordinate the entire pipeline through a mix of scheduled and event-driven jobs, with custom operators written in Python for our specific workflow needs.\n\n### Results\nThe pipeline processes half a terabyte of test data monthly, supporting 50+ engineers across multiple test stands and development programs. Data access that previously required hours of manual data extraction now completes in seconds, allowing engineers to spend less time hunting for data and more time iterating on designs.\n\nThe best outcome: learnings from this project extend well beyond this single pipeline. By building reusable BETA-specific CDK constructs and establishing common architectural patterns, we've created a platform that accelerates development across all our data sources, from manufacturing sensors to flight test telemetry. This unified approach reduces development time and makes our entire data ecosystem more consistent and maintainable.", "recording_license": "", "do_not_record": false, "persons": [{"code": "3LYZFF", "name": "Sarah Tabor", "avatar": "https://pretalx.com/media/avatars/3LYZFF_SYY3PP9.webp", "biography": "Sarah Tabor is a Data Scientist and Data Engineer at BETA Technologies, a Vermont based aerospace company designing and building the future of electric flight.\n\nSarah designs and builds cloud-native data pipelines from source to analysis using Python and open-source tools, all in service of a \"Data For All\" philosophy that democratizes access to data and insights for anyone at BETA, from engineers to executives.\n\nSarah holds BBAs in Economics, Finance and Business Analytics from the University of Iowa, and an MS in Complex Systems and Data Science from the University of Vermont.", "public_name": "Sarah Tabor", "guid": "39f1682f-92d2-59db-bbed-71bd231c4347", "url": "https://pretalx.com/scipy-2026/speaker/3LYZFF/"}], "links": [{"title": "BETA's website", "url": "https://beta.team/", "type": "related"}, {"title": "The Most Beautiful Ferry Flight in America (and It\u2019s Electric)", "url": "https://youtu.be/CLCCUl_r5TY?si=1NIVqPNIM45KfdcB", "type": "related"}, {"title": "Meet BETA Video", "url": "https://youtu.be/NuKBeiHmGJA?si=I_PTE0orA1hA3brU", "type": "related"}, {"title": "We Fly What We Build", "url": "https://youtu.be/tuKxwv0LrNM?si=rHY_XTYPOmyqKTSu", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/SUPRRW/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/SUPRRW/", "attachments": []}, {"guid": "e8f5b48f-ea7b-5140-b068-bab3586952dc", "code": "EJSCLM", "id": 91837, "logo": null, "date": "2026-07-15T16:05:00-05:00", "start": "16:05", "end": "2026-07-15T16:35:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-91837-gpu-accelerated-awkward-arrays-with-cuda-python", "url": "https://pretalx.com/scipy-2026/talk/EJSCLM/", "title": "GPU-Accelerated Awkward Arrays with CUDA Python", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "Awkward Array is a Python library widely used in high-energy physics for representing and manipulating nested, variable-length data. As analysis workloads increasingly rely on GPU acceleration, there is a need for solutions that deliver high performance while remaining accessible to Python developers. In this joint talk between the Awkward and the NVIDIA teams, we present recent developments in GPU execution for Awkward Array using the new native Python CUDA support for CCCL called `cuda.compute`. This novel Python interface for CCCL that enables users to achieve state-of-the-art GPU performance without dropping down to C++ when building new GPU algorithms.\n\nOur approach fuses sequences of Awkward operations into a minimal set of CUDA kernels, reducing kernel launch overhead and improving memory efficiency. Lazy execution allows entire expression graphs to be optimized before kernel generation, which benefits workflows involving jagged arrays, combinatorial operations, and reductions. In addition, the design enables user-defined Python code to be incorporated into GPU execution paths with minimal boilerplate, lowering the barrier for extending Awkward with custom GPU-accelerated logic.\n\n`cuda.compute`  is a new component in the CUDA Python ecosystem that provides native access to optimized algorithms, such as transforms, reductions, and scans. It also provides a collection of iterators that defer execution of operations and enable fusing multiple operations. We demonstrate performance improvements using this approach over an eager GPU execution strategy for representative analysis patterns and show how it integrates into the existing Python workflows. These developments provide a practical, user-friendly path toward high-performance GPU-accelerated data analysis in Python.\n\nWe thank NVIDIA for support and collaboration in developing the CUDA kernels and providing guidance on GPU optimization strategies. Their contributions are gratefully acknowledged.", "description": "**Outline**\n**_I. Background & Motivation: The Reality of \"Messy\" Awkward Data_**\n\n- Beyond Rectangular Tensors: Real-world data is rarely a simple 2D matrix. Whether it's nested JSON, variable-length genomic sequences, or particle tracks in physics, \"jagged\" data is everywhere.\n\n- The Hardware Bottleneck: Standard GPU libraries often require padding jagged data to fixed lengths, which wastes memory and compute cycles.\n\n- The \"Memory Wall\": Even when using existing GPU kernels, executing them one by one (eagerly) forces the GPU to constantly move data between fast registers and slow global memory.\n\n**_II. Methods: Introducing `cuda.compute` and CCCL_**\n\n- A New Python Interface: We introduce cuda.compute, a novel component in the CUDA Python ecosystem that provides native access to CCCL primitives \u2014 transforms, reductions, and scans.\n\n- The Integration: How the Awkward Array team collaborated with NVIDIA to bridge the gap between high-level Python abstractions and low-level CUDA performance.\n\n- From \"Fixed\" to \"Fused\": Moving from a library of pre-written, static kernels to a system where users write their own kernel logic in Python, fused into an efficient CUDA kernel tailored to the task at hand.\n\n**_III. Deep Dive: Kernel Fusion and Lazy Execution_**\n\n- The Expression Graph: How Awkward Array now captures a user's intent \u2014 e.g., \"filter these events, then calculate a mean\" \u2014 as a graph rather than executing each step immediately.\n\n- Dynamic Compilation: Using `cuda.compute` to fuse this graph into a minimal set of CUDA kernels.\n\n- Efficiency Gains: Fusing operations reduces kernel launch overhead and keeps data on-chip (in L1 cache and registers) as long as possible.\n\n**_IV. Results: Performance in the Real World_**\n\n- Benchmarking Complexity: We demonstrate performance gains on representative analysis patterns \u2014 such as combinatorial matching and nested reductions \u2014 common in both high-energy physics (HEP) and large-scale data engineering.\n\n- Performance vs. Effort: This approach achieves C++-level performance while requiring zero C++ code from the end user.\n\n- Workflow Integration: How this fits into existing ecosystems like the broader SciPy stack.\n\n**_V. Conclusion & Outlook_**\n\n- Impact: This collaboration makes high-performance GPU analysis accessible to any scientist working with complex data structures.\n\n- Next Steps: Current availability in the Awkward Array ecosystem, with future plans to expand the `cuda.compute` primitive set.\n\n- Acknowledgements: We gratefully acknowledge NVIDIA's support in developing these kernels and optimization strategies.", "recording_license": "", "do_not_record": false, "persons": [{"code": "ECE7HR", "name": "Ianna Osborne", "avatar": "https://pretalx.com/media/avatars/ECE7HR_eCgKjVY.webp", "biography": "Research Software Engineer, Princeton University\nIanna Osborne is an open-science advocate and research software engineer specializing in particle physics and high-performance computing. She builds scalable, open-source tools for scientific discovery, maintains the Awkward Array project, and leads international efforts that foster collaboration and sustainable research software.", "public_name": "Ianna Osborne", "guid": "209ef22b-e4ff-5c58-a27a-4e032292126e", "url": "https://pretalx.com/scipy-2026/speaker/ECE7HR/"}, {"code": "3ZSL8Z", "name": "Ashwin Srinath", "avatar": "https://pretalx.com/media/avatars/GBFK83_xreEI3X.webp", "biography": "Ashwin Srinath is a Senior Software Engineer at NVIDIA, where he works on making GPU programming from Python delightful.", "public_name": "Ashwin Srinath", "guid": "080ee3e4-50bd-59c8-8127-5218c6d1fdd1", "url": "https://pretalx.com/scipy-2026/speaker/3ZSL8Z/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/EJSCLM/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/EJSCLM/", "attachments": []}], "Thomas Swain Room": [{"guid": "ca4248d1-a18d-55d0-a2c1-45034c05d5bf", "code": "7PSPJP", "id": 92347, "logo": null, "date": "2026-07-15T10:45:00-05:00", "start": "10:45", "end": "2026-07-15T11:15:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92347-fairer-data-the-case-for-data-advertising-in-the-age-of-agentic-ai", "url": "https://pretalx.com/scipy-2026/talk/7PSPJP/", "title": "FAIRer Data: The case for Data Advertising in the age of Agentic AI", "subtitle": "", "track": "Environmental, Earth, and Climate Sciences", "type": "Talk", "language": "en", "abstract": "For scientists wanting to work with and analyse earth science data, the standard remains delivering tooling via python packages, and data via HPC or the cloud. For data siloed on an HPC system, this presents a barrier to findability and accessibility. However, with agentic AI now widely available, the cost of learning a new tech stack or toolchain to deliver this data has plummeted. \n\nIn this talk, I'll outline how we utilised agentic AI to translate an intake catalog into an interactive, single page web application, maximising data discoverability whilst leaning on our existing data infrastructure and established Python API's to constrain the scope, keep the wrapper thin, and the code from becoming spaghettified, unmaintainable AI slop.", "description": "FAIR (Findable, Accessible, Interoperable, Reusable) data is widely agreed on as the benchmark for data management and stewardship. Unfortunately, findability is often an afterthought - and datasets that can't be found are difficult to access!\nIn the earth sciences, we commonly assume that if a dataset can be found by someone who already knows what they are looking for or where to look for it, then it is findable. For small and tight knit research communities, or cloud distributed datasets, this might be the case.\nHowever, within the earth sciences, many datasets remained siloed on HPC systems, where it is often assumed that tribal knowledge that can be obtained from a supervisor, colleague, or collaborator is sufficient to guide new users through these systems. \nWorse yet, users are often expected to obtain login access to an HPC simply to discover which datasets are available, before they can even determine whether the data are relevant to their needs.\nIn practice, this assumption means that datasets are only available to an in group of users who are already familiar with them.\n\nWhy is this so often the case? The defacto tool for data analysis in the earth sciences is Python, but the best way to advertise and distribute datasets is through the web. If we want to distribute the data ourselves, without getting experienced web developers involved, this leaves us with a few options: static site generation through tools like readthedocs, writing a python web server, or going all in and learning enough JavaScript to create an interactive data exploration tool.\n\nThe key issue? The better the interface, the more time and effort you'd need to sink into learning a new tech stack, toolchain, and way of thinking. The result of this - lots of clunky interfaces to explore and obtain data.\n\nWhilst this is still true, with AI agents now widely available, the cost of learning a new tech stack or creating new data delivery tools has plummeted. For an experienced developer with a hoard of well structured data, 'vibe-coding' a wrapper to advertise and distribute that data is now a serious option.\n\nIn this talk, I'll walk through how we created a tool for advertising Australia's trove of earth science data, making it easy to find and discover for anyone with a browser and an internet connection - not just those who already had the right HPC login. Expect to learn:\n\n- Why well structured data, metadata, and documentation are more important than ever - not less - in this new data landscape.\n- How we used intake, duckdb-wasm, polars and Vue to create a tool that blends cloud and HPC data delivery.\n- Why the proliferation of social media and gamification of content has made data advertising more important than ever.\n- How keeping wrappers thin and focusing on the interactive experience lets the data do the talking.\n- How we went about testing, gathering feedback, and iterating on an interactive tool in an area where users expect to be provided with static content or a Python API.\n\nIntended Audience: Earth Scientists, people interested in data sharing, people looking to use emerging tools to make their work more impactful", "recording_license": "", "do_not_record": false, "persons": [{"code": "7M7MXZ", "name": "Charles Turner", "avatar": null, "biography": "Charles is a Research Software Engineer at ACCESS-NRI, where he works in the Model Evaluation and Diagnostics team, helping make it easier to access and analyse climate data. He has a PhD in Oceanography, where he first discovered his love of wrangling and disseminating data.\n\nWhen not in front of a computer, he enjoys routinely injuring himself in a variety of sports.", "public_name": "Charles Turner", "guid": "d488d69a-4579-5b28-9707-35a6968ccec1", "url": "https://pretalx.com/scipy-2026/speaker/7M7MXZ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/7PSPJP/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/7PSPJP/", "attachments": []}, {"guid": "462f7743-d8a4-57a6-bc59-178c3a0cdb09", "code": "ZTAB8K", "id": 92367, "logo": null, "date": "2026-07-15T11:25:00-05:00", "start": "11:25", "end": "2026-07-15T11:55:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92367-computational-biodiversity-accounting-for-agricultural-systems-with-python", "url": "https://pretalx.com/scipy-2026/talk/ZTAB8K/", "title": "Computational Biodiversity Accounting for Agricultural Systems with Python", "subtitle": "", "track": "Environmental, Earth, and Climate Sciences", "type": "Talk", "language": "en", "abstract": "The Ecosystem Services Market Consortium (ESMC) is expanding its agricultural sustainability programs to include biodiversity outcomes across the United States. To support this effort, we developed a Python-based Biodiversity Metric Module that estimates biodiversity gains associated with agricultural best management practices. The module integrates national land cover data, species occurrence records, protected lands datasets, and soil microbial biomass information within a unified geospatial workflow to generate standardized biodiversity unit estimates. This presentation outlines the ecological framework, computational architecture, and lessons learned while scaling biodiversity assessment across thousands of spatially explicit agricultural fields.", "description": "The Ecosystem Services Market Consortium (ESMC) works with agricultural producers across the United States to incentivize sustainable land management. As biodiversity increasingly becomes part of climate and sustainability reporting frameworks, ESMC identified the need for a consistent, scalable method to quantify biodiversity gains associated with agricultural best management practices. Unlike carbon accounting, biodiversity does not reduce to a single stock or flux. It reflects habitat condition, ecological function, spatial connectivity, and recovery through time. Estimating biodiversity change across working agricultural landscapes requires both a sound ecological foundation and robust computational design.\n\nTo meet this need, we developed the Biodiversity Metric Module, a Python-based tool that supports biodiversity quantification within ESMC\u2019s Monitoring, Reporting, and Verification platform. The module evaluates agricultural fields using a structured ecological framework and it calculates biodiversity units by integrating five interacting components: habitat quality, functional diversity, conservation context, habitat size, and time-dependent ecological response.\n\nWe derived habitat quality from national land cover datasets (CDL, NLCD) and translated land cover classes into ecological condition scores using structured parameter tables aligned with program objectives. We represented functional diversity by analyzing species occurrence records (GBIF) to characterize ecological guild presence for plants, insects, and birds, and we supplemented those data with publicly available soil microbial biomass datasets. The model evaluates landscape context by calculating proximity to protected areas to reflect conservation priority and connectivity. It accounts for habitat size by incorporating the spatial footprint of each management practice. Time-dependent response functions model ecological recovery following practice implementation. The system computes biodiversity units for baseline and practice-change conditions and quantifies net biodiversity gain as their difference.\n\nWe operationalized this framework by integrating publicly available, geospatial datasets at multiple spatial resolutions, including raster and vector. Scientific Python tools support spatial processing, numerical computation, and reproducible data management throughout the workflow (e.g. geopandas, rasterio, rasterstats). \n\nWe deployed the Biodiversity Metric Module as a Flask-based application within the broader Monitoring, Reporting, and Verification platform architecture already in place. The application accepts spatial field boundaries and management attributes as inputs, executes geospatial and numerical workflows, and returns standardized geospatial outputs. This design enables consistent evaluation across fields while maintaining clear separation between ecological logic and user interface components.\n\nThroughout development, I focused on raster processing, habitat quality scoring, and integration of species occurrence data within spatial buffers. Aligning national land cover rasters at differing resolutions required deliberate aggregation and consistency checks across baseline and practice-change scenarios. Developing both the habitat quality and species function scores reinforced the importance of explicitly documenting ecological assumptions and spatial bias. These challenges highlighted the importance of modular design and transparent assumptions when building biodiversity metrics at national scale. \n\nThe development of this tool highlights broader challenges in applied biodiversity modeling, including data limitations, spatial bias in species occurrence records, and temporal mismatches between ecological processes and national-scale datasets. It also demonstrates how scientific Python enables integration of diverse environmental data into scalable, reproducible, geospatial workflows. As biodiversity accounting continues to evolve, computational frameworks like this will play an increasingly important role in connecting ecological science with large-scale environmental decision-making.", "recording_license": "", "do_not_record": false, "persons": [{"code": "T33PGE", "name": "Hannah Ferriby", "avatar": "https://pretalx.com/media/avatars/7G37UM_WBCZaFX.webp", "biography": "Hannah Ferriby is an Environmental Data Scientist at Tetra Tech based out of the Lansing, MI area. She received her BSE in Environmental Engineering from the University of Michigan and her MS in Biosystems Engineering from Michigan State University. She specializes in geospatial and remote sensing analysis with a focus on water quality applications. Her work ranges from cyanobacteria harmful algal bloom forecasting to numeric nutrient water quality criteria development to biodiversity metric creation. Hannah is newer to coding in Python but has extensive background in R and JavaScript (Google Earth Engine).", "public_name": "Hannah Ferriby", "guid": "f50873e0-17af-51b6-a903-591115185c53", "url": "https://pretalx.com/scipy-2026/speaker/T33PGE/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/ZTAB8K/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/ZTAB8K/", "attachments": []}, {"guid": "5739678e-1506-56c5-88aa-5183bc1ea4f7", "code": "E8QGYT", "id": 90928, "logo": "https://pretalx.com/media/scipy-2026/submissions/E8QGYT/image_4szFT2I.webp", "date": "2026-07-15T13:15:00-05:00", "start": "13:15", "end": "2026-07-15T13:45:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-90928-from-lidar-to-action-detecting-upland-gullies-to-combat-erosion-and-forest-fires", "url": "https://pretalx.com/scipy-2026/talk/E8QGYT/", "title": "From LiDAR to action: detecting upland gullies to combat erosion and forest fires", "subtitle": "", "track": "Environmental, Earth, and Climate Sciences", "type": "Talk", "language": "en", "abstract": "UChicago's Data Science Institute (DSI) partners with 11th Hour Project to turn data insights into action. In this talk, I'll focus on our collaboration with Occidental Arts & Ecology Center (OAEC)'s Fuels to Flows program, which stabilizes upland waterways by adding brushwood that would otherwise fuel forest fires. Gullies are hidden by trees, so we used publicly available LiDAR to cleanly identify gullies by shape with a lightweight convolutional model. I'll show how Numba made it possible to convolve hundreds of gigabytes of images with unusually large kernels and how we delivered these map layers via static hosting using PMTiles, even with interactive features like computing elevation profiles along hand-drawn lines.", "description": "**Who this is for:** GIS data scientists who work with rasters, problems of scale, and shipping interactive results to non-technical stakeholders with minimum infrastructure. The general SciPy audience may also be interested in this example of a \"right-sized\" ML approach (the \"lightweight convolutional model\"), as well as ways data science can contribute to nonprofits and the public good.\n\n**Motivation/context:** The UChicago DSI 11th Hour group (https://datascience.uchicago.edu/outreach/11th-hour-project/) partners with 11th Hour Project grantees, spanning energy, food & agriculture, human rights, and marine ecology, to build software and data products for social and environmental impact. This talk focuses on one environmental case study within the broader pattern of supporting mission-driven organizations with tools that reduce manual work and scaling beyond ad-hoc analytics.\n\n**Problem:** OAEC has implemented the Fuels to Flows program (https://oaec.org/our-work/wildlands/fuels-to-flows/) on their own site and Monte Rio Redwoods Regional Park (both in Sonoma County, CA), but expanding the program requires identifying new sites, working with land-owners to secure the right permits, and hiring contractors. Our work addresses the first step by making gullies, ladder fuels, and erosion patterns visible on a county-wide interactive map.\n\n**Data:** Sonoma County publicly provides LiDAR-derived products: high-resolution DEMs scanned in 2013 and 2022 (1 m and 0.5 m grids), as well as proxies of ladder fuels that allow ground fires to climb to the forest canopy.\n\n**Method:** Standard gully-finding heuristics produce rasters to search by eye; we extended this technique to (1) reduce noise by convolving images with trough-shaped, rather than point-like, kernels, (2) approximate a CNN with engineered, rather than learned, features due to the paucity of hand-labeled data, and (3) build a vector-based \"road network\" of gullies, rather than an image. This technique has a spin-off used in the DSI's Clinic course: a UChicago student adopted it to vectorize blood vessels in MRI images to predict breast cancer treatment response.\n\n**Delivery:** We provide GIS-ready layers, but the intended users of this work are not GIS experts and the files are unwieldy (400 GB total). Therefore, we built a specialized map app as a website that loads data on demand as the user zooms into it. We also need to minimize our maintenance burden, since this is one of many projects, so we formatted the data as PMTiles, which are flat files that can be served with static web hosting (CloudFlare, in our case), with no application-specific server logic.\n\n**Map app:** https://oaec-found-gully.vercel.app/\n**GitHub:** https://github.com/dsi-clinic/oaec-found-gully\n(currently private; I'll see if I can make it public before submitting)\n\n**What attendees will learn:**\n1. A practical middle ground between simple filters and deep learning when labels are scarce.\n2. How to insert a custom optimization with Numba when standard functions (`convolve2d` in various libraries) restrict performance due to unusual conditions (unusually large kernels in our case).\n3. How to deliver large maps in tiles without requiring a custom server.", "recording_license": "", "do_not_record": false, "persons": [{"code": "BGX8FE", "name": "Jim Pivarski", "avatar": "https://pretalx.com/media/avatars/BGX8FE_lkJQBU6.webp", "biography": "Jim was trained as a particle physicist with a Ph.D. from Cornell and helped commission the CMS experiment at the Large Hadron Collider (LHC). He has worked as a data scientist (at Open Data Group) and a software developer (at Princeton), and was the founder of the Awkward Array project. Jim is now at the University of Chicago's Data Science Institute, where he solves data analysis problems for nonprofit organizations.", "public_name": "Jim Pivarski", "guid": "4831e653-73cc-5b87-8fd9-3ad6177051a3", "url": "https://pretalx.com/scipy-2026/speaker/BGX8FE/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/E8QGYT/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/E8QGYT/", "attachments": []}, {"guid": "0f27f9cf-c4d9-555e-9c30-d41f9c2de1bf", "code": "97QQ8D", "id": 92397, "logo": null, "date": "2026-07-15T13:55:00-05:00", "start": "13:55", "end": "2026-07-15T14:25:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92397-accelerating-geospatial-analysis-with-gpus", "url": "https://pretalx.com/scipy-2026/talk/97QQ8D/", "title": "Accelerating Geospatial Analysis with GPUs", "subtitle": "", "track": "Environmental, Earth, and Climate Sciences", "type": "Talk", "language": "en", "abstract": "Geospatial analysis relies on raster data \u2014 n-dimensional arrays where each cell holds a spatial measurement. The scale of modern remote sensing data makes CPU-based workflows impractical, but raster operations are naturally parallelizable and well suited for GPU acceleration. This talk walks through a GPU-accelerated end-to-end workflow to classify satellite imagery into land cover types, covering data access via [STAC](https://stacspec.org/en), preprocessing (cloud masking, compositing, spectral index computation), training a Random Forest classifier on millions of pixels, and running inference on unseen tiles. The pipeline uses familiar APIs from Xarray, Dask, pandas, and scikit-learn, accelerated with RAPIDS. No prior geospatial or GPU experience is required.", "description": "## Motivation\nMonitoring land use and land cover (LULC) change is essential for understanding deforestation, urban growth, and the effects of climate change. Satellite missions like ESA's Sentinel-2 provide openly available multispectral imagery at up to 10-meter resolution with global coverage and a revisit frequency of roughly 5 days. However, a single tile contains millions of pixels across multiple spectral bands, and producing accurate LULC maps requires preprocessing raw imagery (cloud removal, temporal compositing, index computation), training a classifier, and running inference across large regions. On a CPU, each of these stages can take tens of minutes per scene, making regional-scale analysis slow and difficult to iterate on.\n\nBecause each pixel in a satellite image can be processed independently, these raster operations are naturally parallelizable and well suited for GPU computation. This talk shows how existing Python tools in the geospatial ecosystem can be combined with GPU-accelerated libraries to make geospatial workflows significantly faster, often with minimal code changes. \n\n## Methodology\nWe present an end-to-end LULC classification pipeline built on publicly available datasets. Sentinel-2 Level-2A imagery provides the input features and ESA WorldCover provides per-pixel land cover labels, both accessed through the SpatioTemporal Asset Catalogs (STAC) specification. The pipeline covers several stages common to remote sensing workflows such as querying and loading cloud-hosted imagery into Xarray using Dask for chunked computation, masking clouds using the Sentinel-2 Scene Classification Layer, mosaicing overlapping tiles, computing an annual median composite, and deriving spectral indices such as NDVI and NDWI as additional features. The resulting data cube and matched labels are then used to train a Random Forest classifier on millions of labelled pixels, and the trained model is applied to previously unseen satellite tiles to generate LULC maps.\n\nWe compare wall times for each stage (preprocessing, training, inference) against a CPU baseline (using scikit-learn) to quantify the practical benefits and ease of using GPUs when working with data in the geospatial domain. \n\n## Results\nAcross the full pipeline, GPU-accelerated stages run 3x to 5x faster than their CPU equivalents, with the largest gains in model training and full-scene inference. The trained model performs especially well at distinguishing major land cover classes like built area and water bodies. We also present visual comparisons of model predictions against reference maps over held-out regions for qualitative analysis. \n\n## Conclusion\nAttendees will come away with a practical understanding of how to efficiently leverage GPUs when working with geospatial data in Python. We will also discuss the design choices made, challenges we encountered, and potential improvements to provide a complete understanding to attendees which they can leverage in their own work. The full notebook for this exercise with detailed explanations for attendees to follow is available at https://docs.rapids.ai/deployment/stable/examples/lulc-classification-gpu/notebook/\n\n## Talk Outline (25 min + 5 min Q&A)\n**Introduction (5 min):** What LULC classification is and why it matters for environmental monitoring, how satellite imagery is structured (tiles, bands, resolution, coordinate reference systems), and the datasets used in this talk (Sentinel-2 and ESA WorldCover).\n**Data access and preprocessing (10 min):** Querying cloud-hosted imagery via STAC, loading into Xarray/Zarr, cloud masking, temporal compositing, and computing spectral indices (NDVI, NDWI).\n**Model training and inference (5 min):** Building a feature cube from preprocessed imagery, training a Random Forest classifier on millions of labelled pixels, and generating LULC maps as inference over unseen tiles using the trained model.\n**Performance comparison and potential improvements(5 min):** CPU vs GPU wall-time comparisons across preprocessing, training and inference. Discussion of problems like class imbalance approaches on how to solve these issues. A brief discussion about best practices for chunking and memory management. \n**Q&A (5 min)**", "recording_license": "", "do_not_record": false, "persons": [{"code": "WHWU9R", "name": "Jaya Venkatesh", "avatar": "https://pretalx.com/media/avatars/RTDCAL_IR6nvLd.webp", "biography": "Jaya Venkatesh is a Software Engineer at NVIDIA, working on the RAPIDS ecosystem to streamline the deployment of GPU-accelerated data science workflows across cloud and distributed systems. Previously, he was a Machine Learning Engineer at Pixxel Space, where he developed large-scale, real-time data processing and inference pipelines for Earth observation using GPU-accelerated Python libraries.", "public_name": "Jaya Venkatesh", "guid": "33becb3b-74ef-592e-af56-01c14d57b9be", "url": "https://pretalx.com/scipy-2026/speaker/WHWU9R/"}, {"code": "AURRUC", "name": "Jacob Tomlinson", "avatar": null, "biography": null, "public_name": "Jacob Tomlinson", "guid": "7d5794a8-e43e-58a6-9a19-8751d101fde1", "url": "https://pretalx.com/scipy-2026/speaker/AURRUC/"}, {"code": "VXQXZP", "name": "Naty Clementi", "avatar": null, "biography": "Naty Clementi is a senior software engineer at [NVIDIA](https://www.nvidia.com/). She is a former academic with a Masters in Physics and PhD in Mechanical and Aerospace Engineering to her name. Her work involves contributing to [RAPIDS](https://rapids.ai/), and in the past she has also contributed and maintained other open source projects such as [Ibis](https://ibis-project.org/) and [Dask](https://www.dask.org/). She is an active member of [PyLadies](https://pyladies.com/) and an active volunteer and organizer of [Women and Gender Expansive Coders DC meetups](https://www.meetup.com/women-and-gender-expansive-coders-dc-wgxc-dc/).", "public_name": "Naty Clementi", "guid": "039b72d7-581e-5888-8cb9-a36d8a2ca95a", "url": "https://pretalx.com/scipy-2026/speaker/VXQXZP/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/97QQ8D/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/97QQ8D/", "attachments": []}, {"guid": "dc545e88-ecc5-5e23-88be-bb3fbf648dc8", "code": "N9YDEL", "id": 92513, "logo": "https://pretalx.com/media/scipy-2026/submissions/N9YDEL/image_dx4cUo0.webp", "date": "2026-07-15T14:35:00-05:00", "start": "14:35", "end": "2026-07-15T15:05:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92513-nepho-a-workflow-for-using-mllms-for-atmospheric-data-plot-exploration", "url": "https://pretalx.com/scipy-2026/talk/N9YDEL/", "title": "Nepho: A workflow for using mLLMs for atmospheric data plot exploration", "subtitle": "", "track": "Environmental, Earth, and Climate Sciences", "type": "Talk", "language": "en", "abstract": "The advent of multimodal large language models (mLLMs) provides new opportunities for automated data exploration tasks on multi-petabyte atmospheric data sets. In this presentation, we present Nepho, a Python package for parallel mLLM prompting on collections of atmospheric quicklook data plots. We then evaluate the accuracy of several mLLMs in answering questions about Atmospheric Radiation Measurement (ARM) atmospheric datasets. We demonstrate that the GPT 4/5 and llama3-vision models were the most accurate models for quicklook plot exploration and recommend prompt engineering and retrieval-augmented generation for such data exploration workflows.", "description": "Atmospheric datasets, such as the U.S. Department of Energy Atmospheric Radiation Measurement Facility\u2019s archive span several petabytes and decades. This makes exploring such datasets difficult for users that are interested in specific weather phenomena. However, multimodal LLMs such as GPT 5.0 now support basic analyses of atmospheric data plots. Given that quicklooks are available on ARM\u2019s dqplotbrowser website for most of ARM\u2019s instrument and value added product data, mLLMs present a potential new opportunity for automated data exploration using agents. \n\nIn this presentation, we present a feasibility study for using mLLMs for data exploration. In order to perform this study, we developed Nepho, a Python package that supports parallel mLLM inference of prompts on sets of quicklook plots. Nepho supports a wide variety of mLLMs using OpenAI, RESTful API, and ollama endpoints through a backend abstraction. Nepho encodes image timeseries into an embedding along with the prompt and performs inference of specific prompt-data plot pairs automatically for the user, making automated mLLM workflows easier on image collections. Nepho supports parallel inference for faster processing and therefore can scale to multiple processors. \n\nNepho was used for a feasibility study for using mLLMs to explore atmospheric datasets through quicklook plots. As a part of this effort, atmospheric scientists developed a testing dataset of 132 prompt-data plot-answer triplets from a wide array of atmospheric datasets. An example of such a triplet is shown in Figure 1. In this example, we use an mLLM to explore spikes in eddy correlation flux data from the ARM Southern Great Plains site. We provide the multiple choice question about the plot and then assess accuracy by comparing against human-generated answers about the plot. We evaluated 12 mLLMs in total. GPT-4.1 and GPT-5 provided the best accuracy, both around 68%. The best open source model performance we evaluated was llama3.2-vision:90b with 57.58% accuracy. This shows that, without any effort to provide domain-specific information to the mLLMs, that mLLMs have fair accuracy on answering multiple-choice questions for this testing dataset. Since we did not include any domain-specific information in our prompt, we recommend methods to increase the accuracy for specific datastreams by including domain-specific information through retrieval-augmented generation to improve accuracy. \n\nNepho has enabled other community efforts exploring the feasibility of mLLM-assisted data exploration. For example, the ARM Facility plans further feasibility studies on weather radar scene classification and exploration of data quality issues in atmospheric plots for the ARM Data Quality Office, incorporating these recommendations. LLM-Assisted Radar Scenes (LARS), a weather radar classification package based on Nepho, is already under development.", "recording_license": "", "do_not_record": false, "persons": [{"code": "ZPTUHY", "name": "Bobby Jackson", "avatar": null, "biography": "Bobby Jackson is an atmospheric scientist at Argonne National Laboratory. His research interests include radar meteorology, using AI and edge computing to improve atmospheric observations, and open source software development for the atmospheric sciences. He is a lead developer on PySP2 and PyDDA, two open source Python packages for aerosol and radar wind retrievals. In addition, he is a contributing developer to numerous packages in the Pangeo and Open Source Radar communities, including PyART.", "public_name": "Bobby Jackson", "guid": "650b0c51-563a-54f8-886b-6bd53f9a25a2", "url": "https://pretalx.com/scipy-2026/speaker/ZPTUHY/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/N9YDEL/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/N9YDEL/", "attachments": []}, {"guid": "60a9221b-4299-5c23-aa4d-6e9a99d72ce7", "code": "XZGSV7", "id": 92113, "logo": null, "date": "2026-07-15T15:25:00-05:00", "start": "15:25", "end": "2026-07-15T15:55:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92113-adapt-prototyping-a-real-time-reproducible-data-analysis-framework-for-adaptive-radar-scanning", "url": "https://pretalx.com/scipy-2026/talk/XZGSV7/", "title": "Adapt: Prototyping a Real-Time, Reproducible Data Analysis Framework for Adaptive Radar Scanning", "subtitle": "", "track": "Environmental, Earth, and Climate Sciences", "type": "Talk", "language": "en", "abstract": "_Adapt v0.1_ is a real-time, reproducible data-analysis framework developed to support adaptive radar scanning within the U.S. Department of Energy Atmospheric Radiation Measurement (ARM) facility. It implements a declarative, store-driven architecture that separates acquisition, processing, and visualization into independent, thread-safe components. A continuous ingestion worker registers incoming radar data, while processing workers poll a central DataStore for newly available items and execute configured analysis chains. Visualization and external systems interact only with completed outputs, preventing interference with internal logic. The framework is built on the Scientific Python ecosystem, including Py-ART, Xarray, Scikit-learn, OpenCV, and SciPy, and is designed for maintainability and extensibility through well-defined input\u2013output protocols.\n\nAdaptive radar scanning enables real-time response to evolving convective storms, overcoming limitations of fixed, omnidirectional volume scans that often miss rapid microphysical transitions. Because radar beam physics constrains full-volume update rates, dynamically focusing on sectors of interest can significantly improve temporal resolution. Achieving this requires low-latency analysis, forecasting, and decision support integrated directly into operational workflows. While legacy systems such as TITAN demonstrated real-time storm tracking decades ago, most modern Python-based radar and tracking tools were designed for offline analysis. Campaign-driven ARM operations require continuous ingestion, event-driven execution, streaming outputs, flexible configuration, and robust integration with operational infrastructure. Adapt addresses these needs through a lightweight, modular design that cleanly separates orchestration, scientific logic, and downstream consumers.\n\nThe architecture consists of three loosely coupled layers. The scientific layer contains deterministic modules for detection, analysis, projection, and tracking that operate on structured inputs and produce explicit outputs. The orchestration layer manages item lifecycles, scheduling, and metadata state transitions including creation, queuing, processing, completion, or failure, enabling recoverability and preventing race conditions. The data access layer provides a client abstraction over the repository so downstream systems query structured metadata rather than raw files. Configuration files and CLI arguments define algorithm selection, runtime parameters, radar sources, and product definitions, supporting campaign-specific objectives.\n\nTo prevent silent numerical corruption, Adapt enforces algorithm contracts that validate outputs immediately after execution. Segmentation products are checked for dimensional consistency, contiguous labeling, and mask integrity; projection products are verified for spatial alignment, finite motion vectors, and forecast horizon consistency; analytical outputs undergo schema and metadata validation. Violations halt processing for that item and record diagnostic state in the catalog, ensuring fail-fast behavior and reproducible debugging.\n\nThe processing pipeline operates as an external script transitioning toward modular CLI tools. A downloader thread monitors configured sources and constructs items containing scan metadata, input paths, and expected outputs. Processor threads consume queued items, resolve dependencies through the catalog, execute scientific modules, validate outputs, write results atomically, and update state. Threads communicate exclusively through queues without shared mutable state, and algorithm modules remain stateless. The orchestrator supervises queue depth and dependency conditions without directly controlling thread execution.\n\nMultidimensional grids are stored in NetCDF, while tabular analysis and tracking outputs use Parquet for efficient columnar access. Partitioned directory structures enable scalable time-range queries. A metadata catalog records radar inventories, processing runs, product definitions, and lineage relationships. A data client supports batch queries and streaming mode, polling for newly completed products so dashboards can visualize segmentation masks, projected motion, and lifecycle metrics without disrupting active processing. Each execution is registered as a uniquely identified run storing configuration, radar selection, and product relationships, enabling deterministic replay of historical datasets using the same logic as real-time operation.\n\nXarray provides labeled multidimensional data structures that preserve spatial coordinates and metadata, preventing index misalignment common in raw array workflows. Pydantic enforces strict configuration schemas and validates runtime parameters before execution. Dense motion fields are estimated using OpenCV\u2019s Farneb\u00e4ck optical flow on consecutive reflectivity frames, and cell geometries are derived using SciPy spatial triangulation methods. Py-ART provides Level-II decoding, coordinate transforms, and radar-specific processing foundations.\n\nAdapt remains in an alpha stage. Key development priorities include stronger dataset versioning and provenance tracking within the repository layer, improved support for concurrent reads during active writes, exploration of structured streaming and event-driven orchestration models, and development of interactive dashboards for operational visualization. Future work will also address containerized and distributed deployment for cloud-native scalability and object-storage\u2013first architectures. The modular separation between orchestration, scientific computation, and data APIs allows independent evolution of components and invites community contributions in data management, streaming frameworks, visualization systems, distributed execution, and reproducibility practices.\n\nIn summary, Adapt provides a modular, real-time architecture for adaptive radar scanning that enforces deterministic state management, contract-based validation, and repository abstraction. By eliminating thread entanglement and clearly separating system layers, it supports both historical reprocessing and operational guidance for live adaptive radar campaigns.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "8DARNL", "name": "Bhupendra Raut", "avatar": "https://pretalx.com/media/avatars/R9CQNX_ft4yXtg.webp", "biography": "Bhupendra A. Raut is a Computational Environmental Scientist at Argonne National Laboratory, where his research focuses on the analysis of clouds and precipitation in large-scale remote sensing datasets and numerical model outputs. He has developed convection identification, tracking, and analysis algorithms and applied various clustering and machine learning methods to produce value-added products for multi-platform observational campaigns. Currently, he is co-developing an adaptive sensing framework for the U.S. Department of Energy\u2019s Atmospheric Radiation Measurement (ARM) user facility and leads the Chicago Urban Flux Network.\n\nHis interdisciplinary research leverages statistics, computer vision, machine learning, and edge computing. An active member of the scientific community, he contributes to several prominent open-source atmospheric tools, including Py-ART, TINT, and tobac. Dr. Raut holds a Ph.D. and an M.Sc. in Atmospheric and Space Sciences from the University of Pune, following a B.Sc. from J. B. College of Science.", "public_name": "Bhupendra Raut", "guid": "2ef57307-7997-562d-8753-c5cfc345140d", "url": "https://pretalx.com/scipy-2026/speaker/8DARNL/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/XZGSV7/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/XZGSV7/", "attachments": []}, {"guid": "aad6cac1-a032-5a2c-b6ea-be934d11de2f", "code": "ZZYN3X", "id": 92410, "logo": "https://pretalx.com/media/scipy-2026/submissions/ZZYN3X/image_UsgztaW.webp", "date": "2026-07-15T16:05:00-05:00", "start": "16:05", "end": "2026-07-15T16:35:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92410-navigating-the-storm-software-orchestration-and-pipelines-for-ai-driven-weather-forecasting", "url": "https://pretalx.com/scipy-2026/talk/ZZYN3X/", "title": "Navigating the Storm: Software Orchestration and Pipelines for AI-Driven Weather Forecasting", "subtitle": "", "track": "Environmental, Earth, and Climate Sciences", "type": "Talk", "language": "en", "abstract": "Artificial Intelligence (AI) is reshaping meteorological science across two distinct frontiers. On one end, foundation-scale generative models, large-scale distributed training, and massive ensembles push the limits of high-performance computing and big-data orchestration. On the other, a \"democratized edge\" is emerging, where lightweight, heterogeneous inference workflows broaden access for experimentation. This dual expansion introduces a new class of software challenges spanning distributed training, ensemble-scale orchestration, and efficient, flexible inference pipelines.\n\nThis talk will introduce Earth2Studio and PhysicsNeMo from NVIDIA, two software packages designed to enable and scale AI weather forecasting. By exploring their architectures, we will discuss the broader development journey of building AI-driven meteorological tools and share key lessons learned in managing the intersection of high-performance computing, data science and operational reliability.", "description": "This is a talk for software engineers, data scientists, and climate researchers navigating the transition from classical simulation to AI-driven meteorology. The following core topics will be presented:\n\n**Framework Spotlight: NVIDIA PhysicsNeMo and Earth2Studio**\nWe will introduce and compare two pivotal frameworks from NVIDIA's Earth-2 stack:\n\n- PhysicsNeMo: An open-source Python framework designed for developing AI-physics models at scale. We will discuss its architecture for high-throughput training specifically optimized for weather and climate datasets.\n- Earth2Studio: A modular inference and pipeline toolkit. We explore how Earth2Studio allows developers to chain together diverse data sources (ERA5, GFS, satellite) with pre-trained models to create production-ready AI workflows.\n\n**Architectural Paradigms in AI Weather**\nThis talk dissects the various model paradigms currently dominating the field and the unique software requirements of each:\n\n- Prognostic Forecast Models: Such as StormScope, FourCastNet or GraphCast, which require stateful time-integration loops that autoregress forward in time, generating forecasts.\n- Diagnostic Models: Used for high-resolution downscaling (e.g., CorrDiff) or predicting new products from a forecast system relevant to a particular use case.\n- Data Assimilation Models: The bridge between raw satellite/sensor observations and model states, representing an emerging class of AI models accelerating weather and climate data assimilation.\n\n**The Challenges of the AI-Weather Stack**\nMoving from a research notebook to an operational service introduces significant challenges, which this session will address including:\n\n- Data Gravity & Structures: We will discuss the challenges of managing multi-petabyte datasets like ERA5 and the nuances of data formats (Zarr, NetCDF) when moving between high-bandwidth training and low-latency inference.\n- Scalability During Training: Designing models must have scalability in mind, navigating both the requirements for data pipelines as well as underlying architectures. State-of-the-art skill and impact often involves high-resolution and/or ensemble training, necessitating advanced parallelism techniques.\n- Operational Deployment: Lessons learned in deploying these models into production for users.\n- API Standardization & Model Interoperability: We will also discuss the challenges and solutions surrounding offering a large and diverse class of AI models under the same package(s) and providing a unified API for users.", "recording_license": "", "do_not_record": false, "persons": [{"code": "HDBAD8", "name": "Nicholas Geneva", "avatar": null, "biography": "Nicholas Geneva is a Senior Software Engineer in HPC/AI at NVIDIA, specializing in the development of platforms that integrate deep learning with physics and climate science. With over a decade of experience in scientific software development, he currently focuses on developing NVIDIA\u2019s software for Earth-2 to enable AI driven weather / climate prediction for everyone. Nicholas has been a core developer of NVIDIA\u2019s PhysicsNeMo Python packages for the past four years.", "public_name": "Nicholas Geneva", "guid": "15a11ad2-8af9-58ac-97e5-61b02e9fd34f", "url": "https://pretalx.com/scipy-2026/speaker/HDBAD8/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/ZZYN3X/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/ZZYN3X/", "attachments": []}], "University Hall": [{"guid": "d625a023-66db-55d1-830d-3daf5d758462", "code": "TADDJP", "id": 92398, "logo": "https://pretalx.com/media/scipy-2026/submissions/TADDJP/image_Cas7m7A.webp", "date": "2026-07-15T10:45:00-05:00", "start": "10:45", "end": "2026-07-15T11:15:00-05:00", "duration": "00:30", "room": "University Hall", "slug": "scipy-2026-92398-a-lean-and-kind-ome-zarr-toolkit-for-bioimaging", "url": "https://pretalx.com/scipy-2026/talk/TADDJP/", "title": "A Lean and Kind OME-Zarr Toolkit for Bioimaging", "subtitle": "", "track": "Biological and Medical Sciences", "type": "Talk", "language": "en", "abstract": "Bioimaging generates massive datasets in fragmented, proprietary formats that are difficult to share and align with FAIR principles. ngff-zarr is a lightweight, open-source Python toolkit implementing the OME-Zarr specification -- the community-driven, cloud-native bioimaging standard. With minimal dependencies and a simple pipeline interface, ngff-zarr converts, validates, and generates multiscale representations of extremely large images out-of-core via Dask. Features include multiple downscaling methods, OME-Zarr Zip archives (.ozx), RFC-4 anatomical orientation, and High Content Screening support. This talk also covers ngff-zarr's Model Context Protocol (MCP) server, which enables AI agents to perform bioimaging tasks through natural language, and lessons learned from its deployment at EMBL.", "description": "**The problem.** Modern bioimaging instruments produce datasets that are large, multidimensional, and stored in vendor-specific proprietary formats. These monolithic files are not cloud-ready, are difficult to stream or share, and hinder reproducible, collaborative science. The community needs an open, chunked, cloud-native format backed by robust, accessible tooling.\n\n**OME-Zarr and the community.** OME-Zarr (OME-NGFF) addresses this need as a community-driven open standard built on Zarr's chunked, compressed, n-dimensional array storage. The specification and its ecosystem are described in Moore et al., \"[OME-NGFF: a next-generation file format for expanding bioimaging data-access strategies](https://doi.org/10.1038/s41592-021-01326-w),\" *Nature Methods*, 2021; Moore et al., \"[OME-Zarr: a cloud-optimized bioimaging file format with international community support](https://doi.org/10.1007/s00418-023-02209-1),\" *Histochemistry and Cell Biology*, 2023; and L\u00fcthi et al., \"[2024 OME-NGFF workflows hackathon](https://doi.org/10.37044/osf.io/5uhwz_v2),\" *BioHackrXiv*, 2025. ngff-zarr is developed within and for this community.\n\n**ngff-zarr features.** [ngff-zarr](https://github.com/thewtex/ngff-zarr) is a lean, minimal-dependency implementation that is lazy, parallel, and web-ready -- no local filesystem required. Its features include:\n\n- A *simple Python interface* following a four-step pipeline: array to NgffImage to Multiscales to OME-Zarr store, accepting any Python Array API Standard input (NumPy, Dask, CuPy, PyTorch).\n- *Out-of-core multiscale generation* via Dask for processing extremely large datasets that exceed available memory.\n- *Multiple downscaling methods*: SIMD-accelerated Gaussian filtering via ITK-Wasm (default), bin shrink, label-image mode, and scipy-based fallbacks.\n- *OME-Zarr Zip (.ozx)* single-file archives for easy sharing and archival (RFC-9).\n- *RFC-4 anatomical orientation* metadata for medical and neuroimaging interoperability.\n- High Content Screening (HCS) plate/well support, TIFF/OME-TIFF and Leica LIF conversion, Zarr v3 sharding, and a command-line interface for batch workflows.\n\n**Python usage.** A typical conversion requires just a few lines:\n\n```python\nimport ngff_zarr as nz\n\nimage = nz.to_ngff_image(array, dims=[\"z\", \"y\", \"x\"], scale={\"z\": 2.0, \"y\": 0.5, \"x\": 0.5})\nmultiscales = nz.to_multiscales(image, scale_factors=[2, 4], chunks=64)\nnz.to_ngff_zarr(\"output.ome.zarr\", multiscales)\n```\n\nCloud stores (S3, GCS, Azure) are supported via fsspec, and the CLI (`ngff-zarr -i input.nrrd -o output.ome.zarr`) handles common batch workflows with memory-aware scheduling.\n\n**MCP server and lessons learned.** The `ngff-zarr-mcp` package exposes conversion, validation, inspection, and optimization tools to AI coding agents via the [Model Context Protocol](https://modelcontextprotocol.io/) (MCP). Researchers interact in natural language -- asking an AI assistant to convert a file, examine OME-Zarr metadata, validate spec compliance, or generate a batch processing script -- and the MCP server handles execution. Lessons learned include the importance of structured tool parameters for reliable agent interaction, designing functions that map to researcher intent rather than low-level API calls, and how natural language interfaces lower the barrier for scientists to adopt cloud-native formats and reproducible workflows.\n\n**Audience and takeaways.** Attendees will learn how to convert and manage bioimaging data with ngff-zarr's Python API and CLI, understand the OME-Zarr ecosystem, and see how MCP servers can bring AI-assisted automation to scientific data workflows.\n\nSource code: [github.com/fideus-labs/ngff-zarr](https://github.com/fideus-labs/ngff-zarr) | Documentation: [ngff-zarr.readthedocs.io](https://ngff-zarr.readthedocs.io)", "recording_license": "", "do_not_record": false, "persons": [{"code": "8N9ANH", "name": "Matt McCormick", "avatar": "https://pretalx.com/media/avatars/LCSK7L_JWLcV7u.webp", "biography": "I am a research software engineer who helps scientists perform computational image analysis for reproducible research.", "public_name": "Matt McCormick", "guid": "e7bd8404-935f-5fed-ab51-2604ac3b510e", "url": "https://pretalx.com/scipy-2026/speaker/8N9ANH/"}], "links": [{"title": "Source code", "url": "https://github.com/fideus-labs/ngff-zarr", "type": "related"}, {"title": "Documentation", "url": "https://ngff-zarr.readthedocs.io", "type": "related"}, {"title": "MCP Server", "url": "https://pypi.org/project/ngff-zarr-mcp/", "type": "related"}, {"title": "PyPI Package", "url": "https://pypi.org/project/ngff-zarr/", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/TADDJP/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/TADDJP/", "attachments": []}, {"guid": "c7be1b40-5978-59e1-b749-4d69cff4431b", "code": "ETT3W9", "id": 93248, "logo": null, "date": "2026-07-15T11:25:00-05:00", "start": "11:25", "end": "2026-07-15T11:55:00-05:00", "duration": "00:30", "room": "University Hall", "slug": "scipy-2026-93248-xarray-datastructures-in-biology-examples-and-best-practices", "url": "https://pretalx.com/scipy-2026/talk/ETT3W9/", "title": "Xarray DataStructures in Biology \u2013 Examples and Best Practices", "subtitle": "", "track": "Biological and Medical Sciences", "type": "Talk", "language": "en", "abstract": "In the past year [Xarray](https://xarray.dev/blog/xarray-biology) has seen increased usage across various sub-fields of biology, revealing interesting challenges. It can be difficult to determine the best way to represent a data structure (e.g. anndata, NGFF-Zarr) as an Xarray object. Furthermore, some use cases such as whole brain imaging require the use of lesser known Xarray features such as custom indexes.\n\nIn this talk I will showcase examples of how to encode common biological data structures as Xarray objects. Finally, I will demonstrate how the custom index infrastructure has expanded what types of data can be usefully encoded in Xarray.", "description": "## Background\n\nBiological datasets come in a wide variety of shapes, sizes, and types. However, there are common challenges faced across biology when dealing with complex structured data, such as keeping track of real-world coordinates. [Xarray](https://docs.xarray.dev/en/stable/getting-started-guide/why-xarray.html) provides a powerful solution to these issues. Additionally, Xarray provides first class support for HDF and Zarr files, formats already in wide use in biology. \n\n## Issues\n\nIncreased usage in various projects has revealed issues around converting existing data structures into Xarray. For example some Napari developers use Xarray to keep track of physical units from images, but they struggled with the fact that various libraries had different conventions for encoding metadata into Xarray.\n\nThat struggle is exemplary of a larger issue: The best way to convert an existing data structure (on disk or in memory) to Xarray may not be obvious, especially for newer users of Xarray. Or it is possible to be unaware of functionality (e.g. Custom Indexes) necessary to fully represent a data structure.\n\n## Success Stories\n\nThese conversion difficulties are solvable.\n\nI will present three examples of successful conversion of common biological data structures to Xarray. Through these I will discuss, what worked, what was hard, and recommendations for anyone interested in using Xarray for biology.\n\n- Microscope Images: [OME-Zarr (NGFF)](https://ngff.openmicroscopy.org/)\n- Omics Data: [AnnData](https://anndata.readthedocs.io/en/stable/)\n- Multimodal data (Single cell Raman Spectroscopy + Microscope Images + Lipidomics)\n\n## Indexes\n\nA key enabling technology to allow some biological data structures to be represented in Xarray is the ability to write custom indexes. Custom indexes are powerful tools that can also encode complex interconnected relationships in metadata data structures and allow sophisticated selection queries. However they are not yet well known in the community. \n\nTo showcase their use I will demonstrate the [indexes](https://ianhuntisaak.com/xarray-linked-indexes) developed for a real world use case of combined speech and intracranial EEG data. These indexes also show the benefits of cross field collaboration as they are useful in non-biological applications as well.\n\nXarray also has newly built-in Indexes built using the custom index infrastructure. These indexes allow for opening huge data sets, such as whole brain images, which would previously have resulted in  out of memory errors. I will show how these indexes enable opening a sectioned brain image in Xarray.\n\n## Conclusion\n\nTo conclude I will summarize the advice on how to convert a biological data structure into an Xarray object, and how to fully leverage Xarray\u2019s functionality.\n\nThis will include how to think through:\n\n- How metadata maps to Xarray\n- What kinds of selection queries you need\n- The practicalities of data loading\n\nFinally, and most importantly, advice on how to do this as a community, and where to get help.\n\n### Context\n\nBlog posts:\nhttps://xarray.dev/blog/xarray-napari-plan\nhttps://xarray.dev/blog/flexible-indexing\nhttps://xarray.dev/blog/xarray-biology\n\n\nPrior SciPy Talk about Xarray and Biology:\n\nhttps://www.youtube.com/watch?v=ujOseM1Bk1g\n\nThat talk focused on introducing the idea of Xarray - this talk is more concrete with examples and advice on loading data into xarray and what to do with it once there. \n\n\n**Speaker**\nI am a multimodal-microscopist who has since branched out to support multiple areas of Biology in my role as the Xarray Community Developer where I focus on ensuring Xarray has the tools biologists need and educating biologists about how Xarray might be useful for them.", "recording_license": "", "do_not_record": false, "persons": [{"code": "Y3HT8K", "name": "Ian Hunt-Isaak", "avatar": null, "biography": "I am working as an Xarray community developer at Earthmover. In this role I am focused on improving Xarray\u2019s support and documentation for the biology/biomedical community. Prior to this I completed my PhD in which I extensively used Xarray, zarr and the Pydata stack to implement custom microscope control software and analyze multimodal timelapse single cell microscopy data. I loved the open source scientific software so much that now I get to work full time improving it and sharing it with others.", "public_name": "Ian Hunt-Isaak", "guid": "71e2c4db-d4ce-5032-8d7c-b813c88572fd", "url": "https://pretalx.com/scipy-2026/speaker/Y3HT8K/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/ETT3W9/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/ETT3W9/", "attachments": []}, {"guid": "522fab25-9630-5535-b07e-de1458dc1c94", "code": "BG3PSA", "id": 92409, "logo": null, "date": "2026-07-15T13:15:00-05:00", "start": "13:15", "end": "2026-07-15T13:45:00-05:00", "duration": "00:30", "room": "University Hall", "slug": "scipy-2026-92409-discovering-particles-how-we-analyze-petabytes-of-particle-collision-data-using-python", "url": "https://pretalx.com/scipy-2026/talk/BG3PSA/", "title": "Discovering Particles: How we analyze petabytes of particle collision data using python", "subtitle": "", "track": "Physics and Astronomy", "type": "Talk", "language": "en", "abstract": "At CERN's Large Hadron Collider, we collide protons at near light-speed to discover new particles and understand fundamental physics. Python is becoming the primary language for analyzing this data, marking a significant evolution from the Fortran and C++ workflows of previous decades.\n\nThis talk explores the modern Python-based analysis pipeline of High-Energy Physics (HEP) and the technical challenges it addresses. We'll present how we handle nested, jagged data structures and work with data at the petabyte scale using the community-driven Scikit-HEP ecosystem of specialized tools for efficient and high-performance data analysis.\n\nWe'll show how we're building a Python stack that integrates with distributed computing frameworks and leverages GPU acceleration. Beyond domain-specific analysis tools, HEP's transition to Python has driven improvements to the broader Python packaging ecosystem, including contributions to cibuildwheel, the development of scikit-build-core, and advances in pybind11, benefiting anyone building Python packages with compiled extensions.", "description": "This talk takes you inside the data analysis pipeline at CERN's Large Hadron Collider, where physicists are transitioning from decades of Fortran and C++ workflows to Python-based analysis. We'll explore the technical challenges of working with petabyte-scale, nested data, and show how the solutions developed for High-Energy Physics (HEP) have become valuable tools for the broader Python community.\n\nWe will begin with understanding why HEP computing evolved the way it did. Fortran dominated for decades, then C++ and the ROOT framework became standard in the 90s. We'll explain what triggered the recent shift toward Python: the maturation of NumPy and the scientific stack, the need for faster iteration, and the desire to make analysis more accessible. This history explains the design constraints and opportunities that shaped today's tools.\n\nAt the heart of modern HEP analysis is Scikit-HEP, a community-driven collection of Python packages. We'll dive into the key components: uproot enables pure-Python access to ROOT files without C++ dependencies, Awkward Array provides NumPy-like operations on jagged data structures, hist delivers high-performance histogramming, and additional libraries handle vector math and statistical fitting. Through code examples, we'll demonstrate how these pieces fit together in an actual analysis workflow.\n\nOne of the most interesting technical problems is the structure of collision data itself. When protons collide, each event produces a different number of particles, each with multiple properties. Traditional rectilinear arrays can't represent this naturally. You need nested, variable-length arrays. This isn't just a physics problem; it's the same challenge you face with nested JSON-like data. We'll show how Awkward Array's approach to jagged data offers an elegant solution that's applicable far beyond physics.\n\nScale presents another major challenge. The High-Luminosity LHC upgrade will require analyzing petabytes in under an hour. We'll present our approach: leveraging distributed computing systems (like Dask) across clusters, using GPU acceleration where it provides the most benefit, and designing analysis facilities that colocate computation with data storage. These patterns are relevant to anyone tackling large-scale data problems.\n\nHEP's relatively late adoption of Python created an interesting dynamic: we needed production-quality infrastructure for building binary extensions but didn't have legacy tools to maintain. This drove significant contributions to the Python packaging ecosystem. We needed reliable cross-platform wheel building for packages like boost-histogram, awkward, and iminuit, which led to major improvements in cibuildwheel. We needed better build systems for C++ extensions, which resulted in scikit-build-core. We pushed forward pybind11 development and originally created the Scientific Python development guide and cookie template. These infrastructure improvements now benefit anyone distributing Python packages with compiled code.\n\nThe broader theme is how domain-specific needs can drive general-purpose innovation. The tools and infrastructure HEP has developed address problems common across scientific computing and data engineering.", "recording_license": "", "do_not_record": false, "persons": [{"code": "3E3SVE", "name": "Iason Krommydas", "avatar": "https://pretalx.com/media/avatars/3E3SVE_gPPzXZE.webp", "biography": "I'm a PhD student in the Department of Physics and Astronomy at Rice University, conducting research in high-energy physics as a member of the CMS experiment at the Large Hadron Collider at CERN. My work focuses on studying Higgs boson decays into two photons, analyzing data collected by the CMS detector, and contributing to software development for large-scale scientific analyses. I'm passionate about scientific computing and open-source tools that enable reproducible and efficient research. I\u2019m maintainer of Awkward Array, an array library for nested, variable-sized data, using NumPy-like idioms, and an author and maintainer of Coffea, a toolkit designed to simplify data analysis in particle physics. With deep experience in the scientific Python ecosystem, I enjoy building tools that drive insight and accelerate scientific discovery.", "public_name": "Iason Krommydas", "guid": "acba2ad9-cd31-50a0-a998-aea6be3632e7", "url": "https://pretalx.com/scipy-2026/speaker/3E3SVE/"}, {"code": "AYBNYT", "name": "Henry Schreiner", "avatar": "https://pretalx.com/media/avatars/3SCCSK_KGvI6cP.webp", "biography": "Henry Schreiner is a Computational Physicist / Research Software Engineer in High Energy Physics at Princeton University. He specializes in the interface between high-performance compiled codes and interactive computation in Python, in software distribution, and in interface design. He has previously worked on computational cosmic-ray tomography for archaeology and high performance GPU model fitting. He is currently a member of the IRIS-HEP project, developing tools for the next era of the Large Hadron Collider (LHC).\n\nHe is a maintainer/core developer for packaging, build, scikit-build, cibuildwheel, pybind11, meson-python, nox, and plumbum for Python. He is an admin of Scikit-HEP, and a lead designer on boost-histogram, hist, UHI, vector, uproot-browser, Particle, and DecayLanguage packages there. He is also the lead author of the Scientific-Python Development guide and Scientific-Python/cookie. He is the primary author of CLI11, a C++ library used by Microsoft terminal and many others. He is also the lead web developer for IRIS-HEP. He is also the author of Modern CMake and a variety of CMake, GPU, and Python training courses and classes.", "public_name": "Henry Schreiner", "guid": "c4cb3a37-16ab-5a5f-87dc-eaf183d57c45", "url": "https://pretalx.com/scipy-2026/speaker/AYBNYT/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/BG3PSA/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/BG3PSA/", "attachments": []}, {"guid": "61b042a8-cf1e-5962-b953-037593abc4bc", "code": "SX9977", "id": 101378, "logo": null, "date": "2026-07-15T13:55:00-05:00", "start": "13:55", "end": "2026-07-15T14:25:00-05:00", "duration": "00:30", "room": "University Hall", "slug": "scipy-2026-101378-qdk-chemistry-a-composable-python-toolkit-for-end-to-end-quantum-chemistry-on-quantum-computers", "url": "https://pretalx.com/scipy-2026/talk/SX9977/", "title": "QDK/Chemistry: A Composable Python Toolkit for End-to-End Quantum Chemistry on Quantum Computers", "subtitle": "", "track": "Physics and Astronomy", "type": "Talk", "language": "en", "abstract": "Quantum computers promise to tackle strongly correlated molecular systems that defeat classical electronic-structure methods, but realizing quantum utility depends on every stage of the pipeline, not just the quantum algorithm. QDK/Chemistry, an open-source package in the Microsoft Quantum Development Kit, treats this entire pipeline as a single, modular Python framework. Immutable data classes and stateless algorithms with fixed interfaces let researchers swap backends without changing application code. This talk introduces QDK/Chemistry's composable architecture, shows how classical and quantum stages interoperate to minimize quantum resources, and offers design patterns applicable beyond quantum computing.", "description": "QDK/Chemistry is an open-source package in the Microsoft Quantum Development Kit that provides a composable, end-to-end framework for quantum chemistry on quantum computers. It spans every stage of the quantum-classical workflow, from molecular setup and classical reference calculations through active-space reduction, Hamiltonian construction, fermion-to-qubit encoding, state preparation, and measurement. These stages are connected through a unified Python API backed by a high-performance C++ core.\n\nThe design rests on immutable data classes and stateless algorithms with fixed interfaces. A factory/registry plugin system makes every algorithm slot interchangeable: a researcher can swap a native backend for third party packages (e.g. PySCF, Qiskit, OpenFermion), or a custom implementation by changing a single string, with no rewiring of application code. Benchmarking, backend mixing, and custom extension are first-class operations rather than rewrites.\n\nBecause every stage is an interchangeable module, classical methods generate the high-quality inputs that quantum algorithms depend on, and the same classical results serve as baselines for judging where quantum methods offer genuine utility over the classical state of the art. The emphasis throughout is on minimizing quantum resources at every step and on making workflows reproducible and shareable. Reproducible serialization in XYZ, JSON, and HDF5 formats supports shareable, benchmarkable workflows across groups.\n\nQDK/Chemistry is available on PyPI, with documentation, examples, and companion datasets openly available.", "recording_license": "", "do_not_record": false, "persons": [{"code": "WCYPH7", "name": "David Williams-Young", "avatar": "https://pretalx.com/media/avatars/YJ7UQZ_ruOyTdk.webp", "biography": "David Williams-Young is a Principal Quantum Software Architect at Microsoft Quantum, where he serves as software lead for quantum applications. His work focuses on quantum computing applications in chemistry and materials science, including quantum algorithms, classical simulation methods, and the development of tools that bridge quantum computing and computational many-body theory. Prior to joining Microsoft, he was as a Scientist in the Applied Mathematics and Computational Research Division at Lawrence Berkeley National Laboratory, where he developed exascale electronic structure methods and software for DOE Leadership Computing Facilities. He received his Ph.D. in Chemistry from the University of Washington, specializing in relativistic electronic structure theory. He is the author of numerous open-source computational chemistry libraries and has served as a major contributor numerous quantum chemistry software packages.", "public_name": "David Williams-Young", "guid": "c559712d-07c9-5593-8a5a-a0e2d3e1f372", "url": "https://pretalx.com/scipy-2026/speaker/WCYPH7/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/SX9977/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/SX9977/", "attachments": []}, {"guid": "2f11906f-52ed-53f4-92af-b6edf23a254f", "code": "BFQAPR", "id": 93247, "logo": "https://pretalx.com/media/scipy-2026/submissions/BFQAPR/image_DRsXSyg.webp", "date": "2026-07-15T14:35:00-05:00", "start": "14:35", "end": "2026-07-15T15:05:00-05:00", "duration": "00:30", "room": "University Hall", "slug": "scipy-2026-93247-derivkit-end-to-end-derivative-based-inference-in-scientific-python", "url": "https://pretalx.com/scipy-2026/talk/BFQAPR/", "title": "DerivKit: End-to-End Derivative-Based Inference in Scientific Python", "subtitle": "", "track": "Physics and Astronomy", "type": "Talk", "language": "en", "abstract": "Many scientific workflows rely on derivatives of complex models: Fisher forecasts, sensitivity analysis, gradient-based inference, and emulator construction. In practice, these derivatives are often difficult to compute reliably and integrate into end-to-end inference pipelines.\n\nDerivKit is an open-source Python toolkit that provides a unified framework for derivative-based scientific inference. It supports multiple derivative backends and connects model evaluation directly to downstream inference tools, including Fisher analyses and higher-order likelihood approximations. The framework also provides diagnostics and visualization tools for exploring parameter sensitivities and degeneracies.\n\nOriginally developed for cosmological forecasting pipelines, DerivKit is designed to be domain-agnostic and easily integrated into scientific Python workflows.", "description": "Many scientific workflows rely on derivatives of complex computational models. Derivatives are central to Fisher forecasting, sensitivity analysis, gradient-based inference, emulator construction, and uncertainty propagation. In practice, however, derivative calculations are often implemented in ad-hoc ways within individual projects. This makes them difficult to reproduce, hard to diagnose when they fail, and challenging to integrate with downstream inference tools.\n\nDerivKit is an open-source Python toolkit designed to provide a structured framework for derivative-based scientific inference. The goal of the project is to connect model evaluation, derivative computation, and inference tools into a coherent workflow that is easy to use and inspect. Rather than focusing on a single derivative technique, DerivKit provides a unified interface for multiple derivative backends and supports flexible strategies for computing derivatives of arbitrary scientific models.\n\nThe framework allows users to wrap an existing model function and automatically construct derivative operators with respect to model parameters. These derivatives can then be used directly in inference pipelines, including Fisher matrix forecasts and higher-order likelihood approximations (DALI). In particular, DerivKit provides implementations of higher-order likelihood expansions that extend beyond the Gaussian Fisher approximation, enabling users to explore parameter degeneracies and non-Gaussian structure in likelihood surfaces.\nAn important design goal of DerivKit is to make derivative-based inference transparent and diagnostic-friendly. The toolkit includes utilities for evaluating derivative stability, exploring parameter sensitivities, and visualizing degeneracies in model parameter spaces. These diagnostics help users identify when derivatives are unreliable or when parameter combinations produce nearly degenerate model responses. DerivKit also supports a direct model-to-plot workflow that allows users to move seamlessly from derivative computation to visual analysis of inference results.\n\nAlthough DerivKit was originally developed for cosmological forecasting pipelines used in large astrophysical collaborations, the design of the framework is intentionally domain-agnostic. Many areas of scientific computing face similar challenges when working with derivatives of expensive or complex models. These include climate modeling, epidemiological simulations, materials science, and simulation-based inference workflows. By separating derivative infrastructure from domain-specific modeling code, DerivKit aims to provide a reusable tool that can integrate naturally into a wide range of scientific Python environments.\nThis talk will introduce the design principles behind DerivKit and demonstrate how derivative infrastructure can be organized to support robust scientific inference workflows. We will discuss common pitfalls in numerical derivative calculations, present the architecture of the DerivKit framework, and show examples of derivative-based inference applied to realistic models.\n\nAttendees will learn how to structure derivative computations in a reproducible way, how to diagnose instability and parameter degeneracies, and how derivative-based methods such as Fisher analyses and higher-order likelihood approximations can be incorporated into scientific Python pipelines.", "recording_license": "", "do_not_record": false, "persons": [{"code": "LNYXJW", "name": "Niko Sarcevic", "avatar": "https://pretalx.com/media/avatars/QKSNYV_3WERqwS.webp", "biography": "Nikolina \u201cNiko\u201d \u0160ar\u010devi\u0107 is a cosmologist at Duke University working on cosmological inference and large-scale structure. She is a member of the LSST Dark Energy Science Collaboration (DESC) and the NASA Roman Space Telescope science collaborations. Her research focuses on statistical methods, astrophysical systematics, and scientific software for cosmology. Previously, she worked on dark matter searches as part of the XENON experiment.", "public_name": "Niko Sarcevic", "guid": "6301e5cc-46f0-5f16-872a-1f605fae6851", "url": "https://pretalx.com/scipy-2026/speaker/LNYXJW/"}, {"code": "EW3PCP", "name": "Matthijs van der Wild", "avatar": null, "biography": "", "public_name": "Matthijs van der Wild", "guid": "a8977046-d296-52ab-9ddd-72899a04b799", "url": "https://pretalx.com/scipy-2026/speaker/EW3PCP/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/BFQAPR/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/BFQAPR/", "attachments": []}, {"guid": "0b62da27-2217-531f-8786-cf36e7d67b22", "code": "NEKFB8", "id": 92475, "logo": null, "date": "2026-07-15T15:25:00-05:00", "start": "15:25", "end": "2026-07-15T15:55:00-05:00", "duration": "00:30", "room": "University Hall", "slug": "scipy-2026-92475-declare-don-t-parse-composable-genomic-analysis-with-giql-and-oxbow", "url": "https://pretalx.com/scipy-2026/talk/NEKFB8/", "title": "Declare, Don't Parse: Composable genomic analysis with GIQL and Oxbow", "subtitle": "", "track": "Biological and Medical Sciences", "type": "Talk", "language": "en", "abstract": "Genomic workflows remain tightly coupled to specialized file formats, forcing researchers to build brittle pipelines of format-specific CLI tools. We present projects that help shift this emphasis away from file parsing and towards declarative querying. Oxbow is a library that projects common genomic formats into Apache Arrow, enabling zero-copy integration with data frame libraries and analytics engines. GIQL (Genomic Interval Query Language) is an extended SQL dialect supporting genomic interval operations and semantics that transpiles to standard SQL, making genomic queries composable, readable, and backend-agnostic. Together, this architecture also facilitates the integration of genomic data into data warehouse and lakehouse platforms as well as agentic MCP workflows.", "description": "Genomic data tools remain tightly coupled to specialized file formats, forcing researchers to build brittle pipelines of format-specific CLI tools connected by ad hoc serialization. Meanwhile, standard SQL -- the lingua franca of data analytics -- lacks the vocabulary to express genomic interval relationships and operations that are fundamental to the field. To address both of these issues, we present a pair of projects that together shift the emphasis in genomics from file parsing towards declarative querying.\n\nThe first project, Oxbow, is a Rust-based adapter library that projects common genomic file formats, including BAM, VCF, BED, GTF, BigWig, and others, into Apache Arrow, a standard columnar in-memory representation for tabular analytics. By leveraging Arrow's C Data Interface, Oxbow streams records to Python with zero copy overhead, integrating directly with Polars, DuckDB, and Dask without intermediate serialization. Oxbow supports indexed range queries, column projection push-down, and remote data access via HTTP and object storage, enabling researchers to query genomic files hosted in the cloud without downloading them locally.\n\nThe second project, GIQL (Genomic Interval Query Language, pronounced \u201cJEE-quel\u201d) is an extended SQL dialect and transpiler for genomic interval operations. GIQL introduces domain-specific operators, such as INTERSECTS, WITHIN, and NEAREST, that let researchers express genomic interval logic and spatial joins declaratively. For example, `WHERE a.interval INTERSECTS b.interval` transpiles into standard SQL predicates that any engine can execute. Because the transpiler targets standard SQL, it is backend-agnostic: the same query runs on DuckDB, Polars, SQLite, or any SQL-compliant engine. GIQL provides a declarative alternative to bedtools-style scripting, making genomic queries composable, readable, and reproducible.\n\nThese libraries work together, where Oxbow streams legacy genomic files as Arrow record batches into a SQL engine, and GIQL provides the extended query semantics to interrogate them. We will demonstrate this composition in practice. By building on open, domain-agnostic standards, this architecture also facilitates the integration of genomic data into modern data warehouse and lakehouse platforms as well as agentic MCP workflows.", "recording_license": "", "do_not_record": false, "persons": [{"code": "8EV3GY", "name": "Nezar Abdennur", "avatar": "https://pretalx.com/media/avatars/TU8RF7_cqEhUGF.webp", "biography": "I am an Assistant Professor in the Department of Genomics and Computational Biology and the Department of Systems Biology at UMass Chan Medical School.\n\nI lead a computational research group (https://abdenlab.org) with a dual mandate. My group's biological research focuses on the 3D organization of the genome (3C/Hi-C technologies), its relationship to the epigenome, and the resulting manifold influences on cellular fate, differentiation, aging, and disease. My group's open-source interests are in supporting foundational infrastructure to improve AI and data science for genomics and multi-omics, especially in the scientific Python ecosystem.", "public_name": "Nezar Abdennur", "guid": "c6b2a1ab-45a0-5d58-a37d-25a555ca25fc", "url": "https://pretalx.com/scipy-2026/speaker/8EV3GY/"}, {"code": "VYDQUR", "name": "Conrad Bzura", "avatar": null, "biography": "", "public_name": "Conrad Bzura", "guid": "075d76b0-b881-51b9-8382-2b39c2566483", "url": "https://pretalx.com/scipy-2026/speaker/VYDQUR/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/NEKFB8/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/NEKFB8/", "attachments": []}, {"guid": "791ced0f-5260-5578-8ee6-d99895552ebb", "code": "RE9ETJ", "id": 92488, "logo": null, "date": "2026-07-15T16:05:00-05:00", "start": "16:05", "end": "2026-07-15T16:35:00-05:00", "duration": "00:30", "room": "University Hall", "slug": "scipy-2026-92488-simulation-informed-machine-learning-workflows-for-petase-engineering", "url": "https://pretalx.com/scipy-2026/talk/RE9ETJ/", "title": "Simulation-Informed Machine Learning Workflows for PETase Engineering", "subtitle": "", "track": "Biological and Medical Sciences", "type": "Talk", "language": "en", "abstract": "Engineering enzymes with improved catalytic activity remains a central challenge in biotechnology. In this research, we focus on engineering PETase, a plastic-degrading enzyme, as a testbed for developing a simulation-informed machine learning workflow. We present a Python framework that integrates molecular simulations, docking, and structural analysis with modern machine learning methods to predict enzyme activity from sequence and structure. By combining simulation-derived descriptors\u2014including active-site geometry, electrostatics, stability metrics, dynamics, and docking scores\u2014with sequence embeddings, we generate interpretable predictions that guide rational mutation strategies. While developed for PETase engineering, the workflow is extensible to broader de novo enzyme design efforts.", "description": "Polyethylene terephthalate (PET) plastic degradation has emerged as a major environmental challenge. The discovery of PETase, originally identified in Ideonella sakaiensis, opened new possibilities for enzymatic plastic recycling. However, improving PETase stability, activity, and substrate specificity remains an open problem in protein engineering.\n\nIn this presentation, we introduce a modular Python workflow designed specifically to engineer improved PETase variants. The workflow integrates molecular modeling tools\u2014including Rosetta, FoldX, and AMBER molecular dynamics simulations\u2014with docking and modern machine learning frameworks (scikit-learn and PyTorch). Rather than relying purely on sequence-based ML, We incorporate simulation-informed descriptors such as:\n\n- Electrostatic potential and catalytic residue environment\n- Stability and \u0394\u0394G predictions\n- Molecular dynamics\u2013derived flexibility metrics\n- Docking scores with PET oligomers\n\nThese simulation-derived features are combined with sequence embeddings to predict enzyme activity in an interpretable manner. This enables rational mutation prioritization rather than black-box screening.\n\nKey components include:\n\n**1. Data Pipelines**\n    Standardized processing of sequence variants, simulation outputs, structural descriptors, and  \n    docking results in an automated and reproducible workflow.\n**2. Simulation-Informed Feature Engineering**\n    Integration of structural, dynamic, and energetic descriptors with learned sequence embeddings.\n**3. Machine Learning Modeling**\n    Cross-validation, uncertainty estimation, and careful evaluation to ensure robust predictive \n    performance.\n**4. Interpretability for Engineering**\n    Feature attribution methods to identify which structural or dynamic properties most strongly \n    influence predicted activity \u2014 directly informing mutation strategies.\n\nWe demonstrate the workflow by engineering PETase variants with predicted improvements in catalytic efficiency and stability. By integrating docking of PET oligomers, molecular dynamics simulations, and ML prediction, we show how simulation-informed features improve predictive performance compared to sequence-only baselines.\n\nThis PETase-focused approach illustrates how tightly integrating physics-based simulations with machine learning enables actionable design decisions.\n\nWhile PETase is the immediate application, the framework generalizes to other enzyme families, offering a reproducible and extensible foundation for computational protein engineering.\n\n**What Attendees Gain**\n- A concrete PETase engineering case study\n- A reproducible Python-based workflow integrating simulations and ML\n- Practical strategies for combining docking, MD, and ML\n- Methods for interpretable prediction and rational mutation design\n- An extensible framework adaptable to other enzyme systems", "recording_license": "", "do_not_record": false, "persons": [{"code": "NVD3VR", "name": "Sai Sanjana Prakash", "avatar": "https://pretalx.com/media/avatars/HJJ99F_ldfQ04r.webp", "biography": "Sanju is an independent scientist building computational tools for scientific discovery. Her work spans machine learning, molecular simulation, protein engineering, and scientific software, with an emphasis on developing computational systems that make scientific research more scalable, reproducible, and accessible. She is interested in the principles of intelligence and complex biological systems, and in building general-purpose computational frameworks that expand how science is explored, understood, and accelerated.", "public_name": "Sai Sanjana Prakash", "guid": "35471ff0-13ea-5182-b0a4-a594e2077419", "url": "https://pretalx.com/scipy-2026/speaker/NVD3VR/"}, {"code": "YZECPK", "name": "Charlie Hou", "avatar": null, "biography": null, "public_name": "Charlie Hou", "guid": "5adcf728-7600-54a4-9f90-a80f973b92ea", "url": "https://pretalx.com/scipy-2026/speaker/YZECPK/"}, {"code": "HDSSFT", "name": "Justin Kashi", "avatar": "https://pretalx.com/media/avatars/C7TS9Q_xrgjxzy.webp", "biography": "Hi there! I did my Bachelor in Bioengineering at McGill University (2019-2023), after which I did my Master's thesis in Synthetic Biology & Systems Biology in the Ignea Lab at McGill University (2023-2025) where I studied transcriptomics and metabolomics in Tacca plant species. I fell in love with foundational protein language models and modeling enzymes functions using structural and sequence information. After participating in the Align Bio 2025 PETase protein engineering tournament in Fall 2025 with my teammates, we developed a protein engineering framework using our methodology which we are presenting at the SciPy 2026 conference!", "public_name": "Justin Kashi", "guid": "dfab13e1-856b-5dc9-8f52-1105ca6fa397", "url": "https://pretalx.com/scipy-2026/speaker/HDSSFT/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/RE9ETJ/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/RE9ETJ/", "attachments": []}, {"guid": "e97a1aff-2e75-58b4-ab9e-77af9c94ccca", "code": "QF7KXB", "id": 97796, "logo": null, "date": "2026-07-15T18:00:00-05:00", "start": "18:00", "end": "2026-07-15T19:00:00-05:00", "duration": "01:00", "room": "University Hall", "slug": "scipy-2026-97796-poster-session", "url": "https://pretalx.com/scipy-2026/talk/QF7KXB/", "title": "Poster Session", "subtitle": "", "track": "Poster Session", "type": "Poster Session", "language": "en", "abstract": "The Poster session will be in University Hall from 6:00-7:00pm. Meet with the poster authors to ask questions and learn about the posters that will be on display throughout the main conference.", "description": "1. **Hannes Hapke, David Cardozo, Triveni Gandhi**\t- Opening the Black Box: Mechanistic Interpretability of Agent Tool Selection with Sparse Autoencoders (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n2. **Gita Mohammadi** - Using Scientific Python to Study Trigger Efficiencies in Searches for New Higgs Bosons at CERN (Spirit of SciPy)\n3. **Rudraksh Karpe, Shivay Lamba, Suvrakamal Das, Satyam Soni** - Python Carbon Loops: Closing the Feedback Loop Between Your Code and Its Climate Impact (Environmental, Earth, and Climate Sciences)\n4. **Venkateswaran Shekar** - RECAP: A Python framework for reproducible experiment capture and provenance (General)\n5. **Emmanuel I. Obi** - Teaching Python the Difference Between Radiation Dose and Damage (Biological and Medical Sciences)\n6. **Alexander Luebbert** - Data-Driven Optimization Framework for Competitive Performance in FIRST Robotics Competition (Scientific Computing in Education)\n9. **Carlos Garc\u00eda Jurado Suarez** - Efficient Federated Inference on Entomology Images with PyTorch (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n10. **Allison Ding** - Minimizing Noise Clusters in Topic Modeling: A Scalarized Hyperparameter Optimization Approach with GPU Acceleration (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n11. **Nick Hodgskin** - Modernising Parcels for the era of Cloud-Native Geospatial data\t(Environmental, Earth, and Climate Sciences)\n12. **Daniel McCloy, Eric Larson, Britta Westner** - On-boarding and retaining maintainer talent for MNE-Python\t(Maintainers and Community)\n13. **Noor Aftab** - Building with Agents: The Open Source Story of the Scientific Repo-Agent (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n14. **Deven Maheshwari** - Climate is not a straight line: Scalable Python-based GAMM Workflows for Wildlife Conservation (Environmental, Earth, and Climate Sciences)\n15. **Avik Basu** - Right Predictions, Wrong Reasons: Explanation Drift Monitoring in Production (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n16. **Erik Bolch, Mahsa Jami** - Multi-Sensor Earth Science Made Easy: NASA VITALS\t(Environmental, Earth, and Climate Sciences)\n17. **Rachael Sexton** - Trimming the Hairball: Three Libraries for Better Network Recovery & Metrology (General)\n18. **Abby Mitchell**\t - Unravelling the mystery of free threading for scientific computing (General)\n19. **Joe Cheng, on behalf of Isabella Vel\u00e1squez** - Merging without fear: Using validation to protect your Python workflows (General)\n20. **Aishwarya Chander, Christian La France, Alexander** - A Cloud-Native Single-Cell Data Analysis pipeline with Zarr, Icechunk, and RAPIDS-singlecell (Biological and Medical Sciences)\n21. **Richard Iannone** - Creating beautiful documentation sites for Python libraries with Great Docs (Maintainers and Community)\n22. **Tarun Gandrathi** - Building Trustworthy Scientific Python Workflows in Pharma (Biological and Medical Sciences)\n23. **Jesse Loi** - Bridging the Technical Gap: A Student-Led RAG Pipeline for Community-Driven Document Analysis (Scientific Computing in Education)\n24. **Dylan Madisetti** - Hash all the things: Caching for fast notebook restarts (General)\n25. **Bhupendra Raut** - Adapt: Prototyping a Real-Time, Reproducible Data Analysis Framework for Adaptive Radar Scanning (Environmental, Earth, and Climate Sciences)\n26. ** Adam Theisen** - From Towers to Lidars: ACT Unifies Atmospheric Data into Reproducible Python Workflows (Environmental, Earth, and Climate Sciences)\n27. **Marc Berliner**\t - 5x Fewer Stored Time Steps with Certified Accuracy: A Streaming Compression Algorithm for Adaptive Differential Equation Solvers (Environmental, Earth, and Climate Sciences)\n28. **Lucas Sterzinger**  - Improving access of HDF5/NetCDF4 data in S3 cloud storage: a case study using NASA Land Surface Model data (Environmental, Earth, and Climate Sciences)\n29. **Sruthi Pisipati, Haris Javed** - Everything That Breaks When You Put an LLM Agent in Production (Data-Driven Discovery, Machine Learning and Artificial Intelligence)", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/QF7KXB/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/QF7KXB/", "attachments": []}], "Virtual Sessions": [{"guid": "36db5939-e6e5-56cb-87bb-79014307a3bc", "code": "UZB8CN", "id": 99515, "logo": null, "date": "2026-07-15T18:00:00-05:00", "start": "18:00", "end": "2026-07-15T19:00:00-05:00", "duration": "01:00", "room": "Virtual Sessions", "slug": "scipy-2026-99515-virtual-poster-session", "url": "https://pretalx.com/scipy-2026/talk/UZB8CN/", "title": "Virtual Poster Session", "subtitle": "", "track": "Poster Session", "type": "Poster Session", "language": "en", "abstract": "The Virtual Poster session will be hosted on Gather from 6:00-7:00pm. Gather is a browser-based virtual conference platform that allows participants to move around a digital conference space using customizable avatars. As attendees walk through the poster hall, they can view poster thumbnails, open full-size posters, and start video or audio conversations with presenters nearby. No software installation is required; Gather runs directly in your web browser.", "description": "1. **Georg Heiler, Daniil Gafni** - Versioning Multimodal Data with Metaxy (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n2. **Emmanuel I. Obi** - Define Your Own Dimensions: Algebraic Unit Conversion Beyond SI, CGS, and Natural Units (General)\n3. **Yu-Lin Chen, Tyng-Ruey Chuang  | \u838a\u5ead\u745e, Cheng-Jen Lee | \u674e\u627f\u9331** - Toward Reliable Localization of Free and Open Source Software: LLM-assisted Translation Workflows for QGIS (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n4. **Pavan BG** - NODEFit - Fit time-series data with a Neural Differential Equation (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n5. **Jeroen Janssens** - From Script to Tool: Leveling Up Your Python Workflow (General)\n6. **Taewoon Kim** - From Transactions to Vectors: Embedded Multi-Model Data Workflows in Scientific Python (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n7. **Rodrigo Silva Ferreira** - 25 Years of Interactive Scientific Computing: From IPython and Jupyter to IDE-Native Notebooks (Spirit of SciPy)\n8. **Rene Lagos** - A Reproducible \"Data Lakehouse\" for High-Resolution Gastric Cancer Epidemiology Study in Chile (Biological and Medical Sciences)\n9. **Srilakshmi Bharadwaj** - When \u201cScalable\u201d Isn\u2019t Scalable: Real Lessons from Production Data Systems (General)\n10. **Rylie Weaver** - alphagenome-pt: Training AlphaGenome Models in PyTorch (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n11. **Pankaj Arora** - AI-Driven Inventory Redistribution Between Hospitals to Reduce Waste and Shortages Using Predictive Analytics \n12. **Gift  Ojeabulu** - Why Reproducibility Still Fails in Modern Machine Learning (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n13. **Shaurya Agarwal** - The Silmaril strikes again - Practical Ontology Engineering for AI, Reasoning Engines and Real-World Applications (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n14. **Prashant Badiger, Gajendra Deshpande, Mallikarjun mrityunjaya** - math - Real-Time AI/ML-Based Phishing Detection and Prevention Using the Python (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n15. **Mohd Toukir Khan** - Beyond First PRs: Converting Students into Long-Term Open Source Contributors (Maintainers and Community)\n16. **Aayush Gauba** - Numerical Stability Pitfalls in Scientific Optimization Pipelines (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n17. **Kedar Dabhadkar** - Self-Evolving Skill Graphs: Using Reflective Optimization for AI Agent Skill Organization (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n18. **Ahmad El Hajj** - Density Functions and Random Number Generators of Alpha-Stable Distributions (Scientific Computing in Education)\n19. **Daniel Samuel Etukudo** - Using Food and AI to Manage Chronic Conditions (General)\n20. **Aayush Gauba** - Detecting Anomalies in Scientific Data Using SciPy\u2019s Statistical and Signal Tools (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n21. **Conor Hoekstra** - Parrot Python:  Fused Array Operations for the GPU (General)\n22. **Johannes Plambeck** - Optimising HCP Sample Allocation in Pharma: Combining Non-Linear Ensemble Learning, Spatial Lags, and Integer Programming (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n23. **Rudraksh Karpe, Shivay Lamba, Suvrakamal Das, Satyam Soni** - Recursive Language Models (RLMs): Scaling to Infinite Context via Programmatic Decomposition (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n24. **Gauri Sarode** - When Search Becomes Intelligent: The Rise of LLMs and AI Agents in Discovery Systems (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n25. **Sanjiban Sengupta** - A Unified Inference Interface for Low-Latency Machine Learning in High-Energy Physics (Physics and Astronomy)\n26. **Sho Tanaka** - Avoiding Zero-Trade Policies in RL with a Decoupled MLOps Architecture (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n27. **Lucas Squarize Chagas, Avik Basu** - The Missing Lever in ML Deployment: Threshold Tuning using Regression Discontinuity (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n28. **Arunkumar Amaran** - Conversational AI Interfaces for Retail Data Engineering and Business Intelligence (General)\n29. **Shivika Bisen** - Solving the AI Eval Gap: Domain-Aware Evals for Production AI Agents (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n30. **Petr Andreev** - JIT in the Wild: CPython\u2019s Next Step vs PyPy and V8 (With Real Benchmarks) (Scientific Computing in Education)\n31. **Petr Andreev** - CPython Under Load: NoGIL, Green Threads, AsyncIO vs Other Langs: deep-dive and benchmarks (Scientific Computing in Education)\n32. **GUSTAVO COELHO HAASE, PAULO DOURADO** - PanelBox: A Comprehensive Python Library for Panel Data Econometrics (General)\n33. **Vinay Vyas** - Benchmarking Edge-Accelerated Genomics: A Pilot Study of Unified Memory Architectures in Deep-Sea Metagenomics (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n34. **A Seshaditya** - Large Language Models and Physics-AI for Fluid Dynamic Simulations (Physics and Astronomy)\n35. **Sauhard Bhatt** - Mr. (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n36. **Ruben Huidekoper, Camila Birocchi** - Be Your Own Consultant: Start Self-Diagnozing Your BI Tech Stack (General)\n37.  **Viraj Sharma** - XAI - MechInterp and Causal Visualizations (Data-Driven Discovery, Machine Learning and Artificial Intelligence)\n38. **Shaurya Agarwal** - Vogon Poetry - Columnar Data, Zero-Copy, etc. etc.: key ideas for data and AI teams to up their game\u2026 (Data-Driven Discovery, Machine Learning and Artificial Intelligence)", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/UZB8CN/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/UZB8CN/", "attachments": []}]}}, {"index": 4, "date": "2026-07-16", "day_start": "2026-07-16T04:00:00-05:00", "day_end": "2026-07-17T03:59:00-05:00", "rooms": {"Memorial Hall": [{"guid": "077c5c37-f82a-5f8a-98d5-a99df6dfa02e", "code": "7UJUMK", "id": 97807, "logo": null, "date": "2026-07-16T09:15:00-05:00", "start": "09:15", "end": "2026-07-16T10:00:00-05:00", "duration": "00:45", "room": "Memorial Hall", "slug": "scipy-2026-97807-keynote-amber-case-calm-technology-and-the-history-of-ai", "url": "https://pretalx.com/scipy-2026/talk/7UJUMK/", "title": "Keynote: Amber Case, \"Calm Technology and the History of AI\"", "subtitle": "", "track": "Keynotes", "type": "Keynote", "language": "en", "abstract": "Research Director at the Metagovernance Project and founder of The Calm Tech Institute", "description": "Amber Case's work explores the intersection of humans and technology, challenging us to design systems that inform rather than overburden. \n\nCase is redefining the relationship between humans and technology. As the founder of the Calm Tech Institute and a former fellow at MIT and Harvard, Case brings a profound perspective on how we can design complex systems to be calm: interfaces that work with peripheral attention and inform at different resolution levels.\n\nIn her keynote, \"Calm Technology and the History of AI,\" she will explore moving from \"smart things\" to \"smarter people,\" how to design systems that inform us without overwhelming us, why the future of interface design might involve bringing back the button, and how to ensure modern systems are built in line with how the different parts of our brains interpret information.", "recording_license": "", "do_not_record": false, "persons": [{"code": "FFWTK8", "name": "Amber Case", "avatar": "https://pretalx.com/media/avatars/ARPF3W_lVSwBAf.webp", "biography": "Amber Case is a design advocate, speaker, and author of four books including Calm Technology and A Kids Book About Technology. Fellow at MIT\u2019s Center for Civic Media and Harvard\u2019s Berkman Klein Center for Internet & Society, co-founder and CEO of Geoloqi (acquired by Esri). Named 30 under 30, Fast Company Most Influential Women in Tech, National Geographic Emerging Explorer, she received Bell Labs Claude Shannon Innovation Award. She studies human-technology interaction, culture, design, governance, and AI. At Metagovernance Project, she founded the Calm Tech Institute", "public_name": "Amber Case", "guid": "694300fd-39ef-56d9-9dde-aa5962bb5ef9", "url": "https://pretalx.com/scipy-2026/speaker/FFWTK8/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/7UJUMK/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/7UJUMK/", "attachments": []}, {"guid": "a4b5e0c9-5b34-5abb-ae07-4924156f8e67", "code": "B9GAJH", "id": 97805, "logo": null, "date": "2026-07-16T10:00:00-05:00", "start": "10:00", "end": "2026-07-16T10:25:00-05:00", "duration": "00:25", "room": "Memorial Hall", "slug": "scipy-2026-97805-scipy-tools-plenary", "url": "https://pretalx.com/scipy-2026/talk/B9GAJH/", "title": "SciPy Tools Plenary", "subtitle": "", "track": "SciPy Tools", "type": "Tools Plenary", "language": "en", "abstract": "A session featuring updates and roadmaps from maintainers of core Scientific Python libraries and tools.", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/B9GAJH/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/B9GAJH/", "attachments": []}, {"guid": "d0fba73c-d0f0-549c-aebc-6ad452cc4d36", "code": "PSQLHP", "id": 93260, "logo": null, "date": "2026-07-16T10:45:00-05:00", "start": "10:45", "end": "2026-07-16T11:15:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-93260-pun-intended-consequences", "url": "https://pretalx.com/scipy-2026/talk/PSQLHP/", "title": "Pun Intended Consequences", "subtitle": "", "track": "Spirit of SciPy", "type": "Talk", "language": "en", "abstract": "Did you know that waffles were invented in the 14th century? Is Acetaminophen gluten free? If you said \"yes\" to both questions, you must have seen [Damon McDougall's legendary SciPy 2014 lightning talk](https://www.youtube.com/watch?v=ln4nE_EVDCg&t=3255s). \n\nLet's distill the lore of lightning talks and touch on SciPy culture over the  years and \"make sure we get all the history\" (or a yeast squares sparse low rank approximation of it)\n\nGather 'round, slithering scientists, and ye shall hear\na beer-ful of stories, of yesteryear\n\nThe Spirit of SciPy is the Track\nMC Pi (that's me), has got your back\n\nBeen coming to the conference since 2009\nSharing memories and photos, which we'll all combine\nsome will be profound, others asinine\n\nHoney, do you mead more proof?", "description": "Lighting talks are a perennial favorite for SciPy attendees. Let's celebrate the connection we make here, and give others a glimpse into our amazingly resourceful and creative community. \n\nI started coming to SciPy as a sponsored graduate student (2009-2011), gave talks in '13 and '14, started hosting lightning talks with Anthony Scopatz '17-'19, also volunteered as Communications Chair '18-'19, Program Co-Chair '20, '23, '24. \n\nSome of the SciPy lightning talks I co-hosted with Anthony Scopatz are linked in the middle of this\npage: https://pirsquared.org/talks/ (2017-2019). I also [gave my first and only SciPy Lighting talk in 2022](https://youtu.be/m3JbmBxKPBY?t=2898)\n\nSome of the photos I have I've also previously shared and talked about at the inaugural \"Another Open Source Podcast\" hosted by when I was a guest along with Madicken Munk\nhttps://open.spotify.com/episode/4LArGQQtRqGrixS9vnpNZk\n\n- [SciPy 2009](https://www.flickr.com/photos/tags/scipy2009) - last one at CalTech in Pasadena, CA\n- [SciPy 2010](https://www.flickr.com/photos/tags/scipy2010) - first one in Austin, Texas\n- [SciPy 2011](https://pirsquared.org/scipy2011/)", "recording_license": "", "do_not_record": false, "persons": [{"code": "JMLRLS", "name": "Paul Ivanov", "avatar": "https://pretalx.com/media/avatars/7LYBCA_RPIC2xh.webp", "biography": "Paul Ivanov's been proudly coming to SciPy since 2009. He got his start in the community in Matplotlib, and went on to also gain the commit bit to IPython and Jupyter projects. Away from keyboard, he likes biking and beekeeping.", "public_name": "Paul Ivanov", "guid": "a2a94efa-dd92-54ac-9798-f773206c39fc", "url": "https://pretalx.com/scipy-2026/speaker/JMLRLS/"}], "links": [{"title": "SciPy 2009 videos on Archive.org", "url": "https://archive.org/search?query=+++++++++SciPy+2009+&and%5B%5D=year%3A%222009%22", "type": "related"}, {"title": "SciPy 2010 Python Evangelism 101 - Peter Wang", "url": "https://www.youtube.com/watch?v=fK6E9tq-KjM", "type": "related"}, {"title": "SciPy 2014 Waffles Lightning Talk - Damon McDougal", "url": "https://www.youtube.com/watch?v=ln4nE_EVDCg&t=3255s", "type": "related"}, {"title": "Robert Kern's excellent audio-visual combination from 2021", "url": "https://www.youtube.com/watch?v=nFeYAd_9jW4&t=3039s", "type": "related"}, {"title": "SciPy Five debut @ SciPy 2022 (Here at SciPy - tell me why!)", "url": "https://www.youtube.com/watch?v=yhwXDaaq09s", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/PSQLHP/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/PSQLHP/", "attachments": []}, {"guid": "c7a6b0ac-2b1d-52f5-a1f4-43fa0293d6e2", "code": "MEKP9F", "id": 92076, "logo": null, "date": "2026-07-16T11:25:00-05:00", "start": "11:25", "end": "2026-07-16T11:55:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-92076-scipy-numpy-xarray-and-python-all-have-a-pixi-toml-why", "url": "https://pretalx.com/scipy-2026/talk/MEKP9F/", "title": "Scipy, Numpy, Xarray and Python all have a pixi.toml. Why?", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "After 3 years, Pixi is widely adopted in the scientific Python ecosystem. At SciPy 2026, we want to show why.\n\nScientific Python has specific challenges that Pixi can solve well; a lot of our beloved packages contain C, C++, Rust, CUDA or even Fortran code. With Pixi, a single tool can install the compilers, different Python versions and other build tools in one go, thanks to piggy backing on the years of development that the Conda ecosystem has seen.\n\nThanks to Pixi\u2019s task system and native multi-platform capabilities, the contributor experience is also enhanced. Daunting tasks like running CMake, installing the correct Rust version or C++ compilers are all hidden away behind a magical: `pixi run foobar`.\n\nAre you interested to see how you could improve your own workflow and learn from what these big open-source projects are doing? Then you should join this talk! You'll be amazed by what is possible these days.", "description": "Pixi is getting widely adopted in the Scientific Python community. Projects such as [Python](https://github.com/python/cpython/tree/main/Tools/pixi-packages) itself, [NumPy](https://github.com/numpy/numpy/tree/main/pixi-packages), [SciPy](https://github.com/scipy/scipy/blob/main/pixi.toml), [cuda-python](https://github.com/NVIDIA/cuda-python) and [Xarray](https://github.com/pydata/xarray) have a `pixi.toml` file in their repository. Through the heroic work of Lucas Colley and other contributors, even CPython has a pixi.toml now. In this talk we want to explain what this means and what improvements this brings for users and contributors!\n\nPixi helps for the following reasons:\n\nPrimarily Pixi creates one or more environments on the developer machine containing Conda and Python packages (under the hood, uv is used to resolve and install Python packages). All packages are added to a lockfile that is used to recreate environments in a reproducible way. Pixi can bootstrap the entire development environment in seconds, including a consistent set of compilers, shared libraries, and other low-level pieces.\n\nPixi\u2019s task system makes it easy for contributors (old and new) to get started. Developers can add tasks such as lint, build, start, \u2026 to the pixi.toml file. This simplifies the commands that need to be remembered when starting out with a project. It makes it also easy to have \u201cportable CI\u201d. Pixi can run these tasks on Github, Gitlab, CircleCI on any operating system.\n\nAdvanced use cases:\n\nThe `pixi.toml` files in the CPython project are mainly used for advanced tasks such as building CPython itself with address sanitization turned on. Thanks to Pixi, downstream projects (Numpy, SciPy, \u2026) can depend on CPython from source. This is enabled by the powerful `pixi build`. Pixi build brings building projects from source into packages to Pixi itself. Usually, package consumers and builders are quite disjoint in the Conda ecosystem! With Pixi you can now run crazy things like `pixi global install --git https://github.com/python/cpython --subdir Tools/pixi-packages/asan python`  to obtain the latest version of Python built from main installed globally on your system.\n\nOur talk will also cover the following topics:\n\n- What is Pixi and the conda-ecosystem?\n- How do these big open source projects use Pixi?\n- What steps can one take to benefit from Pixi in their workflow?\n\nPixi itself is open source under the BSD3 Clause, written in Rust and embeds astral-sh's uv to help with combining conda and Python packages into one virtual environment. Pixi is built on the rattler base library that is used in all sorts of different conda tools and is also making it's way into conda and conda-build.\n\nSome of the previously mentioned projects started to use Pixi because of one specific feature: cross-platform source building of Git packages into a local development environment. This experience is similar to depending on a package from source in a python environment but Pixi also takes care of all the complex compiler and low level system libraries that a user might require to have on their system. This feature has proven very useful for testing the latest (pre-release) versions of projects in their upstream environments. \n\nThese workflows come with a few key steps:\n\n- Building packages from source code, from git or paths\n- Installing virtual environments on any platform, Windows, macOS, Linux\n- Reproducible environments with lockfiles\n- Cross-platform Makefile-like task system with Pixi tasks\n- Deployment with easy to share artifacts\n\nRelevant links:\n\n- Pixi repository: https://github.com/prefix-dev/pixi/\n- Pixi documentation: https://pixi.prefix.dev/latest/\n- Rattler repository: https://github.com/conda/rattler\n- SciPy 2025 talk: https://www.youtube.com/watch?v=UeyMkK5MzcA&t=5s\n- SciPy 2025 workshop: https://www.youtube.com/watch?v=8AYp3MlRSNA\n- EuroPython 2025 talk: https://www.youtube.com/watch?v=HOqv3kh4z_c", "recording_license": "", "do_not_record": false, "persons": [{"code": "EKHRSA", "name": "Ruben Arts", "avatar": "https://pretalx.com/media/avatars/EKHRSA_R8Mr2L8.webp", "biography": "Ruben is part of the Prefix.dev core team, builing Pixi and other tools in the package management space. Originally he's a Robotics engineer working on industrial robots, but quickly figuring out that solving development and deployment problems were one of the bigger issues that robotics developers had to deal with. Joining Prefix.dev allowed him to focus on improving the UX/DX of a large group of software engineers. Over the years he's been doing multiple talks and workshops on how to properly manage software and development workflows.", "public_name": "Ruben Arts", "guid": "116518af-d223-54bc-b652-fa3ad14a6897", "url": "https://pretalx.com/scipy-2026/speaker/EKHRSA/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/MEKP9F/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/MEKP9F/", "attachments": []}, {"guid": "57a74241-4e66-5a79-a6f6-f13e0afa6a53", "code": "JMFW8M", "id": 103185, "logo": null, "date": "2026-07-16T12:15:00-05:00", "start": "12:15", "end": "2026-07-16T13:05:00-05:00", "duration": "00:50", "room": "Memorial Hall", "slug": "scipy-2026-103185-instro-an-open-source-python-library-for-interfacing-with-hardware-test-equipment-in-heritage-gallery", "url": "https://pretalx.com/scipy-2026/talk/JMFW8M/", "title": "Instro: An open-source Python library for interfacing with hardware test equipment (in Heritage Gallery)", "subtitle": "", "track": "Lunch and Learn", "type": "Lunch and Learn", "language": "en", "abstract": "Instro is an open-source Python library that puts one typed API in front of power supplies, DAQs, multimeters, oscilloscopes, and more. Write your test once, swap the driver, and your code stays put. We drive a power supply live using the built-in simulator, no hardware required, and show how to add your own.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "LFWXCF", "name": "John Hoehner", "avatar": "https://pretalx.com/media/avatars/MFUGJZ_ziEWwLI.webp", "biography": "John Hoehner is an Instrumentation Engineer at Nominal.io.", "public_name": "John Hoehner", "guid": "557b4565-ade0-5a68-b79c-a3d2a16bd3dc", "url": "https://pretalx.com/scipy-2026/speaker/LFWXCF/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/JMFW8M/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/JMFW8M/", "attachments": []}, {"guid": "cec62aea-c46c-5fe2-ad5b-e49601f48fd3", "code": "VLD7LX", "id": 92516, "logo": null, "date": "2026-07-16T13:15:00-05:00", "start": "13:15", "end": "2026-07-16T13:45:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-92516-just-throw-it-away-class-imbalance-lessons-from-molecular-machine-learning-to-meatballs", "url": "https://pretalx.com/scipy-2026/talk/VLD7LX/", "title": "Just throw it away? Class imbalance lessons from molecular machine learning to meatballs", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "Imbalanced datasets are common across science and industry: most screened molecules are inactive and most batted balls in baseball result in outs. One standard practice is to downsample the majority class or avoid collecting more of it. But majority-class examples are not interchangeable. Some are closely related to other examples, while others are distinct from any other example in the dataset. Others define the boundary between success and failure.\n\nThis talk asks two practical questions:\n1.\tHow much majority-class data is actually necessary for a performative machine learning model?\n2.\tIf we cannot collect all of it, which majority-class examples should we collect?\n\nUsing three wildly different datasets\u2014antibacterial molecular screening, sandwich taste ratings, and Major League Baseball at-bat outcomes\u2014I compare random downsampling to strategies that retain harder or more diverse majority-class examples, and evaluate the impact on generalization and performance for real-world machine learning models.", "description": "**Motivation**\nThe goal of this talk is pragmatic. Rather than assume that majority-class data is disposable, I measure its value in different domains and discuss how to retain the right subset under budget constraints. I also evaluate whether those choices improve performance where it matters most: generalization and discrimination on a decision boundary.\n\n**Intended Audience**\nThis talk is aimed at:\n* Python data scientists working with imbalanced datasets\n* scikit-learn + other ML package users building applied ML systems\n* Anyone who has wondered whether all that negative data is actually necessary\n\nIt assumes familiarity with basic machine learning concepts (classification, regression, cross-validation), but does not require deep theoretical background. The focus is on applied ML.\n\n**Datasets**\nI explore these questions across three domains.\n\n1) Antibacterial screening:\nThis dataset consists of ~40,000 small molecules experimentally screened for antibacterial activity. Only a small fraction (3%) show measurable activity. Evaluation uses both random splits and scaffold splits, where entire structural families of molecules are held out to test generalization under distribution shift.\n\n2) MLB batted-ball outcomes:\nUsing features such as exit velocity and launch angle, the task is to predict outcomes (out, single, double, home run). The majority of at bats result in outs. Rare but desirable events like home runs occupy a small section of feature space and can have similar features to near-misses.\n\n3) \u201cRoll for Sandwich\u201d ratings:\nThis dataset contains ingredient combinations (bread, meat, cheese, toppings) and a human rating from 0\u201310 from the TikTok series \"Roll For Sandwich\". Roughly half of sandwiches score above 7, while very low scores are rare (only ~11% have scores <3). The space of possible combinations is large and sparsely explored. This provides a regression setting where \u201cnegative\u201d examples are low-rated sandwiches.\n\n**Evaluation**\nAcross all three datasets, I run two main experiments.\n\nFirst, data saturation experiments: hold the minority-class examples fixed, and gradually increase the number of majority-class examples to determine where performance plateaus.\n\nSecond, fixed-budget data selection: vary how the majority-class examples are chosen:\n* Random down-sampling\n* Hard examples near the decision boundary (e.g., inactive molecules structurally similar to actives, near-miss home runs, or sandwich variants that differ by one ingredient)\n* Diversity-oriented selection that maximizes coverage of feature space\n\nEvaluation includes classic ML metrics (e.g., F1 score). We also use matched pairs: pairs of examples that are highly similar in features but differ in outcome. In chemistry, these are matched molecular pairs that differ by a small structural modification yet flip activity. In sandwiches, these are nearly identical ingredient sets with different ratings. In baseball, these are batted balls with similar exit velocity and launch angle but different outcomes. Performance on these pairs measures whether a model captures meaningful decision boundaries rather than broad class separation. I also report top-k metrics (e.g., precision@k) to reflect practical decision-making scenarios.", "recording_license": "", "do_not_record": false, "persons": [{"code": "JCQXWS", "name": "Jackie Valeri", "avatar": "https://pretalx.com/media/avatars/ZJEZXE_t0dfxUi.webp", "biography": "Hi! I am a Senior Data Scientist at Moderna working on machine learning and data science for pre-clinical research initiatives. I love custom algorithm development, iteratively designing libraries of molecules, and working with experimentalists to execute drug discovery campaigns. When I'm not in front of a screen, I love puzzling, baking, skiing, watching baseball, and reading sci fi/fantasy novels.", "public_name": "Jackie Valeri", "guid": "b3c24dc3-05ea-5161-be8b-b74d6d8e5594", "url": "https://pretalx.com/scipy-2026/speaker/JCQXWS/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/VLD7LX/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/VLD7LX/", "attachments": []}, {"guid": "394186f2-0b9b-5e54-9062-8f51d6a197a6", "code": "ECYVWR", "id": 93158, "logo": "https://pretalx.com/media/scipy-2026/submissions/ECYVWR/image_u7IWR0f.webp", "date": "2026-07-16T13:55:00-05:00", "start": "13:55", "end": "2026-07-16T14:25:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-93158-compressing-lstm-networks-for-scalable-retail-demand-forecasting-a-python-based-approach-to-efficient-time-series-prediction", "url": "https://pretalx.com/scipy-2026/talk/ECYVWR/", "title": "Compressing LSTM Networks for Scalable Retail Demand Forecasting: A Python-Based Approach to Efficient Time-Series Prediction", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "Deploying deep learning models for time-series forecasting at retail scale presents a fundamental tension between prediction accuracy and computational cost. This talk presents a Python-based framework combining structured pruning, quantization-aware training, and knowledge distillation to compress LSTM networks for demand forecasting. Using NumPy, TensorFlow/Keras, and scikit-learn, we achieved 47% accuracy improvement over baseline models while reducing model size by 73% and inference costs by 92%. We discuss practical implementation patterns, reproducibility considerations, and how these compression techniques generalize beyond retail to any domain requiring efficient sequential prediction at scale.", "description": "**Background**\nMany teams that use LSTM networks for time-series forecasting hit the same wall: as models get more complex, they become too slow and costly to run in production. In retail, for example, you may need to forecast demand for thousands of products every day. The same challenge shows up in energy, healthcare, logistics, and other fields.\nModel compression , making models smaller while keeping them useful , is well studied for image models (CNNs), but less explored for recurrent models like LSTMs used in time-series work. This talk fills that gap using tools from the Python ecosystem.\n\n**What We Built**\nWe developed a three-step compression pipeline, all in Python:\n\n**Structured Pruning**: We used TensorFlow/Keras and NumPy to find and remove LSTM units that contribute the least. Unlike random pruning, this gives you a truly smaller model , not a sparse one that still takes up memory.\n\n**Quantization:** We converted model weights from 32-bit floats to 8-bit integers using TensorFlow Lite, which cuts memory use and speeds up predictions with minimal loss in quality.\nKnowledge Distillation: We trained a small \"student\" LSTM to learn from the larger \"teacher\" model. The student learns not just the final predictions but also the internal patterns the teacher uses, through custom Keras loss functions.\n\nData processing used pandas and NumPy. We tracked experiments with scikit-learn pipelines and visualized results with Matplotlib.\n\n**Results**\nThe compressed model delivered strong improvements:\n\n47% better accuracy (lower RMSE) than the uncompressed model\n73% smaller model size\n92% lower inference cost (wall-clock time)\n\nAn interesting finding: moderate compression acted like a regularizer, helping the model generalize better. This is consistent with the lottery ticket hypothesis , smaller networks can often outperform larger ones.\n\n**Who Should Attend**\nThis talk is for data scientists, ML engineers, and researchers who deploy deep learning models in production and care about efficiency. You do not need to be a retail expert , the techniques apply to any sequential prediction task.\n\n**What You Will Learn**\nHow to prune, quantize, and distill LSTM models using Python tools you already know\nWhen compression helps vs. hurts forecast quality\nPractical patterns for setting up reproducible compression experiments\nHow to adapt these methods to your own forecasting domain\n\n**Why This Matters for the SciPy Community**\nThis work shows that the standard Python scientific stack (TensorFlow, NumPy, scikit-learn, Matplotlib) is enough to build production-ready model optimization , no special proprietary tools needed. As more teams scale up ML inference, efficient models become essential.\n\nLinks : https://ieeexplore.ieee.org/abstract/document/11380599", "recording_license": "", "do_not_record": false, "persons": [{"code": "XU83Y7", "name": "Ravi Teja Pagidoju", "avatar": "https://pretalx.com/media/avatars/UNNG9J_Jm4lv0E.webp", "biography": "**About My Background:**\nI'm a Senior Software Engineer with 9+ years of experience in software engineering and AI/ML research. I pursued MS in Applied Computer Science and am also pursuing PMBA currently. My research focuses on practical applications of machine learning, optimization techniques, and generative AI models across various domains.\n\n**Published Work:**\nMy recent publications include:\nPlanogram Synthesis using Diffusion Models - Published by Springer (constraint-aware generative models for spatial optimization)\nLSTM Compression Techniques - Accepted at IEEE ICUIS 2025 (neural network optimization for resource-constrained deployment)\nGenerative AI for MES Optimization: LLM-Driven Digital Manufacturing Configuration Recommendation- Published in International Journal of Applied Mathematics (LLM-based optimization for manufacturing systems)\nComparative Analysis of Optimized GCD and Hybrid LLM-GCD Approaches for Retail Shelf Space Allocation - Published in European Journal of Information Technologies and Computer Science (hybrid approaches combining LLMs with classical optimization)\nCost-Performance Analysis of Cloud-Based Retail Point-of-Sale Systems: A Comparative Study of Google Cloud Platform and Microsoft Azure\n**ResearchGate**: https://www.researchgate.net/profile/Ravi-Teja-Pagidoju/research \n\n**Peer review Experience:**\nI have reviewed papers at IEEE Transactions on Industrial Informatics Journal(Q1).\nI have judged multiple hackathons, DECA startup pitches, business intelligence awards,.\nI\u2019m also a mentor at Fuel accelerator (\nhttps://www.fuelaccelerator.com) , an active member in Retail AI Council.\n\n**My Speaking Experience:**\nPresented at Generative AI Expo 2026\nPresented at NWA Tech Fest\nPresented Keynote at SCRS ConferenceSCRS Conference\nRegular knowledge sharing within engineering teams\n\nEmail: Pagidojuraviteja1@gmail.com", "public_name": "Ravi Teja Pagidoju", "guid": "2ba9bbf0-ce38-5928-8f8e-88cecaec762c", "url": "https://pretalx.com/scipy-2026/speaker/XU83Y7/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/ECYVWR/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/ECYVWR/", "attachments": []}, {"guid": "efd698f8-e092-5600-9496-0bce5e3bb0f9", "code": "GSBQXK", "id": 93250, "logo": null, "date": "2026-07-16T14:35:00-05:00", "start": "14:35", "end": "2026-07-16T15:05:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-93250-enabling-agentic-ai-infrastructure-for-scientific-data-ecosystems", "url": "https://pretalx.com/scipy-2026/talk/GSBQXK/", "title": "Enabling Agentic AI Infrastructure for Scientific Data Ecosystems", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "The Atmospheric Radiation Measurement (ARM) User Facility Data Center (ADC) capable of supporting scalable, secure, and reproducible engagement with atmospheric research data is evolving towards AI-ready ecosystem. We will discuss architectural designs utilized in production scientific data setting including open-source technologies to further multi-agent coordination, agentic retrieval-augmented generation (A-RAG), shared contextual memory via vector stores, and model-agnostic inference orchestration within Kubernetes infrastructure. We will go over ARM's foundational stack designed to support agentic AI workflows for data discovery, metadata research, reasoning, and user engagement. Additionally, we will go over architectural decisions, trade-offs, and security measures pertinent to research computing environments with some demonstrations.", "description": "With large language models (LLMs) and agentic AI system becoming more prevalent, scientific data centers are investigating how this can improve data discovery, metadata interpretation/automation, and overall data-researcher interaction. Deploying LLMs alone aren't enough for advancing AI-enabled capabilities in scientific environments. We must think of a cohesive architecture that integrates with our existing research infrastructure and facilitates interoperability, reproducibility, scalability, and governance.\n\nThis talk describes the design and implementation of an agentic AI infrastructure developed within the Atmospheric Radiation Measurement (ARM) User Facility Data Center (ADC) to support AI-enabled workflows across atmospheric science data systems. Rather than developing a single application, the effort provides a foundational stack that standardizes how AI agents engage with data, tools, and users throughout the ARM ecosystem.\n\nThe architecture is organized as a layered system that facilitates modular and interoperable AI services. At its foundation is a centralized inference infrastructure providing model-agnostic access to LLMs deployed on GPU-enabled research systems. The framework introduces an Agentic Retrieval-Augmented Generation (A-RAG) approach tailored for scientific data workflows. Traditionally retrieval-augmented generation improves the accuracy of language models by grounding responses in externally retrieved information. With A-RAG, each specialized agent can retrieve domain-relevant information from ARM data services, metadata catalogs, documentation, and web services, enabling evidence-driven responses that reflect the structure and context of atmospheric research data.\n\nThe framework adopts emerging protocols such as Model Context Protocol (MCP) for structured tool access, Agent-to-Agent (A2A) for coordinated communication among agents, and Agent\u2013User Interaction (AG-UI) protocol that support traceable conversational workflows. These protocols allow conversational interfaces, tools and applications to integrate with the framework while reusing shared services. At the central of these capabilities is shared contextual memory layer implemented through persistent vector stores that hold embeddings of structured scientific artifacts and documentation. Through this contextual layer the agents can operate over a consistent state which in turn supports coherent reasoning across sessions and workflows.\n\nAttendees will learn about architectural patterns for building and developing agentic AI infrastructure, strategies for extending traditional RAG into coordinated multi-agent systems, and practical considerations for deploying open-source LLM tooling in environments that require security, governance, and reproducibility. \n\nIntended audience: Software Engineers, Architects, Maintainers or Practitioners interested in AI and enabling that in scientific platforms.\n\nWhile the implementation is grounded towards atmospheric science domain, the architectural principles presented are broadly applicable to other scientific data repositories, national laboratory computing environments, university research platforms, and open-source projects that aim to create interoperable and trustworthy AI-enabled workflows. Towards the end of presentation will have a demonstration illustrating how these architectural components enable coordinated AI agents to facilitate scientific data exploration in a production setting such ADC.", "recording_license": "", "do_not_record": false, "persons": [{"code": "UKKTZ3", "name": "Chirag Shah", "avatar": "https://pretalx.com/media/avatars/33SY3V_XK3x8Ov.webp", "biography": "Chirag Shah is an Environmental Data Science Engineer and full-stack software developer working with the U.S. Department of Energy\u2019s Atmospheric Radiation Measurement (ARM) User Facility Data Center. His work focuses on building scalable scientific data systems that improve the discoverability, accessibility, and usability of large-scale atmospheric and environmental observations.\n\nAt the ARM Data Center, Chirag leads the design and development of modern research software platforms used by scientists to explore, analyze, and interact with complex observational datasets.\nChirag's technical interests span scientific data management, distributed systems, artificial intelligence, machine learning, and advanced data visualization. His work emphasizes building robust infrastructure and user-centric tools that enable researchers to efficiently work with large observational datasets and accelerate scientific discovery in Earth and environmental systems research.\n\nCommitted to advancing modern research software practices, Chirag actively explores emerging technologies that enhance the way scientists interact with complex data ecosystems.", "public_name": "Chirag Shah", "guid": "5852db6c-5e67-5c4f-ad9e-3b8235bedee4", "url": "https://pretalx.com/scipy-2026/speaker/UKKTZ3/"}, {"code": "XYXVGK", "name": "Utkarsh Mahai", "avatar": "https://pretalx.com/media/avatars/NWJ7NN_fwqfkHX.webp", "biography": "Utkarsh Mahai is a full-stack software engineer at the Department of Energy's Atmospheric Radiation Measurement (ARM) User Facility Data Center. He works on building software, tools, and applications that help scientists and researchers access data, streamline workflows, and focus more on advancing their science. \n\nHis work spans the full software development lifecycle, from designing user experiences and building web applications to developing backend services and integrating emerging technologies where they can provide meaningful value. More recently, he has been involved in building agentic systems, modernizing user interfaces in the age of AI, and improving the ways information and context flow through applications.\n\nOutside of work, Utkarsh is interested in conversations around AI ethics, governance, and the broader impact of emerging technologies. He enjoys continuous learning and exploring new ideas, tools, and approaches to solving problems.\n\nBefore joining the ARM Data Center, he worked on software products and business processes in the financial technology (fintech) and entertainment industries.", "public_name": "Utkarsh Mahai", "guid": "743ab684-56c4-5022-8999-d26c7e84b6b5", "url": "https://pretalx.com/scipy-2026/speaker/XYXVGK/"}, {"code": "KBF8UJ", "name": "Austin Aguilar", "avatar": "https://pretalx.com/media/avatars/PQQTSG_XrSPaWh.webp", "biography": "Austin Aguilar is a Software Engineer for the Atmospheric Radiation Measurement (ARM) Data Center at Oak Ridge National Laboratory. Austin helps develop and maintains a variety of user facing applications that allow researchers to focus on their science rather than data discovery. Austin is also very passionate about researching and developing agentic AI workflows to be leveraged within the scope of science.", "public_name": "Austin Aguilar", "guid": "dfeb51a4-0ce6-5006-8058-dd5ae951db9f", "url": "https://pretalx.com/scipy-2026/speaker/KBF8UJ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/GSBQXK/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/GSBQXK/", "attachments": []}, {"guid": "2db4a97e-06c2-517d-821b-185e26cc5952", "code": "JVKEV7", "id": 97788, "logo": null, "date": "2026-07-16T15:30:00-05:00", "start": "15:30", "end": "2026-07-16T16:30:00-05:00", "duration": "01:00", "room": "Memorial Hall", "slug": "scipy-2026-97788-lightning-talk", "url": "https://pretalx.com/scipy-2026/talk/JVKEV7/", "title": "Lightning Talk", "subtitle": "", "track": "Lightning Talks", "type": "Lightning Talk", "language": "en", "abstract": "Lightning talks are 5-minute talks on any topic of interest for the SciPy community. We encourage spontaneous and prepared talks from everyone, but we can\u2019t guarantee spots. Sign ups are at the NumFOCUS booth during the conference.", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/JVKEV7/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/JVKEV7/", "attachments": []}, {"guid": "84224f20-6fc7-51a2-81c7-f82727b10fd4", "code": "WMAQPQ", "id": 102463, "logo": null, "date": "2026-07-16T16:40:00-05:00", "start": "16:40", "end": "2026-07-16T17:35:00-05:00", "duration": "00:55", "room": "Memorial Hall", "slug": "scipy-2026-102463-scientific-python-ecosystem-coordination-maintainer-support-in-heritage-gallery-room", "url": "https://pretalx.com/scipy-2026/talk/WMAQPQ/", "title": "Scientific Python: Ecosystem Coordination & Maintainer Support (in Heritage Gallery Room)", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "\"The Scientific Python project aims to support maintainers and grow the maintainer community.\nWe do so via, e.g., the Scientific Python Ecosystem Coordination process (https://scientific-python.org/specs/), by building tools (https://tools.scientific-python.org/: `spin`, `lazy-loader`, web theme, etc.), and by hosting annual developer summits. When an impactful opportunity presents itself, we take on bespoke technical initiatives such as the SciPy Sparse Array API refactor, or maintaining the myst documentation engine.", "description": "In this BoF, we want to connect with the community to:\n\n- Learn about maintainer needs\n- Explore ecosystem-wide ideas that can be captured as SPECs\n- Connect with maintainers who are interested in participating\n- Discuss domain stacks: groups of field-specific packages\n\nPlease join us to share your ideas for improving the ecosystem!", "recording_license": "", "do_not_record": false, "persons": [{"code": "SRNWW9", "name": "St\u00e9fan van der Walt", "avatar": null, "biography": null, "public_name": "St\u00e9fan van der Walt", "guid": "a87494bf-4c60-5a4d-bffb-1b3a05f753eb", "url": "https://pretalx.com/scipy-2026/speaker/SRNWW9/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/WMAQPQ/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/WMAQPQ/", "attachments": []}, {"guid": "4e2c113c-284a-5b41-abc3-56ec4bf99f4a", "code": "RKS93M", "id": 102435, "logo": null, "date": "2026-07-16T17:45:00-05:00", "start": "17:45", "end": "2026-07-16T18:40:00-05:00", "duration": "00:55", "room": "Memorial Hall", "slug": "scipy-2026-102435-securing-the-scientific-python-supply-chain-in-heritage-gallery-room", "url": "https://pretalx.com/scipy-2026/talk/RKS93M/", "title": "Securing the Scientific Python Supply Chain  (in Heritage Gallery Room)", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "Supply chain attacks on Python, including recent compromises of popular packages and CI workflows, have exposed structural weaknesses in the scientific Python ecosystem. This BoF will bring together library maintainers, downstream users, and security practitioners to discuss practical strategies for securing scientific Python stacks, from core packages (NumPy/SciPy) to domain libraries and analysis workflows. We will share current efforts (e.g., SPEC 8, Trusted Publishing, SBOM generation, GitHub Actions hardening), identify pain points and gaps, and brainstorm actionable steps the community can take over the next year to make scientific Python releases more trustworthy by default. Join us to share your experiences, challenges, and ideas on fortifying our open-source projects against potential threats and ensuring the integrity of scientific research.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "9HRXKH", "name": "Juanita Gomez", "avatar": null, "biography": null, "public_name": "Juanita Gomez", "guid": "b72485f4-a6d1-57ca-bcf3-d31178cc4ae6", "url": "https://pretalx.com/scipy-2026/speaker/9HRXKH/"}, {"code": "EKWFU8", "name": "Jarrod Millman", "avatar": null, "biography": "Jarrod Millman is the Executive Director for Berkeley's Open Source Program Office (OSPO). With a background in computer science, mathematics, and statistics, and degrees from Cornell and Berkeley, Millman is a founding member of the scientific Python ecosystem. His primary focus is on developing and sustaining open-source, community-owned scientific software tools. Millman serves on the steering council of NetworkX, is a core developer of scikit-image, and was an early contributor to NumPy, SciPy, and scikit-learn. He has co-founded several influential initiatives to advance open and reproducible research, including the Scientific Python project, the nonprofit NumFOCUS, and the Neuroimaging in Python project.", "public_name": "Jarrod Millman", "guid": "24565714-bc69-52cc-91e3-cbf205d3c36e", "url": "https://pretalx.com/scipy-2026/speaker/EKWFU8/"}, {"code": "H8ZFYG", "name": "Matthew Feickert", "avatar": "https://pretalx.com/media/avatars/H8ZFYG_0B3tdYV.webp", "biography": "Matthew is a research scientist in experimental high energy physics and data science at the University of Wisconsin-Madison Data Science Institute (a \u201cdata physicist\u201d). He works as a member of the ATLAS collaboration on searches for physics beyond the standard model with experiments performed at CERN's Large Hadron Collider (LHC) in Geneva, Switzerland. He also serves on the executive board of the Institute for Research and Innovation in Software for High Energy Physics (IRIS-HEP) where he is a researcher and the Analysis Systems Area lead. He is also a topical editor for physics and data science for the Journal of Open Source Software. He previously did his Ph.D. (2019) research at Southern Methodist University, also on the ATLAS experiment, and was a postdoc at the University of Illinois at Urbana-Champaign, and the University of Wisconsin-Madison.", "public_name": "Matthew Feickert", "guid": "24bbffd4-66b8-5c0f-8c5f-c38640f4d575", "url": "https://pretalx.com/scipy-2026/speaker/H8ZFYG/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/RKS93M/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/RKS93M/", "attachments": []}], "Johnson Great Room": [{"guid": "24611bef-2731-52d7-b8d5-2c6976489544", "code": "UHUVMM", "id": 92504, "logo": null, "date": "2026-07-16T10:45:00-05:00", "start": "10:45", "end": "2026-07-16T11:15:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-92504-agents-for-correct-transparent-and-reproducible-data-analysis", "url": "https://pretalx.com/scipy-2026/talk/UHUVMM/", "title": "Agents for Correct, Transparent, and Reproducible Data Analysis", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "How do we build competent data analysis agents? Data analysis requires a willingness to pause, question conclusions, and dig into subtleties. Frontier LLMs, however, are optimized to push tasks toward completion, not to slow down when something seems off. This tendency works well for coding agents, where success is often verifiable. But for data analysis, verification is more complicated, and autonomous work by the agent can be at odds with the spirit of the discipline. Drawing on our experience building data analysis agents, we'll share evaluations that expose where LLM-driven analysis goes wrong and design patterns that keep analyses correct, transparent, and reproducible.", "description": "LLM-powered agents are increasingly used for software development and data analysis. However, LLMs are non-deterministic, have uneven competencies, and can lack important context for realistic tasks. For software development, models can typically leverage tight feedback loops. It is often clear if code accomplishes its goal, and the model can also write both code and tests for that code, using the test results to iterate on its work. For data analysis, however, it\u2019s often less clear if the model has done the task well or provided a correct result. \n\nHow, then, do we make competent data analysis agents? In this talk, we will discuss strategies for creating data analysis agents that produce correct, transparent, and reproducible results. We will use examples from Posit Assistant, Posit\u2019s general-purpose coding and data analysis agent. The intended audience includes scientists or data practitioners interested in using AI in data analysis workflows. \n\nFirst, we will discuss the importance of empirical evaluation. Because LLM capabilities can be difficult to predict, we created a series of evaluations, some using the Python library Inspect, to measure the capabilities of the skills we care about. These evaluations help us make decisions about model choice, tool design, and prompting, as well as identify any critical issues in the models\u2019 abilities to carry out data science tasks. As an example, we will discuss bluffbench, an evaluation that measures LLMs\u2019 ability to interpret plots that conflict with their priors. We will also discuss a developmental benchmark that measures agents\u2019 ability to surface subtle data quality issues across long contexts.\n\nSecond, we will discuss design choices to make agent-assisted analyses transparent and reproducible. Data analysis involves a variety of tasks, and different tasks require different levels of human awareness, input, and understanding. For example, exploratory data analysis still typically requires input and understanding from the user by nature of the task. Thus, when doing EDA, our agents produce briefer responses and ask the user for more input. For coding tasks with clear goals, however, we can often trust the agents to act more autonomously. \n\nData analysis agents introduce both risks and opportunities for rigorous data analysis. Our aim for this talk is to introduce practical guidance for evaluating and creating data analysis agents that can be integrated into scientific workflows, while preserving accuracy, transparency, and reproducibility. \n\n Related work:\n\n* Bluffbench and plot interpretation: [Bluffbench repo](https://simonpcouch.github.io/bluffbench/), [Introducing bluffbench](https://posit.co/blog/introducing-bluffbench/), and [How well do LLMs interpret plots?](https://posit.co/blog/llm-plot-interpretation/)\n* [Introducing Databot](https://posit.co/blog/introducing-databot/) and [Databot is not a flotation device](https://posit.co/blog/databot-is-not-a-flotation-device/). Posit Assistant will be released in March and so does not yet have public documentation. \n* [Next edit suggestions (code completion) evaluations](https://github.com/posit-dev/nesevals)\n* Evidence of public speaking ability: \n    * [Is that LLM feature any good? Simon Couch @ posit::conf(2025)](https://www.youtube.com/watch?v=HciRoc9TzMc)\n    * [Getting Started with LLM APIs in R. Sara Altman @ RPharma 2025](https://www.youtube.com/watch?v=1efPTy4TQ4Q)", "recording_license": "", "do_not_record": false, "persons": [{"code": "8MZMVC", "name": "Sara Altman", "avatar": "https://pretalx.com/media/avatars/QSAA33_QVd6y0c.webp", "biography": "Sara Altman is a developer advocate on the AI Core team at Posit, where she focuses on how AI can be effectively and thoughtfully used for data science. She co-authors the Posit AI newsletter with Simon Couch. Previously, she helped build Posit Academy and taught data science and R at Stanford.", "public_name": "Sara Altman", "guid": "81784e4e-2ead-5f26-b33b-44532ccc025c", "url": "https://pretalx.com/scipy-2026/speaker/8MZMVC/"}, {"code": "MWX3YZ", "name": "Simon Couch", "avatar": "https://pretalx.com/media/avatars/ZUNZVF_1NTqiRR.webp", "biography": "Simon Couch builds tools that make the work of data science more joyful and effective. As an engineer on the AI Core Team at Posit, his work spans coding assistants, model evaluations, and next-edit-suggestion systems. Drawing on his background in statistics, Simon spent several years authoring and maintaining core packages in the open-source tidymodels framework\u2014like stacks, broom, and infer\u2014before shifting his focus to LLMs. He blogs about his work at simonpcouch.com.", "public_name": "Simon Couch", "guid": "ef4f9c0d-32eb-527f-a974-6d204cb56ff3", "url": "https://pretalx.com/scipy-2026/speaker/MWX3YZ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/UHUVMM/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/UHUVMM/", "attachments": []}, {"guid": "e2bd6a0b-524c-5d1d-b8e2-c30c588ca10b", "code": "3GRQ87", "id": 92515, "logo": null, "date": "2026-07-16T11:25:00-05:00", "start": "11:25", "end": "2026-07-16T11:55:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-92515-vibes-meet-rigor-evaluating-and-improving-ai-performance-on-complex-scientific-code", "url": "https://pretalx.com/scipy-2026/talk/3GRQ87/", "title": "Vibes, meet rigor: Evaluating and improving AI performance on complex scientific code", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "Scientists apply rigorous methods to their research, but rarely to the AI tools they use to write code. We tested different LLM models in combination with domain-specific tools (including MCP servers and skills) to find the optimal combination for writing complex domain-specific code. We created a quantitative proficiency test for Starsim, a disease modeling framework, and evaluated different combinations of models and tools. While Claude Opus outperformed other models, access to tools improved performance more than choosing the best model. Thus, to improve LLM performance on domain-specific problems, we recommend developing a set of tools with the help of quantitative evaluation.", "description": "**Background**  \nScientists are often decidedly unscientific about choosing AI tools to help them write code. They know these tools are helpful, but except for trying out different models, they rarely perform controlled evaluations to check whether other changes to their AI workflow produce significantly better results. This is because writing and executing these evaluations is typically time-consuming, the results of the evaluations are difficult to interpret and quantify, and AI workflows and tooling are evolving rapidly. Here we describe our process for quantitatively testing our assumptions about how to build a good AI assistant.\n\n**Methods**  \nOur team models infectious diseases using [Starsim](https://starsim.org/), a high-performance agent-based modeling library built on NumPy, SciPy, and Numba. Specifically, Starsim includes modules for different diseases, transmission networks, and interventions (such as vaccines). Starsim has been used to model domains ranging from family planning and primary health care to HIV and tuberculosis. Since the diseases themselves are often very complicated, the Starsim models built to model them can also be very complicated. This presents a challenge to AI tools due to limited context windows and out-of-date information.\n\nWe created a \"Starsim exam\" [evaluation suite](https://github.com/starsimhub/scipy2026_starsim_ai/tree/main/problems) based on Starsim\u2019s online documentation. This benchmark is administered using [Inspect.ai](http://Inspect.ai) and follows the structure of the [SciCode](https://arxiv.org/abs/2407.13168) benchmark with a modular approach to question building and evaluation via unit tests.\n\nNext, we created a set of agent tools to improve domain-specific performance, called [Starsim-AI](https://github.com/starsimhub/starsim_ai). Specifically, we added MCP servers for Starsim and [Sciris](https://docs.sciris.org/en/latest/) (a scientific Python library used widely in the codebase). We also created a set of \"skills\" for Starsim, which consist of problem-solving and feature-oriented Markdown files covering topics including statistical distributions, simulation construction, and calibration. These skills were created by Claude Code based on the Starsim [tutorials](https://docs.starsim.org/tutorials) and [user guide](https://docs.starsim.org/user_guide). They were then manually reviewed and revised by Starsim core developers for accuracy and completeness.\n\nFinally, we ran the evaluation suite using two Anthropic models (Claude Sonnet 4.6 and Claude Opus 4.6), both with and without access to the Starsim-AI tools, and two OpenAI models (GPT-5.2 and GPT-5 mini, which did not have access to the tools).\n\n**Results**  \nPerformance on the evaluation varied widely among the no-tool models, from 17% with GPT-5 mini to 70% with Claude Opus 4.6. Adding the full skillset in agent mode increased performance to 78% for Sonnet 4.6 and 91% for Opus 4.6. When given unlimited solving time, adding skills reduced task completion time by up to 20%. Conversely, when given limited solving time (2 minutes), Starsim-AI increased Opus 4.6's performance from 13% to 65%. Across models, task performance was strongly correlated with token usage (R\u00b2=0.61), but adding skills only marginally increased token usage (1-3%).\n\n**Conclusions**  \nFor our domain-specific problem, providing custom skills and MCP servers reduced the error rate by a factor of three (from 30% to 9%) and reduced task completion time by 20%. We recommend creating a structured problem set for use with a quantitative evaluation tool, as this can help develop the set of domain-specific tools that most effectively improves LLM performance.", "recording_license": "", "do_not_record": false, "persons": [{"code": "SMYXLT", "name": "Cliff Kerr", "avatar": "https://pretalx.com/media/avatars/KFUMRH_6mlSu1I.webp", "biography": "Dr. Cliff Kerr is a Senior Software Engineer at the Institute for Disease Modeling, part of the Gates Foundation, where he works on HIV, STIs, tuberculosis, and family planning. Previously, he completed a B.Sc. in neuroscience and a Ph.D. in physics, was a lecturer in scientific computing at the University of Sydney, co-founded two startups (on data analytics and health economics), worked on a DARPA project teaching robots to pick up balls, and developed an algorithm that composes music in real time based on brain activity recordings. He lives in New York.", "public_name": "Cliff Kerr", "guid": "ee2e318d-385f-521b-be91-d0fad5ba1b0b", "url": "https://pretalx.com/scipy-2026/speaker/SMYXLT/"}], "links": [{"title": "SciPy 2026 submission", "url": "https://github.com/starsimhub/scipy2026_starsim_ai", "type": "related"}, {"title": "Starsim-AI", "url": "https://github.com/starsimhub/starsim_ai", "type": "related"}, {"title": "Starsim", "url": "https://starsim.org", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/3GRQ87/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/3GRQ87/", "attachments": []}, {"guid": "1ab29dee-db2e-5300-a27e-d7ea7e250e35", "code": "GT9Y9U", "id": 92240, "logo": null, "date": "2026-07-16T13:15:00-05:00", "start": "13:15", "end": "2026-07-16T13:45:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-92240-when-vectorized-arrays-aren-t-enough-array-optimization-from-bytecode-to-assembly", "url": "https://pretalx.com/scipy-2026/talk/GT9Y9U/", "title": "When Vectorized Arrays Aren't Enough: Array Optimization from Bytecode to Assembly", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "Few of us come to scientific computing with an understanding of how to write a system kernel or build a transistor. But we often downplay the benefits of going just one or two layers of abstraction below our comfort zone, strengthening our foundations and expanding our options. \n\nThis talk explores the meaning, utility, and optimization of vectorized array operations, fundamental to NumPy, from Python bytecode down to x86 assembly. We'll build a physical intuition for how array operations work, when they turn out to be less performant than we might expect, and how to find the right balance between effort and performance for your needs.", "description": "**Intended Audience**\nScientific Python developers who use NumPy arrays habitually, but find themselves concerned that the efficiency of their code is impacted by implementation details further down the stack of abstractions. \n\n**What We'll Cover**\n - How NumPy's array operations are implemented compared to lists\n - Pitfalls of naive NumPy use\n - Other options for numerical array operations in Python\n - Writing bespoke extensions in Rust\n - x86 assembly in a nutshell\n - What is 'vectorization' really?", "recording_license": "", "do_not_record": false, "persons": [{"code": "SAQZWN", "name": "Nicolas R Posner", "avatar": "https://pretalx.com/media/avatars/FUFHPG_FR3p3MJ.webp", "biography": "I am a data scientist and software engineer specializing in Python+Rust integration. I currently work at the Rose Center for Earth and Space optimizing astrophysical simulations and writing high-performance library code.", "public_name": "Nicolas R Posner", "guid": "d4561986-72d0-5592-8526-48c5156d380f", "url": "https://pretalx.com/scipy-2026/speaker/SAQZWN/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/GT9Y9U/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/GT9Y9U/", "attachments": []}, {"guid": "f5fee67a-d194-5e26-a6f3-daf6f28ae40b", "code": "AVLJ7K", "id": 92306, "logo": null, "date": "2026-07-16T13:55:00-05:00", "start": "13:55", "end": "2026-07-16T14:25:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-92306-bridging-data-discovery-and-analysis-using-web-components-and-jupyterlite", "url": "https://pretalx.com/scipy-2026/talk/AVLJ7K/", "title": "Bridging data discovery and analysis using web components and JupyterLite", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "JupyterLite takes the simplicity of the Jupyter notebook interface and hosts it entirely in the browser, eliminating the need to setup a JupyterHub, making the Jupyter notebook environment much more accessible to users.  In this talk we'll explore how our team harnessed JupyterLite and web components to make the distance between browsing for data and coding against that data 10 seconds and a new tab.", "description": "We work at the Goddard Earth Sciences (GES) Data and Information Services Center (DISC), one of NASA's earth science data archives. Earth science data at NASA is freely available to anyone with an internet connection. NASA offers high quality, curated, validated datasets representing decades of measurements from a variety of remote sensing instruments and models.\n\nUnfortunately, making data available is not the same as making data easy to use. For years, a major sticking point for our users has been transitioning from the in-browser experience of our search engines and visualization tools to compute environments on their own systems. Suddenly users are confronted with data files in weird binary formats with unpredictable metadata, which can be a challenge for a wide range of users, from scientists, students, policy professionals to highly experienced developers.\n\nJupyter is a great tool for making computational workflows more approachable, particularly for new and occasional programmers. Jupyter notebooks provide the perfect vehicle for combining documentation with runnable code examples. Unfortunately, not all users have easy access to their own Jupyter server..\n\nOur team tackled this problem directly, starting with one of our simpler visualization tools, the [Hydrology Time Series Service](https://disc.gsfc.nasa.gov/information/tools?title=Hydrology%20Time%20Series). This tool allows users to plot long time series from hydrology-focused, high temporal resolution data. For some users, the time series plot may be enough for their needs. But if it isn't, we offer a button to jump them directly into a JupyterLite notebook with their selected data loaded into python pandas and ready for further analysis. JupyterLite runs right in their browser, so there's no need for a server or any setup.\n\nWe think this solution is just about the most seamless jump from a pure GUI data exploration environment to a coding environment that we've seen. In this talk, we'll demo the integration and cover what the website is doing behind the scenes to make this jump happen, bringing the user's data along for the ride. Come join us to see how a little [javascript](https://github.com/gesdisc/jupyter-notebook-from-json-extension/) can enable a whole lot of [python](https://gesdisc.github.io/jupyterlite/lab/index.html).", "recording_license": "", "do_not_record": false, "persons": [{"code": "R7U9BC", "name": "Christine Smit", "avatar": "https://pretalx.com/media/avatars/AAMR9G_lKR07GG.webp", "biography": "I'm a principal software engineer at the NASA Goddard Earth Sciences (GES) Data and Information Services Center (DISC). Our prime directive is to archive earth science data and make that data available to the public for free. Since joining the GES DISC, I've mainly focused on the services end of public data access, working on tools that allow users to do some initial data exploration and visualization without having to download, understand, and open raw data files. I'm happy to wax poetic about metadata, interoperability, and well designed colorbars.", "public_name": "Christine Smit", "guid": "56ef7b59-befd-5309-bd93-2aaa85af94d0", "url": "https://pretalx.com/scipy-2026/speaker/R7U9BC/"}, {"code": "FBBS3L", "name": "Jon Carlson", "avatar": null, "biography": null, "public_name": "Jon Carlson", "guid": "5226865d-fd9e-5c0a-b25c-7f7f424ee537", "url": "https://pretalx.com/scipy-2026/speaker/FBBS3L/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/AVLJ7K/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/AVLJ7K/", "attachments": []}, {"guid": "834d12ec-e04e-538a-a2e0-bab00e99e949", "code": "QVKGLW", "id": 92392, "logo": "https://pretalx.com/media/scipy-2026/submissions/QVKGLW/image_PzVesz0.webp", "date": "2026-07-16T14:35:00-05:00", "start": "14:35", "end": "2026-07-16T15:05:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-92392-dagster-slurm-bringing-modern-data-orchestration-to-slurm-managed", "url": "https://pretalx.com/scipy-2026/talk/QVKGLW/", "title": "Dagster-slurm: Bringing Modern Data Orchestration to Slurm-Managed", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "dagster-slurm is an open-source Python integration that allows data scientists and research software engineers to run Dagster pipeline assets on both a laptop and Slurm-managed HPC supercomputers without making any code changes. It automatically handles SSH transport, environment packaging via pixi-pack, and Slurm job submission, while streaming logs and scheduler metrics back to the Dagster UI in real time. The talk covers the full workflow, from local development to staging and production deployment on a real HPC cluster, using a live demo with a self-contained Docker Compose environment. It has been validated on VSC-5 in Austria and CINECA Leonardo in Italy.", "description": "Motivation\nScientific and data engineering pipelines often span multiple compute tiers, such as preprocessing on a workstation, model training on an HPC cluster, and downstream analytics on a cloud VM. However, this is typically managed poorly, with the HPC step being a hand-written sbatch script that is disconnected from the rest of the pipeline. This results in no shared lineage, no unified observability, and no automated trigger of downstream work when the job finishes. As a result, research software engineers have to maintain two separate codebases and two mental models of the same workflow.\n\nWhat dagster-slurm does\n\ndagster-slurm is a Dagster ComputeResource and PipesClient that enables Dagster Software-Defined Assets to run on Slurm HPC clusters. You can redirect a Python function decorated with @dg.asset to a supercomputer by simply setting ExecutionMode.SLURM, without needing to make any other code changes. The library handles tasks such as SSH connection management, automatic environment packaging, Slurm job submission, and log and metadata streaming back to the Dagster UI.\nDagster-slurm is designed for teams that need an orchestrator that handles HPC.  Allowing HPC workloads to be integrated into the same asset graph with full lineage, scheduling, and observability.\n\nTalk structure (25 minutes)\nThe problem: why HPC and data orchestration are still separate (3 min)\nArchitecture overview: ComputeResource, Dagster Pipes over SSH, and pixi-pack (5 min)\nLive demo: running an asset locally and then submitting it to a containerized Slurm cluster with real-time log streaming (10 min)\nLessons from production use: environment portability, air-gapped clusters, and site-specific authentication (4 min)\nRoadmap and how to contribute (3 min)\nThe demo uses a self-contained Docker Compose stack that runs on a laptop, with no need for external cluster connectivity.\n\nAudience and outcomes\nThis talk is intended for research software engineers, data engineers, and ML practitioners who work with Python pipelines and occasionally need HPC resources. Attendees will learn how to connect an existing Dagster project to a Slurm cluster and understand the design tradeoffs between task-level HPC frameworks and asset-based data orchestration.\nLinks: https://github.com/ascii-supply-networks/dagster-slurm | https://dagster-slurm.geoheil.com | JOSS paper (under review)", "recording_license": "", "do_not_record": false, "persons": [{"code": "FAFBTT", "name": "Hernan Picatto", "avatar": "https://pretalx.com/media/avatars/HS37TR_1J4M8l6.webp", "biography": "Hernan Picatto is a Data Engineer at the Supply Chain Intelligence Institute Austria (ASCII) and a PhD student in Informatics at TU Wien. His work bridges the gap between modern data orchestration and High-Performance Computing (HPC), focusing on reproducible workflows for web-scale NLP. Hernan is a core contributor to dagster-slurm and currently manages pipelines that process petabytes of Common Crawl data to reconstruct global supply chain networks. Before returning to academia, he worked as a Senior Software Engineer at JPMorgan Chase and an Algorithm Engineer at ZhiShou Technology in Beijing.", "public_name": "Hernan Picatto", "guid": "caa232f8-9cd8-5405-aa45-5054c0568b49", "url": "https://pretalx.com/scipy-2026/speaker/FAFBTT/"}, {"code": "G3R7A3", "name": "Georg Heiler", "avatar": null, "biography": "Georg is a Senior data expert at Magenta and a ML-ops engineer at ASCII. He is solving challenges with data. His interests include geospatial graphs and time series. Georg transitions the data platform of Magenta to the cloud and is handling large scale multi-modal ML-ops challenges at ASCII.", "public_name": "Georg Heiler", "guid": "98283466-ccc8-5992-92cf-723881babca2", "url": "https://pretalx.com/scipy-2026/speaker/G3R7A3/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/QVKGLW/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/QVKGLW/", "attachments": []}, {"guid": "1c400965-7b3e-56cb-ab02-d363eee89370", "code": "ZNPYX9", "id": 102432, "logo": null, "date": "2026-07-16T16:40:00-05:00", "start": "16:40", "end": "2026-07-16T17:35:00-05:00", "duration": "00:55", "room": "Johnson Great Room", "slug": "scipy-2026-102432-gpu-accelerated-python", "url": "https://pretalx.com/scipy-2026/talk/ZNPYX9/", "title": "GPU-Accelerated Python", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "This Birds of a Feather session will bring together developers, users, researchers, and educators interested in GPU-accelerated Python.  The discussion will explore the current state of the ecosystem, new library developments, and strategies for making GPU acceleration more accessible to a broader scientific audience.  Topics may include performance optimization, debugging and profiling, education and training, and opportunities for collaboration across projects and communities.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "Z9ENP8", "name": "Katrina Riehl", "avatar": "https://pretalx.com/media/avatars/Z9ENP8_ml6Zw2v.webp", "biography": "Dr. Katrina Riehl is a Principal Technical Product Manager at NVIDIA leading the CUDA Education program. For over two decades, Katrina has worked extensively in the fields of scientific computing, machine learning, data science, and visualization. Most notably, she has helped lead data initiatives at the University of Texas Austin Applied Research Laboratory, Anaconda, Apple, Expedia Group, Cloudflare, and Snowflake. She is an active volunteer in the Python open-source scientific software community and currently serves on the Advisory Council for NumFOCUS.", "public_name": "Katrina Riehl", "guid": "885c34b1-3992-5e82-988f-01bce678c58b", "url": "https://pretalx.com/scipy-2026/speaker/Z9ENP8/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/ZNPYX9/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/ZNPYX9/", "attachments": []}, {"guid": "baa1c4af-1e48-5347-bdb6-ca91f2d1285c", "code": "QBWEZ7", "id": 102434, "logo": null, "date": "2026-07-16T17:45:00-05:00", "start": "17:45", "end": "2026-07-16T18:40:00-05:00", "duration": "00:55", "room": "Johnson Great Room", "slug": "scipy-2026-102434-building-scientific-approaches-to-generative-ai", "url": "https://pretalx.com/scipy-2026/talk/QBWEZ7/", "title": "Building Scientific Approaches to Generative AI", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "Generative AI seems like it\u2019s everywhere and attendees of this very conference have built the technical foundations that have enabled its explosive growth. However, unlike the scientific computing software that we typically build, the rapid adoption of generative AI has not been met with the type of rigorous quality control that is required of powerful systems and expected of scientific endeavors. While there are many efforts around assessing the performance of generative AI systems, (e.g., benchmarks, human-in-the-loop AI red teaming, etc.) these methods often lack the rigor and context to make them truly scientific evaluations. In this BoF, we will host a community conversation to discuss the requirements to claim that a generative AI evaluation is scientifically sound while also maintaining subject matter expertise, relevance, and actionability. The SciPy Conference is an excellent forum for this discussion, bringing together scientists, developers, and practitioners.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "LEGKJZ", "name": "Julie Hollek", "avatar": "https://pretalx.com/media/avatars/MZVXN3_4f4q4tM.webp", "biography": "", "public_name": "Julie Hollek", "guid": "ce2160f5-5369-5841-911d-65999c58b85f", "url": "https://pretalx.com/scipy-2026/speaker/LEGKJZ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/QBWEZ7/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/QBWEZ7/", "attachments": []}], "Thomas Swain Room": [{"guid": "50d7d0ff-bd24-5aad-af31-83a5b147fe23", "code": "GBW7DW", "id": 92239, "logo": null, "date": "2026-07-16T10:45:00-05:00", "start": "10:45", "end": "2026-07-16T11:15:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92239-learning-in-the-open-integrating-open-source-contributions-into-the-classroom", "url": "https://pretalx.com/scipy-2026/talk/GBW7DW/", "title": "Learning in the Open: Integrating Open Source Contributions into the Classroom", "subtitle": "", "track": "Scientific Computing in Education", "type": "Talk", "language": "en", "abstract": "In Fall 2025, the UConn School of Mechanical, Aerospace, and Manufacturing Engineering launched Open Source Experiences, an elective course developed in partnership with six NumFOCUS-supported projects (napari, BiocPy, Blosc, MNE-Python, mlpack, JuliaHub). The course embedded students directly into active open source communities, where they contributed to the codebases, collaborated with project maintainers, and learned about community-driven open source software development. In this talk, we will share the lessons learned from piloting this collaboration model, and how these experiences benefit students, open source and open science communities, and educators alike. Attendees will take away actionable insights for integrating open source contributions into their own classrooms and programs.", "description": "The scientific open source community thrives on shared knowledge and welcoming communities, the very system of values that the annual EuroSciPy conference celebrates. At the same time, educators in computational sciences and engineering seek ways to help students move beyond traditional assignments and into experiential learning. Where experiential is a combination of skill-building, networking, and understanding of how science and software happen in the real world. Open Source Experiences was designed to meet both of these needs.\n\nIn this talk, we will share how we worked with students and open source community mentors to structure a semester-long course where students made valuable contributions to existing scientific Python projects. Students participated in issue triage, bug fixes, documentation improvements, and feature contributions, guided by project maintainers. Through this format, students gained experience with tooling (version control, CI/CD, code formatting, testing), in community practices (contributing guidelines, communication norms), and long-term project planning (design decisions, roadmap alignment), while participating projects gained valuable contributions and new contributors.\n\nStudent participation and contributions were assessed with regular progress updates. As instructors, we facilitated discussions to guide the Open Source Experience learning process: working on bugs and issues in an open environment, community expectations, GitHub best practices, etc.\n\n**Talk outline:**\n\n- Course design and goals: balancing academic learning objectives with community needs, assessment strategies.\n- Collaboration with maintainers: selecting projects, preparing onboarding documentation, setting expectations, and creating a mentorship model that respects both students\u2019 learning and maintainer time.\n- Student outcomes: reflections on learning gains around technical skills, professional communication, and confidence engaging in open source ecosystems.\n- Challenges and lessons learned.\n\nWe\u2019ll also share examples of student contributions and how they augmented both the ecosystem and the students\u2019 portfolios.\n\nWhether you\u2019re an educator thinking about how to bring open source into your curriculum or a project leader looking for ways to engage with academic institutions to widen your project\u2019s contributor pipeline, this talk will give you concrete ideas to adapt.", "recording_license": "", "do_not_record": false, "persons": [{"code": "Y3AYSQ", "name": "Inessa Pawson", "avatar": "https://pretalx.com/media/avatars/XNBC37_rao1qHM.webp", "biography": "Inessa is building bridges between people, open source software, and open science. Over the years, she has launched and continues to support several educational initiatives focused on widening the open source contributor pipeline. Inessa is Director of Open Source Program Office at OpenTeams and guest faculty at University of Connecticut. She also serves on the NumPy Steering Council, Scientific Python Ecosystem Coordination Steering Committee, and the pyOpenSci Advisory Council. Inessa is perpetually fascinated by incentive design, collaborative intelligence, and jazz.", "public_name": "Inessa Pawson", "guid": "66a47ef5-755d-58fc-97e5-9c1956817aac", "url": "https://pretalx.com/scipy-2026/speaker/Y3AYSQ/"}, {"code": "7WX39X", "name": "Ryan C Cooper", "avatar": "https://pretalx.com/media/avatars/AHJYAJ_hcOmJWf.webp", "biography": "Ryan C. Cooper is an Associate Professor-in-Residence at the University of Connecticut. His background is in mechanics and materials science with an emphasis on numerical simulations and engineering education. He has been using Jupyter and GitHub to enhance the classroom experience for over six years. Prof. Cooper has developed and free open source materials for computational work in engineering and volunteered with the NumPy documentation team SciPy track chair. Ryan is an integral part of the AI in the School of Engineering committee. He has a Ph.D. from Columbia University and spent two and a half years at Oak Ridge National Laboratory as a Postdoctoral researcher.", "public_name": "Ryan C Cooper", "guid": "52d6720e-6c1c-5399-a4f8-4acc29c8fb8d", "url": "https://pretalx.com/scipy-2026/speaker/7WX39X/"}, {"code": "TVK79T", "name": "Mohammad Mundiwala", "avatar": "https://pretalx.com/media/avatars/TVK79T_0HrOXFW.webp", "biography": "", "public_name": "Mohammad Mundiwala", "guid": "3eeedbf6-6c55-5f8f-93f4-123cdd8a4de7", "url": "https://pretalx.com/scipy-2026/speaker/TVK79T/"}, {"code": "J39VEX", "name": "Ryan Curtin", "avatar": null, "biography": null, "public_name": "Ryan Curtin", "guid": "77d237cd-528d-53e8-9cd3-a439b14ff2c8", "url": "https://pretalx.com/scipy-2026/speaker/J39VEX/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/GBW7DW/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/GBW7DW/", "attachments": []}, {"guid": "bc5b5b2f-c722-58ef-8e6f-fdfe042a7ca4", "code": "8GVHWU", "id": 92373, "logo": null, "date": "2026-07-16T11:25:00-05:00", "start": "11:25", "end": "2026-07-16T11:55:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92373-accessible-python-powered-web-apps-for-the-classroom", "url": "https://pretalx.com/scipy-2026/talk/8GVHWU/", "title": "Accessible Python Powered Web Apps for the Classroom", "subtitle": "", "track": "Scientific Computing in Education", "type": "Talk", "language": "en", "abstract": "Introducing novel software tools into the classroom is increasingly challenging. Fortunately, the richness of the modern web platform, and the proliferation of free static web hosting, provides a low friction way to introduce powerful software into the classroom. Coupled with the maturity of the Pyodide project, the possibilities of introducing scientific Python powered web-apps into classroom are limitless. This presentation demonstrates these possibilities through a case study of the open-source EngineeringPaper.xyz project that gives students instant access to the SymPy library. The unique interactive capabilities enabled by the web platform, such as math expression editing, will also be discussed.", "description": "I've been using Jupyter notebooks for example calculations in my mechanical engineering classes at the University of Minnesota Duluth for many years. However, due to my students' limited coding experience, these Python powered calculations were often impenetrable to my students and provided limited value since I wasn't able to have my students create their own calculations. I simply didn't have the course time available to get them up to speed on Python coding while also covering the core topics of my course, especially with the challenges of getting a working scientific Python stack on the plethora of student computers. The introduction of the Pyodide project, and the power of the modern web platform, have been a game changer in what's possible in terms of bringing powerful Python powered apps into the classroom. This presentation will use my EngineeringPaper.xyz open-source project as a case study in what's possible in terms of bringing intuitive scientific Python powered apps into the classroom. With EngineeringPaper.xyz, I'm now able to have my students create, modify, and submit their own Python powered calculations, using their own devices, with minimal training.\n\nThis talk will cover the technical stack that powers EngineeringPaper.xyz, which includes Pyodide to run the scientific Python stack in the browser and the MathLive interactive math notation editor used to provide the user an intuitive way to enter mathematical expressions. The talk will also cover the parsing strategy used to convert the LaTeX expressions obtained from MathLive into Python expressions that can be interpreted by the SymPy symbolic math library.\n\nIn addition to the technical aspects of EngineeringPaper.xyz, this talk will also discuss key usability features that are used to allow students who are technical, but are not necessarily coders, to take advantage of Python powered scientific computing. These principles include using math notation as a common language and leaning into declarative logic rather than imperative logic in order to minimize confusion and tripping points. Finally, the all-important issue of how students submit their work to a learning management system (LMS), such as Canvas, is addressed. A strategy that uses Markdown as an intermediate format and Pandoc to convert this Markdown into the DOCX or PDF files that can be submitted to the LMS is presented.\n\nEngineeringPaper.xyz is likely more complex than most scientific Python powered apps for the classroom need to be. However, I think the friction points addressed and the overall approach taken can be instructive for the builders of more narrowly focused apps. These app builders can pick and choose from the technologies and approaches used in EngineeringPaper.xyz.\n\nRelevant Links:\n[EngineeringPaper.xyz GitHub Repository](https://github.com/mgreminger/EngineeringPaper.xyz)\n[My Previous SciPy 2021 talk](https://youtu.be/KrlqQBH84x4?si=F7flRHXp1026ViX4)\n[Blog Post Describing EngineeringPaper.xyz's use in the education](https://blog.engineeringpaper.xyz/an-open-source-tool-for-teaching-analytical-calculations-in-engineering-education)", "recording_license": "", "do_not_record": false, "persons": [{"code": "BWMGBF", "name": "Michael Greminger", "avatar": "https://pretalx.com/media/avatars/VU9JKP_RnxCxN4.webp", "biography": "Associate Professor at the University of Minnesota Duluth in the Mechanical and Industrial Engineering Department and developer of the EngineeringPaper.xyz engineering calculation web app.", "public_name": "Michael Greminger", "guid": "2c2fe3b6-68c2-5ee3-8f70-c75ddea811e4", "url": "https://pretalx.com/scipy-2026/speaker/BWMGBF/"}], "links": [{"title": "EngineeringPaper.xyz GitHub Repository", "url": "https://github.com/mgreminger/EngineeringPaper.xyz", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/8GVHWU/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/8GVHWU/", "attachments": []}, {"guid": "1b29bc20-e48d-555c-a26b-71fe2676379a", "code": "GVQECR", "id": 92115, "logo": null, "date": "2026-07-16T13:15:00-05:00", "start": "13:15", "end": "2026-07-16T13:45:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92115-down-the-rabbit-hole-history-of-the-readme-and-why-you-should-care", "url": "https://pretalx.com/scipy-2026/talk/GVQECR/", "title": "Down the Rabbit Hole: History of the README and Why You Should Care", "subtitle": "", "track": "Maintainers and Community", "type": "Talk", "language": "en", "abstract": "When early programmers needed to share code on punch cards and magnetic tape in the 1970s, they needed to explain how to use it, warn about bugs, and provide context. The code on its own wasn't enough, so the README file was born. But READMEs have never been entirely utilitarian forms of documentation. Instead they became (and remain) very human. A 1974 README ends with \"Good luck!\", and in 1978, The Jargon File connected the name itself to Alice in Wonderland, suggesting that \"Read Me\" should stand beside \"Eat Me\" and \"Drink Me\" in a surreal, hidden world.\n\nThis talk reveals how READMEs have always been where developers get to be human. Be it an exasperated warning from the 1970s, a 2009 README that became a complete fairy tale, or today's projects built solely to help developers add jokes to their docs, the pattern holds across five decades: READMEs are where we connect, welcome, and guide each other.\n\nYou'll leave with practical principles for writing READMEs that invite contribution and build community, grounded in this history. If you want contributors to your open source project, your README is likely their first impression and invitation. Make it count.", "description": "Rather than talking about using README files as a mechanism to make it easier for people to contribute, this talk focuses on how README files surface community culture and can make people want to contribute.\n\nThis talk uses historical examples spanning five decades to reveal patterns that remain relevant: READMEs have always been where developers connect with each other, not just with code. That connection makes people want to contribute. If you want people to contribute to your project, the README is likely their first impression. Make it human. Make it welcoming. Make it a door, not a wall. The history of computing shows us that developers have always known this. They've just expressed it in different ways across the decades.", "recording_license": "", "do_not_record": false, "persons": [{"code": "CRHP3W", "name": "Daina Bouquin", "avatar": "https://pretalx.com/media/avatars/PZEMN7_zB176St.webp", "biography": "Daina Bouquin is Senior Developer Relations Engineer at Anaconda with over 12 years of experience spanning astrophysics, library science, and software development. She previously served as Head Librarian at the Harvard-Smithsonian Center for Astrophysics, where she led projects on software citation, preservation, and recovering the contributions of early women in computing. This work gave her deep familiarity with historical computing collections in addition to experience supporting scientists doing computational research. At Anaconda, she creates educational content and strengthens connections between engineering teams and the broader open source community. She believes documentation isn't just about clarity, it's about building communities where people want to participate.", "public_name": "Daina Bouquin", "guid": "c43e0529-8870-5f9d-8e1b-a8b076578c73", "url": "https://pretalx.com/scipy-2026/speaker/CRHP3W/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/GVQECR/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/GVQECR/", "attachments": []}, {"guid": "ba10afd4-eab4-5eea-bc7d-7a662c84acda", "code": "TFEA7N", "id": 92028, "logo": null, "date": "2026-07-16T13:55:00-05:00", "start": "13:55", "end": "2026-07-16T14:25:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92028-building-for-the-road-ahead-transferable-lessons-from-the-front-lines-of-open-source-maintenance", "url": "https://pretalx.com/scipy-2026/talk/TFEA7N/", "title": "Building for the Road Ahead: Transferable Lessons from the Front Lines of Open Source Maintenance", "subtitle": "", "track": "Maintainers and Community", "type": "Talk", "language": "en", "abstract": "Open source maintainers are the lifeblood of the cloud-native ecosystem, balancing the rapid pace of technical innovation with the crucial need for project stability and sustainable community growth. Having served in leadership roles for foundational projects like **XGBoost, KServe, Kubeflow, Argo, and the Kubernetes**, this session moves beyond technical deep-dives to share the hard-won, non-obvious lessons of maintaining and scaling a successful open source project.", "description": "We will explore the maintainer\u2019s journey: from building a neutral foundation for multi-vendor collaboration to managing challenging governance decisions and successfully onboarding new waves of contributors. Attendees will gain a clear, practical framework for:\n\n- **Balancing Control and Collaboration**: Deciding when to extend project primitives versus delegating functionality to the wider ecosystem.\n- **Sustainable Governance**: Creating inclusive contribution pipelines that scale with project maturity.\n- **Community as Innovation Engine**: Using community feedback and cross-project partnerships (e.g., vLLM, Envoy AI Gateway) to drive a roadmap that is both cutting-edge and enterprise-ready.\n\nThis is a session for both current and aspiring maintainers looking for honest stories and actionable, transferable strategies to secure the long-term health and impact of their own open-source projects.\n\n**Key Takeaways:**\n\n- **Actionable Strategies for Community Growth**: Learn proven techniques for converting end-users into contributors and building a diverse maintainer base, leveraging case studies from the Kubeflow and KServe communities.\n- **Maintainer Decision-Making Frameworks**: Gain insight into the process for critical project decisions, such as adopting new standards (like Kubernetes Gateway API) or managing core vs. extension boundaries, that balance stability with innovation.\n- **The Power of Open Collaboration**: Understand the practical benefits and challenges of multi-company/vendor neutral collaboration and how it is essential for tackling complex, shared infrastructure problems like Generative AI model serving.", "recording_license": "", "do_not_record": false, "persons": [{"code": "ECKLLZ", "name": "Yuan Tang", "avatar": "https://pretalx.com/media/avatars/XFTTHZ_OsiD3kW.webp", "biography": "Yuan is a Senior Principal Software Engineer at [Red Hat AI](https://www.redhat.com/en/products/ai). Previously, he has led AI infrastructure and platform teams at [various companies](https://terrytangyuan.github.io/cv#experience). He holds [leadership positions](https://terrytangyuan.github.io/cv#services) in open source communities, including [Argo](https://argoproj.github.io/), [Kubeflow](https://github.com/kubeflow), [KServe](https://github.com/kserve/kserve), [Kubernetes](https://github.com/kubernetes/community/tree/1cd8d239089e777d0e2f70d665e7db153f040a80/sig-list.md), and [CNCF](https://contribute.cncf.io/community/tags/workloads-foundation/). He's also a maintainer and author of many popular [open source projects](https://terrytangyuan.github.io/cv#projects). In addition, Yuan [authored](https://terrytangyuan.github.io/cv#publications) three technical books as well as numerous papers and patents. He's a frequent [conference speaker](https://terrytangyuan.github.io/cv#talks), technical advisor, leader, and mentor at [various organizations](https://terrytangyuan.github.io/cv#services).", "public_name": "Yuan Tang", "guid": "8f76d403-79cf-5b03-9366-27dcc6d4a08d", "url": "https://pretalx.com/scipy-2026/speaker/ECKLLZ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/TFEA7N/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/TFEA7N/", "attachments": []}, {"guid": "5a15bec5-9d93-56a5-bf67-f15b2a973508", "code": "FAKEUM", "id": 91121, "logo": null, "date": "2026-07-16T14:35:00-05:00", "start": "14:35", "end": "2026-07-16T15:05:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-91121-grammars-of-data-lessons-from-20-years-of-the-tidyverse", "url": "https://pretalx.com/scipy-2026/talk/FAKEUM/", "title": "Grammars of Data: lessons from ~20 years of the tidyverse", "subtitle": "", "track": "Maintainers and Community", "type": "Talk", "language": "en", "abstract": "The tidyverse is a collection of R packages designed to facilitate data science. My team and I have been working on it for nearly 20 years, and in this talk, I\u2019ll share some of what we\u2019ve learned about software development and open source community building in that time. \n\nIt\u2019s very clear that AI is having a profound impact on how we develop software and do data science, so I\u2019ll also offer a look into the (near) future, discussing how we\u2019re updating our thinking about how people will do data science, and speculating on what work is likely to have the biggest impact.", "description": "My team and I have spent the last almost 20 years building a collection of R packages known as the [tidyverse](https://tidyverse.org/). The tidyverse includes packages like ggplot2 (for visualisation) and dplyr and tidyr (for data manipulation) and is designed to make data science easier to learn by embracing a consistent design across makes. The overall aim of the tidyverse is make data science faster, more effective, more fun, and more accessible to more people.\n\nThe tidyverse was named and created in 2016, but the core ideas started development in 2006 with ggplot and reshape, predecessors of the core ggplot2 and tidyr packages. We\u2019ve learned a lot about software development and open source community building over those 20 years and I\u2019d love to share some of what we\u2019ve learned with the scipy community.\n\nI\u2019ll also talk about how we\u2019re thinking about coding data science today: it\u2019s clear that AI is having and will continue to have a profound impact the practice of data science. What are the implications for open source tool builders? What does it mean for our identities as programmers and data scientists? It\u2019s hard to speculate too much, but I will discuss the changes we\u2019re seeing (and making!) and offer some very near term predictions.", "recording_license": "", "do_not_record": false, "persons": [{"code": "JHDFAW", "name": "Hadley Wickham", "avatar": "https://pretalx.com/media/avatars/JHDFAW_8eNQbas.webp", "biography": "Hadley is Chief Scientist at Posit PBC, winner of the 2019 COPSS award, and a member of the R Foundation. He builds tools (both computational and cognitive) to make data science easier, faster, and more fun. His work includes packages for data science (like the tidyverse, which includes ggplot2, dplyr, and tidyr)and principled software development (e.g. roxygen2, testthat, and pkgdown). He is also a writer, educator, and speaker promoting the use of R for data science. Learn more on his website, <http://hadley.nz>.", "public_name": "Hadley Wickham", "guid": "a38bfd1c-e60a-55db-baa9-c27ad6b984e7", "url": "https://pretalx.com/scipy-2026/speaker/JHDFAW/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/FAKEUM/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/FAKEUM/", "attachments": []}, {"guid": "f384b06b-b602-5fdb-bbe9-29ca4c68be92", "code": "8WTNKR", "id": 102433, "logo": null, "date": "2026-07-16T16:40:00-05:00", "start": "16:40", "end": "2026-07-16T17:35:00-05:00", "duration": "00:55", "room": "Thomas Swain Room", "slug": "scipy-2026-102433-funding-scientific-open-source-in-the-age-of-ai-new-challenges-and-opportunities", "url": "https://pretalx.com/scipy-2026/talk/8WTNKR/", "title": "Funding Scientific Open Source in the Age of AI: New Challenges and Opportunities", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "Open source software has fueled every major scientific discovery of the last two decades. Yet as scientific practice races toward agentic workflows, no-code interfaces, and AI-driven hypothesis testing, the open source infrastructure (and the maintainer communities who keep it alive) remain systemically underfunded and not yet designed for AI-native use.", "description": "Many foundational tools were built for human-in-the-loop workflows and need modernization to support AI applications or data-intensive model training workflows. At the same time, LLMs and agentic frameworks are increasingly becoming the frontend through which scientists access core capabilities provided by open source libraries, forcing many communities to adapt to use cases that were never part of their original roadmap. AI has also dramatically impacted software engineering practices and the ability for open source projects to vet and incorporate community contributions.\n\nIn May 2026, we launched the Open Source for Science Fund, a new multi-donor initiative designed with the precise goal of sustaining and evolving the open source stack that underpins science in the AI era. The Fund builds on six years of funding through the Chan Zuckerberg Initiative's Essential Open Source Software for Science (EOSS) program, which deployed $58M in funding and supported a significant number of software projects in the scientific Python ecosystem.\n\nWith this BoF, we want to share early insights from the launch of the Fund and engage the SciPy community in identifying opportunities to design funding programs tailored to the evolving needs of scientists and the maintainer communities that support them\"", "recording_license": "", "do_not_record": false, "persons": [{"code": "L7DWZJ", "name": "Dario Taraborelli", "avatar": "https://pretalx.com/media/avatars/R7JKEK_WwU9l01.webp", "biography": "Dario is the founder and director of the [**Open Source for Science Fund**](https://os4science.org), a multi-donor fund by Renaissance Philanthropy launched in 2026 aiming to sustain and evolve the open source stack that powers scientific discovery. He previously led a portfolio of philanthropic investments in open science and open source at the Chan Zuckerberg Initiative, including the [**Essential Open Source Software for Science**](https://czi.co/EOSS) (EOSS) program, which supported for six years multiple libraries and communities in the scientific python ecosystem.", "public_name": "Dario Taraborelli", "guid": "4281ed5d-4d04-50e7-a14c-ee863c8ee33c", "url": "https://pretalx.com/scipy-2026/speaker/L7DWZJ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/8WTNKR/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/8WTNKR/", "attachments": []}], "Virtual Sessions": [{"guid": "acbc7833-27f7-514b-926e-38c8059b360e", "code": "LKAHLP", "id": 102009, "logo": null, "date": "2026-07-16T16:40:00-05:00", "start": "16:40", "end": "2026-07-16T17:35:00-05:00", "duration": "00:55", "room": "Virtual Sessions", "slug": "scipy-2026-102009-virtual-bof-resilient-data-software-science-and-culture", "url": "https://pretalx.com/scipy-2026/talk/LKAHLP/", "title": "Virtual BoF: Resilient data, software, science, and culture", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "The scientific Python community, like the rest of the world, faces a set of interlocking crises. Longstanding questions over how to sustainably develop, fund and maintain open-source scientific software, open data, reproducible research and collaborative training are being magnified by a variety of forces from the AI boom, to government disinvestment, to economic disruption. Combining a short panel with audience discussion and Q&A, this virtual BoF will reflect on these challenges and crowdsource ideas for how the SciPy community (including future conferences) can serve as a vehicle for bolstering scientific open ecosystems.\n\n_This is a virtual Birds of a Feather section. It will take place on the virtual platform for the conference, Airmeet. All attendees will have access to Airmeet. **NOTE: This session will observe Chatham House rules.**_\n\nHybrid committee co-chairs Puneet Kollipara and David Nicholson will host and moderate this panel discussion.", "description": "The scientific Python community, like the rest of the world, faces a set of interlocking crises. In addition to longstanding questions of how to sustainably develop open-source scientific software, we now face attacks on science, data and research infrastructure that would have once been unthinkable. And as if maintainers were not already stretched thin, they must now agree on how to deal with a firehose of pull requests generated using AI.\n\nNew ways of working have sprung up in response to these crises. These point the way toward building *resilience* into data, software, science and culture. The goal of this virtual BoF is to discuss what all these efforts have in common, to strengthen existing connections and share information. Additionally we will discuss proposing a track on resilient data, software, science, and culture for the SciPy 2027 conference. This BoF will be a panel discussion that brings together the SciPy conference community with a broader set of leaders involved in these efforts across scientific disciplines and open source software ecosystems.\n\nPanelists will first introduce themselves and their area of focus. This will be followed by a general discussion and Q&A. \nWe are pleased to welcome these panelists to discuss the following topics:\n- [Jonny Saunders](https://jon-e.net/), post-doctoral researcher, UCLA; [SciOp](https://sciop.net/), [data preservation](https://www.librarypunk.gay/e/160-sciopnet-feat-jonny-and-jez-part-1/); [NeuroMatch](https://neuromatch.io/) and [decentralized infrastructure](https://arxiv.org/pdf/2209.07493)\n- [Brianna (Pag\u00e1n) Corremonte](https://www.briannapagan.com/), technical lead, [Development Seed](https://developmentseed.org/); [\"Beyond Open Data\"](https://cloudnativegeo.org/beyond-open-data-white-paper.pdf) and [Incentivising open science through powerful free and open tooling](https://meetingorganizer.copernicus.org/EGU26/EGU26-20056.html?pdf)\n- [Juan Nunez-Iglesias](https://image.coop/people/juan), co-creator of [napari](https://napari.org/stable/), core [scikit-image](https://scikit-image.org/) team member; [Image Cooperative](https://image.coop/)\n- [Kris Armeni](https://www.kristijanarmeni.net/), research scientist; [civic tech contributor](https://pretalx.com/scipy-2026/talk/AFWXAU/)\n- [Yanina Bellini Saibene](https://yabellini.netlify.app/about/), community manager, rOpenSci; [creation and reinforcing open software communities in Latin America](https://yabellini.netlify.app/talk/2025_latinr-champions/)\n\nThis is a virtual Birds of a Feather section. It will take place on the virtual platform for the conference, Airmeet. All attendees will have access to Airmeet. **NOTE: _This session will observe Chatham House rules._**\n\nHybrid committee co-chairs Puneet Kollipara and David Nicholson will host this panel discussion and moderate.", "recording_license": "", "do_not_record": false, "persons": [{"code": "SLQZZW", "name": "Puneet Kollipara", "avatar": "https://pretalx.com/media/avatars/RLGZFC_3T0GMqv.webp", "biography": "**[Puneet Kollipara](https://www.linkedin.com/in/pkollipara)** is a climate/geospatial data scientist and outreach/communications specialist in Washington, DC. Leveraging open source, open data and open science, he helps clients find, analyze, provide, translate and integrate data for the benefit of people and the planet. \n\nHe works with **[Fulton Ring](https://www.fultonring.com)** and the **[Public Environmental Data Partners](https://publicenvirodata.org)** on **[HIFLD Next](https://hifld.publicenvirodata.org)**. This community-shaped geospatial data hub preserves and improves data layers from HIFLD Open, a public-facing federal aggregator of infrastructure datasets shut down in 2025. \n\nTo support open ecosystems, Puneet mentors aspiring technologists and contributes to open-source projects via communities like the Upskilling Labs and Civic Tech DC. He also helps plan open-source gatherings like **[FedGeoDay](https://fedgeo.us)** and **[FOSS4G North America](https://foss4gna.org)**. He joined the SciPy Conference Hybrid Committee in 2026.\n\nPreviously a science, environmental and public-policy journalist and blogger, Puneet holds a bachelor\u2019s degree in physics and economics from Washington University in St. Louis, MO.", "public_name": "Puneet Kollipara", "guid": "b528c683-056d-5fe8-ad92-d41be81756b4", "url": "https://pretalx.com/scipy-2026/speaker/SLQZZW/"}, {"code": "7GEBXP", "name": "David Nicholson", "avatar": null, "biography": "", "public_name": "David Nicholson", "guid": "f2482af7-a2be-5d28-8c72-1c9917335fce", "url": "https://pretalx.com/scipy-2026/speaker/7GEBXP/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/LKAHLP/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/LKAHLP/", "attachments": []}]}}, {"index": 5, "date": "2026-07-17", "day_start": "2026-07-17T04:00:00-05:00", "day_end": "2026-07-18T03:59:00-05:00", "rooms": {"Memorial Hall": [{"guid": "162af5ef-5d83-5e1d-a9b4-f59b5771ebb2", "code": "BKZVXU", "id": 97808, "logo": null, "date": "2026-07-17T09:15:00-05:00", "start": "09:15", "end": "2026-07-17T10:00:00-05:00", "duration": "00:45", "room": "Memorial Hall", "slug": "scipy-2026-97808-keynote-dr-joseph-h-kennedy-snakes-in-the-microwaves-how-python-is-powering-the-golden-age-of-sar", "url": "https://pretalx.com/scipy-2026/talk/BKZVXU/", "title": "Keynote: Dr. Joseph H. Kennedy, \"Snakes in the Microwaves: How Python is Powering the Golden Age of SAR\"", "subtitle": "", "track": "Keynotes", "type": "Keynote", "language": "en", "abstract": "Staff Scientist at the Alaska Satellite Facility", "description": "Synthetic Aperture Radar (SAR) is transforming how we observe our planet. It sees through clouds, smoke, and darkness, measures millimeter-scale changes to Earth's surface from space, and is rapidly becoming a cornerstone of Earth observation. With a decade of Sentinel-1 observations, the launch of NISAR, and fleets of commercial satellites, we're entering the Golden Age of SAR.\n\nAt the Alaska Satellite Facility, we steward more than 30 PB of freely-accessible SAR data for NASA Earthdata, with the archive expected to exceed 100 PB as calibrated NISAR data becomes available. But turning that flood of data into scientific insight requires far more than storage\u2014it demands an ecosystem of software that enables scientists to discover, access, process, and analyze data at an unprecedented scale.\n\nDrawing on examples from the Alaska Satellite Facility and the broader NASA Earthdata ecosystem, I'll explore how the Scientific Python ecosystem, through tools like NumPy, Xarray, Zarr, Jupyter, and countless community-built libraries, has become the foundation powering everything from cloud-native data access and open-source scientific libraries to large-scale processing platforms and \u201cnear\u201d-real-time Earth monitoring projects like ITS_LIVE. Along the way, we'll see how the Python community has helped transform SAR from a specialized research tool into a global scientific resource, moving beyond individual images toward continuous streams of Earth observations\u2014and why the next decade of Earth observation will be defined as much by open-source software as by the satellites themselves.", "recording_license": "", "do_not_record": false, "persons": [{"code": "BW8BNW", "name": "Joseph H. Kennedy", "avatar": "https://pretalx.com/media/avatars/NKY8UB_pnUXE69.webp", "biography": "Dr. Joseph H. Kennedy is a Staff Scientist for the Alaska Satellite Facility (ASF) and Geophysical Institute at the University of Alaska Fairbanks. He is a computational glaciologist by training but has since transitioned primarly into PB-scale remote sensing/Earth observing data processing and is best know for is work on the ITS_LIVE global glacier velocity project and the development of ASF's on-demand processing system, HyP3.  He specializes in bridging the gap between scientists and software engineers, building ground-up Cloud and HPC data processing/analysis platforms for users, managing global-scale processing campaigns, and generating analysis-ready EarthObserving/Modeling data.", "public_name": "Joseph H. Kennedy", "guid": "998313c6-effb-5ef7-aa52-537bf69220dd", "url": "https://pretalx.com/scipy-2026/speaker/BW8BNW/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/BKZVXU/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/BKZVXU/", "attachments": []}, {"guid": "bd9b24a5-8211-5f14-a9a2-7589c71b4e6d", "code": "REGLJW", "id": 97806, "logo": null, "date": "2026-07-17T10:00:00-05:00", "start": "10:00", "end": "2026-07-17T10:25:00-05:00", "duration": "00:25", "room": "Memorial Hall", "slug": "scipy-2026-97806-scipy-tools-plenary", "url": "https://pretalx.com/scipy-2026/talk/REGLJW/", "title": "SciPy Tools Plenary", "subtitle": "", "track": "SciPy Tools", "type": "Tools Plenary", "language": "en", "abstract": "A session featuring updates and roadmaps from maintainers of core Scientific Python libraries and tools.", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/REGLJW/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/REGLJW/", "attachments": []}, {"guid": "899c31bf-c237-54e2-9a77-a4b6d8797027", "code": "YWHVF7", "id": 93245, "logo": null, "date": "2026-07-17T10:45:00-05:00", "start": "10:45", "end": "2026-07-17T11:15:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-93245-canvas-chat-non-linear-workflows-for-ai-assisted-data-science", "url": "https://pretalx.com/scipy-2026/talk/YWHVF7/", "title": "Canvas Chat - non-linear workflows for AI-assisted data science", "subtitle": "", "track": "Spirit of SciPy", "type": "Talk", "language": "en", "abstract": "Canvas Chat is a browser-based tool that combines Python's data science stack with large language model connectivity, enabling natural language interaction with data. Built on Pyodide, it runs entirely in the browser with no server-side computation required. Users bring their own API keys for LLM access, while all session data persists locally in IndexedDB. The visual, non-linear interface represents conversations as nodes on an infinite canvas, supporting branching, merging, and stateful exploration of data analysis workflows. This talk demonstrates how browser-based Python plus LLMs can democratize data science by removing infrastructure barriers while preserving privacy and reproducibility.", "description": "### Motivation\n\nData scientists face significant infrastructure hurdles when exploring data. Setting up Python environments, managing dependencies, and configuring cloud resources create friction before any actual analysis begins. Meanwhile, large language models have transformed how we think about interacting with code and data, yet most LLM-powered tools require cloud infrastructure and raise privacy concerns.\n\nWhat if we could bring Python's full data science stack into the browser, connect it to LLMs for natural language interaction, and keep everything local and private?\n\n### Canvas Chat - architecture and approach\n\nCanvas Chat addresses these challenges through three key design decisions:\n\n1. **Pyodide for browser-native Python**: The entire Python runtime, including NumPy, pandas, and Matplotlib, runs via WebAssembly in the browser. No installation, no server, no compute costs beyond the client machine.\n\n2. **LLM connectivity with local-first privacy**: Users bring their own API keys (OpenAI, Anthropic, Google, Groq, or local Ollama). Session data, conversation history, and analysis state persist in IndexedDB. Nothing is sent to third-party servers beyond the user's chosen LLM provider.\n\n3. **Visual, non-linear workflows**: Unlike traditional chat interfaces, Canvas Chat represents conversations as a directed acyclic graph (DAG) on an infinite canvas. Users can branch from any point, merge multiple context branches, and explore analysis paths in parallel. This matches how data scientists actually think about problems.\n\n### Key capabilities\n\n- **Natural language to code**: Users describe what they want in plain English, and the LLM generates executable Python code\n- **Stateful sessions**: Unlike stateless notebooks, the canvas maintains full conversation history and data lineage\n- **Multi-modal input**: Images, PDFs, and web content can be incorporated into analysis workflows\n- **Extensible via plugins**: Custom node types allow domain-specific extensions without modifying core code\n- **Zero deployment**: A single `uvx canvas-chat` command launches everything\n\n### What attendees will learn\n\n1. How Pyodide enables full Python data science in the browser\n2. Architectural patterns for connecting browser-based Python to LLMs\n3. Design principles for non-linear, stateful analysis interfaces\n4. Privacy-preserving approaches to AI-assisted data science\n5. How to extend Canvas Chat with custom plugins for their domain\n\n### Relevance to the SciPy community\n\nCanvas Chat represents an unconventional application of the Python scientific stack, repurposing Pyodide for interactive AI-assisted workflows. It addresses a core SciPy value, lowering barriers to scientific computing, while introducing novel interaction patterns that could influence future tool development. The talk will include live demonstrations and practical guidance for attendees who want to experiment with browser-based Python plus LLM workflows.\n\n### Links\n\n- Live demo: https://ericmjl--canvas-chat-fastapi-app.modal.run/\n- Source code: https://github.com/ericmjl/canvas-chat\n- Documentation: https://ericmjl.github.io/canvas-chat/", "recording_license": "", "do_not_record": false, "persons": [{"code": "9NRRJH", "name": "Eric Ma", "avatar": "https://pretalx.com/media/avatars/EP39HL_u5E0WzK.webp", "biography": "As Senior Principal Data Scientist at Moderna Eric leads the Data Science and Artificial Intelligence (Research) team to accelerate science to the speed of thought. Prior to Moderna, he was at the Novartis Institutes for Biomedical Research conducting biomedical data science research with a focus on using Bayesian statistical methods in the service of discovering medicines for patients. Prior to Novartis, he was an Insight Health Data Fellow in the summer of 2017 and defended his doctoral thesis in the Department of Biological Engineering at MIT in the spring of 2017.\n\nEric is also an open-source software developer and has led the development of pyjanitor, a clean API for cleaning data in Python, and nxviz, a visualization package for NetworkX. He is also on the core developer team of NetworkX and PyMC. In addition, he gives back to the community through code contributions, blogging, teaching, and writing.\n\nHis personal life motto is found in the Gospel of Luke 12:48.", "public_name": "Eric Ma", "guid": "0c3ba6a2-c8b9-5f73-9828-d3217a7a15e6", "url": "https://pretalx.com/scipy-2026/speaker/9NRRJH/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/YWHVF7/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/YWHVF7/", "attachments": []}, {"guid": "2d610a27-3f04-5cbc-a08e-a04ff8c3393f", "code": "W9QWGR", "id": 92503, "logo": null, "date": "2026-07-17T11:25:00-05:00", "start": "11:25", "end": "2026-07-17T11:55:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-92503-open-exchange-architecture-from-computational-narrative-to-interactive-preprint", "url": "https://pretalx.com/scipy-2026/talk/W9QWGR/", "title": "Open Exchange Architecture: From computational narrative to interactive preprint", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "Computational narratives like Jupyter, MyST Markdown, R-Markdown, and Quarto are amazing for doing science. You can combine narrative, code, data and images, conducting your analysis while also creating information to share. However, the workflow has been that notebooks are where you do the work, but you need to publish a pdf article to advertise the work, and this is the research output that most people see. That process not only creates extra work, but we're losing key information, amazing graphics, interactive visualizations, and a connection to the code and data.\n\nFlattening science into a published pdf sacrifices reproducibility and valuable context for others to build on the research. We\u2019re continuing to share our science in 19th century ways, as if we need to send printed, physical copies of our work to people in the mail. This is both a boring and ineffective way to communicate science and also reduces the visibility and value to the modular components of research. The data, images, and code all have individual value, especially as we think about new ways for humans and machines to build on existing science for new impact. \n\nThe Open Exchange Infrastructure (OXA, https://oxa.dev) is a community standard for scientific publishing built for modular and computational science. Initial contributors include Stencila, eLife, Posit, PLOS, openRxiv, Curvenote, NeuroLibre, and Creative Commons \u2014 representing a new document format that brings together the best of Jupyter Notebooks, Quarto, MyST Markdown, and publishing/archiving standards to enable new scientific publishing experiences and workflows. OXA additionally allows many tools and existing formats to connect with each other and into traditional publishing workflows, like Journal Article Tag Suite (JATS XML) and Manuscript Exchange Common Approach (MECA). This means that what you share is interactive and engaging and your research products, like large scale microscopy images (e.g. OME-Zarr), are first-class citizens where image datasets, notebooks, and other research products are highlighted not hidden.\n\nIn this talk we\u2019re sharing more on the technical architecture of the format and a pilot between openRxiv (the non-profit organization behind the largest biomedical preprint servers: bioRxiv and medRxiv) and Curvenote (a scientific content management system that also hosts the SciPy Proceedings) to migrate 500k preprints (8.1TB) to OXA and show real-world examples of interactive scientific content, modular attribution, and what\u2019s possible when the pieces are connected and scientific research can be open, engaging and match what\u2019s possible with our current technology - to change the way we share and do science. This isn\u2019t a future vision, this is what is already happening today.", "description": "This talk is for: \n* People who are scientists creating and sharing research, especially using computational notebooks (e.g. Jupyter Notebooks, Jupyter Book, Quarto, MyST Markdown)\n* People developing tools related to scientific communications, that could more easily be connected with each other through OXA\n* People working on formats and standards for computational notebooks and scientific publication\n\nSome relevant previous speaking experience includes: \n- Talk at [SciPy 2023](https://www.youtube.com/watch?v=7nkUcwBgoME) on \"Scientific and technical publishing with Python and Quarto\"\n- Talk at [PyData Seattle 2023](https://www.youtube.com/watch?v=CiXhTA6zkjA) on \"It's not just code: managing an open source project\"\n- Talk at [posit::conf 2022](https://www.youtube.com/watch?v=ttLnLdU1-CQ) on \"These are a few of my favorite things (about Quarto presentations)\"", "recording_license": "", "do_not_record": false, "persons": [{"code": "ASPM3D", "name": "Tracy Teal", "avatar": "https://pretalx.com/media/avatars/DMSP98_rHrgxLe.webp", "biography": "Dr. Tracy K. Teal has been the Open Source Program Director at RStudio/Posit and Nixtla, Executive Director of Dryad, and a co-founder and Executive Director of The Carpentries and is now the CEO at openRxiv. She developed open source bioinformatics software as an assistant professor at Michigan State University and holds a PhD in Computation and Neural Systems from California Institute of Technology. Tracy is involved in the open source software and reproducible research communities, including serving on advisory committees for NumFOCUS, pyOpenSci, R Consortium, and CarbonPlan, and has been working with open source communities, developing curriculum, and teaching people how to work with data and code as a developer, instructor and project leader throughout her career.", "public_name": "Tracy Teal", "guid": "ff1ac42c-72d4-5924-800d-96a9e89ac69e", "url": "https://pretalx.com/scipy-2026/speaker/ASPM3D/"}, {"code": "SZCRM8", "name": "Rowan Cockett", "avatar": null, "biography": null, "public_name": "Rowan Cockett", "guid": "a4ad8b60-7461-5e9e-8d5c-c0d551379e15", "url": "https://pretalx.com/scipy-2026/speaker/SZCRM8/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/W9QWGR/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/W9QWGR/", "attachments": []}, {"guid": "759e11db-6254-5403-beaa-b308225649eb", "code": "BQDMZH", "id": 93239, "logo": null, "date": "2026-07-17T13:15:00-05:00", "start": "13:15", "end": "2026-07-17T13:45:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-93239-how-is-python-transforming-materials-modeling-with-machine-learning", "url": "https://pretalx.com/scipy-2026/talk/BQDMZH/", "title": "How Is Python Transforming Materials Modeling with Machine Learning?", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "What is the best way to study materials for modern devices? For example, designing better batteries means understanding how lithium ions move at the atomic level. This, in turn, requires building a model of the electrolyte and the electrodes, and observing how the system evolves along a dynamic trajectory. Until a few years ago, we would have approached this problem by first oversimplifying it into its core components and then using quantum mechanical simulations. However, these calculations are computationally very demanding: for a system like this, it could take hours on a supercomputer just to analyze a single trajectory step!\n\nWith the AI boom, machine-learning interatomic potentials (MLIPs) have become one of the most promising alternatives. Instead of running expensive quantum-mechanical calculations at every step, we can now perform only a small number of them and use the results as a training set for neural networks. Once trained, the MLIP can look at the complex atomic configuration of a system and immediately predict the energies and forces acting on each atom, without solving the underlying physics equations. This allows the simulation to evolve in milliseconds rather than hours, opening the door to simulations that were previously impractical.\n\nWithout the Scientific Python ecosystem, the development of machine learning methods in quantum chemistry would have a very hard time, since libraries such as PyTorch, Scikit-learn, and TensorFlow, combined with atomistic workflow tools like the Atomistic Simulation Environment (ASE), form the backbone of these methods. Importantly, most MLIPs are also open-source projects, whether developed by universities (MACE, CHGNet, and M3GNet) or by research groups at large technology companies such as Google DeepMind and Meta FAIR (UMA).\n\nIn this talk, we will explore how Scientific Python libraries power modern MLIP workflows, from dataset generation and model training to large-scale atomistic simulations. We will introduce the key ideas behind them in an intuitive way and discuss the current state of the field. Finally, we will highlight where current research is heading: from predicting how atoms move to learning the behavior of electrons, which ultimately determine those motions as well as many other fundamental properties, a much more challenging task.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "YHZMMH", "name": "Filippo Balzaretti", "avatar": "https://pretalx.com/media/avatars/MHAZ8G_DIubXnK.webp", "biography": "Mathematician, Physicist, and Computational Chemist in love with ab initio modeling, surface science, and catalysis. Developer of improved quantum chemistry methods spanning Density Functional Theory (DFT), Density Functional Tight Binding (DFTB), and Machine Learning Interatomic Potentials (MLIPs). Advocate and early-stage contributor to open-source projects for transparent and collaborative science.", "public_name": "Filippo Balzaretti", "guid": "64e7e72b-dff1-578d-a9ab-95e6c9aae230", "url": "https://pretalx.com/scipy-2026/speaker/YHZMMH/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/BQDMZH/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/BQDMZH/", "attachments": []}, {"guid": "19d3f2d7-b4c8-5d58-a60d-ae2b00bc1051", "code": "39NQ3Y", "id": 92509, "logo": null, "date": "2026-07-17T14:35:00-05:00", "start": "14:35", "end": "2026-07-17T15:05:00-05:00", "duration": "00:30", "room": "Memorial Hall", "slug": "scipy-2026-92509-retrieval-augmented-generation-with-raghilda", "url": "https://pretalx.com/scipy-2026/talk/39NQ3Y/", "title": "Retrieval Augmented Generation with Raghilda", "subtitle": "", "track": "Data-Driven Discovery, Machine Learning and Artificial Intelligence", "type": "Talk", "language": "en", "abstract": "LLMs are powerful, but their knowledge is frozen \u2014 they can't access your private documents or recent information. Retrieval-Augmented Generation (RAG) solves this by searching relevant documents and including them in the prompt, grounding responses in real information. But building a good retrieval system involves many steps: reading diverse file formats, chunking text at sensible boundaries, computing embeddings, and combining search strategies. This talk introduces raghilda, a Python framework that handles the full retrieval pipeline. We'll cover how RAG works, how to build a retrieval system with raghilda, and how to connect it to an LLM with a practical example.", "description": "LLMs are powerful, but their knowledge is frozen \u2014 they can't access your private documents or recent information. When asked about topics outside their training data, they either refuse to answer or hallucinate confident-sounding responses. Retrieval-Augmented Generation (RAG) solves this by searching relevant documents and including them in the prompt, grounding responses in real information. While long context windows (100K+ tokens) might seem to make retrieval unnecessary, research on \"lost in the middle\" effects shows that LLMs lose track of information buried in long prompts. RAG provides precision: the model sees a handful of relevant paragraphs instead of hundreds of irrelevant pages. But building a good retrieval system involves many steps: reading diverse file formats, chunking text at sensible boundaries, computing embeddings, and combining search strategies. Each step has pitfalls \u2014 HTML-to-text conversion is messy, naive fixed-size chunking splits code blocks and paragraphs in half, and pure vector search misses exact keyword matches. This talk introduces raghilda, an open-source Python framework that handles the full retrieval pipeline with sensible defaults while keeping every step exposed and replaceable. We'll build a RAG system from scratch, walking through each stage of the pipeline:\n\nIngestion: turning raw documents into a searchable store. We'll cover how to read diverse sources (URLs, PDFs, DOCX files) and convert them to a common format, how to crawl websites to discover pages automatically, how to chunk text at semantic boundaries (headings, paragraphs, sentences) rather than at arbitrary character offsets, why preserving heading hierarchy as context metadata matters for retrieval quality, and how embeddings are computed and stored alongside the text.\n\nRetrieval: finding the right chunks given a query. We'll explore why pure vector similarity search isn't enough, how BM25 keyword matching complements semantic search, how attribute filters let you scope queries by metadata (source URL, document type, custom fields), and how these strategies combine into hybrid retrieval.\n\nIntegration: connecting retrieval to an LLM and measuring how well it works. We'll show how to register a search function as a tool that the LLM calls when it needs information, and demonstrate the difference in answer quality between an augmented and unaugmented model on domain-specific questions. We'll also discuss how to evaluate a RAG system: both the retrieval component and the end-to-end generation, and how tuning chunking parameters, search strategies, and reranking affects downstream answer quality.\n\nThroughout, we'll use raghilda to implement each step, showing both the high-level one-liner workflow and the lower-level components so attendees understand what's happening at each stage and how to customize it for their own use cases.\n\n- Source code: https://github.com/posit-dev/raghilda\n- Documentation: https://posit-dev.github.io/raghilda/", "recording_license": "", "do_not_record": false, "persons": [{"code": "ZDSTNC", "name": "Carson Sievert", "avatar": null, "biography": "Carson is currently a Principal Software Engineer at [Posit Software, PBC](https://posit.co/). He's an original author and maintainer of projects such as [shiny](https://shiny.posit.co/py/), [shinywidgets](https://shiny.posit.co/py/docs/jupyter-widgets.html), [shinylive](https://shinylive.io/py/examples/), and [chatlas](https://github.com/posit-dev/chatlas/). Prior to joining Posit, Carson was an engineer at [Plotly](https://plotly.com/) for numerous years, won the [ASA's Chambers Award](https://community.amstat.org/jointscsg-section/awards/john-m-chambers), and received his PhD in 2017.", "public_name": "Carson Sievert", "guid": "547196f7-b1af-54f2-814a-cb598828c1c5", "url": "https://pretalx.com/scipy-2026/speaker/ZDSTNC/"}, {"code": "ZFGT8U", "name": "Daniel Falbel", "avatar": null, "biography": "", "public_name": "Daniel Falbel", "guid": "d8790191-e8e4-5fd8-a800-5428be599fb0", "url": "https://pretalx.com/scipy-2026/speaker/ZFGT8U/"}, {"code": "SVBAVJ", "name": "Tomasz Kalinowski", "avatar": "https://pretalx.com/media/avatars/VAATH8_DPgGkY8.webp", "biography": "Tomasz Kalinowski is a scientist turned software engineer at Posit, building open-source tools for data and scientific computing across Python and R. His work focuses on cross-language interoperability, machine learning workflows, and performance. He maintains reticulate and the tensorflow/keras R interfaces, coauthored *Deep Learning with R*, and helps lead Posit\u2019s open-source PyData team (Great Tables, Plotnine, pointblank).", "public_name": "Tomasz Kalinowski", "guid": "6e7b397f-2dde-5957-8376-8047d1998f07", "url": "https://pretalx.com/scipy-2026/speaker/SVBAVJ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/39NQ3Y/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/39NQ3Y/", "attachments": []}, {"guid": "7c0734b4-f7c3-5382-a681-d7a649f3a201", "code": "DVGCMK", "id": 97787, "logo": null, "date": "2026-07-17T15:30:00-05:00", "start": "15:30", "end": "2026-07-17T16:30:00-05:00", "duration": "01:00", "room": "Memorial Hall", "slug": "scipy-2026-97787-lightning-talks", "url": "https://pretalx.com/scipy-2026/talk/DVGCMK/", "title": "Lightning Talks", "subtitle": "", "track": "Lightning Talks", "type": "Lightning Talk", "language": "en", "abstract": "Lightning talks are 5-minute talks on any topic of interest for the SciPy community. We encourage spontaneous and prepared talks from everyone, but we can\u2019t guarantee spots. Sign ups are at the NumFOCUS booth during the conference.", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/DVGCMK/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/DVGCMK/", "attachments": []}, {"guid": "5cb1687a-0021-53fb-b91f-48f6a06573f2", "code": "QCWPTS", "id": 102438, "logo": null, "date": "2026-07-17T16:40:00-05:00", "start": "16:40", "end": "2026-07-17T17:35:00-05:00", "duration": "00:55", "room": "Memorial Hall", "slug": "scipy-2026-102438-beyond-the-hype-ai-tools-in-scientific-open-source-in-heritage-gallery-room", "url": "https://pretalx.com/scipy-2026/talk/QCWPTS/", "title": "Beyond the Hype: AI Tools in Scientific Open Source (in Heritage Gallery Room)", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "AI tool adoption is outpacing our ability to thoughtfully decide how, when, and whether to use it. Researchers, maintainers, and contributors are reacting in real time both to the use of AI tools in open source development and to the flood of AI-assistance contributions that continue to strain human open source infrastructure.  Peer-review programs like pyOpenSci and JOSS, along with maintainer teams across the ecosystem, are being forced to react by creating guardrails and protection systems on the fly. The result of this is a difficult combination of introduced technical debt caused by the unguided use of AI tools in software development, burnout across volunteer teams who are fielding rapid AI-assisted contributions, and polarization around whether AI tools have a productive place in open source community at all.", "description": "pyOpenSci has received support from the Sloan Foundation to better understand the challenges and opportunities that AI tools present for scientific open source. Grounded in the idea that AI represents a collaboration between humans and analytic tools \u2014 where human judgment drives which tool, when, and how \u2014 we'll facilitate small-group discussions to collect real stories of impacts, from across our collective community. Whether you're a researcher considering AI tools in your workflow, a maintainer fielding AI-assisted contributions, or a contributor navigating new expectations, this session is for you. Help us shape the resources and frameworks we'll develop over the next six months \u2014 and learn how to get involved.\n\nAbout pyOpenSci\npyOpenSci broadens participation in scientific open source by breaking down social and technical barriers. Our community works together to make participation in open source more accessible to everyone, everywhere. We run an open peer review process for scientific Python software and develop accessible, open learning resources that tackle common challenges\u2014like software development, packaging, and the use of AI tools in scientific open source\u2014in support of open and reproducible scientific discovery.", "recording_license": "", "do_not_record": false, "persons": [{"code": "F9RULK", "name": "Leah Wasser", "avatar": null, "biography": null, "public_name": "Leah Wasser", "guid": "f1131476-68d0-5d1d-a1fe-da9a0a881f1f", "url": "https://pretalx.com/scipy-2026/speaker/F9RULK/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/QCWPTS/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/QCWPTS/", "attachments": []}, {"guid": "ee78a75c-790b-5563-af5c-447b2291a754", "code": "UR8WPT", "id": 102467, "logo": null, "date": "2026-07-17T17:45:00-05:00", "start": "17:45", "end": "2026-07-17T18:40:00-05:00", "duration": "00:55", "room": "Memorial Hall", "slug": "scipy-2026-102467-scipy-2026-sprint-prep-bof-in-heritage-gallery-room", "url": "https://pretalx.com/scipy-2026/talk/UR8WPT/", "title": "SciPy 2026 Sprint Prep BoF (in Heritage Gallery Room)", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "Come join the BoF to do a practice run on contributing to a GitHub project. We will walk through how to open a Pull Request for a bugfix, using the workflow most libraries participating at the weekend sprints use (hosted by the sprint chairs)", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/UR8WPT/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/UR8WPT/", "attachments": []}], "Johnson Great Room": [{"guid": "140c87ad-b7ab-5fee-b219-641a612b33f8", "code": "TBDWJZ", "id": 93240, "logo": "https://pretalx.com/media/scipy-2026/submissions/TBDWJZ/image_KxI4XY0.webp", "date": "2026-07-17T10:45:00-05:00", "start": "10:45", "end": "2026-07-17T11:15:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-93240-derivations-not-just-simulations-teaching-applied-mathematics-with-scientific-python", "url": "https://pretalx.com/scipy-2026/talk/TBDWJZ/", "title": "Derivations, Not Just Simulations: Teaching Applied Mathematics with Scientific Python", "subtitle": "", "track": "Scientific Computing in Education", "type": "Talk", "language": "en", "abstract": "Graduate textbooks in applied mathematics are notoriously inscrutable, dense with symbolic derivations never connected to intuition, application, or executable code. This talk presents a teaching pattern: Motivate, Symbolize, Derive, Lambdify, Simulate, Validate. The infrastructure is stable self-contained marimo notebooks with tests and automated publishing via GitHub Actions. Within the notebook, SymPy handles the symbolic stages; NumPy, SciPy, and matplotlib handle numerics and visualization. We demonstrate the pattern through a complete interactive treatment of Isaacs' Homicidal Chauffeur, a classical pursuit-evasion differential game, and close with an invitation to collaborate on open-source educational content in advanced applied mathematics.", "description": "**Why: The Gap in Advanced Applied Mathematics Education**\nThe scientific Python ecosystem has transformed computational education. QuantEcon demonstrated that graduate-level economics can be taught through executable notebooks; the Executable Books Project built supporting infrastructure; the Scientific Python Lecture Notes teach scientific programming. These contributions work well when the subject is inherently numerical: the insight is in the computational behavior.\n\nAdvanced applied mathematics is different. In calculus of variations, optimal control, and differential game theory, the central insights are irreducibly symbolic: coordinate reductions, optimality conditions, conservation laws, geometric classifications of solution structure. A student who runs an ODE solver learns what the system does, not why the solution takes the form it does. The standard textbooks, Kirk's Optimal Control, Bryson and Ho's Applied Optimal Control, and Isaacs' Differential Games, present derivations as static prose to be followed, not arguments to be executed. The gap between following a derivation on the page and computing with it is left entirely to the reader.\n\n**What: A Six-Stage Didactic Pattern**\nWe present a six-stage pattern structured around the learner's experience:\n\n1. Motivate \u2014 ground the problem in physical intuition and applications before any formalism\n2. Symbolize \u2014 define state, parameters, and dynamics as SymPy expressions before touching numerics\n3. Derive \u2014 execute the mathematical argument symbolically; learners witness structural results rather than being asked to trust them\n4. Lambdify \u2014 convert symbolic expressions to numerical functions via sp.lambdify, eliminating manual transcription and the silent errors it produces\n5. Simulate \u2014 integrate the lambdified dynamics and explore behavior interactively through marimo sliders\n6. Visualize \u2014 plot trajectories, reachable sets, and phase portraits to connect symbolic results to physical intuition\n\nAbove is the learner's arc. The author's responsibilities are separate: verification (does the code correctly implement the mathematics) and validation (is the numerical demonstration provide intuition). SymPy makes verification tractable: symbolic identities become pytest assertions that run in CI. Validation is achieved through the same simulations and visualizations the learner experiences. Crucially, this architecture affords a learner-to-author transition: a learner who forks the repo moves from consuming the author's V&V to owning it, extending derivations, updating tests, and validating through their own simulations. A static PDF cannot support this. A version-controlled repo with CI can.\n\n**How: The Toolchain**\nSymPy provides computer algebra for the Symbolize, Derive, and Lambdify stages. NumPy and SciPy supply the numerical substrate, in particular solve_ivp for trajectory integration. matplotlib handles visualization. marimo provides a reactive notebook environment: cells re-execute automatically when dependencies change, eliminating hidden state and keeping interactive controls consistent with the derivation. pytest and GitHub Actions close the loop: tests verify correctness on every commit and the notebook publishes automatically to GitHub Pages.\n\nWe demonstrate the full stack through the Homicidal Chauffeur problem (Isaacs, RAND, 1951; Merz, Stanford, 1971), a pursuit-evasion differential game between a fast-but-constrained pursuer and a slow-but-agile evader. The symbolic layer handles the 5-DOF to 2-DOF coordinate reduction, bang-bang optimal control, costate conservation, and Merz's singular surface taxonomy. The numerical layer simulates and visualizes what the analysis established. [Source Code](https://github.com/mzargham/hc-marimo). [Hosted Live](https://mzargham.github.io/hc-marimo/.)", "recording_license": "", "do_not_record": false, "persons": [{"code": "NNYAMC", "name": "Michael Zargham", "avatar": "https://pretalx.com/media/avatars/QQGZ3J_2mPV8fo.webp", "biography": "Dr. Michael Zargham is the Chief Engineer at BlockScience, a systems engineering firm that operationalizes emerging technology for high reliability organizations. His work focuses on digital infrastructures supporting ecosystems which span many geographies and jurisdictions. He serves roles in various non-profit organizations: advisor to Humane Intelligence, Board Member & Research Director at Metagov, Trustee for the Superset Data Trust, and is an advisor to NumFocus. Dr. Zargham received his PhD in Electrical and Systems engineering from the University of Pennsylvania with focus on optimal control for dynamic resource allocation in networks.", "public_name": "Michael Zargham", "guid": "8f6a6d35-ef3b-5f50-a2e1-152ab0671ec2", "url": "https://pretalx.com/scipy-2026/speaker/NNYAMC/"}], "links": [{"title": "Example Repository", "url": "https://github.com/mzargham/hc-marimo", "type": "related"}, {"title": "Deployed Notebook", "url": "https://mzargham.github.io/hc-marimo/", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/TBDWJZ/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/TBDWJZ/", "attachments": []}, {"guid": "edfb0f7f-105f-593d-96d0-041b5057a8ed", "code": "E8XVHN", "id": 92276, "logo": "https://pretalx.com/media/scipy-2026/submissions/E8XVHN/image_kajp9l3.webp", "date": "2026-07-17T11:25:00-05:00", "start": "11:25", "end": "2026-07-17T11:55:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-92276-horton-hears-a-word-building-ai-infrastructure-for-children-s-speech-recognition", "url": "https://pretalx.com/scipy-2026/talk/E8XVHN/", "title": "\u201cHorton hears a word\u201d: Building AI Infrastructure for Children\u2019s Speech Recognition", "subtitle": "", "track": "Scientific Computing in Education", "type": "Talk", "language": "en", "abstract": "Improving automatic speech recognition (ASR) for children is needed to enhance education and early childhood development. When ASR fails for children, reading assessments mis-score, speech therapy tools become unreliable, and many classroom tools cannot be built at all. The ASR gap exists because data sensitivity complicates the collection and sharing of transcribed child audio.\n\n**In this presentation, we\u2019ll share how to unblock progress by creating public useful AI infrastructure even when data can\u2019t be shared openly.** We\u2019ll discuss what makes child ASR so hard, how we advanced the field with an AI modeling competition, and best practices for sharing pretrained models.", "description": "Better child-centered automated speech recognition (ASR) is needed to unlock new research and tools to help teachers teach and students learn. ASR is largely solved for adults but remains a challenge for children, especially in noisy, real-world learning environments. Children\u2019s speech presents distinct modeling challenges: greater acoustic variability, inconsistent pronunciation, uneven speech and linguistic development, and unpredictable grammar and vocabulary. There are also wide differences across age, accents, and speech tasks. Yet the development of robust ASR models for children is fundamental to universal screening, personalized literacy and reading instruction, speech therapy, and educational games. The applicability of these models extends to communication, commercial, and medical contexts.\n\n**This talk will present new, benchmarked ASR models developed through a [crowdsourced AI competition](https://www.drivendata.org/competitions/group/childrens-asr-competition/).** The competition draws on a combined corpus of pre-existing child speech datasets and newly curated, annotated recordings, comprising 560,000 transcribed utterances and 519 hours of child speech. We will discuss how hosting a competition can enable progress in a domain where the data is sensitive, difficult to collect, and difficult to share.\n\n**Then, we will break down what we learned from the competition about effective modeling approaches for child ASR.** We will discuss the strengths of various transformer-based architectures, such as Parakeet, Canary, Whisper, and Qwen, as well as fine-tuning strategies to produce transcription outputs suitable for diagnostic and speech-screening applications. We will also discuss why competitions remain useful in the age of AI agents.\n\n**Beyond modeling results, we will describe the broader AI infrastructure challenge at the center of this work.** Improving child ASR requires access to large, representative datasets, but children\u2019s speech raises difficult questions around privacy, identifiability, and responsible model release. We will discuss the tradeoffs involved in using private data to evaluate public approaches, collecting demographic information to assess bias while limiting privacy risk, and deciding what parts of an AI system can be made shareable when the underlying data cannot be fully open.\n\n**The goal of this talk is to present a pathway to unblocking progress by creating pre-trained models as a public good when the underlying data cannot be shared.** The presentation is suitable for anyone interested in ASR and its use in educational contexts, as well as people in any field working with sensitive data, thorny data ethics questions, or the challenge of building shared AI infrastructure when datasets cannot simply be released publicly.", "recording_license": "", "do_not_record": false, "persons": [{"code": "9SNUNR", "name": "Katie Wetstone", "avatar": "https://pretalx.com/media/avatars/XU8KJC_OBJMOMZ.webp", "biography": "Katie Wetstone is a data scientist with a passion for leveraging machine learning tools to promote sustainable, ethical, and just change. At DrivenData, she works to implement open-source machine learning competitions and direct consulting projects that support mission-driven organizations. Her projects have spanned a variety of issues including public health, conservation, and education. She holds a BA in chemistry from Harvard University, and a Masters of Development Practice from the University of California, Berkeley.", "public_name": "Katie Wetstone", "guid": "6f2f48f6-04d6-5278-8bae-34ef4a78f811", "url": "https://pretalx.com/scipy-2026/speaker/9SNUNR/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/E8XVHN/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/E8XVHN/", "attachments": []}, {"guid": "5fa15f41-9677-508b-8002-b68ed490ab30", "code": "KHRTU8", "id": 93148, "logo": "https://pretalx.com/media/scipy-2026/submissions/KHRTU8/image_3O4zpk0.webp", "date": "2026-07-17T13:15:00-05:00", "start": "13:15", "end": "2026-07-17T13:45:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-93148-gofish-a-grammar-of-more-graphics", "url": "https://pretalx.com/scipy-2026/talk/KHRTU8/", "title": "GoFish: A Grammar of More Graphics!", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "Visualization libraries like Altair are based on the Grammar of Graphics (GoG), a theory of visualization that moved beyond fixed chart types towards a composable graphical language. But while the GoG makes simple charts easy, custom graphics still require low-level libraries like matplotlib. We present GoFish, a grammar of *more* graphics! GoFish formalizes patterns of visual structure (like connecting shapes with lines or spreading them out in space) letting you create diagrams, annotated charts, and infographics piece by declarative piece. In this talk, we'll see some fun and funky GoFish charts, and I'll uncover the hidden structure behind everyday visualizations.", "description": "Libraries like Altair and Plotly brought the Grammar of Graphics to Python, making it easy to map data to marks (bar, line, area, etc.) and channels (size, color, position, etc.). But when you need annotations, custom layouts, or pictographic designs, you're back to wrestling with matplotlib and manually computing coordinates.\n\nGoFish is a new, open-source Python library we\u2019ve developed at MIT for making custom, data-driven graphics. It is MIT-licensed and currently in alpha, but will hit beta before the conference. It's available on pypi as gofish-graphics.\n\nWhile most visualization libraries are built on marks and channels, GoFish is also built around _visual structure_, like spreading shapes out in space, connecting shapes with lines, or containing them in a common region. We call these primitives _graphical operators_. In conjunction with marks and channels, graphical operators allow GoFish users to easily produce a wide range of graphics: richly annotated bar charts and scatter plots; nested charts like scatterpies; polar ribbon charts; and composited images that layer and intersect shapes. They also give us a new understanding of more typical charts like stacked bars, waffles, and ribbons, which turn out to be simple combinations of just a few operators.\n\n**About Me**\nI'm a last-year PhD student at MIT in the VIS group. My research applies programming language theory to visualization design. I presented GoFish as a full paper at the IEEE VIS conference in November, 2025 to a standing-room only crowd.\n\n- Paper, website, and code: https://vis.csail.mit.edu/pubs/gofish/.\n- VIS talk: https://youtu.be/S3LGLxyblpM?si=dlhSoPpHHuXWp7m6&t=660.\n\n**Audience and Takeaways**\nThe audience for the talk is anyone who's hit the limits of a library like Altair, Plotly, or seaborn, built a scientific figure in Matplotlib, or is just curious about the theory behind visualization. The audience will leave the talk with both a practical understanding of how GoFish can be used to build visualizations and a new conceptual understanding of how graphics are structured.\n\n**Talk Outline**\n_The Grammar of Graphics and its limits (~5 min)._ I'll introduce the marks-and-channels approach to specifying charts, some of its history, and its use in Python. I'll then frame the core tension: high-level libraries like Altair are easy but restrictive, while low-level libraries like Matplotlib are expressive but tedious. What if there was a better way?\n\n_Building up GoFish by example (~15 min)._ I'll start by showing examples of visual structure in familiar charts to build intuition for what graphical operators capture. I'll then introduce GoFish's API piece by piece: first marks and channels, then graphical operators that compose marks into glyphs and charts. Along the way I'll show how users can use libraries like pandas with GoFish for sorting and aggregation; how GoFish's compositional approach naturally supports nested charts like scatterpies; and how selecting marks in an existing chart makes it easy to add annotations and connecting ribbons.\n\n_The fun stuff and a call to action (~5 min)._ I'll wrap up with a showcase of cool graphics GoFish enables, and I'll end with an invitation to try GoFish and contribute. We want to help more people create expressive visualizations!", "recording_license": "", "do_not_record": false, "persons": [{"code": "RXVFHV", "name": "Josh Pollock", "avatar": "https://pretalx.com/media/avatars/W7D8D9_ASYumr6.webp", "biography": "Josh is a sixth (and last!) year PhD student in the MIT Visualization Group. He builds new theories and libraries for charts and diagrams and is obsessed with how a well-designed picture can create a new insight. In his free time (any time really), Josh may be seen doing contact improv, singing, or playing guitar.", "public_name": "Josh Pollock", "guid": "53f1d372-1663-5fec-b7b6-7fafb3d2cbd2", "url": "https://pretalx.com/scipy-2026/speaker/RXVFHV/"}], "links": [{"title": "GitHub Repo", "url": "https://github.com/starfish-graphics/gofish-graphics/", "type": "related"}, {"title": "Website", "url": "https://gofish.graphics/", "type": "related"}, {"title": "PyPi", "url": "https://pypi.org/project/gofish-graphics/", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/KHRTU8/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/KHRTU8/", "attachments": []}, {"guid": "e3633fda-4a4e-540d-9cae-f9908781614e", "code": "38FQ9D", "id": 93202, "logo": null, "date": "2026-07-17T13:55:00-05:00", "start": "13:55", "end": "2026-07-17T14:25:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-93202-ask-more-of-your-notebook-what-can-anywidgets-do-for-you", "url": "https://pretalx.com/scipy-2026/talk/38FQ9D/", "title": "Ask more of your notebook: what can anywidgets do for you?", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "You're staring at a plot in a notebook. A subset of points doesn't look right. You want to select them, inspect them, understand _why_. In a traditional notebook, that means stopping to write more code and re-run cells. The exploration becomes an exercise in programming, not insight.\n\nThis talk comes in two parts. First, I introduce two primitives: reactive cell execution (marimo) and widgets (anywidget) that bridge Python and the browser. A brush stroke on a scatter plot becomes a Python selection. A slider flows through your analysis.\n\nSecond, I build intuition for composing these primitives\u2014from quick explorations to reusable, domain-specific instruments that let you craft the interaction to match your scientific question.", "description": "Interactive widgets connect Python objects to browser-based UIs, letting you explore and manipulate data beyond static output. Composing widgets in traditional notebooks, however, means writing callback-based code (event handlers, state management, update coordination), a style that is error-prone and differs from the cell-based, REPL-like style most familiar to notebook users.\n\nReactive execution offers a simpler model. marimo (https://marimo.io) models a notebook as a dataflow graph. When a value changes, dependent cells re-execute automatically. The system ensures consistency. anywidget (https://anywidget.dev) provides a specification for creating custom widgets with Python and JavaScript, giving you access to any browser API from within a notebook. In a reactive notebook environment, these widgets participate in the dataflow graph like any other value.\n\nThis talk introduces these two primitives and builds up a mental model for working with them. I start with how existing widgets compose in a reactive environment: a slider updates a parameter, dependent cells react, a chart selection filters a dataframe. These are patterns that work out of the box.\n\nFrom there, I show how to compose off-the-shelf widgets with custom ones to build domain-specific interactions, such as inspecting outliers or comparing experimental conditions. These don't need to be polished applications; they can live in a notebook and be shared with collaborators when useful.\n\nAttendees will leave with a working understanding of these primitives and practical patterns for building interactive tools that help them and their collaborators make their data feel more tangible.", "recording_license": "", "do_not_record": false, "persons": [{"code": "7D3LYU", "name": "Trevor Manz", "avatar": "https://pretalx.com/media/avatars/XZRHLS_DpLy6QB.webp", "biography": "Trevor is a researcher and software engineer working on interactive computing tools in Python. He completed his PhD in the HIDIVE Lab at Harvard Medical School, where he developed interactive visualization and analysis tools for biological and AI applications. He created anywidget and contributes to open-source data tooling. Trevor now works at marimo, building a reactive notebook environment for Python. He lives in Brooklyn, NY with his two cats, Laird and Minnow, whom he\u2019s very fond of.", "public_name": "Trevor Manz", "guid": "0ee2c78d-97cc-5d77-8aa4-0013a2c86f26", "url": "https://pretalx.com/scipy-2026/speaker/7D3LYU/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/38FQ9D/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/38FQ9D/", "attachments": []}, {"guid": "e5c785f7-8a08-55ff-a5f2-23fef9fa9385", "code": "U7SRHU", "id": 93217, "logo": null, "date": "2026-07-17T14:35:00-05:00", "start": "14:35", "end": "2026-07-17T15:05:00-05:00", "duration": "00:30", "room": "Johnson Great Room", "slug": "scipy-2026-93217-remote-access-to-scientific-data-with-tiled", "url": "https://pretalx.com/scipy-2026/talk/U7SRHU/", "title": "Remote Access to Scientific Data with Tiled", "subtitle": "", "track": "General", "type": "Talk", "language": "en", "abstract": "Tiled is a full-fledged data management service designed specifically to help scientists store, find, and access scientific data at scale easily. \n\n \n\nThe concept of data structures is the cornerstone of Tiled; it allows us to abstract the inherent diversity of various file formats and data storage types to a handful of scientifically meaningful representations: arrays, tables, nested hierarchies, and even awkward, ragged, and sparse arrays. Tiled provides a consistent API to such disparate datasets and naturally integrates with the SciPy ecosystem, including NumPy, pandas, xarray, Dask, and more. The users can slice, convert, and retrieve only the data they need, or even subscribe to live streams from external instruments and send updates to a dashboard. Importantly, Tiled supports operations with rich metadata \u2013 including search \u2013 making the data registered in Tiled discoverable and interactable with minimal overhead, by the human users and AI agents alike. Tiled runs equally well on a private laptop or in a large facility\u2019s data center. Its built-in authentication and authorization mechanisms make the data access controllable and secure. Finally, Tiled is a fully open-source project developed under a multi-institutional governance model, which reflects our commitment to open science and the FAIR principles in scientific computing. \n\n \n\nIn this talk we will introduce Tiled\u2019s architecture, demonstrate its most popular use cases using the native Python client, discuss deployment and integration strategies, and show how it can simplify practical scientific data workflows.", "description": "## Motivation and Background \n\nScientific datasets, especially from large experimental facilities, are growing in size and complexity. Researchers increasingly face the challenges of securely sharing large volumes of data between institutions while juggling a multitude of storage formats. These obstacles emphasize the need for decoupling the storage infrastructure from the analytical workflows, so that scientists can focus on computation and interpretation rather than data plumbing. Traditional access patterns, where entire files are transferred and parsed locally, strain bandwidth, memory, and compute resources. The recent advent and widespread adoption of agentic workflows further highlight this problem: to operate efficiently, an AI agent often benefits from having a direct access to certain dataset slices enriched with metadata \u2013 a requirement, which is difficult to fulfil with the file-centric approach. \n\nTiled was developed within the synchrotron light source community (e.g., National Synchrotron Light Source II and other facilities using the Bluesky ecosystem) to address these challenges, but it is agnostic to the specifics of the application domain. The project aims to provide a unified, high-performance, feature-rich service that lets users interact with their data without making any considerations about the underlying storage formats and infrastructure. It lets users slice, search, and stream only the pieces of data they need \u2013 whether arrays, tables, or hierarchical datasets \u2013 and treat them as familiar NumPy or pandas objects. \n\n\n## What is Tiled? \n\nTiled\u2019s core offering is a data access and management service with: \n\n* A web server that exposes structured datasets via HTTP APIs. \n\n* A Python client that seamlessly integrates with popular tools in scientific computing, such as NumPy, pandas, xarray, Dask, AwkwardArray, etc.; users of h5py or zarr, for example, would find Tiled\u2019s interface familiar. \n\n* Support for multiple underlying data sources: filesystems, databases, remote servers, blob storages, or combinations thereof. \n\n* Efficient format transcoding and chunked data access: users can retrieve just the subset of data they need, reducing I/O and network costs. \n\n* An easily expandable set of supported storage formats (e.g. zarr, parquet, csv, hdf5, etc.) and extensible data structures beyond simple arrays and tables (e.g. sparse, awkward, and ragged arrays). \n\n* Integrated caching, both client-side and server-side, to accelerate repeated access and interactive exploration. \n\n* Streaming capabilities via WebSockets, enabling real-time data updates and interactive workflows, which is particularly valuable for live experiments, monitoring dashboards, and adaptive analysis pipelines. \n\n* Built-in authentication and authorization (authN/authZ) mechanisms, allowing deployments to enforce fine-grained access control. Tiled supports multiple authentication providers and role-based permissions, making it suitable for multi-user facilities, collaborative research groups, and cloud deployments. \n\n\n## Relevance to SciPy Community and Broader Audience \n\nEven though Tiled has originated from and is used widely in the synchrotron light source community, the problems it solves are universal wherever large or complex datasets are involved, from genomics to environmental science to astronomy. Tiled fills a gap by offering a flexible, secure, and convenient data access abstraction layer that complements computational tools. Its integration with the SciPy ecosystem and standards makes it directly applicable to users in practically any scientific domain.", "recording_license": "", "do_not_record": false, "persons": [{"code": "RAZK8J", "name": "Yevgen Matviychuk", "avatar": null, "biography": "", "public_name": "Yevgen Matviychuk", "guid": "caa2c461-f33b-504e-86ca-8e5fabbb0a39", "url": "https://pretalx.com/scipy-2026/speaker/RAZK8J/"}], "links": [{"title": "Project Documentation", "url": "https://blueskyproject.io/tiled/", "type": "related"}, {"title": "Source Code on GitHub", "url": "https://github.com/bluesky/tiled", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/U7SRHU/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/U7SRHU/", "attachments": []}, {"guid": "de3f9ce0-f03e-5af9-8331-f996e5142412", "code": "XFA9VF", "id": 102466, "logo": null, "date": "2026-07-17T16:40:00-05:00", "start": "16:40", "end": "2026-07-17T17:35:00-05:00", "duration": "00:55", "room": "Johnson Great Room", "slug": "scipy-2026-102466-scipy-2027", "url": "https://pretalx.com/scipy-2026/talk/XFA9VF/", "title": "SciPy 2027", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "Come share your ideas for next year's SciPy. Participants will have an opportunity to sign up to be on next year's organizing committee.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "TBVGQE", "name": "Madicken", "avatar": "https://pretalx.com/media/avatars/W9B3CQ_IaS5eMD.webp", "biography": "Co-chair of SciPy 2026", "public_name": "Madicken", "guid": "561939fe-34f0-581f-bbff-e6d4f926029d", "url": "https://pretalx.com/scipy-2026/speaker/TBVGQE/"}, {"code": "KZGFDE", "name": "Gil Forsyth", "avatar": "https://pretalx.com/media/avatars/KZGFDE_QGtViPC.webp", "biography": "Co-chair of SciPy 2026", "public_name": "Gil Forsyth", "guid": "b111b595-7662-5c7d-85d5-44299892fdc4", "url": "https://pretalx.com/scipy-2026/speaker/KZGFDE/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/XFA9VF/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/XFA9VF/", "attachments": []}, {"guid": "3045fcd2-738d-5127-982a-e8f6401113e2", "code": "WFXBKQ", "id": 102465, "logo": null, "date": "2026-07-17T17:45:00-05:00", "start": "17:45", "end": "2026-07-17T18:40:00-05:00", "duration": "00:55", "room": "Johnson Great Room", "slug": "scipy-2026-102465-lockfile-based-development-and-applications", "url": "https://pretalx.com/scipy-2026/talk/WFXBKQ/", "title": "Lockfile-based development and applications", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "Until very recently, producing and using reproducible scientific software environments required advanced knowledge and a strict adherence to best practices (e.g. DOI: 10.25080/majora-212e5952-028). Now, with the advent of modern tooling with lockfile-first workflows (i.e. Pixi and uv), and the emergence of lockfile standards across scientific open source, applications can be made reproducible at the digest level through tooling decisions. As this technology and practices become increasingly common there is an opportunity to define common best practices around lockfile based software development that can further reduce developer overhead and maintenance burden. This Birds of a Feather panel will focus on how experienced developers are leveraging lockfiles across software development, applications, and deployment while providing best practices and practical recommendations, while also highlighting continuing challenges and opportunities for improvement.\n\nGoogle Form for questions for the panel: https://forms.gle/1YP4951Yb9U4r2md6", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "VXQXZP", "name": "Naty Clementi", "avatar": null, "biography": "Naty Clementi is a senior software engineer at [NVIDIA](https://www.nvidia.com/). She is a former academic with a Masters in Physics and PhD in Mechanical and Aerospace Engineering to her name. Her work involves contributing to [RAPIDS](https://rapids.ai/), and in the past she has also contributed and maintained other open source projects such as [Ibis](https://ibis-project.org/) and [Dask](https://www.dask.org/). She is an active member of [PyLadies](https://pyladies.com/) and an active volunteer and organizer of [Women and Gender Expansive Coders DC meetups](https://www.meetup.com/women-and-gender-expansive-coders-dc-wgxc-dc/).", "public_name": "Naty Clementi", "guid": "039b72d7-581e-5888-8cb9-a36d8a2ca95a", "url": "https://pretalx.com/scipy-2026/speaker/VXQXZP/"}, {"code": "H8ZFYG", "name": "Matthew Feickert", "avatar": "https://pretalx.com/media/avatars/H8ZFYG_0B3tdYV.webp", "biography": "Matthew is a research scientist in experimental high energy physics and data science at the University of Wisconsin-Madison Data Science Institute (a \u201cdata physicist\u201d). He works as a member of the ATLAS collaboration on searches for physics beyond the standard model with experiments performed at CERN's Large Hadron Collider (LHC) in Geneva, Switzerland. He also serves on the executive board of the Institute for Research and Innovation in Software for High Energy Physics (IRIS-HEP) where he is a researcher and the Analysis Systems Area lead. He is also a topical editor for physics and data science for the Journal of Open Source Software. He previously did his Ph.D. (2019) research at Southern Methodist University, also on the ATLAS experiment, and was a postdoc at the University of Illinois at Urbana-Champaign, and the University of Wisconsin-Madison.", "public_name": "Matthew Feickert", "guid": "24bbffd4-66b8-5c0f-8c5f-c38640f4d575", "url": "https://pretalx.com/scipy-2026/speaker/H8ZFYG/"}, {"code": "EKHRSA", "name": "Ruben Arts", "avatar": "https://pretalx.com/media/avatars/EKHRSA_R8Mr2L8.webp", "biography": "Ruben is part of the Prefix.dev core team, builing Pixi and other tools in the package management space. Originally he's a Robotics engineer working on industrial robots, but quickly figuring out that solving development and deployment problems were one of the bigger issues that robotics developers had to deal with. Joining Prefix.dev allowed him to focus on improving the UX/DX of a large group of software engineers. Over the years he's been doing multiple talks and workshops on how to properly manage software and development workflows.", "public_name": "Ruben Arts", "guid": "116518af-d223-54bc-b652-fa3ad14a6897", "url": "https://pretalx.com/scipy-2026/speaker/EKHRSA/"}, {"code": "KZGFDE", "name": "Gil Forsyth", "avatar": "https://pretalx.com/media/avatars/KZGFDE_QGtViPC.webp", "biography": "Co-chair of SciPy 2026", "public_name": "Gil Forsyth", "guid": "b111b595-7662-5c7d-85d5-44299892fdc4", "url": "https://pretalx.com/scipy-2026/speaker/KZGFDE/"}, {"code": "AYBNYT", "name": "Henry Schreiner", "avatar": "https://pretalx.com/media/avatars/3SCCSK_KGvI6cP.webp", "biography": "Henry Schreiner is a Computational Physicist / Research Software Engineer in High Energy Physics at Princeton University. He specializes in the interface between high-performance compiled codes and interactive computation in Python, in software distribution, and in interface design. He has previously worked on computational cosmic-ray tomography for archaeology and high performance GPU model fitting. He is currently a member of the IRIS-HEP project, developing tools for the next era of the Large Hadron Collider (LHC).\n\nHe is a maintainer/core developer for packaging, build, scikit-build, cibuildwheel, pybind11, meson-python, nox, and plumbum for Python. He is an admin of Scikit-HEP, and a lead designer on boost-histogram, hist, UHI, vector, uproot-browser, Particle, and DecayLanguage packages there. He is also the lead author of the Scientific-Python Development guide and Scientific-Python/cookie. He is the primary author of CLI11, a C++ library used by Microsoft terminal and many others. He is also the lead web developer for IRIS-HEP. He is also the author of Modern CMake and a variety of CMake, GPU, and Python training courses and classes.", "public_name": "Henry Schreiner", "guid": "c4cb3a37-16ab-5a5f-87dc-eaf183d57c45", "url": "https://pretalx.com/scipy-2026/speaker/AYBNYT/"}], "links": [{"title": "Google Form for asking the panel questions", "url": "https://forms.gle/1YP4951Yb9U4r2md6", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/WFXBKQ/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/WFXBKQ/", "attachments": []}], "Thomas Swain Room": [{"guid": "eef692ca-16d9-562f-88ce-5d5a68d439f7", "code": "HXFDCX", "id": 93249, "logo": null, "date": "2026-07-17T10:45:00-05:00", "start": "10:45", "end": "2026-07-17T11:15:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-93249-brassy-palatable-multi-institution-release-notes", "url": "https://pretalx.com/scipy-2026/talk/HXFDCX/", "title": "Brassy: Palatable Multi-Institution Release Notes", "subtitle": "", "track": "Maintainers and Community", "type": "Talk", "language": "en", "abstract": "Nobody loves writing release notes... and it only gets worse when multiple institutions are editing the same file. We learned the hard way that using a single file, manual RST editing, and no validation leads to repeated merge conflicts every release cycle. In response, we built Brassy, a CLI tool that replaces single-file changelogs with per-change YAML files, assembles them into formatted release notes, and lints entries in CI. We also built PinkRST (an RST formatter) and a Python-based Sphinx build system to tie our large multi-institution and many repo software documentation together. This talk covers the tools, the integration, and what we learned about getting scientists to _actually_ write documentation.", "description": "We work as part of a multi-institutional team developing a newly open-sourced 17-year-old package for processing geolocated satellite and weather data. The package has dozens of plugin repositories, contributors across institutions, and operational users in the US Navy who depend on it for near-real-time tropical cyclone imagery. \n\nOnce GeoIPS (Geolocated Information Processing System) was open-sourced, release notes became a huge pain point. Contributors would all edit the same reStructuredText (or at times, markdown) file by hand in different pull requests. There was no auto-enforceable standard format. No linting. Lots of merge conflicts. Consequentially, important changes slipped through undocumented and formatting varied wildly.  \n\nWe built three pieces of infrastructure to fix this, each of which is open-source, pip-installable, and lightweight. \n\nFirst we built Brassy (Build Release Assembler for Sane Software with YAML). Brassy swaps a shared changelog file for individual YAML files with one per change. For each , the contributor fills out a structured template (title, description, category, affected files, linked issues, etc.) generated by Brassy. At release time, Bbrassy assembles the changelog fragments into sphinx-compatible formatted RST. One file per change means near-zero merge conflicts and easy application of a change to this release or the next. Brassy also provides quality of life functionality by generating templates pre-populated with git-tracked file changes, pruning empty sections, and running as a CI linter to catch formatting problems on every pull request before they land. \n\nSecondly, we created pinkrst, an opinionated RST formatter in the spirit of Black for Python. It handles tedious autoformatting of reStructuredText for doc8 compatibility (line wrapping, whitespace cleanup, and consistent formatting of lists, headers and codeblocks) of the generated release notes. \n\nThird, we developed a more robust Python-based Sphinx build system. GeoIPS previously relied on a complex bash script to build documentation across its core package and many plugin repos. We replaced it with a Python builder that calls Brassy to assemble release notes from YAML directories, runs pinkrst to format the output, generates API docs for multiple packages via sphinx-apidoc, and builds multiple repos into final HTML. This pipeline handles docs for both the core GeoIPS package and any plugin, using shared templates, CI workflows and configuration. \n\nTools alone don't solve documentation problems... For better or worse, people must actually use them! GeoIPS plugin writers are primarily scientists, not software engineers. Like many scientific projects, the codebase grew a lot faster than its docs and did so for for years. We'll talk about what worked: lowering the barrier, clear guidelines on \"what\" goes \"where,\" catching problems early, and making standards obvious enough that contributors rarely need to ask. \n\nThis talk is for anyone maintaining a multi-team open-source project . We will cover how per-change changelogs outperform single file release notes in distributed teams, how CI linting of non-code artifacts enforce standards without slowing people down, and do our best to offer practical advice for introducing new tooling into a project where no single team is the \u201cleader.\u201d", "recording_license": "", "do_not_record": false, "persons": [{"code": "PTL9WM", "name": "Gwyn Uttmark", "avatar": "https://pretalx.com/media/avatars/DJLPVN_cdPPFkD.webp", "biography": "Gwyn Uttmark serves at Colorado State University and has more than a decade of expertise in open-source scientific software development. Gwyn currently works with the Cooperative Institute for Research in the Atmosphere to make the US Navy-born GeoIPS an accessible and effective platform for open-source development communities and academic researchers.", "public_name": "Gwyn Uttmark", "guid": "d5c2b0c9-ab72-53d5-bede-c6da8eaab026", "url": "https://pretalx.com/scipy-2026/speaker/PTL9WM/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/HXFDCX/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/HXFDCX/", "attachments": []}, {"guid": "b9dfa8c0-dcd0-50d4-b6a7-199268d47cb0", "code": "UNFTFU", "id": 92502, "logo": null, "date": "2026-07-17T11:25:00-05:00", "start": "11:25", "end": "2026-07-17T11:55:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92502-re-connecting-foundational-libraries-with-their-communities-successes-failures-and-surprises-in-building-the-napari-plugin-sustainability-initiative", "url": "https://pretalx.com/scipy-2026/talk/UNFTFU/", "title": "(Re)-connecting foundational libraries with their communities: Successes, failures, and surprises in building the napari plugin sustainability initiative", "subtitle": "", "track": "Maintainers and Community", "type": "Talk", "language": "en", "abstract": "Foundational Python libraries provide critical functionality that diverse communities of downstream developers and users depend on. Often, gaps in awareness between a core project and its broader community silently erode trust, collaboration, and sustainability. This talk shares lessons from a community-driven initiative to (re)-connect [napari](https://napari.org/)\u2014a foundational library for multi-dimensional image viewing built on the scientific Python stack\u2014with its ecosystem of over 580 community-developed plugins. Through a working group of core contributors, plugin developers, and end users, the napari plugin sustainability initiative discovered that creating new avenues for communication and collaboration leads to shared ownership of ecosystem progress.", "description": "Many scientific Python projects follow a familiar arc: early excitement, rapid adoption, a burst of community-built extensions\u2014and then a slow drift apart. This talk is for anyone maintaining a Python project with a broader community, developing downstream tools, or interested in practical approaches to open-source sustainability. Attendees will learn concrete strategies for community engagement, automated quality tooling, and shared infrastructure that can be adapted to any Python project ecosystem.\n\nIn late 2025, with support from a [URSSI Early Career Fellowship](https://urssi.us/), [napari](https://napari.org/) launched the plugin sustainability initiative to rekindle the relationship between the core project and its downstream plugin community. A [working group](https://napari.org/stable/community/meeting_schedule.html) brought together core contributors, plugin developers, and users\u2014novice to experienced\u2014across roles, time zones, and disciplines. This talk will share what worked: engaging the global community, openness to community creativity, and creating space for domain scientists to share real workflows. It will also share real challenges: reaching folks who had already disengaged and including voices that don't have bandwidth for regular meetings.\n\nThe most impactful finding has been how much the community *wants* to shape solutions once given the opportunity. The conversation was never \"what should the core team do for us?\" but \"how can we work on this together?\" This shift\u2014from a service relationship to shared ownership\u2014has been the single most valuable outcome. The biggest barriers remain social: not knowing whether contributions were welcome, not knowing who else was working on similar problems, and not having a channel that felt heard.\n\nThe working group has converged on [three interconnected efforts](https://napari.org/island-dispatch/blog/plugin-sustainability-initiative.html) shaped by community priorities:\n\n**1. Automated and human review systems.** We're building automated tooling\u2014inspired by [SciPy's repo-review](https://repo-review.readthedocs.io/en/latest/)\u2014that checks plugin repositories for packaging quality, test coverage, and dependency health. Compatibility checks via [npe2api](https://github.com/napari/npe2api) detect when plugins break against new napari releases *before* users hit the problem. Alongside automation, human peer review modeled on [PyOpenSci](https://www.pyopensci.org/about-peer-review/) will pair experienced community members with plugin developers for domain-aware feedback.\n\n**2. Modernized packaging infrastructure.** We're updating the [napari-plugin-template](https://github.com/napari/napari-plugin-template) and [plugin documentation](https://napari.org/stable/plugins/index.html) based on firsthand accounts from working group members who upgraded their own plugins, with a focus on creating beginner-friendly and advanced tracks. This includes guidance on reproducible environments with [pixi](https://pixi.sh/) and [uv](https://docs.astral.sh/uv/), clearer separation of computation from UI code, and curated plugin bundles to address dependency conflicts.\n\n**3. Discoverability and stewardship.** We're surfacing maintenance status, compatibility, and quality signals on the [napari hub](https://napari-hub.org/). A plugin donation program would let maintainers hand off plugins to community stewards rather than abandoning them, and a shared GitHub organization will enable collective maintenance.\n\nThese efforts are works in progress, but we have found bi-directional impact: downstream developers gain improved tooling and documentation, while investing back into the core napari project. Everything is open source and documented for other communities to adapt. Ultimately, investing in listening and shared ownership *while* building technical infrastructure is what engages a broad community and builds trust that spending time in the ecosystem is worthwhile.", "recording_license": "", "do_not_record": false, "persons": [{"code": "TXRYUQ", "name": "Tim Monko", "avatar": "https://pretalx.com/media/avatars/ANMQRF_B4QUTlT.webp", "biography": "I am a full-time maintainer and community manager of napari, an interactive multi-dimensional Python image and data viewer, and its plugin ecosystem. I work to extend the plugin ecosystem and help scientists achieve their goals with image analysis.", "public_name": "Tim Monko", "guid": "6e7a85e7-db43-5cee-a74a-aa6865f5e5ea", "url": "https://pretalx.com/scipy-2026/speaker/TXRYUQ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/UNFTFU/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/UNFTFU/", "attachments": []}, {"guid": "a7afc663-46bf-542e-b084-3f744d62c219", "code": "KNDR8V", "id": 92444, "logo": null, "date": "2026-07-17T13:15:00-05:00", "start": "13:15", "end": "2026-07-17T13:45:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92444-finding-the-right-time-collaborating-across-time-zones", "url": "https://pretalx.com/scipy-2026/talk/KNDR8V/", "title": "Finding the right time: Collaborating across Time Zones", "subtitle": "", "track": "Maintainers and Community", "type": "Talk", "language": "en", "abstract": "Building software is often presented as the ultimate in asynchronous collaboration - open a PR, wait for a review, work on something else, and come back when it's a good time for you. The reality can often be... messier. \n\nAs someone who lives in UTC+8, works in UTC+10, and collaborates globally, I'll share my experience of why 7AM meetings aren't all bad, how to deal with the itch to respond to reviews on a Saturday morning, and how I finally learnt to listen to my wife and learn to switch off when there was no good reason to be on.", "description": "I'm notionally a senior developer, but I only finished my PhD four years ago. What this means in practice for me is that I feel a lot of pressure to understand tools & libraries I've only just come across, figure out issues nobody else has (or can), and constantly dig deeper whilst maintaining a productive output. \n\nThe added complication? I work remotely, a 38 hour drive from an office 2 timezones ahead of me. have a shed at the bottom of the garden where I work. This might seem great as a WFH work-life separator, but I have a gym in there too, so it's also where I exercise and tinker with things.\n\n In this talk, I'll outline:\n- Why there's nothing wrong with a 7AM meeting - so long as you're willing (and able!) to shut the computer off early too.\n- Why I **don't** bring my laptop into the house.\n- Why it's harder - not easier - to stop working when the office hours no longer line up.\n- How a nap in the hammock or a walk with the dog can be the right move for productivity\n- Why you shouldn't have Github, Slack, or Zulip on your phone - and why I do anyway.\n- How to forgive yourself for ignoring your own rules and opening a PR at 10PM on a Thursday night - and why you shouldn't berate yourself for it!\n\nThis is not going to be a technical talk, but one about how to make peace with your compulsion to be useful, how to listen to your wife and switch off when you shouldn't be working, and how the dynamics of open source, time zones, and how the messy nature international collaboration makes it harder to say no to yet another project you don't have time for.", "recording_license": "", "do_not_record": false, "persons": [{"code": "7M7MXZ", "name": "Charles Turner", "avatar": null, "biography": "Charles is a Research Software Engineer at ACCESS-NRI, where he works in the Model Evaluation and Diagnostics team, helping make it easier to access and analyse climate data. He has a PhD in Oceanography, where he first discovered his love of wrangling and disseminating data.\n\nWhen not in front of a computer, he enjoys routinely injuring himself in a variety of sports.", "public_name": "Charles Turner", "guid": "d488d69a-4579-5b28-9707-35a6968ccec1", "url": "https://pretalx.com/scipy-2026/speaker/7M7MXZ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/KNDR8V/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/KNDR8V/", "attachments": []}, {"guid": "723dd171-4401-5fa1-a88d-d8e2f253b870", "code": "AFWXAU", "id": 92116, "logo": null, "date": "2026-07-17T13:55:00-05:00", "start": "13:55", "end": "2026-07-17T14:25:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92116-commit-to-community-open-source-practices-as-social-infrastructure-in-volunteer-civic-tech", "url": "https://pretalx.com/scipy-2026/talk/AFWXAU/", "title": "Commit to Community: Open Source Practices as Social Infrastructure in Volunteer Civic Tech", "subtitle": "", "track": "Maintainers and Community", "type": "Talk", "language": "en", "abstract": "Most scientific python maintainers build for users who will `pip install` the code. In civic tech, your community members are policy researchers, journalists, or NGO advocates who may never touch a Python environment. This cross-disciplinary context changes maintainership: traditional open source practices serve double duty as engineering *and* social infrastructure. I'll share lessons learned in maintaining the [CIB Mango Tree](https://cibmangotree.org), a civic tech Python toolkit for detecting inauthentic behavior in social media. I\u2019ll show how in civic tech context familiar practices, like release cycles and continuous integration, can be repurposed to surface the otherwise invisible developer work to the broader community.", "description": "I'll start by briefly introducing the [Civic Tech DC](https://www.civictechdc.org), a non-partisan, non-profit community of volunteer technologists, policy thinkers, researchers, designers, and community leaders passionate about using open-source technology for public good in the Washington, DC area. I'll point out the unique aspect of community design centered around the biweekly in-person project nights.\n\n### The double duty of open source maintainership in civic tech\nDrawing on my experience as a maintainer of the CIB Mango Tree project, I'll discuss three examples of familiar open source practices. I\u2019ll highlight how in the civic tech context each of these serve a social function in addition to the engineering purpose.\n\n**Release schedule as community planning.** A regular and frequent release cycle primarily streamlines code distribution for the users. But there is a community angle to it as well: it boosts the visibility of ongoing volunteers who see their contributions ship when they can't commit long-term. Similarly, versioning code streamlines conversations about project development across diverse team members: saying `v0.10.0` becomes as much a reference to code version by maintainers as well as a community signal by project managers to coordinate around for future plans.\n\n**Continuous integration as progress visibility.** Among developers, continuous integration (CI) primarily ensures ongoing code integrity. In our project, CI also helps us with external progress visibility to the broader community beyond maintainers alone. We use CI to build executable previews of the development version. Our project and product managers can thus try out new features right as maintainers put them into the development branch.\n\n**Dependency selection as onboarding policy.** Choosing right-sized dependencies is primarily about balancing code complexity and performance, but equally about right-sizing the onboarding ramps for volunteer contributors. Choosing a dashboard framework that does not offer production-grade capabilities but comes with a simpler mental model to navigate makes it easier for new volunteers to get up to speed and contribute. When volunteer bandwidth is fleeting and turnover rate high, this becomes a non-negligible decision factor.\n\n### Learning from the design constraints of volunteer civic tech\nIn civic tech, code and technical choices serve the broader community from the start. The civic tech lens forces a much more explicit and continuous emphasis on the community needs than I anticipated coming from the scientific Python background. This led to realization that collaborative open source practices we all know need not be siphoned away as invisible labor and can form a stronger bridge between the work of the developer and the broader community.", "recording_license": "", "do_not_record": false, "persons": [{"code": "NAYCL7", "name": "Kristijan Armeni", "avatar": "https://pretalx.com/media/avatars/YQGEZR_XazUHxI.webp", "biography": "Kristijan Armeni is a research scientist with doctoral and postdoctoral training in computational neuroscience, investigating language processing in the human brain (EEG/MEG) and in artificial cognitive systems (language models). He is an advocate of open science and maintains an interest in public interest technology and civic tech. He currently helps building the CIB Mango Tree project, an interactive open source tool for analyses of social media datasets.", "public_name": "Kristijan Armeni", "guid": "9ae36192-e822-5faf-90e5-855fca07a21a", "url": "https://pretalx.com/scipy-2026/speaker/NAYCL7/"}], "links": [{"title": "Project Website", "url": "https://cibmangotree.org/", "type": "related"}, {"title": "Technical Documentation Page", "url": "https://civictechdc.github.io/cib-mango-tree/", "type": "related"}, {"title": "Project GitHub Repository", "url": "https://github.com/civictechdc/cib-mango-tree", "type": "related"}], "feedback_url": "https://pretalx.com/scipy-2026/talk/AFWXAU/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/AFWXAU/", "attachments": []}, {"guid": "4eed822d-13b9-5a66-a139-bc46d3f5500d", "code": "J9EHEQ", "id": 92473, "logo": null, "date": "2026-07-17T14:35:00-05:00", "start": "14:35", "end": "2026-07-17T15:05:00-05:00", "duration": "00:30", "room": "Thomas Swain Room", "slug": "scipy-2026-92473-on-boarding-and-retaining-maintainer-talent-for-mne-python", "url": "https://pretalx.com/scipy-2026/talk/J9EHEQ/", "title": "On-boarding and retaining maintainer talent for MNE-Python", "subtitle": "", "track": "Maintainers and Community", "type": "Talk", "language": "en", "abstract": "MNE-Python is open-source software for analyzing electrophysiological data in neuroscience. Like many projects, we struggle to retain maintainers. Finding maintainers in our user community is hard; most have little formal training in programming. To address this, we organized progressive training sprints with open applications and a participation stipend. Currently, we are onboarding four alumni of those sprints as new maintainers. We\u2019ve seen positive outcomes from this approach, but at a high cost. We are now developing a curriculum for future onboarding efforts. We hope to spark discussions with other project leaders about their efforts toward educating and retaining talented maintainers.", "description": "_Background_\nMNE-Python [1] is open-source software for analyzing electrophysiological data in neuroscience. We have a broad user base spanning neuroscience research, clinical neurology, and applied neurotechnology.\n\n_Problem statement_\nLike many open-source software projects, MNE-Python is struggling to retain maintainers and reach a comfortable Truck Factor [2]. This is aggravated by academic incentive systems which devalue open source work compared to scientific publications [3], and the fact that many MNE-Python users are not formally trained in programming. Moreover, MNE-Python\u2019s status as domain software makes it difficult for capable programmers lacking neuroscience backgrounds to fill the maintenance gap: there are too many domain-specific details that one must know to effectively maintain the codebase.\n\nInterventions\nTo increase our contributor pool, we organized two New Developer Sprints and one Intermediate Developer Sprint. These fully-remote one-week courses were open to applications from the community, and participants received a stipend. Both types of sprint involved participants pair-programming with each other or with seasoned maintainers. In the New Developer Sprints, participants chose from a list of curated issues, complemented by short presentations from invited senior community members about how they benefitted from being MNE-Python contributors earlier in their careers. For the Intermediate Sprint, participants chose larger contributions in advance and spent the whole week on them, complemented by short presentations on pertinent topics (running and writing tests, building documentation, deprecations, CIs, etc). Currently, we are onboarding four alumni of those sprints as maintainers (and providing stipends during the two-year onboarding period), and writing a reusable curriculum to support future onboarding efforts. When complete, the domain-general parts will be extracted and published separately from the MNE-Python-specific curriculum.\n\n_Comparison to previous efforts_\nPast contributors and maintainers mostly came from labs where the lab director had a vested interest in MNE-Python, or were recruited at conferences to contribute their methodological developments. In contrast, our current approach has been bottom-up: first training users how to contribute, then upskilling contributors to facilitate repeat contributions, and finally providing intensive training in maintainer-specific skills. This approach also allowed us to prioritize inclusivity in our recruitment, leading to a slight increase in the diversity of our regular contributors and maintainers. On the other hand, the sprints and maintainer onboarding were funded by three separate grants over a six-year period, and were a huge investment of existing maintainers\u2019 time.\n\n_Preliminary results_\nIn our experience, providing education on how to contribute to open source, especially information specific to our project, greatly lowers the threshold for our users to be willing to attempt a contribution. However, the incentive structure of academia still works against retaining our contributors and maintainers long-term. We hope that by publicizing our onboarding curriculum and creating other \u201ccontributor ladder\u201d resources, we will empower more users to self-educate about open-source contribution. This will hopefully increase the \u201cinput stream\u201d of contributors, and may also increase retention: by making contribution easier through upskilling, hopefully each single contribution becomes less effortful and thus more likely to be attempted.\n\n_Open questions to community_\nWith this contribution, we hope to spark a discussion among open source software maintainers about their efforts toward educating and retaining talented maintainers.\n\n_Funding acknowledgment_\nThis project has been made possible in part by grant numbers 2020-219006 and 2021-237679 from the Chan Zuckerberg Initiative DAF, an advised fund of Silicon Valley Community Foundation, and by NSF POSE award 2449064.\n\n_References_\n[1]: https://mne.tools/ and https://github.com/mne-tools/mne-python/\n[2]: Avelino, G., Passos, L., Hora, A., & Valente, M. T. (2016). A Novel Approach for Estimating Truck Factors. 2016 IEEE 24th International Conference on Program Comprehension (ICPC), 1\u201310. https://doi.org/10.1109/ICPC.2016.7503718\n[3]: Westner, B. U., McCloy, D. R., Larson, E., Gramfort, A., Katz, D. S., Smith, A. M., Delorme, A., Litvak, V., Makeig, S., Oostenveld, R., Schoffelen, J.-M., & Tierney, T. M. (2025). Cycling on the Freeway: The perilous state of open-source neuroscience software. Imaging Neuroscience, 3, imag_a_00554. https://doi.org/10.1162/imag_a_00554", "recording_license": "", "do_not_record": false, "persons": [{"code": "WJ8WEY", "name": "Daniel McCloy", "avatar": "https://pretalx.com/media/avatars/CRQRH8_WpRYWRO.webp", "biography": "I am a developer of open-source scientific software, and a scientist trained in acoustic phonetics, speech perception, and auditory neuroscience. My scientific interest broadly centers on the perception and representation of speech sounds. I'm (probably) most known for my work on MNE-Python.", "public_name": "Daniel McCloy", "guid": "dc81d5a5-60e1-5682-bc99-eab0ae6364ea", "url": "https://pretalx.com/scipy-2026/speaker/WJ8WEY/"}, {"code": "UTN9BQ", "name": "Eric Larson", "avatar": "https://pretalx.com/media/avatars/JXSDCU_1pBVmV1.webp", "biography": "Research Scientist at the Institute for Learning and Brain Sciences, University of Washington, Seattle, WA.", "public_name": "Eric Larson", "guid": "9725bbd0-f428-5a8f-ae4b-da5d5f4e7b70", "url": "https://pretalx.com/scipy-2026/speaker/UTN9BQ/"}, {"code": "HPSEMY", "name": "Britta Westner", "avatar": null, "biography": "I am an Assistant Professor at the Donders Institute and Radboudumc in The Netherlands. My current research focuses mainly on the intersection of language and memory. I work with human electrophysiological data and have a focus on data analysis methods such as source reconstruction and decoding. I am enthusiastic about open source and am part of the core developer team of MNE-Python since 2019. Since 2024, I am also part of the newly-formed steering council of MNE-Python.", "public_name": "Britta Westner", "guid": "8346f5f2-794a-530f-86c1-a6fc39719eab", "url": "https://pretalx.com/scipy-2026/speaker/HPSEMY/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/J9EHEQ/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/J9EHEQ/", "attachments": []}, {"guid": "6c10b458-bfc7-55e9-bbfe-59180c005055", "code": "CKB9FU", "id": 102464, "logo": null, "date": "2026-07-17T17:45:00-05:00", "start": "17:45", "end": "2026-07-17T18:40:00-05:00", "duration": "00:55", "room": "Thomas Swain Room", "slug": "scipy-2026-102464-the-academy-and-industry-building-interdisciplinary-relationships", "url": "https://pretalx.com/scipy-2026/talk/CKB9FU/", "title": "The Academy and Industry: Building Interdisciplinary Relationships", "subtitle": "", "track": "Birds of a Feather (BoFs)", "type": "Birds-of-a-Feather (Bof)", "language": "en", "abstract": "This session will be a community discussion centered around developing interdisciplinary relationships between academic research institutions and non-academic institutions (industry, government, etc.).  Using a new initiative from the Center for Interdisciplinary Exploration and Research in Astrophysics (CIERA) at Northwestern University as a model to build upon, develop, and learn from, we hope to build a shared understanding of how those engaged in scientific and technological development broadly would benefit from such efforts. The initiative \u2013 CIERA\u2019s Tech Council \u2013 brings together a group of professionals (many of which with academic backgrounds) to serve as scientific collaborators, technical experts, community liaisons, and mentors. Starting from this point, we ask: What does it mean to create a rich and thriving ecosystem around an academic institution that translates technological expertise into scientific progress, increases accessibility of advanced tools and research, and builds community across varying career paths? As the SciPy Conference is a hub for interdisciplinary knowledge and skill sharing, it is a perfect place to hold such a discussion.", "description": "", "recording_license": "", "do_not_record": false, "persons": [{"code": "TLWUXR", "name": "Alexandra Mannings", "avatar": null, "biography": null, "public_name": "Alexandra Mannings", "guid": "372400bc-9d6a-5d8f-ba41-7057e0da9035", "url": "https://pretalx.com/scipy-2026/speaker/TLWUXR/"}, {"code": "YNPFNQ", "name": "Caleb Krueger", "avatar": null, "biography": "Post-Baccalaureate Research Fellow at Northwestern University, Center for the Interdisciplinary Exploration and Research in Astrophysics (CIERA). Assistant Program Coordinator for the new CIERA Tech Council. Studying environmental science at the University of Chicago (M.S. 2027 exp.).", "public_name": "Caleb Krueger", "guid": "802a1f09-4d0c-5dba-80ca-62f671633473", "url": "https://pretalx.com/scipy-2026/speaker/YNPFNQ/"}], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/CKB9FU/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/CKB9FU/", "attachments": []}], "Virtual Sessions": [{"guid": "e940cf59-7f9e-5229-ba48-bd4fac1c6772", "code": "CF3PMY", "id": 103032, "logo": null, "date": "2026-07-17T16:40:00-05:00", "start": "16:40", "end": "2026-07-17T17:45:00-05:00", "duration": "01:05", "room": "Virtual Sessions", "slug": "scipy-2026-103032-exclusively-on-zoom-virtual-speed-networking", "url": "https://pretalx.com/scipy-2026/talk/CF3PMY/", "title": "(Exclusively on Zoom) Virtual Speed Networking", "subtitle": "", "track": null, "type": "Social Event", "language": "en", "abstract": "You'll be randomly paired with another conference attendee for a 5-minute chat. Non-cheesy icebreakers will be provided. Virtual and in-person attendees welcome!\n\nZoom link will be provided in the SciPy 2026 conference Slack team", "description": "", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://pretalx.com/scipy-2026/talk/CF3PMY/feedback/", "origin_url": "https://pretalx.com/scipy-2026/talk/CF3PMY/", "attachments": []}]}}, {"index": 6, "date": "2026-07-18", "day_start": "2026-07-18T04:00:00-05:00", "day_end": "2026-07-19T03:59:00-05:00", "rooms": {}}, {"index": 7, "date": "2026-07-19", "day_start": "2026-07-19T04:00:00-05:00", "day_end": "2026-07-20T03:59:00-05:00", "rooms": {}}]}}}