{"$schema": "https://c3voc.de/schedule/schema.json", "generator": {"name": "pretalx", "version": "2026.3.0.dev0", "url": "https://cfp.pydata.org"}, "schedule": {"url": "https://cfp.pydata.org/pydata-eindhoven-2025/schedule/", "version": "0.11", "base_url": "https://cfp.pydata.org", "conference": {"acronym": "pydata-eindhoven-2025", "title": "PyData Eindhoven 2025", "start": "2025-12-09", "end": "2025-12-09", "daysCount": 1, "timeslot_duration": "00:05", "time_zone_name": "UTC", "colors": {"primary": "#4c9cb4"}, "rooms": [{"name": "Ernst-Curie", "slug": "5038-ernst-curie", "guid": "beaaa232-c1cd-5e55-9486-89a54685807e", "description": "PySport Room", "capacity": 130}, {"name": "Auditorium", "slug": "5037-auditorium", "guid": "910e86ee-7648-59d0-9aaa-1276772e8e5d", "description": "Main Room", "capacity": 300}, {"name": "Planck-Bohr", "slug": "5039-planck-bohr", "guid": "d766d663-e4a5-5fa6-a64a-747c6928700e", "description": null, "capacity": 80}], "tracks": [{"name": "Data Engineering", "slug": "6089-data-engineering", "color": "#F50244"}, {"name": "AI/Machine Learning/GenAI", "slug": "6090-aimachine-learninggenai", "color": "#000000"}, {"name": "Sports Analytics hosted PySport", "slug": "6095-sports-analytics-hosted-pysport", "color": "#084CB1"}], "days": [{"index": 1, "date": "2025-12-09", "day_start": "2025-12-09T04:00:00+00:00", "day_end": "2025-12-10T03:59:00+00:00", "rooms": {"Auditorium": [{"guid": "c08fbf75-c60c-5e85-8c8d-d82329df2630", "code": "LZRVAL", "id": 84665, "logo": null, "date": "2025-12-09T09:00:00+00:00", "start": "09:00", "end": "2025-12-09T09:10:00+00:00", "duration": "00:10", "room": "Auditorium", "slug": "pydata-eindhoven-2025-84665-opening", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/LZRVAL/", "title": "Opening", "subtitle": "", "track": null, "type": "Opening", "language": "en", "abstract": "Welcome to PyData Eindhoven 2025", "description": "Welcome to PyData Eindhoven 2025", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/LZRVAL/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/LZRVAL/", "attachments": []}, {"guid": "54373263-a549-5890-80bf-cf632f8ce90f", "code": "R7NMXB", "id": 84667, "logo": null, "date": "2025-12-09T09:10:00+00:00", "start": "09:10", "end": "2025-12-09T09:55:00+00:00", "duration": "00:45", "room": "Auditorium", "slug": "pydata-eindhoven-2025-84667-opening-keynote-by-maurits-hendriks-the-science-of-showing-up-why-confrontation-selection-and-data-matter-in-the-journey-to-elite-performance", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/R7NMXB/", "title": "Opening Keynote by Maurits Hendriks: \u201cThe Science of Showing Up\u201d: Why Confrontation, Selection, and Data Matter in the Journey to Elite\u00a0Performance", "subtitle": "", "track": null, "type": "Keynote", "language": "en", "abstract": "In elite sport, winning depends on the ability to deliver at the precise moment it matters. The Science of Showing Up examines how athletes build systems that enable performance on demand\u2014through daily iteration, selective focus, and confronting real progress. Technology and data now amplify this process. But the key question remains: does data help us win? Only when it sharpens decision-making, accelerates learning, and keeps us connected to the essence of the sport. Data is a tool\u2014not the destination.", "description": "Drawing on decades in Olympic and professional high-performance environments, this keynote reveals how the best athletes and teams design structures to peak at exactly the right moment. Ultimately, the science of showing up is the art of combining human resilience with intelligent systems\u2014being ready, physically, mentally, and organizationally, when the world\u00a0is\u00a0watching.", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/R7NMXB/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/R7NMXB/", "attachments": []}, {"guid": "ec18c0c1-5d36-57d6-9298-65a19e455786", "code": "XMJ9LX", "id": 82530, "logo": null, "date": "2025-12-09T10:00:00+00:00", "start": "10:00", "end": "2025-12-09T10:30:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-82530-scaling-retail-planning-at-ikea-orchestrating-sales-fulfillment-and-capacity-assessment-with-metaflow", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/XMJ9LX/", "title": "Scaling Retail Planning at IKEA: Orchestrating Sales, Fulfillment and Capacity Assessment with Metaflow", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "At IKEA, retail planning is a complex chain of processes, from sales forecasting to fulfillment and capacity assessment, that involve multiple teams. Each team builds their own predictive models independently, yet their outputs depend on one another to ensure a concise planning chain. \n\nIn this talk, we will show how IKEA uses Metaflow, an open-source framework for building and managing real-life ML, to orchestrate and connect the forecasting pipelines for more than thirty countries. We\u2019ll discuss how Metaflow helps align independent teams, improve readability, and enable reproducible workflows and scale. \n\nYou will leave with practical approaches for an aligned team workflow and concrete patterns for orchestrating ML/AI pipelines.", "description": "Retail planning at IKEA is more than just predicting sales, it is about connecting various forecasts that inform and depend on one another: \n\nSales forecasting: How much will customers buy? \nFulfillment planning: Can we ensure availability? \nCapacity assessment: Do our stores and distribution centers have enough capacity to handle the volume? \nEach of these domains has its own models and ways of working. Previously, these teams built and deployed everything independently. Making it difficult to align predictions and maintain visibility across the retail chain. To improve this process, IKEA adopted Metaflow as a data science platform. Metaflow provides a framework for developing, running, and monitoring pipelines. \n\nThe patterns and solutions shared during the presentation can apply to any organization dealing with (complex) workflows, interconnected predictions, or resource/scaling challenges. \n\n \n\nTarget audience: \n\nData scientists, data engineers, ML practitioners, and technical leads interested in: \n\nWorkflow orchestration and reproducibility. \nCross-team collaboration in data science. \nScalable forecasting systems in production environments. \n \n\nBackground knowledge: Basic understanding of data pipelines and workflow orchestration. No Metaflow experience required. \n\n \n\nTalk outline: \n\n0-10 minutes: \n\nIntroduction to IKEA\u2019s Retail Planning \nProblems that you encounter coordinating these processes in 30+ countries \nWhy traditional pipelines did not succeed \n10-15 minutes: \n\nWhat we needed: A system that could handle complex dependencies, scale and let non-engineers contribute safe and easy \nIntroduce Metaflow as an OS framework solution \n \n15-25 minutes: \n\nQuick introduction how Metaflow works in practice \nShow how Metaflow handles the Retail Planning chain \nShow how Metaflow works together with Argo Workflows \nShow how Metaflow helps us scale and manage our resources at the same time \n \n25-30 minutes: recap and questions", "recording_license": "", "do_not_record": false, "persons": [{"code": "RAPLWQ", "name": "Yannick Mariman", "avatar": "https://cfp.pydata.org/media/avatars/RAPLWQ_yCTfANc.webp", "biography": "I'm Yannick Mariman, a data engineer working at Pipple, working with IKEA to build an improved Retail Planning platform. Earlier in my career, I saw good data solutions failing because of a lack of proper workflows and orchestration. This led me to become more interested in tangible solutions that focus on robustness and stability.", "public_name": "Yannick Mariman", "guid": "997f674e-6b69-587a-b634-0e6f44852edb", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/RAPLWQ/"}, {"code": "XSS3KZ", "name": "Harlley de Lima", "avatar": "https://cfp.pydata.org/media/avatars/XSS3KZ_EzOhDNj.webp", "biography": "Harley is a Machine Learning Engineering at IKEA, where he works on developing data-driven solutions that support and optimize retail planning processes. With a solid foundation in both software engineering and applied machine learning, he enjoys turning complex business challenges into scalable systems that improve decision-making and deliver real impact.", "public_name": "Harlley de Lima", "guid": "7442799e-fa91-565f-baf2-93cd5c3433c8", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/XSS3KZ/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/XMJ9LX/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/XMJ9LX/", "attachments": []}, {"guid": "0f122366-5cdb-5acc-bcaf-90fc6346e402", "code": "NHFJJW", "id": 82089, "logo": null, "date": "2025-12-09T10:45:00+00:00", "start": "10:45", "end": "2025-12-09T11:15:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-82089-ai-powered-web-scraping-from-data-collection-to-strategic-insights", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/NHFJJW/", "title": "AI-Powered Web Scraping: From Data Collection to Strategic Insights", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "Companies today are hungry for external data to stay competitive, but actually getting and making sense of that data isn\u2019t easy. Standard web scraping often produces messy or incomplete results, and modern anti-bot systems make reliable collection even tougher.\n\nIn this talk, I\u2019ll share how pairing Python\u2019s scraping frameworks (like Scrapy, Playwright, and Selenium) with AI/ML can turn raw, unstructured data into clear, actionable insights.\n\nWe\u2019ll look at:\n\n1) How to build scrapers that still work in 2025.\n\n2) Ways to use AI to automatically clean, enrich, and classify data.\n\n3) Real-world applications of sentiment analysis for reviews and social media.\n\n4) Case studies showing how SMEs have used these pipelines to sharpen marketing and product strategies.\n\nBy the end, you\u2019ll see how to design pipelines that don\u2019t just gather data, but deliver real strategic value. The session will focus on practical Python tools, scalable deployment (Airflow, Kubernetes, cloud platforms), and key lessons learned from hands-on projects at the intersection of scraping and AI.", "description": "Collecting web data is getting harder\u2014between messy datasets and stronger bot defenses, traditional scraping often falls short. This talk shows how combining Python tools (Scrapy, Playwright, Selenium) with AI/ML can turn raw, unstructured data into clear insights. We\u2019ll explore practical best practices, real-world case studies, and how SMEs use sentiment analysis pipelines to make smarter marketing and product decisions.", "recording_license": "", "do_not_record": false, "persons": [{"code": "QJFG93", "name": "Yevhenii", "avatar": "https://cfp.pydata.org/media/avatars/QJFG93_pSC53Le.webp", "biography": "Senior Python Developer and team lead with over 8 years of experience building large-scale data acquisition systems for international companies. Led teams in developing resilient scrapers, AI-powered sentiment analysis platforms, and predictive models for industries ranging from e-commerce to finance. Passionate about turning raw web data into actionable insights, sharing hands-on lessons from real-world projects at the intersection of Python, data engineering, and machine learning.", "public_name": "Yevhenii", "guid": "7106314b-d6e1-5106-93ee-b33debc0eaf1", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/QJFG93/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/NHFJJW/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/NHFJJW/", "attachments": []}, {"guid": "db845e86-347d-5f82-b38d-fb05526ac6f8", "code": "LURJEK", "id": 82556, "logo": null, "date": "2025-12-09T11:20:00+00:00", "start": "11:20", "end": "2025-12-09T11:50:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-82556-efficient-time-series-forecasting-with-thousands-of-local-models-on-databricks", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/LURJEK/", "title": "Efficient Time-Series Forecasting with Thousands of Local Models on Databricks", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "In industries like energy and retail, forecasting often requires local models when each time series has unique behavior \u2014 though training thousands of them can be overwhelming. However, training and managing thousands of such models presents scalability and operational challenges. This talk shows how we scaled local models on Databricks by leveraging the Pandas API on Spark, and shares practical lessons on storage, reuse, and scaling challenges to make this approach efficient when it\u2019s truly needed", "description": "Industries like energy, retail, and logistics often face a critical problem: forecasting demand or consumption for thousands of entities, each with unique patterns. Global models may miss local nuances, while training individual models can overwhelm traditional pipelines.\nIn this talk, we\u2019ll explore a practical and scalable solution built on Databricks\u2014using pandas API on spark to train and predict thousands of local models in parallel.\nYou\u2019ll discover how to:\n- Train and serialize ML models per group efficiently.\n- Use binary model storage to overcome MLflow\u2019s RPS limits.\n- Generate and forecast future data asynchronously at scale.\nWe\u2019ll discuss the trade-offs between MLflow Registry, Unity Catalog, and inline model execution, and how this approach powers forecasting for thousands of groups in the energy sector.\nWhether you\u2019re a data scientist or ML engineer, you\u2019ll leave with actionable ideas for scaling your own forecasting workflows on Databricks", "recording_license": "", "do_not_record": false, "persons": [{"code": "7U9BCY", "name": "Daria Mustafina", "avatar": null, "biography": null, "public_name": "Daria Mustafina", "guid": "eb62eb3a-16d3-5453-9a79-8820b1aa5a9d", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/7U9BCY/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/LURJEK/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/LURJEK/", "attachments": []}, {"guid": "0889e849-14f9-5582-b485-441c5f9f7e2e", "code": "9FRSYC", "id": 82392, "logo": "https://cfp.pydata.org/media/pydata-eindhoven-2025/submissions/9FRSYC/20250826100916_Z1vy6UO.jpg", "date": "2025-12-09T12:50:00+00:00", "start": "12:50", "end": "2025-12-09T13:20:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-82392-finding-trash-in-waste", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9FRSYC/", "title": "Finding trash in waste", "subtitle": "", "track": "AI/Machine Learning/GenAI", "type": "Talk", "language": "en", "abstract": "At waste transfer stations for source separated packaging waste incoming waste trucks are visually inspected on objects that could disturb the sorting and recycling of the truck load. This is a manual procedure and in case the number of disturbing items is too high, the part of the  truck load needs to be removed. Currently, 8.5 % of the truck loads is rejected. This leads to loss of valuable plastics for recycling. We have investigated the automation of this inspection using cameras and vision foundation models. Inhouse, we developed a data pipeline where waste items are first detected, then segmented and eventually classified whether they belong to this waste stream using anomaly detection. The accepted material continues to a plastic recovery facility. This approach has led to a proof-of-principle with the potential to be implemented as a pilot-scale at a waste transfer station. The project is part of the research program \u2018MultiPurpose Plastic Sorting\u2019 subsidised by TKI Energy & Industry.", "description": "After waste collection of lightweight packaging waste, in the Netherlands known as PMD (plastic-, metaal- en drankkartons), the incoming material is visually inspected by an operator to extract contaminants that disturb the sorting and recycling. Typical contaminants are non-packaging material and large rigids and foils.\n\nAt the applied research institute NTCP, we investigated how to automate the detection process in order to have an objective method to inspect a PMD waste stream. This improves consistency and thereby quality and reduces required time needed for a thorough assessment. We decided to use cameras in a top view of the pile of items, as the inspection is performed. For every item, we first need to locate it in the image and segment the pixels of it. We built a data pipeline using vision foundation models.\n\nTo determine whether an item is considered a contaminant or PMD packaging, we found that the class of contaminants had a too broad variety to build a classification network. Therefore, we chose to explore the feasibility of anomaly detection models. Several dedicated anomaly detection models were trained. Finally, we learned that using a combination of foundation models for both visual as well textual features, resulted in a suitable distribution to distinguish contaminants from PMD packaging.\n\nWith a prototype set-up, we have showcased a proof-of-principle of a well working and promising automatic detection system on actual waste, given the variety in both PMD and\u00a0contaminants. images, given the variety in both valid waste and unwanted items.", "recording_license": "", "do_not_record": false, "persons": [{"code": "U873CM", "name": "Tom Koopen", "avatar": "https://cfp.pydata.org/media/avatars/U873CM_HqRCLMw.webp", "biography": "Tom Koopen is a highly experienced professional in computer vision and deep learning. He holds a Master\u2019s Degree in Applied Physics from the University of Twente, specializing in optical measurement systems. \n\nWith over 25 years of experience in computer vision, Tom has worked with various companies in the Netherlands. Since 2013 he is an entrepeneur at \u201cde tijdelijke expert\u201d, assisting customers with the application of computer vision technology, focusing on measurements, identification, and sorting of products. Recently he founded \u201ctextilemining.eco\u201d to build innovative machines for textile recycling.  \n\nHe has designed lighting systems, selected and optimized cameras, and written software for thousands of hours. Some of his notable projects include inspecting plastic crates for contamination, improving the sorting of plastics, metals and flower bulbs.  He developed a 3D scanner to recognize roof tiles for Luijtgaarden B.V. and measured colors during high-speed printing processes at QI Press Controls. Oh, and don\u2019t forget the beer bottle inspection with 10 per second about 20 years ago.", "public_name": "Tom Koopen", "guid": "5beb07fb-bebd-5771-9c15-82dc1168bcd1", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/U873CM/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9FRSYC/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9FRSYC/", "attachments": []}, {"guid": "a2820937-4ece-5b2f-b913-8a84c00c7ea8", "code": "RUTXQ3", "id": 82472, "logo": null, "date": "2025-12-09T13:25:00+00:00", "start": "13:25", "end": "2025-12-09T13:55:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-82472-from-1m-license-to-in-house-success-how-we-built-a-real-time-recommendation-system-and-saved-millions-doing-it", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/RUTXQ3/", "title": "From \u20ac1M License to In-House Success: How We Built a Real-Time Recommendation System and Saved Millions Doing It", "subtitle": "", "track": "AI/Machine Learning/GenAI", "type": "Talk", "language": "en", "abstract": "When we at Bol decided to personalize campaign banners, we did what many companies do: bought an expensive solution. As a software engineering team with zero data science experience, we integrated a third-party recommender system for \u20ac1 million annually, built the cloud infrastructure, and waited for results. After our first season, the data told a harsh truth\u2014the third-party tool wasn't delivering value proportional to its cost. We faced a crossroads: accept mediocrity or build our own solution from scratch, tailored to our requirements and architecture.\nWe'll walk you through our journey of building a more intelligent and flexible recommendation system from the ground up, and how this journey saved us over a million euros per year. We will share the incremental steps that shaped our journey, alongside the valuable lessons learned along the way", "description": "We started our journey of developing an in-house personalized banner recommendation system at Bol, by replacing an expensive third-party tool. We are going to talk about our decisions from ML to explore-exploit balance of our recommendations, sharing our story of building the system from scratch.", "recording_license": "", "do_not_record": false, "persons": [{"code": "TAM9CE", "name": "ALI KOHAN", "avatar": null, "biography": "Ali Kohan: \nData Scientist and AI consultant with a background in computer vision and deep learning. After several years working as a computer vision engineer, I am now focused on building recommendation and personalization systems. Passionate about applied AI, I enjoy exploring how intelligent systems can better understand and adapt to human behavior.\n\nReza Ebrahimpour:\nSoftware engineer with a decade of experience building scalable, cloud-native systems in the JVM ecosystem. Passionate about microservice architectures, data-driven design, and high-availability services. Recently focused on integrating machine learning and AI solutions into production systems at Bol.", "public_name": "ALI KOHAN", "guid": "7e916a17-453d-51c9-b17f-cee3cca947ea", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/TAM9CE/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/RUTXQ3/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/RUTXQ3/", "attachments": []}, {"guid": "28bc8284-aedd-5a68-ad19-6b90340a780d", "code": "JBPQS9", "id": 85836, "logo": null, "date": "2025-12-09T14:10:00+00:00", "start": "14:10", "end": "2025-12-09T14:40:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-85836-federated-data-centralized-action-a-governance-model-powered-by-lakewatch-and-databricks-system-tables", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/JBPQS9/", "title": "Federated Data, Centralized Action: A Governance Model Powered by Lakewatch and Databricks System Tables", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "In federated data architectures, balancing team autonomy with accountability is a critical challenge. This presentation introduces our 3 pillar-based governance model that transforms raw Databricks System Tables into actionable scorecards for cost efficiency and best practices.", "description": "Learn how this approach drives centralized action, empowers domain teams, and eliminates governance bottlenecks while optimizing Databricks-related costs.", "recording_license": "", "do_not_record": false, "persons": [{"code": "EZYTL7", "name": "Cristiano Cortez Da Rocha", "avatar": "https://cfp.pydata.org/media/avatars/EZYTL7_Aa4xi2S.webp", "biography": "Cristiano Cortez Da Rocha", "public_name": "Cristiano Cortez Da Rocha", "guid": "699dee35-e714-5c6e-b047-da473ce3b39d", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/EZYTL7/"}, {"code": "JJWNQR", "name": "Frank Chidi Mbonu", "avatar": null, "biography": "Frank Chidi Mbonu", "public_name": "Frank Chidi Mbonu", "guid": "9b718ec6-aaf9-5c8e-a1a3-f6a55002c073", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/JJWNQR/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/JBPQS9/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/JBPQS9/", "attachments": []}, {"guid": "e6312fd3-bd18-5df6-ba2e-3c007d32ea4d", "code": "TN83BD", "id": 85302, "logo": null, "date": "2025-12-09T14:45:00+00:00", "start": "14:45", "end": "2025-12-09T15:15:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-85302-building-deploying-and-managing-ai-agents-at-scale", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/TN83BD/", "title": "Building, Deploying and Managing AI Agents at Scale", "subtitle": "", "track": "AI/Machine Learning/GenAI", "type": "Talk", "language": "en", "abstract": "This session delivers a blueprint for building, deploying, and managing agents in a secure, scalable, and cost-effective manner on Google Cloud, bridging the critical gap between development and operations.", "description": "The hype around Agentic AI is real, but so are the challenges. Creating a single agent is straightforward.. But how do you scale effectively? If you don't do it properly, critical issues will arise: spiraling costs, fragmented oversight, and major governance risks. Your agents need to be governed by a set of fundamental design principles. In other words, you need a platform. But what does a \u201cplatform\u201d truly mean in practice? Sander will provide a pragmatic deep dive\u2014no hype, no superficial demos\u2014into the key pillars of this foundational platform: - Infrastructure & Serving: reliable serving patterns with Vertex AI Agent Engine. - State & Memory: Implementing session management and long-term memory for personalized interactions using Google Cloud's MemoryBank. - Orchestration: Using frameworks like Google's Agent-Development-Kit (ADK) - Integration: We will take a look into the different options to connect your agents with external services and data. - Security: Highlighting current authentication and authorization challenges for autonomous agents. - Governance & Cost Control: Establishing robust guardrails for cost, security, and compliance from day one. - Observability & Evaluation: Implementing rigorous monitoring and evaluation to prove business value and ensure reliability. This talk offers a pragmatic, engineering-focused approach to building a resilient, manageable, and impactful AI agent ecosystem on Google Cloud.", "recording_license": "", "do_not_record": false, "persons": [{"code": "KTJHNJ", "name": "Sander van Donkelaar", "avatar": "https://cfp.pydata.org/media/avatars/KTJHNJ_rpKjSzl.webp", "biography": null, "public_name": "Sander van Donkelaar", "guid": "86abddf3-f35d-5415-b334-4d89a5fa4035", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/KTJHNJ/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/TN83BD/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/TN83BD/", "attachments": []}, {"guid": "ecb7edc2-c1d4-50c4-9517-6f84cff1bd83", "code": "8YNNH9", "id": 85703, "logo": null, "date": "2025-12-09T15:20:00+00:00", "start": "15:20", "end": "2025-12-09T15:50:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-85703-responsible-human-agent-ecosystem-making-mission-critical-systems-situationally-aware", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/8YNNH9/", "title": "Responsible Human-Agent Ecosystem: Making Mission Critical Systems Situationally Aware", "subtitle": "", "track": "AI/Machine Learning/GenAI", "type": "Talk", "language": "en", "abstract": "Modern mission-critical systems operate in environments that change by the second. To keep up, they need more than static maps and siloed data, they need true situational awareness. This talk explores how we are building a Responsible Human-Agent Ecosystem that combines high-resolution 3D geospatial data, real-time sensor fusion, and AI-driven agents to help mission-critical platforms understand the world the way humans do- but faster and at scale.", "description": "Using technologies such as 3D digital twins, computer vision, and predictive ML models, our systems create a live, interactive view of the internal working of a large, multi-system operational environment. AI agents can detect anomalies, track threats, reason about context, and support operators with transparent, trustworthy insights.\nFor developers and technologists, this work sits at the intersection of AI, geospatial engineering, high-performance computing, and human-centered design offering a chance to build systems that truly matter. Join us to see how cutting-edge tech is shaping the next generation of situationally aware capabilities.", "recording_license": "", "do_not_record": true, "persons": [{"code": "WZTMTC", "name": "Ipsit Dash", "avatar": "https://cfp.pydata.org/media/avatars/WZTMTC_3KJlyZ7.webp", "biography": "Ipsit Dash is a Director Consulting Expert at CGI Netherlands, specializing in AI and data analytics. As a SAFe SPC with nearly ten years of experience at the intersection of digital transformation, data analytics, and AI-driven innovation, he helps organizations design and implement scalable, future-proof data and AI ecosystems. He translates strategic ambitions into clear roadmaps, measurable KPIs, and agile execution, with strong stakeholder alignment and robust governance.\n\nIpsit began his career at CGI in 2016 as a software engineer and progressed to solution architect and then program lead consulting for clients in financial services, government, aerospace, and high-tech. He builds use-case portfolios, develops targeted learning paths for teams, and drives multidisciplinary delivery\u2014from pilot to scalable rollout\u2014with demonstrable value creation and effective risk management.", "public_name": "Ipsit Dash", "guid": "e4d85b91-4321-53b4-bc65-6fdf689ad1f3", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/WZTMTC/"}, {"code": "MJKQTT", "name": "Bart-Peter Smit", "avatar": "https://cfp.pydata.org/media/avatars/MJKQTT_AhsgCAK.webp", "biography": "Bart-Peter is an enthusiastic and inquisitive Geo-ICT specialist who enjoys applying his knowledge to innovative projects. His interests in Situation Awareness, 3D scanning, Human Factors, Data Governance, and Data-Driven Working are particularly valuable. In his role as Business Consultant, he often finds himself working on projects at the intersection of people and technology. He often works with clients to chart a course, developing a vision for the project and its value. Bart-Peter is most enthusiastic about helping improve human decision-making processes, enabling people and organizations to make even better decisions based on data.\n\nFinally, Bart-Peter enjoys being involved in Business Development, mentoring students and startups, Corporate Social Responsibility, and Geo-ICT practice. He gives guest lectures at universities and is also involved in the 3D scanning of tombs in Egypt.", "public_name": "Bart-Peter Smit", "guid": "ad2111c5-bc9b-5ba1-839a-0729e1f1b793", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/MJKQTT/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/8YNNH9/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/8YNNH9/", "attachments": []}, {"guid": "847a2755-a969-5b3c-b108-55ef00a20e5d", "code": "XTSZYZ", "id": 84668, "logo": null, "date": "2025-12-09T15:55:00+00:00", "start": "15:55", "end": "2025-12-09T16:25:00+00:00", "duration": "00:30", "room": "Auditorium", "slug": "pydata-eindhoven-2025-84668-hier-is-vincent-een-en-al-oor", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/XTSZYZ/", "title": "Hier is Vincent. E\u00e9n en al oor.", "subtitle": "", "track": null, "type": "Keynote", "language": "en", "abstract": "Can we unlock heritage by engaging in dialogue with it?", "description": "Can the public engage directly with Vincent van Gogh? Based on more than 800 letters in which he describes his circumstances, work and emotions, a so-called Digital Twin has been developed. Enriched with additional sources and skills, and created with respect for copyrights, ethics and privacy. This gives the public the opportunity to experience his thoughts, struggles, vulnerability, and passion in a vivid and meaningful encounter.", "recording_license": "", "do_not_record": false, "persons": [{"code": "CAREKC", "name": "Rob Mulder", "avatar": "https://cfp.pydata.org/media/avatars/CAREKC_SNLqB7K.webp", "biography": "Rob Mulder", "public_name": "Rob Mulder", "guid": "cfb2a646-2c62-5d07-90e7-7d646c57ed4e", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/CAREKC/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/XTSZYZ/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/XTSZYZ/", "attachments": []}, {"guid": "e1839d6f-6e8e-5ba4-969c-10cbc982d51b", "code": "PHUKC8", "id": 84670, "logo": null, "date": "2025-12-09T16:25:00+00:00", "start": "16:25", "end": "2025-12-09T16:40:00+00:00", "duration": "00:15", "room": "Auditorium", "slug": "pydata-eindhoven-2025-84670-closing", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/PHUKC8/", "title": "Closing", "subtitle": "", "track": null, "type": "Closing", "language": "en", "abstract": "Closing", "description": "Closing", "recording_license": "", "do_not_record": false, "persons": [], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/PHUKC8/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/PHUKC8/", "attachments": []}], "Ernst-Curie": [{"guid": "7442d293-6c23-5f52-8fb3-30572dd91143", "code": "7GEDFP", "id": 79874, "logo": null, "date": "2025-12-09T10:00:00+00:00", "start": "10:00", "end": "2025-12-09T10:30:00+00:00", "duration": "00:30", "room": "Ernst-Curie", "slug": "pydata-eindhoven-2025-79874-developing-a-nation-wide-padel-rating-system-a-data-driven-approach", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/7GEDFP/", "title": "Developing a Nation-Wide Padel Rating System: A Data-Driven Approach", "subtitle": "", "track": "Sports Analytics hosted PySport", "type": "Talk", "language": "en", "abstract": "Padel has been one of the fastest-growing sports in the Netherlands in recent years. While it initially benefited from the rating facilities of its \u2018big brother\u2019 tennis, the KNLTB decided in 2024 to develop a dedicated, tailor-made rating system for padel, which has been in effect since 2025. The development process involved extensive analyses, simulations, and probability modeling on data from more than 300,000 padel matches, complemented by recommendations from the field.\n\nIn this presentation, the audience will be taken through the technical development process, as well as the unique characteristics of padel that were crucial in creating an effective rating system.", "description": "Elo systems are widely used to estimate player levels in situations where everyone plays against different opponents. The principle is simple: after each match, your rating shifts based on the expected win probability. Beating a stronger team increases your rating more than beating a weaker one.\nIn the Netherlands, padel initially used the existing tennis rating system. But tennis and padel have different characteristics that make this challenging. In tennis, a rating gap often translates quite predictably into a win probability. In padel, more factors come into play: mixed-gender teams, asymmetric pairings (e.g., two men versus a man and a woman), and the sport\u2019s specific match dynamics. \nDeveloping an effective padel rating system is a real-world problem. We found that data analysis is crucial, but without sport-specific knowledge and context, conclusions can be misleading.\nWe analyzed data from more than 300,000 padel matches and tested a range of models\u2014from standard Elo variants to fully custom-built algorithms. Each system was evaluated on, amongst other things,  two core metrics: speed (how quickly a system converges to a realistic player level) and efficiency (how stable ratings remain without unnecessary fluctuations).\nThis presentation will walk through the technical process, the model comparisons, and the padel-specific decisions that led to a rating system designed to work in real competition.", "recording_license": "", "do_not_record": false, "persons": [{"code": "N9BZEL", "name": "Max Brouwer", "avatar": "https://cfp.pydata.org/media/avatars/N9BZEL_T1EY6Qo.webp", "biography": "Max Brouwer (MSc) is a Data Scientist at the Royal Dutch Lawn Tennis Association (KNLTB). He graduated cum laude from the University of Groningen in 2021, earning a Master\u2019s degree in Sport Science with a focus on applying computer vision and machine learning techniques to tennis.\n\nAt the KNLTB, Max contributes to both the Sport Science Team and the Data Team. In the Sport Science Team, he works on projects involving sensor technology, load monitoring, and match analysis. In the Data Team, his expertise extends to recreational tennis and padel, including the development of the new padel rating system.", "public_name": "Max Brouwer", "guid": "2ea35c9e-9116-56b5-8af9-a672c31ab151", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/N9BZEL/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/7GEDFP/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/7GEDFP/", "attachments": []}, {"guid": "751df793-9198-5995-81ce-65a070a441fd", "code": "38RZBP", "id": 82527, "logo": null, "date": "2025-12-09T11:20:00+00:00", "start": "11:20", "end": "2025-12-09T11:50:00+00:00", "duration": "00:30", "room": "Ernst-Curie", "slug": "pydata-eindhoven-2025-82527-identifying-playstyles-in-football-through-spatial-networks", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/38RZBP/", "title": "Identifying playstyles in football through spatial networks", "subtitle": "", "track": "Sports Analytics hosted PySport", "type": "Talk", "language": "en", "abstract": "Breaking away from traditional manual video analysis, this talk introduces a data-driven approach to automatically identify football playstyles in key moments before a shot on goal , using tracking and event data. By applying network science , which studies relationships and interactions within complex systems, we objectively analyze attacking and defensive strategies. Key spatial network metrics are used to reveal diverse playstyles through clustering techniques. The session concludes with insights into the results and possible applications of these findings in football analysis.", "description": "Football analysts and coaches rely on manual video analysis to identify playstyles and tactical behaviors, a process that is time-consuming and often subjective. This talk introduces an innovative approach that leverages tracking and event data to automatically identify playstyles in the critical moments leading up to a shot on goal. By employing advanced network science and football analytics, this method offers a comprehensive and objective analysis of both attacking and defensive strategies.\n\nIn this presentation, we will start with the introduction of spatial networks, where we approach football players as an undirected graph. Then, we will look into the developed spatial network metrics that are used in the identification of playstyles. These are summarized into three groups, measuring space control, pressure, and connectivity. With these metrics, we can use clustering techniques to obtain different attacking and defending playstyles. We conclude with the results of the clustering techniques and the possible interpretation and applications of the results.", "recording_license": "", "do_not_record": false, "persons": [{"code": "TVGKR7", "name": "Annemarijn Blom", "avatar": "https://cfp.pydata.org/media/avatars/TVGKR7_dONFGgV.webp", "biography": "With a background in mathematics and econometrics, Annemarijn combines analytical skills with practical solutions. As a data scientist, she is passionate about solving complex puzzles with mathematics and applying her knowledge to real-world problems. At Pipple, she uses her skills to develop practical, effective and creative solutions in various sectors.", "public_name": "Annemarijn Blom", "guid": "0ccd76b4-478b-5b07-a912-d12799457bac", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/TVGKR7/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/38RZBP/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/38RZBP/", "attachments": []}, {"guid": "74395091-486b-5771-95e9-99b34a285a0f", "code": "9KFFYT", "id": 82438, "logo": "https://cfp.pydata.org/media/pydata-eindhoven-2025/submissions/9KFFYT/footballbert_UB71YnD.png", "date": "2025-12-09T12:50:00+00:00", "start": "12:50", "end": "2025-12-09T13:20:00+00:00", "duration": "00:30", "room": "Ernst-Curie", "slug": "pydata-eindhoven-2025-82438-footballbert-encoding-player-identity-in-vectors-with-transformers", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9KFFYT/", "title": "FootballBERT: Encoding player identity in vectors with Transformers.", "subtitle": "", "track": "Sports Analytics hosted PySport", "type": "Talk", "language": "en", "abstract": "FootballBERT introduces a new way of representing football players \u2014 not as static IDs or statistical aggregates that fluctuate wildly over short periods, but as contextual embeddings learned directly from match data.\nBuilt on a Transformer architecture and trained through a Masked Player Prediction (MPP) objective, FootballBERT captures how a player\u2019s identity emerges from teammates, opponents, and coaches tactical demands \u2014 much like BERT learns word meaning from sentences.\nOpenly released on Hugging Face, FootballBERT is a plug-and-play foundation model whose embeddings can be integrated into any downstream system, paving the way for player-aware analytics across performance modeling, recruitment and prediction.", "description": "In football analytics, player identity is still often encoded using one-hot vectors or individual statistics \u2014 highly volatile over short time periods. Such approaches ignore the relational context between teammates and opponents, and coaches tactical demands.\n\nFootballBERT changes that paradigm. Inspired by NLP breakthroughs, it applies the Transformer architecture to lineup data, learning contextual player embeddings through a Masked Player Prediction (MPP) objective. Each embedding encodes a player\u2019s identity through patterns of who they play with, against, and under which tactical setups.\n\nIn this talk, I\u2019ll walk through:\n\n- Why embedding player identity into dense vectors matters for football analytics.\n\n- How FootballBERT is trained end-to-end from raw lineup data and positional features across 170K+ matches.\n\n- Empirical insights on what these embeddings capture \u2014 especially their ability to generalize across leagues.\n\n- How such representations enable downstream applications like BALLER-Transfer Portal, an AI model that predicts how any player would perform in any tactical context with unprecedented granularity.\n\nThis session bridges deep learning, transformers, and sports data science \u2014 showing how domain-specific foundation models like FootballBERT can power the next generation of context-aware analytics.\nWhether you\u2019re an ML researcher, AI engineer, or football data scientist, you\u2019ll leave with a concrete understanding of how to build, train, and apply transformer models beyond text \u2014 starting from real-world, messy data.", "recording_license": "", "do_not_record": false, "persons": [{"code": "HHJTJ9", "name": "Achraff ADJILEYE", "avatar": "https://cfp.pydata.org/media/avatars/HHJTJ9_VxiE8Vf.webp", "biography": "Achraff Adjileye is a research engineer passionate about football analytics and artificial intelligence. He is the founder of the BALLER project, which aims to build a foundational model for football analytics\u2014powering the next generation of context-aware football analysis, much like GPT revolutionized text understanding.\n\nHis vision: Football is the ultimate team sport, yet most analytics treat players as isolated individuals. Players are often represented by radar charts of individual statistics, ignoring the rich collective context that shapes their identity. While this approach transformed data-driven scouting, it is inherently prone to misinterpretation, leading to costly mistakes in transfers and strategic decisions. Achraff works every day to create a football analytics world that respects the collective DNA of the beautiful game.", "public_name": "Achraff ADJILEYE", "guid": "6a7479a1-6e88-52e2-94d8-dfe8fd990e89", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/HHJTJ9/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9KFFYT/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9KFFYT/", "attachments": []}, {"guid": "8deed598-b0f2-5d09-84f5-a16f639df3d3", "code": "AHXR33", "id": 82187, "logo": null, "date": "2025-12-09T13:25:00+00:00", "start": "13:25", "end": "2025-12-09T13:55:00+00:00", "duration": "00:30", "room": "Ernst-Curie", "slug": "pydata-eindhoven-2025-82187-football-is-complex-but-your-code-doesn-t-have-to-be-meet-databallpy-and-a-practical-deep-dive-into-pressing", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/AHXR33/", "title": "Football is complex, but your code doesn\u2019t have to be \u2014 meet DataBallPy and a practical deep dive into pressing", "subtitle": "", "track": "Sports Analytics hosted PySport", "type": "Talk", "language": "en", "abstract": "DataBallPy is an open-source Python package that quickly starts your analysis of a football-related question. In the current talk, we will introduce the core features and functionalities of DataBallPy using code examples with compelling visualisations. The second part of the talk will showcase a practical example of how the Royal Belgian Football Association (RBFA) has used components of DataBallPy to analyse the effectiveness and efficiency of pressuring the opponent in over 200 games. Taken together, this talk will give you a clear starting point of how to start answering your football-related questions.", "description": "Answering complex data-driven questions about football tactics is a lot of fun. Trying to parse football data, synchronize tracking and event data, and determine which individual player actually has ball possession is just tedious work, which limits the time to actually answer the questions you wanted to answer in the first place. DataBallPy is an open-source Python package designed to abstract away these repetitive tasks, allowing analysts and developers to focus on building models and answering their questions.\nIn the first half of this talk, we introduce DataBallPy\u2019s architecture and core functionality (and how it differs from existing packages like Kloppy). Using code examples, we will show how you can perform multiple preprocessing steps and visualise the data, and we will showcase some of the implemented features. Equally important, we value transparency and education. Therefore, we have some elaborate documentation on how all features are implemented. In short, DataBallPy will quickstart your process in answering your questions.\nThe second half of the talk presents a practical case study: how the Royal Belgian Football Federation (RBFA) uses open-source packages, among which DataBallPy, to analyse the efficiency and effectiveness of pressuring the opponent. The RBFA was inspired by the Common Data Format (CDF) (Anzer et al., 2025) to document and store data. Similarly, DataBallPy functionalities were made to work with the CDF as intended. The research builds on Bekkers\u2019 pressing model, which uses the Time To Intercept (TTI) concept to estimate how quickly a defender can reach a specific target location on the pitch, allowing it to represent pressure. During the talk, we will showcase how the RBFA got to answer the following two questions: (1) Does external load differ significantly between successful and unsuccessful pressing actions? and (2) How does external load vary across pressing actions initiated from high-, mid-, and low-block defensive positions?", "recording_license": "", "do_not_record": false, "persons": [{"code": "SVDMQ7", "name": "Alexander Oonk", "avatar": "https://cfp.pydata.org/media/avatars/SVDMQ7_qQHwaod.webp", "biography": "PhD student at the University of Groningen investigating the one-on-one dribbles.", "public_name": "Alexander Oonk", "guid": "6f6860db-97e5-5c07-81bb-c52de2235361", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/SVDMQ7/"}, {"code": "3LKVHH", "name": "Tygo Nikamp", "avatar": "https://cfp.pydata.org/media/avatars/3LKVHH_wKBoulg.webp", "biography": "24 year old Sport Science and Business Administration student at the University of Groningen, currently doing an internship as Data Scientist at the Royal Belgian Football Association (RBFA). Passionate about working with (soccer) data, and driven to excel in the fast-paced, performance-driven world of football.", "public_name": "Tygo Nikamp", "guid": "5a64ce07-f34b-565e-9bd9-dec9e23c45dc", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/3LKVHH/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/AHXR33/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/AHXR33/", "attachments": []}, {"guid": "c82ffd21-514b-54d2-868e-c3bd050c9021", "code": "GGZ9TA", "id": 81070, "logo": null, "date": "2025-12-09T14:10:00+00:00", "start": "14:10", "end": "2025-12-09T14:40:00+00:00", "duration": "00:30", "room": "Ernst-Curie", "slug": "pydata-eindhoven-2025-81070-planning-hockey-careers-with-python", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/GGZ9TA/", "title": "Planning Hockey Careers With Python", "subtitle": "", "track": "Sports Analytics hosted PySport", "type": "Talk", "language": "en", "abstract": "How can data science help young athletes navigate their careers? In this talk, I\u2019ll share my experience building a career path planner for aspiring ice hockey players. The project combines player performance data, career path patterns, and predictive modeling to suggest possible development paths and milestones. Along the way, I\u2019ll discuss the challenges of messy sports data and communicating insights in a way that resonates with non-technical users like coaches, parents, and players.", "description": "## Objective\n\nTo share the process of building a data-driven career path planner for young hockey players, highlighting both the technical challenges and the broader impact of applying data science in real-world decision-making.\n\n## Outline\n\n- Why hockey (and why career planning matters in sports)\n- Data collection & wrangling: turning messy, incomplete performance records into usable data\n- Modeling progression: techniques to identify likely career trajectories and milestones\n- Communicating results: designing tools that resonate with non-technical stakeholders (players, coaches, parents)\n- Lessons learned: challenges, surprises, and what translates to other domains beyond sports\n\n## Central Thesis\n\nData science can provide valuable guidance in high-stakes, personal decision-making \u2014 but building tools for non-technical users requires more than just models. It demands thoughtful data cleaning, careful feature selection, and communication strategies that make insights accessible and actionable.\n\n## Audience\n\nThis talk is for data scientists, analysts, and practitioners who are curious about applying machine learning to sports. It will appeal both to those interested in the technical process (data wrangling, predictive modeling) and those interested in building data products that impact real people.\n\n## Tone / Type of Talk\n\nInformative and light-hearted \u2014 grounded in practical data challenges, with a touch of storytelling from the hockey world. This is not a heavily mathematical talk at all, suitable for broader audience as well.\n\n## Key Takeaways\n\n- How to wrangle and model messy sports data for career planning use cases.\n- How to design predictive models that communicate uncertainty and possible paths.\n- Lessons on building data tools for non-technical stakeholders.\n- Inspiration for applying data science to creative and impactful real-world problems.\n\n## Background Knowledge Expected\n\nNo prior knowledge of hockey is required \u2014 we\u2019ll keep the sports side light and fun.", "recording_license": "", "do_not_record": false, "persons": [{"code": "HJXFN7", "name": "Jaroslav Bezdek", "avatar": "https://cfp.pydata.org/media/avatars/HJXFN7_rYzFmMY.webp", "biography": "I graduated in statistics in 2018 and have been working in data analytics ever since, across a variety of roles. I am currently the Head of Sport Sciences at GRAET, a company building a platform to help aspiring ice hockey players develop their potential. Alongside this, I work as a part-time data analyst for one of the major Czech ice hockey clubs. My work combines a passion for sports with a love of data. Outside of work, I enjoy playing the drums, riding my bike, and debating whether pineapple belongs on pizza \u2014 though as a fan of pizza Hawaii, I\u2019m firmly in the \u201cyes\u201d camp.", "public_name": "Jaroslav Bezdek", "guid": "47e43be9-34c4-513f-a1df-5b7f59d649b4", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/HJXFN7/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/GGZ9TA/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/GGZ9TA/", "attachments": []}, {"guid": "55f9d9b3-0dfb-557a-94a3-328f5b336874", "code": "EGJV7E", "id": 82416, "logo": null, "date": "2025-12-09T14:45:00+00:00", "start": "14:45", "end": "2025-12-09T15:15:00+00:00", "duration": "00:30", "room": "Ernst-Curie", "slug": "pydata-eindhoven-2025-82416-optimizing-fantasy-basketball-decisions-with-python-linear-integer-programming-for-roster-management", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/EGJV7E/", "title": "Optimizing fantasy basketball decisions with Python: linear & integer programming for roster management", "subtitle": "", "track": "Sports Analytics hosted PySport", "type": "Talk", "language": "en", "abstract": "Fantasy basketball involves daily decisions: which players to start, who to pick up from free agency, and how to balance competing objectives across multiple statistical categories. This talk demonstrates how linear programming and integer programming can help solving those problems.\n\nUsing Python library PuLP  we'll explore when to use linear programming versus integer programming, how to formulate constraints for roster decisions, and how to handle different league formats. Through practical examples, we'll build optimizers for start/sit decisions and free agency streaming.", "description": "Core Content:\nThis talk teaches Linear and Integer Programming fundamentals through fantasy basketball examples, demonstrating when to use each approach for optimization problems.\n\nSession Structure:\n\n -Linear programming fundamentals\n- Integer programming fundamentals \n- Demo: start/sit optimization\n- Demo: free agency optimization \n- Real life applications\n\nI will also explain rules of different fantasy basketball formats.\nPrior Knowledge Expected:\nPython (no basketball knowledge needed but not hating on sports will help)", "recording_license": "", "do_not_record": false, "persons": [{"code": "7BJRGM", "name": "Pawel Kapuscinski", "avatar": "https://cfp.pydata.org/media/avatars/7BJRGM_k2gCqC7.webp", "biography": "Sports nerd with over 30 years of experience as a fan, 10 years of experience working in data, and 5 years combining both to support professional teams (across football, baseball and ice hockey) in better understanding their sports.", "public_name": "Pawel Kapuscinski", "guid": "b170c6ec-d1e7-5083-acec-2fddcdea9821", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/7BJRGM/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/EGJV7E/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/EGJV7E/", "attachments": []}, {"guid": "30326fe9-9768-5222-8dc4-eca155ced0af", "code": "K9LDVT", "id": 82550, "logo": null, "date": "2025-12-09T15:20:00+00:00", "start": "15:20", "end": "2025-12-09T15:50:00+00:00", "duration": "00:30", "room": "Ernst-Curie", "slug": "pydata-eindhoven-2025-82550-xreceiver-a-gnn-approach-to-the-evaluation-of-the-decision-making-process-of-passing-options-in-football", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/K9LDVT/", "title": "xReceiver: a GNN approach to the evaluation of the decision-making process of passing options in football", "subtitle": "", "track": "Sports Analytics hosted PySport", "type": "Talk", "language": "en", "abstract": "The process of decision-making in football is characterized by a complex interplay between spatial positioning, opponent pressure, and player intent. In this research, we introduce xReceiver, a real-time Graph Neural Network (GNN) framework designed to predict the optimal passing target by modeling on-field interactions as dynamic graphs. Each player is represented as a node with positional and contextual features, while potential passing lines form weighted edges characterized by distance, angle, and pressure metrics. We have developed a Message-Passing Neural Network (MPNN) that is trained using a combination of tracking data and event data from professional matches. Our model achieves 65.22% accuracy in identifying the actual chosen receiver and 95.65% accuracy within its top three suggestions. xReceiver further offers quantification of each option's likelihood, threat, and creativity, enabling performance analysts to evaluate over 1,000 passes in seconds.", "description": "During my internship at the Royal Belgian FA, my supervisor received a request from the performance analyst: how can we analyse the quality of passes using AI? \nWe generally consider a player to be a good passer if most of their passes are successful, but sometimes passing the ball to that player is not the right option if there is someone else who is less marked and in a better position! Can players really identify the right teammate in the right situation? \n\nOutline of what will be discussed during the talk:\n\n1. Introduction (2\u20133 minutes).\n2. From data to graph through synchronization (5 minutes).\n3. Graph Neural Networks (5 minutes)\n4. The xReceiver model (7-8 minutes)\n5. Applications in Opponent Analysis and Scouting (5 minutes)\n\nQuestions and feedback will be welcomed at the end, since these can really help with the improvements of this project.", "recording_license": "", "do_not_record": false, "persons": [{"code": "TNURBM", "name": "Gabriel Masella", "avatar": "https://cfp.pydata.org/media/avatars/TNURBM_hwg7ta1.webp", "biography": "Fotball Data Scientist/AI Engineer @SportAnalytics\nData Science and AI's Master Student @UNITS\nFormer Data Science Intern @RBFA", "public_name": "Gabriel Masella", "guid": "96b940f8-1960-50e4-be33-742bda3f979e", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/TNURBM/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/K9LDVT/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/K9LDVT/", "attachments": []}], "Planck-Bohr": [{"guid": "3233f876-ee51-5be8-97a3-2e425b3ac73e", "code": "9KLLZY", "id": 81860, "logo": "https://cfp.pydata.org/media/pydata-eindhoven-2025/submissions/9KLLZY/ML_pipeline_t_6Bwqz7S.webp", "date": "2025-12-09T10:00:00+00:00", "start": "10:00", "end": "2025-12-09T10:30:00+00:00", "duration": "00:30", "room": "Planck-Bohr", "slug": "pydata-eindhoven-2025-81860-beyond-one-model-scaling-orchestrating-monitoring", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9KLLZY/", "title": "Beyond One Model: Scaling, Orchestrating & Monitoring", "subtitle": "", "track": "AI/Machine Learning/GenAI", "type": "Talk", "language": "en", "abstract": "Training one model is fun. Running thousands without everything catching fire? That\u2019s the real challenge. In this talk, we\u2019ll show how we \u2014 two data scientists turned accidental ML engineers \u2014 scaled anomaly detection at Vanderlande. Expect a peek into our orchestration setup, a quick code snippet, a look at our monitoring dashboard and how we scale to a thousand models.", "description": "# Description\n\nBuilding a single machine learning model is one challenge. Running thousands of them in production \u2014 reliably, efficiently, and transparently \u2014 is another. As Vanderlande adopted its use of anomaly detection, we faced the reality that success wasn\u2019t about the accuracy of one model, but about the MLOps infrastructure to scale, orchestrate, and monitor many.\n\nWe are two data scientists who gradually found ourselves tackling problems that look a lot more like ML engineering. In this session, we\u2019ll share our journey in a practical and informal way, showing the real steps we took, the hurdles we hit, and the solutions we built.\n\nWe\u2019ll walk through the orchestration layer that coordinates training and deployment for thousands of models, highlight practical solutions for scaling pipelines, and discuss how we keep the system resilient when failures occur. A short code snippet will illustrate how lightweight the deployment process can be, and we\u2019ll close with a look at the monitoring framework that provides visibility and reliability across the entire fleet.\n\nWhether you are just starting to scale your ML initiatives or already struggling with operational complexity, this talk will offer lessons learned and concrete practices for moving beyond one model.\n \n# Outline \n**1. Vanderlande\u2019s use case (5 min)**  \n- Vanderlande specific use case\n- Anomaly detection at scale \n- Why *beyond one model* became a necessity  \n \n**2. Machine learning orchestration & pipeline (13 min)**  \n- Intro: One model ML Architecture\n- Architecture & orchestration design for multiple models\n- Isolated pipelines to avoid cascading failures   \n \n**3. Code snippet & peek into monitoring (7 min)**  \n- Inference orchestrator code example  \n- CI/CD deployment pipeline \n- Monitoring framework: jobs, models, anomalies at scale  \n- Lessons learned & what we\u2019d do differently  \n \n**4. Wrap-up & Q&A (5 min)**  \n- Key takeaways  \n- Quick audience questions", "recording_license": "", "do_not_record": false, "persons": [{"code": "TLGMZN", "name": "Azucena Morales", "avatar": "https://cfp.pydata.org/media/avatars/TLGMZN_RDWpckP.webp", "biography": "TBD", "public_name": "Azucena Morales", "guid": "3a080037-a955-5526-a298-35f1e4d7536a", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/TLGMZN/"}, {"code": "K9VMFS", "name": "Vi Chu", "avatar": "https://cfp.pydata.org/media/avatars/K9VMFS_6Jczaqr.webp", "biography": "**Vi Chu** is a data scientist at **Vanderlande**, where she develops scalable predictive maintenance solutions that help keep complex automated logistics systems running smoothly. She\u2019s motivated by collaborative problem-solving and loves when a solution becomes simple, elegant, and easy for everyone to put into practice.", "public_name": "Vi Chu", "guid": "d3f394c4-b055-5b8f-ac73-caa1beffe40c", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/K9VMFS/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9KLLZY/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/9KLLZY/", "attachments": []}, {"guid": "bd4c27ac-934d-50f7-a29d-eb6c381c7fe7", "code": "EMUNDT", "id": 82545, "logo": null, "date": "2025-12-09T11:20:00+00:00", "start": "11:20", "end": "2025-12-09T11:50:00+00:00", "duration": "00:30", "room": "Planck-Bohr", "slug": "pydata-eindhoven-2025-82545-from-data-lake-entanglement-to-data-mesh-decoupling-scaling-a-self-service-data-platform", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/EMUNDT/", "title": "From Data Lake Entanglement to Data Mesh Decoupling: Scaling a Self-Service Data Platform", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "Our data platform journey started with a classic data lake \u2014 easy to ingest, hard to evolve. As domains scaled, tight coupling across source systems, pipelines, and data products slowed everything down. In this talk, we share how we re-architected toward a domain-oriented data mesh using PySpark, Delta Lake and DQX to achieve true decoupling. Expect practical lessons on designing independent data products, managing lineage and governance, and scaling self-service without chaos.", "description": "1. What exactly is architectural decoupling\n2. Identify how hidden coupling creeps into data lakes and data meshes.\n3. Learn technical design patterns for decoupling ingestion, transformation, and ownership using open tools.\n4. Understand how to scale self-service data mesh principles without breaking governance or lineage integrity.", "recording_license": "", "do_not_record": false, "persons": [{"code": "HHTVQH", "name": "Geert Jongen", "avatar": null, "biography": "Geert Jongen is a System Architect for Data & Analytics at Vanderlande, where he designs and evolves the company\u2019s data platform toward a federated data mesh. Before that, he worked as a data consultant at Pipple for five years. With a background in data engineering, analytics, and data science, he focuses on building scalable data architectures that empower teams through autonomy and governance.", "public_name": "Geert Jongen", "guid": "82aaec45-11a4-52dc-8598-313351fca131", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/HHTVQH/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/EMUNDT/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/EMUNDT/", "attachments": []}, {"guid": "2a2afa02-0c98-59c8-811f-2a53b18db1fc", "code": "J3ECG7", "id": 82421, "logo": null, "date": "2025-12-09T12:50:00+00:00", "start": "12:50", "end": "2025-12-09T13:20:00+00:00", "duration": "00:30", "room": "Planck-Bohr", "slug": "pydata-eindhoven-2025-82421-scaling-python-to-thousands-of-nodes-with-ray", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/J3ECG7/", "title": "Scaling Python to thousands of nodes with Ray", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "Python is the language of choice for anything to do with AI and ML. While that has made it easy to write code for one machine, it's much more difficult to run workloads across clusters of thousands of nodes. Ray allows you to do just that. I'll demonstrate how to implement this open source tool with a few lines of code. As a demo project, I'll show how I built a RAG for the Wheel of Time series.", "description": "DESCRIPTION\n\nI don\u2019t read as much as I used to. As a result, ploughing through the Wheel of Time series by Robert Jordan has become a multi-year project for me. With 2787 named characters across 14 books, I often have to search on wikis for what happened six books ago. And who was that one minor character again\u2026?\n\nWith AI we can do better. What if I fine-tuned an LLM with RAG to help me keep track of the entire saga? I could ask questions and it would give me spoiler-free answers\u2026 \u2728\u2028\n\nWhipping up a few lines of Python to train and tune models is easier than ever. But while executing scripts on a laptop or VM is trivial, it becomes much more difficult when you want to parallellize such a workload across an entire cluster. That\u2019s where Ray comes in. With a few lines of code, we can modify our code to distribute our model training to tens or hundreds of machines. If you ever wondered how you companies like OpenAI train foundational models: this is how.\u2028\n\nIn this talk, I\u2019ll cover Ray\u2019s core concepts and show how to bring it in action. I\u2019ll introduce Ray tasks and actors, and show how to set up large-scale pipelines with Ray Data. As a demo project, I\u2019ll finetune an LLM to help me with my Wheel of Time read-through. And I\u2019ll show how can scale your workloads from a single machine to thousands of nodes.\n\nAGENDA\n\n1. Introduction (2m)\n2. Problem (3m)\n3. Ray explained (10m)\n4. Demo project (10m)\n5. Conclusion and takeaways (5m)\n\nKEY TAKEAWAYS\n\n- For AI workloads we need frameworks that are hardware-agnostic and work on both CPUs and GPUs\n- With Ray tasks and actors, we can achieve distributed computing with just a few decorators\n- Automatic up- and downscaling is a must if you want to keep your cloud bill in check\n- While a literary masterpiece, The Wheel of Time has too many characters and some pacing issues", "recording_license": "", "do_not_record": false, "persons": [{"code": "EG3CNY", "name": "Rob de Wit-Liezenga", "avatar": "https://cfp.pydata.org/media/avatars/EG3CNY_WCS5zUx.webp", "biography": "Rob de Wit-Liezenga is a freelance engineer who has worked in different data-related roles throughout his career. He's been a data analyst, data platform engineer, developer advocate, and customer success engineer. He enjoys work where he can combine hard tech with a people-focused role.\n\nHe has spoken at and co-organized various conferences in the past, including PyData Eindhoven. He enjoys meeting fellow pythonistas, learning from them, and sharing some of his own learnings.\n\nNowadays, Rob works as a CSE for Anyscale, where he helps customers deploy and optimize their infrastructure for large-scale AI workloads. Outside work, he enjoys photography, learning Spanish, hiking, and climbing.", "public_name": "Rob de Wit-Liezenga", "guid": "dd80e247-c699-5947-bbb7-b93f45d4fc22", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/EG3CNY/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/J3ECG7/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/J3ECG7/", "attachments": []}, {"guid": "7704351d-062c-5f81-9264-254b0c1518f9", "code": "E7QMDN", "id": 81604, "logo": null, "date": "2025-12-09T13:25:00+00:00", "start": "13:25", "end": "2025-12-09T13:55:00+00:00", "duration": "00:30", "room": "Planck-Bohr", "slug": "pydata-eindhoven-2025-81604-extending-sql-databases-with-python", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/E7QMDN/", "title": "Extending SQL Databases with Python", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "What if your database could run Python code inside SQL? In this talk, we\u2019ll explore how to extend popular databases using Python, without needing to write a line of C.\n\nWe\u2019ll cover three systems\u2014SQLite, DuckDB, and PostgreSQL\u2014and show how Python can be used in each to build custom SQL functions, accelerate data workflows, and prototype analytical logic. Each database offers a unique integration path:\n- SQLite and DuckDB allow you to register Python functions directly into SQL via sqlite3.create_function, making it easy to inject business logic or custom transformations.\n- PostgreSQL offers PL/Python, a full-featured procedural language for writing SQL functions in Python. We\u2019ll also touch on advanced use cases, including embedding the Python interpreter directly into a PostgreSQL extension for deeper integration.\n\nBy the end of this talk, you\u2019ll understand the capabilities, limitations, and gotchas of Python-powered extensions in each system\u2014and how to choose the right tool depending on your use case, whether you\u2019re analyzing data, building pipelines, or hacking on your own database.", "description": "1. Introduction (3 min)\n\t\u2022\tWho this talk is for: devs, data engineers, extension hackers\n\t\u2022\tMotivation: why embed Python in databases?\n\t\u2022\tOverview of the 3 systems (SQLite, DuckDB, PostgreSQL)\n\n2. SQLite & DuckDB: Python Functions via sqlite3.create_function (7 min)\n\t\u2022\tHow sqlite3.create_function() works in SQLite\n\t\u2022\tExample: creating a simple text-processing SQL function in Python\n\t\u2022\tUse cases: rapid prototyping, lightweight data pipelines\n\n3. PostgreSQL with PL/Python (6 min)\n\t\u2022\tEnabling and using the PL/Python extension\n\t\u2022\tWriting SQL functions in Python\n\t\u2022\tPros and limitations (e.g., sandboxing, permissions, virtualenvs)\n\n4. Advanced: Embedding Python into PostgreSQL Extensions (7 min)\n\t\u2022\tWriting PostgreSQL extensions in C that embed Python\n\t\u2022\tUse cases: integrating ML models, custom procedural logic\n\t\u2022\tShort demo or diagram: C + Python working inside Postgres\n\n5. Trade-offs and Comparison (3 min)\n\t\u2022\tPerformance, deployment, complexity, ecosystem\n\t\u2022\tWhen to use which approach\n\n6. Q&A (4 min)\n\t\u2022\tInvite questions and deeper discussion from the audience", "recording_license": "", "do_not_record": false, "persons": [{"code": "XDWEBT", "name": "Florents Tselai", "avatar": "https://cfp.pydata.org/media/avatars/XDWEBT_6CA6Hm6.webp", "biography": "Florents (Flo) Tselai is a data generalist, open-source developer, and PostgreSQL contributor. He has authored multiple database extensions, primarily for PostgreSQL, and also for SQLite and DuckDB.\n\nHis work sits at the intersection of database engineering, AI, web crawling, and data journalism - blending research, engineering, and leadership along the way.\n\nHe is also the creator of diofanti.org, a civic-tech platform that promotes government transparency and citizen empowerment, enabling people to track public spending and decisions in Greece. https://github.com/Florents-Tselai", "public_name": "Florents Tselai", "guid": "b752bed8-4944-5758-9592-178e303d5b23", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/XDWEBT/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/E7QMDN/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/E7QMDN/", "attachments": []}, {"guid": "a2fd8090-6ee8-58f8-9431-52f7bc7ef0dd", "code": "GJNEKL", "id": 85298, "logo": null, "date": "2025-12-09T14:10:00+00:00", "start": "14:10", "end": "2025-12-09T14:40:00+00:00", "duration": "00:30", "room": "Planck-Bohr", "slug": "pydata-eindhoven-2025-85298-compactifai-quantum-inspired-ai-model-compression", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/GJNEKL/", "title": "CompactifAI: Quantum-Inspired AI Model Compression", "subtitle": "", "track": "AI/Machine Learning/GenAI", "type": "Talk", "language": "en", "abstract": "Large AI models have become powerful but increasingly impractical; with escalating training costs, bloated memory requirements, and latency bottlenecks that limit real-world deployments. This talk introduces CompactifAI: a quantum-inspired compression framework that uses tensor networks to surgically shrink large models while preserving their accuracy and capabilities.", "description": "We will begin with the story of how Multiverse came to be in 2019 with the mission to solve today\u2019s problems through quantum technologies. Along this path, we discovered during a project for Bosch that quantum-inspired algorithms running entirely on classical hardware could ultra-compress AI models. In 2024, we realized that these same techniques could be applied to Large Language Models. This insight gave birth to CompactifAI. From there, we\u2019ll walk through CompactifAI and its compression pipeline, highlighting how it outperforms naive pruning or quantization approaches in both precision and control leveraging Tensor Networks.\n\nAttendees will see how this enables new deployment scenarios: running powerful LLMs on edge devices, routing queries between local and cloud models, and even removing or restoring specific behaviors (e.g. safety filters or domain knowledge).\n\nAdditionally, we\u2019ll show how to integrate our compactifAI compressed models via API with minimal code changes and provide relevant developer resources for those interested in benefitting from faster, cheaper, more efficient models.\nThis talk is aimed at ML engineers, researchers, and technical leads working with LLMs, vision models, or constrained deployment targets who are ready to think beyond just \u201cbigger is better.\u201d", "recording_license": "", "do_not_record": false, "persons": [{"code": "FSMMEV", "name": "Jon Lei\u00f1ena Otamendi", "avatar": "https://cfp.pydata.org/media/avatars/FSMMEV_GkVeWKJ.webp", "biography": "Solution Architect @ Multiverse Computing", "public_name": "Jon Lei\u00f1ena Otamendi", "guid": "59af755b-06e0-5f4a-ab36-ae123559a5a6", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/FSMMEV/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/GJNEKL/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/GJNEKL/", "attachments": []}, {"guid": "9ec5f13b-9574-5400-9344-c40ff15de7ab", "code": "FVGMBL", "id": 85702, "logo": null, "date": "2025-12-09T14:45:00+00:00", "start": "14:45", "end": "2025-12-09T15:15:00+00:00", "duration": "00:30", "room": "Planck-Bohr", "slug": "pydata-eindhoven-2025-85702-from-experiment-to-enterprise-architecting-ai-for-stability-and-scale", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/FVGMBL/", "title": "From Experiment to Enterprise: Architecting AI for Stability and Scale", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "AI teams iterate at the speed of innovation, while organizations require platforms that are reliable, governed, and cost\u2011efficient. This session presents pragmatic patterns and reference architectures that align rapid development with production requirements\u2014so data scientists and developers can move fast without breaking stability.", "description": "Using the Dell AI Factory as an example, we\u2019ll show how an open ecosystem, validated designs, and expert services help teams build, deploy, and scale AI securely and repeatably.\n\nWe\u2019ll illustrate how silicon, model, and software diversity matters and can be included in a production environment.", "recording_license": "", "do_not_record": false, "persons": [{"code": "ZLC78N", "name": "Roland Kunz", "avatar": null, "biography": null, "public_name": "Roland Kunz", "guid": "ab5a3295-d473-5f34-b97d-fbf056817291", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/ZLC78N/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/FVGMBL/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/FVGMBL/", "attachments": []}, {"guid": "a66b3d8e-60fe-54a4-bc79-19f401efaec6", "code": "7FTJYA", "id": 86074, "logo": null, "date": "2025-12-09T15:20:00+00:00", "start": "15:20", "end": "2025-12-09T15:50:00+00:00", "duration": "00:30", "room": "Planck-Bohr", "slug": "pydata-eindhoven-2025-86074-ingestify-rethinking-ingestion-for-complex-data", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/7FTJYA/", "title": "Ingestify: Rethinking Ingestion for Complex Data", "subtitle": "", "track": "Data Engineering", "type": "Talk", "language": "en", "abstract": "Traditional data pipelines often tie ingestion and transformation together, forcing data into rows and columns early in the process. But modern workloads - from large XML documents to high-resolution video files - have transformations that are far from trivial and require very different compute resources than ingestion. Separating these concerns becomes essential. When a transform fails, you shouldn\u2019t have to re-download data or hit the source system again.", "description": "Ingestify is a Metadata-First Ingestion Layer that stores raw data as-is, enriched with structured metadata, and ingests only when the source data has actually changed. This makes the ingest step fast, durable and independent of downstream processing. Transformations can evolve, fail or be retried freely while the original data remains intact.\n\nIn this talk, I\u2019ll show how metadata-first ingestion forms a clean and reliable foundation for working with large, complex or non-tabular data - and why it\u2019s a good fit for modern\u00a0data\u00a0workflows.", "recording_license": "", "do_not_record": false, "persons": [{"code": "M87AD9", "name": "Koen Vossen", "avatar": null, "biography": null, "public_name": "Koen Vossen", "guid": "a464eed5-5d55-5d9d-90d4-408e73f52213", "url": "https://cfp.pydata.org/pydata-eindhoven-2025/speaker/M87AD9/"}], "links": [], "feedback_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/7FTJYA/feedback/", "origin_url": "https://cfp.pydata.org/pydata-eindhoven-2025/talk/7FTJYA/", "attachments": []}]}}]}}}