1058 lines
88 KiB
HTML
1058 lines
88 KiB
HTML
<!doctype html>
|
||
<html lang="en">
|
||
<head>
|
||
<meta charset="utf-8">
|
||
<title>AritmoLab — Role of AI in the Analysis of Unstructured Clinical Databases</title>
|
||
<link rel="stylesheet" href="vendor/reveal/dist/reveal.css">
|
||
<link rel="stylesheet" href="fonts.css?v=8">
|
||
<link rel="stylesheet" href="deck.css?v=41">
|
||
<link rel="stylesheet" href="slide-lens.css?v=7">
|
||
<link rel="stylesheet" href="ai-callouts.css?v=3">
|
||
<link rel="stylesheet" href="screenshot-tour.css?v=1">
|
||
</head>
|
||
<body>
|
||
<div class="reveal"><div class="slides">
|
||
|
||
|
||
<section class="arit title-slide">
|
||
<header class="head">
|
||
<img src="logo.png" alt="Policlinico San Donato">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">AritmoLab · Clinical Data Platform</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<h1>Role of AI in the Analysis of Unstructured Clinical Databases</h1>
|
||
<div class="hgap"></div>
|
||
<p class="lead">How we turned a mix of structured and free-text clinical records into a research data platform — the AritmoLab platform experience.</p>
|
||
<div class="title-meta">
|
||
<span class="pill">AritmoLab</span>
|
||
<span class="what">A clinical data platform hosting cardiology data and analysis tools</span>
|
||
</div>
|
||
<div class="speakers">
|
||
<span>Dr. Marco Pancotti - MultiPhysixLab</span>
|
||
<span>Dr. Sara Paratico - I.R.C.C.S. Policlinico San Donato</span>
|
||
</div>
|
||
<p class="event">“Multidimensional Characterization of Cardiac Arrhythmias: Role of Electrocardiology in the Artificial Intelligence Era”</p>
|
||
<p class="event-where">San Donato Milanese, Milan, Italy · 2–3 October 2026</p>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">01 / 13</span></footer>
|
||
<aside class="notes">
|
||
Good morning. My name is Marco Pancotti, and I led the part of the PAMP-FA project dedicated to building the AritmoLab portal at Policlinico San Donato, with support from Sara Paratico, who will co-present with me today.
|
||
The scope of the project included a data warehouse and tools for machine learning and predictive statistics to support research by the Arrhythmology Unit, directed by Professor Pappone and Professor Locati.
|
||
Today we'll focus on the use of AI to extract structured information from clinical text, followed by a brief tour of the portal and ThothII, a tool for building datamarts from natural-language requests.
|
||
</aside>
|
||
</section>
|
||
|
||
|
||
<section class="arit">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">Where we started</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<h2>Where we started - Four islands and four missing pieces</h2>
|
||
<div class="scatter">
|
||
<div class="todo-stack" style="left:30%; top:22%; width:40%">
|
||
<div class="todo" data-missing="ci" style="margin-left:0"><span class="q">?</span><span class="tx"><b>Clinical Intelligence</b><span>dashboards on the clinical history of the Unit and its patients</span></span></div>
|
||
<div class="todo" data-missing="dwh" style="margin-left:9%"><span class="q">?</span><span class="tx"><b>Datawarehouse</b><span>the single source of truth</span></span></div>
|
||
<div class="todo" data-missing="ml" style="margin-left:4%"><span class="q">?</span><span class="tx"><b>ML-ready data</b><span>cohort tables, ready for model training</span></span></div>
|
||
<div class="todo" data-missing="portal" style="margin-left:12%"><span class="q">?</span><span class="tx"><b>Professional Portal</b><span>AritmoLab — the grown-up Omics Portal</span></span></div>
|
||
</div>
|
||
<button class="isle" data-start="cardioref" style="--rot:-3deg; --size:114px; left:calc(1% + 39px); top:4%; width:230px">
|
||
<svg class="ico" viewBox="0 0 48 48" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><path d="M24 33 c-6-3.5-9-7-9-10.5 0-3 4-5.5 7-3 1 .8 1.6 1.8 2 3 .4-1.2 1-2.2 2-3 3-2.5 7 0 7 3 0 3.5-3 7-9 10.5z" fill="currentColor" stroke="none"/><path d="M22.5 21 h3 v3 h3 v3 h-3 v3 h-3 v-3 h-3 v-3 h3 z" fill="#fff" stroke="none"/></svg>
|
||
<span class="nm">Cardioref</span><span class="sc" style="width:300px">visits · operations · reports — twenty years of management on structured forms</span></button>
|
||
<button class="isle" data-start="genetic" style="--rot:3deg; --size:114px; left:calc(73% + 39px); top:4%; width:230px">
|
||
<svg class="ico" viewBox="0 0 48 48" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><path d="M18 8 C28 14 28 20 18 26 C12 30 12 35 18 40"/><path d="M30 8 C20 14 20 20 30 26 C36 30 36 35 30 40"/><path d="M17 12 H31"/><path d="M15.5 18 H32.5"/><path d="M17 24 H31"/><path d="M14.5 31 H22"/><rect x="25" y="29" width="12" height="12" rx="1.5" stroke-width="2"/><path d="M25 34 H37"/><path d="M30.5 29 V41"/></svg>
|
||
<span class="nm">Genetic data</span><span class="sc">DNA instruments — results typed into Excel by hand</span></button>
|
||
<button class="isle" data-start="omics" style="--rot:-2deg; --size:114px; left:calc(1% + 39px); top:55%; width:230px">
|
||
<svg class="ico" viewBox="0 0 48 48" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><circle cx="24" cy="24" r="20" stroke-width="2.6"/><circle cx="24" cy="24" r="4" fill="currentColor" stroke="none"/><circle cx="24" cy="10" r="3.4"/><circle cx="11.5" cy="31" r="3.4"/><circle cx="36.5" cy="31" r="3.4"/><path d="M24 20.5 V13.5"/><path d="M21 26.5 L14 29.5"/><path d="M27 26.5 L34 29.5"/></svg>
|
||
<span class="nm">Aritmolab Portal</span><span class="sc">meant to host Cardioref + genetics — still immature</span></button>
|
||
<button class="isle" data-start="ecg" style="--rot:2deg; --size:114px; left:calc(73% + 39px); top:55%; width:230px">
|
||
<svg class="ico" viewBox="0 0 48 48" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><circle cx="24" cy="24" r="20" stroke-width="2.6"/><path d="M9 24 h6 l3-9 5 16 3-7 h14" stroke-width="2.4"/><circle cx="24" cy="24" r="1.8" fill="currentColor" stroke="none"/></svg>
|
||
<span class="nm">ECG</span><span class="sc">paper strip + a CSV on request</span></button>
|
||
<div class="missing-title" style="left:calc(30% + 14px); top:calc(22% - 46px); width:40%">Missing Pieces</div>
|
||
<p class="hint" style="position:absolute; left:26%; width:48%; bottom:2px; margin:0; text-align:center">▸ click a system: how it was used — and what held it back · click a missing piece: what it would have given</p>
|
||
</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">02 / 13</span></footer>
|
||
<aside class="notes">
|
||
The starting point was four disconnected systems:
|
||
1 - <strong>Cardioref</strong> held twenty years of electrophysiology records and supported clinical operations, but much of the information needed for research was in free text.
|
||
2 - <strong>Genetic data</strong> were manually entered into Excel, without integration with Cardioref.
|
||
3 - The <strong>Aritmolab Portal</strong> was an early prototype, not yet integrated with the source systems.
|
||
4 - <strong>ECG</strong> data were available as paper records or CSV exports on request.
|
||
|
||
Whe lacked:
|
||
5 - some longitudinal <strong>clinical analytics</strong>
|
||
6 - a shared <strong>data warehouse</strong>,
|
||
7 - a set of structured datasets for model training,
|
||
8 - and a unified <strong>clinical portal</strong>.
|
||
|
||
</aside>
|
||
</section>
|
||
|
||
|
||
<section class="arit flow-slide">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">The project</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">The plan</div>
|
||
<h2>What we wanted to build</h2>
|
||
<p class="lead">One platform, two destinations: the 360° patient portal and a research-ready warehouse. Built on open-source pillars, with AI coding agents wherever they helped. Next, we’ll look at the hardest part: unstructured data.</p>
|
||
<div class="flow">
|
||
<button type="button" class="zone existed lens-target" id="plan-existed" data-lens="plan-existed" data-lens-title="What existed" data-lens-number="1" aria-label="Explore What existed" aria-expanded="false" aria-controls="slide-lens">
|
||
<span class="zlabel">What existed</span>
|
||
<span class="fcol">
|
||
<span class="fbox src"><b>Cardioref</b><span>cardiology records · procedures · letters</span></span>
|
||
<span class="fbox src"><b>Genetic data</b><span>labs · variants · nomenclature</span></span>
|
||
<span class="fbox src"><b>ECG</b><span>signals · device follow-up</span></span>
|
||
<span class="fbox src future"><b><span class="plus">+</span>Future sources</b><span>new subsystems can be connected as sources</span></span>
|
||
</span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">The starting material offered complementary views of the patient.</p>
|
||
<p><b>Clinical course · Cardioref</b>Visits, procedures and reports describe the course of care. Dates and narrative details give each event its clinical context.</p>
|
||
<p><b>Genetic findings</b>Laboratory results and variant descriptions add the genetic perspective, recorded separately from the clinical history.</p>
|
||
<p><b>Electrical activity · ECG</b>Tracings and exported signals document cardiac electrical activity, complementing the written account of the patient’s condition.</p>
|
||
<p><b>Different forms of evidence</b>Structured fields, free text and signals must retain their clinical meaning when connected. Future sources would extend this initial set.</p>
|
||
</template>
|
||
</button>
|
||
<div class="fcol ai-hit">
|
||
<button class="brain brain-btn" data-ai="ingestion" aria-label="AI contribution: the mappings" aria-pressed="false"><span class="ai-label" aria-hidden="true">AI</span><img class="artificial-brain" src="artificial-brain.svg" alt=""><span class="ai-number">6</span></button>
|
||
<div class="farrow">→</div>
|
||
</div>
|
||
<div class="zone built">
|
||
<span class="zlabel">What we built</span>
|
||
<div style="display:flex; align-items:center; gap:10px; height:100%;">
|
||
<div class="fcol" style="flex:0.8; justify-content:center"><button type="button" class="fbox lens-target" id="plan-staging" data-lens="plan-staging" data-lens-number="2" aria-label="Explore Staging" aria-expanded="false" aria-controls="slide-lens"><b>Staging</b><span>raw replica of the sources</span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">The landing area: a faithful working copy of the data brought in from each source system.</p>
|
||
<p><b>What it contains</b>Original tables, identifiers, dates and clinical text, still in the source’s format.</p>
|
||
<p><b>Why it matters</b>Subsequent processing works on this copy. The original clinical system continues its daily work, and transformations can be checked against the imported data.</p>
|
||
<p class="detail-example"><b>Clinical example</b>A Cardioref letter arrives with its original wording. “No syncope” is still text; this layer does not yet turn it into a clinical variable.</p>
|
||
</template>
|
||
</button></div>
|
||
<div class="farrow">→</div>
|
||
<div class="fcol ai" style="flex:1.25">
|
||
<button class="brain brain-btn" data-ai="integration" aria-label="AI contribution: clinical text reading" aria-pressed="false"><span class="ai-label" aria-hidden="true">AI</span><img class="artificial-brain" src="artificial-brain.svg" alt=""><span class="ai-number">7</span></button>
|
||
<button type="button" class="fbox lens-target" id="plan-integration" data-lens="plan-integration" data-lens-number="3" aria-label="Explore Integration" aria-expanded="false" aria-controls="slide-lens"><b>Integration</b><span>cleaned, normalized, deduplicated</span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">The reconciliation layer: source records become consistent, connected clinical information.</p>
|
||
<p><b>What happens here</b>Formats and terminology are standardized, duplicates reconciled, and records linked through patient and event identifiers.</p>
|
||
<p><b>Where AI contributes</b>It extracts conditions, procedures and drug-challenge outcomes from clinical text. Negation and context matter; extraction quality needs validation.</p>
|
||
<p class="detail-example"><b>Clinical example</b>“No syncope” is not a positive finding. A family history of Brugada must remain distinct from the patient’s own diagnosis.</p>
|
||
</template>
|
||
</button>
|
||
<div class="fcap">AI reads the clinical text: pathologies, procedures, drug-challenge outcomes</div>
|
||
</div>
|
||
<div class="farrow">→</div>
|
||
<div class="fcol" style="flex:1.05; justify-content:center"><button type="button" class="fbox lens-target" id="plan-star-schema" data-lens="plan-star-schema" data-lens-number="4" aria-label="Explore Data warehouse" aria-expanded="false" aria-controls="slide-lens"><b>Data warehouse</b><span>star schema · organized for analysis</span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">The shared analytical database: integrated clinical data reorganized for research across patients and over time.</p>
|
||
<p><b>How a star schema works</b>“Facts” represent events or measurements, such as a procedure or test. “Dimensions” describe their context, such as the patient, date and procedure type.</p>
|
||
<p><b>Why it matters</b>Researchers can filter, group and compare records using common definitions, without reconstructing every relationship from the original hospital tables.</p>
|
||
<p class="detail-example"><b>Clinical question it can support</b>How many patients underwent a given procedure each year, and how does that distribution vary by age group?</p>
|
||
</template>
|
||
</button></div>
|
||
<div class="farrow">→</div>
|
||
<div class="fcol ai" style="flex:1.25">
|
||
<button class="brain brain-btn" data-ai="datamarts" aria-label="AI contribution: datamart generation" aria-pressed="false"><span class="ai-label" aria-hidden="true">AI</span><img class="artificial-brain" src="artificial-brain.svg" alt=""><span class="ai-number">8</span></button>
|
||
<button type="button" class="fbox lens-target" id="plan-datamarts" data-lens="plan-datamarts" data-lens-number="5" aria-label="Explore Datamarts" aria-expanded="false" aria-controls="slide-lens"><b>Datamarts</b><span>research-ready marts</span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">Focused datasets derived from the warehouse for a specific research question, study or dashboard.</p>
|
||
<p><b>What they define</b>The cohort, time window, variables and level of detail: for example, one row per patient or one row per procedure.</p>
|
||
<p><b>How ThothII helps</b>A plain-English question becomes a proposed SQL query through a guided workflow with human review. The resulting dataset supports analysis and portal dashboards.</p>
|
||
<p class="detail-example"><b>Illustrative study dataset</b>Patients who underwent a drug-challenge test, with test date, result and selected clinical characteristics. Its cohort and variable definitions must be agreed before interpreting results.</p>
|
||
</template>
|
||
</button>
|
||
<div class="fcap">AI builds them on demand from plain-English questions (ThothII)</div>
|
||
</div>
|
||
<div class="farrow">→</div>
|
||
<div class="fcol" style="flex:1">
|
||
<div class="fbox end"><b>AritmoLab Data Warehouse</b><span>queried for research</span></div>
|
||
<div class="fbox end"><b>AritmoLab Portal</b><span>management & exploration</span></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<div class="pillars">
|
||
<span class="agents">AI coding agents — Anthropic · OpenAI — used wherever they helped, 360° in the code</span>
|
||
<span class="pillar"><b>Airflow</b> · orchestration</span>
|
||
<span class="pillar"><b>Superset</b> · dashboards</span>
|
||
<span class="pillar"><b>Django</b> · portal scaffolding</span>
|
||
<span class="pillar"><b>Authentik</b> · auth — GSD LDAP</span>
|
||
</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">03 / 13</span></footer>
|
||
<aside class="notes">
|
||
Here is the whole project on one slide.
|
||
|
||
On the left, what already existed, three hospital data sources:
|
||
(1) <strong>What existed</strong>: <strong>Cardioref</strong>, our electrophysiology records dataset, the <strong>genetic data</strong> coming from the labs, the <strong>ECG signals</strong>, and, in the future, other sources.
|
||
|
||
On the right, what we built in AritmoLab:
|
||
(2) a staging copy of all relevant Cardioref data, (3) an integration layer where the data is cleaned and normalized, (4) a star-schema warehouse, where the data are reorganized as facts and dimensions, and (5) research datamarts, generated from the data warehouse and presented through dashboards embedded in the portal.
|
||
|
||
Everywhere you see the brain symbol, that's where AI works for us.
|
||
6 - The <strong>mappings</strong> that bring the sources in were themselves drafted by AI, then revised by humans
|
||
7 - At <strong>integration</strong>, AI reads the clinical text.
|
||
8 - At the end, AI builds the <strong>datamarts</strong> on demand from plain-English questions.
|
||
That was the plan. And to build it, we used AI everywhere it helped — drafting configs, reading clinical text, and at the very end, producing the datamarts themselves.
|
||
|
||
Next, we’ll look at the hardest part: unstructured data. My colleague, Sara Paratico,do will show you exactly how the text reading works. But first, the first step of the climb: the mappings.
|
||
</aside>
|
||
</section>
|
||
|
||
|
||
<section class="arit center-v problem">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">The problem</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">From free text to a research platform</div>
|
||
<h2>The clinical truth lives in unstructured columns</h2>
|
||
<div class="quote">
|
||
<div class="tr-grid">
|
||
<p class="it">“Il paziente riferisce sincope ricorrente; ECG basale con sopraslivellamento ST in V1–V3; test provocativo con flecainide positivo per pattern Brugada.”</p>
|
||
<p class="en"><span class="en-tag">EN</span>“The patient reports recurrent syncope; baseline ECG with ST-segment elevation in V1–V3; flecainide provocation test positive for a Brugada pattern.”</p>
|
||
</div>
|
||
</div>
|
||
<div class="pipeline" id="pipeline">
|
||
<div class="stage"><b>Sources</b><span>Cardioref · Genetic data<br>Omics Portal · ECG</span><span class="brain-ai"><svg viewBox="0 0 32 32" fill="none" stroke="currentColor" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round"><path d="M16 4 C11 2 5.5 4 6 8.5 C2.5 10 2 14.5 4.5 17 C2.5 20 4 24.5 8 25 C9 28 14 29.5 16 27"/><path d="M16 4 C21 2 26.5 4 26 8.5 C29.5 10 30 14.5 27.5 17 C29.5 20 28 24.5 24 25 C23 28 18 29.5 16 27"/><path d="M16 4.5 V26.5"/><path d="M16 10 h4.5"/><circle cx="23" cy="10" r="1.7" fill="currentColor" stroke="none"/><path d="M16 15.5 h-4.5"/><circle cx="8.5" cy="15.5" r="1.7" fill="currentColor" stroke="none"/><path d="M16 21 h4.5"/><circle cx="23" cy="21" r="1.7" fill="currentColor" stroke="none"/></svg></span></div>
|
||
<div class="arrow">→</div>
|
||
<div class="stage"><b>Staging</b><span>raw replica</span><span class="brain-ai"><svg viewBox="0 0 32 32" fill="none" stroke="currentColor" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round"><path d="M16 4 C11 2 5.5 4 6 8.5 C2.5 10 2 14.5 4.5 17 C2.5 20 4 24.5 8 25 C9 28 14 29.5 16 27"/><path d="M16 4 C21 2 26.5 4 26 8.5 C29.5 10 30 14.5 27.5 17 C29.5 20 28 24.5 24 25 C23 28 18 29.5 16 27"/><path d="M16 4.5 V26.5"/><path d="M16 10 h4.5"/><circle cx="23" cy="10" r="1.7" fill="currentColor" stroke="none"/><path d="M16 15.5 h-4.5"/><circle cx="8.5" cy="15.5" r="1.7" fill="currentColor" stroke="none"/><path d="M16 21 h4.5"/><circle cx="23" cy="21" r="1.7" fill="currentColor" stroke="none"/></svg></span></div>
|
||
<div class="arrow">→</div>
|
||
<div class="stage ai"><b>Integration</b><span>3NF · text mining</span><span class="brain-ai"><svg viewBox="0 0 32 32" fill="none" stroke="currentColor" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round"><path d="M16 4 C11 2 5.5 4 6 8.5 C2.5 10 2 14.5 4.5 17 C2.5 20 4 24.5 8 25 C9 28 14 29.5 16 27"/><path d="M16 4 C21 2 26.5 4 26 8.5 C29.5 10 30 14.5 27.5 17 C29.5 20 28 24.5 24 25 C23 28 18 29.5 16 27"/><path d="M16 4.5 V26.5"/><path d="M16 10 h4.5"/><circle cx="23" cy="10" r="1.7" fill="currentColor" stroke="none"/><path d="M16 15.5 h-4.5"/><circle cx="8.5" cy="15.5" r="1.7" fill="currentColor" stroke="none"/><path d="M16 21 h4.5"/><circle cx="23" cy="21" r="1.7" fill="currentColor" stroke="none"/></svg></span></div>
|
||
<div class="arrow">→</div>
|
||
<div class="stage"><b>Data Warehouse</b><span>star schema</span></div>
|
||
<div class="arrow">→</div>
|
||
<div class="stage hot" id="marts"><b>Marts</b><span>dbt · Superset</span></div>
|
||
</div>
|
||
<div class="midrow">
|
||
<div class="concepts">
|
||
<div class="lbl">four challenges to face · the answers come next</div>
|
||
<div class="citem"><span class="n">01</span><div><b>Volume without structure</b><span class="d">twenty years of records locked in free-text columns</span></div></div>
|
||
<div class="citem"><span class="n">02</span><div><b>Ambiguous language</b><span class="d">negation · family history · bilingual shorthand</span></div></div>
|
||
<div class="citem"><span class="n">03</span><div><b>No shared ontology</b><span class="d">every system names conditions its own way</span></div></div>
|
||
<div class="citem"><span class="n">04</span><div><b>Proving reliability</b><span class="d">extraction must be demonstrably correct</span></div></div>
|
||
</div>
|
||
<div class="railwrap" id="railwrap">
|
||
<div class="conn-drop" id="conn-drop"></div>
|
||
<div class="conn-elbow" id="conn-elbow"></div>
|
||
<div class="rail" id="rail">
|
||
<div class="lbl">outputs of the datamarts</div>
|
||
<div class="row"><svg class="oicon" viewBox="0 0 34 34" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round"><path d="M4 29 h26"/><rect x="7" y="17" width="5" height="12" rx="1"/><rect x="15" y="9" width="5" height="20" rx="1"/><rect x="23" y="13" width="5" height="16" rx="1"/></svg><b>Dashboards</b><span class="d">clinical review on Superset</span></div>
|
||
<div class="row"><svg class="oicon" viewBox="0 0 34 34" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round"><circle cx="7" cy="7" r="2.6"/><circle cx="7" cy="27" r="2.6"/><circle cx="17" cy="17" r="3.4"/><circle cx="27" cy="7" r="2.6"/><circle cx="27" cy="27" r="2.6"/><path d="M9.3 8.8 L14.6 14.6"/><path d="M9.3 25.2 L14.6 19.4"/><path d="M19.4 14.6 L24.7 8.8"/><path d="M19.4 19.4 L24.7 25.2"/></svg><b>Machine Learning</b><span class="d">cohort tables for model training</span></div>
|
||
<div class="row"><svg class="oicon" viewBox="0 0 34 34" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round"><path d="M3 27 L11 19 L17 22 L23 13"/><path d="M23 13 L30 6" stroke-dasharray="3 3.4"/><circle cx="30" cy="6" r="2.2" fill="currentColor" stroke="none"/><path d="M3 31 h28" opacity=".35"/></svg><b>Predictive Statistics</b><span class="d">outcomes & risk analysis</span></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<div class="thoth-wrap" id="thothwrap">
|
||
<div class="t-conn" id="t-conn"></div>
|
||
<div class="t-head" id="t-head"></div>
|
||
<div class="thoth-col">
|
||
<div class="thoth-box"><span class="brain-ai"><svg viewBox="0 0 32 32" fill="none" stroke="currentColor" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round"><path d="M16 4 C11 2 5.5 4 6 8.5 C2.5 10 2 14.5 4.5 17 C2.5 20 4 24.5 8 25 C9 28 14 29.5 16 27"/><path d="M16 4 C21 2 26.5 4 26 8.5 C29.5 10 30 14.5 27.5 17 C29.5 20 28 24.5 24 25 C23 28 18 29.5 16 27"/><path d="M16 4.5 V26.5"/><path d="M16 10 h4.5"/><circle cx="23" cy="10" r="1.7" fill="currentColor" stroke="none"/><path d="M16 15.5 h-4.5"/><circle cx="8.5" cy="15.5" r="1.7" fill="currentColor" stroke="none"/><path d="M16 21 h4.5"/><circle cx="23" cy="21" r="1.7" fill="currentColor" stroke="none"/></svg></span>ThothII — natural-language → SQL over the DWH, builds the final datamarts</div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">04 / 13</span></footer>
|
||
<aside class="notes">
|
||
Here is one real sentence from a discharge letter — in Italian, as the clinicians wrote it. In this single sentence there is a diagnosis (syncope), an ECG finding (ST elevation in V1–V3), and a drug-challenge result (flecainide positive for Brugada pattern). For a human cardiologist this is readable in two seconds. For a database, this is just a blob of text in a column.
|
||
The structured tables — demographics, procedures, dates — only tell half the story. The rest is locked inside these free-text fields, in every hospital system we have.
|
||
</aside>
|
||
</section>
|
||
|
||
|
||
<section class="arit center-v stats">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">Text analysis</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">What the text mining produced</div>
|
||
<h2>Deterministic AI, audited numbers</h2>
|
||
<div class="strip">
|
||
<div class="sstat"><span class="n">58,438</span><span class="l">letters read</span></div>
|
||
<div class="sstat"><span class="n">73,389</span><span class="l">pathologies extracted</span></div>
|
||
<div class="sstat"><span class="n">10,908</span><span class="l">tests parsed</span></div>
|
||
<div class="sstat"><span class="n">2,307</span><span class="l">Brugada patients</span></div>
|
||
<div class="sstat"><span class="n">≈ 240</span><span class="l">letters / month</span></div>
|
||
</div>
|
||
<div class="cycle" id="cycle">
|
||
<svg class="cycle-svg" viewBox="0 0 1184 270">
|
||
<defs><marker id="arrw" viewBox="0 0 10 10" refX="8" refY="5" markerWidth="6.5" markerHeight="6.5" orient="auto-start-reverse"><path d="M0 0L10 5L0 10z" fill="#cb333b"/></marker></defs>
|
||
<g id="cycle-arcs" fill="none" stroke="#cb333b" stroke-width="2">
|
||
<path marker-end="url(#arrw)"/>
|
||
<path marker-end="url(#arrw)"/>
|
||
<path marker-end="url(#arrw)"/>
|
||
<path marker-end="url(#arrw)"/>
|
||
</g>
|
||
</svg>
|
||
<div class="vcomp" style="left:592px; top:50px"><svg class="cicon" viewBox="0 0 34 34" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round" stroke-linejoin="round"><rect x="7" y="4" width="20" height="26" rx="2.5"/><path d="M12 11h10"/><path d="M12 16h10"/><path d="M12 21h6"/></svg><b>Letter reader</b><span class="d">every discharge letter, Italian & English</span></div>
|
||
<div class="vcomp" style="left:992px; top:135px"><svg class="cicon" viewBox="0 0 34 34" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round" stroke-linejoin="round"><circle cx="14.5" cy="14.5" r="8"/><path d="M20.5 20.5 L29 29"/><path d="M11 14.5 h7"/><path d="M14.5 11 v7"/></svg><b>Clinical matcher</b><span class="d">recognises diagnoses, procedures, test outcomes</span></div>
|
||
<div class="vcomp" style="left:592px; top:220px"><svg class="cicon" viewBox="0 0 34 34" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round" stroke-linejoin="round"><path d="M17 3.5 L28 8 v8.5 c0 7-4.7 11.6-11 14 C10.7 28.1 6 23.5 6 16.5 V8 Z"/><path d="M12.5 16.5 l3.2 3.2 L23.5 12"/></svg><b>Context guard</b><span class="d">negations & family history kept apart</span></div>
|
||
<div class="vcomp" style="left:192px; top:135px"><svg class="cicon" viewBox="0 0 34 34" fill="none" stroke="currentColor" stroke-width="2.2" stroke-linecap="round" stroke-linejoin="round"><circle cx="17" cy="7" r="3"/><circle cx="7" cy="26.5" r="3"/><circle cx="27" cy="26.5" r="3"/><path d="M17 10 v7"/><path d="M17 17 L8 23.5"/><path d="M17 17 L26 23.5"/></svg><b>Ontology sorter</b><span class="d">every finding into its clinical category</span></div>
|
||
</div>
|
||
<div class="note-row">
|
||
<div class="nitem"><b>Traceable.</b> Every finding carries the version of the rules that produced it (<code>pattern_version</code>): any number can be rebuilt years later.</div>
|
||
<div class="nitem"><b>Audited.</b> A new rule set goes live only after it proves ≥95% accuracy on 100 records reviewed by hand.</div>
|
||
</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">05 / 13</span></footer>
|
||
<aside class="notes">
|
||
[DRAFT — Sara] I'll take you inside these numbers. We read 58,438 discharge letters — every single one, every night. From them the text miner extracted 73,389 pathology records, classified into our clinical ontology. It parsed 10,908 drug-challenge tests, distinguishing the therapy from the actual test. And at the end of the chain: 2,307 patients with a confirmed Brugada pattern — a cohort nobody could have built by hand.
|
||
This is not a one-off migration: the pipeline runs every night, and today it processes on average two hundred and forty letters a month. Four simple pieces do the work: a letter reader for the Italian and English text, a clinical matcher that recognises diagnoses, procedures and test outcomes, a context guard that keeps negations and family history apart, and the ontology sorter that files every finding into its clinical category.
|
||
The key word here is deterministic: no black box. Every number can be traced back to the rule that produced it.
|
||
</aside>
|
||
</section>
|
||
|
||
|
||
<section class="arit nlp">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">Text analysis</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">Clinical NLP</div>
|
||
<h2>Reading clinical text is not keyword matching</h2>
|
||
<div class="hgap"></div>
|
||
<div class="split">
|
||
<div class="left">
|
||
<ul class="points">
|
||
<li><b>Negation</b> — “fibrillazione atriale <i>esclusa</i>”, “<i>non</i> FA” <span class="g">(“AF ruled out”, “no AF”)</span>: the rule scans a window around every match before deciding</li>
|
||
<li><b>Family ≠ patient</b> — “padre con FA” <span class="g">(“father with AF”)</span> is attributed to the family member, not to the patient</li>
|
||
<li><b>Two languages</b> — every pattern matches IT and EN forms: fibrillazione atriale / atrial fibrillation</li>
|
||
<li><b>Abbreviations & noise</b> — FA, f.a., “TA 140/90” (blood pressure, not a diagnosis)</li>
|
||
<li><b>One field, many statements</b> — the text is split into clauses before matching</li>
|
||
</ul>
|
||
</div>
|
||
<div class="right">
|
||
<div class="code">
|
||
<div class="code-head"><span>what the rules see</span><span>examples from the corpus</span></div>
|
||
<pre>“<i>padre con</i> fibrillazione atriale”
|
||
<span class="en">“father with atrial fibrillation”</span>
|
||
→ patologia: FA · attribuzione: <b>familiare</b>
|
||
<span class="en">pathology: AF · attribution: family</span>
|
||
|
||
“fibrillazione atriale <i>esclusa</i>”
|
||
<span class="en">“atrial fibrillation ruled out”</span>
|
||
→ patologia: FA · <b>negato</b>
|
||
<span class="en">pathology: AF · negated</span>
|
||
|
||
“test provocativo con flecainide
|
||
positivo per pattern Brugada”
|
||
<span class="en">“flecainide provocation test, positive for Brugada pattern”</span>
|
||
→ test: flecainide · esito: <b>POSITIVO</b>
|
||
<span class="en">test: flecainide · outcome: POSITIVE</span></pre>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">06 / 13</span></footer>
|
||
<aside class="notes">
|
||
[DRAFT — Sara] Why can't we just search for "FA"? Because clinical text lies to naive search. "Fibrillazione atriale esclusa" contains the words of a diagnosis but negates it — so every rule scans a window around the match, looking for negation cues. "Padre con FA" is real atrial fibrillation — but in the father, not the patient: we record it as family history. Letters mix Italian and English, abbreviations collide ("TA" is blood pressure, not a therapy), and one field can contain five different statements — so we split the text into clauses first.
|
||
Each of these problems has a specific, versioned solution. Sara to expand with real corpus examples.
|
||
</aside>
|
||
</section>
|
||
|
||
|
||
<section class="arit center-v">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">Text analysis</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">The clinical ontology</div>
|
||
<h2>Two tiers, one clinical order</h2>
|
||
<div class="onto-split">
|
||
<div class="onto-left">
|
||
<div class="tier-h">TIER 1 · arrhythmic<span class="cnt">11</span></div>
|
||
<div class="cats">
|
||
<span class="cat fill"><b>AF</b><span class="it">FA</span></span><span class="cat fill"><b>Brugada</b></span><span class="cat fill"><b>Flutter</b></span><span class="cat fill"><b>AT</b><span class="it">TA</span></span><span class="cat fill"><b>AVNRT</b><span class="it">TRN</span></span><span class="cat fill"><b>PSVT</b><span class="it">WPW / TPSV</span></span><span class="cat fill"><b>VT</b><span class="it">TV</span></span><span class="cat fill"><b>VF</b><span class="it">FV</span></span><span class="cat fill"><b>Ventricular ectopy</b><span class="it">Extrasistolia V.</span></span><span class="cat fill"><b>Syncope</b><span class="it">Sincope</span></span><span class="cat fill"><b>Long QT</b><span class="it">QT lungo</span></span>
|
||
</div>
|
||
<div class="tier-h">TIER 2 · structural<span class="cnt">5</span></div>
|
||
<div class="cats">
|
||
<span class="cat"><b>AV block</b><span class="it">Blocco AV</span></span><span class="cat"><b>Bundle branch block</b><span class="it">Blocco di branca</span></span><span class="cat"><b>Cardiomyopathy</b><span class="it">Cardiomiopatia</span></span><span class="cat"><b>Heart failure</b><span class="it">Scompenso</span></span><span class="cat"><b>Valvular disease</b><span class="it">Valvulopatia</span></span>
|
||
</div>
|
||
<div class="onto-points">
|
||
<div class="oitem"><span class="n">01</span><div><b>Order encodes clinical precedence</b> <span class="d">— “Brugada” is matched before “TV”, so “substrato per TV” <span class="g">(“substrate for VT”)</span> can't mask a Brugada pattern</span></div></div>
|
||
<div class="oitem"><span class="n">02</span><div><b>Synonyms live inline</b> <span class="d">— FA / f.a. / fib. atriale / atrial fibrillation → one canonical label</span></div></div>
|
||
<div class="oitem"><span class="n">03</span><div><b>Versioned like software</b> <span class="d">— semver for the pattern library, <code>pattern_version</code> on every extracted row</span></div></div>
|
||
</div>
|
||
</div>
|
||
<div class="out-panel">
|
||
<div class="out-h">The output: text becomes recorded data</div>
|
||
<div class="out-item"><span class="n">01</span><div><b>Quantitative data on the records</b><span class="d">every procedure and implant the text analysis reads is written back as structured, quantitative values on the record that describes it</span></div></div>
|
||
<div class="out-item"><span class="n">02</span><div><b>Straight into the DWH flow</b><span class="d">these records enter the data-warehouse generation like any other source</span></div></div>
|
||
<div class="out-item"><span class="n">03</span><div><b>As if typed at the visit</b><span class="d">the data lands exactly as if clinicians had keyed it in themselves during the visits</span></div></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">07 / 13</span></footer>
|
||
<aside class="notes">
|
||
[DRAFT — Sara] Extraction needs a target vocabulary — that's the ontology. Tier 1 holds the eleven arrhythmological categories that matter most for our research; Tier 2 holds five structural conditions. The order of the rules is itself clinical knowledge: Brugada patterns are tested before TV, because "substrato per TV" often appears in Brugada reports and would otherwise mask the diagnosis.
|
||
The ontology is versioned like software: a semantic version for the pattern library, stamped on every extracted row. When we add a synonym or fix a rule, the change is traceable — and the data can be rebuilt.
|
||
And this is the output of the whole work: reading a letter writes structured, quantitative values back onto the records that describe the procedures and implants it contains — and those records enter the warehouse generation exactly as if the clinicians had typed them during the visits. Text becomes data, indistinguishable from bedside data entry.
|
||
Sara: review the category list and the Italian labels before final. Labels are now English-first (audience is mostly foreign) — please confirm the EN terms, especially PSVT as the umbrella for WPW/TPSV and AT/AVNRT for TA/TRN.
|
||
</aside>
|
||
</section>
|
||
|
||
|
||
<section class="arit center-v validation">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">Text analysis</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<!-- Verified against aritmolab/chirone-etl on Gitea, commit 2ab00106188f298c1a0c1e2c48b3a4dbc66c45b5.
|
||
Exact sources and interpretation limits: ../slides/08-validation-sources.md.
|
||
SC-003/004 are targets; ~893 refers to false-positive new-letter records, not distinct patients. -->
|
||
<div class="kicker">Validation</div>
|
||
<h2>Trusting the text is a process, not a promise</h2>
|
||
<div class="onto-split">
|
||
<div class="onto-left">
|
||
<div class="onto-points">
|
||
<button type="button" class="oitem lens-target" id="validation-quality" data-lens="validation-quality" data-lens-number="1" data-lens-placement="center" data-lens-caption="Validation · 01 / 04" aria-label="Explore Quality gates" aria-expanded="false" aria-controls="slide-lens"><span class="n">01</span><span><b>Quality gates</b> <span class="d">— targets: ≥95% procedure accuracy on 100 manual reviews; ≥85% pathology coverage</span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">We separate correct classification from how often the rules find a condition.</p>
|
||
<p><b>Procedure classification</b>The specification sets a target of at least 95% agreement with a manual review of 100 randomly selected records: did the system assign the right procedure type?</p>
|
||
<p><b>Pathology coverage</b>The 85% target asks whether at least one recognized arrhythmia is extracted from a record, assuming most procedures concern known arrhythmias. It does not measure whether every diagnosis is correct or every disease is found.</p>
|
||
<p class="detail-example"><b>How to interpret the numbers</b>These are acceptance targets. Clinical review is still needed to detect incorrect labels and missed findings; coverage alone cannot establish clinical accuracy.</p>
|
||
</template>
|
||
</button>
|
||
<button type="button" class="oitem lens-target" id="validation-review" data-lens="validation-review" data-lens-number="2" data-lens-placement="center" data-lens-caption="Validation · 02 / 04" aria-label="Explore Clinical criteria" aria-expanded="false" aria-controls="slide-lens"><span class="n">02</span><span><b>Clinical criteria</b> <span class="d">— explicit criteria determine which patients belong in the analysis; a disease name alone is not enough</span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">Finding a disease name is the first step. Inclusion in an analysis depends on what the text means and the criteria for that cohort.</p>
|
||
<p><b>Read the context</b>“Father with Brugada” concerns a relative; “test negative for Brugada” reports a negative result; “suspected Brugada” expresses uncertainty. None of these phrases alone establishes a diagnosis in the patient.</p>
|
||
<p><b>Make the selection explicit</b>Queries apply the cohort criteria to prepare the dataset. Superset displays the result. The diagnosis view first excludes negated findings and findings attributed to relatives.</p>
|
||
<p class="detail-example"><b>A rule used in this project</b>For Brugada and long QT syndrome, that view also requires a positive provocative test or an ablation for the condition, at patient level. Other pathologies use only the first filter. These are project inclusion rules, not a universal diagnostic standard.</p>
|
||
</template>
|
||
</button>
|
||
<button type="button" class="oitem lens-target" id="validation-feedback" data-lens="validation-feedback" data-lens-number="3" data-lens-placement="center" data-lens-caption="Validation · 03 / 04" aria-label="Explore Feedback loop" aria-expanded="false" aria-controls="slide-lens"><span class="n">03</span><span><b>Feedback loop</b> <span class="d">— v1.3.1 fixed missed negations behind ~893 false-positive Brugada records</span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">The 11 June 2026 audit identified about 893 false-positive Brugada records in the newer letter format.</p>
|
||
<p><b>The error and the fix</b>“Test alla flecainide negativo per sindrome di Brugada” was treated as an affirmed finding: the rules recognized “negato”, but missed “negativo”. Version 1.3.1 added “negativo”, “negativa” and “negativi” before and after the condition.</p>
|
||
<p><b>Make the correction repeatable</b>Regression tests require that this sentence still produces a Brugada finding, now marked as negated. A separate test checks that an affirmed Brugada diagnosis remains positive. Reprocessing applies the revised rules to historical letters.</p>
|
||
<p class="detail-example"><b>What changed</b>The finding is reclassified, not erased. The ~893 figure counts affected records, not necessarily distinct patients; the cohort filters can now exclude those negated findings.</p>
|
||
</template>
|
||
</button>
|
||
<button type="button" class="oitem lens-target" id="validation-context" data-lens="validation-context" data-lens-number="4" data-lens-placement="center" data-lens-caption="Validation · 04 / 04" aria-label="Explore Nothing is silently dropped" aria-expanded="false" aria-controls="slide-lens"><span class="n">04</span><span><b>Nothing is silently dropped</b> <span class="d">— negated and family-attributed findings are stored too, filtered only at the mart layer</span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">A recognized finding can be kept even when it is negated or refers to someone else.</p>
|
||
<p><b>Store context alongside the finding</b>The extractor looks up to 60 characters before and after the match, stopping at a full stop, semicolon or line break. It records whether the finding is negated and whether it concerns the patient or a relative.</p>
|
||
<p><b>Filter when building the research dataset</b>Those flags travel through integration and the warehouse to the mentions dataset. The confirmed-diagnosis view excludes negated and family findings, while the underlying dataset remains available for other analyses.</p>
|
||
<p class="detail-example"><b>Preservation has a defined scope</b>Repeated matches for the same pathology are consolidated, preferring an affirmed patient finding. “Suspected” or “to exclude” is not automatically treated as a negation. These limits remain explicit for clinical review.</p>
|
||
</template>
|
||
</button>
|
||
</div>
|
||
<div class="roi roi-quiet"><b class="k">Quality gate</b>Manual comparison checks procedure accuracy. Pathology coverage and clinical correctness are assessed separately.</div>
|
||
</div>
|
||
<button type="button" class="out-panel lens-target" id="validation-advantages" data-lens="validation-advantages" data-lens-number="5" data-lens-title="The advantages" data-lens-placement="center" data-lens-caption="Validation · Advantages" aria-label="Explore the advantages: speed, precision, determinism" aria-expanded="false" aria-controls="slide-lens">
|
||
<span class="out-h">The advantage: speed, precision, determinism</span>
|
||
<span class="out-item"><span class="n">01</span><span><b>Machine speed</b><span class="d">twenty years of letters are read by rules, not by hand; when a pattern improves, the whole archive can be reprocessed</span></span></span>
|
||
<span class="out-item"><span class="n">02</span><span><b>Explicit quality targets</b><span class="d">≥95% procedure accuracy on a manual sample; ≥85% pathology coverage, with clinical review of the resulting cohort</span></span></span>
|
||
<span class="out-item"><span class="n">03</span><span><b>Deterministic by design</b><span class="d">the same letter and rule version yield the same extraction: no LLM sampling, no randomness, results traceable to versioned rules</span></span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">Speed, precision and determinism each address a different part of making clinical text usable for research.</p>
|
||
<p><b>01 · Speed: apply a correction across the archive</b>Once defined, the extraction rules process letters automatically. An improved rule can be applied again to historical text, without manually relabelling every record.</p>
|
||
<p><b>02 · Precision: make errors visible and correctable</b>Explicit quality targets, clinical context and cohort filters make the results inspectable. The Brugada correction shows how an audit can identify a recurring error, and regression tests can protect the fix. Quality still needs measurement and clinical review.</p>
|
||
<p><b>03 · Determinism: reproduce the extraction</b>With the same input, rule version and configuration, extraction produces the same findings. There is no language-model sampling at this step. Versioning explains why a result changes when the rules change.</p>
|
||
</template>
|
||
</button>
|
||
</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">08 / 13</span></footer>
|
||
<aside class="notes">
|
||
<p><button type="button" data-note-lens="validation-quality">1. Quality gates.</button> We set a 95% procedure-accuracy target, against 100 manual reviews. The 85% pathology target measures coverage, not diagnostic accuracy.</p>
|
||
<p><button type="button" data-note-lens="validation-review">2. Clinical criteria.</button> Finding a disease name in a letter is only the first step. We apply explicit clinical criteria to decide which patients belong in the analysis.</p>
|
||
<p><button type="button" data-note-lens="validation-feedback">3. Feedback loop.</button> Version 1.3.1 corrected missed negations in roughly 900 Brugada records, with regression tests protecting the fix.</p>
|
||
<p><button type="button" data-note-lens="validation-context">4. Nothing is silently dropped.</button> Negated and family findings retain their context. Research datasets filter them explicitly, so preserving a finding does not mean counting it as the patient’s diagnosis.</p>
|
||
<p><button type="button" data-note-lens="validation-advantages">5. The advantages.</button> We can reprocess the archive quickly, inspect the rules behind each result, and reproduce the same extraction with the same rules.</p>
|
||
</aside>
|
||
</section>
|
||
|
||
<!-- ============ 09 · ARITMOLAB TODAY ============ -->
|
||
<section class="arit">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">AritmoLab today</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">The portal, today</div>
|
||
<h2>AritmoLab — a quick tour</h2>
|
||
<div class="screenshot-tour" aria-label="AritmoLab tour, screenshots 1 to 7">
|
||
<button type="button" data-screenshot="home" data-screenshot-number="1" data-screenshot-title="Home page" aria-haspopup="dialog"><img src="screenshots/1-HomePage.png" alt="" loading="lazy"><span><b>1</b> Home page</span></button>
|
||
<button type="button" data-screenshot="patients" data-screenshot-number="2" data-screenshot-title="Patient list" aria-haspopup="dialog"><img src="screenshots/2-PatientList.png" alt="" loading="lazy"><span><b>2</b> Patient list</span></button>
|
||
<button type="button" data-screenshot="profile" data-screenshot-number="3" data-screenshot-title="Patient profile" aria-haspopup="dialog"><img src="screenshots/3-PatientGeneralData.png" alt="" loading="lazy"><span><b>3</b> Patient profile</span></button>
|
||
<button type="button" data-screenshot="history" data-screenshot-number="4" data-screenshot-title="Clinical history" aria-haspopup="dialog"><img src="screenshots/4-PatientDetail.png" alt="" loading="lazy"><span><b>4</b> Clinical history</span></button>
|
||
<button type="button" data-screenshot="procedure" data-screenshot-number="5" data-screenshot-title="Procedure details" aria-haspopup="dialog"><img src="screenshots/5-PatientProcedureDetails.png" alt="" loading="lazy"><span><b>5</b> Procedure details</span></button>
|
||
<button type="button" data-screenshot="dashboards" data-screenshot-number="6" data-screenshot-title="Dashboard catalogue" aria-haspopup="dialog"><img src="screenshots/6-DashboardsList.png" alt="" loading="lazy"><span><b>6</b> Dashboard catalogue</span></button>
|
||
<button type="button" data-screenshot="brugada" data-screenshot-number="7" data-screenshot-title="Brugada dashboard" aria-haspopup="dialog"><img src="screenshots/7-Brugada-dashboard.png" alt="" loading="lazy"><span><b>7</b> Brugada dashboard</span></button>
|
||
</div>
|
||
<p class="hint">Select a screen to enlarge · Follow the tour from 1 to 7</p>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">09 / 13</span></footer>
|
||
<aside class="notes">
|
||
<p><button type="button" data-note-screenshot="home">1. Home page.</button> The home page summarises around 57,000 patients, by age, sex and geographical origin. It is the starting point for exploring the archive.</p>
|
||
<p><button type="button" data-note-screenshot="patients">2. Patient list.</button> Search filters help us find a patient or study participant, open their record, or export the results.</p>
|
||
<p><button type="button" data-note-screenshot="profile">3. Patient profile.</button> The profile brings demographic and clinical fields together, with access to procedures, devices, diagnostic examinations and genetics.</p>
|
||
<p><button type="button" data-note-screenshot="history">4. Clinical history.</button> A dated timeline brings together clinical notes, discharge letters and procedures, retaining the source of each event.</p>
|
||
<p><button type="button" data-note-screenshot="procedure">5. Procedure details.</button> Here, an ablation record shows the treated arrhythmias, procedural details, recorded complications and conclusions.</p>
|
||
<p><button type="button" data-note-screenshot="dashboards">6. Dashboard catalogue.</button> We then move from individual records to dashboards covering departmental activity, procedures, devices and genetics.</p>
|
||
<p><button type="button" data-note-screenshot="brugada">7. Brugada dashboard.</button> The funnel separates text mentions, filtered mentions, confirmed cases and ablation outcomes. Other charts describe sex, age, annual diagnoses and the timing of pre- and post-assessments.</p>
|
||
</aside>
|
||
</section>
|
||
<!-- ============ 10 · CRISIS ============ -->
|
||
<section class="arit crisis">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">The crisis</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">All good? Not yet.</div>
|
||
<h2>The warehouse speaks SQL. Research needs more.</h2>
|
||
<div class="hgap"></div>
|
||
<div class="crisis-points" aria-label="Four barriers between the warehouse and research">
|
||
<button type="button" class="oitem lens-target" id="crisis-language" data-lens="crisis-language" data-lens-number="1" data-lens-title="From clinical question to SQL" data-lens-placement="center" data-lens-caption="The research gap · 01 / 04" aria-expanded="false" aria-controls="slide-lens"><span class="n">01</span><span><b>Clinical questions need translation</b><span class="d">Clinicians define the question; a query must express it across linked tables.</span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">Knowing what to ask is different from knowing how the database stores the answer.</p>
|
||
<p><b>The clinical question</b>“How many patients with confirmed Brugada underwent an ablation?” requires agreed definitions of the cohort and procedure.</p>
|
||
<p><b>The engineering task</b>SQL must connect diagnoses, patients and procedures, apply the time window and count each patient once, even when several records describe the same person.</p>
|
||
<p class="detail-example"><b>The bridge</b>Clinical expertise defines the meaning. A reviewed query turns that meaning into an explicit, checkable selection.</p>
|
||
</template>
|
||
</button>
|
||
<button type="button" class="oitem lens-target" id="crisis-intelligence" data-lens="crisis-intelligence" data-lens-number="2" data-lens-title="Health Intelligence: describe what happened" data-lens-placement="center" data-lens-caption="The research gap · 02 / 04" aria-expanded="false" aria-controls="slide-lens"><span class="n">02</span><span><b>Health Intelligence needs shared definitions</b><span class="d">Volumes, diagnoses and outcomes need consistent groups, periods and denominators.</span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">A dashboard needs agreed indicators, not simply a chart drawn over raw records.</p>
|
||
<p><b>Define what is counted</b>Patients, admissions and procedures answer different questions. For an outcome percentage, specify which patients are eligible and the observation period.</p>
|
||
<p><b>Prepare comparable summaries</b>Aggregate by year, procedure or patient group using the same definitions. Keep missing information visible so that changes in documentation are not mistaken for changes in care.</p>
|
||
<p class="detail-example"><b>Example</b>Annual ablation volumes describe activity. An outcome percentage also needs a defined denominator and follow-up window.</p>
|
||
</template>
|
||
</button>
|
||
<button type="button" class="oitem lens-target" id="crisis-prediction" data-lens="crisis-prediction" data-lens-number="3" data-lens-title="Predictive research: build the study table" data-lens-placement="center" data-lens-caption="The research gap · 03 / 04" aria-expanded="false" aria-controls="slide-lens"><span class="n">03</span><span><b>Predictive research needs a study dataset</b><span class="d">A defined cohort, consistent variables and outcomes measured over an agreed period.</span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">Linked clinical records must become a table shaped around the study question.</p>
|
||
<p><b>Define one row</b>Choose the unit of analysis: for example, one patient or one procedure. Place the selected characteristics in columns, with consistent units and explicit handling of missing values.</p>
|
||
<p><b>Respect the timeline</b>Define when prediction would occur and when the outcome is assessed. Predictor variables must contain only information available at that prediction time.</p>
|
||
<p class="detail-example"><b>Example</b>To study outcomes after ablation, separate pre-procedure characteristics from later observations. This prevents future information from leaking into the prediction.</p>
|
||
</template>
|
||
</button>
|
||
<button type="button" class="oitem lens-target" id="crisis-datamarts" data-lens="crisis-datamarts" data-lens-number="4" data-lens-title="Datamarts: the preparation bottleneck" data-lens-placement="center" data-lens-caption="The research gap · 04 / 04" aria-expanded="false" aria-controls="slide-lens"><span class="n">04</span><span><b>Hand-made datamarts are the bottleneck</b><span class="d">Each question requires selection, joins, checks and a reproducible dataset.</span></span>
|
||
<template class="lens-details">
|
||
<p class="detail-intro">A datamart is a focused dataset prepared for a particular analysis.</p>
|
||
<p><b>More than writing SQL</b>Someone must agree the cohort, connect the sources, resolve duplicate records and check missing values and patient counts. A query can run successfully and still answer the wrong question.</p>
|
||
<p><b>Make the work repeatable</b>Keep the selection rules, query and checks together, so the dataset can be rebuilt when the data or study definition changes.</p>
|
||
<p class="detail-example"><b>Where assistance helps</b>AI can draft the query. Clinicians and engineers still review its meaning and results before using the dataset.</p>
|
||
</template>
|
||
</button>
|
||
</div>
|
||
<div class="roi"><b class="k">The gap</b>Twenty years of data, one warehouse — and no fast road from a research question to an answer.</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">10 / 13</span></footer>
|
||
<aside class="notes">
|
||
<p><strong>0. What is SQL?</strong><br>SQL means Structured Query Language. It tells a database what to select, connect and count.</p>
|
||
<p><button type="button" data-note-lens="crisis-language">1. Clinical questions.</button><br>“How many patients with confirmed Brugada underwent an ablation?” We must define confirmation and count each patient once.</p>
|
||
<p><button type="button" data-note-lens="crisis-intelligence">2. Health Intelligence.</button><br>Dashboards need agreed definitions, time periods and denominators to make comparisons meaningful.</p>
|
||
<p><button type="button" data-note-lens="crisis-prediction">3. Predictive research.</button><br>Study tables separate characteristics known before prediction from outcomes observed afterwards.</p>
|
||
<p><button type="button" data-note-lens="crisis-datamarts">4. Datamarts.</button><br>A datamart is an analysis dataset with explicit selection rules and repeatable checks.</p>
|
||
<p><strong>5. The gap.</strong><br>The gap is between clinical meaning and database instructions: storing data does not automatically make a question answerable.</p>
|
||
</aside>
|
||
</section>
|
||
|
||
<!-- ============ 11 · THOTHII TOUR ============ -->
|
||
<section class="arit">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">A tour of ThothII</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">ThothII, step by step</div>
|
||
<h2>From question to datamart — the app</h2>
|
||
<div class="shots">
|
||
<div class="shot">screenshot</div><div class="shot">screenshot</div><div class="shot">screenshot</div>
|
||
<div class="shot">screenshot</div><div class="shot">screenshot</div><div class="shot">screenshot</div>
|
||
</div>
|
||
<p class="hint">[MP: 6-7 real screenshots of ThothII — ask, review SQL, datamart, dashboard]</p>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">11 / 13</span></footer>
|
||
<aside class="notes">[DRAFT] Let me walk you through ThothII: the researcher asks the question in plain English; the AI proposes the SQL; the query is reviewed; the datamart is assembled; the dashboard comes alive. Six screens, a few minutes. MP: real screenshots.</aside>
|
||
</section>
|
||
<!-- ============ 12 · HAPPY ENDING ============ -->
|
||
<section class="arit">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">The happy ending</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">From datamart to discovery</div>
|
||
<h2>One datamart — many questions answered</h2>
|
||
<div class="charts">
|
||
<div class="chart"><b>Multivariate analysis</b><span>[SVG chart — next step]</span></div>
|
||
<div class="chart"><b>Machine learning</b><span>[SVG chart — next step]</span></div>
|
||
</div>
|
||
<p class="hint">[Illustrative data, modeled on published arrhythmology predictors — Brugada focus]</p>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">12 / 13</span></footer>
|
||
<aside class="notes">[DRAFT — next step: two SVG charts, illustrative data modeled on real literature] And this is the happy ending. From one datamart, the Unit can run a multivariate analysis — which factors truly drive arrhythmic risk in Brugada patients — and train machine-learning models on the same table. What used to take weeks of manual data preparation now takes minutes. The AI did not replace the researcher: it gave the researcher back their time.</aside>
|
||
</section>
|
||
|
||
<section class="arit title-slide">
|
||
<header class="head">
|
||
<img src="logo.png" alt="">
|
||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||
</header>
|
||
<nav class="band">
|
||
<span class="band-section">Thank you</span>
|
||
</nav>
|
||
<div class="sbody">
|
||
<div class="kicker">Questions?</div>
|
||
<h1>Thank you</h1>
|
||
<div class="hgap"></div>
|
||
<p class="lead">Want the deep dives? The technical walkthroughs of the text miner, the mapping agents and ThothII are coming as videos on our Substack.</p>
|
||
<div class="speakers" style="margin-top:32px">
|
||
<span>Dr. Marco Pancotti - MultiPhysixLab</span>
|
||
<span>Dr. Sara Paratico - I.R.C.C.S. Policlinico San Donato</span>
|
||
</div>
|
||
</div>
|
||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">13 / 13</span></footer>
|
||
<aside class="notes">
|
||
[DRAFT] Thank you for your attention. If you want to go deeper — the text miner, the mapping agents, ThothII — we are publishing the technical walkthroughs as videos on our Substack; the link is on the final version of this deck. And now, happy to take your questions.
|
||
</aside>
|
||
</section>
|
||
|
||
</div></div>
|
||
|
||
<!-- AI contribution popup -->
|
||
<div class="ai-pop" id="aiPop">
|
||
<svg class="ai-line"><line x1="0" y1="0" x2="0" y2="0"></line><circle cx="0" cy="0" r="7"></circle></svg>
|
||
<div class="ai-pop-card" role="dialog" aria-modal="true">
|
||
<button class="ai-pop-x" aria-label="Close">✕</button>
|
||
<div class="ai-pop-kicker" id="aiPopKicker">AI contribution</div>
|
||
<div class="ai-pop-title"><span class="brain brain-btn" id="aiPopBrain"><img class="artificial-brain" src="artificial-brain.svg" alt=""></span><b id="aiPopTitle"></b></div>
|
||
<div id="aiPopBody"></div>
|
||
</div>
|
||
</div>
|
||
|
||
<script src="vendor/reveal/dist/reveal.js"></script>
|
||
<script src="vendor/reveal/plugin/notes/notes.js?v=3"></script>
|
||
<script src="slide-lens.js?v=4"></script>
|
||
<script src="screenshot-tour.js?v=1"></script>
|
||
<script>
|
||
Reveal.initialize({
|
||
width: 1280, height: 720, margin: 0,
|
||
hash: true, controls: false, progress: false,
|
||
transition: 'fade', overview: true,
|
||
plugins: [RevealNotes],
|
||
});
|
||
|
||
// jump dropdown — one select per slide, in the red band
|
||
const JUMP_TITLES = ['Title', 'Where we started', 'What we wanted to build',
|
||
'The problem', 'What the text miner reads', 'Hard problems of clinical NLP', 'The clinical ontology',
|
||
'Trust the text', 'AritmoLab today', 'All good? Not yet', 'A tour of ThothII',
|
||
'From datamarts to models', 'Thank you'];
|
||
document.querySelectorAll('.arit .band').forEach((band) => {
|
||
const sel = document.createElement('select');
|
||
sel.className = 'jump';
|
||
sel.setAttribute('aria-label', 'Jump to slide');
|
||
sel.innerHTML = JUMP_TITLES.map((t, i) => `<option value="${i}">${String(i + 1).padStart(2, '0')} · ${t}</option>`).join('');
|
||
sel.addEventListener('change', () => { Reveal.slide(parseInt(sel.value, 10)); sel.blur(); });
|
||
band.appendChild(sel);
|
||
});
|
||
Reveal.on('slidechanged', () => {
|
||
const idx = Reveal.getIndices().h;
|
||
document.querySelectorAll('.arit .band select.jump').forEach((s) => { s.selectedIndex = idx; s.blur(); });
|
||
});
|
||
|
||
// slide "The problem": rail drops from Marts center, ThothII arrow rises to Marts.
|
||
// Reveal scales the canvas — all rect offsets are divided by the current scale.
|
||
function alignProblem() {
|
||
const pipe = document.getElementById('pipeline');
|
||
if (!pipe || !pipe.closest('section').classList.contains('present')) return;
|
||
const s = Reveal.getScale() || 1;
|
||
const pr = pipe.getBoundingClientRect();
|
||
const marts = document.getElementById('marts').getBoundingClientRect();
|
||
const rail = document.getElementById('rail').getBoundingClientRect();
|
||
const rwrap = document.getElementById('railwrap').getBoundingClientRect();
|
||
const drop = document.getElementById('conn-drop');
|
||
const elbow = document.getElementById('conn-elbow');
|
||
const cx = (marts.left + marts.width / 2 - rwrap.left) / s;
|
||
const railLeft = (rail.left - rwrap.left) / s;
|
||
const railTop = (rail.top - rwrap.top) / s;
|
||
const mBottom = (marts.bottom - rwrap.top) / s;
|
||
drop.style.left = (cx - 1) + 'px';
|
||
drop.style.top = mBottom + 'px';
|
||
drop.style.height = Math.max(0, railTop - mBottom) + 'px';
|
||
elbow.style.left = railLeft + 'px';
|
||
elbow.style.width = Math.max(0, cx - railLeft) + 'px';
|
||
elbow.style.top = (railTop - 1) + 'px';
|
||
const twrap = document.getElementById('thothwrap');
|
||
const box = twrap.querySelector('.thoth-box');
|
||
const tr = twrap.getBoundingClientRect();
|
||
const br = box.getBoundingClientRect();
|
||
const conn = document.getElementById('t-conn');
|
||
const head = document.getElementById('t-head');
|
||
const x = tr.width / s - 10;
|
||
const top = (pr.bottom - tr.top) / s + 2;
|
||
const boxTop = (br.top - tr.top) / s;
|
||
conn.style.left = x + 'px';
|
||
conn.style.top = top + 'px';
|
||
conn.style.height = Math.max(0, boxTop - top) + 'px';
|
||
head.style.left = x + 'px';
|
||
head.style.top = (top - 7) + 'px';
|
||
}
|
||
addEventListener('load', alignProblem);
|
||
addEventListener('resize', alignProblem);
|
||
Reveal.on('slidechanged', alignProblem);
|
||
Reveal.on('ready', alignProblem);
|
||
if (document.fonts && document.fonts.ready) document.fonts.ready.then(alignProblem);
|
||
|
||
// slide "What the text miner reads": curved arrows between the four components,
|
||
// drawn from the measured edges of each block so every arrow starts and lands precisely.
|
||
function alignCycle() {
|
||
const cyc = document.getElementById('cycle');
|
||
if (!cyc || !cyc.closest('section').classList.contains('present')) return;
|
||
const s = Reveal.getScale() || 1;
|
||
const cr = cyc.getBoundingClientRect();
|
||
const boxes = [...cyc.querySelectorAll('.vcomp')].map(c => {
|
||
const r = c.getBoundingClientRect();
|
||
const l = (r.left - cr.left) / s, t = (r.top - cr.top) / s, w = r.width / s, h = r.height / s;
|
||
return { l, t, w, h, cx: l + w / 2, cy: t + h / 2 };
|
||
});
|
||
function edgePoint(from, to, box) {
|
||
const dx = to.cx - from.cx, dy = to.cy - from.cy;
|
||
let t = 1;
|
||
if (dx > 0) t = Math.min(t, ((box.l + box.w) - from.cx) / dx);
|
||
if (dx < 0) t = Math.min(t, (box.l - from.cx) / dx);
|
||
if (dy > 0) t = Math.min(t, ((box.t + box.h) - from.cy) / dy);
|
||
if (dy < 0) t = Math.min(t, (box.t - from.cy) / dy);
|
||
return { x: from.cx + dx * t, y: from.cy + dy * t };
|
||
}
|
||
const paths = cyc.querySelectorAll('#cycle-arcs path');
|
||
const seq = [0, 1, 2, 3];
|
||
for (let k = 0; k < 4; k++) {
|
||
const a = boxes[seq[k]], b = boxes[seq[(k + 1) % 4]];
|
||
const start = edgePoint(a, b, a), end = edgePoint(b, a, b);
|
||
const mid = { x: (start.x + end.x) / 2, y: (start.y + end.y) / 2 };
|
||
let ox = mid.x - 592, oy = mid.y - 135;
|
||
const len = Math.hypot(ox, oy) || 1;
|
||
const cpx = mid.x + (ox / len) * 46, cpy = mid.y + (oy / len) * 46;
|
||
paths[k].setAttribute('d',
|
||
`M ${start.x.toFixed(1)} ${start.y.toFixed(1)} Q ${cpx.toFixed(1)} ${cpy.toFixed(1)} ${end.x.toFixed(1)} ${end.y.toFixed(1)}`);
|
||
}
|
||
}
|
||
addEventListener('load', alignCycle);
|
||
addEventListener('resize', alignCycle);
|
||
Reveal.on('slidechanged', alignCycle);
|
||
Reveal.on('ready', alignCycle);
|
||
if (document.fonts && document.fonts.ready) document.fonts.ready.then(alignCycle);
|
||
|
||
const AI_CONTRIBS = {
|
||
ingestion: {
|
||
title: 'The mappings',
|
||
body: `
|
||
<p class="lead">The mappings that turn three raw sources into clean data are <b>crafted by AI</b> — <b>revised by humans</b>.</p>
|
||
<ul class="ai-pop-list">
|
||
<li><b>AI drafts</b> — LLM agents write the mappings from prompts and schema samples</li>
|
||
<li><b>What is a mapping?</b> — a plain instruction sheet that tells the system, field by field, where each piece of data comes from and where it has to land</li>
|
||
<li><b>At scale</b> — <span class="n">19,764</span> lines of instructions, all machine-validated</li>
|
||
<li><b>Humans revise</b> — every line is checked, then recorded in <b>git</b>, the archive that keeps every version of the instructions and can undo any change</li>
|
||
</ul>
|
||
<div class="roi"><b class="k">Positive effects</b>Weeks of hand-writing became days of reviewing.</div>`,
|
||
},
|
||
integration: {
|
||
title: 'Reading the clinical text',
|
||
body: `
|
||
<p class="lead">A deterministic, bilingual (IT/EN) pattern library reads the free text inside the nightly pipeline.</p>
|
||
<ul class="ai-pop-list">
|
||
<li><b>Pathologies</b> — <span class="n">73,389</span> extracted from <span class="n">58,438</span> discharge letters, into a two-tier clinical ontology</li>
|
||
<li><b>Procedures ↔ pathologies</b> — linked at clause level</li>
|
||
<li><b>Drug-challenge tests</b> — <span class="n">10,908</span> parsed (flecainide, ajmaline, adrenaline, isoprenaline)</li>
|
||
<li><b>No black box</b> — semver rules, <code>pattern_version</code> on every row, <span class="n">≥95%</span> accuracy gate on 100 manual reviews</li>
|
||
</ul>
|
||
<div class="roi"><b class="k">Positive effects</b>58,438 letters of free text became an analysable research asset — automatically, every night.</div>`,
|
||
},
|
||
datamarts: {
|
||
title: 'Datamarts from plain English',
|
||
body: `
|
||
<p class="lead">Researchers ask in plain English; the AI writes the SQL over the star schema.</p>
|
||
<ul class="ai-pop-list">
|
||
<li><b>Ask</b> — “how many Brugada patients had an effective ablation?”</li>
|
||
<li><b>Generate</b> — AI builds the SQL and assembles a curated datamart</li>
|
||
<li><b>Serve</b> — results flow to Superset for statistics and machine learning</li>
|
||
<li><b>In the loop</b> — the researcher reviews every proposed query before it runs</li>
|
||
</ul>
|
||
<div class="roi"><b class="k">Positive effects</b>A new research datamart in minutes instead of weeks of hand-written SQL — with the researcher approving every query.</div>`,
|
||
},
|
||
};
|
||
const pop = document.getElementById('aiPop');
|
||
const card = pop.querySelector('.ai-pop-card');
|
||
const line = pop.querySelector('.ai-line line');
|
||
const dot = pop.querySelector('.ai-line circle');
|
||
let activePopup = null;
|
||
const closePop = () => {
|
||
pop.classList.remove('open');
|
||
document.querySelectorAll('.brain-btn[aria-pressed="true"]').forEach(button => button.setAttribute('aria-pressed', 'false'));
|
||
activePopup = null;
|
||
SlideLens.close();
|
||
ScreenshotTour.close();
|
||
window.dispatchEvent(new Event('presentationchange'));
|
||
};
|
||
Reveal.on('slidechanged', closePop);
|
||
const placeNear = (brainEl) => {
|
||
const b = brainEl.getBoundingClientRect();
|
||
const vw = window.innerWidth, vh = window.innerHeight;
|
||
card.style.width = '';
|
||
card.style.maxHeight = '';
|
||
if (pop.classList.contains('is-ai')) {
|
||
const margin = 24, gap = Math.min(150, Math.max(80, vw * .12));
|
||
const roomRight = vw - margin - b.right - gap;
|
||
const roomLeft = b.left - gap - margin;
|
||
const onRight = roomRight >= roomLeft;
|
||
const by = b.top + b.height / 2;
|
||
let left, top, sx, sy, ex, ey;
|
||
if (Math.max(roomRight, roomLeft) >= 260) {
|
||
card.style.width = `${Math.min(540, Math.max(roomRight, roomLeft))}px`;
|
||
const cw = card.offsetWidth, ch = card.offsetHeight;
|
||
left = onRight ? b.right + gap : b.left - gap - cw;
|
||
top = Math.max(margin, Math.min(vh - margin - ch, by - ch / 2));
|
||
sx = onRight ? b.right + 8 : b.left - 8; sy = by;
|
||
ex = onRight ? left : left + cw;
|
||
ey = Math.max(top + 24, Math.min(by, top + ch - 24));
|
||
} else {
|
||
// Narrow windows: use a vertical connector instead of covering the icon.
|
||
const below = vh - b.bottom >= b.top;
|
||
const verticalGap = 54;
|
||
card.style.maxHeight = `${Math.max(80, (below ? vh - b.bottom : b.top) - verticalGap - margin)}px`;
|
||
left = Math.max(margin, Math.min(vw - margin - card.offsetWidth, b.left + b.width / 2 - card.offsetWidth / 2));
|
||
top = below ? b.bottom + verticalGap : b.top - verticalGap - card.offsetHeight;
|
||
sx = b.left + b.width / 2; sy = below ? b.bottom + 8 : b.top - 8;
|
||
ex = Math.max(left + 24, Math.min(sx, left + card.offsetWidth - 24));
|
||
ey = below ? top : top + card.offsetHeight;
|
||
}
|
||
card.style.left = `${left}px`; card.style.top = `${top}px`;
|
||
line.setAttribute('x1', sx); line.setAttribute('y1', sy);
|
||
line.setAttribute('x2', ex); line.setAttribute('y2', ey);
|
||
dot.setAttribute('cx', sx); dot.setAttribute('cy', sy);
|
||
return;
|
||
}
|
||
const cw = card.offsetWidth, ch = card.offsetHeight;
|
||
const bx = b.left + b.width / 2, by = b.top + b.height / 2;
|
||
const gap = 52;
|
||
let left;
|
||
if (bx + gap + cw < vw - 12) left = bx + gap;
|
||
else if (bx - gap - cw > 12) left = bx - gap - cw;
|
||
else left = Math.max(12, Math.min(vw - cw - 12, bx - cw / 2));
|
||
const top = Math.max(12, Math.min(vh - ch - 12, by - ch / 2));
|
||
card.style.left = Math.round(left) + 'px';
|
||
card.style.top = Math.round(top) + 'px';
|
||
const ly = Math.max(top + 24, Math.min(by, top + ch - 24));
|
||
const ax = left < bx ? left + cw : left;
|
||
line.setAttribute('x1', bx); line.setAttribute('y1', by);
|
||
line.setAttribute('x2', ax); line.setAttribute('y2', ly);
|
||
dot.setAttribute('cx', bx); dot.setAttribute('cy', by);
|
||
};
|
||
const MISSING_CONTRIBS = {
|
||
ci: { kicker: 'The missing piece · 20 years ago', title: 'Clinical Intelligence — the history, made visible',
|
||
body: `<p class="lead">Twenty years of clinical history, with no way to show it.</p>
|
||
<ul class="ai-pop-list"><li><b>What it is</b> — dashboards on the Unit's activity and its patients: volumes, incidence, outcomes</li>
|
||
<li><b>Who needed it</b> — management, audit, clinical research</li>
|
||
<li><b>Why it never existed</b> — the data was in prose; building dashboards on text was impractical</li></ul>
|
||
<div class="roi"><b class="k">The gap</b>No fast answer to “how are we doing?” — every question meant a manual query.</div>` },
|
||
dwh: { kicker: 'The missing piece · 20 years ago', title: 'Datawarehouse — one analysis-ready home',
|
||
body: `<p class="lead">One home for every record the Unit produces: text, structured fields, genetics, ECG.</p>
|
||
<ul class="ai-pop-list"><li><b>What it is</b> — a single schema joining all the sources, designed for analysis</li>
|
||
<li><b>Why it never existed</b> — each system kept its own silo; joining them meant hand-stitching exports</li></ul>
|
||
<div class="roi"><b class="k">The gap</b>Every research question started with manual exports — and ended in fragile spreadsheets.</div>` },
|
||
ml: { kicker: 'The missing piece · 20 years ago', title: 'ML-ready data — cohorts for research',
|
||
body: `<p class="lead">Tables shaped for statistics and model training: clean, wide, reproducible.</p>
|
||
<ul class="ai-pop-list"><li><b>What it is</b> — one row per patient, features beside outcomes, rebuildable at will</li>
|
||
<li><b>Why it never existed</b> — building it by hand from free text and separate silos took weeks</li></ul>
|
||
<div class="roi"><b class="k">The gap</b>Predictive research on the Unit's data stayed out of reach.</div>` },
|
||
portal: { kicker: 'The missing piece · 20 years ago', title: 'Professional Portal — the 360° view',
|
||
body: `<p class="lead">The 360° view of the patient: Cardioref and genetics finally in one place.</p>
|
||
<ul class="ai-pop-list"><li><b>What it is</b> — the grown-up Omics Portal: management, care and research on the same platform</li>
|
||
<li><b>Why it never existed</b> — the idea was there; the integrated data to feed it was not</li></ul>
|
||
<div class="roi"><b class="k">The gap</b>A single door for patients and clinicians — promised, never delivered.</div>` },
|
||
};
|
||
const START_CONTRIBS = {
|
||
cardioref: {
|
||
kicker: 'The starting point · 20 years ago',
|
||
title: 'Cardioref — the daily workhorse',
|
||
body: `
|
||
<p class="lead">For twenty years, the database running the Arrhythmology Unit.</p>
|
||
<ul class="ai-pop-list">
|
||
<li><b>Used for</b> — patient records, the Unit's daily activity, planning and logging of everything that was done</li>
|
||
<li><b>Good at it</b> — as a management database, it did (and still does) its job</li>
|
||
<li><b>The catch</b> — most clinical information lives in free text, outside the structured fields</li>
|
||
</ul>
|
||
<div class="roi"><b class="k">Problem</b>Almost unusable for research and data extraction: the knowledge is there — locked in prose.</div>`,
|
||
},
|
||
genetic: {
|
||
kicker: 'The starting point · 20 years ago',
|
||
title: 'Genetic data — an island',
|
||
body: `
|
||
<p class="lead">A dedicated system collecting the genetic results of our patients.</p>
|
||
<ul class="ai-pop-list">
|
||
<li><b>Used for</b> — storing genetic tests, variants and lab reports</li>
|
||
<li><b>The catch</b> — its only output is Excel sheets, with no integration into Cardioref</li>
|
||
</ul>
|
||
<div class="roi"><b class="k">Problem</b>Genotypes on one side, clinical records on the other — no way to join them.</div>`,
|
||
},
|
||
omics: {
|
||
kicker: 'The starting point · 20 years ago',
|
||
title: 'Aritmolab Portal — born too early',
|
||
body: `
|
||
<p class="lead">A 360° patient-management system, still in its infancy.</p>
|
||
<ul class="ai-pop-list">
|
||
<li><b>Used for</b> — the first patient-management features beyond the hospital electronic health record</li>
|
||
<li><b>The catch</b> — embryonic functionality, and no integration with Cardioref</li>
|
||
</ul>
|
||
<div class="roi"><b class="k">Problem</b>A promising idea without a bridge: it could not see the clinical history stored elsewhere.</div>`,
|
||
},
|
||
ecg: {
|
||
kicker: 'The starting point · 20 years ago',
|
||
title: 'ECG subsystems — one per device',
|
||
body: `
|
||
<p class="lead">A constellation of ECG systems, each with its own software and its own silo.</p>
|
||
<ul class="ai-pop-list">
|
||
<li><b>Used for</b> — acquiring and storing ECG signals</li>
|
||
<li><b>The catch</b> — no machine-to-machine communication: the only export is a CSV file, on request</li>
|
||
</ul>
|
||
<div class="roi"><b class="k">Problem</b>Every ECG meant a manual download — automation was impossible by design.</div>`,
|
||
},
|
||
};
|
||
const openPop = (key, brainEl, dict) => {
|
||
const c = (dict || AI_CONTRIBS)[key];
|
||
if (!c) return;
|
||
SlideLens.close();
|
||
ScreenshotTour.close();
|
||
const isAI = (dict || AI_CONTRIBS) === AI_CONTRIBS;
|
||
pop.classList.toggle('is-ai', isAI);
|
||
document.querySelectorAll('.brain-btn[aria-pressed="true"]').forEach(button => button.setAttribute('aria-pressed', 'false'));
|
||
if (isAI && brainEl) brainEl.setAttribute('aria-pressed', 'true');
|
||
document.getElementById('aiPopBrain').style.display = isAI ? 'flex' : 'none';
|
||
document.getElementById('aiPopKicker').textContent = c.kicker || 'AI contribution';
|
||
document.getElementById('aiPopTitle').textContent = c.title;
|
||
document.getElementById('aiPopBody').innerHTML = c.body;
|
||
pop.classList.add('open');
|
||
if (brainEl) placeNear(brainEl);
|
||
activePopup = { kind: isAI ? 'ai' : dict === START_CONTRIBS ? 'start' : 'missing', key };
|
||
window.dispatchEvent(new Event('presentationchange'));
|
||
};
|
||
document.addEventListener('click', (e) => {
|
||
const screenshot = e.target.closest('[data-screenshot]');
|
||
if (screenshot) {
|
||
closePop();
|
||
ScreenshotTour.open(screenshot);
|
||
return;
|
||
}
|
||
const lens = e.target.closest('[data-lens]');
|
||
if (lens) {
|
||
const wasOpen = SlideLens.active?.key === lens.dataset.lens;
|
||
closePop();
|
||
if (!wasOpen) SlideLens.open(lens);
|
||
return;
|
||
}
|
||
if (e.target.closest('.slide-lens-close')) { closePop(); return; }
|
||
const b = e.target.closest('.brain-btn[data-ai]');
|
||
if (b) { openPop(b.dataset.ai, b, AI_CONTRIBS); return; }
|
||
const s = e.target.closest('.isle[data-start]');
|
||
if (s) { openPop(s.dataset.start, s, START_CONTRIBS); return; }
|
||
const m = e.target.closest('.todo[data-missing]');
|
||
if (m) { openPop(m.dataset.missing, m, MISSING_CONTRIBS); return; }
|
||
if (e.target.closest('.ai-pop-x') || e.target === pop) closePop();
|
||
if (SlideLens.active && !e.target.closest('.slide-lens')) closePop();
|
||
});
|
||
// Escape closes the popup first (capture), without toggling Reveal's overview
|
||
document.addEventListener('keydown', (e) => {
|
||
if (e.key === 'Escape' && (pop.classList.contains('open') || SlideLens.active || ScreenshotTour.active)) {
|
||
closePop(); e.preventDefault(); e.stopPropagation();
|
||
}
|
||
const block = e.target.closest('button[data-lens], button[data-screenshot]');
|
||
if (block && (e.key === 'Enter' || e.key === ' ')) {
|
||
e.preventDefault(); e.stopPropagation();
|
||
if (!e.repeat) block.click();
|
||
return;
|
||
}
|
||
if (e.repeat || e.altKey || e.ctrlKey || e.metaKey || e.target.closest('input, select, textarea, [contenteditable="true"]')) return;
|
||
const lens = [...(Reveal.getCurrentSlide()?.querySelectorAll('[data-lens-number], [data-screenshot-number]') || [])]
|
||
.find(button => (button.dataset.lensNumber || button.dataset.screenshotNumber) === e.key);
|
||
if (lens) { lens.click(); e.preventDefault(); e.stopPropagation(); }
|
||
}, true);
|
||
|
||
// Small shared interface for the presenter console and its synchronized audience view.
|
||
const popupSelector = '[data-lens], [data-screenshot], .brain-btn[data-ai], .isle[data-start], .todo[data-missing]';
|
||
const popupIdentity = el => {
|
||
const kind = ['lens', 'screenshot', 'ai', 'start', 'missing'].find(k => el.dataset[k]);
|
||
return { kind, key: el.dataset[kind] };
|
||
};
|
||
const popupDictionary = kind => ({ ai: AI_CONTRIBS, start: START_CONTRIBS, missing: MISSING_CONTRIBS })[kind];
|
||
window.PresentationControls = {
|
||
snapshot: () => ({ index: Reveal.getIndices().h, popup: ScreenshotTour.active || SlideLens.active || activePopup }),
|
||
slides: () => Reveal.getSlides().map(slide => ({
|
||
title: slide.querySelector('h1, h2')?.textContent.trim() || '',
|
||
notes: slide.querySelector('aside.notes')?.innerHTML || '',
|
||
popups: [...slide.querySelectorAll(popupSelector)].map(el => {
|
||
const id = popupIdentity(el);
|
||
if (id.kind === 'screenshot') return { ...id, title: el.dataset.screenshotTitle };
|
||
const source = id.kind === 'lens' ? document.getElementById(id.key) : null;
|
||
return { ...id, title: source ? source.dataset.lensTitle || source.querySelector('b').textContent : popupDictionary(id.kind)[id.key].title };
|
||
}),
|
||
})),
|
||
openPopup: id => {
|
||
const el = [...Reveal.getCurrentSlide().querySelectorAll(popupSelector)]
|
||
.find(el => { const candidate = popupIdentity(el); return candidate.kind === id.kind && candidate.key === id.key; });
|
||
if (el && id.kind === 'screenshot') {
|
||
if (ScreenshotTour.active?.key !== id.key) { closePop(); ScreenshotTour.open(el); }
|
||
} else if (el && id.kind === 'lens') {
|
||
if (SlideLens.active?.key !== id.key) { closePop(); SlideLens.open(el); }
|
||
else SlideLens.layout();
|
||
} else if (el) openPop(id.key, el, popupDictionary(id.kind));
|
||
},
|
||
closePopup: closePop,
|
||
};
|
||
addEventListener('resize', () => {
|
||
if (activePopup) window.PresentationControls.openPopup(activePopup);
|
||
});
|
||
</script>
|
||
<script src="presenter-bridge.js"></script>
|
||
</body>
|
||
</html>
|