Refine presentation flow and speaker notes
This commit is contained in:
+113
-240
@@ -8,7 +8,7 @@
|
||||
<link rel="stylesheet" href="deck.css?v=41">
|
||||
<link rel="stylesheet" href="slide-lens.css?v=7">
|
||||
<link rel="stylesheet" href="ai-callouts.css?v=3">
|
||||
<link rel="stylesheet" href="screenshot-tour.css?v=5">
|
||||
<link rel="stylesheet" href="screenshot-tour.css?v=7">
|
||||
</head>
|
||||
<body>
|
||||
<div class="reveal"><div class="slides">
|
||||
@@ -37,7 +37,7 @@
|
||||
<p class="event">“Multidimensional Characterization of Cardiac Arrhythmias: Role of Electrocardiology in the Artificial Intelligence Era”</p>
|
||||
<p class="event-where">San Donato Milanese, Milan, Italy · 2–3 October 2026</p>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">01 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">01 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
Good morning. My name is Marco Pancotti, and I led the part of the PAMP-FA project dedicated to building the AritmoLab portal at Policlinico San Donato, with support from Sara Paratico, who will co-present with me today.
|
||||
The scope of the project included a data warehouse and tools for machine learning and predictive statistics to support research by the Arrhythmology Unit, directed by Professor Pappone and Professor Locati.
|
||||
@@ -46,55 +46,7 @@
|
||||
</section>
|
||||
|
||||
|
||||
<section class="arit">
|
||||
<header class="head">
|
||||
<img src="logo.png" alt="">
|
||||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||||
</header>
|
||||
<nav class="band">
|
||||
<span class="band-section">Where we started</span>
|
||||
</nav>
|
||||
<div class="sbody">
|
||||
<h2>Where we started - Four islands and four missing pieces</h2>
|
||||
<div class="scatter">
|
||||
<div class="todo-stack" style="left:30%; top:22%; width:40%">
|
||||
<div class="todo" data-missing="ci" style="margin-left:0"><span class="q">?</span><span class="tx"><b>Clinical Intelligence</b><span>dashboards on the clinical history of the Unit and its patients</span></span></div>
|
||||
<div class="todo" data-missing="dwh" style="margin-left:9%"><span class="q">?</span><span class="tx"><b>Datawarehouse</b><span>the single source of truth</span></span></div>
|
||||
<div class="todo" data-missing="ml" style="margin-left:4%"><span class="q">?</span><span class="tx"><b>ML-ready data</b><span>cohort tables, ready for model training</span></span></div>
|
||||
<div class="todo" data-missing="portal" style="margin-left:12%"><span class="q">?</span><span class="tx"><b>Professional Portal</b><span>AritmoLab — the grown-up Omics Portal</span></span></div>
|
||||
</div>
|
||||
<button class="isle" data-start="cardioref" style="--rot:-3deg; --size:114px; left:calc(1% + 39px); top:4%; width:230px">
|
||||
<svg class="ico" viewBox="0 0 48 48" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><path d="M24 33 c-6-3.5-9-7-9-10.5 0-3 4-5.5 7-3 1 .8 1.6 1.8 2 3 .4-1.2 1-2.2 2-3 3-2.5 7 0 7 3 0 3.5-3 7-9 10.5z" fill="currentColor" stroke="none"/><path d="M22.5 21 h3 v3 h3 v3 h-3 v3 h-3 v-3 h-3 v-3 h3 z" fill="#fff" stroke="none"/></svg>
|
||||
<span class="nm">Cardioref</span><span class="sc" style="width:300px">visits · operations · reports — twenty years of management on structured forms</span></button>
|
||||
<button class="isle" data-start="genetic" style="--rot:3deg; --size:114px; left:calc(73% + 39px); top:4%; width:230px">
|
||||
<svg class="ico" viewBox="0 0 48 48" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><path d="M18 8 C28 14 28 20 18 26 C12 30 12 35 18 40"/><path d="M30 8 C20 14 20 20 30 26 C36 30 36 35 30 40"/><path d="M17 12 H31"/><path d="M15.5 18 H32.5"/><path d="M17 24 H31"/><path d="M14.5 31 H22"/><rect x="25" y="29" width="12" height="12" rx="1.5" stroke-width="2"/><path d="M25 34 H37"/><path d="M30.5 29 V41"/></svg>
|
||||
<span class="nm">Genetic data</span><span class="sc">DNA instruments — results typed into Excel by hand</span></button>
|
||||
<button class="isle" data-start="omics" style="--rot:-2deg; --size:114px; left:calc(1% + 39px); top:55%; width:230px">
|
||||
<svg class="ico" viewBox="0 0 48 48" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><circle cx="24" cy="24" r="20" stroke-width="2.6"/><circle cx="24" cy="24" r="4" fill="currentColor" stroke="none"/><circle cx="24" cy="10" r="3.4"/><circle cx="11.5" cy="31" r="3.4"/><circle cx="36.5" cy="31" r="3.4"/><path d="M24 20.5 V13.5"/><path d="M21 26.5 L14 29.5"/><path d="M27 26.5 L34 29.5"/></svg>
|
||||
<span class="nm">Aritmolab Portal</span><span class="sc">meant to host Cardioref + genetics — still immature</span></button>
|
||||
<button class="isle" data-start="ecg" style="--rot:2deg; --size:114px; left:calc(73% + 39px); top:55%; width:230px">
|
||||
<svg class="ico" viewBox="0 0 48 48" fill="none" stroke="currentColor" stroke-width="2.6" stroke-linecap="round" stroke-linejoin="round"><circle cx="24" cy="24" r="20" stroke-width="2.6"/><path d="M9 24 h6 l3-9 5 16 3-7 h14" stroke-width="2.4"/><circle cx="24" cy="24" r="1.8" fill="currentColor" stroke="none"/></svg>
|
||||
<span class="nm">ECG</span><span class="sc">paper strip + a CSV on request</span></button>
|
||||
<div class="missing-title" style="left:calc(30% + 14px); top:calc(22% - 46px); width:40%">Missing Pieces</div>
|
||||
<p class="hint" style="position:absolute; left:26%; width:48%; bottom:2px; margin:0; text-align:center">▸ click a system: how it was used — and what held it back · click a missing piece: what it would have given</p>
|
||||
</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">02 / 13</span></footer>
|
||||
<aside class="notes">
|
||||
The starting point was four disconnected systems:
|
||||
1 - <strong>Cardioref</strong> held twenty years of electrophysiology records and supported clinical operations, but much of the information needed for research was in free text.
|
||||
2 - <strong>Genetic data</strong> were manually entered into Excel, without integration with Cardioref.
|
||||
3 - The <strong>Aritmolab Portal</strong> was an early prototype, not yet integrated with the source systems.
|
||||
4 - <strong>ECG</strong> data were available as paper records or CSV exports on request.
|
||||
|
||||
Whe lacked:
|
||||
5 - some longitudinal <strong>clinical analytics</strong>
|
||||
6 - a shared <strong>data warehouse</strong>,
|
||||
7 - a set of structured datasets for model training,
|
||||
8 - and a unified <strong>clinical portal</strong>.
|
||||
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
|
||||
<section class="arit flow-slide">
|
||||
@@ -119,11 +71,11 @@
|
||||
<span class="fbox src future"><b><span class="plus">+</span>Future sources</b><span>new subsystems can be connected as sources</span></span>
|
||||
</span>
|
||||
<template class="lens-details">
|
||||
<p class="detail-intro">The starting material offered complementary views of the patient.</p>
|
||||
<p><b>Clinical course · Cardioref</b>Visits, procedures and reports describe the course of care. Dates and narrative details give each event its clinical context.</p>
|
||||
<p><b>Genetic findings</b>Laboratory results and variant descriptions add the genetic perspective, recorded separately from the clinical history.</p>
|
||||
<p><b>Electrical activity · ECG</b>Tracings and exported signals document cardiac electrical activity, complementing the written account of the patient’s condition.</p>
|
||||
<p><b>Different forms of evidence</b>Structured fields, free text and signals must retain their clinical meaning when connected. Future sources would extend this initial set.</p>
|
||||
<p class="detail-intro">Four disconnected systems held complementary views of the patient.</p>
|
||||
<p><b>Clinical course · Cardioref</b>Twenty years of visits, procedures and reports supported daily clinical work. Much of the information needed for research remained in free text.</p>
|
||||
<p><b>Genetic findings</b>Laboratory results and variants were entered manually into Excel, without integration with Cardioref.</p>
|
||||
<p><b>Electrical activity · ECG</b>Separate device systems provided paper records or CSV exports on request, requiring manual collection.</p>
|
||||
<p><b>The early portal</b>The AritmoLab prototype, originally the Omics Portal, lacked integration with the sources. Connecting clinical history, genetics and ECG was the prerequisite for a shared patient view.</p>
|
||||
</template>
|
||||
</button>
|
||||
<div class="fcol ai-hit">
|
||||
@@ -157,9 +109,9 @@
|
||||
<div class="farrow">→</div>
|
||||
<div class="fcol" style="flex:1.05; justify-content:center"><button type="button" class="fbox lens-target" id="plan-star-schema" data-lens="plan-star-schema" data-lens-number="4" aria-label="Explore Data warehouse" aria-expanded="false" aria-controls="slide-lens"><b>Data warehouse</b><span>star schema · organized for analysis</span>
|
||||
<template class="lens-details">
|
||||
<p class="detail-intro">The shared analytical database: integrated clinical data reorganized for research across patients and over time.</p>
|
||||
<p class="detail-intro">A shared analytical database replaces separate exports and fragile spreadsheets with one source of truth.</p>
|
||||
<p><b>How a star schema works</b>“Facts” represent events or measurements, such as a procedure or test. “Dimensions” describe their context, such as the patient, date and procedure type.</p>
|
||||
<p><b>Why it matters</b>Researchers can filter, group and compare records using common definitions, without reconstructing every relationship from the original hospital tables.</p>
|
||||
<p><b>Clinical Intelligence</b>Common definitions make the Unit’s activity and patients’ histories comparable over time. Dashboards can show volumes, incidence and outcomes for management, audit and research.</p>
|
||||
<p class="detail-example"><b>Clinical question it can support</b>How many patients underwent a given procedure each year, and how does that distribution vary by age group?</p>
|
||||
</template>
|
||||
</button></div>
|
||||
@@ -168,10 +120,11 @@
|
||||
<button class="brain brain-btn" data-ai="datamarts" aria-label="AI contribution: datamart generation" aria-pressed="false"><span class="ai-label" aria-hidden="true">AI</span><img class="artificial-brain" src="artificial-brain.svg" alt=""><span class="ai-number">8</span></button>
|
||||
<button type="button" class="fbox lens-target" id="plan-datamarts" data-lens="plan-datamarts" data-lens-number="5" aria-label="Explore Datamarts" aria-expanded="false" aria-controls="slide-lens"><b>Datamarts</b><span>research-ready marts</span>
|
||||
<template class="lens-details">
|
||||
<p class="detail-intro">Focused datasets derived from the warehouse for a specific research question, study or dashboard.</p>
|
||||
<p><b>What they define</b>The cohort, time window, variables and level of detail: for example, one row per patient or one row per procedure.</p>
|
||||
<p><b>How ThothII helps</b>A plain-English question becomes a proposed SQL query through a guided workflow with human review. The resulting dataset supports analysis and portal dashboards.</p>
|
||||
<p class="detail-example"><b>Illustrative study dataset</b>Patients who underwent a drug-challenge test, with test date, result and selected clinical characteristics. Its cohort and variable definitions must be agreed before interpreting results.</p>
|
||||
<p class="detail-intro">Reproducible cohorts provide the structured data previously missing for statistics and model training.</p>
|
||||
<p><b>What they define</b>The cohort, time window, variables and level of detail, such as one row per patient with features and outcomes.</p>
|
||||
<p><b>How ThothII helps</b>A plain-English question becomes proposed SQL through a guided workflow with human review, reducing manual dataset preparation.</p>
|
||||
<p><b>The professional portal</b>AritmoLab brings integrated patient information and research dashboards into one place, extending the early Omics Portal.</p>
|
||||
<p class="detail-example"><b>Illustrative study dataset</b>Drug-challenge patients, test dates, results and clinical characteristics, with agreed cohort and variable definitions.</p>
|
||||
</template>
|
||||
</button>
|
||||
<div class="fcap">AI builds them on demand from plain-English questions (ThothII)</div>
|
||||
@@ -192,23 +145,19 @@
|
||||
<span class="pillar"><b>Authentik</b> · auth — GSD LDAP</span>
|
||||
</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">03 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">02 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
Here is the whole project on one slide.
|
||||
|
||||
On the left, what already existed, three hospital data sources:
|
||||
(1) <strong>What existed</strong>: <strong>Cardioref</strong>, our electrophysiology records dataset, the <strong>genetic data</strong> coming from the labs, the <strong>ECG signals</strong>, and, in the future, other sources.
|
||||
|
||||
On the right, what we built in AritmoLab:
|
||||
(2) a staging copy of all relevant Cardioref data, (3) an integration layer where the data is cleaned and normalized, (4) a star-schema warehouse, where the data are reorganized as facts and dimensions, and (5) research datamarts, generated from the data warehouse and presented through dashboards embedded in the portal.
|
||||
|
||||
Everywhere you see the brain symbol, that's where AI works for us.
|
||||
6 - The <strong>mappings</strong> that bring the sources in were themselves drafted by AI, then revised by humans
|
||||
7 - At <strong>integration</strong>, AI reads the clinical text.
|
||||
8 - At the end, AI builds the <strong>datamarts</strong> on demand from plain-English questions.
|
||||
That was the plan. And to build it, we used AI everywhere it helped — drafting configs, reading clinical text, and at the very end, producing the datamarts themselves.
|
||||
|
||||
Next, we’ll look at the hardest part: unstructured data. My colleague, Sara Paratico,do will show you exactly how the text reading works. But first, the first step of the climb: the mappings.
|
||||
<p>We had four goals: clinical dashboards to understand the Unit’s activity and patients’ histories, a shared data warehouse, structured cohorts for machine learning and predictive statistics, and a unified professional portal.</p>
|
||||
<p>(1) We started with disconnected systems: twenty years of Cardioref records, much of their research value in free text, genetic results entered manually into Excel, ECGs on paper or exported on request, and an early portal without source integration. Each source answered a different need, but combining them for a clinical study meant collecting exports and rebuilding patient histories by hand.</p>
|
||||
<p>(2) To connect them, staging preserves a working copy of the sources.</p>
|
||||
<p>(3) Integration then cleans, normalizes and links the records.</p>
|
||||
<p>(4) The data warehouse organizes events and their context for analysis.</p>
|
||||
<p>(5) From there, datamarts turn that shared information into research datasets and dashboards within AritmoLab.</p>
|
||||
<p>(6) The brain symbols show where AI contributes. It drafts the mappings, which humans review.</p>
|
||||
<p>(7) During integration, AI extracts structured information from clinical text.</p>
|
||||
<p>(8) ThothII then builds datamarts from plain-English questions, with human approval.</p>
|
||||
<p>Open-source tools support the platform, and AI coding agents helped throughout development. The popups connect each component to the gap it addresses.</p>
|
||||
<p>Next, we turn to the hardest part: unstructured data. Sara will explain how we extract clinical meaning and check its reliability.</p>
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
@@ -268,10 +217,12 @@
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">04 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">03 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
Here is one real sentence from a discharge letter — in Italian, as the clinicians wrote it. In this single sentence there is a diagnosis (syncope), an ECG finding (ST elevation in V1–V3), and a drug-challenge result (flecainide positive for Brugada pattern). For a human cardiologist this is readable in two seconds. For a database, this is just a blob of text in a column.
|
||||
The structured tables — demographics, procedures, dates — only tell half the story. The rest is locked inside these free-text fields, in every hospital system we have.
|
||||
<p>Over twenty years, we have collected clinical information in discharge letters and procedure reports. Much remains in free text.</p>
|
||||
<p>Clinicians connect these details and form hypotheses. Here, one sentence combines syncope, ECG changes and a positive flecainide test. Research needs defined variables that preserve this meaning.</p>
|
||||
<p>Our text miner extracts them in the integration layer, before the data enter the warehouse.</p>
|
||||
<p>We face four challenges: capturing diverse clinical histories, interpreting context correctly, mapping different expressions to common terms, and validating accuracy. Reliable research and patient care depend on getting these details right.</p>
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
@@ -314,11 +265,11 @@
|
||||
<div class="nitem"><b>Audited.</b> A new rule set goes live only after it proves ≥95% accuracy on 100 records reviewed by hand.</div>
|
||||
</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">05 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">04 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
[DRAFT — Sara] I'll take you inside these numbers. We read 58,438 discharge letters — every single one, every night. From them the text miner extracted 73,389 pathology records, classified into our clinical ontology. It parsed 10,908 drug-challenge tests, distinguishing the therapy from the actual test. And at the end of the chain: 2,307 patients with a confirmed Brugada pattern — a cohort nobody could have built by hand.
|
||||
This is not a one-off migration: the pipeline runs every night, and today it processes on average two hundred and forty letters a month. Four simple pieces do the work: a letter reader for the Italian and English text, a clinical matcher that recognises diagnoses, procedures and test outcomes, a context guard that keeps negations and family history apart, and the ontology sorter that files every finding into its clinical category.
|
||||
The key word here is deterministic: no black box. Every number can be traced back to the rule that produced it.
|
||||
<p>Our text miner uses explicit clinical rules to identify diagnoses and procedures, including catheter ablation and drug challenge tests, while checking their context.</p>
|
||||
<p>These figures cover over 58,000 letters and around 73,000 records of clinical conditions. We identified almost 11,000 drug challenge test entries, including positive Brugada tests in 2,307 patients.</p>
|
||||
<p>Each extracted record retains the rule version used. This lets us compare classifications with clinical review and investigate errors. For procedure classification, our target is at least 95 percent accuracy against 100 manually reviewed records.</p>
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
@@ -367,10 +318,11 @@
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">06 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">05 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
[DRAFT — Sara] Why can't we just search for "FA"? Because clinical text lies to naive search. "Fibrillazione atriale esclusa" contains the words of a diagnosis but negates it — so every rule scans a window around the match, looking for negation cues. "Padre con FA" is real atrial fibrillation — but in the father, not the patient: we record it as family history. Letters mix Italian and English, abbreviations collide ("TA" is blood pressure, not a therapy), and one field can contain five different statements — so we split the text into clauses first.
|
||||
Each of these problems has a specific, versioned solution. Sara to expand with real corpus examples.
|
||||
<p>A report may describe atrial fibrillation in the patient, rule it out, or mention it in the father's history. The same term must lead to different classifications.</p>
|
||||
<p>Our analyser checks for negation and family references, recording these attributes separately. It recognises Italian and English terms and common abbreviations. For example, "TA" may mean atrial tachycardia, but followed by a blood pressure value, it should not trigger that diagnosis.</p>
|
||||
<p>When reports describe several events, the system separates the text into clauses. For drug challenge tests, this helps distinguish a positive result before ablation from a negative result afterwards, preserving each observation and its clinical context.</p>
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
@@ -410,12 +362,12 @@
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">07 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">06 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
[DRAFT — Sara] Extraction needs a target vocabulary — that's the ontology. Tier 1 holds the eleven arrhythmological categories that matter most for our research; Tier 2 holds five structural conditions. The order of the rules is itself clinical knowledge: Brugada patterns are tested before TV, because "substrato per TV" often appears in Brugada reports and would otherwise mask the diagnosis.
|
||||
The ontology is versioned like software: a semantic version for the pattern library, stamped on every extracted row. When we add a synonym or fix a rule, the change is traceable — and the data can be rebuilt.
|
||||
And this is the output of the whole work: reading a letter writes structured, quantitative values back onto the records that describe the procedures and implants it contains — and those records enter the warehouse generation exactly as if the clinicians had typed them during the visits. Text becomes data, indistinguishable from bedside data entry.
|
||||
Sara: review the category list and the Italian labels before final. Labels are now English-first (audience is mostly foreign) — please confirm the EN terms, especially PSVT as the umbrella for WPW/TPSV and AT/AVNRT for TA/TRN.
|
||||
<p>Our clinical ontology, a shared set of terms, helps identify arrhythmia-related diagnoses in the narrative.</p>
|
||||
<p>The first tier covers conditions such as atrial fibrillation and Brugada. If none are found within a text field, the system checks a second tier for cardiovascular comorbidities, such as cardiomyopathy or heart failure, that may influence the patient's arrhythmic presentation.</p>
|
||||
<p>Different expressions map to one concept: for example, "fibrillazione atriale" and "atrial fibrillation" receive the same label.</p>
|
||||
<p>The extracted records enter the warehouse alongside data clinicians entered in structured fields. We can then define patient groups and research datasets, retaining each finding's source and the version of the extraction rules.</p>
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
@@ -486,13 +438,13 @@
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">08 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">07 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
<p><button type="button" data-note-lens="validation-quality">1. Quality gates.</button> We set a 95% procedure-accuracy target, against 100 manual reviews. The 85% pathology target measures coverage, not diagnostic accuracy.</p>
|
||||
<p><button type="button" data-note-lens="validation-review">2. Clinical criteria.</button> Finding a disease name in a letter is only the first step. We apply explicit clinical criteria to decide which patients belong in the analysis.</p>
|
||||
<p><button type="button" data-note-lens="validation-feedback">3. Feedback loop.</button> Version 1.3.1 corrected missed negations in roughly 900 Brugada records, with regression tests protecting the fix.</p>
|
||||
<p><button type="button" data-note-lens="validation-context">4. Nothing is silently dropped.</button> Negated and family findings retain their context. Research datasets filter them explicitly, so preserving a finding does not mean counting it as the patient’s diagnosis.</p>
|
||||
<p><button type="button" data-note-lens="validation-advantages">5. The advantages.</button> We can reprocess the archive quickly, inspect the rules behind each result, and reproduce the same extraction with the same rules.</p>
|
||||
<p>(1) To check the extraction, we set a 95% procedure-accuracy target against 100 manual reviews. The 85% pathology target measures coverage, not diagnostic accuracy.</p>
|
||||
<p>(2) Finding a disease name in a letter is only the first step. We also apply explicit clinical criteria to decide which patients belong in the analysis.</p>
|
||||
<p>(3) Review feeds back into the rules. Version 1.3.1 corrected missed negations in roughly 900 Brugada records, with regression tests protecting the fix.</p>
|
||||
<p>(4) Throughout this process, negated and family findings retain their context. Research datasets filter them explicitly, so preserving a finding does not mean counting it as the patient’s diagnosis.</p>
|
||||
<p>(5) This lets us reprocess the archive quickly, inspect the rules behind each result, and reproduce the same extraction with the same rules.</p>
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
@@ -508,26 +460,29 @@
|
||||
<div class="sbody">
|
||||
<div class="kicker">The portal, today</div>
|
||||
<h2>AritmoLab — a quick tour</h2>
|
||||
<div class="screenshot-tour" data-screenshot-auto-open aria-label="AritmoLab tour, screenshots 1 to 7">
|
||||
<div class="screenshot-tour screenshot-tour-four" data-screenshot-auto-open aria-label="AritmoLab tour, screenshots 1 to 4">
|
||||
<button type="button" data-screenshot="home" data-screenshot-number="1" data-screenshot-title="Home page" aria-haspopup="dialog"><img src="screenshots/1-HomePage.png" alt="" loading="lazy"><span><b>1</b> Home page</span></button>
|
||||
<button type="button" data-screenshot="patients" data-screenshot-number="2" data-screenshot-title="Patient list" aria-haspopup="dialog"><img src="screenshots/2-PatientList.png" alt="" loading="lazy"><span><b>2</b> Patient list</span></button>
|
||||
<button type="button" data-screenshot="profile" data-screenshot-number="3" data-screenshot-title="Patient profile" aria-haspopup="dialog"><img src="screenshots/3-PatientGeneralData.png" alt="" loading="lazy"><span><b>3</b> Patient profile</span></button>
|
||||
<button type="button" data-screenshot="history" data-screenshot-number="4" data-screenshot-title="Clinical history" aria-haspopup="dialog"><img src="screenshots/4-PatientDetail.png" alt="" loading="lazy"><span><b>4</b> Clinical history</span></button>
|
||||
<button type="button" data-screenshot="procedure" data-screenshot-number="5" data-screenshot-title="Procedure details" aria-haspopup="dialog"><img src="screenshots/5-PatientProcedureDetails.png" alt="" loading="lazy"><span><b>5</b> Procedure details</span></button>
|
||||
<button type="button" data-screenshot="dashboards" data-screenshot-number="6" data-screenshot-title="Dashboard catalogue" aria-haspopup="dialog"><img src="screenshots/6-DashboardsList.png" alt="" loading="lazy"><span><b>6</b> Dashboard catalogue</span></button>
|
||||
<button type="button" data-screenshot="brugada" data-screenshot-number="7" data-screenshot-title="Brugada dashboard" aria-haspopup="dialog"><img src="screenshots/7-Brugada-dashboard.png" alt="" loading="lazy"><span><b>7</b> Brugada dashboard</span></button>
|
||||
<button type="button" data-screenshot="profile" data-screenshot-number="2" data-screenshot-title="Patient data" aria-haspopup="dialog"><img src="screenshots/3-PatientGeneralData.png" alt="" loading="lazy"><span><b>2</b> Patient data</span></button>
|
||||
<button type="button" data-screenshot="dashboards" data-screenshot-number="3" data-screenshot-title="Dashboard catalogue" aria-haspopup="dialog"><img src="screenshots/6-DashboardsList.png" alt="" loading="lazy"><span><b>3</b> Dashboard catalogue</span></button>
|
||||
<button type="button" data-screenshot="brugada" data-screenshot-number="4" data-screenshot-title="Brugada: one of many dashboards" aria-haspopup="dialog"><img src="screenshots/7-Brugada-dashboard.png" alt="" loading="lazy"><span><b>4</b> Brugada: one of many dashboards</span></button>
|
||||
</div>
|
||||
<p class="hint">Select a screen to enlarge · Follow the tour from 1 to 7</p>
|
||||
<p class="hint">Select a screen to enlarge · Follow the tour from 1 to 4</p>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">09 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">08 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
<p><button type="button" data-note-screenshot="home">1. Home page.</button> The home page summarises around 57,000 patients, by age, sex and geographical origin. It is the starting point for exploring the archive.</p>
|
||||
<p><button type="button" data-note-screenshot="patients">2. Patient list.</button> Search filters help us find a patient or study participant, open their record, or export the results.</p>
|
||||
<p><button type="button" data-note-screenshot="profile">3. Patient profile.</button> The profile brings demographic and clinical fields together, with access to procedures, devices, diagnostic examinations and genetics.</p>
|
||||
<p><button type="button" data-note-screenshot="history">4. Clinical history.</button> A dated timeline brings together clinical notes, discharge letters and procedures, retaining the source of each event.</p>
|
||||
<p><button type="button" data-note-screenshot="procedure">5. Procedure details.</button> Here, an ablation record shows the treated arrhythmias, procedural details, recorded complications and conclusions.</p>
|
||||
<p><button type="button" data-note-screenshot="dashboards">6. Dashboard catalogue.</button> We then move from individual records to dashboards covering departmental activity, procedures, devices and genetics.</p>
|
||||
<p><button type="button" data-note-screenshot="brugada">7. Brugada dashboard.</button> The funnel separates text mentions, filtered mentions, confirmed cases and ablation outcomes. Other charts describe sex, age, annual diagnoses and the timing of pre- and post-assessments.</p>
|
||||
<div data-popup-notes="home"><div data-popup-notes-body>
|
||||
<p>(1) AritmoLab is an extensive portal with many interconnected features. Our limited time prevents us from presenting it in detail, so we will show just four screens. Sara Paratico and I are available for a more in-depth presentation on request, either during the conference or afterwards through a remote connection.</p>
|
||||
<p>The home page summarises around 57,000 patients by age, sex and geographical origin. It is the starting point for exploring the archive.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="profile"><div data-popup-notes-body>
|
||||
<p>(2) From this overview, we can open a patient profile, which brings demographic and clinical information together, with access to procedures, devices, diagnostic examinations and genetics. It provides a single starting point for exploring an individual patient's record.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="dashboards"><div data-popup-notes-body>
|
||||
<p>(3) Moving from individual patient records to an overview of the department, the dashboard catalogue gives us access to clinical activity, procedures, devices and genetics.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="brugada"><div data-popup-notes-body>
|
||||
<p>(4) Among the many dashboards we could present, we chose Brugada as an example. The funnel separates text mentions, filtered mentions, confirmed cases and ablation outcomes. Other charts show sex, age, annual diagnoses and the timing of pre- and post-assessments.</p>
|
||||
</div></div>
|
||||
</aside>
|
||||
</section>
|
||||
<!-- ============ 10 · CRISIS ============ -->
|
||||
@@ -579,14 +534,14 @@
|
||||
</div>
|
||||
<div class="roi"><b class="k">The gap</b>Twenty years of data, one warehouse — and no fast road from a research question to an answer.</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">10 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">09 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
<p><strong>0. What is SQL?</strong><br>SQL means Structured Query Language. It tells a database what to select, connect and count.</p>
|
||||
<p><button type="button" data-note-lens="crisis-language">1. Clinical questions.</button><br>“How many patients with confirmed Brugada underwent an ablation?” We must define confirmation and count each patient once.</p>
|
||||
<p><button type="button" data-note-lens="crisis-intelligence">2. Health Intelligence.</button><br>Dashboards need agreed definitions, time periods and denominators to make comparisons meaningful.</p>
|
||||
<p><button type="button" data-note-lens="crisis-prediction">3. Predictive research.</button><br>Study tables separate characteristics known before prediction from outcomes observed afterwards.</p>
|
||||
<p><button type="button" data-note-lens="crisis-datamarts">4. Datamarts.</button><br>A datamart is an analysis dataset with explicit selection rules and repeatable checks.</p>
|
||||
<p><strong>5. The gap.</strong><br>The gap is between clinical meaning and database instructions: storing data does not automatically make a question answerable.</p>
|
||||
<p>The warehouse speaks SQL, which means Structured Query Language. It tells a database what to select, connect and count.</p>
|
||||
<p>(1) To answer a clinical question such as “How many patients with confirmed Brugada underwent an ablation?”, we must define confirmation and count each patient once.</p>
|
||||
<p>(2) The same need for clarity applies to Health Intelligence: dashboards need agreed definitions, time periods and denominators to make comparisons meaningful.</p>
|
||||
<p>(3) For predictive research, study tables separate characteristics known before prediction from outcomes observed afterwards.</p>
|
||||
<p>(4) A datamart provides an analysis dataset with explicit selection rules and repeatable checks.</p>
|
||||
<p>The gap is therefore between clinical meaning and database instructions: storing data does not automatically make a question answerable.</p>
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
@@ -600,136 +555,53 @@
|
||||
<span class="band-section">A tour of ThothII</span>
|
||||
</nav>
|
||||
<div class="sbody">
|
||||
<div class="kicker">ThothII, step by step</div>
|
||||
<h2>From question to datamart — the app</h2>
|
||||
<div class="screenshot-tour screenshot-tour-compact" data-screenshot-auto-open aria-label="ThothII tour, screenshots 1 to 12">
|
||||
<button type="button" data-screenshot="thothii-start" data-screenshot-number="1" data-screenshot-title="Starting point" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/01-StartingPoint.png?v=4" alt="" loading="lazy"><span><b>1</b> Starting point</span>
|
||||
<div class="kicker">Human in the Loop · Eight phases, six screens</div>
|
||||
<h2>From clinical question to datamart, with human review</h2>
|
||||
<div class="screenshot-tour screenshot-tour-compact" data-screenshot-auto-open aria-label="ThothII tour, screenshots 1 to 6">
|
||||
<button type="button" data-screenshot="thothii-start" data-screenshot-number="1" data-screenshot-title="Why ThothII exists" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/01-StartingPoint.png?v=4" alt="" loading="lazy"><span><b>1</b> Why ThothII exists</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-disambiguation-1" data-screenshot-number="2" data-screenshot-title="Disambiguation · 1" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/02-Disambiguation01.png?v=4" alt="" loading="lazy"><span><b>2</b> Disambiguation · 1</span>
|
||||
<button type="button" data-screenshot="thothii-disambiguation-1" data-screenshot-number="2" data-screenshot-title="Agreeing on the question" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/02-Disambiguation01.png?v=4" alt="" loading="lazy"><span><b>2</b> Agreeing on the question</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-disambiguation-final" data-screenshot-number="3" data-screenshot-title="Question clarified" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/03-DisambiguationFinal.png?v=4" alt="" loading="lazy"><span><b>3</b> Question clarified</span>
|
||||
<button type="button" data-screenshot="thothii-schema-final" data-screenshot-number="3" data-screenshot-title="Connecting to the data" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/05-CloseSchemaLinking.png?v=4" alt="" loading="lazy"><span><b>3</b> Connecting to the data</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-schema-1" data-screenshot-number="4" data-screenshot-title="Schema linking · 1" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/04-SchemaLinking01.png?v=4" alt="" loading="lazy"><span><b>4</b> Schema linking · 1</span>
|
||||
<button type="button" data-screenshot="thothii-cte-1" data-screenshot-number="4" data-screenshot-title="Checking each step" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/07-CTE01.png?v=4" alt="" loading="lazy"><span><b>4</b> Checking each step</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-schema-final" data-screenshot-number="5" data-screenshot-title="Schema confirmed" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/05-CloseSchemaLinking.png?v=4" alt="" loading="lazy"><span><b>5</b> Schema confirmed</span>
|
||||
<button type="button" data-screenshot="thothii-sql" data-screenshot-number="5" data-screenshot-title="Approving the final query" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/08-FinalSQL.png?v=5" alt="" loading="lazy"><span><b>5</b> Approving the final query</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-cte-plan" data-screenshot-number="6" data-screenshot-title="CTE planning" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/06-CTEPlanning.png?v=4" alt="" loading="lazy"><span><b>6</b> CTE planning</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-cte-1" data-screenshot-number="7" data-screenshot-title="CTE · 1" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/07-CTE01.png?v=4" alt="" loading="lazy"><span><b>7</b> CTE · 1</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-sql" data-screenshot-number="8" data-screenshot-title="Final SQL" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/08-FinalSQL.png?v=5" alt="" loading="lazy"><span><b>8</b> Final SQL</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-datamart" data-screenshot-number="9" data-screenshot-title="Datamart" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/09-DatamartProduction.png?v=5" alt="" loading="lazy"><span><b>9</b> Datamart</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-memory" data-screenshot-number="10" data-screenshot-title="Saving memory" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/10-MemorySaving.png?v=5" alt="" loading="lazy"><span><b>10</b> Saving memory</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-finish" data-screenshot-number="11" data-screenshot-title="Final step" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/11-FinalStep.png?v=5" alt="" loading="lazy"><span><b>11</b> Final step</span>
|
||||
</button>
|
||||
<button type="button" data-screenshot="thothii-return" data-screenshot-number="12" data-screenshot-title="Back to start" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/12-BackToStartingPoint.png?v=5" alt="" loading="lazy"><span><b>12</b> Back to start</span>
|
||||
<button type="button" data-screenshot="thothii-datamart" data-screenshot-number="6" data-screenshot-title="Reusing the result" aria-haspopup="dialog">
|
||||
<img src="screenshots/thothii/09-DatamartProduction.png?v=5" alt="" loading="lazy"><span><b>6</b> Reusing the result</span>
|
||||
</button>
|
||||
</div>
|
||||
<p class="hint">Select a screen to enlarge · Follow the tour from 1 to 12</p>
|
||||
<p class="hint"><strong>AI proposes. People review, correct and approve.</strong> · Select a screen to enlarge</p>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">11 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">10 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
<div data-popup-notes="thothii-start"><h3>From a research question to SQL</h3><div data-popup-notes-body><p>The task has changed. We are no longer extracting structured, coded values from the text within clinical records. We are now starting from a natural-language request to retrieve the records that contain the relevant values.</p><p>Research questions are often difficult to express clearly and unambiguously, and their clinical concepts do not map directly to the underlying database structure. As a result, writing the right SQL query can take hours, with repeated testing and refinement before it accurately reflects the intended research question.</p></div></div>
|
||||
<div data-popup-notes="thothii-disambiguation-1"><h3>Clarifying the question with the researcher</h3><div data-popup-notes-body>
|
||||
<p><strong>1. Workflow phase.</strong> At the top, the phase indicator shows where we are in the process. Here, phase one is highlighted: clarifying the research question.</p>
|
||||
<p><strong>2. AI working time.</strong> The timer measures how long the AI has been working, excluding the time the human reviewer spends considering the options and making decisions.</p>
|
||||
<p><strong>3. Model activity.</strong> On the left, we can follow the AI's running explanation of its analysis: the interpretations it considers, the evidence it consults, and the rationale for its proposals.</p>
|
||||
<p><strong>4. Proposed interpretations.</strong> In the centre, the AI asks the researcher to choose between alternative interpretations. Here, the question is how to identify an ablation for atrial fibrillation in the database. This choice determines which patients enter the study.</p>
|
||||
<p><strong>5. Reviewer actions.</strong> The reviewer can select a proposed option, or use <strong>Go back</strong> to revisit the previous step, <strong>Exit</strong> to stop the current workflow, or <strong>Other — specify</strong> to describe a different interpretation in free text.</p>
|
||||
<div data-popup-notes="thothii-start"><div data-popup-notes-body>
|
||||
<p>(1) ThothII combines a wide range of capabilities in a guided, eight-phase workflow. A full presentation deserves at least thirty minutes. Here, we focus on the main steps.</p>
|
||||
<p>It helps researchers turn clinical questions into checked SQL and reusable analysis datasets, bridging clinical meaning and database structure. AI proposes, and people review, correct and approve. This is Human in the Loop throughout the workflow, combining clinical judgment and database expertise.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="thothii-disambiguation-final"><h3>From clarification to an agreed study question</h3><div data-popup-notes-body>
|
||||
<p><strong>Bringing the decisions together.</strong> The AI combines the original request with the researcher's answers into an explicit study definition. It records what we mean by each clinical concept, which patients to include, the time window, and the result we want.</p>
|
||||
<p><strong>The clarified question.</strong> In this example: how many distinct patients had their first recorded ablation for atrial fibrillation between 2020 and 2023? We identify the procedure from the treated pathology, and find each patient's first AF ablation across the entire available database history before applying the date filter.</p>
|
||||
<p><strong>Making the interpretation actionable.</strong> The summary links these choices to the relevant database fields and selection rules, and displays the checks completed during clarification. These agreed criteria guide the subsequent schema linking and SQL construction.</p>
|
||||
<p><strong>Human confirmation.</strong> The researcher reviews the consolidated definition. <strong>Save and proceed</strong> confirms it and closes phase one; <strong>Reject</strong> sends it back for revision.</p>
|
||||
<div data-popup-notes="thothii-disambiguation-1"><div data-popup-notes-body>
|
||||
<p>(2) We begin by clarifying the question: here, what counts as an atrial fibrillation ablation? We then review relevant knowledge from earlier work before approving a precise reformulation: how many patients had their first recorded AF ablation between 2020 and 2023? We find the first procedure across the available history before filtering dates. The reviewer can challenge the interpretation.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="thothii-schema-1"><h3>Finding the data needed to answer the question</h3><div data-popup-notes-body>
|
||||
<p><strong>Linking clinical meaning to data.</strong> We are now in phase four, schema linking: identifying where the information needed to answer the agreed question is stored. Tables group related records; columns hold specific details, such as a patient identifier, a procedure date, or the condition treated.</p>
|
||||
<p><strong>Proposed tables.</strong> The AI presents candidate tables and explains why each is relevant. Here, one table provides the ablation episodes and their dates; another identifies the pathology treated, allowing us to distinguish atrial fibrillation ablations from other procedures.</p>
|
||||
<p><strong>Selecting the columns.</strong> The column controls let the reviewer inspect and adjust the fields selected for the query. We need the information required to identify patients, recognise the relevant procedures, and apply the agreed time criteria.</p>
|
||||
<p><strong>Review before proceeding.</strong> The researcher and a technical reviewer can check this mapping together, adjust the selection, and confirm it. This establishes which data will support the SQL query; the relationships between the selected tables are reviewed next.</p>
|
||||
<div data-popup-notes="thothii-schema-final"><div data-popup-notes-body>
|
||||
<p>(3) With the question agreed, we connect its concepts to tables, fields and relationships. We then review the mapping and selection rules together before building SQL. The reviewer checks which patients, procedures and dates will count.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="thothii-schema-final"><h3>Schema linking complete: from clinical meaning to database structure</h3><div data-popup-notes-body>
|
||||
<p><strong>What schema linking means.</strong> Schema linking connects the concepts in the clarified research question to the database: which tables contain the information, which columns represent each concept, and how records from different tables must be connected.</p>
|
||||
<p><strong>Our clinical example.</strong> Here, we connect the treated pathology to the corresponding ablation episode, associate that episode with its date and patient, and specify the rules for identifying each patient's first recorded AF ablation and applying the 2020–2023 window. The intended result is a count of distinct patients.</p>
|
||||
<p><strong>What completion means.</strong> These choices are now consolidated into a documented mapping, with the relationships, selection rules, and validation checks shown for review. We have an explicit specification of where the answer will come from and how the relevant data fit together.</p>
|
||||
<p><strong>The final approval.</strong> By selecting <strong>Save and proceed</strong>, the reviewer approves this mapping for the next stage: planning and building the SQL query. Query execution and verification of the resulting patient count still follow.</p>
|
||||
<div data-popup-notes="thothii-cte-1"><div data-popup-notes-body>
|
||||
<p>(4) Once these choices are clear, we build and test smaller query steps, called CTEs. Each has a purpose, SQL and sample results for review. Here, we inspect patient and procedure records. Successful execution alone does not establish clinical correctness, so the reviewer can accept or request changes.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="thothii-cte-plan"><h3>CTEs: building the query step by step</h3><div data-popup-notes-body><p><strong>CTE stands for Common Table Expression.</strong> It is a named intermediate result within an SQL query.</p><p>CTEs break a complex research question into smaller, logical steps—for example, identifying AF ablations, finding the first procedure for each patient, and selecting the study population.</p><p>This makes the query easier to understand, check, and modify, helping us verify that each step reflects the intended clinical criteria.</p></div></div>
|
||||
<div data-popup-notes="thothii-cte-1"><h3>Reviewing a CTE: purpose, SQL, and results</h3><div data-popup-notes-body>
|
||||
<p><strong>Purpose and position.</strong> Each CTE is presented as a reviewable step. At the top, we see its name and position in the sequence: here, the first of three. A short explanation describes its clinical purpose and the reasoning behind the selection rules.</p>
|
||||
<p><strong>The SQL implementation.</strong> The code shows how that purpose is translated into database operations. In this example, it connects the treated pathology to the ablation episode and selects AF ablations, returning the patient identifier, episode identifier, and procedure date.</p>
|
||||
<p><strong>Tests and a data preview.</strong> The screen reports the test status and execution time, followed by a preview of the returned records. The ten rows shown are a limited preview, not the total study population. Successful execution still requires a check that the results make clinical sense.</p>
|
||||
<p><strong>Human review.</strong> The reviewer can compare the explanation, code, and sample data before choosing <strong>Save and proceed</strong> or <strong>Reject</strong>. This makes each intermediate step inspectable before it contributes to the final query.</p>
|
||||
<div data-popup-notes="thothii-sql"><div data-popup-notes-body>
|
||||
<p>(5) We can now assemble and verify the final SQL against the agreed question. The reviewer approves it or requests changes, checking that the query answers the original clinical intent.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="thothii-sql"><h3>Titolo8</h3><div data-popup-notes-body><p>testo8</p></div></div>
|
||||
<div data-popup-notes="thothii-datamart"><h3>Reusing the query: daily datamarts or research on demand</h3><div data-popup-notes-body>
|
||||
<p>Once the query generation and review workflow is complete, the validated SQL query can be reused beyond the current session.</p>
|
||||
<p><strong>Daily datamart generation.</strong> The query can be integrated into the broader ETL pipeline and scheduled to run every day. This allows the datamart to be rebuilt or refreshed systematically as new source data become available, using the same agreed selection rules.</p>
|
||||
<p><strong>Reuse by researchers.</strong> Alternatively, the query can simply be saved as an SQL file and made available to researchers, who can inspect it, run it when needed, or adapt it for a subsequent study.</p>
|
||||
<p>In both cases, the workflow produces a reusable query that captures the reviewed interpretation of the research question.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="thothii-memory"><h3>Saving clinical clarifications for future queries</h3><div data-popup-notes-body>
|
||||
<p><strong>What we retain.</strong> Clarifying a research question can produce knowledge that is useful beyond the current study: for example, an agreed interpretation of a clinical term or how a procedure is represented in the local data.</p>
|
||||
<p><strong>The researcher chooses.</strong> At the end of the workflow, ThothII presents candidate clarifications for reuse. The reviewer selects which ones to save as shared knowledge for the workspace.</p>
|
||||
<p><strong>How this helps next time.</strong> When a related question is asked, ThothII can retrieve these clarifications and propose them to the reviewer. The reviewer checks whether they apply to the new question before using them. This helps avoid repeating the same clarification work while keeping each study's interpretation under human control.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="thothii-finish"><h3>A saved session documents how the result was reached</h3><div data-popup-notes-body>
|
||||
<p>At the end of the workflow, ThothII finalizes a session that brings together the generated SQL statement and the documented process that led to it.</p>
|
||||
<p>The session preserves the research question, its agreed interpretation, the mapping to the database, the intermediate query steps, and the recorded validation results.</p>
|
||||
<p>Human involvement is captured through the recorded clarifications and review decisions: what the reviewer selected, approved, or rejected along the way. These records document how the interaction between the system and the researcher shaped the final query.</p>
|
||||
<p>Researchers can revisit the session to inspect the SQL and understand the choices behind it. This provides a traceable record of how the original question became the final result.</p>
|
||||
<p><strong>Time and cost for this example.</strong> The AI processing took less than five minutes, excluding the time spent on human review. Using DeepSeek V4 Flash, the estimated model cost for this run was about five cents.</p>
|
||||
</div></div>
|
||||
<div data-popup-notes="thothii-return"><h3>Expert supervision and deployment options</h3><div data-popup-notes-body>
|
||||
<p><strong>Supervision requires knowledge of the data.</strong> The workflow must be supervised by someone who understands the meaning and content of the database tables and fields. AI can make plausible guesses about what a field represents or how records should be connected, and these guesses can become hallucinations. The reviewer's role is to challenge those assumptions and check them against the actual data and its clinical meaning.</p>
|
||||
<p><strong>The researcher defines the dataset.</strong> Human judgement is essential to decide which information the study needs, which records meet the required quality criteria, and which time windows and temporal rules should apply. The responsible researcher must confirm these choices so that the resulting dataset addresses the research question. Successful SQL execution alone does not establish that the dataset is suitable for the study.</p>
|
||||
<p><strong>Server or authorised workstation.</strong> At Policlinico San Donato, ThothII runs on a server. The application can also be installed on the PC of a qualified staff member who is authorised to access the database remotely. The application runs on that workstation, while the database server executes the SQL through the authorised connection.</p>
|
||||
<p><strong>A local deployment option.</strong> Open models such as Qwen3.8-27B or Gemma 4 26B A4B are candidates for running the AI component on premises. Their suitability for this workflow should be assessed on representative research questions, including the quality of the generated SQL and the reliability of the review steps.</p>
|
||||
<p><strong>Hardware within reach.</strong> With quantized models, a compact server equipped with a suitable GPU is a practical deployment option. A GPU budget of a few thousand euros is a planning target; the required memory and final cost depend on the model, context length, and number of concurrent sessions.</p>
|
||||
<p><strong>Where the query runs.</strong> SQL execution takes place on the internal database server through controlled application code. The language model helps construct and review the query; the database engine executes it.</p>
|
||||
<p><strong>Keeping model processing local.</strong> Hosting the session model on premises allows its prompts and responses to remain within the organisation. With an external model, the information included in prompts and tool results must be checked and filtered to prevent sensitive data from leaving the environment.</p>
|
||||
<p><small>Model references: <a href="https://huggingface.co/Qwen/Qwen3.8-27B" target="_blank" rel="noopener noreferrer">Qwen3.8-27B model card</a>; <a href="https://ai.google.dev/gemma/docs/core" target="_blank" rel="noopener noreferrer">Gemma 4 models and memory requirements</a>.</small></p>
|
||||
<div data-popup-notes="thothii-datamart"><div data-popup-notes-body>
|
||||
<p>(6) With the query approved, we decide whether to produce a datamart: an analysis dataset that can be refreshed through the ETL pipeline. We also choose which clarifications to retain for future questions. The saved artifacts and review decisions document how the result was reached.</p>
|
||||
</div></div>
|
||||
</aside>
|
||||
</section>
|
||||
<!-- ============ 12 · HAPPY ENDING ============ -->
|
||||
<section class="arit">
|
||||
<header class="head">
|
||||
<img src="logo.png" alt="">
|
||||
<span class="head-org">AritmoLab · Policlinico San Donato</span>
|
||||
</header>
|
||||
<nav class="band">
|
||||
<span class="band-section">The happy ending</span>
|
||||
</nav>
|
||||
<div class="sbody">
|
||||
<div class="kicker">From datamart to discovery</div>
|
||||
<h2>One datamart — many questions answered</h2>
|
||||
<div class="charts">
|
||||
<div class="chart"><b>Multivariate analysis</b><span>[SVG chart — next step]</span></div>
|
||||
<div class="chart"><b>Machine learning</b><span>[SVG chart — next step]</span></div>
|
||||
</div>
|
||||
<p class="hint">[Illustrative data, modeled on published arrhythmology predictors — Brugada focus]</p>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">12 / 13</span></footer>
|
||||
<aside class="notes">[DRAFT — next step: two SVG charts, illustrative data modeled on real literature] And this is the happy ending. From one datamart, the Unit can run a multivariate analysis — which factors truly drive arrhythmic risk in Brugada patients — and train machine-learning models on the same table. What used to take weeks of manual data preparation now takes minutes. The AI did not replace the researcher: it gave the researcher back their time.</aside>
|
||||
</section>
|
||||
|
||||
<section class="arit title-slide">
|
||||
<header class="head">
|
||||
<img src="logo.png" alt="">
|
||||
@@ -742,15 +614,17 @@
|
||||
<div class="kicker">Questions?</div>
|
||||
<h1>Thank you</h1>
|
||||
<div class="hgap"></div>
|
||||
<p class="lead">Want the deep dives? The technical walkthroughs of the text miner, the mapping agents and ThothII are coming as videos on our Substack.</p>
|
||||
<p class="lead">We are available to explore AI in data reorganization and clinical text analysis,<br>and datamart generation from natural-language requests with ThothII.<br>Meet us during the conference or arrange a remote session.</p>
|
||||
<div class="speakers" style="margin-top:32px">
|
||||
<span>Dr. Marco Pancotti - MultiPhysixLab</span>
|
||||
<a href="mailto:mpancotti@mpxlab.org" style="font-size:32px;font-weight:700;color:var(--bordeaux)">mpancotti@mpxlab.org</a>
|
||||
<span>Dr. Sara Paratico - I.R.C.C.S. Policlinico San Donato</span>
|
||||
<a href="mailto:sara.paratico@grupposandonato.it" style="font-size:32px;font-weight:700;color:var(--bordeaux)">sara.paratico@grupposandonato.it</a>
|
||||
</div>
|
||||
</div>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">13 / 13</span></footer>
|
||||
<footer class="foot"><span class="g">Role of AI in the Analysis of Unstructured Clinical Databases</span><span class="g">Dr. Marco Pancotti - MultiPhysixLab</span><span class="g">Dr. Sara Paratico - Gruppo San Donato</span><span class="g">San Donato Milanese, Milan, Italy · 2–3 October 2026</span><span class="g num">11 / 11</span></footer>
|
||||
<aside class="notes">
|
||||
[DRAFT] Thank you for your attention. If you want to go deeper — the text miner, the mapping agents, ThothII — we are publishing the technical walkthroughs as videos on our Substack; the link is on the final version of this deck. And now, happy to take your questions.
|
||||
Thank you for your attention. Sara Paratico and I are available for a closer look at how we use AI to reorganize data, analyse clinical text, and generate datamarts from natural-language requests with ThothII. We can discuss these topics and demonstrate the tools during the conference or remotely in the coming days. Please contact us at the email addresses on this slide.
|
||||
</aside>
|
||||
</section>
|
||||
|
||||
@@ -780,10 +654,10 @@
|
||||
});
|
||||
|
||||
// jump dropdown — one select per slide, in the red band
|
||||
const JUMP_TITLES = ['Title', 'Where we started', 'What we wanted to build',
|
||||
const JUMP_TITLES = ['Title', 'What we wanted to build',
|
||||
'The problem', 'What the text miner reads', 'Hard problems of clinical NLP', 'The clinical ontology',
|
||||
'Trust the text', 'AritmoLab today', 'All good? Not yet', 'A tour of ThothII',
|
||||
'From datamarts to models', 'Thank you'];
|
||||
'Thank you'];
|
||||
document.querySelectorAll('.arit .band').forEach((band) => {
|
||||
const sel = document.createElement('select');
|
||||
sel.className = 'jump';
|
||||
@@ -1128,7 +1002,6 @@
|
||||
notes: slide.querySelector('aside.notes')?.innerHTML || '',
|
||||
popupNotes: [...slide.querySelectorAll('aside.notes [data-popup-notes]')].map(note => ({
|
||||
key: note.dataset.popupNotes,
|
||||
title: note.querySelector('h3').textContent,
|
||||
html: note.querySelector('[data-popup-notes-body]').innerHTML,
|
||||
})),
|
||||
popups: [...slide.querySelectorAll(popupSelector)].map(el => {
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
display: grid; grid-template-columns: repeat(4, 1fr); grid-template-rows: repeat(2, minmax(0, 1fr));
|
||||
gap: 16px; flex: 1; min-height: 0; margin-top: 10px;
|
||||
}
|
||||
.arit .screenshot-tour-four { grid-template-columns: repeat(2, minmax(0, 1fr)); }
|
||||
.arit .screenshot-tour button {
|
||||
display: flex; flex-direction: column; min-width: 0; min-height: 0; padding: 0;
|
||||
overflow: hidden; border: 1px solid var(--line-strong); border-radius: 8px;
|
||||
@@ -20,7 +21,7 @@
|
||||
}
|
||||
.arit .screenshot-tour b { color: var(--bordeaux); font: 700 22px/1 var(--sans); }
|
||||
.arit .screenshot-tour-compact {
|
||||
grid-template-columns: repeat(6, minmax(0, 1fr));
|
||||
grid-template-columns: repeat(3, minmax(0, 1fr));
|
||||
grid-template-rows: repeat(2, minmax(0, 1fr)); gap: 10px;
|
||||
}
|
||||
.arit .screenshot-tour-compact span { gap: 8px; padding: 6px 10px; font-size: 14px; }
|
||||
|
||||
Reference in New Issue
Block a user