From 4b1f09f3e490aee88c29c15dd07d654f00d29b26 Mon Sep 17 00:00:00 2001
From: "George G. Vega Yon"
Date: Tue, 18 Aug 2026 18:42:35 -0600
Subject: [PATCH 1/2] Publish a page per talk, generated from TOML records
Talks were only recorded in the Google Calendar, so links pointed at calendar
entries and nothing about a past talk survived on the site. Each talk now has a
record in `_data/talks/*.toml` (title, speakers, bios, abstract, location, Zoom,
slides, recording, ...) and its own page under `/talks/-/`.
GitHub Pages builds Jekyll in safe mode, so the pages cannot be produced by a
plugin at build time: `scripts/generate_talks.py` turns the TOML records into
`_talks/*.md` and both are committed. A new workflow re-runs the generator with
`--check` on every PR so the two can't drift apart.
- `scripts/import_calendar_talks.py` seeds records from the public calendar feed
(178 talks, 2020-2026); imported records are flagged `needs_review = true`
- `_layouts/talk.html` renders a talk: abstract, speaker bios and links, slides,
recording; Zoom links are only shown while a talk is still upcoming
- `/talks/` lists upcoming talks, then past talks by year
- `seminar.md` and `_includes/next_talks.html` now read from these records
instead of fetching the Google Calendar with JavaScript
- `future: true` in `_config.yml`, otherwise Jekyll hides upcoming talks
- README documents how to add a talk
Co-Authored-By: Claude Opus 5
---
.github/workflows/talks.yml | 26 +
Makefile | 11 +-
README.md | 71 +++
_config.yml | 9 +
_data/talks/2020-01-09-chris-musco.toml | 33 ++
_data/talks/2020-01-16-alexander-lex.toml | 27 +
_data/talks/2020-01-23-harish-maringanti.toml | 33 ++
_data/talks/2020-01-30-qingyao-ai.toml | 27 +
_data/talks/2020-02-06-gail-zasowski.toml | 27 +
_data/talks/2020-02-20-bei-wang.toml | 27 +
_data/talks/2020-02-27-taylor-sparks.toml | 33 ++
_data/talks/2020-03-05-john-horel.toml | 31 +
_data/talks/2020-08-28-vivek-gupta.toml | 32 +
_data/talks/2020-09-04-dheeraj-mekala.toml | 27 +
_data/talks/2020-09-11-parthe-pandit.toml | 38 ++
_data/talks/2020-09-18-vishnu-lokhande.toml | 27 +
_data/talks/2020-10-02-ellen-riloff.toml | 57 ++
_data/talks/2020-10-09-swaroop-mishra.toml | 27 +
_data/talks/2020-10-16-varun-gangal.toml | 27 +
_data/talks/2020-10-23-nancy-wang.toml | 34 ++
_data/talks/2020-10-30-alberto-cairo.toml | 27 +
.../talks/2020-10-30-daniel-scharfstein.toml | 30 +
.../talks/2020-11-06-bhargavi-paranjape.toml | 27 +
_data/talks/2020-11-13-akanksha-atrey.toml | 27 +
_data/talks/2020-11-20-akhil-arora.toml | 27 +
_data/talks/2020-12-04-danish-pruthi.toml | 27 +
.../2021-01-22-grad-student-spotlights.toml | 32 +
_data/talks/2021-01-29-sanghamitra-dutta.toml | 27 +
_data/talks/2021-02-05-michal-moshkovitz.toml | 27 +
_data/talks/2021-02-12-fritz-lekschas.toml | 44 ++
_data/talks/2021-02-19-samson-zhou.toml | 31 +
_data/talks/2021-02-26-rajesh-jayaram.toml | 36 ++
_data/talks/2021-03-12-yaoqing-yang.toml | 27 +
_data/talks/2021-03-19-vivek-gupta.toml | 27 +
_data/talks/2021-04-09-emily-beth-wall.toml | 31 +
_data/talks/2021-04-16-arun-sai-suggala.toml | 31 +
_data/talks/2021-04-23-lizzie-kumar.toml | 27 +
_data/talks/2021-08-27-jeff-phillips.toml | 27 +
_data/talks/2021-09-03-shandian-zhe.toml | 31 +
_data/talks/2021-09-10-pierre-lermusiaux.toml | 35 ++
_data/talks/2021-09-17-ross-whitaker.toml | 33 ++
_data/talks/2021-09-24-c-seshadhri.toml | 35 ++
_data/talks/2021-10-01-tony-h-grubesic.toml | 27 +
_data/talks/2021-10-08-anna-little.toml | 27 +
.../talks/2021-10-22-erin-wolf-chambers.toml | 49 ++
_data/talks/2021-10-29-bao-wang.toml | 32 +
_data/talks/2021-11-05-marina-kogan.toml | 27 +
_data/talks/2021-11-12-mikhail-belkin.toml | 44 ++
_data/talks/2021-11-19-michael-yeh.toml | 27 +
_data/talks/2021-12-03-julia-silge.toml | 31 +
_data/talks/2021-12-10-sameer-singh.toml | 32 +
_data/talks/2022-01-14-yi-zhou.toml | 27 +
_data/talks/2022-01-21-chinmay-hedge.toml | 27 +
_data/talks/2022-01-28-swaroop-mishra.toml | 27 +
_data/talks/2022-02-04-tuhin-chakrabarty.toml | 27 +
.../2022-02-11-vaggos-chatziafratis.toml | 37 ++
_data/talks/2022-02-18-jeff-phillips.toml | 27 +
_data/talks/2022-02-25-sunipa-dev.toml | 27 +
.../2022-03-04-anirudh-goyal-of-montreal.toml | 27 +
.../2022-03-25-chad-topaz-jude-higdon.toml | 34 ++
_data/talks/2022-04-01-debanjan-mahata.toml | 27 +
_data/talks/2022-04-08-tao-li.toml | 27 +
_data/talks/2022-04-15-abhinav-kumar.toml | 27 +
_data/talks/2022-04-22-khyati-chandu.toml | 31 +
_data/talks/2022-04-29-james-brundage.toml | 27 +
_data/talks/2022-08-24-bei-wang-phillips.toml | 40 ++
_data/talks/2022-08-31-casey-greene.toml | 27 +
_data/talks/2022-09-07-kevin-moon.toml | 27 +
_data/talks/2022-09-14-elliot-smith.toml | 27 +
_data/talks/2022-09-21-jes-ford.toml | 27 +
_data/talks/2022-09-28-prashant-pandey.toml | 27 +
_data/talks/2022-10-05-jessica-shi.toml | 31 +
_data/talks/2022-10-19-jie-zhang.toml | 31 +
_data/talks/2022-11-02-alex-chin.toml | 27 +
_data/talks/2022-11-09-shireen-elhabian.toml | 27 +
_data/talks/2022-11-16-bernadette-stolz.toml | 27 +
_data/talks/2022-11-30-aaron-quinlan.toml | 27 +
_data/talks/2022-12-07-tao-yang.toml | 30 +
_data/talks/2023-01-11-george-vega-yon.toml | 27 +
_data/talks/2023-01-18-echo-warner.toml | 27 +
_data/talks/2023-01-25-shweta-jain.toml | 31 +
_data/talks/2023-02-01-ana-marsovic.toml | 31 +
_data/talks/2023-02-08-aaron-clauset.toml | 35 ++
_data/talks/2023-02-15-orly-alter.toml | 27 +
_data/talks/2023-02-22-vivek-gupta.toml | 27 +
_data/talks/2023-03-01-emily-hadley.toml | 27 +
_data/talks/2023-03-15-nate-veldt.toml | 27 +
_data/talks/2023-03-22-bailey-fosdick.toml | 27 +
_data/talks/2023-03-29-justin-baker.toml | 27 +
_data/talks/2023-04-05-yao-yaun-mao.toml | 27 +
_data/talks/2023-04-12-titus-brown.toml | 27 +
_data/talks/2023-04-19-sumana-basu.toml | 27 +
.../2023-05-30-bodhisattwa-majumder.toml | 33 ++
...023-08-23-data-science-lecture-series.toml | 27 +
...0-data-science-lecture-series-speaker.toml | 27 +
...023-08-30-data-science-lecture-series.toml | 27 +
...6-data-science-lecture-series-speaker.toml | 27 +
...3-data-science-lecture-series-speaker.toml | 27 +
...0-data-science-lecture-series-speaker.toml | 27 +
...2023-09-25-sandia-information-session.toml | 27 +
...7-data-science-lecture-series-speaker.toml | 27 +
...4-data-science-lecture-series-speaker.toml | 31 +
...8-data-science-lecture-series-speaker.toml | 35 ++
...5-data-science-lecture-series-speaker.toml | 32 +
...1-data-science-lecture-series-speaker.toml | 34 ++
...2-data-science-lecture-series-speaker.toml | 31 +
...9-data-science-lecture-series-speaker.toml | 27 +
...6-data-science-lecture-series-speaker.toml | 33 ++
...3-data-science-lecture-series-speaker.toml | 31 +
...024-01-10-data-science-lecture-series.toml | 27 +
_data/talks/2024-01-31-pratik-soni.toml | 30 +
.../talks/2024-03-13-swabha-swayamdipta.toml | 27 +
.../talks/2024-04-10-ucds-lecture-series.toml | 37 ++
_data/talks/2024-08-27-jeff-phillips.toml | 33 ++
_data/talks/2024-09-03-esha-datta.toml | 27 +
_data/talks/2024-09-10-guanhong-tao.toml | 30 +
_data/talks/2024-09-17-aurora-clark.toml | 27 +
_data/talks/2024-09-24-rebecca-barter.toml | 30 +
_data/talks/2024-10-01-simon-brewer.toml | 27 +
.../talks/2024-10-15-raghav-venkatraman.toml | 35 ++
_data/talks/2024-10-29-vivek-gupta.toml | 35 ++
_data/talks/2024-11-12-amir-abdullah.toml | 27 +
_data/talks/2024-11-19-hoaning-xue.toml | 27 +
_data/talks/2024-11-26-zhichao-xu.toml | 27 +
_data/talks/2025-01-17-fengjiao-wang.toml | 27 +
.../talks/2025-01-24-data-science-ai-day.toml | 34 ++
_data/talks/2025-02-07-omkar-bhalerao.toml | 32 +
_data/talks/2025-02-14-bao-wang.toml | 31 +
_data/talks/2025-02-21-chenglu-li.toml | 27 +
_data/talks/2025-02-28-bernardo-modenesi.toml | 32 +
_data/talks/2025-03-21-jeff-phillips.toml | 37 ++
_data/talks/2025-03-28-seth-pettie.toml | 51 ++
_data/talks/2025-04-04-sabyasachi-basu.toml | 33 ++
_data/talks/2025-04-11-peter-jacobs.toml | 31 +
_data/talks/2025-04-18-tucker-hermans.toml | 33 ++
_data/talks/2025-08-20-varun-shankar.toml | 27 +
_data/talks/2025-08-27-konstantin-genin.toml | 27 +
_data/talks/2025-09-03-daniel-brown.toml | 27 +
_data/talks/2025-09-10-bei-wang-phillips.toml | 27 +
_data/talks/2025-09-17-aditya-bhaskara.toml | 27 +
.../2025-09-24-kenneth-blake-vernon.toml | 35 ++
_data/talks/2025-10-01-luis-garcia.toml | 27 +
_data/talks/2025-10-15-vineet-pandey.toml | 27 +
_data/talks/2025-10-22-kenneth-marino.toml | 27 +
_data/talks/2025-10-29-anna-fariha.toml | 27 +
_data/talks/2025-11-05-jenny-lin.toml | 27 +
.../2025-11-12-kyle-dawson-tyler-hagen.toml | 39 ++
_data/talks/2025-11-19-vivek-srikumar.toml | 27 +
.../2025-12-03-data-visualization-101.toml | 27 +
_data/talks/2026-01-09-marina-kogan.toml | 27 +
_data/talks/2026-01-23-andrew-mcnutt.toml | 27 +
_data/talks/2026-02-06-makoto-kelp.toml | 27 +
_data/talks/2026-02-17-juliana-freire.toml | 39 ++
_data/talks/2026-02-19-dinesh-manocha.toml | 27 +
_data/talks/2026-02-24-paul-parsons.toml | 27 +
_data/talks/2026-02-26-zezhong-wang.toml | 27 +
_data/talks/2026-02-27-kenny-marino.toml | 27 +
_data/talks/2026-03-02-bogdan-raita.toml | 27 +
_data/talks/2026-03-03-grace-guo.toml | 27 +
_data/talks/2026-03-05-josh-levine.toml | 27 +
_data/talks/2026-03-16-tenghao-huang.toml | 27 +
_data/talks/2026-03-20-kate-isaacs.toml | 27 +
_data/talks/2026-03-20-xueguang-ma.toml | 27 +
_data/talks/2026-03-23-benjie-wang.toml | 31 +
_data/talks/2026-03-25-dick-sadler.toml | 27 +
_data/talks/2026-03-26-bailing-lyu.toml | 27 +
_data/talks/2026-03-27-erdogan-kaya.toml | 27 +
_data/talks/2026-03-30-xiaoling-hu.toml | 33 ++
_data/talks/2026-03-31-he-yin.toml | 35 ++
.../2026-04-01-md-mostafijur-rahman.toml | 27 +
_data/talks/2026-04-06-qiang-ji.toml | 39 ++
_data/talks/2026-04-07-si-chen.toml | 27 +
_data/talks/2026-04-08-fahim-faisal.toml | 27 +
_data/talks/2026-04-10-chase-neumann.toml | 40 ++
_data/talks/2026-04-13-jihyun-rho.toml | 27 +
_data/talks/2026-04-14-sameer-honwad.toml | 31 +
_data/talks/2026-04-15-amirali-abdullah.toml | 30 +
_data/talks/2026-04-15-chengbin-deng.toml | 27 +
_data/talks/2026-04-15-shiqi-yu.toml | 27 +
_data/talks/2026-04-17-daniel-sieta.toml | 27 +
_data/talks/2026-08-28-warren-pettine.toml | 27 +
_data/talks/2026-09-04-george-vega-yon.toml | 27 +
_data/talks/_TEMPLATE.toml | 40 ++
_includes/next_talks.html | 75 ++-
_layouts/talk.html | 137 +++++
_talks/2020-01-09-chris-musco.md | 27 +
_talks/2020-01-16-alexander-lex.md | 22 +
_talks/2020-01-23-harish-maringanti.md | 27 +
_talks/2020-01-30-qingyao-ai.md | 23 +
_talks/2020-02-06-gail-zasowski.md | 23 +
_talks/2020-02-20-bei-wang.md | 23 +
_talks/2020-02-27-taylor-sparks.md | 28 +
_talks/2020-03-05-john-horel.md | 24 +
_talks/2020-08-28-vivek-gupta.md | 28 +
_talks/2020-09-04-dheeraj-mekala.md | 23 +
_talks/2020-09-11-parthe-pandit.md | 29 +
_talks/2020-09-18-vishnu-lokhande.md | 24 +
_talks/2020-10-02-ellen-riloff.md | 40 ++
_talks/2020-10-09-swaroop-mishra.md | 24 +
_talks/2020-10-16-varun-gangal.md | 24 +
_talks/2020-10-23-nancy-wang.md | 29 +
_talks/2020-10-30-alberto-cairo.md | 21 +
_talks/2020-10-30-daniel-scharfstein.md | 24 +
_talks/2020-11-06-bhargavi-paranjape.md | 24 +
_talks/2020-11-13-akanksha-atrey.md | 24 +
_talks/2020-11-20-akhil-arora.md | 23 +
_talks/2020-12-04-danish-pruthi.md | 24 +
_talks/2021-01-22-grad-student-spotlights.md | 25 +
_talks/2021-01-29-sanghamitra-dutta.md | 23 +
_talks/2021-02-05-michal-moshkovitz.md | 23 +
_talks/2021-02-12-fritz-lekschas.md | 37 ++
_talks/2021-02-19-samson-zhou.md | 24 +
_talks/2021-02-26-rajesh-jayaram.md | 32 +
_talks/2021-03-12-yaoqing-yang.md | 23 +
_talks/2021-03-19-vivek-gupta.md | 24 +
_talks/2021-04-09-emily-beth-wall.md | 26 +
_talks/2021-04-16-arun-sai-suggala.md | 26 +
_talks/2021-04-23-lizzie-kumar.md | 24 +
_talks/2021-08-27-jeff-phillips.md | 24 +
_talks/2021-09-03-shandian-zhe.md | 25 +
_talks/2021-09-10-pierre-lermusiaux.md | 25 +
_talks/2021-09-17-ross-whitaker.md | 27 +
_talks/2021-09-24-c-seshadhri.md | 26 +
_talks/2021-10-01-tony-h-grubesic.md | 23 +
_talks/2021-10-08-anna-little.md | 24 +
_talks/2021-10-22-erin-wolf-chambers.md | 44 ++
_talks/2021-10-29-bao-wang.md | 27 +
_talks/2021-11-05-marina-kogan.md | 24 +
_talks/2021-11-12-mikhail-belkin.md | 37 ++
_talks/2021-11-19-michael-yeh.md | 23 +
_talks/2021-12-03-julia-silge.md | 25 +
_talks/2021-12-10-sameer-singh.md | 27 +
_talks/2022-01-14-yi-zhou.md | 25 +
_talks/2022-01-21-chinmay-hedge.md | 23 +
_talks/2022-01-28-swaroop-mishra.md | 22 +
_talks/2022-02-04-tuhin-chakrabarty.md | 24 +
_talks/2022-02-11-vaggos-chatziafratis.md | 31 +
_talks/2022-02-18-jeff-phillips.md | 23 +
_talks/2022-02-25-sunipa-dev.md | 24 +
.../2022-03-04-anirudh-goyal-of-montreal.md | 24 +
_talks/2022-03-25-chad-topaz-jude-higdon.md | 26 +
_talks/2022-04-01-debanjan-mahata.md | 24 +
_talks/2022-04-08-tao-li.md | 24 +
_talks/2022-04-15-abhinav-kumar.md | 24 +
_talks/2022-04-22-khyati-chandu.md | 27 +
_talks/2022-04-29-james-brundage.md | 24 +
_talks/2022-08-24-bei-wang-phillips.md | 34 ++
_talks/2022-08-31-casey-greene.md | 21 +
_talks/2022-09-07-kevin-moon.md | 25 +
_talks/2022-09-14-elliot-smith.md | 24 +
_talks/2022-09-21-jes-ford.md | 24 +
_talks/2022-09-28-prashant-pandey.md | 24 +
_talks/2022-10-05-jessica-shi.md | 27 +
_talks/2022-10-19-jie-zhang.md | 25 +
_talks/2022-11-02-alex-chin.md | 24 +
_talks/2022-11-09-shireen-elhabian.md | 24 +
_talks/2022-11-16-bernadette-stolz.md | 24 +
_talks/2022-11-30-aaron-quinlan.md | 24 +
_talks/2022-12-07-tao-yang.md | 26 +
_talks/2023-01-11-george-vega-yon.md | 25 +
_talks/2023-01-18-echo-warner.md | 24 +
_talks/2023-01-25-shweta-jain.md | 27 +
_talks/2023-02-01-ana-marsovic.md | 25 +
_talks/2023-02-08-aaron-clauset.md | 27 +
_talks/2023-02-15-orly-alter.md | 24 +
_talks/2023-02-22-vivek-gupta.md | 25 +
_talks/2023-03-01-emily-hadley.md | 24 +
_talks/2023-03-15-nate-veldt.md | 24 +
_talks/2023-03-22-bailey-fosdick.md | 24 +
_talks/2023-03-29-justin-baker.md | 22 +
_talks/2023-04-05-yao-yaun-mao.md | 24 +
_talks/2023-04-12-titus-brown.md | 25 +
_talks/2023-04-19-sumana-basu.md | 24 +
_talks/2023-05-30-bodhisattwa-majumder.md | 29 +
.../2023-08-23-data-science-lecture-series.md | 20 +
...-30-data-science-lecture-series-speaker.md | 22 +
.../2023-08-30-data-science-lecture-series.md | 20 +
...-06-data-science-lecture-series-speaker.md | 21 +
...-13-data-science-lecture-series-speaker.md | 22 +
...-20-data-science-lecture-series-speaker.md | 21 +
.../2023-09-25-sandia-information-session.md | 21 +
...-27-data-science-lecture-series-speaker.md | 23 +
...-04-data-science-lecture-series-speaker.md | 24 +
...-18-data-science-lecture-series-speaker.md | 25 +
...-25-data-science-lecture-series-speaker.md | 23 +
...-01-data-science-lecture-series-speaker.md | 24 +
...-22-data-science-lecture-series-speaker.md | 24 +
...-29-data-science-lecture-series-speaker.md | 22 +
...-06-data-science-lecture-series-speaker.md | 25 +
...-13-data-science-lecture-series-speaker.md | 24 +
.../2024-01-10-data-science-lecture-series.md | 19 +
_talks/2024-01-31-pratik-soni.md | 23 +
_talks/2024-03-13-swabha-swayamdipta.md | 22 +
_talks/2024-04-10-ucds-lecture-series.md | 26 +
_talks/2024-08-27-jeff-phillips.md | 28 +
_talks/2024-09-03-esha-datta.md | 24 +
_talks/2024-09-10-guanhong-tao.md | 23 +
_talks/2024-09-17-aurora-clark.md | 22 +
_talks/2024-09-24-rebecca-barter.md | 25 +
_talks/2024-10-01-simon-brewer.md | 22 +
_talks/2024-10-15-raghav-venkatraman.md | 28 +
_talks/2024-10-29-vivek-gupta.md | 31 +
_talks/2024-11-12-amir-abdullah.md | 23 +
_talks/2024-11-19-hoaning-xue.md | 24 +
_talks/2024-11-26-zhichao-xu.md | 25 +
_talks/2025-01-17-fengjiao-wang.md | 24 +
_talks/2025-01-24-data-science-ai-day.md | 25 +
_talks/2025-02-07-omkar-bhalerao.md | 25 +
_talks/2025-02-14-bao-wang.md | 24 +
_talks/2025-02-21-chenglu-li.md | 24 +
_talks/2025-02-28-bernardo-modenesi.md | 27 +
_talks/2025-03-21-jeff-phillips.md | 32 +
_talks/2025-03-28-seth-pettie.md | 46 ++
_talks/2025-04-04-sabyasachi-basu.md | 28 +
_talks/2025-04-11-peter-jacobs.md | 26 +
_talks/2025-04-18-tucker-hermans.md | 28 +
_talks/2025-08-20-varun-shankar.md | 21 +
_talks/2025-08-27-konstantin-genin.md | 20 +
_talks/2025-09-03-daniel-brown.md | 20 +
_talks/2025-09-10-bei-wang-phillips.md | 20 +
_talks/2025-09-17-aditya-bhaskara.md | 20 +
_talks/2025-09-24-kenneth-blake-vernon.md | 28 +
_talks/2025-10-01-luis-garcia.md | 22 +
_talks/2025-10-15-vineet-pandey.md | 20 +
_talks/2025-10-22-kenneth-marino.md | 20 +
_talks/2025-10-29-anna-fariha.md | 20 +
_talks/2025-11-05-jenny-lin.md | 20 +
_talks/2025-11-12-kyle-dawson-tyler-hagen.md | 27 +
_talks/2025-11-19-vivek-srikumar.md | 20 +
_talks/2025-12-03-data-visualization-101.md | 20 +
_talks/2026-01-09-marina-kogan.md | 24 +
_talks/2026-01-23-andrew-mcnutt.md | 23 +
_talks/2026-02-06-makoto-kelp.md | 23 +
_talks/2026-02-17-juliana-freire.md | 33 ++
_talks/2026-02-19-dinesh-manocha.md | 22 +
_talks/2026-02-24-paul-parsons.md | 22 +
_talks/2026-02-26-zezhong-wang.md | 22 +
_talks/2026-02-27-kenny-marino.md | 23 +
_talks/2026-03-02-bogdan-raita.md | 21 +
_talks/2026-03-03-grace-guo.md | 22 +
_talks/2026-03-05-josh-levine.md | 22 +
_talks/2026-03-16-tenghao-huang.md | 22 +
_talks/2026-03-20-kate-isaacs.md | 23 +
_talks/2026-03-20-xueguang-ma.md | 23 +
_talks/2026-03-23-benjie-wang.md | 22 +
_talks/2026-03-25-dick-sadler.md | 21 +
_talks/2026-03-26-bailing-lyu.md | 23 +
_talks/2026-03-27-erdogan-kaya.md | 22 +
_talks/2026-03-30-xiaoling-hu.md | 26 +
_talks/2026-03-31-he-yin.md | 24 +
_talks/2026-04-01-md-mostafijur-rahman.md | 22 +
_talks/2026-04-06-qiang-ji.md | 26 +
_talks/2026-04-07-si-chen.md | 21 +
_talks/2026-04-08-fahim-faisal.md | 22 +
_talks/2026-04-10-chase-neumann.md | 28 +
_talks/2026-04-13-jihyun-rho.md | 22 +
_talks/2026-04-14-sameer-honwad.md | 24 +
_talks/2026-04-15-amirali-abdullah.md | 24 +
_talks/2026-04-15-chengbin-deng.md | 22 +
_talks/2026-04-15-shiqi-yu.md | 23 +
_talks/2026-04-17-daniel-sieta.md | 24 +
_talks/2026-08-28-warren-pettine.md | 21 +
_talks/2026-09-04-george-vega-yon.md | 21 +
.../import_calendar_talks.cpython-313.pyc | Bin 0 -> 23077 bytes
scripts/generate_talks.py | 235 ++++++++
scripts/import_calendar_talks.py | 551 ++++++++++++++++++
seminar.md | 13 +-
talks.md | 80 +++
368 files changed, 10885 insertions(+), 34 deletions(-)
create mode 100644 .github/workflows/talks.yml
create mode 100644 _data/talks/2020-01-09-chris-musco.toml
create mode 100644 _data/talks/2020-01-16-alexander-lex.toml
create mode 100644 _data/talks/2020-01-23-harish-maringanti.toml
create mode 100644 _data/talks/2020-01-30-qingyao-ai.toml
create mode 100644 _data/talks/2020-02-06-gail-zasowski.toml
create mode 100644 _data/talks/2020-02-20-bei-wang.toml
create mode 100644 _data/talks/2020-02-27-taylor-sparks.toml
create mode 100644 _data/talks/2020-03-05-john-horel.toml
create mode 100644 _data/talks/2020-08-28-vivek-gupta.toml
create mode 100644 _data/talks/2020-09-04-dheeraj-mekala.toml
create mode 100644 _data/talks/2020-09-11-parthe-pandit.toml
create mode 100644 _data/talks/2020-09-18-vishnu-lokhande.toml
create mode 100644 _data/talks/2020-10-02-ellen-riloff.toml
create mode 100644 _data/talks/2020-10-09-swaroop-mishra.toml
create mode 100644 _data/talks/2020-10-16-varun-gangal.toml
create mode 100644 _data/talks/2020-10-23-nancy-wang.toml
create mode 100644 _data/talks/2020-10-30-alberto-cairo.toml
create mode 100644 _data/talks/2020-10-30-daniel-scharfstein.toml
create mode 100644 _data/talks/2020-11-06-bhargavi-paranjape.toml
create mode 100644 _data/talks/2020-11-13-akanksha-atrey.toml
create mode 100644 _data/talks/2020-11-20-akhil-arora.toml
create mode 100644 _data/talks/2020-12-04-danish-pruthi.toml
create mode 100644 _data/talks/2021-01-22-grad-student-spotlights.toml
create mode 100644 _data/talks/2021-01-29-sanghamitra-dutta.toml
create mode 100644 _data/talks/2021-02-05-michal-moshkovitz.toml
create mode 100644 _data/talks/2021-02-12-fritz-lekschas.toml
create mode 100644 _data/talks/2021-02-19-samson-zhou.toml
create mode 100644 _data/talks/2021-02-26-rajesh-jayaram.toml
create mode 100644 _data/talks/2021-03-12-yaoqing-yang.toml
create mode 100644 _data/talks/2021-03-19-vivek-gupta.toml
create mode 100644 _data/talks/2021-04-09-emily-beth-wall.toml
create mode 100644 _data/talks/2021-04-16-arun-sai-suggala.toml
create mode 100644 _data/talks/2021-04-23-lizzie-kumar.toml
create mode 100644 _data/talks/2021-08-27-jeff-phillips.toml
create mode 100644 _data/talks/2021-09-03-shandian-zhe.toml
create mode 100644 _data/talks/2021-09-10-pierre-lermusiaux.toml
create mode 100644 _data/talks/2021-09-17-ross-whitaker.toml
create mode 100644 _data/talks/2021-09-24-c-seshadhri.toml
create mode 100644 _data/talks/2021-10-01-tony-h-grubesic.toml
create mode 100644 _data/talks/2021-10-08-anna-little.toml
create mode 100644 _data/talks/2021-10-22-erin-wolf-chambers.toml
create mode 100644 _data/talks/2021-10-29-bao-wang.toml
create mode 100644 _data/talks/2021-11-05-marina-kogan.toml
create mode 100644 _data/talks/2021-11-12-mikhail-belkin.toml
create mode 100644 _data/talks/2021-11-19-michael-yeh.toml
create mode 100644 _data/talks/2021-12-03-julia-silge.toml
create mode 100644 _data/talks/2021-12-10-sameer-singh.toml
create mode 100644 _data/talks/2022-01-14-yi-zhou.toml
create mode 100644 _data/talks/2022-01-21-chinmay-hedge.toml
create mode 100644 _data/talks/2022-01-28-swaroop-mishra.toml
create mode 100644 _data/talks/2022-02-04-tuhin-chakrabarty.toml
create mode 100644 _data/talks/2022-02-11-vaggos-chatziafratis.toml
create mode 100644 _data/talks/2022-02-18-jeff-phillips.toml
create mode 100644 _data/talks/2022-02-25-sunipa-dev.toml
create mode 100644 _data/talks/2022-03-04-anirudh-goyal-of-montreal.toml
create mode 100644 _data/talks/2022-03-25-chad-topaz-jude-higdon.toml
create mode 100644 _data/talks/2022-04-01-debanjan-mahata.toml
create mode 100644 _data/talks/2022-04-08-tao-li.toml
create mode 100644 _data/talks/2022-04-15-abhinav-kumar.toml
create mode 100644 _data/talks/2022-04-22-khyati-chandu.toml
create mode 100644 _data/talks/2022-04-29-james-brundage.toml
create mode 100644 _data/talks/2022-08-24-bei-wang-phillips.toml
create mode 100644 _data/talks/2022-08-31-casey-greene.toml
create mode 100644 _data/talks/2022-09-07-kevin-moon.toml
create mode 100644 _data/talks/2022-09-14-elliot-smith.toml
create mode 100644 _data/talks/2022-09-21-jes-ford.toml
create mode 100644 _data/talks/2022-09-28-prashant-pandey.toml
create mode 100644 _data/talks/2022-10-05-jessica-shi.toml
create mode 100644 _data/talks/2022-10-19-jie-zhang.toml
create mode 100644 _data/talks/2022-11-02-alex-chin.toml
create mode 100644 _data/talks/2022-11-09-shireen-elhabian.toml
create mode 100644 _data/talks/2022-11-16-bernadette-stolz.toml
create mode 100644 _data/talks/2022-11-30-aaron-quinlan.toml
create mode 100644 _data/talks/2022-12-07-tao-yang.toml
create mode 100644 _data/talks/2023-01-11-george-vega-yon.toml
create mode 100644 _data/talks/2023-01-18-echo-warner.toml
create mode 100644 _data/talks/2023-01-25-shweta-jain.toml
create mode 100644 _data/talks/2023-02-01-ana-marsovic.toml
create mode 100644 _data/talks/2023-02-08-aaron-clauset.toml
create mode 100644 _data/talks/2023-02-15-orly-alter.toml
create mode 100644 _data/talks/2023-02-22-vivek-gupta.toml
create mode 100644 _data/talks/2023-03-01-emily-hadley.toml
create mode 100644 _data/talks/2023-03-15-nate-veldt.toml
create mode 100644 _data/talks/2023-03-22-bailey-fosdick.toml
create mode 100644 _data/talks/2023-03-29-justin-baker.toml
create mode 100644 _data/talks/2023-04-05-yao-yaun-mao.toml
create mode 100644 _data/talks/2023-04-12-titus-brown.toml
create mode 100644 _data/talks/2023-04-19-sumana-basu.toml
create mode 100644 _data/talks/2023-05-30-bodhisattwa-majumder.toml
create mode 100644 _data/talks/2023-08-23-data-science-lecture-series.toml
create mode 100644 _data/talks/2023-08-30-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-08-30-data-science-lecture-series.toml
create mode 100644 _data/talks/2023-09-06-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-09-13-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-09-20-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-09-25-sandia-information-session.toml
create mode 100644 _data/talks/2023-09-27-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-10-04-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-10-18-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-10-25-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-11-01-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-11-22-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-11-29-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-12-06-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2023-12-13-data-science-lecture-series-speaker.toml
create mode 100644 _data/talks/2024-01-10-data-science-lecture-series.toml
create mode 100644 _data/talks/2024-01-31-pratik-soni.toml
create mode 100644 _data/talks/2024-03-13-swabha-swayamdipta.toml
create mode 100644 _data/talks/2024-04-10-ucds-lecture-series.toml
create mode 100644 _data/talks/2024-08-27-jeff-phillips.toml
create mode 100644 _data/talks/2024-09-03-esha-datta.toml
create mode 100644 _data/talks/2024-09-10-guanhong-tao.toml
create mode 100644 _data/talks/2024-09-17-aurora-clark.toml
create mode 100644 _data/talks/2024-09-24-rebecca-barter.toml
create mode 100644 _data/talks/2024-10-01-simon-brewer.toml
create mode 100644 _data/talks/2024-10-15-raghav-venkatraman.toml
create mode 100644 _data/talks/2024-10-29-vivek-gupta.toml
create mode 100644 _data/talks/2024-11-12-amir-abdullah.toml
create mode 100644 _data/talks/2024-11-19-hoaning-xue.toml
create mode 100644 _data/talks/2024-11-26-zhichao-xu.toml
create mode 100644 _data/talks/2025-01-17-fengjiao-wang.toml
create mode 100644 _data/talks/2025-01-24-data-science-ai-day.toml
create mode 100644 _data/talks/2025-02-07-omkar-bhalerao.toml
create mode 100644 _data/talks/2025-02-14-bao-wang.toml
create mode 100644 _data/talks/2025-02-21-chenglu-li.toml
create mode 100644 _data/talks/2025-02-28-bernardo-modenesi.toml
create mode 100644 _data/talks/2025-03-21-jeff-phillips.toml
create mode 100644 _data/talks/2025-03-28-seth-pettie.toml
create mode 100644 _data/talks/2025-04-04-sabyasachi-basu.toml
create mode 100644 _data/talks/2025-04-11-peter-jacobs.toml
create mode 100644 _data/talks/2025-04-18-tucker-hermans.toml
create mode 100644 _data/talks/2025-08-20-varun-shankar.toml
create mode 100644 _data/talks/2025-08-27-konstantin-genin.toml
create mode 100644 _data/talks/2025-09-03-daniel-brown.toml
create mode 100644 _data/talks/2025-09-10-bei-wang-phillips.toml
create mode 100644 _data/talks/2025-09-17-aditya-bhaskara.toml
create mode 100644 _data/talks/2025-09-24-kenneth-blake-vernon.toml
create mode 100644 _data/talks/2025-10-01-luis-garcia.toml
create mode 100644 _data/talks/2025-10-15-vineet-pandey.toml
create mode 100644 _data/talks/2025-10-22-kenneth-marino.toml
create mode 100644 _data/talks/2025-10-29-anna-fariha.toml
create mode 100644 _data/talks/2025-11-05-jenny-lin.toml
create mode 100644 _data/talks/2025-11-12-kyle-dawson-tyler-hagen.toml
create mode 100644 _data/talks/2025-11-19-vivek-srikumar.toml
create mode 100644 _data/talks/2025-12-03-data-visualization-101.toml
create mode 100644 _data/talks/2026-01-09-marina-kogan.toml
create mode 100644 _data/talks/2026-01-23-andrew-mcnutt.toml
create mode 100644 _data/talks/2026-02-06-makoto-kelp.toml
create mode 100644 _data/talks/2026-02-17-juliana-freire.toml
create mode 100644 _data/talks/2026-02-19-dinesh-manocha.toml
create mode 100644 _data/talks/2026-02-24-paul-parsons.toml
create mode 100644 _data/talks/2026-02-26-zezhong-wang.toml
create mode 100644 _data/talks/2026-02-27-kenny-marino.toml
create mode 100644 _data/talks/2026-03-02-bogdan-raita.toml
create mode 100644 _data/talks/2026-03-03-grace-guo.toml
create mode 100644 _data/talks/2026-03-05-josh-levine.toml
create mode 100644 _data/talks/2026-03-16-tenghao-huang.toml
create mode 100644 _data/talks/2026-03-20-kate-isaacs.toml
create mode 100644 _data/talks/2026-03-20-xueguang-ma.toml
create mode 100644 _data/talks/2026-03-23-benjie-wang.toml
create mode 100644 _data/talks/2026-03-25-dick-sadler.toml
create mode 100644 _data/talks/2026-03-26-bailing-lyu.toml
create mode 100644 _data/talks/2026-03-27-erdogan-kaya.toml
create mode 100644 _data/talks/2026-03-30-xiaoling-hu.toml
create mode 100644 _data/talks/2026-03-31-he-yin.toml
create mode 100644 _data/talks/2026-04-01-md-mostafijur-rahman.toml
create mode 100644 _data/talks/2026-04-06-qiang-ji.toml
create mode 100644 _data/talks/2026-04-07-si-chen.toml
create mode 100644 _data/talks/2026-04-08-fahim-faisal.toml
create mode 100644 _data/talks/2026-04-10-chase-neumann.toml
create mode 100644 _data/talks/2026-04-13-jihyun-rho.toml
create mode 100644 _data/talks/2026-04-14-sameer-honwad.toml
create mode 100644 _data/talks/2026-04-15-amirali-abdullah.toml
create mode 100644 _data/talks/2026-04-15-chengbin-deng.toml
create mode 100644 _data/talks/2026-04-15-shiqi-yu.toml
create mode 100644 _data/talks/2026-04-17-daniel-sieta.toml
create mode 100644 _data/talks/2026-08-28-warren-pettine.toml
create mode 100644 _data/talks/2026-09-04-george-vega-yon.toml
create mode 100644 _data/talks/_TEMPLATE.toml
create mode 100644 _layouts/talk.html
create mode 100644 _talks/2020-01-09-chris-musco.md
create mode 100644 _talks/2020-01-16-alexander-lex.md
create mode 100644 _talks/2020-01-23-harish-maringanti.md
create mode 100644 _talks/2020-01-30-qingyao-ai.md
create mode 100644 _talks/2020-02-06-gail-zasowski.md
create mode 100644 _talks/2020-02-20-bei-wang.md
create mode 100644 _talks/2020-02-27-taylor-sparks.md
create mode 100644 _talks/2020-03-05-john-horel.md
create mode 100644 _talks/2020-08-28-vivek-gupta.md
create mode 100644 _talks/2020-09-04-dheeraj-mekala.md
create mode 100644 _talks/2020-09-11-parthe-pandit.md
create mode 100644 _talks/2020-09-18-vishnu-lokhande.md
create mode 100644 _talks/2020-10-02-ellen-riloff.md
create mode 100644 _talks/2020-10-09-swaroop-mishra.md
create mode 100644 _talks/2020-10-16-varun-gangal.md
create mode 100644 _talks/2020-10-23-nancy-wang.md
create mode 100644 _talks/2020-10-30-alberto-cairo.md
create mode 100644 _talks/2020-10-30-daniel-scharfstein.md
create mode 100644 _talks/2020-11-06-bhargavi-paranjape.md
create mode 100644 _talks/2020-11-13-akanksha-atrey.md
create mode 100644 _talks/2020-11-20-akhil-arora.md
create mode 100644 _talks/2020-12-04-danish-pruthi.md
create mode 100644 _talks/2021-01-22-grad-student-spotlights.md
create mode 100644 _talks/2021-01-29-sanghamitra-dutta.md
create mode 100644 _talks/2021-02-05-michal-moshkovitz.md
create mode 100644 _talks/2021-02-12-fritz-lekschas.md
create mode 100644 _talks/2021-02-19-samson-zhou.md
create mode 100644 _talks/2021-02-26-rajesh-jayaram.md
create mode 100644 _talks/2021-03-12-yaoqing-yang.md
create mode 100644 _talks/2021-03-19-vivek-gupta.md
create mode 100644 _talks/2021-04-09-emily-beth-wall.md
create mode 100644 _talks/2021-04-16-arun-sai-suggala.md
create mode 100644 _talks/2021-04-23-lizzie-kumar.md
create mode 100644 _talks/2021-08-27-jeff-phillips.md
create mode 100644 _talks/2021-09-03-shandian-zhe.md
create mode 100644 _talks/2021-09-10-pierre-lermusiaux.md
create mode 100644 _talks/2021-09-17-ross-whitaker.md
create mode 100644 _talks/2021-09-24-c-seshadhri.md
create mode 100644 _talks/2021-10-01-tony-h-grubesic.md
create mode 100644 _talks/2021-10-08-anna-little.md
create mode 100644 _talks/2021-10-22-erin-wolf-chambers.md
create mode 100644 _talks/2021-10-29-bao-wang.md
create mode 100644 _talks/2021-11-05-marina-kogan.md
create mode 100644 _talks/2021-11-12-mikhail-belkin.md
create mode 100644 _talks/2021-11-19-michael-yeh.md
create mode 100644 _talks/2021-12-03-julia-silge.md
create mode 100644 _talks/2021-12-10-sameer-singh.md
create mode 100644 _talks/2022-01-14-yi-zhou.md
create mode 100644 _talks/2022-01-21-chinmay-hedge.md
create mode 100644 _talks/2022-01-28-swaroop-mishra.md
create mode 100644 _talks/2022-02-04-tuhin-chakrabarty.md
create mode 100644 _talks/2022-02-11-vaggos-chatziafratis.md
create mode 100644 _talks/2022-02-18-jeff-phillips.md
create mode 100644 _talks/2022-02-25-sunipa-dev.md
create mode 100644 _talks/2022-03-04-anirudh-goyal-of-montreal.md
create mode 100644 _talks/2022-03-25-chad-topaz-jude-higdon.md
create mode 100644 _talks/2022-04-01-debanjan-mahata.md
create mode 100644 _talks/2022-04-08-tao-li.md
create mode 100644 _talks/2022-04-15-abhinav-kumar.md
create mode 100644 _talks/2022-04-22-khyati-chandu.md
create mode 100644 _talks/2022-04-29-james-brundage.md
create mode 100644 _talks/2022-08-24-bei-wang-phillips.md
create mode 100644 _talks/2022-08-31-casey-greene.md
create mode 100644 _talks/2022-09-07-kevin-moon.md
create mode 100644 _talks/2022-09-14-elliot-smith.md
create mode 100644 _talks/2022-09-21-jes-ford.md
create mode 100644 _talks/2022-09-28-prashant-pandey.md
create mode 100644 _talks/2022-10-05-jessica-shi.md
create mode 100644 _talks/2022-10-19-jie-zhang.md
create mode 100644 _talks/2022-11-02-alex-chin.md
create mode 100644 _talks/2022-11-09-shireen-elhabian.md
create mode 100644 _talks/2022-11-16-bernadette-stolz.md
create mode 100644 _talks/2022-11-30-aaron-quinlan.md
create mode 100644 _talks/2022-12-07-tao-yang.md
create mode 100644 _talks/2023-01-11-george-vega-yon.md
create mode 100644 _talks/2023-01-18-echo-warner.md
create mode 100644 _talks/2023-01-25-shweta-jain.md
create mode 100644 _talks/2023-02-01-ana-marsovic.md
create mode 100644 _talks/2023-02-08-aaron-clauset.md
create mode 100644 _talks/2023-02-15-orly-alter.md
create mode 100644 _talks/2023-02-22-vivek-gupta.md
create mode 100644 _talks/2023-03-01-emily-hadley.md
create mode 100644 _talks/2023-03-15-nate-veldt.md
create mode 100644 _talks/2023-03-22-bailey-fosdick.md
create mode 100644 _talks/2023-03-29-justin-baker.md
create mode 100644 _talks/2023-04-05-yao-yaun-mao.md
create mode 100644 _talks/2023-04-12-titus-brown.md
create mode 100644 _talks/2023-04-19-sumana-basu.md
create mode 100644 _talks/2023-05-30-bodhisattwa-majumder.md
create mode 100644 _talks/2023-08-23-data-science-lecture-series.md
create mode 100644 _talks/2023-08-30-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-08-30-data-science-lecture-series.md
create mode 100644 _talks/2023-09-06-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-09-13-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-09-20-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-09-25-sandia-information-session.md
create mode 100644 _talks/2023-09-27-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-10-04-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-10-18-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-10-25-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-11-01-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-11-22-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-11-29-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-12-06-data-science-lecture-series-speaker.md
create mode 100644 _talks/2023-12-13-data-science-lecture-series-speaker.md
create mode 100644 _talks/2024-01-10-data-science-lecture-series.md
create mode 100644 _talks/2024-01-31-pratik-soni.md
create mode 100644 _talks/2024-03-13-swabha-swayamdipta.md
create mode 100644 _talks/2024-04-10-ucds-lecture-series.md
create mode 100644 _talks/2024-08-27-jeff-phillips.md
create mode 100644 _talks/2024-09-03-esha-datta.md
create mode 100644 _talks/2024-09-10-guanhong-tao.md
create mode 100644 _talks/2024-09-17-aurora-clark.md
create mode 100644 _talks/2024-09-24-rebecca-barter.md
create mode 100644 _talks/2024-10-01-simon-brewer.md
create mode 100644 _talks/2024-10-15-raghav-venkatraman.md
create mode 100644 _talks/2024-10-29-vivek-gupta.md
create mode 100644 _talks/2024-11-12-amir-abdullah.md
create mode 100644 _talks/2024-11-19-hoaning-xue.md
create mode 100644 _talks/2024-11-26-zhichao-xu.md
create mode 100644 _talks/2025-01-17-fengjiao-wang.md
create mode 100644 _talks/2025-01-24-data-science-ai-day.md
create mode 100644 _talks/2025-02-07-omkar-bhalerao.md
create mode 100644 _talks/2025-02-14-bao-wang.md
create mode 100644 _talks/2025-02-21-chenglu-li.md
create mode 100644 _talks/2025-02-28-bernardo-modenesi.md
create mode 100644 _talks/2025-03-21-jeff-phillips.md
create mode 100644 _talks/2025-03-28-seth-pettie.md
create mode 100644 _talks/2025-04-04-sabyasachi-basu.md
create mode 100644 _talks/2025-04-11-peter-jacobs.md
create mode 100644 _talks/2025-04-18-tucker-hermans.md
create mode 100644 _talks/2025-08-20-varun-shankar.md
create mode 100644 _talks/2025-08-27-konstantin-genin.md
create mode 100644 _talks/2025-09-03-daniel-brown.md
create mode 100644 _talks/2025-09-10-bei-wang-phillips.md
create mode 100644 _talks/2025-09-17-aditya-bhaskara.md
create mode 100644 _talks/2025-09-24-kenneth-blake-vernon.md
create mode 100644 _talks/2025-10-01-luis-garcia.md
create mode 100644 _talks/2025-10-15-vineet-pandey.md
create mode 100644 _talks/2025-10-22-kenneth-marino.md
create mode 100644 _talks/2025-10-29-anna-fariha.md
create mode 100644 _talks/2025-11-05-jenny-lin.md
create mode 100644 _talks/2025-11-12-kyle-dawson-tyler-hagen.md
create mode 100644 _talks/2025-11-19-vivek-srikumar.md
create mode 100644 _talks/2025-12-03-data-visualization-101.md
create mode 100644 _talks/2026-01-09-marina-kogan.md
create mode 100644 _talks/2026-01-23-andrew-mcnutt.md
create mode 100644 _talks/2026-02-06-makoto-kelp.md
create mode 100644 _talks/2026-02-17-juliana-freire.md
create mode 100644 _talks/2026-02-19-dinesh-manocha.md
create mode 100644 _talks/2026-02-24-paul-parsons.md
create mode 100644 _talks/2026-02-26-zezhong-wang.md
create mode 100644 _talks/2026-02-27-kenny-marino.md
create mode 100644 _talks/2026-03-02-bogdan-raita.md
create mode 100644 _talks/2026-03-03-grace-guo.md
create mode 100644 _talks/2026-03-05-josh-levine.md
create mode 100644 _talks/2026-03-16-tenghao-huang.md
create mode 100644 _talks/2026-03-20-kate-isaacs.md
create mode 100644 _talks/2026-03-20-xueguang-ma.md
create mode 100644 _talks/2026-03-23-benjie-wang.md
create mode 100644 _talks/2026-03-25-dick-sadler.md
create mode 100644 _talks/2026-03-26-bailing-lyu.md
create mode 100644 _talks/2026-03-27-erdogan-kaya.md
create mode 100644 _talks/2026-03-30-xiaoling-hu.md
create mode 100644 _talks/2026-03-31-he-yin.md
create mode 100644 _talks/2026-04-01-md-mostafijur-rahman.md
create mode 100644 _talks/2026-04-06-qiang-ji.md
create mode 100644 _talks/2026-04-07-si-chen.md
create mode 100644 _talks/2026-04-08-fahim-faisal.md
create mode 100644 _talks/2026-04-10-chase-neumann.md
create mode 100644 _talks/2026-04-13-jihyun-rho.md
create mode 100644 _talks/2026-04-14-sameer-honwad.md
create mode 100644 _talks/2026-04-15-amirali-abdullah.md
create mode 100644 _talks/2026-04-15-chengbin-deng.md
create mode 100644 _talks/2026-04-15-shiqi-yu.md
create mode 100644 _talks/2026-04-17-daniel-sieta.md
create mode 100644 _talks/2026-08-28-warren-pettine.md
create mode 100644 _talks/2026-09-04-george-vega-yon.md
create mode 100644 scripts/__pycache__/import_calendar_talks.cpython-313.pyc
create mode 100644 scripts/generate_talks.py
create mode 100644 scripts/import_calendar_talks.py
create mode 100644 talks.md
diff --git a/.github/workflows/talks.yml b/.github/workflows/talks.yml
new file mode 100644
index 0000000..db664b2
--- /dev/null
+++ b/.github/workflows/talks.yml
@@ -0,0 +1,26 @@
+name: Talk pages
+
+on:
+ pull_request:
+ branches:
+ - main
+ - master
+ push:
+ branches:
+ - main
+ - master
+
+jobs:
+ check:
+ runs-on: ubuntu-latest
+ steps:
+ - name: Checkout repository
+ uses: actions/checkout@v4
+
+ - name: Set up Python
+ uses: actions/setup-python@v5
+ with:
+ python-version: "3.11"
+
+ - name: Check that _talks/ matches _data/talks/
+ run: python3 scripts/generate_talks.py --check
diff --git a/Makefile b/Makefile
index 88bf9f3..56dac72 100644
--- a/Makefile
+++ b/Makefile
@@ -1,4 +1,4 @@
-.PHONY: help setup serve build clean
+.PHONY: help setup serve build clean talks talks-check import-talks
.DEFAULT_GOAL := help
@@ -21,3 +21,12 @@ build: ## Build the site for production
clean: ## Remove generated site files
rm -rf _site .jekyll-cache
+
+talks: ## Generate the talk pages in _talks/ from _data/talks/*.toml
+ python3 scripts/generate_talks.py
+
+talks-check: ## Verify the talk pages match _data/talks/*.toml (used by CI)
+ python3 scripts/generate_talks.py --check
+
+import-talks: ## Seed new talk records from the seminar Google Calendar
+ python3 scripts/import_calendar_talks.py
diff --git a/README.md b/README.md
index c6a55f1..0ac56d3 100644
--- a/README.md
+++ b/README.md
@@ -68,6 +68,77 @@ The available variables are:
> **_NOTE:_** The subfolders (affiliated, core, and leadership) under `_members` have no effects. They exist only for organizing these files. To show member under a role, set the role variable in its .md file with a right value.
+### Talks
+
+Every talk in the Data Science & AI Lecture Series has its own page on the site
+(for example `/talks/2026-09-04-george-vega-yon/`). Those pages are **generated**;
+the source of truth for each talk is a TOML file in `_data/talks/`.
+
+To add a talk:
+
+1. Copy `_data/talks/_TEMPLATE.toml` to `_data/talks/YYYY-MM-DD-speaker-name.toml`
+ and fill it in. The file name becomes the page URL, so keep the
+ `date-speaker` shape.
+2. Run `make talks` (equivalently `python3 scripts/generate_talks.py`). This
+ writes `_talks/YYYY-MM-DD-speaker-name.md`.
+3. Commit **both** the TOML file and the generated markdown, and open a pull
+ request. CI (`.github/workflows/talks.yml`) re-runs the generator and fails if
+ the two are out of sync.
+
+Updating a talk later (adding slides, a recording link, or a speaker photo) is
+the same loop: edit the TOML, run `make talks`, commit both files.
+
+The generated pages are wired into the site automatically:
+
+* `seminar.md` shows the next talk and the next few upcoming talks.
+* `/talks/` (`talks.md`) lists everything, upcoming first, then past talks by year.
+
+A minimal record looks like this:
+
+```toml
+[talk]
+title = "Data Science of Tracking Measles in Utah"
+date = 2026-09-04
+start_time = "13:30"
+end_time = "14:30"
+location = "WEB 2250"
+zoom = "https://utah.zoom.us/j/85983626630"
+abstract = """
+What the talk is about, in markdown.
+"""
+
+[[speakers]]
+name = "George Vega Yon"
+affiliation = "Division of Epidemiology, University of Utah"
+website = "https://ggvy.cl"
+bio = """
+A short bio, in markdown.
+"""
+```
+
+Only `[talk] title`, `[talk] date`, and one `[[speakers]] name` are required;
+everything else is optional and simply omitted from the page when empty. Add a
+`[[speakers]]` block per speaker for joint talks, and set `canceled = true`
+rather than deleting a record for a talk that did not happen.
+
+Why TOML plus a generator? GitHub Pages builds Jekyll in safe mode, so custom
+plugins (which could read TOML at build time) are not available: the pages have
+to be generated ahead of time and committed.
+
+#### Seeding records from the Google Calendar
+
+`scripts/import_calendar_talks.py` reads the series' public Google Calendar and
+writes TOML records for talks that do not have one yet:
+
+```shell
+make import-talks # or: python3 scripts/import_calendar_talks.py
+```
+
+Calendar descriptions are free-form, so this is best effort — imported records
+are marked `needs_review = true` under `[meta]` and should be checked (titles,
+affiliations, and abstracts especially) before they are considered final. It
+never overwrites an existing file unless you pass `--overwrite`.
+
### progrmas
Add/delete/edit .md files in `_progrmas` folder to add/delete/edit members.
diff --git a/_config.yml b/_config.yml
index 0d6a3ad..f41c59a 100644
--- a/_config.yml
+++ b/_config.yml
@@ -6,11 +6,18 @@ encoding: utf-8
# Disable the default primer theme from github-pages
theme: null
+# Talk pages are dated in the future by design (that is what "upcoming" means),
+# and Jekyll hides future-dated documents unless this is set.
+future: true
+
collections:
programs:
output: false
members:
output: false
+ # One page per talk, generated from _data/talks/*.toml by scripts/generate_talks.py.
+ talks:
+ output: true
defaults:
-
@@ -34,6 +41,8 @@ exclude:
- README.md
- Gemfile
- Gemfile.lock
+ - Makefile
+ - scripts
# - node_modules
# - vendor/bundle/
# - vendor/cache/
diff --git a/_data/talks/2020-01-09-chris-musco.toml b/_data/talks/2020-01-09-chris-musco.toml
new file mode 100644
index 0000000..a2bcef3
--- /dev/null
+++ b/_data/talks/2020-01-09-chris-musco.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Randomized FunctionalAnalysis"
+date = 2020-01-09
+start_time = "12:15"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Sketching and subsampling are central algorithmic tools in scaling statisticalmethods to very large datasets. These techniques seek to quickly compress datadown to a compact set of informative features or examples, which can then beprocessed in place of the original data, at much lower computational cost. Thecentral question of this talk is what sketching methods can teach us abouteffective machine learning and data analysis in the small data regime. Inapplications where high quality data examples remain a rare luxury, can ourknowledge of data sketching guide more efficient initial data collection?
+
+We study this problem by focusing specifically on techniques for large matrixcomputations. In the field of randomized numerical linear algebra, importancesampling has emerged as an important tool for dataset compression. Statisticalleverage scores and related measures are used to judge the importance of rowsor columns in a matrix, which are then non-uniformly subsampled, leading tofaster algorithms for regression, low-rank approximation, kernel methods, andmany other data problems.
+
+I will introduce a simple generalization of leverage score sampling to infinitedimensional linear operators and show the potential of this generalization indeveloping sample efficient algorithms for small data applications.Specifically, I will survey a number of recent results on robust polynomialcurve fitting, bandlimited function interpolation, off-grid sparse Fouriertransforms, and sample efficient covariance estimation. I will illustrateconnections between these new results and classical tools in approximationtheory and signal processing, and will discuss several open researchdirections.
+"""
+
+[[speakers]]
+name = "Chris Musco"
+affiliation = "NYU"
+website = "https://www.chrismusco.com"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "42t5o0blhd85v46a8f5h2s2dba@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-01-16-alexander-lex.toml b/_data/talks/2020-01-16-alexander-lex.toml
new file mode 100644
index 0000000..e55fe84
--- /dev/null
+++ b/_data/talks/2020-01-16-alexander-lex.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Literate Visualization: Making Visual Analysis Sessions Reproducible and Reusable"
+date = 2020-01-16
+start_time = "12:15"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Interactive visualization is an important part of the data science process. It enables analysts to directly interact with the data, exploring it with minimal effort. Unlike code, however, an interactive visualization session is ephemeral and can’t be easily shared, revisited, or reused. Computational notebooks, such as Jupyter Notebooks, R Markdown, or Observable are a perfect match for many data science applications. They are also the most popular embodiment of Knuth’s “Literate Programming”, where the logic of a program is explained in natural language, figures, and equations. In this talk, I will sketch approaches to “Literate Visualization”. I will show how we can leverage provenance data of an analysis session to create well-documented and annotated visualization stories that enable reproducibility and sharing. I will also introduce early work on semi-automatically inferring mid-level analysis goals, which allows us to understand the analysis process at a higher level. Understanding analysis goals enables us to speed up interactions and even re-used visual analysis processes."
+
+[[speakers]]
+name = "Alexander Lex"
+affiliation = ""
+website = "https://vdl.sci.utah.edu/team/lex/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "42t5o0blhd85v46a8f5h2s2dba_R20200116T191500@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-01-23-harish-maringanti.toml b/_data/talks/2020-01-23-harish-maringanti.toml
new file mode 100644
index 0000000..964243f
--- /dev/null
+++ b/_data/talks/2020-01-23-harish-maringanti.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Data Science projects in Marriott Library"
+date = 2020-01-23
+start_time = "12:15"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+At research intensive universities, libraries have traditionally supported data science activities in various ways including acquiring datasets that researchers need, hosting workshops and training sessions on data science tools, and offering data support tools for creation of persistent identifiers(dois), etc. At Marriott Library, in addition to supporting data science programs on campus, we have embarked on a suite of data science projects to add value to our culturally-rich collections. Our efforts are focused on enriching our collection data, and making this collection data available for computational use (collections as data [1]) so that developers, scientists, and digital humanists can programmatically interact with the data in myriad ways and undertake projects related to data mining & text analysis, advanced visualizations, and geospatial analysis. In this presentation, I will talk about two specific projects - Utah Digital Newspapers [2] and machine learning meets archives [3] - to highlight these efforts.
+
+Utah Digital Newspapers (UDN): Marriott Library was an early pioneer in digitizing newspapers and making the content available to historians, researchers, and lifelong learners. UDN program has been operating since 2002 and is recognized as one of the leaders in newspaper digitization in the United States. We have continued to partner with universities, colleges, state agencies, county and city libraries, and other agencies to digitize, deliver, and archive historical newspaper collections; As of 2019, UDN has well over 22.5 million newspaper articles and 3.5 million pages in the repository platform. In this presentation, we will talk about the importance of looking at collections as data, our API work with UDN and demonstrate the usefulness of this approach with specific examples.
+
+Machine learning meets archives: Metadata is the bedrock of library archives and Digital Library systems, as it helps in users discovering the unique content in various collections housed in digital libraries. But creating metadata is a time-intensive process. We are working with machine learning algorithms to generate descriptive metadata for digital images. I will share the results of our work, and also lessons learned from working with digital library data.
+"""
+
+[[speakers]]
+name = "Harish Maringanti"
+affiliation = "Associate Dean for IT & Digital Library Services, Marriott Library"
+website = "https://collectionsasdata.github.io/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "42t5o0blhd85v46a8f5h2s2dba_R20200116T191500@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-01-30-qingyao-ai.toml b/_data/talks/2020-01-30-qingyao-ai.toml
new file mode 100644
index 0000000..af830bd
--- /dev/null
+++ b/_data/talks/2020-01-30-qingyao-ai.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Unbiased Learning to Rank: Theory and Practice"
+date = 2020-01-30
+start_time = "12:15"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Implicit feedback (e.g., user clicks) is an important source of data for modern search engines. While heavily biased, it is cheap to collect and particularly useful for user-centric retrieval applications such as search ranking. Therefore, a learning-to-rank algorithm that can effectively learn from implicit user feedback without affected by its inherent biases could fundamentally change the design of ranking systems and significantly improve the quality of modern search engines. To develop an unbiased learning-to-rank system with biased feedback, previous studies have focused on constructing probabilistic graphical models (e.g., click models) with user behavior hypothesis to extract and train ranking systems with unbiased relevance signals. Recently, a novel counterfactual learning framework that estimates and adopts examination propensity for unbiased learning to rank has attracted much attention. In this talk, we aim to provide an overview of the fundamental mechanism for unbiased learning to rank. We describe the theory behind existing frameworks, and give instructions on how to conduct unbiased learning to rank in practice."
+
+[[speakers]]
+name = "Qingyao Ai"
+affiliation = "Utah SoC"
+website = "http://ir.aiqingyao.org/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "42t5o0blhd85v46a8f5h2s2dba_R20200116T191500@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-02-06-gail-zasowski.toml b/_data/talks/2020-02-06-gail-zasowski.toml
new file mode 100644
index 0000000..c7155ca
--- /dev/null
+++ b/_data/talks/2020-02-06-gail-zasowski.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Big Data, Big Universe: Data-Driven Discoveries in Astrophysics"
+date = 2020-02-06
+start_time = "12:15"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "The stars in the night sky have inspired questions about our place in the Universe throughout history. The development of telescopes showed us that the stars visible to the naked eye are but a tiny fraction of their vast numbers within our own Galaxy, and revealed energy signatures invisible to the human senses. We now know that there are billions of stars in our galaxy, billions of galaxies in our Universe, and nearly 14 billion years of cosmic evolution that have led to where and what we are today. As the volume of astronomical data grows at an ever quickening rate, new discoveries increasingly come from careful mining and analysis of existing data, often used in unforeseen ways. This talk will describe some of the major unanswered questions in astrophysics, and how new data-driven analysis techniques are uncovering new insights into solving them."
+
+[[speakers]]
+name = "Gail Zasowski"
+affiliation = "Utah Physics & Astronomy"
+website = "http://www.physics.utah.edu/~zasowski/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "42t5o0blhd85v46a8f5h2s2dba_R20200116T191500@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-02-20-bei-wang.toml b/_data/talks/2020-02-20-bei-wang.toml
new file mode 100644
index 0000000..8d21f2d
--- /dev/null
+++ b/_data/talks/2020-02-20-bei-wang.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "TopoAct: Exploring the Shape of Activations in Deep Learning"
+date = 2020-02-20
+start_time = "12:15"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Deep neural networks such as GoogLeNet and ResNet have achieved superhuman performance in tasks like image classification. To understand how such superior performance is achieved, we can probe a trained deep neural network by studying neuron activations, that is, combinations of neuron firings, at any layer of the network in response to a particular input. With a large set of input images, we aim to obtain a global view of what neurons detect by studying their activations. We ask the following questions: What is the shape of the space of activations? That is, what is the organizational principle behind neuron activations, and how are the activations related within a layer and across layers? Applying tools from topological data analysis, we present TopoAct, a visual exploration system used to study topological summaries of activation vectors for a single layer as well as the evolution of such summaries across multiple layers. We present visual exploration scenarios using TopoAct that provide valuable insights towards learned representations of an image classifier."
+
+[[speakers]]
+name = "Bei Wang"
+affiliation = "Utah SoC, SCI"
+website = "http://www.sci.utah.edu/~beiwang/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "42t5o0blhd85v46a8f5h2s2dba_R20200116T191500@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-02-27-taylor-sparks.toml b/_data/talks/2020-02-27-taylor-sparks.toml
new file mode 100644
index 0000000..c213f44
--- /dev/null
+++ b/_data/talks/2020-02-27-taylor-sparks.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "New Algorithms, Descriptors, and Machine Learning Techniques Tailored to the Challenges of Materials Informatics"
+date = 2020-02-27
+start_time = "12:15"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+New materials are required to address many of the energy, environmental, and technological needs of the present and future. Materials Informatics, or the application of data science techniques to solve materials research challenges, is expected to play a key role in materials development and discovery given the infinite palette available for new materials. Interestingly, the requirements and tasks of Materials Informatics do not always overlap with general machine learning. Therefore, adopting existing data science tools including visualization, algorithms, featurization schemas etc may not provide the ideal outcomes for the specific needs of Materials Informatics.
+
+In this talk I will focus on some of our recent work to bring tailored data science approaches to actual materials research problems. Specifically, I will introduce how we have created an attention-based neural network architecture for the prediction of materials properties. We show that this novel algorithm outperforms other methods in the absence of chemical information, even when the statistical and ensemble learning techniques are given domain-specific chemical knowledge about the materials.
+
+Dr. Sparks is an Associate Professor and Associate Chair of the Materials Science and Engineering Department at the University of Utah. He is originally from Utah and an alumni of the department he now teaches in. Before graduate school he worked at Ceramatec Inc. He did his MS in Materials at UCSB and his PhD in Applied Physics at Harvard University in David Clarke’s laboratory and then did a postdoc with Ram Seshadri in the Materials Research Laboratory at UCSB. His current research centers on the discovery, synthesis, characterization, and properties of new materials for energy applications. He is a pioneer in the emerging field of materials informatics whereby big data, data mining, and machine learning are leveraged to solve challenges in materials science. He also hosts a podcast entitled “Materialism” where he discusses the past, present, and future of Materials Science.
+"""
+
+[[speakers]]
+name = "Taylor Sparks"
+affiliation = "Utah Materials Science & Engineering"
+website = "https://my.eng.utah.edu/~sparks/"
+photo = ""
+bio = "Dr. Sparks is an Associate Professor and Associate Chair of the Materials Science and Engineering Department at the University of Utah. He is originally from Utah and an alumni of the department he now teaches in. Before graduate school he worked at Ceramatec Inc. He did his MS in Materials at UCSB and his PhD in Applied Physics at Harvard University in David Clarke’s laboratory and then did a postdoc with Ram Seshadri in the Materials Research Laboratory at UCSB. His current research centers on the discovery, synthesis, characterization, and properties of new materials for energy applications. He is a pioneer in the emerging field of materials informatics whereby big data, data mining, and machine learning are leveraged to solve challenges in materials science. He also hosts a podcast entitled “Materialism” where he discusses the past, present, and future of Materials Science."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "42t5o0blhd85v46a8f5h2s2dba_R20200116T191500@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-03-05-john-horel.toml b/_data/talks/2020-03-05-john-horel.toml
new file mode 100644
index 0000000..62f402b
--- /dev/null
+++ b/_data/talks/2020-03-05-john-horel.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "The Promise and Perils of Big Data in the Cloud - Examples from the Atmospheric Sciences"
+date = 2020-03-05
+start_time = "12:15"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+From the inception of numerical weather prediction in the 1950’s, atmospheric scientists have stretched the envelope on the hardware and procedures available to write, store, and use data on mass storage systems. The opportunities now to rely on cloud resources to process, access, and disseminate environmental data offer improved capabilities for data science applications in the atmospheric sciences but also introduce complexities for university researchers.
+
+The Big Data Project of the National Oceanographic and Atmospheric Administration is assessing the potential benefits of storing in the cloud observations and weather and climate model output that are generating petabytes of data daily. Retrieving, archiving, analyzing, and disseminating only a fraction of this environmental information has required us to move beyond computational approaches traditionally used within the atmospheric science community. For example, hundreds of users rely on a 140+ Tbyte archive we maintain of High Resolution Rapid Refresh (HRRR) model output on the Pando system of the University’s Center for High Performance Computing. Computing resources available nationwide as part of the Open Science Grid- a high-throughput computing resource- have been used to analyze wildland fire events. Data analytical methods are being explored that rely on Zarr data compression to provide classes and functions for working with N-dimensional arrays.
+"""
+
+[[speakers]]
+name = "John Horel"
+affiliation = "Professor, Chair, Department of Atmospheric Sciences"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "42t5o0blhd85v46a8f5h2s2dba_R20200116T191500@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-08-28-vivek-gupta.toml b/_data/talks/2020-08-28-vivek-gupta.toml
new file mode 100644
index 0000000..d3b4b38
--- /dev/null
+++ b/_data/talks/2020-08-28-vivek-gupta.toml
@@ -0,0 +1,32 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09 (https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+date = 2020-08-28
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = "https://www.youtube.com/redirect?q=https%3A%2F%2Fvgupta123.github.io%2F&v=YhfU1BON8EI&event=video_description&redir_token=QUFFLUhqbFVhUUxOS2RDSHBPYWJPMjNqaFpoalpzY2E3d3xBQ3Jtc0tuWlZRQ0tfSTdPNkxsTzFuWDN2Y3lvdXdwRXJrOTJ6bGI3NXhrc2x4QVYxalNTTUVXWE10V3VncklsZ2Q1RnltN1VaWkpsek9OY0EtdWwtQWZNVHR0Q2k4Rng3eVE4N3FWYTk5dUtETV9aZXRSaGlEcw%3D%3D"
+canceled = false
+abstract = """
+Vivek Gupta (Utah School of Computing)
+(https://vgupta123.github.io/ (https://www.youtube.com/redirect?q=https%3A%2F%2Fvgupta123.github.io%2F&v=YhfU1BON8EI&event=video_description&redir_token=QUFFLUhqbFVhUUxOS2RDSHBPYWJPMjNqaFpoalpzY2E3d3xBQ3Jtc0tuWlZRQ0tfSTdPNkxsTzFuWDN2Y3lvdXdwRXJrOTJ6bGI3NXhrc2x4QVYxalNTTUVXWE10V3VncklsZ2Q1RnltN1VaWkpsek9OY0EtdWwtQWZNVHR0Q2k4Rng3eVE4N3FWYTk5dUtETV9aZXRSaGlEcw%3D%3D))
+
+Experience of the everyday language indicates the use of complicated reasonings both for people and the AI systems. Natural Language Inference (NLI) is the process of reasoning about inferential relationships, meaning to establish whether a hypothesis is a true (entailment), false (contradiction), or undetermined (neutral) given a premise. Previous works have generated inference corpora, such as the SNLI and the MNLI, which comprise only unstructured representations of text in the form of sentences in which relationships between words are explicitly expressed, and often need information extraction. However, text can also occur universally in other structured forms like tables, graphs, and databases.Building upon previous work on large-scale datasets for inference, we introduce a new dataset called INFOTABS, comprising of human-written textual hypotheses based on premises that are tables extracted from Wikipedia info-boxes. Our analysis shows that the semi-structured, multi-domain, and heterogeneous nature of the premises admits complex, multi-faceted reasoning. Experiments reveal that, while human annotators agree on the relationships between a table-hypothesis pair, several standard modeling strategies are unsuccessful at the task, suggesting that reasoning about tables can pose a new modeling challenge. For more details on InfoTabS visit http://infotabs.github.io (http://infotabs.github.io/)
+"""
+
+[[speakers]]
+name = "Vivek Gupta"
+affiliation = "UoU"
+website = "https://vgupta123.github.io/"
+photo = ""
+bio = "Vivek is a Ph.D. student at the School of Computing, University of Utah. Previously, he was working as a Research Fellow in Microsoft Research Lab, India, in Machine Learning and Natural Language Processing group. He graduated as a dual degree student in the Department of Computer Science and Engineering at IIT Kanpur in 2016. He is broadly interested in research in the field of Machine Learning and Natural Language Processing. To know more about his current research interest, you can visit https://vgupta123.github.io/ (https://www.youtube.com/redirect?q=https%3A%2F%2Fvgupta123.github.io%2F&v=YhfU1BON8EI&event=video_description&redir_token=QUFFLUhqbFVhUUxOS2RDSHBPYWJPMjNqaFpoalpzY2E3d3xBQ3Jtc0tuWlZRQ0tfSTdPNkxsTzFuWDN2Y3lvdXdwRXJrOTJ6bGI3NXhrc2x4QVYxalNTTUVXWE10V3VncklsZ2Q1RnltN1VaWkpsek9OY0EtdWwtQWZNVHR0Q2k4Rng3eVE4N3FWYTk5dUtETV9aZXRSaGlEcw%3D%3D)"
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-09-04-dheeraj-mekala.toml b/_data/talks/2020-09-04-dheeraj-mekala.toml
new file mode 100644
index 0000000..5a50820
--- /dev/null
+++ b/_data/talks/2020-09-04-dheeraj-mekala.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Contextualized Weak Supervision for Text Classification"
+date = 2020-09-04
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Weakly supervised text classification based on a few user-provided seed words has recently attracted much attention from researchers. Existing methods mainly generate pseudo-labels in a context-free manner (e.g., string matching), therefore, the ambiguous, context-dependent nature of human language has been long overlooked. In this paper, we propose a novel framework ConWea, providing contextualized weak supervision for text classification. Specifically, we leverage contextualized representations of word occurrences and seed word information to automatically differentiate multiple interpretations of the same word, and thus create a contextualized corpus. This contextualized corpus is further utilized to train the classifier and expand seed words in an iterative manner. This process not only adds new contextualized, highly label-indicative keywords but also disambiguates initial seed words, making our weak supervision fully contextualized. Extensive experiments and case studies on real-world datasets demonstrate the necessity and significant advantages of using contextualized weak supervision, especially when the class labels are fine-grained."
+
+[[speakers]]
+name = "Dheeraj Mekala"
+affiliation = "UCSD"
+website = ""
+photo = ""
+bio = "I am a Master’s student in the Computer Science department at the University of California, San Diego working with Prof. Jingbo Shang. I am broadly interested in Natural Language Processing, Text Mining, and Graph Mining. My current research revolves around developing principled data-driven approaches with minimal human effort. Specifically, there are huge amounts of unlabeled data available on the Internet and I try to leverage them to minimize the human effort. I completed my Bachelor of Technology in Computer Science And Engineering from the Indian Institute of Technology, Kanpur in 2017, where I worked with Prof. Harish Karnick, Prof. Purushottam Kar on Hierarchical Classification and Text Document Representation. I worked as a Data Scientist and Product Engineer at Sprinklr for 2 years and I interned at Microsoft India in the summer of 2016."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "5ddq64pkus241fvkcb7qk6orqj@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-09-11-parthe-pandit.toml b/_data/talks/2020-09-11-parthe-pandit.toml
new file mode 100644
index 0000000..5f81776
--- /dev/null
+++ b/_data/talks/2020-09-11-parthe-pandit.toml
@@ -0,0 +1,38 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Characterizing the asymptotic performance of inverse problems over Deep Networks"
+date = 2020-09-11
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+At the heart of Machine Learning lies the question of generalizability of learned models over previously unseen data. While over-parameterized models based on neural networks are now ubiquitous in machine learning applications, our understanding of their generalization capabilities remains incomplete. This task is made harder by the non-convexity of the underlying estimation/learning problems.
+
+The solutions to some of these estimation problems can be analyzed using a class of algorithms called Approximate Message Passing, even in the presence of the non-convexity. The dynamics of this algorithm follow a simplified macroscopic description often called the State Evolution. This analytical tool allows us to provide some insights into two broad classes of problems related to Neural Networks:
+
+1. Generalization error in 1 and 2-layer Networks
+2. Image Reconstruction error with Deep Image Priors
+"""
+
+[[speakers]]
+name = "Parthe Pandit"
+affiliation = "UCLA"
+website = "https://parthe.github.io/"
+photo = ""
+bio = """
+Parthe is a 5th year Ph.D. candidate at UCLA in ECE and an MS candidate in Statistics, where he is advised by Allie K. Fletcher, Sundeep Rangan, and Arash A. Amini. He is currently a research intern with Amazon AWS and Amazon Search, working on problems in conditional text generation. He is interested in analyzing optimization problems arising out of Machine Learning, Statistics, and Information Theory. His past research includes theoretical results in Network Economics, Mechanism Design, and Approximating NP-hard problems in graph theory.
+
+Parthe is the winner of the 2019 Jack K. Wolf Best Paper award, the Gurukrupa Foundation Fellowship, and scholarships from the J. N. Tata Endowment, and K. C. Mahindra foundation. In 2015, he received a B.Tech. and M.Tech. in Electrical Engineering at IIT Bombay, where he worked on Speech and Music Processing.
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-09-18-vishnu-lokhande.toml b/_data/talks/2020-09-18-vishnu-lokhande.toml
new file mode 100644
index 0000000..91351e6
--- /dev/null
+++ b/_data/talks/2020-09-18-vishnu-lokhande.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Optimization methods for imposing Fairness in Computer Vision Models"
+date = 2020-09-18
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "In this talk, we will study a mechanism to impose fairness in computer vision models concurrently while training the model and informed by standard fairness measures. While existing fairness based approaches in vision have largely relied on training adversarial modules together with the primary classification/regression task, in an effort to remove the influence of the protected attribute or variable, we will discuss how ideas based on well-known optimization concepts can provide a simpler alternative. In our proposed scheme, imposing fairness just requires specifying the protected attribute and utilizing our optimization routine. We will discuss experiments, that are interpretable, demonstrating that several fairness measures from the literature can be reliably imposed on standard vision tasks. We will also discuss technical analysis on the convergence guarantees of the said optimization routine."
+
+[[speakers]]
+name = "Vishnu Lokhande"
+affiliation = "University of Wisconsin-Madison"
+website = "https://lokhande-vishnu.github.io"
+photo = ""
+bio = "Vishnu Lokhande is a fourth year PhD student in Computer Sciences at the University of Wisconsin-Madison. He is currently completing his research internship at Microsoft Research in the Interactive Media Group. His research interests include Algorithmic Fairness, Semi-Supervised Learning, Constrained and Stochastic Optimization problems. Prior to his graduate studies, he received his bachelor’s in Electrical Engineering at the Indian Institute of Technology Kanpur."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-10-02-ellen-riloff.toml b/_data/talks/2020-10-02-ellen-riloff.toml
new file mode 100644
index 0000000..73371f2
--- /dev/null
+++ b/_data/talks/2020-10-02-ellen-riloff.toml
@@ -0,0 +1,57 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Identifying Affective Events and the Reasons for their Polarity"
+date = 2020-10-02
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Recognizing affective states is essential for narrative text
+understanding and for applications such as conversational dialogue,
+summarization, and sarcasm recognition. Many tools have been developed
+to recognize explicit expressions of sentiment, but affective states
+can also be inferred from events. This talk will focus on "affective
+events", which are generally desirable or undesirable experiences that
+implicitly suggest an affective state for the experiencer. For
+example, buying a home is usually desirable and associated with a
+positive affective state, but being laid off is undesirable and
+associated with a negative state. First, we will describe a weakly
+supervised learning method to induce affective events from a text
+corpus by optimizing for semantic consistency. Second, we aim to
+characterize affective events based on Human Needs Categories, which
+often explain people's motivations, goals, and desires. We will
+present a co-training model for Human Needs categorization that uses
+an event expression classifier and an event context classifier to
+learn from both labeled and unlabeled texts.
+"""
+
+[[speakers]]
+name = "Ellen Riloff"
+affiliation = "Utah Computer Science"
+website = "http://www.cs.utah.edu/~riloff/"
+photo = ""
+bio = """
+Ellen Riloff is a Professor in the School of Computing at the
+University of Utah. Her primary research area is natural language
+processing, with an emphasis on information extraction, affective text
+analysis, semantic class induction, and bootstrapping methods that
+learn from unannotated texts. Prof. Riloff has served as the General
+Chair for the EMNLP 2018 conference, Program Co-Chair for the NAACL
+HLT 2012 and CoNLL 2004 conferences, on the NAACL Executive Board for
+2004-2005 and 2017-2018, the Computational Linguistics Editorial
+Board, and the Transactions of the Association for Computational
+Linguistics (TACL) Editorial Board. In 2018, Prof. Riloff was named a
+Fellow of the Association for Computational Linguistics (ACL).
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-10-09-swaroop-mishra.toml b/_data/talks/2020-10-09-swaroop-mishra.toml
new file mode 100644
index 0000000..b03e63e
--- /dev/null
+++ b/_data/talks/2020-10-09-swaroop-mishra.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "DQI: Measuring Data Quality in NLP"
+date = 2020-10-09
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Neural language models have achieved human-level performance across several NLP datasets. However, recent studies have shown that these models are not truly learning the desired task; rather, their high performance is attributed to overfitting using spurious biases, which suggests that the capabilities of AI systems have been over-estimated. We introduce a generic formula for Data Quality Index (DQI) to help dataset creators create datasets with minimal unwanted biases. We propose a new data creation paradigm using DQI to create higher quality data. The data creation paradigm consists of several data visualizations to help data creators (i) understand the quality of data and (ii) visualize the impact of the created data instance on the overall quality. It also has a couple of automation methods to (i) assist data creators and (ii) make the model more robust to adversarial attacks. We use DQI along with these automation methods to renovate biased examples in SNLI. We show that models trained on the renovated SNLI dataset generalize better to out of distribution tasks. Renovation results in reduced model performance, exposing a large gap with respect to human performance. DQI systematically helps in creating harder benchmarks using active learning. Our work takes the process of dynamic dataset creation forward, wherein datasets evolve together with the evolving state of the art, therefore serving as a means of benchmarking the true progress of AI. Finally, we also show that DQI helps in pruning a dataset without compromising IID and OOD performance significantly."
+
+[[speakers]]
+name = "Swaroop Mishra"
+affiliation = "Arizona State University"
+website = "https://scholar.google.com/citations?user=-7LK2SwAAAAJ&hl=en"
+photo = ""
+bio = "Swaroop is currently a 2nd year PhD student at ASU. Prior to this, he received an M.S. degree from IIT Kanpur in 2016, post which he worked as a software engineer at MathWorks for 2 years, and as a Technical Consultant in the Ministry of Electronics and Information Technology, Govt. of India for a year."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-10-16-varun-gangal.toml b/_data/talks/2020-10-16-varun-gangal.toml
new file mode 100644
index 0000000..6acc717
--- /dev/null
+++ b/_data/talks/2020-10-16-varun-gangal.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Examining Extra Sentential Abilities of Contextual Embeddings"
+date = 2020-10-16
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "In the first third of our talk, we try to understand what and how much does BERT already know about event arguments (including cross-sentence ones)?. We observe that BERT's attention heads have modest but well above-chance ability to spot event arguments sans any training. Furthermore, we investigate how our methods do for cross-sentence event arguments, proposing a procedure to isolate \"best heads\" for cross-sentence argument detection separately of those for intra-sentence arguments. In the second third, we take a closer look at the infilling abilities of BERT. We know BERT is good at Word-level Infilling (obviously!) . How good is it though at Sentence-level Infilling a.k.a Cloze ? We introduce a human-created sentence cloze dataset, collected from public school English examinations. Our task requires a model to fill up multiple blanks in a passage from a shared candidate set with distractors designed by English teachers. Our experiments show a significant performance gap between BERT (72%) and humans (87%), encouraging future models to bridge this gap.We conclude our talk by going through a somewhat unrelated, recent foray into data augmentation for finetuning pretrained generators on low resource domains."
+
+[[speakers]]
+name = "Varun Gangal"
+affiliation = "LTI, CMU"
+website = "https://scholar.google.com/citations?user=rWZq2nQAAAAJ&hl=en"
+photo = ""
+bio = "Varun is a PhD student at CMU LTI, advised by Eduard Hovy. His research is primarily on language generation, with specific interests in style transfer, data-to-text generation and document/story-level generation tasks. He has recently also been exploring probing and data augmentation questions motivated by his primary interests. His research has appeared at ACL, EMNLP and AAAI."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-10-23-nancy-wang.toml b/_data/talks/2020-10-23-nancy-wang.toml
new file mode 100644
index 0000000..aa19b36
--- /dev/null
+++ b/_data/talks/2020-10-23-nancy-wang.toml
@@ -0,0 +1,34 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Global Table Extractor (GTE): A Framework for Joint Table Identification and Cell Structure Recognition Using Visual Context"
+date = 2020-10-23
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Documents are often the format of choice for knowledge sharing and preservation in business and science, within which are tables that capture most of the critical data. Unfortunately, most documents are stored and distributed as PDF or scanned images, which fail to preserve table formatting.
+Recent vision-based deep learning approaches have been proposed to address this gap, but most still cannot achieve state-of-the-art results.
+
+ We present Global Table Extractor (GTE), a vision-guided systematic framework for joint table detection and cell structured recognition, which could be built on top of any object detection model. With GTE-Table, we invent a new penalty based on the natural cell containment constraint of tables to train our table network aided by cell location predictions. GTE-Cell is a new hierarchical cell detection network that leverages table styles. Further, we design a method to automatically label table and cell structure in existing documents to cheaply create a large corpus of training and test data. We use this to enhance PubTabNet with cell labels and create FinTabNet, real-world and complex scientific and financial datasets with detailed table structure annotations to help train and test structure recognition.
+
+ Our deep learning framework surpasses previous state-of-the-art results on the ICDAR 2013 and ICDAR 2019 table competition test dataset in both table detection and cell structure recognition. Further experiments demonstrate a greater than 45% improvement in cell structure recognition when compared to a vanilla RetinaNet object detection model in our new financial dataset (FinTabNet).
+"""
+
+[[speakers]]
+name = "Nancy Wang"
+affiliation = "IBM"
+website = "https://researcher.watson.ibm.com/researcher/view.php?person=ibm-wangnxr"
+photo = ""
+bio = "Nancy Wang is a Researcher with IBM Research - Almaden who is currently working on applying deep learning and computer vision methods for table extraction and table understanding from documents. She graduated from the University of Washington with her PhD in Computer Science in 2018 in the area of computer vision for computational neuroscience. Her new table extraction work is under review at top-level AI conferences and is in the process of being incorporated into Watson Discovery. She was also one of the presenting tutors for the Table Extraction and Understanding Tutorial at ICDM 2019 and VLDB 2020."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-10-30-alberto-cairo.toml b/_data/talks/2020-10-30-alberto-cairo.toml
new file mode 100644
index 0000000..eea8b70
--- /dev/null
+++ b/_data/talks/2020-10-30-alberto-cairo.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Data Visualization: How to Make Good Decisions"
+date = 2020-10-30
+start_time = "15:30"
+end_time = "16:30"
+series = "Data Science Seminar"
+location = ""
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Data visualization, the display of data through graphs, charts, maps, and diagrams, is a skill in great demand in many disciplines, from the sciences to communication or business analytics. However, visualization is often misunderstood. For instance, it's often taught as the application of a series of strict rules. This talk argues that visualization is more akin to writing: yes, we do need to understand visualization's grammar but, beyond that, visualization design is flexible, and can't be based on rules that are set in stone. Instead, designers need to develop a good decision-making framework based on asking themselves a series of questions."
+
+[[speakers]]
+name = "Alberto Cairo"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Alberto Cairo is a journalist and designer with many years of experience leading graphics and visualization teams in several countries. He is the Knight Chair at the School of Communication of the University of Miami, where he teaches courses on infographics and data visualization. He is also director of the Center for Visualization at UM’s Institute for Data Science and Computing, and a Faculty Fellow at the Abess Center for Ecosystem Science and Policy.In the past decade, Cairo has taught and consulted in nearly thirty countries, working for Microsoft, Google, the U.S. National Guard, and many other companies and institutions. Cairo has also written for The New York Times and Scientific American magazine and he is the author of numerous books, the latest one being 'How Charts Lie: Getting Smarter About Visual Information' (W.W. Norton, 2019)"
+
+[meta]
+source = "google-calendar"
+calendar_uid = "5s6i1fjsvem8lctt89djp8t9a9@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-10-30-daniel-scharfstein.toml b/_data/talks/2020-10-30-daniel-scharfstein.toml
new file mode 100644
index 0000000..d4ca4bd
--- /dev/null
+++ b/_data/talks/2020-10-30-daniel-scharfstein.toml
@@ -0,0 +1,30 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Semiparametrics: A Biostatistician’s Toolbox"
+date = 2020-10-30
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "In this talk, I will discuss the theory of semiparametrics that I use to estimate causal effects at root-n rates. Estimators of these effects depend on estimators of nuisance parameters that can be estimated at rates slower than root-n; I provide sufficient conditions for these rates. I will seek advice on the machine learning estimation techniques that satisfy these conditions. I will illustrate the theory in the context of estimating the causal contrast of two competing treatments based on data from a comprehensive cohort study in which clinically eligible individuals are first asked to enroll in a randomized trial and, if they refuse, are then asked to enroll in a parallel observational study in which they can choose treatment according to their own preference."
+
+[[speakers]]
+name = "Daniel Scharfstein"
+affiliation = "Utah, Population Health Sciences"
+website = "http://www.biostat.jhsph.edu/~dscharf/about.html"
+photo = ""
+bio = """
+Daniel Scharfstein is a Professor of Biostatistics in the Department of Population Health Sciences, at the University of Utah School of Medicine.
+He joined the U in August 2020 after spending 23 years on the faculty in the Department of Biostatistics at the Johns Hopkins Bloomberg School of Public Health.
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-11-06-bhargavi-paranjape.toml b/_data/talks/2020-11-06-bhargavi-paranjape.toml
new file mode 100644
index 0000000..34b15d9
--- /dev/null
+++ b/_data/talks/2020-11-06-bhargavi-paranjape.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09 (https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+date = 2020-11-06
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Decisions of complex models for language understanding can be explained by limiting the inputs they are provided to a relevant sub-sequence of the original text — a rationale. Models that condition predictions on a concise rationale, while being more interpretable, tend to be less accurate than models that are able to use the entire context. In this paper, we show that it is possible to better manage the trade-off between concise explanations and high task accuracy by optimizing abound on the Information Bottleneck (IB) objective. Our approach jointly learns an explainer that predicts sparse binary masks over input sentences without explicit supervision and an end-task predictor that considers only the residual sentences. Using IB, we derive a learning objective that allows direct control of mask sparsity levels through a tunable sparse prior. Experiments on the ERASER benchmark demonstrate significant gains over previous work for both task performance and agreement with human rationales"
+
+[[speakers]]
+name = "Bhargavi Paranjape"
+affiliation = "University of Washington"
+website = "https://bhargaviparanjape.github.io"
+photo = ""
+bio = "Bhargavi is a second-year Ph.D. student in the Paul G. Allen School of Computer Science & Engineering, University of Washington, where she is advised by Hannaneh Hajishirzi and Luke Zettlemoyer. Her research interests include interpretability and explainability of neural models, managing model and dataset bias, and pre-training techniques for NLP. She earned an MS in Language Technology from LTI, Carnegie Mellon University, and a B.Tech in Computer Science and Engineering from IIT Kharagpur."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-11-13-akanksha-atrey.toml b/_data/talks/2020-11-13-akanksha-atrey.toml
new file mode 100644
index 0000000..7db609f
--- /dev/null
+++ b/_data/talks/2020-11-13-akanksha-atrey.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Towards High-Performance Machine Learning on the Edge"
+date = 2020-11-13
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Modern day distributed technologies, such as mobile systems and the Internet of Things (IoT), enable the global integration of heterogeneous smart devices via wireless networks. A common characteristic across these technologies is their ability to collect and communicate continuously streaming data. The generation of high bandwidth data makes machine learning (ML) and artificial intelligence (AI) appealing for processing, reasoning, and predicting about the environment, but low network latency requirements make offloading intelligence to the cloud undesirable. This raises an important question: how can we design, develop and evaluate ML algorithms that perform beyond predictive accuracy (e.g., generalizable, explainable, privacy-aware, and efficient) while being accessible and scalable in resource-constrained edge environments? In this talk, I will cover two aspects, explainability and privacy, when deploying such ML models in modern distributed technologies. The focus of the talk will be two-fold: (1) counterfactual evaluation of the explanations generated using saliency maps in deep reinforcement learning for applications such as autonomous vehicles, and (2) privacy implications of personalized ML models in context-aware mobility applications. The talk will be concluded with a discussion on where the future of ML lies in evolving distributed technologies."
+
+[[speakers]]
+name = "Akanksha Atrey"
+affiliation = "UMass"
+website = "https://akanksha-atrey.github.io"
+photo = ""
+bio = "Akanksha Atrey is a Ph.D. student in the College of Information and Computer Sciences at the University of Massachusetts Amherst. She is a member of the Laboratory of Advanced Software Systems, where she is advised by Prof. Prashant Shenoy. Her research interests lie at the intersection of machine learning, edge computing, and privacy with a focus on applications in IoT and mobile computing. Her work seeks to make machine learning more available, usable, and scalable for applications in edge systems. Prior to joining UMass, she was a software engineer at IBM where she worked on the IBM z/OS Mainframe. In 2016, she received her Bachelor of Science degree in Mathematics and Computer Science from the State University of New York at Albany. Among her achievements, she has been named a CRA-W Research Scholar, received an Honorable Mention for the NSF Graduate Research Fellowship Program, and received the Lori A. Clarke Scholarship in Computer Science."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-11-20-akhil-arora.toml b/_data/talks/2020-11-20-akhil-arora.toml
new file mode 100644
index 0000000..e8f823e
--- /dev/null
+++ b/_data/talks/2020-11-20-akhil-arora.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Low-rank Subspaces for Unsupervised Entity Linking"
+date = 2020-11-20
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Entity linking is an important problem with many applications. Most previous solutions were designed for settings where annotated training data is available, which is, however, not the case in numerous domains. We propose a light-weight and scalable entity linking method, Eigenthemes, that relies solely on the availability of entity names and a referent knowledge base. Eigenthemes exploits the fact that the entities that are truly mentioned in a document (the ``gold entities'') tend to form a semantically dense subset of the set of all candidate entities in the document. Geometrically speaking, when representing entities as vectors via some given embedding, the gold entities tend to lie in a low-rank subspace of the full embedding space. Eigenthemes identifies this subspace using the singular value decomposition and scores candidate entities according to their proximity to the subspace. Extensive experiments on benchmark datasets from a variety of real-world domains showcase the effectiveness of our approach."
+
+[[speakers]]
+name = "Akhil Arora"
+affiliation = "EPFL"
+website = ""
+photo = ""
+bio = "Akhil Arora is a PhD student affiliated with the Data Science Lab (dlab) at EPFL. Prior to this, Akhil spent close to five years in industry working with the research labs of American Express and Xerox. Akhil’s research interests include large scale data management, graph mining, and machine learning. He is a recipient of the prestigious “EDIC Doctoral Fellowship” for the academic year 2018-19, and the “Most Reproducible Paper” award at SIGMOD 2018. He has published his research in prestigious data mining and database conferences, served as a reviewer, and co-organized workshops in these conferences."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2020-12-04-danish-pruthi.toml b/_data/talks/2020-12-04-danish-pruthi.toml
new file mode 100644
index 0000000..a6ccc83
--- /dev/null
+++ b/_data/talks/2020-12-04-danish-pruthi.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "A Tale of Evidence and Explanations"
+date = 2020-12-04
+start_time = "11:50"
+end_time = "13:10"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "I would present a brief overview of the state of research in explainability and its evaluation (or lack thereof). Then, I would offer a new lens into explanations, viewing them as a communication channel between a teacher and a student. This view enables us to quantitatively evaluate different attribution methods in a principled way at scale. Shifting gears, in the second part of the talk, I would introduce new techniques to supplement predictions with evidence to enable stakeholders to verify the outcomes readily."
+
+[[speakers]]
+name = "Danish Pruthi"
+affiliation = "LTI, CMU"
+website = "https://www.cs.cmu.edu/~ddanish/"
+photo = ""
+bio = "Danish Pruthi is a senior Ph.D. student at School of Computer Science in Carnegie Mellon University. Broadly, his research aims to enable machines to understand and explain natural language phenomena. He completed his bachelors degree in computer science from BITS Pilani, Pilani in 2015. He has also spent time doing research at Google AI, Facebook AI Research, Microsoft Research, and Indian Institute of Science. He is a recipient of the Siebel Scholarship, and CMU Presidential Fellowship."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "25im2jdereukeqbrt5oequoank_R20200904T175000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-01-22-grad-student-spotlights.toml b/_data/talks/2021-01-22-grad-student-spotlights.toml
new file mode 100644
index 0000000..6c3fa8e
--- /dev/null
+++ b/_data/talks/2021-01-22-grad-student-spotlights.toml
@@ -0,0 +1,32 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "1. Introduction and Logistics"
+date = 2021-01-22
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Archit Rathore: Exploring the Shape of Activations - A BERT case studyBenwei Shi: At-the-time and Back-in-time Persistent Sketches
+Joe Vinu: What all it takes for Performant Deep Nets to be Reliable and Fair?
+Brian Lavallee: Rounding Out Structural Rounding
+Vivek Gupta: Inference on Tables as Semi-Structured Data
+"""
+
+[[speakers]]
+name = "Grad Student Spotlights"
+affiliation = "10-minute talks by 4 current graduate students"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "55bhk7qcmf0fjahlntbinoquue@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-01-29-sanghamitra-dutta.toml b/_data/talks/2021-01-29-sanghamitra-dutta.toml
new file mode 100644
index 0000000..89e35b0
--- /dev/null
+++ b/_data/talks/2021-01-29-sanghamitra-dutta.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "A Systematic Understanding of Exempt and Non-Exempt Algorithmic Biases"
+date = 2021-01-29
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "With the growing use of machine learning algorithms in highly consequential domains, the quantification and removal of bias with respect to gender, race, etc., is becoming increasingly important. While quantifying bias is essential, sometimes the needs of a business (e.g., hiring) may require the use of certain features that are critical in a way that any bias that can be explained by them might need to be exempted (inspired from the business necessity defense of Title VII of Civil Rights Act). For instance, in hiring a software engineer, a standardized coding-test score may be a critical feature that is weighed strongly in the decision even if it introduces bias, whereas other features, such as name, zip code, or reference letters may be used to improve decision-making, but only to the extent that they do not introduce bias. In this work, we propose a novel information-theoretic measure of non-exempt bias, which quantifies the part of the bias that cannot be accounted for by the critical features. This measure can be applied for (i) Auditing trained models to check if the bias arose purely due to the critical features; and also for (ii) Training with selective removal of the non-exempt bias if desired. We arrive at this decomposition through canonical examples that lead to a set of desirable properties (axioms) that any measure of non-exempt bias should satisfy. We then propose a causal measure of non-exempt bias that satisfies all of them. We also propose observational measures that only satisfy some of these properties (including an impossibility result on observational measures being able to satisfy all properties). Then, we perform case studies using them to show how one can train models while reducing non-exempt bias. Our quantification bridges ideas of causality, Simpson's paradox, and a body of work from information theory called Partial Information Decomposition (PID). The talk will be fairly accessible, no knowledge of information theory is required."
+
+[[speakers]]
+name = "Sanghamitra Dutta"
+affiliation = "CMU"
+website = ""
+photo = ""
+bio = "Sanghamitra Dutta (B. Tech. IIT Kharagpur) is a doctoral candidate in the Department of Electrical and Computer Engineering at Carnegie Mellon University, PA, USA. Her research interests revolve around machine learning and information theory. She is currently focussed on addressing the emerging trust issues in machine learning concerning fairness, privacy and reliability. Her work bridges the fields of information theory, causality, reliability and machine learning. In her prior work, she has also examined problems in reliable computing, proposing novel algorithmic solutions for large-scale machine-learning in the presence of faults and failures, using tools from coding theory (an emerging area called “coded computing”). Her results on coded computing address problems that have been open for several decades and have received substantial attention from across communities. She is a recipient of the 2019 K&L Gates Presidential Fellowship, 2019 Axel Berny Presidential Graduate Fellowship, 2017 Tan Endowed Graduate Fellowship, 2016 Prabhu and Poonam Goel Graduate Fellowship, and the 2014 HONDA Young Engineer and Scientist Award."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "2p9fq6r80ck9so1hrs2hkj0h01@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-02-05-michal-moshkovitz.toml b/_data/talks/2021-02-05-michal-moshkovitz.toml
new file mode 100644
index 0000000..21e12a9
--- /dev/null
+++ b/_data/talks/2021-02-05-michal-moshkovitz.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Unexpected Effects of Online no-Substitution k-means Clustering"
+date = 2021-02-05
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Offline k-means clustering was studied extensively and algorithms with a constant approximation are available. However, online clustering is still uncharted. New factors come into play: the ordering of the dataset and whether the number of points, n, is known in advance or not. Their exact effects are unknown. In this work, we focus on the online setting where the decisions are irreversible: after a point arrives the algorithm needs to decide whether to take the point as a center or not, and this decision is final. How many centers are needed and sufficient to achieve constant approximation in this setting? We show upper and lower bounds for all the different cases. These bounds are exactly the same up to a constant, thus achieving optimal bounds. For example, for k-means cost with constant k>1 and random order, Θ(logn) centers are enough to achieve a constant approximation, while the mere a priori knowledge of n reduces the number of centers to a constant. These bounds hold for any distance function that obeys a triangle-type inequality."
+
+[[speakers]]
+name = "Michal Moshkovitz"
+affiliation = "UCSD"
+website = ""
+photo = ""
+bio = "Michal is a postdoctoral fellow at the Qualcomm Institute of the University of California, San Diego. Her interests lie in the foundations of AI, exploring how different constraints affect learning. She works on explainable machine learning, bounded memory learning, and online no-substitution clustering. Michal received her PhD from the Hebrew University and an MSc from Tel-Aviv University. She was the recipient of an Anita Borg scholarship from Google and a Hoffman scholarship from the Hebrew University."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "1fplk468i67vq4rao19d5872b8@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-02-12-fritz-lekschas.toml b/_data/talks/2021-02-12-fritz-lekschas.toml
new file mode 100644
index 0000000..99567fd
--- /dev/null
+++ b/_data/talks/2021-02-12-fritz-lekschas.toml
@@ -0,0 +1,44 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Visual Pattern Exploration At and Across Scales"
+date = 2021-02-12
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Visually exploring data is a powerful approach to discover,
+understand, and interpret novel or not-well defined patterns. It
+allows us to gain insights and generate hypotheses for subsequent
+analyses. However, visual exploration can become challenging when the
+patterns of interest are sparsely-distributed, several orders of
+magnitude smaller than the entire dataset, or detected with high
+uncertainty. In this talk, I will discuss challenges in visually
+exploring multi-modal and multi-scale data, and present new
+visualization systems for efficiently browsing, comparing, and finding
+patterns in the context of genomic, geospatial, and time-series data.
+Specifically, I will describe a web platform for browsing multi-modal
+and multi-scale datasets, as well as their guided navigation. I will
+present a generalized framework and toolkit for interactively
+arranging, grouping, and aggregating thousands of pattern instances.
+And I will demonstrate how interactive visual machine learning can
+enhance our ability to find patterns effectively.
+"""
+
+[[speakers]]
+name = "Fritz Lekschas"
+affiliation = "Harvard"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "6t3vaf2b4krthj70ntv2s5snfb@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-02-19-samson-zhou.toml b/_data/talks/2021-02-19-samson-zhou.toml
new file mode 100644
index 0000000..388292d
--- /dev/null
+++ b/_data/talks/2021-02-19-samson-zhou.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Tight Bounds for Adversarially Robust Streams and Sliding Windows via Difference Estimators"
+date = 2021-02-19
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+We introduce difference estimators for data stream computation, which provide approximations to F(v)-F(u) for frequency vectors v,u and a given function F. We show how to use such estimators to carefully trade error for memory in an iterative manner. The function F is generally non-linear, and we give the first difference estimators for the frequency moments F_p for p between 0 and 2, as well as for integers p>2. Using these, we resolve a number of central open questions in adversarial robust streaming and sliding window models.
+
+For both models, we obtain algorithms for norm estimation whose dependence on epsilon is 1/epsilon^2, which shows, up to logarithmic factors, that there is no overhead over the standard insertion-only data stream model for these problems.
+"""
+
+[[speakers]]
+name = "Samson Zhou"
+affiliation = "CMU"
+website = ""
+photo = ""
+bio = "Samson is a postdoctoral researcher at Carnegie Mellon University, hosted by David P. Woodruff. He received his PhD from Purdue, where he was advised by Greg Frederickson and Elena Grigorescu. He spent a year as a postdoctoral researcher at Indiana University, hosted by Grigory Yaroslavtsev. His research focuses on the theoretical foundations of data science, including sublinear algorithms with an emphasis on streaming algorithms, machine learning, and numerical linear algebra."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4mruo6knu7djmmv8mfub2hfadi@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-02-26-rajesh-jayaram.toml b/_data/talks/2021-02-26-rajesh-jayaram.toml
new file mode 100644
index 0000000..6af46c3
--- /dev/null
+++ b/_data/talks/2021-02-26-rajesh-jayaram.toml
@@ -0,0 +1,36 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "An Improved Analysis of the Quadtree for High Dimensional EMD"
+date = 2021-02-26
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = "https://rajeshjayaram.com/EarthMoverCJLW.pdf"
+recording = ""
+canceled = false
+abstract = """
+The Earth Mover Distance (EMD) between two multi-sets A,B in R^d of size s is the min-cost of bipartite matchings between points in A and B, where cost is measured by distance between points. In this talk, we discuss a classic divide-and-conquer algorithm known as Quadtree for approximating EMD.
+We give a new analysis of the Quadtree, showing that it gives a Õ(log s) approximation. This improves on the previous known O(min{log s , log d} * log s)-approximation of Andoni, Indyk, and Krauthgamer [SODA 08], and Backurs, Dong, Indyk, Razenshteyn, and Wagner (ICML 20).
+
+We also give new space efficient sketching and streaming algorithms for estimating EMD with the improved approximation factor. The main conceptual contribution is an analytical framework for studying the Quadtree which goes beyond worst-case distortion of randomized tree embeddings.
+
+Based on a joint work with Xi Chen, Amit Levi, and Erik Waingarten.
+
+Paper: https://rajeshjayaram.com/EarthMoverCJLW.pdf (https://rajeshjayaram.com/EarthMoverCJLW.pdf)
+"""
+
+[[speakers]]
+name = "Rajesh Jayaram"
+affiliation = "CMU"
+website = "https://rajeshjayaram.com/EarthMoverCJLW.pdf"
+photo = ""
+bio = "Rajesh Jayaram is a PhD student at Carnegie Mellon University, advised by David Woodruff. His research focuses on the design of sketching algorithms, especially streaming and distributed algorithms, for problems in big-data. A central theme of his work is the usage of sketching techniques to speed up algorithmic tasks across various applications, such as in machine learning, optimization, databases, and numerical linear algebra. Additionally, he is interested in aspects of robustness in machine learning and streaming. His work in sketching has received two Best Paper Awards at the Symposium on Principles of Database Systems (PODS) in 2019 and 2020."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "74coomn4fvfpv0avekrihuvbbc@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-03-12-yaoqing-yang.toml b/_data/talks/2021-03-12-yaoqing-yang.toml
new file mode 100644
index 0000000..c076592
--- /dev/null
+++ b/_data/talks/2021-03-12-yaoqing-yang.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Boundary thickness and robustness in learning models"
+date = 2021-03-12
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Robustness of machine learning models to various adversarial and non-adversarial corruptions continues to be of interest. In this talk, we present the notion of the \"boundary thickness\" of a classifier, and we describe its connection with and usefulness for model robustness. Thick decision boundaries lead to improved performance, while thin decision boundaries lead to overfitting (e.g., measured by the robust generalization gap between training and testing) and lower robustness. We show that a thicker boundary helps improve robustness against adversarial examples (e.g., improving the robust test accuracy of adversarial training) as well as so-called out-of-distribution (OOD) transforms, and we show that many commonly-used regularization and data augmentation procedures can increase boundary thickness. On the theoretical side, we establish that maximizing boundary thickness during training is akin to the so-called mixup training procedure. Using these observations, we show that noise-augmentation on mixup training further increases boundary thickness, thereby combating vulnerability to various forms of adversarial attacks and OOD transforms. We can also show that the performance improvement in several lines of recent work happens in conjunction with a thicker boundary."
+
+[[speakers]]
+name = "Yaoqing Yang"
+affiliation = "UC Berkeley"
+website = ""
+photo = ""
+bio = "Yaoqing Yang obtained the PhD degree from Carnegie Mellon University. Now he is a postdoctoral researcher in RISE Lab, UC Berkeley. His primary research interest lies in applying theoretical approaches to robustness issues in large-scale distributed machine learning, as well as designing learning algorithms on structured data such as point clouds and graphs."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "5nbrqe43a6onjh1u16vlqd3b54@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-03-19-vivek-gupta.toml b/_data/talks/2021-03-19-vivek-gupta.toml
new file mode 100644
index 0000000..5b25cad
--- /dev/null
+++ b/_data/talks/2021-03-19-vivek-gupta.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Logic based classification for Low Resource Setting"
+date = 2021-03-19
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = "https://www.aclweb.org/anthology/2020.aacl-main.71.pdf"
+recording = ""
+canceled = false
+abstract = "An NLP model’s ability to reason should be independent of language. Previous works utilize Natural Language Inference(NLI) to understand the reasoning ability of models, mostly focusing on high resource languages like English. To address scarcity of data in low-resource languages such as Hindi, we use data recasting to create four NLI datasets from existing four text classification datasets in Hindi language. Through experiments, we show that our recasted dataset is devoid of statistical irregularities and spurious patterns. We study the consistency in predictions of the textual entailment models and propose a consistency regulariser to remove pairwise-inconsistencies in predictions. Furthermore, we propose a novel two-step classification method which uses textual-entailment predictions for classification tasks. We further improve the classification performance by jointly training the classification and textual entailment tasks together. We therefore highlight the benefits of data recasting and our approach with supporting experimental results. You can access the dataset and paper here: https://www.aclweb.org/anthology/2020.aacl-main.71.pdf (https://www.aclweb.org/anthology/2020.aacl-main.71.pdf). Joint work with BloomBerg AI."
+
+[[speakers]]
+name = "Vivek Gupta"
+affiliation = "University of Utah"
+website = "https://www.aclweb.org/anthology/2020.aacl-main.71.pdf"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "2uivvek9nh1f7d891vfsgedtum@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-04-09-emily-beth-wall.toml b/_data/talks/2021-04-09-emily-beth-wall.toml
new file mode 100644
index 0000000..036531e
--- /dev/null
+++ b/_data/talks/2021-04-09-emily-beth-wall.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "As We Are: Detecting and Mitigating Human Bias in Visual Analytics"
+date = 2021-04-09
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "Virtual"
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Visual Analytics combines the complementary strengths of humans (perception and sensemaking capabilities) and machines (fast and accurate information processing). However, people are susceptible to inherent limitations and biases, including cognitive biases (e.g., anchoring bias), social biases borne of cultural stereotypes and prejudices (e.g., gender bias), and perceptual biases (e.g., illusions). These biases can impact data analysis and decision making in critical ways, leading to inaccurate or inefficient choices, or even propagating long-standing institutional and systemic biases.
+
+Given our knowledge of these biases and the increased use of data visualization to support decision making in data science, the goal of this research is to detect and mitigate human biases in visual data analysis. In this talk, I describe (1) which types of bias are particularly relevant in the process of visual data analysis, (2) how user interactions with data can be used to approximate human biases, and (3) how visualization systems can be designed to increase user awareness of potentially unconscious or implicit biases. By creating systems that promote real-time awareness of bias, people can reflect on their behavior and decision making and ultimately engage in a less-biased analysis and decision making process.
+"""
+
+[[speakers]]
+name = "Emily Beth Wall"
+affiliation = "Emory"
+website = ""
+photo = ""
+bio = "Dr. Emily Wall is an Assistant Professor in the Computer Science department at Emory University (beginning Summer 2021). She completed her PhD in the School of Interactive Computing at Georgia Tech in 2020 and is currently a Postdoctoral Scholar at Northwestern University. Her research interests lie at the intersection of cognitive science and data visualization. Particularly, her research has focused on increasing awareness of unconscious and implicit human biases through the design and evaluation of (1) computational approaches to quantify bias from user interaction and (2) interfaces to support visual data analysis. Her research has been supported by NSF, Pacific Northwest National Laboratory, and Siemens, among others."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "7l6e4jsrifbviegue7r9simqsh@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-04-16-arun-sai-suggala.toml b/_data/talks/2021-04-16-arun-sai-suggala.toml
new file mode 100644
index 0000000..9f1e513
--- /dev/null
+++ b/_data/talks/2021-04-16-arun-sai-suggala.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Game Theoretic Statistics"
+date = 2021-04-16
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "Virtual"
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Game theory and statistics are often regarded as disparate research areas. This is because typical statistical estimation settings are non-adversarial, and the samples are assumed to be generated by some stationary non-reactive source. However, there is a great degree of commonality between the two fields. Classically, the mathematical philosophy of statistics, particularly frequentist statistics, posits that the source of samples is potentially adversarial. This resulted in the rich theory of minimax statistical games and estimation. Boosting algorithms, which are often regarded as best off-the-shelf classifiers, can be viewed as playing a zero-sum game against a weak learner. To allow for various departures of ``test environment'' from ``train environments'', the emerging field of robust machine learning allows for adversarial manipulation of the train or test environments. Finally, an emerging class of density estimators (GANs) in modern machine learning use an adversarial ``critic'' of the density estimator to improve the final density estimation. The common theme among these classical and modern developments is an interplay between statistical estimation and two player games.
+
+In this talk, I will present some of my recent work at the intersection of statistics and game theory and show how game theory can help statistics. In particular, I will present my work on minimax statistical estimation, where we develop algorithmic techniques for constructing minimax estimators. Our algorithms rely on tools from online nonconvex learning and help us construct minimax estimators for fundamental estimation problems such as covariance estimation and entropy estimation.
+"""
+
+[[speakers]]
+name = "Arun Sai Suggala"
+affiliation = "CMU"
+website = ""
+photo = ""
+bio = "Arun Sai Suggala is a final year Machine Learning PhD student at Carnegie Mellon University (CMU), advised by Pradeep Ravikumar. Arun is broadly interested in online learning, game theory, and their applications to machine learning and statistics. He is particularly interested in designing new algorithmic and analytic tools in game theory for solving statistical and machine learning problems. His work has received the best student paper award at ALT'20. Prior to CMU, Arun completed his undergraduate studies in Computer Science and Engineering in the Indian Institute of Technology, Bombay."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "100up85j42e3lvd7g8i991n4cb@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-04-23-lizzie-kumar.toml b/_data/talks/2021-04-23-lizzie-kumar.toml
new file mode 100644
index 0000000..a6ccd94
--- /dev/null
+++ b/_data/talks/2021-04-23-lizzie-kumar.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Epistemic values in feature importance methods: Lessons from feminist epistemology"
+date = 2021-04-23
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "Virtual"
+zoom = "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "As the public seeks greater accountability and transparency from machine learning algorithms, the research literature on methods to explain algorithms and their outputs has rapidly expanded. Feature importance, or the practice of assigning quantitative importance values to the input features of a machine learning model, form a popular class of such methods. Much of the research on feature importance rests on formalizations that attempt to capture universally desirable properties. We investigate the ways in which epistemic values are implicitly embedded in these methods and analyze the ways in which they conflict with ideas from feminist philosophy. We offer some suggestions on how to conduct research on explanations that respects feminist epistemic values, taking into account the importance of social context, the epistemic privileges of subjugated knowers, and adopting more interactional ways of knowing."
+
+[[speakers]]
+name = "Lizzie Kumar"
+affiliation = "Utah"
+website = ""
+photo = ""
+bio = "Lizzie Kumar is a second-year Computing Ph.D. student advised by Suresh Venkatasubramanian at the University of Utah where her work has previously been supported by the ARCS Foundation. She is interested in the practice of analyzing the social impact of machine learning systems and developing responsible AI law and policy. Previously, she developed risk models on the Data Science team at MassMutual while completing her M.S. in Computer Science at the University of Massachusetts, and also holds a B.A. in Mathematics from Scripps College."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4isdaa3q7kil2psmp7b3vqb3su@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-08-27-jeff-phillips.toml b/_data/talks/2021-08-27-jeff-phillips.toml
new file mode 100644
index 0000000..9f6780e
--- /dev/null
+++ b/_data/talks/2021-08-27-jeff-phillips.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "A Visual tour of Bias Mitigation"
+date = 2021-08-27
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Word vector embeddings have been shown to contain and amplify biases in data they are extracted from. Consequently, many techniques have been proposed to identify, mitigate, and attenuate these biases in word representations. In this talk, I will review a collection of state-of-the-art debiasing techniques. To aid this, we provide an open source web-based visualization tool VERB (Visualization of Embedding Representations for deBiasing) and offer hands-on experience in exploring the effects of these debiasing techniques on the geometry of high-dimensional word vectors. To help understand how various debiasing techniques change the underlying geometry, I will show how to decompose each technique into interpretable sequences of primitive operations and study their effect on the word vectors using dimensionality reduction and interactive visual exploration."
+
+[[speakers]]
+name = "Jeff Phillips"
+affiliation = "University of Utah"
+website = "https://www.cs.utah.edu/~jeffp/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "630867A7-3CD6-490A-81C3-4AD8A2D0D7F7"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-09-03-shandian-zhe.toml b/_data/talks/2021-09-03-shandian-zhe.toml
new file mode 100644
index 0000000..8c3a1cb
--- /dev/null
+++ b/_data/talks/2021-09-03-shandian-zhe.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Multi-fidelity Learning and Optimization for Physical Simulation and AutoML"
+date = 2021-09-03
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Multi-fidelity learning involves using training examples at different fidelities or resolutions. High-fidelity examples are of high-quality but often are much more costly to collect than inaccurate, low-fidelity examples. How to retrieve and leverage examples at multiple fidelities is the key to reduce the learning cost while maximizing efficiency.
+
+This talk will introduce our recent work in multi-fidelity learning and optimization. First, I will introduce our deep auto-regressive models that can capture complex correlations across the fidelities to integrate examples of high-dimensional outputs. These are common in applications of physical simulation. Second, I will introduce our work of deep multi-fidelity active learning and Bayesian optimization that can improve the learning and optimization efficiency while reducing the cost of generating training examples, namely maximizing the benefit-cost ratio. Finally, a batch version of the active learning and optimization technique will be presented, which can reduce the query redundancy, improve diversity, and further boost the benefit-cost ratio. I will showcase the advantage of our methods in standard benchmarks of physical simulation, topology structure optimization, and typical tasks in hyper-parameter tuning/AutoML.
+"""
+
+[[speakers]]
+name = "Shandian Zhe"
+affiliation = "Utah SoC"
+website = "https://www.cs.utah.edu/~zhe/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FFA9CFC-C56B-47A9-8680-50C0DE87E0F0"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-09-10-pierre-lermusiaux.toml b/_data/talks/2021-09-10-pierre-lermusiaux.toml
new file mode 100644
index 0000000..5400d8a
--- /dev/null
+++ b/_data/talks/2021-09-10-pierre-lermusiaux.toml
@@ -0,0 +1,35 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Neural Closure Models for Dynamical Systems"
+date = 2021-09-10
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Complex dynamical systems are used for predictions in many domains. Because of computational costs, models are truncated, coarsened or aggregated. As the neglected and unresolved terms become important, the utility of model predictions diminishes. We develop a novel, versatile and rigorous methodology to learn non-Markovian closure parametrizations for known-physics/low-fidelity models using data from high-fidelity simulations. The new neural closure models augment low-fidelity models with neural delay differential equations (nDDEs), motivated by the Mori–Zwanzig formulation and the inherent delays in complex dynamical systems. We demonstrate that neural closures efficiently account for truncated modes in reduced-order models, capture the effects of subgrid-scale processes in coarse models and augment the simplification of complex biological and physical-biogeochemical models. We find that using non-Markovian over Markovian closures improves long-term prediction accuracy and requires smaller networks. We derive adjoint equations and network architectures needed to efficiently implement the new discrete and distributed nDDEs, for any time-integration schemes and allowing non-uniformly spaced temporal training data. The performance of discrete over-distributed delays in closure models is explained using information theory, and we find an optimal amount of past information for a specified architecture. Finally, we analyze computational complexity and explain the limited additional cost due to neural closure models.
+
+Paper: https://royalsocietypublishing.org/doi/10.1098/rspa.2020.1004 (https://royalsocietypublishing.org/doi/10.1098/rspa.2020.1004?fbclid=IwAR1DZy5lHuf68lYwkk5Z_kUYzbHXdgRi0zUhoy5bohGy-w0BjZuwjqPhytQ)
+"""
+
+[[speakers]]
+name = "Pierre Lermusiaux"
+affiliation = ""
+website = "http://meche.mit.edu/people/faculty/pierrel@mit.edu"
+photo = ""
+bio = """
+Abhinav is a 5th year Ph.D. candidate in Mechanical Engineering and Computation at MIT. He received his Bachelor's degree and Master's degree in Mechanical Engineering from the Indian Institute of Technology, Kanpur. At MIT he was a fellow of the MIT-Tata Center for Technology & Design from 2018-20, and recipient of the 2020-21 MathWorks Mechanical Engineering Fellowship.Abhinav is currently developing state-of-the-art scientific machine learning algorithms, with applications to predictive ocean modeling. Apart from his present work, he has specifically worked on uncertainty quantification, data assimilation, Bayesian model learning, and optimal sampling for high-dimensional systems. The algorithms he develops are problem agnostic and can be widely applied. He believes that his unique background in mechanical engineering, applied mathematics, machine learning, and computing position him to identify and implement cross-disciplinary solutions to problems.
+
+https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09 (https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09)
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-09-17-ross-whitaker.toml b/_data/talks/2021-09-17-ross-whitaker.toml
new file mode 100644
index 0000000..8b62e02
--- /dev/null
+++ b/_data/talks/2021-09-17-ross-whitaker.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Air Quality Mapping Using Sensor Networks and Statistical Regression"
+date = 2021-09-17
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+This talk describes an air quality mapping system that is currently deployed in the Salt Lake Valley. We begin with the motivations for the system and then briefly describe the cyberphysical infrastructure. We then review the Gaussian process modeling approach we are using and discuss several important practical considerations around challenges of data wrangling and numerical implementations. We present some examples of AQ estimates associated with specific events of bad air quality. Finally we talk about current directions in research associated with this approach.
+
+COI Disclaimer: Ross Whitaker has a financial interest in the company Tetrad, which has business interests related to the technologies in this talk.
+
+MEB 3147
+"""
+
+[[speakers]]
+name = "Ross Whitaker"
+affiliation = "Utah SoC & SCI"
+website = "http://www.cs.utah.edu/~whitaker/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-09-24-c-seshadhri.toml b/_data/talks/2021-09-24-c-seshadhri.toml
new file mode 100644
index 0000000..a336e10
--- /dev/null
+++ b/_data/talks/2021-09-24-c-seshadhri.toml
@@ -0,0 +1,35 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Studying the (in)effectiveness of low dimensional graph embeddings"
+date = 2021-09-24
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Low dimensional graph embeddings are a fundamental and popular tool used for machine learning on graphs. Given a graph, the basic idea is to produce a low-dimensional vector for each vertex, such that "similarity" in geometric space corresponds to "proximity" in the graph. These vectors can then be used as features in a plethora of machine learning tasks, such as link prediction, community labeling, recommendations, etc. Despite many results emerging in this area over the past few years, there is less study on the core premise of these embeddings. Can such low-dimensional embeddings effectively capture the structure of real-world (such as social) networks? Contrary to common wisdom, we mathematically prove and empirically demonstrate that popular low-dimensional graph embeddings do not capture salient properties of real-world networks. We mathematically prove that common low-dimensional embeddings cannot generate graphs with both low average degree and large clustering coefficients, which have been widely established to be empirically true for real-world networks. Empirically, we observe that the embeddings generated by popular methods fail to recreate the triangle structure of real-world networks, and do not perform well on certain community labeling tasks.
+
+Joint work with Ashish Goel, Caleb Levy, Aneesh Sharma, and Andrew Stolman
+"""
+
+[[speakers]]
+name = "C. Seshadhri"
+affiliation = "UC Santa Cruz"
+website = "https://users.soe.ucsc.edu/~sesh/"
+photo = ""
+bio = """
+C. Seshadhri (Sesh) is a professor of Computer Science at the University of California, Santa Cruz. Prior to joining UCSC, he was a researcher at Sandia National Labs, Livermore. His primary interests are in theoretical computer science and the mathematical foundations of big data algorithms. His work spans many areas: sublinear algorithms, graph algorithms, graph modeling, scalable computation, and data mining. In the theory world, his work has resolved numerous open problems in property testing and sublinear algorithms. A number of his papers in the interface of TCS and applied algorithms have received paper awards at KDD, WWW, ICDM, SDM, and WSDM. He received the 2019 SDM/IBM Early Career Award for Excellence in Data Analytics.
+
+https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09 (https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09)
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-10-01-tony-h-grubesic.toml b/_data/talks/2021-10-01-tony-h-grubesic.toml
new file mode 100644
index 0000000..35c7a5b
--- /dev/null
+++ b/_data/talks/2021-10-01-tony-h-grubesic.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Estimating Potential Oil Spill Trajectories and Coastal Impacts from Near-Shore Storage Facilities: A Case Study of FSO Nabarima and the Gulf of Paria"
+date = 2021-10-01
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "The FSO Nabarima is a floating storage facility and offloading vessel in the Gulf of Paria, between Venezuela and the island of Trinidad. During the latter half of 2020, the Nabarima was disabled, holding approximately 1.3 million barrels (55 million gallons) of crude oil on board. In October of 2020, the vessel was tilting and potentially at risk of spilling its payload into open water. Although all of the oil on the Nabarima was successfully offloaded by April 2021, the threat of large crude oil releases is ubiquitous and persistent in many coastal regions, threatening local ecosystems and livelihoods in coastal communities. The purpose of this presentation is to highlight a geocomputational framework for evaluating the potential spatial vulnerability of coastlines should oil be released from near-shore storage facilities. We use the Nabarima as a broadly representative case study, discuss potential spill cleanup and mitigation strategies, and highlight the challenges of coordinating cross-national responses to these types of spill scenarios."
+
+[[speakers]]
+name = "Tony H. Grubesic"
+affiliation = "UT Austin"
+website = "http://tonygrubesic.net"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-10-08-anna-little.toml b/_data/talks/2021-10-08-anna-little.toml
new file mode 100644
index 0000000..dcba78a
--- /dev/null
+++ b/_data/talks/2021-10-08-anna-little.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "The Mathematics of the Signal-to-Noise Ratio and Insights for Data Science"
+date = 2021-10-08
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Despite the huge variety of data types and goals in data science, there is usually an underlying signal-to-noise ratio that governs the difficulty of the data science task. Analysis of this signal-to-noise ratio leads to important insights about sample size requirements and the impact of the data dimension, which are consistent across various data models and tasks. This talk will illustrate these universal insights in three specific contexts: (1) density-based clustering via graph embeddings, (2) clustering mixture models via multidimensional scaling, and (3) signal recovery from noisy data. The underlying data models are motivated by applications such as imaging, particle physics, single-cell RNA sequencing, and cryo-electron microscopy."
+
+[[speakers]]
+name = "Anna Little"
+affiliation = "Utah Math"
+website = "https://www.anna-little.com"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-10-22-erin-wolf-chambers.toml b/_data/talks/2021-10-22-erin-wolf-chambers.toml
new file mode 100644
index 0000000..5da7cf7
--- /dev/null
+++ b/_data/talks/2021-10-22-erin-wolf-chambers.toml
@@ -0,0 +1,49 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Applications of topology and geometry to root analysis"
+date = 2021-10-22
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Analysis of 3d shapes is a core problem in many fields, and there are
+many tools from topology and geometry that can provide insight and
+understanding. In this talk, we focus on developing significance
+measures for 3d plant structures, primarily root systems of plants.
+Our measures are based on the medial axis transform, which plays a
+fundamental role in shape matching and analysis, but is widely known
+to be unstable to even small boundary perturbations. Methods for
+pruning the medial axis are usually guided by some measure of
+significance, with considerable work done for both 2- and
+3-dimensional shapes. Such significance measures can be used for
+identifying salient features, and hence are useful for simplification,
+comparison, and alignment. In this talk, we will present theoretical
+insights and properties of commonly used significance measures,
+focusing on those in 2D and 3D that are both shape-revealing and
+topology-preserving, as well as being robust to noise on the boundary.
+We'll then discuss several methods that de-noise a shape and identify
+topologically and geometrically prominent features, using both the
+medial axis and other measures commonly used in topological data
+analysis. Our methods are quite successful compared to the state of
+the art, and are available in the package TopoRoot, an automatic
+pipeline for plant architectural analysis from 3D Imaging.
+"""
+
+[[speakers]]
+name = "Erin Wolf Chambers"
+affiliation = "St. Loius University"
+website = "https://cs.slu.edu/~chambers/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-10-29-bao-wang.toml b/_data/talks/2021-10-29-bao-wang.toml
new file mode 100644
index 0000000..0848604
--- /dev/null
+++ b/_data/talks/2021-10-29-bao-wang.toml
@@ -0,0 +1,32 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "How Differential Equations and Random Graph Insights Benefit Deep Learning"
+date = 2021-10-29
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+We will present recent results on developing new deep learning algorithms leveraging differential equations and random graph insights.
+First, we will present a new class of continuous-depth deep neural networks that were motivated by the ODE limit of the classical momentum method, named heavy-ball neural ODEs (HBNODEs). HBNODEs enjoy two properties that imply practical advantages over NODEs: (i) The adjoint state of an HBNODE also satisfies an HBNODE, accelerating both forward and backward ODE solvers, thus significantly accelerate learning and improve the utility of the trained models. (ii) The spectrum of HBNODEs is well structured, enabling effective learning of long-term dependencies from complex sequential data.
+Second, we will extend HBNODE to graph learning leveraging diffusion on graphs, resulting in new algorithms for deep graph learning. The new algorithms are more accurate than existing deep graph learning algorithms and more scalable to deep architectures, and also suitable for learning at low labeling rate regimes. Moreover, we will present a fast multipole method-based efficient attention mechanism for modeling graph nodes interactions.
+Third, if time permits, we will discuss building an efficient and reliable overlay network for decentralized federated learning based on the random graph theory.
+"""
+
+[[speakers]]
+name = "Bao Wang"
+affiliation = "Utah Math, SCI"
+website = "http://www.math.utah.edu/~bwang/index.html"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-11-05-marina-kogan.toml b/_data/talks/2021-11-05-marina-kogan.toml
new file mode 100644
index 0000000..fc5fa1f
--- /dev/null
+++ b/_data/talks/2021-11-05-marina-kogan.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Sequence-based approaches as human-centered data science methods for crisis informatics"
+date = 2021-11-05
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Social media platforms have been increasingly used by the public in crisis situations, partly because they upend the traditional top-down broadcasting model of risk communication. Instead, social media platforms facilitate a two-way information exchange between the official response channels and the general public, enabling more participatory crisis communication, as well as coordination and self-organization among the public. In this more complex information ecosystem, understanding the flow of information is crucial to supporting those affected and preventing malicious actors from capitalizing on the uncertainty. However, the study of such information flows is challenging, because the high-tempo, high-volume convergent nature of crisis events produces vast amounts of social media data, necessitating the use of the data science methods. On the other hand, to glean meaningful insight from the crisis-related social media activity, it is necessary to use methods that account for the complex social context of the user activity. In this talk I will show how the Human-Centered Data Science (HCDS) provides methodological approaches that both harness the power of computational methods and account for the highly situated nature of social media activity in disruption. I will focus on sequence-based approaches as examples of HCDS methods in two empirical studies: analysis of attention-garnering information during a natural disaster and investigation of behavioral signatures in coordinated information operations."
+
+[[speakers]]
+name = "Marina Kogan"
+affiliation = "Utah SoC"
+website = "http://www.mkoganresearch.com"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-11-12-mikhail-belkin.toml b/_data/talks/2021-11-12-mikhail-belkin.toml
new file mode 100644
index 0000000..f3c29c5
--- /dev/null
+++ b/_data/talks/2021-11-12-mikhail-belkin.toml
@@ -0,0 +1,44 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "From classical statistics to modern deep learning"
+date = 2021-11-12
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Recent empirical successes of deep learning have exposed significant gaps in our
+fundamental understanding of learning and optimization mechanisms.
+Modern best practices for model selection are in direct contradiction to the methodologies
+suggested by classical analyses. Similarly, the efficiency of SGD-based local methods
+used in training modern models, appeared at odds with the standard intuitions on optimization.
+
+First, I will present evidence, empirical and mathematical, that necessitates
+revisiting classical statistical notions, such as over-fitting. I will continue to discuss the emerging
+understanding of generalization, and, in particular, the "double descent" risk curve, which extends
+the classical U-shaped generalization curve beyond the point of interpolation.
+
+Second, I will discuss why the landscapes of over-parameterized neural networks are
+generically never convex, even locally. Instead they satisfy the Polyak-Lojasiewicz (PL)
+condition across most of the parameter space instead, presents an powerful framework for optimization in general over-parameterized models and allows SGD-type methods to converge to a global minimum.
+
+While our understanding has significantly grown in the last few years, a key piece of the puzzle remains -- how does optimization align with statistics to form the complete mathematical picture of modern ML?
+"""
+
+[[speakers]]
+name = "Mikhail Belkin"
+affiliation = ""
+website = "http://misha.belkin-wang.org"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-11-19-michael-yeh.toml b/_data/talks/2021-11-19-michael-yeh.toml
new file mode 100644
index 0000000..2cd677b
--- /dev/null
+++ b/_data/talks/2021-11-19-michael-yeh.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Towards a Near Universal Time Series Data Mining Tool: Introducing the Matrix Profile"
+date = 2021-11-19
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Matrix profile is a data structure that annotates a time series by recording the location of and the distance to the nearest neighbors of each subsequences in the time series. The matrix profile stores such information in an efficient and easy-to-access fashion and can be used in a variety of data mining tasks like motif/discord discovery, semantic segmentation, and clustering. In this talk, I will 1) introduce what matrix profile is, 2) discuss the computational challenge associated with matrix profile, and 3) show how it can be used in different time series data mining tasks."
+
+[[speakers]]
+name = "Michael Yeh"
+affiliation = "Visa Research"
+website = "https://mcyeh.github.io/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-12-03-julia-silge.toml b/_data/talks/2021-12-03-julia-silge.toml
new file mode 100644
index 0000000..5639f37
--- /dev/null
+++ b/_data/talks/2021-12-03-julia-silge.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Data visualization for machine learning practitioners"
+date = 2021-12-03
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = "MEB 3147 and Zoom"
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Visual representations of data inform how machine learning practitioners think, understand, and decide. Before charts are ever used for outward communication about a ML system, they are used by the system designers and operators themselves as a tool to make better modeling choices. Practitioners use visualization, from very familiar statistical graphics to creative and less standard plots, at the points of most important human decisions when other ways to validate those decisions can be difficult. Visualization approaches are used to understand both the data that serves as input for machine learning and the models that practitioners create. In this talk, learn about the process of building a ML model in the real world, how and when practitioners use visualization to make more effective choices, and considerations for ML visualization tooling."
+
+[[speakers]]
+name = "Julia Silge"
+affiliation = "RStudio"
+website = "https://juliasilge.com"
+photo = ""
+bio = """
+Julia Silge is a data scientist and software engineer at RStudio PBC where she works on open source modeling tools. She is an author, an international keynote speaker, and a real-world practitioner focusing on data analysis and machine learning practice. Julia loves text analysis, making beautiful charts, and communicating about technical topics with diverse audiences.
+
+MEB 3147 (coffee and snacks provided)
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2021-12-10-sameer-singh.toml b/_data/talks/2021-12-10-sameer-singh.toml
new file mode 100644
index 0000000..aa21acc
--- /dev/null
+++ b/_data/talks/2021-12-10-sameer-singh.toml
@@ -0,0 +1,32 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Evaluating and Testing Natural Language Processing Models"
+date = 2021-12-10
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Current evaluation of the generalization of natural language processing (NLP) systems, and much of machine learning, primarily consists of measuring the accuracy on held-out instances of the dataset. Since the held-out instances are often gathered using similar annotation process as the training data, they include the same biases that act as shortcuts for machine learning models, allowing them to achieve accurate results without requiring actual natural language understanding. Thus held-out accuracy is often a poor proxy for measuring generalization. Further, aggregate metrics have little to say about where the problems may lie, and how to address them.
+In this talk, I will introduce a number of approaches we are investigating to perform a more thorough evaluation of NLP systems. I will first provide a quick overview of automated techniques for perturbing instances in the dataset that identify loopholes and shortcuts in NLP models, including semantic adversaries and universal triggers. I will then describe recent work on creating comprehensive and thorough tests and evaluation benchmarks for NLP using CheckList, that aim to directly evaluate comprehension and understanding capabilities. The talk will include a number of NLP tasks, such as sentiment analysis, textual entailment, paraphrase detection, and question answering.
+
+Dr. Sameer Singh is an Associate Professor of Computer Science at the University of California, Irvine (UCI) and an Allen AI Fellow at Allen Institute for AI. He is working primarily on robustness and interpretability of machine learning algorithms, along with models that reason with text and structure for natural language processing. Sameer was a postdoctoral researcher at the University of Washington and received his PhD from the University of Massachusetts, Amherst. He has received the NSF CAREER award, selected as a DARPA Riser, UCI Distinguished Early Career Faculty award, and the Hellman Faculty Fellowship. His group has received funding from Allen Institute for AI, Amazon, NSF, DARPA, Adobe Research, Hasso Plattner Institute, NEC, Base 11, and FICO. Sameer has published extensively at machine learning and natural language processing venues and received conference paper awards at KDD 2016, ACL 2018, EMNLP 2019, AKBC 2020, and ACL 2020. (https://sameersingh.org/)
+"""
+
+[[speakers]]
+name = "Sameer Singh"
+affiliation = "UC Irvine"
+website = "http://sameersingh.org"
+photo = ""
+bio = "Dr. Sameer Singh is an Associate Professor of Computer Science at the University of California, Irvine (UCI) and an Allen AI Fellow at Allen Institute for AI. He is working primarily on robustness and interpretability of machine learning algorithms, along with models that reason with text and structure for natural language processing. Sameer was a postdoctoral researcher at the University of Washington and received his PhD from the University of Massachusetts, Amherst. He has received the NSF CAREER award, selected as a DARPA Riser, UCI Distinguished Early Career Faculty award, and the Hellman Faculty Fellowship. His group has received funding from Allen Institute for AI, Amazon, NSF, DARPA, Adobe Research, Hasso Plattner Institute, NEC, Base 11, and FICO. Sameer has published extensively at machine learning and natural language processing venues and received conference paper awards at KDD 2016, ACL 2018, EMNLP 2019, AKBC 2020, and ACL 2020. (https://sameersingh.org/)"
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9FC6B798-1BF5-49C1-A760-08701CF251B3+ZOOMMEETINGNUMBER:93778940103"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-01-14-yi-zhou.toml b/_data/talks/2022-01-14-yi-zhou.toml
new file mode 100644
index 0000000..ffcd452
--- /dev/null
+++ b/_data/talks/2022-01-14-yi-zhou.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Understanding the Convergence of Optimization Algorithms for Minimax Machine Learning"
+date = 2022-01-14
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "WEB 1250"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "The past decade has witnessed the great success of deep learning in broad societal and commercial applications. However, conventional deep learning relies on wildly fitting data with neural networks, which is known to produce models that lack resilience. For instance, models used in facial recognition and healthcare are known to be biased toward people of a certain race or gender. Models used in autonomous driving are vulnerable to malicious attacks, i.e., putting an art sticker on a stop sign may force the model to classify it as a speed limit sign. Therefore, the next-generation deep learning paradigm aims to deliver resilient models that promote robustness to malicious attacks, fairness among users, and privacy preservation, and this can be realized by leveraging the emerging minimax machine learning framework. In this talk, I will present three gradient-descent-ascent (GDA) type of optimization algorithms for solving different classes of nonconvex minimax machine learning problems. Then, I will present a principled nonconvex minimax optimization theory that establishes the global convergence and convergence rates of these algorithms."
+
+[[speakers]]
+name = "Yi Zhou"
+affiliation = "Utah ECE"
+website = "https://sites.google.com/site/yizhouhomepage/home"
+photo = ""
+bio = "Yi Zhou is an Assistant Professor affiliated with the Dept. of ECE at The University of Utah. Before joining the University of Utah, he received a Ph.D. in ECE in 2018 from The Ohio State University and worked as a post-doctoral fellow at Information Initiative at Duke University. His research interests include statistical machine learning, nonconvex & distributed optimization, deep learning, reinforcement learning and statistical signal processing."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "0D350DB8-6148-4847-B1F9-691580AFD9B0"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-01-21-chinmay-hedge.toml b/_data/talks/2022-01-21-chinmay-hedge.toml
new file mode 100644
index 0000000..dce7b5d
--- /dev/null
+++ b/_data/talks/2022-01-21-chinmay-hedge.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Designing Neural Networks for Efficient Encrypted Inference"
+date = 2022-01-21
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "As deep neural networks become ever more pervasive, so too are concerns surrounding users' data privacy. Curiously, standard cryptographic encryption approaches for guaranteeing data privacy do not interact well with traditional neural network models. In this talk, I will (a) outline why standard networks are not encryption-efficient, (b) suggest two new approaches for designing deep networks that do support efficient and secure inference, and (c) show results instantiating these approaches on real-world use cases."
+
+[[speakers]]
+name = "Chinmay Hedge"
+affiliation = "NYU"
+website = ""
+photo = ""
+bio = "Chinmay is a faculty member in the CSE and ECE Departments at the NYU Tandon School of Engineering. His research focuses on developing principled, fast, and robust algorithms for diverse problems in machine learning, with applications to imaging and computer vision, materials design, and transportation. Prior to NYU, he was an assistant professor in the Electrical and Computer Engineering Department at Iowa State University, and before that, a post-doctoral associate in the Theory of Computation (TOC) group at MIT, working with Piotr Indyk. He received his Ph.D. at Rice University under the supervision of Rich Baraniuk."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-01-28-swaroop-mishra.toml b/_data/talks/2022-01-28-swaroop-mishra.toml
new file mode 100644
index 0000000..35a40f2
--- /dev/null
+++ b/_data/talks/2022-01-28-swaroop-mishra.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Towards the Development of Models that Learn New Tasks from Instructions"
+date = 2022-01-28
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "MEB 3147 (LCR)"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Swaroop Mishra"
+affiliation = "ASU, Ph.D. Student"
+website = ""
+photo = ""
+bio = "Swaroop Mishra is a 3rd year Ph.D. student at Arizona State University. He finished his Masters from IIT Kanpur in 2016. He worked as a Software Engineer for 2 years at MathWorks and as a Technical Consultant for 1 year at the Information Technology Research Academy, Ministry of Electronics and Information Technology, Govt. of India before starting his Ph.D. He did a research internship with Allen AI and has been collaborating with them for the past year."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-02-04-tuhin-chakrabarty.toml b/_data/talks/2022-02-04-tuhin-chakrabarty.toml
new file mode 100644
index 0000000..946540c
--- /dev/null
+++ b/_data/talks/2022-02-04-tuhin-chakrabarty.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "The Curious Case of Figurative Language"
+date = 2022-02-04
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "MEB 3147 (LCR)"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Despite the ubiquity of figurative language across various forms of speech and writing, the vast majority of NLP research focuses primarily on literal language. Figurative language is challenging because of its implicit nature. In this talk, I will specifically focus on the following questions: 1) Can large language models understand/interpret them? 2) Can computers generate figurative language?"
+
+[[speakers]]
+name = "Tuhin Chakrabarty"
+affiliation = "Columbia University, Ph.D Student"
+website = ""
+photo = ""
+bio = "Tuhin is a Ph.D. student at Columbia University (based in NYC) advised by Smaranda Muresan and an Amazon Ph.D. fellow. Tuhin's interests are language understanding and generation especially non-literal languages, which require world knowledge or commonsense reasoning"
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-02-11-vaggos-chatziafratis.toml b/_data/talks/2022-02-11-vaggos-chatziafratis.toml
new file mode 100644
index 0000000..d053749
--- /dev/null
+++ b/_data/talks/2022-02-11-vaggos-chatziafratis.toml
@@ -0,0 +1,37 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Neural Networks Expressivity through the lens of Dynamical Systems"
+date = 2022-02-11
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Given a target function f, how large must a neural network be in order to approximate f?
+Understanding the representational power of Deep Neural Networks (DNNs) and how their structural properties (e.g., depth, width, type of activation unit) affect the functions they can compute, has been an important yet challenging question in approximation theory and deep learning even in the early days of AI.
+
+In this talk, I want to tell you about some recent progress on this topic that uses ideas from dynamical systems. The main results are exponential depth-width trade-offs for DNNs representing certain families of functions. Our techniques rely on a generalized notion of fixed points, called periodic points that have played a major role in chaos theory (Li-Yorke chaos and Sharkovsky's theorem).
+
+Based on three recent works:
+- with Ioannis Panageas, Sai Ganesh Nagarajan and Xiao Wang from ICLR'20 (spotlight): https://arxiv.org/abs/1912.04378 (https://arxiv.org/abs/1912.04378)
+- with Ioannis Panageas and Sai Ganesh Nagarajan from ICML'20: https://arxiv.org/abs/2003.00777 (https://arxiv.org/abs/2003.00777)
+- with Clayton Sanford from AISTATS'22: https://arxiv.org/abs/2110.10295 (https://arxiv.org/abs/2110.10295)
+"""
+
+[[speakers]]
+name = "Vaggos Chatziafratis"
+affiliation = "UC Santa Cruz"
+website = "https://cs.stanford.edu/~vaggos/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-02-18-jeff-phillips.toml b/_data/talks/2022-02-18-jeff-phillips.toml
new file mode 100644
index 0000000..5f96dc2
--- /dev/null
+++ b/_data/talks/2022-02-18-jeff-phillips.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Some Very Basic Theory of Classification (in low dimensions"
+date = 2022-02-18
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "WEB 1250"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "I plan to talk about some new results on some very fundamental (but overlooked) theory questions in classification. First, how fast can you find a linear classifier that eps-approximates the optimal one in terms of miss-classification? Second, if you want to preserve a Euclidean margin between perfectly classified points for a polynomial classifier, how many samples do you need?"
+
+[[speakers]]
+name = "Jeff Phillips"
+affiliation = "Utah SoC"
+website = "https://www.cs.utah.edu/~jeffp/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-02-25-sunipa-dev.toml b/_data/talks/2022-02-25-sunipa-dev.toml
new file mode 100644
index 0000000..87bfa0b
--- /dev/null
+++ b/_data/talks/2022-02-25-sunipa-dev.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Towards Inclusive and Socially Aware Language Technologies"
+date = 2022-02-25
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Large language models are commonly used in different paradigms of natural language processing and machine learning, and are known for their efficiency as well as their overall lack of interpretability. Their data driven approach for emulating human language often results in human biases being encoded and even amplified, potentially leading to cyclic propagation of representational and allocational harm. We discuss in this talk some aspects of detecting, evaluating, and mitigating biases and associated harms in a holistic, inclusive, and culturally-aware manner. In particular, we discuss the disparate impact on society of common language tools that are not inclusive of all gender identities."
+
+[[speakers]]
+name = "Sunipa Dev"
+affiliation = "Google Research"
+website = "https://sunipa.github.io"
+photo = ""
+bio = "Sunipa Dev is a Research Scientist on the Ethical AI team at Google RAI. Previously, she was an NSF Computing Innovation Fellow at UCLA, before which she completed her PhD at the University of Utah. Her ongoing research focuses on various facets of fairness and interpretability in NLP, including robust measurements of bias, cross-cultural understanding of concepts in NLP, and inclusive language representations."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-03-04-anirudh-goyal-of-montreal.toml b/_data/talks/2022-03-04-anirudh-goyal-of-montreal.toml
new file mode 100644
index 0000000..7367379
--- /dev/null
+++ b/_data/talks/2022-03-04-anirudh-goyal-of-montreal.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "From Specialists to Generalists: Inductive Biases of Deep Learning for Higher Level Cognition"
+date = 2022-03-04
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "MEB 3147 (LCR)"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "A fascinating hypothesis is that human and animal intelligence could be explained by a few principles (rather than an encyclopedic list of heuristics). If that hypothesis was correct, we could more easily both understand our own intelligence and build intelligent machines. Just like in physics, the principles themselves would not be sufficient to predict the behavior of complex systems like brains, and substantial computation might be needed to simulate human-like intelligence. This hypothesis would suggest that studying the kind of inductive biases that humans and animals exploit could help both clarify these principles and provide inspiration for AI research and neuroscience theories. Deep learning already exploits several key inductive biases, and my work considers a larger list, focusing on those which concern mostly higher-level and sequential conscious processing. The objective of clarifying these particular principles is that they could potentially help us build AI systems benefiting from humans' abilities in terms of flexible out-of-distribution and systematic generalization, which is currently an area where a large gap exists between state-of-the-art machine learning and human intelligence."
+
+[[speakers]]
+name = "Anirudh Goyal of Montreal"
+affiliation = ""
+website = "https://anirudh9119.github.io/"
+photo = ""
+bio = "Anirudh Goyal is a student of science advised by Prof. Yoshua Bengio. His current research interests center around understanding how neural learners can compose and abstract their own representations in a way that can be used to better generalize to out of distribution samples. More concretely, his work focuses on designing such models by incorporating in them strong but general assumptions (inductive biases) that enable high-level reasoning about the structure of the world. During his PhD, he has spent time as a visiting researcher at UC Berkeley, MPI Tuebingen and DeepMind. He was also one of the recipients of the 2021 Google PhD Fellowship in Machine Learning."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-03-25-chad-topaz-jude-higdon.toml b/_data/talks/2022-03-25-chad-topaz-jude-higdon.toml
new file mode 100644
index 0000000..f01053d
--- /dev/null
+++ b/_data/talks/2022-03-25-chad-topaz-jude-higdon.toml
@@ -0,0 +1,34 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Quantitative Approaches to Social Justice"
+date = 2022-03-25
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "WEB 1250"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Civil rights leader, educator, and investigative journalist Ida B. Wells said that \"the way to right wrongs is to shine the light of truth upon them.\" This talk will demonstrate how mathematical, statistical, and computational approaches can shine a light on social injustices and help build solutions to remedy them. We will present research-to-action projects on diversity in art museums, inclusion in STEM, equity in criminal sentencing, and other topics. The tools engaged include crowdsourcing, data cleaning, clustering, hypothesis testing, statistical modeling, Markov chains, data visualization, and much more. Overall, we hope that this talk leaves you informed about the breadth of social justice applications that one can tackle using quantitative tools in careful collaboration with other scholars and activists."
+
+[[speakers]]
+name = "Chad Topaz"
+affiliation = "QSIDE"
+website = "https://qsideinstitute.org/#"
+photo = ""
+bio = ""
+
+[[speakers]]
+name = "Jude Higdon"
+affiliation = "QSIDE"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-04-01-debanjan-mahata.toml b/_data/talks/2022-04-01-debanjan-mahata.toml
new file mode 100644
index 0000000..21bf859
--- /dev/null
+++ b/_data/talks/2022-04-01-debanjan-mahata.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Identifying Keyphrases from Text Documents - From Heuristics to Language Models"
+date = 2022-04-01
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "\""
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Automatic identification of keyphrases from text documents is an extreme summarization problem that lies at the intersection of the areas of natural language processing (NLP) and information retrieval (IR). Keyphrases aid in capturing the most salient topics from the input text and are useful in multiple downstream tasks such as classification, clustering, summarization, document recommendation, query expansion, interactive document retrieval, semantic and faceted search. Despite the ground-breaking advancements triggered by deep neural networks, automatically identifying keyphrases from text using machine learning techniques is still a challenging problem that hasn't been explored as much as other related and popular tasks such as named entity extraction, question answering, summarization. This talk will provide an overview of the advances made in keyphrase extraction and generation from text documents and present the approaches, datasets, evaluation strategies, and the associated challenges. It will also dive into the topic of how language models have been effective in pushing state-of-the-art performances in this domain. Lastly, the speaker will present the current trends and future directions for research in this domain."
+
+[[speakers]]
+name = "Debanjan Mahata"
+affiliation = "Moody Analytics"
+website = ""
+photo = ""
+bio = "Debanjan Mahata is a Director of AI at Moody's Analytics, New York, and leads the KYC machine learning team. He is also an Adjunct Faculty at the Department of Computer Science and Engineering, Indraprastha Institute of Information Technology (IIIT-Delhi). He closely collaborates with the Multimodal Digital Media Analysis lab (MIDAS@IIITD). His work lies at the intersection of natural language processing and information retrieval. He is currently interested in keyphrase extraction and generation, understanding code-switched text from social media, and computational social science problems solved using NLP. Before joining Moody's in August 2021, he was a Senior Research Scientist at Bloomberg AI (Nov 2017 - Jul 2021) and Senior Research Associate at Infosys Limited, Palo Alto, California (Aug 2015 - Oct 2017). He holds a Ph.D. in Integrated Computing from Donaghey College of Engineering and Information Technology, University of Arkansas at Little Rock."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-04-08-tao-li.toml b/_data/talks/2022-04-08-tao-li.toml
new file mode 100644
index 0000000..8ad1cb9
--- /dev/null
+++ b/_data/talks/2022-04-08-tao-li.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Improving Data Efficiency of Neural Models using Logic"
+date = 2022-04-08
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "MEB 3147 (LCR)"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "In this talk, we will focus on a simple approach that uses logic to improve neural model performance for natural language processing (NLP) tasks. Many downstream NLP tasks involve domain knowledge that can be easily stated in logical forms. We argue that we can use such knowledge to improve model learning. This results in better data efficiency, i.e., a model that performs better with less annotation. To this end, we propose frameworks that integrate domain knowledge, expressed as declarative constraints, with neural models. We show that such integration substantially improves state-of-the-art neural models in a variety of NLP tasks. To facilitate using our frameworks, we will also propose a PyTorch library that unifies differentiable tensor operations and logical operations."
+
+[[speakers]]
+name = "Tao Li"
+affiliation = "Google Research"
+website = ""
+photo = ""
+bio = "Tao Li is interested in on Natural Language Processing and Machine Learning. He is currently a Research Engineer at Google Research working. Earlier this year, He graduated as a PhD at the the U’s School of Computing where he was advised by Prof. Vivek Srikumar. He was also a MS graduate at the U back in 2014. Besides school, he had research internships at AI2, Amazon A9, and Philips Research."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-04-15-abhinav-kumar.toml b/_data/talks/2022-04-15-abhinav-kumar.toml
new file mode 100644
index 0000000..85ef34a
--- /dev/null
+++ b/_data/talks/2022-04-15-abhinav-kumar.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Mathematical Modeling for Landmark and 3D Object Detection"
+date = 2022-04-15
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "MEB 3147 (LCR)"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Modern computer vision models have excelled on several tasks and beaten several benchmarks. However, many of these models have avoided the mathematical and principled approaches to computer vision, resulting in sub-optimal performance and issues with interpretability. This talk will re-introduce mathematical modeling for computer vision tasks such as facial landmark localization and monocular 3D object detection for autonomous driving. In particular, we discuss the joint estimation of location, uncertainty, and visibility for facial landmark detection and introduce a mathematically differentiable NMS for monocular 3D detection. The mathematical modeling enables end-to-end learning in these tasks, resulting in improved performance."
+
+[[speakers]]
+name = "Abhinav Kumar"
+affiliation = ""
+website = "https://sites.google.com/view/abhinavkumar/"
+photo = ""
+bio = "Abhinav is a Ph.D. student in computer science at Michigan State University working with Prof. Xiaoming Liu. His current research focus is mathematical modeling applied to 3D object detection for autonomous driving. Before joining the graduate program at MSU, he was at the University of Utah and worked at Xerox Research Center India, Bangalore. He holds a master's and bachelor's in electrical engineering from the Indian Institute of Technology (IIT) Bombay and IIT Patna, respectively."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-04-22-khyati-chandu.toml b/_data/talks/2022-04-22-khyati-chandu.toml
new file mode 100644
index 0000000..1491082
--- /dev/null
+++ b/_data/talks/2022-04-22-khyati-chandu.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Anchoring Multimodal Narrative Generation"
+date = 2022-04-22
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "WEB 1250 and Zoom"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Humans inherently learn from and interact with multiple views of information, be it various modalities or languages. So, the expectations from contemporary and 21st-century technology are a testimony to the increasing need to model these multiview contexts better. Natural language generation plays a pivotal role in communicating these contexts in human-understandable languages. This talk brings together both of these transformative technologies to make strides toward a longstanding dream of human-like multiview narrative generation. The critical challenge is identifying the natural-sounding properties of long-form texts and modeling them in tandem with visual contexts. I present anchors for grounding three such properties including content (relevance), structure (coherence), and surface form realization (expression), and anchors them with relevant visual contexts. These anchors also provide us with human interpretable handles for controlling these properties.
+
+Details: To illustrate the effectiveness of the anchors for each of the three properties, I present: Starting with content: In situated multimodal contexts, relevance is the concept of the elements in one modality being connected to the other modality that makes this context informative and complementary. I present visual infilling with curriculum learning as a global objective for content and hierarchically attending over entity skeletons as a local objective for content, to generate visual stories and procedures. To improve the controllability and transferability in English and five other languages, I also introduce a dual-stage model with weakly supervised skeletons and a text-as-side attention mechanism to denoise the content in an image caption. Moving onto structure: The alignment of descriptions in language to the corresponding visual inputs is crucial to generating a logical and coherent narrative. I present a scaffolding technique as a local objective for structure by extracting a layout from vast amounts of unsupervised text to incorporate structure into cooking recipes generated from images. Finally, surface form: The crux of naturalness to automatic generation comes by incorporating individualized and personalized ways of expressing the same content. I present a locally guided, weakly supervised model for generating persona-based visual stories and a dual-staged adversarial technique to generate mixed views from non-parallel data. All of the above work mainly focuses on static multimodal narratives, and finally, I present a case to highlight the significance of transitioning to dynamic grounding. I conclude by presenting the shortcomings of the current approaches in the NLP domain to the grounding problem and offer recommendations along with executable actions for course correction to bridge this gap and enable grounding for machines to resemble human communication.
+"""
+
+[[speakers]]
+name = "Khyati Chandu"
+affiliation = "Meta AI"
+website = "https://www.cs.cmu.edu/~kchandu/"
+photo = ""
+bio = "Dr. Khyathi Chandu is a Research Scientist at Meta AI. Prior to this, she completed her Ph.D. at Carnegie Mellon University, working on controllable generation and multimodality. The avid goal of her research is to enable seamless communication between humans and machines with multiple modalities and languages. Her research focuses on improving vision-and-language generation, by adapting to appropriate content, structure, and persona, particularly in long-form generation. She has also done an array of work in code-switching and biomedical text summarization. She was selected as Rising Stars EECS 2020, won the sixth edition of the BioAsq challenge, organized several workshops at *CL conferences, and co-chaired D&I initiatives at NLP conferences and WiNLP workshops. She has consistently been on the Dean’s Merit list and was awarded the Best All-Rounder Student in her undergraduate. She is also an editor for the Machine Learning Blog at CMU and loves designing creative content in her free time."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-04-29-james-brundage.toml b/_data/talks/2022-04-29-james-brundage.toml
new file mode 100644
index 0000000..d128543
--- /dev/null
+++ b/_data/talks/2022-04-29-james-brundage.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Leveraging Unlabeled Data for Machine Learning in the Electrocardiogram"
+date = 2022-04-29
+start_time = "15:00"
+end_time = "16:00"
+series = "Data Science Seminar"
+location = "MEB 3147 (LCR)"
+zoom = "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Supervised deep learning (DL) has become an increasingly common tool for advanced analysis of the electrocardiogram (ECG). These methods rely heavily on labeled datasets, in which there is a clinical annotation for each ECG. However, real world ECG datasets may not contain enough labeled recordings to facilitate robust feature extraction, preventing DL analysis for clinical problems with small datasets. Self-supervised learning (SSL) seeks to utilize cheaply labeled or unlabeled data to improve performance in a supervised learning task. This process consists of first training a model on a primary task with cheap data labels, followed by a second training process which attempts to learn the downstream task by initializing with weights learned from the first. While SSL has become a popular tool in many machine learning domains, it is only starting to be used in ECG based machine learning. Here, we demonstrate the progress we have made in applying SSL approaches to detect low left ventricular ejection fraction, a complex ECG detection task, using data extracted from the University of Utah."
+
+[[speakers]]
+name = "James Brundage"
+affiliation = "UU HSC"
+website = ""
+photo = ""
+bio = "James Brundage is a second year medical student (MSII) at the University of Utah. He completed a BS and MS in neuroscience at BYU with an emphasis in cellular neuro-electrophysiology. Since starting at the University of Utah, he has worked in the MacLeod lab, coordinating and carrying out a collaborative effort between the Scientific Computing and Imaging Institute (SCII), Nora Eccles Cardiovascular Research and Training Institute (CVRTI) and the cardiology team at University of Utah Hospital focused on analysis of the electrocardiogram (ECG) using machine learning. Following medical school, he hopes to pursue a career as a physician scientist with a focus on applied machine learning in the clinical setting."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB206A01-E8F7-49FE-926B-ADBC53D718C4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-08-24-bei-wang-phillips.toml b/_data/talks/2022-08-24-bei-wang-phillips.toml
new file mode 100644
index 0000000..60e7e4b
--- /dev/null
+++ b/_data/talks/2022-08-24-bei-wang-phillips.toml
@@ -0,0 +1,40 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "On Hypergraph Analysis and Visualization"
+date = 2022-08-24
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Hypergraphs capture multi-way relationships in data, and they have
+consequently seen a number of applications in higher-order network
+analysis, computer vision, geometry processing, and machine learning.
+In this talk, I will discuss hypergraph analysis and visualization.
+In particular, I will focus on developing the theoretical foundations
+in studying the space
+of hypergraphs using ingredients from optimal transport.
+
+This talk is
+based on joint works
+with Youjia Zhou, Archit Rathore, Emilie Purvine, Samir Chowdhury, Tom
+Needham, and Ethan Semrad.
+"""
+
+[[speakers]]
+name = "Bei Wang Phillips"
+affiliation = "Utah SoC/SCI"
+website = "http://www.sci.utah.edu/~beiwang/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "DE624136-76EB-41DB-AAC2-8BBBA498B70A"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-08-31-casey-greene.toml b/_data/talks/2022-08-31-casey-greene.toml
new file mode 100644
index 0000000..88bf4ed
--- /dev/null
+++ b/_data/talks/2022-08-31-casey-greene.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "TBA"
+date = 2022-08-31
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Casey Greene"
+affiliation = "CU Anschutz"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-09-07-kevin-moon.toml b/_data/talks/2022-09-07-kevin-moon.toml
new file mode 100644
index 0000000..78bcd36
--- /dev/null
+++ b/_data/talks/2022-09-07-kevin-moon.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Scalable supervised manifold learning with random forests and neural networks"
+date = 2022-09-07
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "The manifold assumption has been used in many machine learning applications to combat the curse of dimensionality. Most manifold learning methods are unsupervised and typically focus on preserving the dominant structure and variation in the data. In many cases, we wish to analyze the data in a supervised setting with respect to expert-provided data labels. Most supervised manifold learning methods exaggerate the separation between data points of different classes, distorting the true structure of the data. In this talk, I will present RF-PHATE, a supervised dimensionality reduction method that preserves the true structure of the variables that are relevant for the supervised task. RF-PHATE is based upon a diffusion process applied to random forest proximities and is well-suited for data visualization. I will then show how to improve the scalability of RF-PHATE and any other manifold learning algorithm and perform out of sample extension using geometry regularized autoencoders (GRAE)."
+
+[[speakers]]
+name = "Kevin Moon"
+affiliation = "USU"
+website = "https://sites.google.com/a/umich.edu/kevin-r-moon/home"
+photo = ""
+bio = "Kevin Moon is an assistant professor at Utah State University in the department of mathematics and statistics. He received his B.S. and M.S. in Electrical Engineering from BYU while focusing on signal processing with minors in economics and math. He then received an M.S. in Mathematics and a PhD in Electrical Engineering from the University of Michigan where he worked with Dr. Alfred Hero on the problem of nonparametric estimation of distributional functionals. Prior to joining USU in 2018, he worked with Dr. Smita Krishnaswamy and Dr. Ronald Coifman as a postdoc at Yale University in the Genetics department and the Applied Math program where he developed methods for data visualization and exploratory data analysis with a focus in biomedical applications. His current research focuses on the development of theory and applications in machine learning, big data, information theory, deep learning, and data science in general. Applications of interest include biology (including medical), finance, ecology, engineering, and navigation."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-09-14-elliot-smith.toml b/_data/talks/2022-09-14-elliot-smith.toml
new file mode 100644
index 0000000..29d1882
--- /dev/null
+++ b/_data/talks/2022-09-14-elliot-smith.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Human neuronal population encoding of temporal difference learning variables during risky choices"
+date = 2022-09-14
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Recent research in AI showed that agents designed to predict the full distribution of potential rewards, rather than a central estimate of that distribution, generate richer learning distributions that allow them to perform better, especially on risky tasks. Such distributional reinforcement learning (distRL) was also discovered in dopamine neurons in the rodent ventral tegmental area. In this nanosymposium presentation, I will discuss recent work from direct brain recordings in neurosurgical patients undergoing monitoring for treatment of medically refractory epilepsy who performed a risky decision making task called the Balloon Analog Risk Task. Results from two studies will be presented: In the first study, we examined neuronal population recordings (157 neurons) from microelectrodes implanted in the anterior cingulate, orbitofrontal and temporal cortices (15 participants), finding that human prefrontal and mesial temporal neurons exhibited signatures of distRL: correlated diverse optimism in reward coding and diverse asymmetric scaling of reward prediction error. In the second study, we examined correlations between broadband high frequency local field potentials (an established correlate of population neuronal firing) and variables from temporal difference learning models for reward and risk while 37 participants made risky choices during BART. We found differences in which brain areas (3199 stereoelectroencephalography or electrocorticography contacts sampling frontal, temporal, and parietal lobes) encoded temporal difference learning model variables between participants who were more or less risk averse in their choices during BART. These areas included the left dorsolateral prefrontal, anterior cingulate, and orbitofrontal cortices. The results from these studies shed light on the neural underpinnings of human value learning in uncertain environments."
+
+[[speakers]]
+name = "Elliot Smith"
+affiliation = "Utah Neurology"
+website = "http://neurosmiths.org"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-09-21-jes-ford.toml b/_data/talks/2022-09-21-jes-ford.toml
new file mode 100644
index 0000000..1491f04
--- /dev/null
+++ b/_data/talks/2022-09-21-jes-ford.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Model Review: Improving Transparency, Reproducibility, & Knowledge Sharing using MLflow"
+date = 2022-09-21
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Code Review is an integral part of software development, but many teams don’t have similar processes in place for the development and deployment of Machine Learning (ML) models. I will motivate the decision to create a Model Review process, starting from the principles of transparency, reproducibility, and knowledge sharing. MLflow is a useful Python package to help simplify and automate much of the tracking necessary to create detailed records of machine learning experiments. Much of this talk will be spent introducing this tool, and demonstrating the core MLflow Tracking functionality. I’ll discuss how my team is currently running a Model Review process for any ML models that we push to production, and how we use MLflow to streamline this work and learn from each other."
+
+[[speakers]]
+name = "Jes Ford"
+affiliation = "Cash App"
+website = "http://jesford.github.io"
+photo = ""
+bio = "Jes Ford is a sponsored snowboarder turned astrophysicist turned data scientist. She enjoys applying Python data science tools to a wide variety of problems, and teaching skills and best practices to others. Jes completed her PhD in Physics at UBC Vancouver in 2015, and did a Postdoc in Data Science at the University of Washington under Jake VanderPlas. Currently based in Salt Lake City, she works remotely for Cash App (Block) as a Senior Machine Learning Engineer, and previously held local data science positions at Recursion and Backcountry. Jes spends her free time exploring the Wasatch mountains on snowboard, mountain bike, and foot. She has been involved in organizing the Salt Lake PyLadies chapter and the local Women in Data Science Conference, and is always looking for fun ways to be a part of her local tech community."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-09-28-prashant-pandey.toml b/_data/talks/2022-09-28-prashant-pandey.toml
new file mode 100644
index 0000000..05d17f5
--- /dev/null
+++ b/_data/talks/2022-09-28-prashant-pandey.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Scalability Challenges in Large-Scale Sequence Search"
+date = 2022-09-28
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Sequence-level searches on large collections of RNA sequencing experiments, such as the NCBI Sequence Read Archive (SRA), would enable one to ask many questions about the expression or variation of a given transcript in a population. Building an efficient and scalable sequence search index at the scale of SRA data is a challenging task and requires fundamental innovations in compression and scalable indexing. Recently, several tools have been proposed to index and search through SRA data but they offer various trade-offs in terms of space, speed, updatability, and accuracy. In this talk, I will present Mantis, a fast, exact, and updatable sequence search index. Mantis uses recent advancements in fast and compact hash tables, domain-specific data compression techniques, and scalable indexing to build a scalable and updatable index and supports fast sequence searches on ~40K experiments (>100TB) in size from SRA."
+
+[[speakers]]
+name = "Prashant Pandey"
+affiliation = "Utah SoC"
+website = "https://prashantpandey.github.io"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-10-05-jessica-shi.toml b/_data/talks/2022-10-05-jessica-shi.toml
new file mode 100644
index 0000000..1c6b3c2
--- /dev/null
+++ b/_data/talks/2022-10-05-jessica-shi.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Theoretically and Practically Efficient Parallel Nucleus Decomposition"
+date = 2022-10-05
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+We study the nucleus decomposition problem, which has been shown to be useful in finding dense substructures in graphs. We present a novel parallel algorithm that is efficient both in theory and in practice. Our algorithm achieves a work complexity matching the best sequential algorithm while also having low depth (parallel running time), which significantly improves upon the only existing parallel nucleus decomposition algorithm (Sariyuce et al., PVLDB 2018). The key to the theoretical efficiency of our algorithm is a new lemma that bounds the amount of work done when peeling cliques from the graph, combined with the use of theoretically-efficient parallel algorithms for clique listing and bucketing.
+
+We introduce several new practical optimizations, including a new multi-level hash table structure to store information on cliques space-efficiently and a technique for traversing this structure cache-efficiently. On a 30-core machine with two-way hyper-threading on real-world graphs, we achieve up to a 55x speedup over the state-of-the-art parallel nucleus decomposition algorithm by Sariyuce et al., and up to a 40x self-relative parallel speedup. We are able to efficiently compute larger nucleus decompositions than prior work on several million-scale graphs for the first time.
+"""
+
+[[speakers]]
+name = "Jessica Shi"
+affiliation = "MIT"
+website = "https://jeshi96.github.io"
+photo = ""
+bio = "Jessica Shi is a 4th-year PhD student at MIT in the EECS department, where she is advised by Julian Shun. She is also a Student Researcher at Google on the Graph Mining team, where she is mentored by Jakub Łącki. Jessica’s current research interests include developing shared-memory parallel graph algorithms with provable theoretical guarantees and efficient scalable implementations, with a focus on subgraph decomposition and clustering algorithms. She is supported by a 2018 NSF Graduate Research Fellowship, and she has previously received her S.M. in computer science from MIT, and her A.B. in mathematics from Princeton University."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-10-19-jie-zhang.toml b/_data/talks/2022-10-19-jie-zhang.toml
new file mode 100644
index 0000000..6f74481
--- /dev/null
+++ b/_data/talks/2022-10-19-jie-zhang.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Active Sampling for Min-Max Fairness"
+date = 2022-10-19
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Models satisfying Min-max fairness minimizes maximum group specific losses, so that the model has a more equitable performance over all groups. Benefit of min-max fair models over other fairness notions and models include it levels up: meaning it only degrades performance of a group if the degradation improves performance on the worst off group.
+
+In this talk, I will briefly mention the prior works that defined min-max fairness. I will mainly focus on our paper “Active Sampling for Min-Max Fairness” to introduce two algorithms that find min-max fair models with convergence guarantees.
+"""
+
+[[speakers]]
+name = "Jie Zhang"
+affiliation = "U Washington"
+website = ""
+photo = ""
+bio = "Claire Zhang is a PhD student at the Paul G. Allen School of Computer Science and Engineering at the University of Washington. She is fortunate to be advised by Professor Jamie Morgenstern. She completed her undergraduate studies at the University of Utah School of Computing, and was very fortunate to be advised by professor Suresh Venkatasubramanian while at the U. Her research interests include fairness in machine learning and learning theory."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-11-02-alex-chin.toml b/_data/talks/2022-11-02-alex-chin.toml
new file mode 100644
index 0000000..d1c9275
--- /dev/null
+++ b/_data/talks/2022-11-02-alex-chin.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Quantifying supply-demand imbalance in ridesharing systems"
+date = 2022-11-02
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "The status of the rider and driver distributions and how they interact in a two-sided marketplace has implications for market efficiency and policy-making. As such, accurately characterizing the supply-demand state of Lyft's marketplace is a crucial task. I will describe how this task can be framed in terms of an asymmetric optimal transport problem that yields a multi-resolution view of the supply-demand state and how it varies both spatially and temporally. I will then discuss how this approach can be incorporated into applications such as policy optimization and machine learning prediction problems. Time permitting I will also discuss other science efforts we have at Lyft."
+
+[[speakers]]
+name = "Alex Chin"
+affiliation = "Lyft"
+website = "http://alexchin.com"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-11-09-shireen-elhabian.toml b/_data/talks/2022-11-09-shireen-elhabian.toml
new file mode 100644
index 0000000..2aab887
--- /dev/null
+++ b/_data/talks/2022-11-09-shireen-elhabian.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Data-driven Shape Analysis: Methods, Applications, and Future"
+date = 2022-11-09
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Quantitative analysis of shapes is contingent upon defining a metric in the space of shapes to compare shapes and perform shape statistics. A growing consensus in the field that such a metric should be adapted to the specific population under investigation, begging for learning such a metric in a data-driven manner. This talk will cover a state-of-art data-driven approach for statistical shape modeling that provides unbiased, objective, and intuitive evaluation of geometric shapes, emphasizing anatomical structures reconstructed from volumetric images. I will talk about how we applied shape analysis to clinical and scientific questions in various ways. I will also highlight the role of machine learning in mitigating critical bottlenecks and significant barriers to making shape modeling a robust tool for on-demand clinical diagnostics and streamlining its adoption in research and practice. I will end with a future outlook for shape modeling to enable more complex and diverse modeling scenarios."
+
+[[speakers]]
+name = "Shireen Elhabian"
+affiliation = "Utah CS/SCI"
+website = "http://www.sci.utah.edu/~shireen/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-11-16-bernadette-stolz.toml b/_data/talks/2022-11-16-bernadette-stolz.toml
new file mode 100644
index 0000000..be4cb3f
--- /dev/null
+++ b/_data/talks/2022-11-16-bernadette-stolz.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Applications of global and local persistent homology for the shape of biological data"
+date = 2022-11-16
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "In the first part of this talk, I will showcase how persistent homology can be used to spatially characterise structural abnormality in tumour blood vessel networks. More specifically, I will show that the number of vessel loops and their distribution in these networks change over time when tumours undergo treatment with vascular targeting agents and radiation therapy. In the second part of the talk, I will speak about applications of local persistent homology. I will show how local persistent homology can be used to select landmarks from large and noisy data sets. In contrast to existing methods, this subsampling process is robust to outliers and is developed specifically for persistent homology. I will further introduce a novel method that can detect geometric anomalies, such as intersections or boundaries, in point cloud data sampled from intersecting surfaces. This detection is based on the computation of persistent homology in local annular neighbourhoods around points and is less sensitive to the size of the local neighbourhood and surface curvature than an existing method."
+
+[[speakers]]
+name = "Bernadette Stolz"
+affiliation = "Oxford"
+website = "https://www.maths.ox.ac.uk/people/bernadette.stolz"
+photo = ""
+bio = "Bernadette obtained her DPhil in 2020 from the Mathematical Institute at the University of Oxford. She is now a Postdoctoral Researcher at the Laboratory for Topology and Neuroscience at EPFL and a Visiting Research Fellow at the Mathematical Institute, University of Oxford. Previously she was a Postdoctoral Research Assistant at the Centre for Topological Data Analysis at the University of Oxford. In her research, she develops techniques in topological data analysis (TDA) to study biological data, in particular dynamical networks and spatial data. Her research can be broadly categorised into three main groups: 1) Developing TDA techniques to answer biological questions arising from experimental data. 2) Developing novel data science methods based on TDA. 3) Using TDA in combination with mechanistic models to link form and function in biological systems. Her expertise in TDA is complemented by an MSc in Mathematical Modelling and Scientific Computing (University of Oxford, 2014) and undergraduate degrees in Mathematics (University of Bern, 2012) and Molecular Medicine (University of Göttingen, 2009). Her research has been recognised with the L'Oréal-Unesco For Women in Science Rising Talent Award 2022, the Anile-ECMI Prize for best PhD thesis 2021, and the Mathematical Institute (University of Oxford) 2020 DPhil Thesis Prize."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-11-30-aaron-quinlan.toml b/_data/talks/2022-11-30-aaron-quinlan.toml
new file mode 100644
index 0000000..bc0aaf7
--- /dev/null
+++ b/_data/talks/2022-11-30-aaron-quinlan.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Finding signals of genome mutation in the noise of DNA sequencing error"
+date = 2022-11-30
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "The research in our laboratory is focused on the application of computational methods to develop a deeper understanding of genetic variation in diverse contexts. Modern experimental methods allow us to examine entire genomes with exquisite detail. Perhaps not surprisingly, staggering complexity is revealed as we look more closely at how genetic variation (both inherited and somatic) contributes to phenotypes. Modern genomic technologies necessitate efficient approaches for exploring, manipulating and comparing large genomic datasets. We develop such methods so that we and others may apply them to experiments investigating the impact of genetic variation on human disease, evolution, and somatic differentiation. Genome research is difficult - we strive to develop computational means that make it easier."
+
+[[speakers]]
+name = "Aaron Quinlan"
+affiliation = "Utah Human Genetics"
+website = "http://quinlanlab.org"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2022-12-07-tao-yang.toml b/_data/talks/2022-12-07-tao-yang.toml
new file mode 100644
index 0000000..f0daafb
--- /dev/null
+++ b/_data/talks/2022-12-07-tao-yang.toml
@@ -0,0 +1,30 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Optimizing Ranking Effectiveness and Fairness"
+date = 2022-12-07
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Advanced ranking techniques have led to improvements in AI-powered information services that significantly changed people's lives. For example, search engines that rank information according to their utilities to use's queries have helped billions of people better finish their tasks in daily work; recommendation systems that rank products/movies/news according to the user's interests have completely changed the way people discover information everyday. Therefore, how to construct and optimize ranking systems is one of the most important research problems in the field of Information Retrieval (IR). When optimizing ranking systems, there are two important criteria to measure the quality of result rankings in IR systems. The first criterion is ranking effectiveness, which refers to the ability of a ranking system to effectively present results based on their relevance to the users' needs. The second criterion is ranking fairness, which refers to the ability of a ranking system to present results fairly. For example, in job recommendation, if a ranking system only considers ranking effectiveness and ranks items solely according to relevance, a small number of top candidates will always be exposed to users and dominate users' attention as users usually only examine the top ranks. In such case, other candidates will rarely have the chance to be hired even when they are highly qualified for the job. Therefore, it is important to balance the effectiveness of ranked lists with the fairness in ranking optimization.
+In this talk, I will present my recent works on ranking effectiveness and fairness optimization. The talk will be two parts. In the first part of this talk, I will introduce works on sole-effectiveness optimization where I propose uncertainty-aware rank systems based on Bayes modelling. In the second part of this talk, I will introduce works on fairness-effectiveness joint optimization.
+"""
+
+[[speakers]]
+name = "Tao Yang"
+affiliation = "Utah SoC"
+website = "https://www.cs.utah.edu/~taoyang/"
+photo = ""
+bio = "Tao Yang is fourth year Ph.D. student from the University of Utah, supervised by Prof. Qingyao Ai and Prof. Jeff M Phillips. He mainly focuses on Information Retrieval (IR) and Machine Learning related topics. Especially, he mainly focuses on how to construct and optimize ranking systems while considering ranking effectiveness and fairness. His works have been published on top-tier IR conferences and journals, like SIGIR, WWW, CIKM,WSDM,TOIS,...."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "F4F316F8-30D6-4274-8E19-4C9D755200DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-01-11-george-vega-yon.toml b/_data/talks/2023-01-11-george-vega-yon.toml
new file mode 100644
index 0000000..7fd8563
--- /dev/null
+++ b/_data/talks/2023-01-11-george-vega-yon.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Prediction of Gene Functions by Leveraging Biological Insights with Mechanistic Machine Learning"
+date = 2023-01-11
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Biomedical sciences, in particular, bioinformaticians and computational biologists, are in a race to annotate the immense number of genes and gene products we are still learning from. In this talk, I present a new method for predicting gene functions using mechanistic machine learning. Mechanistic machine learning is a relatively new area of research where predictive ML-based algorithms are improved by incorporating domain knowledge via mechanistic models. Here, we use a theoretically-funded function evolution model that relies solely on phylogenetic trees from PantherDB and annotations from the Gene Ontology (GO) to make high-quality predictions. I will illustrate how combining the mentioned model with a large gene expression database (Bgee) in an ML model significantly improves prediction quality."
+
+[[speakers]]
+name = "George Vega Yon"
+affiliation = "Utah Epidemiology"
+website = "https://ggvy.cl/"
+photo = ""
+bio = "Dr. George G. Vega Yon is a research assistant professor of epidemiology at the University of Utah. Dr. Vega Yon is a methodologist that uses statistical computing tools to study complex systems. His research includes statistical models for social network analysis, agent-based models, and phylogenetics. Dr. Vega Yon has over ten years of experience working in data science creating scientific software, including multiple R packages in data visualization, network science, high-performance computing, and bayesian statistics. George is originally from Chile and obtained his Ph.D. in biostatistics from USC, an M.Sc. in Social Science from Caltech, and an M.A. in economics and public policy from Universidad Adolfo Ibáñez. (https://ggv.cl)"
+
+[meta]
+source = "google-calendar"
+calendar_uid = "8074B040-5817-49E8-B11C-D683AFFB00A2"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-01-18-echo-warner.toml b/_data/talks/2023-01-18-echo-warner.toml
new file mode 100644
index 0000000..b4b1ea8
--- /dev/null
+++ b/_data/talks/2023-01-18-echo-warner.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "College of Nursing, Division of Acute and Chronic Care"
+date = 2023-01-18
+start_time = "10:45"
+end_time = "12:00"
+series = "Data Science Seminar"
+location = "Virtual:"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Unproven health claims on the internet may have substantial influence on patient behaviors and decision making. For example, cancer patients with curable disease who pursue unproven cancer treatment in lieu of evidence-based approaches demonstrate 2-4 times higher mortality than patients who avoid unproven cancer treatment. Interventions to mitigate the impact of online health misinformation are desperately needed. The few interventions that have been tested do not consider the extent of exposure to misinformation online, primarily because individual estimates of exposure are based on self-report and are considered highly unreliable. Web-monitoring software is typically used by businesses to monitor remote employee productivity. Our paradigm shifting approach applies web-monitoring to quantify online cancer misinformation exposure. We will discuss the feasibility of using web-monitoring software to quantify exposure to online health information with a special focus on cancer symptom management and unproven cancer treatment misinformation. Specifically, we will review 1) characteristics of online cancer health misinformation and the impacts this exposure may have on cancer patient health outcomes, relationships, and finances 2) Ethical considerations of web-monitoring, and 3) methodological rigor and reproducibility of web-monitoring approaches for studying exposure to other types of health misinformation online."
+
+[[speakers]]
+name = "Echo Warner"
+affiliation = "Utah Nursing, HCI"
+website = "https://faculty.utah.edu/u0600488-ECHO_LYN_WARNER/hm/index.hml"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "16684726-DDEC-4F37-934E-D4B7E2A0489D"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-01-25-shweta-jain.toml b/_data/talks/2023-01-25-shweta-jain.toml
new file mode 100644
index 0000000..418edf2
--- /dev/null
+++ b/_data/talks/2023-01-25-shweta-jain.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Putting Parameterization into Practice"
+date = 2023-01-25
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Graphs are everywhere: social networks, protein interaction networks, citation networks, epidemic spread networks. Applications that use these graphs have progressively become more sophisticated, relying on getting fast and accurate solutions to graph-theoretic problems, many of which are NP-Hard. However, the sizes of today's graphs easily run into millions of vertices and edges, if not more. As a result, many classic algorithms are infeasible for such graphs.
+
+Fortunately, real-world graphs across different domains show a lot of common characteristics such as an abundance of triangles, low average distance between vertices (small-world property), low degeneracy (a measure of the sparsity of edges) etc. which we can leverage to design algorithms that are provably efficient given those parameters. I will demonstrate this in the context of clique counting and decomposition of graphs, which has applications in community detection, spam detection, fraud detection, gene module detection etc. The new algorithms not only massively improve the performance in practice but also improve our theoretical understanding of real-world graphs and help to bridge the gap between theory and practice.
+"""
+
+[[speakers]]
+name = "Shweta Jain"
+affiliation = "Utah SoC"
+website = "https://sjain12.github.io"
+photo = ""
+bio = "Shweta Jain is a Computing Innovation Fellow (CIFellow) at the University of Utah working with Prof. Blair D. Sullivan. She was previously a postdoc at the University of Illinois, Urbana-Champaign and prior to that she completed her Ph.D. in Computer Science at the University of California, Santa Cruz (UCSC), advised by Prof. Seshadhri Comandur. Her research interests are in graph mining, parameterized algorithms, randomized and approximation algorithms, and algorithms for massive data and the goal of her research is to use these tools to design algorithms that work well in practice and have provable guarantees. Her work has been recognized by two Best Paper Awards, the SIGKDD Best Dissertation Runner-Up Award 2021, and the Best Dissertation Award 2020 of the Computer Science Department at UCSC. She was also chosen as a Rising Star of EECS 2020."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "B29C877B-91F5-4381-8AAC-922C420AB72B"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-02-01-ana-marsovic.toml b/_data/talks/2023-02-01-ana-marsovic.toml
new file mode 100644
index 0000000..42ca34a
--- /dev/null
+++ b/_data/talks/2023-02-01-ana-marsovic.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "“AI” That Masters Language Could Reason about Negation"
+date = 2023-02-01
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "We have experienced firsthand the growing impact that AI technologies like ChatGPT have. It is agreed upon that risks involving such technologies must be managed. This is at a glance akin to how people handle safety-critical systems in, e.g., aviation. However, the rigorous principles of safety engineering are not easily applicable to AI technologies. AI-backed solutions are obscured in AI’s internals, and the exact requirements for AI safety are unverifiable since they are neither defined nor regulated. One technique that has emerged as a step to measure an aspect of AI safety is to construct data that represents a specific phenomenon that a safe model must handle well, and that has no spurious correlations. Low performance on the dataset is undesired.In this talk, I’ll focus on one common linguistic phenomenon, negation, without which the full power of human language-based communication cannot be realized. I will show how we carefully constructed a question-answering dataset, CondaQA, to study how well current models (described by the New York Time Magazine as “mastering language”) reason about negated statements. An InstructGPT model (the latest GPT model before Nov 28, 2022) combined with chain-of-thought prompting achieves 66.28% accuracy and 27.28% consistency on CondaQA, way behind human accuracy of 91.94% and consistency of 81.58%."
+
+[[speakers]]
+name = "Ana Marsovic"
+affiliation = "Utah SoC"
+website = "https://www.anamarasovic.com"
+photo = ""
+bio = """
+Ana Marasović is an Assistant Professor in the Kahlert School of Computing at the University of Utah. Her primary research interests are at the confluence of NLP, explainable AI, and multimodality. She aims to rigorously validate AI technologies and make human interaction with AI more intuitive. She was a Young Investigator at the Allen Institute for AI from 2019–2022. During that time, she also had a courtesy appointment in the Paul G. Allen School of Computer Science & Engineering at the University of Washington. She obtained her PhD in 2019 from Heidelberg University.
+
+Name pronunciation: Ah-nah Mara-so-veetch, with “Mara” as the actress “Mara Wilson”
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "31771BA2-4E2B-4FF4-91EC-04D6A8318D65"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-02-08-aaron-clauset.toml b/_data/talks/2023-02-08-aaron-clauset.toml
new file mode 100644
index 0000000..18d6054
--- /dev/null
+++ b/_data/talks/2023-02-08-aaron-clauset.toml
@@ -0,0 +1,35 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Meritocracy or systemic bias? Untangling the drivers the productivity and prominence among scientists"
+date = 2023-02-08
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Simple measures of scholarly productivity and prominence vary enormously across both individual scientists and institutions -- but to what degree do these inequalities represent genuine meritocratic differences vs. systemic biases that limit scientific progress?
+
+In this talk, I'll describe a sequence of results that substantially untangle the underlying systemic drivers of productivity and prominence among scientists. First, I'll show that productivity and prominence are, to a significant degree, environmental variables such that the prestige of a scientist's working environment drives their individual productivity, largely by providing larger research groups to elite scientists. Second, I'll describe a network-based generative model of individual productivity and prominence that untangles these measures from their underlying collaboration networks. These models corroborate the labor-advantage hypothesis of elite institutions, and also reveal both that gendered differences in the productivity and prominence of mid-career researchers can be largely explained by gendered differences in coauthorship networks, and that these networks are partially transferable from senior to junior collaborators. Hence, collaboration networks, and the systemic factors that shape them, play a critical role in driving scholarly inequalities in science, and suggest that these networks are an important form of unequally distributed social capital that shapes who makes what scientific discoveries. I'll close with a discussion of policies that could potentially mitigate the unequal distribution of this social capital and help both diversify the academy and broaden its contributions to society.
+"""
+
+[[speakers]]
+name = "Aaron Clauset"
+affiliation = "UC Boulder"
+website = "https://www.colorado.edu/cs/aaron-clauset"
+photo = ""
+bio = """
+Aaron Clauset is a Professor in the Department of Computer Science and the BioFrontiers Institute at the University of Colorado Boulder, and is External Faculty at the Santa Fe Institute. He received a PhD in Computer Science, with distinction, from the University of New Mexico, a BS in Physics, with honors, from Haverford College, and was an Omidyar Fellow at the prestigious Santa Fe Institute. In 2016, he was awarded the Erdos-Renyi Prize in Network Science, and since 2017, he has been a Deputy Editor responsible for the Social, Computing, and Interdisciplinary Sciences at Science Advances.
+
+Clauset is an internationally recognized expert on network science, data science, and machine learning for complex systems. His research program is around two general themes: identifying fundamental principles of the organization and behavior of complex social and biological systems, and developing approaches for using data and computation to illuminate those ideas. A recent major focus of this work has been on the "science of science," where he studies the shape, origins, and consequences of social and epistemic inequalities on scientific careers, productivity, the spread of ideas, and the composition of the scientific workforce. His research results have appeared in many prestigious scientific venues, including Nature, Science, PNAS, SIAM Review, Science Advances, Nature Communications, AAAI, and ICDM. His work has been covered in the popular press by Quanta Magazine, the Wall Street Journal, The Economist, Discover Magazine, Wired, the Boston Globe and The Guardian.
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "24519C76-7D93-4F65-A4B4-2D165951C6FB"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-02-15-orly-alter.toml b/_data/talks/2023-02-15-orly-alter.toml
new file mode 100644
index 0000000..fddf9ee
--- /dev/null
+++ b/_data/talks/2023-02-15-orly-alter.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Solving Cancer with Data: Mathematical Discovery and Computational and Experimental Validation of Whole-Genome Genotype–Survival and Response to Treatment Phenotype Relationships in Cancer"
+date = 2023-02-15
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "WEB 3780"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "1/2 of men and 1/3 of women will face cancer, a disease of the whole 3B-nucleotide genome. But, despite the availability of open-source data and the $100/1-hour genome, genetic tests remain limited to one to a few hundred genes. Therefore, the prognosis, diagnosis, and treatment of cancer remain unchanged. This is due to the lack of suitable AI/ML. I will describe work in my lab inventing AI/ML that connects the whole genome with a patient’s survival and response to treatment. Our algorithms discover accurate, precise, and interpretable predictors, applicable to the general population, from as few as 50–100 patients. Our predictors outperform all other indicators, where they exist. All other methods miss them. I will describe my international retrospective clinical trial, which validated a genome-wide pattern in tumors from glioblastoma brain cancer patients as the best predictor of life expectancy and response to standard of care. We discovered this, and predictors in, e.g., adult lung, ovarian, and uterine adenocarcinoma tumors and pediatric nerve neuroblastoma tumors, in public data, proving that the algorithms and predictors are uniquely suited to personalized medicine. I will also describe work translating the algorithms and predictors to the clinic."
+
+[[speakers]]
+name = "Orly Alter"
+affiliation = "Utah BME, SCI"
+website = "https://alterlab.org/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9812EE31-49DA-42AB-9699-04C66AC522D1"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-02-22-vivek-gupta.toml b/_data/talks/2023-02-22-vivek-gupta.toml
new file mode 100644
index 0000000..ab862e4
--- /dev/null
+++ b/_data/talks/2023-02-22-vivek-gupta.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Inference and Reasoning for Semi-structured Tables"
+date = 2023-02-22
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Understanding semi-structured tabular data, which is ubiquitous in the real world, requires an understanding of the meaning of text fragments and the implicit connections between them. We believe such data could be used to investigate how individuals and machines reason about semi-structured data. First, we present the InfoTabS dataset, which consists of human-written textual predictions based on tables collected from Wikipedia's infoboxes. Our research demonstrates that the semi-structured, multi-domain, and heterogeneous nature of the premises prompts complicated, multi-faceted reasoning, offering a modeling challenge for traditional modeling techniques. Second, we analyzed these challenges in-depth and developed simple, effective preprocessing strategies to overcome them. Thirdly, despite accurate NLI prediction, we demonstrate through rigorous probing that the existing model does not reason with the provided tabular facts. To address this, we suggest a two-stage evidence extraction and tabular inference technique for enhancing model reasoning and interpretability. We also investigate efficient methods for enhancing tabular inference datasets with semi-automatic data augmentation and pattern-based pre-training. Lastly, to ensure that tabular reasoning models work in more than one language, we introduce XInfoTabS, a unique problem of bilingual tabular inference, and a cost-effective pipeline for translating tables. In the near future, we plan to test the tabular reasoning model for temporal changes, especially for dynamic tables where information changes over time."
+
+[[speakers]]
+name = "Vivek Gupta"
+affiliation = "Utah SoC"
+website = "https://vgupta123.github.io"
+photo = ""
+bio = "Vivek is a fifth-year doctorate candidate at the Utah NLP Group's at Kahlert School of Computing, University of Utah. He is fortunate to be advised by Prof. Vivek Srikumar. He is broadly interested in NLP research in semi-structured data and low-resource languages. He is awarded Bloomberg Data Science Fellowship 2021-23, the Best paper award at the DeeLIO 2022 workshop, and the Outstanding paper award at the NLP4ConvAI 2022 workshop . He currently also serves as the Utah Data Science Club's coordinator. He used to be a Research Fellow (Microsoft Research Fellowship 2016–18) at the Microsoft Research Lab, India, where he worked with the Machine Learning and Natural Language Processing group. In 2016, he graduated from IIT Kanpur as a dual degree (BS-MS) student in the Department of Computer Science and Engineering. He was the inaugural coordinator of IIT Kanpur's Special Interest Group in Machine Learning (SIGML)"
+
+[meta]
+source = "google-calendar"
+calendar_uid = "BDC0531F-2350-4889-AF6F-868D4FEC9570"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-03-01-emily-hadley.toml b/_data/talks/2023-03-01-emily-hadley.toml
new file mode 100644
index 0000000..fb074a1
--- /dev/null
+++ b/_data/talks/2023-03-01-emily-hadley.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Applied Strategies for Advancing Racial Equity and Addressing Bias in Big Data Research"
+date = 2023-03-01
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "In the last decade, big data research studies have proliferated and, in some cases, offered considerable promise for humanity. Yet, numerous incidents have documented that without safeguards, this same research can reproduce and amplify existing societal biases and disparities. Given the increased public awareness of structural racism, bias based on race and ethnicity in big data research is particularly concerning. Big data researchers can mitigate the risk of perpetuating bias based on race and ethnicity by intentionally incorporating best practices that reduce or eliminate racial biases. We synthesize key findings and applied recommendations from over 140 sources for addressing race and ethnicity bias in big data research. We discuss considerations when planning a big data project, including identifying study motivations, centering participatory involvement, and addressing key concerns regarding race and ethnicity in big data collection and quality. We detail issues related to proxy discrimination, algorithmic audits, data completeness, and the Big Data Paradox. We provide real-world examples of advancing racial equity and addressing bias and share recommendations for additional resources and opportunities for further investigation."
+
+[[speakers]]
+name = "Emily Hadley"
+affiliation = "RTI International"
+website = ""
+photo = ""
+bio = "Emily Hadley is a Research Data Scientist with the RTI International Center for Data Science. Her work spans several practice areas including health, education, social policy, and criminal justice. She has experience with machine learning, natural language processing, and predictive analytics, and a passion for antiracism, bias, and equity in data science."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "5DD80F98-8A59-44B0-BA61-8BEA1C0195DE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-03-15-nate-veldt.toml b/_data/talks/2023-03-15-nate-veldt.toml
new file mode 100644
index 0000000..2d6329f
--- /dev/null
+++ b/_data/talks/2023-03-15-nate-veldt.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Measuring homophily in group interactions: hypergraph models and combinatorial impossibilities"
+date = 2023-03-15
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Homophily is the well-known sociological principle that people tend to connect with others who are similar to them, or more informally: \"birds of a feather flock together.\" Although many social interactions occur in groups, homophily is typically measured using a graph, which only accounts for interactions involving two individuals. This talk will present a new hypergraph framework for more directly measuring homophily in group settings. Our measures highlight natural patterns in group homophily that appear with gender in scientific collaboration and political affiliation in legislative bill cosponsorship, and also reveal distinctive gender distributions in group photographs, all of which cannot be fully captured by graph-based measures. We will also discuss subtle combinatorial limits and impossibilities that arise when measuring homophily in hypergraphs, which are completely independent of human behavior and must be properly accounted for in order to understand how homophily can (and cannot) be manifested in group interactions."
+
+[[speakers]]
+name = "Nate Veldt"
+affiliation = "Texas A&M"
+website = "https://veldt.engr.tamu.edu"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "66247255-5B4E-4099-8DFC-472B2C18FDC8"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-03-22-bailey-fosdick.toml b/_data/talks/2023-03-22-bailey-fosdick.toml
new file mode 100644
index 0000000..50a88f8
--- /dev/null
+++ b/_data/talks/2023-03-22-bailey-fosdick.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Modeling Infection Fatality Rates to Assess the Burden of COVID-19 in Developing Countries"
+date = 2023-03-22
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "COVID-19 spread quickly around the world after first being discovered in China in late 2019. It has had devastating impacts, however its impacts, both in terms of infection prevalence and fatalities, have been non-uniformly distributed worldwide. While early studies focused on COVID-19 infection and fatality rates in high-income countries, much less attention has been given to the impacts of COVID-19 in developing countries. In this work, we systematically reviewed the literature to identify all COVID-19 serology studies conducted by early 2021 using population representative samples. We developed a Bayesian hierarchical model for simultaneously modeling serology and death data to make inference on age-specific seroprevalence and age-specific infection fatality rates. This model directly accounts for conventional sampling uncertainty, as well as uncertainty about the serological test assay sensitivity and specificity. Through a careful analysis of data from over thirty developing countries, we found seroprevalence in many developing country locations was markedly higher than in high-income countries early in the pandemic and age-specific infection fatality rates were roughly twice as high as that in high-income countries."
+
+[[speakers]]
+name = "Bailey Fosdick"
+affiliation = "CU Anschutz"
+website = "https://www.baileyfosdick.com"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "9C979B8F-372B-48DA-AF62-A3707E54913E"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-03-29-justin-baker.toml b/_data/talks/2023-03-29-justin-baker.toml
new file mode 100644
index 0000000..f26dbdb
--- /dev/null
+++ b/_data/talks/2023-03-29-justin-baker.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Monotone Implicit Graph Neural Networks for Long-Range Dependency Learning"
+date = 2023-03-29
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "From social networks to chemical engineering, deep graph neural networks play an instrumental role in advancing our industrial and scientific frontier. Of particular interest are networks which can learn long range dependencies in a scalable and expressive manner. In this talk, we will delve into the power of deep learning on graphs, with a focus on implicit graph neural networks (IGNNs) and their scalability. We will also discuss how monotone operator theory enhances the expressivity of IGNNs, overcoming a crucial obstacle to learning long-range dependencies. By doing so, monotone IGNNs can significantly improve graph learning and have the potential to make breakthroughs in various fields."
+
+[[speakers]]
+name = "Justin Baker"
+affiliation = "Utah Math & SCI"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "313AB60F-2B8F-4271-80F1-86D4D96A031D"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-04-05-yao-yaun-mao.toml b/_data/talks/2023-04-05-yao-yaun-mao.toml
new file mode 100644
index 0000000..56bbdae
--- /dev/null
+++ b/_data/talks/2023-04-05-yao-yaun-mao.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Searching for dwarf (small) galaxies in astronomical surveys"
+date = 2023-04-05
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Dwarf galaxies are small fuzzy galaxies that contain much fewer stars than Milky Way-mass galaxies. Observations of these little galaxies can enhance our understanding of galaxy formation and the nature of dark matter. Finding these dwarf galaxies is, however, not an easy task because they are faint and dim from our perspective. I will describe the challenges and recent efforts on the search for nearby dwarf galaxies in astronomical surveys. One particular challenge is to identify potential dwarf galaxies with only image data that do not contain distance information. I will discuss a few traditional methods that are used to identify dwarf galaxies and obtain their astronomical distances, and why these methods are mostly used to find dwarf galaxies in specific patches of the sky. I will then discuss how the distance information can be used as training data in machine learning algorithms, such as convolutional neural networks, to derive distance information from just astronomical images. Finally, I will discuss the remaining challenges we have, and how we may improve the methods in preparation for future observations from the Rubin Observatory Legacy Survey of Space and Time (LSST) and Roman Space Telescope."
+
+[[speakers]]
+name = "Yao-Yaun Mao"
+affiliation = "Utah Astro"
+website = "https://yymao.github.io"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "A76D8FFE-A462-49A1-A7DC-AF4A37B83B46"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-04-12-titus-brown.toml b/_data/talks/2023-04-12-titus-brown.toml
new file mode 100644
index 0000000..445c63b
--- /dev/null
+++ b/_data/talks/2023-04-12-titus-brown.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Is everything everywhere all at once? Asking questions of all public microbiome shotgun data"
+date = 2023-04-12
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = "MEB 3147"
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Public sequence data offers many opportunities for reuse, exploration, and discovery. What happens if you make it really, really easy to search the content of all the public microbiome data sets? It turns out you can enable some interesting science, but you also need to address many technical, social, and policy issues. In this talk I’ll showcase some of our results from being able to search everything, everywhere, all at once; describe our current efforts; and discuss the possible opportunities and challenges of petabyte-scale sequence search."
+
+[[speakers]]
+name = "Titus Brown"
+affiliation = "UC Davis"
+website = "http://ivory.idyll.org/lab/"
+photo = ""
+bio = "C. Titus Brown is a Professor at the School of Veterinary Medicine at UC Davis, where he works on effective large scale sequencing data analysis, methods development, training, and open science. His most recent work focuses enabling on petabase-scale search of all available public microbiome data, with the goal of better hypothesis generation and refinement. He tweets at @ctitusbrown and blogs at http://ivory.idyll.org/blog/. His google scholar profile is reasonably up to date but is nevertheless highly misleading."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "843CEAB8-FFC3-4944-92ED-6AF3DC3A9BD0"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-04-19-sumana-basu.toml b/_data/talks/2023-04-19-sumana-basu.toml
new file mode 100644
index 0000000..43be390
--- /dev/null
+++ b/_data/talks/2023-04-19-sumana-basu.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Towards Reinforcement Learning for Precision Drug Dosing"
+date = 2023-04-19
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Drug dosing is an important application of AI, which can be formulated as a Reinforcement Learning (RL) problem, since every individual’s drug dosing requirement is different. In this talk, we will talk about two major challenges of using RL for drug dosing: delayed and prolonged effects of medications, which break the Markov assumption of the RL framework. We will talk about an approach to solve this problem in a model free reinforcement learning setting, talk further about the challenges of deploying it in real life and sketch the outline of a more realistic Model Based Reinforcement Learning (MBRL) solution to it."
+
+[[speakers]]
+name = "Sumana Basu"
+affiliation = "McGill University"
+website = "https://scholar.google.com/citations?view_op=view_org&hl=en&org=13784427342582529234"
+photo = ""
+bio = "Sumana is a PhD student at McGill University (Mila), researching Deep Reinforcement Learning in Healthcare. Her work focuses on applying Reinforcement Learning to Autonomous Drug Dosing. She completed her Masters at Mila, where she studied deep learning for predicting Alzheimer's disease progression. She has also interned in the past at Meta AI (FAIR) on the fastMRI Active Acquisition project, using Reinforcement Learning to accelerate MRI acquisition, and will be interning at Microsoft Research Labs coming summer, on exploring Reinforcement Learning for optimizing genetic perturbation in Cancer Treatment."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EFA7C0AA-5911-4AA0-A36D-53125411A131"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-05-30-bodhisattwa-majumder.toml b/_data/talks/2023-05-30-bodhisattwa-majumder.toml
new file mode 100644
index 0000000..1598b13
--- /dev/null
+++ b/_data/talks/2023-05-30-bodhisattwa-majumder.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "User-centric Natural Language Processing"
+date = 2023-05-30
+start_time = "15:00"
+end_time = "16:30"
+series = "Data Science Seminar"
+location = "Where: MEB 3147 Large Conference Room Simcast: (Meeting ID: 965 3805 0936, Passcode: 404653)"
+zoom = "https://utah.zoom.us/j/96538050936"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Artificial intelligence (AI) has shown remarkable effectiveness in knowledge-seeking applications (e.g., for recommendations and explanations). However, the increasing expectation of more trust, accessibility, and anthropomorphism in these AI systems requires the underlying components (dialog models, LLMs, classifiers) to be adaptive and adequately knowledge grounded. In reality, the outputs of the constituent models often lack commonsense, explanations, and subjectivity, which motivates us to ask the question: what can we achieve by redesigning AI systems to start with individual needs?
+
+Ideally, an assistive AI system must be aware of the surrounding world, produce faithful explanations, and align with the user's preferences. In this talk, I will discuss a post-hoc knowledge-injection technique that enriches the dialog responses at the decoding time and promotes achieving conversational goals. Then, I will explore how to elevate existing AI systems using a unified framework to map low-level and abstractive explanations by background knowledge. Finally, I will hint at a user-centric interventionist approach that can help users obtain more equitable predictions backed by faithful explanations as compared to a black-box counterpart. I will conclude with the future possibilities and societal impacts of next-generation user-centric systems.
+
+bodhi_flyer.png
+"""
+
+[[speakers]]
+name = "Bodhisattwa Majumder"
+affiliation = "UCSD"
+website = "http://nlp.cs.utah.edu/"
+photo = ""
+bio = "Bodhisattwa Prasad Majumder (https://www.majumderb.com/) recently received his Ph.D. in Computer Science from UC San Diego and was advised by Prof. Julian McAuley. His research goal is to build safe, trustworthy, and user-centric interactive systems. He previously spent time at the Allen Institute of AI, Google AI, Microsoft Research, and FAIR (Meta AI), along with collaborations from U of Oxford, U of British Columbia, and the Alan Turing Institute. His work has been recognized by the UCSD CSE Doctoral Award for Research, Adobe Research Fellowship, Qualcomm Innovation Fellowship, and Highlights of ACM RecSys, among many awards and several media coverages. In 2019, Bodhi led UCSD in the finals of the Amazon Alexa Prize. He also co-authored a best-selling NLP book with O’Reilly Media that is being adopted in universities internationally."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "2216ru1s1mq9h665i4cbdm1f9q@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-08-23-data-science-lecture-series.toml b/_data/talks/2023-08-23-data-science-lecture-series.toml
new file mode 100644
index 0000000..f520587
--- /dev/null
+++ b/_data/talks/2023-08-23-data-science-lecture-series.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Zoom link: https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+date = 2023-08-23
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "Warnock Engineering Building (Room: 3780), 72 Central Campus Dr, Salt Lake City, UT 84112, USA"
+zoom = "https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Data Science Lecture Series"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-08-30-data-science-lecture-series-speaker.toml b/_data/talks/2023-08-30-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..5062c9f
--- /dev/null
+++ b/_data/talks/2023-08-30-data-science-lecture-series-speaker.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Improving Fairness of Information Access in Networks"
+date = 2023-08-30
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "In social networks, node position is a form of social capital which enables faster and more reliable access to diverse information. Structural biases often arise from network formation and can lead to significant disparities in information access based on position. We discuss ways to quantify this social capital through the lens of information flow in the network, focusing on the setting where each node may be a source of distinct desirable information. We define several measures of access advantage, and consider the problem of improving equity by making interventions in the network, focusing on the case of edge augmentation. We describe several heuristic strategies for budgeted intervention, and present the results of an empirical evaluation on a corpus of real-world networks."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Blair D. Sullivan is a Professor in the School of Computing at the University of Utah. Prior to joining Utah, Dr. Sullivan was an Associate Professor at NC State University, and before that a Research Scientist at Oak Ridge National Laboratory. She received her Ph.D. in Mathematics from Princeton University in 2008 as a Department of Homeland Security Graduate Fellow, and B.S. degrees in Applied Mathematics and Computer Science from Georgia Tech in 2003. Sullivan’s research cross-cuts the fields of data-driven science, parameterized graph algorithms, network science, and algorithm engineering with a recent focus on problems arising in computational genomics. In 2014, Sullivan was named one of 14 Moore Investigators in Data-Driven Discovery. She currently serves on the Steering Committee for SODA, and was recently elected Chair of the SIAM SIAG on Applied & Computational Discrete Algorithms."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-08-30-data-science-lecture-series.toml b/_data/talks/2023-08-30-data-science-lecture-series.toml
new file mode 100644
index 0000000..4ff7995
--- /dev/null
+++ b/_data/talks/2023-08-30-data-science-lecture-series.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Zoom link: https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+date = 2023-08-30
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = "https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Data Science Lecture Series"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-09-06-data-science-lecture-series-speaker.toml b/_data/talks/2023-09-06-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..02f794e
--- /dev/null
+++ b/_data/talks/2023-09-06-data-science-lecture-series-speaker.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Blame the data, not the system: how data constraints can help in trustworthy machine learning and explain causes of data-system malfunction"
+date = 2023-09-06
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "The core of modern data-driven systems comprises models learned from large datasets, and they are usually optimized to target particular data and workloads. While these data-driven systems have seen wide adoption and success, their reliability and proper function hinge on the data's continued conformance to the systems initial settings and assumptions. My research focuses on designing mechanisms to assess the trustworthiness of a system's inferences and explain causes of system malfunction due to data nonconformance. The key idea here is that since data is central to data-driven systems, it can guide us to determine whether predictions made by an ML model can be trusted, and to expose the cause of a system's unexpected behavior. In this talk, I will talk about mechanisms and explanation frameworks to facilitate trusting and understanding outcomes involving data and data systems."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-09-13-data-science-lecture-series-speaker.toml b/_data/talks/2023-09-13-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..205f731
--- /dev/null
+++ b/_data/talks/2023-09-13-data-science-lecture-series-speaker.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Dynamic Graph Sketching: To Infinity And Beyond"
+date = 2023-09-13
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Existing graph stream processing systems must store the graph explicitly in RAM which limits the scale of graphs they can process. The graph semi-streaming literature offers algorithms which avoid this limitation via linear sketching data structures that use small (sublinear) space, but these algorithms have not seen use in practice to date. In this talk I will explore what is needed to make graph sketching algorithms practically useful, and as a case study present a sketching algorithm for connected components and a corresponding high-performance implementation. Finally, I will give an overview of the many open problems in this area, focusing on improving query performance of graph sketching algorithms."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = "David is the 2023 Grace Hopper Postdoctoral Fellow at Lawrence Berkeley Lab and his research focuses on compact, dynamic, and memory-hierarchy-aware algorithms for large-scale data science. Prior to that, he was a CRA Computing Innovation Postdoctoral Fellow working with Martin Farach-Colton at Rutgers University. He earned his PhD at UMass Amherst working with Andrew McGregor."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-09-20-data-science-lecture-series-speaker.toml b/_data/talks/2023-09-20-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..4b29678
--- /dev/null
+++ b/_data/talks/2023-09-20-data-science-lecture-series-speaker.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Data Preparation: The Biggest Roadblock in Data Science"
+date = 2023-09-20
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "When building Machine learning (ML) models, data scientists face a significant hurdle: data preparation. ML models are exactly as good as the data we train them on. Unfortunately, data preparation is tedious and laborious because it often requires human judgment on how to proceed. In fact, data scientists spend at least 80% of their time locating the datasets they want to analyze, integrating them together, and cleaning the result.In this talk, I will present my key contributions in data preparation for data science, which address the following problems: (1) data discovery: how to discover data of interest from a large collection of heterogeneous tables (e.g., data lakes); (2) error detection: how to find errors in the input and intermediate data in complex data workflows; and (3) data repairing: how to repair data errors with minimal human intervention. The developed systems are specifically designed to support data science development which poses particular requirements such as interactivity and modularity."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-09-25-sandia-information-session.toml b/_data/talks/2023-09-25-sandia-information-session.toml
new file mode 100644
index 0000000..f1f10af
--- /dev/null
+++ b/_data/talks/2023-09-25-sandia-information-session.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "TBA"
+date = 2023-09-25
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science Seminar"
+location = "WEB 2460"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Sandia Information Session"
+affiliation = "pizza"
+website = "https://tinyurl.com/2244r966"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "A80A4AD1-1725-4C47-97F1-35BB0A9E486A"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-09-27-data-science-lecture-series-speaker.toml b/_data/talks/2023-09-27-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..5fbeb8b
--- /dev/null
+++ b/_data/talks/2023-09-27-data-science-lecture-series-speaker.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Loss Minimization and Multi-group Fairness"
+date = 2023-09-27
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Training a predictor to minimize a loss function fixed in advance is the dominant paradigm in machine learning. However, loss minimization by itself might not guarantee desiderata like fairness and accuracy that one could reasonably expect from a predictor. In contrast, various group-fairness notions have been propsoed that constrain the predictor to share certain statistical properties of the data, even when conditioned on a rich family of subgroups. There is no explicit attempt at loss minimization.In this talk, we will explore some recently discovered connections between loss minimization and notions of multi-group fairness. We will see settings where one can lead to the other, and other settings where this is unlikely."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = "https://parikg.github.io/"
+photo = ""
+bio = "Parikshit Gopalan (https://parikg.github.io/) is a machine learning researcher at Apple (https://machinelearning.apple.com/). His current interests are in fairness in machine learning, unsupervised learning and algorithms/systems for big data. In the past, he has made important contributions to erasure coding for distributed storage, coding theory and computational complexity. His work has been awarded the 2014 Joint IEEE Communication Society & Information Theory Society Paper Prize, 2013 Microsoft TCN Storage Technical Award and the best paper award for the 2012 USENIX Advanced Technology Conference. In the past, he has been a researcher at VMware, Microsoft Research (Silicon valley and Redmond), a postdoc at the University of Washington and UT Austin, a graduate student at Georgia Tech and an undergraduate at IIT Bombay."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-10-04-data-science-lecture-series-speaker.toml b/_data/talks/2023-10-04-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..6670252
--- /dev/null
+++ b/_data/talks/2023-10-04-data-science-lecture-series-speaker.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Leveraging the Structure of Data"
+date = 2023-10-04
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Although predictions from machine learning models influence more and more of our lives, the standard way of posing a ML problem has remained relatively unchanged for decades. In the search for better models, a new and popular family of techniques (sometimes called Graph Machine Learning) has emerged. These techniques rely on expanding beyond the features of an individual entity and instead look to pull information from its relationships. The methods offer a tantalizing way of improving task performance by leveraging previously unused information. However, it is not a free lunch, as these models can be more complex, difficult to train, and may have challenges in interpretability. This talk will discuss the fundamentals of graph machine learning, a few models, and some insights from years of real-world applications."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = "such as NeurIPS, ICML, ICLR, KDD, and WWW"
+website = "https://ai.google/research/teams/algorithms-optimization/"
+photo = ""
+bio = """
+Bryan Perozzi is a Research Scientist in Google Research’s Algorithms and Optimization (https://ai.google/research/teams/algorithms-optimization/) group, where he routinely analyzes some of the world’s largest (and perhaps most interesting) graphs. Bryan’s research (https://scholar.google.com/citations?hl=en&user=rZgbMs4AAAAJ&view_op=list_works) focuses on developing techniques for learning expressive representations of relational data with neural networks. These scalable algorithms are useful for prediction tasks (classification/regression), pattern discovery, and anomaly detection in large networked data sets.
+
+Bryan is an author of 40+ peer-reviewed papers at leading conferences in machine learning and data mining (such as NeurIPS, ICML, ICLR, KDD, and WWW). His doctoral work on learning network representations was awarded the prestigious SIGKDD Dissertation Award. Bryan received his Ph.D. in Computer Science from Stony Brook University in 2016, and his M.S. from the Johns Hopkins University in 2011.
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-10-18-data-science-lecture-series-speaker.toml b/_data/talks/2023-10-18-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..ae278dc
--- /dev/null
+++ b/_data/talks/2023-10-18-data-science-lecture-series-speaker.toml
@@ -0,0 +1,35 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "DBSP: A formal model for streaming computation and its applications to incremental computations and databases"
+date = 2023-10-18
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = "https://utah.zoom.us/j/91737198805pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+DBSP is a simple streaming programming language inspired by Digital Signal Processing [DSP]. DBSP can be used to give a precise definition of incremental computations -- operating on changes (deltas, diffs). Moreover, given a DBSP program, a simple algorithm can convert it to a DBSP program that computes on changes. All practical database query operators (the relational algebra, group-by, aggregations, fixed-points, etc) can be expressed in DBSP. As a consequence we obtain an algorithm which can incrementalize essentially any database query. The DBSP theory has been formally verified using a theorem prover, making it the first verified theory of incremental view maintenance.
+
+The DBSP paper has received the best paper award at the 2023 conference on Very Large Databases [VLDB].
+"""
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = """
+Mihai Budiu is chief scientist at Feldera. He has a Ph.D. in CS from Carnegie Mellon University. He was previously employed at VMware Research, Barefoot Networks, and Microsoft Research. Mihai has worked on reconfigurable hardware, computer architecture, compilers, security, distributed systems, big data platforms, large-scale machine learning, programmable networks and P4, data visualization, and databases; four of his papers have received “test of time” awards. He has also received two technology transfer awards.
+
+Zoom link: https://utah.zoom.us/j/91737198805pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09)
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-10-25-data-science-lecture-series-speaker.toml b/_data/talks/2023-10-25-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..3731745
--- /dev/null
+++ b/_data/talks/2023-10-25-data-science-lecture-series-speaker.toml
@@ -0,0 +1,32 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Computational journeys in a sparse universe"
+date = 2023-10-25
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = "https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Sparsity is a fundamental assumption that allows us to compute efficiently on and find parsimonious solutions to science and engineering problems. Sparsity exists in all basic sciences such as physics, biology, and chemistry. I am going to give a sampling of recent work we have done on sparse computations. My talk will travel across diverse problem domains including randomized linear algebra, graph neural networks, protein family and structure discovery from metagenomic data, and tensor computations. The underlying theme will be the challenges posed by sparsity and the computational techniques we employ to overcome these challenges."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = """
+Aydın Buluç is a Senior Scientist at the Applied Math and Computational Research Division of the Lawrence Berkeley National Laboratory (LBNL) and an Adjunct Faculty at EECS department of UC Berkeley. His research interests include parallel computing, combinatorial scientific computing, high performance graph analysis and machine learning, sparse linear algebra, and computational genomics. He received his Ph.D. in Computer Science from the University of California, Santa Barbara in 2010. After that, he was a Luis W. Alvarez postdoctoral fellow at LBNL. Dr. Buluç is a recipient of the DOE Early Career Award in 2013 and the IEEE TCSC Award for Excellence for Early Career Researchers in 2015. He recently led a team that
+was chosen as a finalist for the 2022 ACM Gordon Bell Prize. He was a founding associate editor of the ACM Transactions on Parallel Computing. He is currently leading a DOE Mathematical Multifaceted Integrated Capabilities Center named Sparsitute.
+
+Zoom: https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09)
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-11-01-data-science-lecture-series-speaker.toml b/_data/talks/2023-11-01-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..1fe9c59
--- /dev/null
+++ b/_data/talks/2023-11-01-data-science-lecture-series-speaker.toml
@@ -0,0 +1,34 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Linear Probing Revisited: How to Get Rid of Clustering"
+date = 2023-11-01
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = "https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+The linear-probing hash table is one of the oldest and most widely used data structures in computer science. However, linear probing also famously comes with a major drawback: as soon as the hash table reaches a high memory utilization, elements within the hash table begin to cluster together, causing insertions to become slow. This clustering phenomenon, which was first discovered by Donald Knuth in 1962, increases the expected time per insertion to $\\Theta(x^2)$ (rather than the more desirable $\\Theta(x)$) in a hash table that is a $1 - 1/x$ fraction full.
+A natural question is whether one can somehow reduce clustering. In this talk, we establish an even stronger statement: the classical linear-probing hash table (even as it was first implemented in the 1950s) already has less clustering than the classical results would seem to suggest. As insertions and deletions are performed over time, the tombstones left behind by deletions cause the combinatorial structure of the hash table to stabilize in a way that eliminates clustering. This means that, for some versions of linear probing, the amortized expected time per operation is actually $\\tilde{O}(x)$. We also present a new version of linear probing that avoids clustering entirely, achieving $O(x)$ expected time per operation.
+"""
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = """
+William Kuszmaul's research focuses on the design and analysis of randomized algorithms and data structures. He is currently the Rabin Postdoctoral Fellow in Theoretical Computer Science at Harvard University, and after that, he will begin as an Assistant Professor in the CS Department at CMU. His research has won numerous awards at both theory and systems conferences, including Distinguished Paper at ASPLOS'23, Best Student Paper at ESA'22, Best Paper Finalist at SPAA'22, Best Paper at FUN'20, and Best Paper Finalist at APOCS'20. Prior to his postdoc, William completed a PhD at MIT, where he was funded by the John and Fannie Hertz Fellowship.
+
+Zoom: https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09)
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-11-22-data-science-lecture-series-speaker.toml b/_data/talks/2023-11-22-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..41ed219
--- /dev/null
+++ b/_data/talks/2023-11-22-data-science-lecture-series-speaker.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "“Leveraging the Structure of Data“"
+date = 2023-11-22
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Although predictions from machine learning models influence more and more of our lives, the standard way of posing a ML problem has remained relatively unchanged for decades. In the search for better models, a new and popular family of techniques (sometimes called Graph Machine Learning) has emerged. These techniques rely on expanding beyond the features of an individual entity and instead look to pull information from its relationships. The methods offer a tantalizing way of improving task performance by leveraging previously unused information. However, it is not a free lunch, as these models can be more complex, difficult to train, and may have challenges in interpretability. This talk will discuss the fundamentals of graph machine learning, a few models, and some insights from years of real-world applications."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = "such as NeurIPS, ICML, ICLR, KDD, and WWW"
+website = "https://ai.google/research/teams/algorithms-optimization/"
+photo = ""
+bio = """
+Bryan Perozzi is a Research Scientist in Google Research’s Algorithms and Optimization (https://ai.google/research/teams/algorithms-optimization/) group, where he routinely analyzes some of the world’s largest (and perhaps most interesting) graphs. Bryan’s research (https://scholar.google.com/citations?hl=en&user=rZgbMs4AAAAJ&view_op=list_works) focuses on developing techniques for learning expressive representations of relational data with neural networks. These scalable algorithms are useful for prediction tasks (classification/regression), pattern discovery, and anomaly detection in large networked data sets.
+
+Bryan is an author of 40+ peer-reviewed papers at leading conferences in machine learning and data mining (such as NeurIPS, ICML, ICLR, KDD, and WWW). His doctoral work on learning network representations was awarded the prestigious SIGKDD Dissertation Award. Bryan received his Ph.D. in Computer Science from Stony Brook University in 2016, and his M.S. from the Johns Hopkins University in 2011.
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "5319a3mlaa97jfhodc5bqitjgm@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-11-29-data-science-lecture-series-speaker.toml b/_data/talks/2023-11-29-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..4695ef7
--- /dev/null
+++ b/_data/talks/2023-11-29-data-science-lecture-series-speaker.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Framework for Parallel Hierarchical Agglomerative Clustering"
+date = 2023-11-29
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "We study the hierarchical clustering problem, where the goal is to produce a dendrogram that represents clusters at varying scales of a data set. We propose the ParChain framework for designing parallel hierarchical agglomerative clustering (HAC) algorithms, and using the framework we obtain novel parallel algorithms for the complete linkage, average linkage, and Ward's linkage criteria. Compared to most previous parallel HAC algorithms, which require quadratic memory, our new algorithms require only linear memory, and are scalable to large data sets. ParChain is based on our parallelization of the nearest-neighbor chain algorithm, and enables multiple clusters to be merged on every round. We introduce two key optimizations that are critical for efficiency: a range query optimization that reduces the number of distance computations required when finding nearest neighbors of clusters, and a caching optimization that stores a subset of previously computed distances, which are likely to be reused. Experimentally, we show that our highly-optimized implementations using 48 cores with two-way hyper-threading achieve 5.8--110.1x speedup over state-of-the-art parallel HAC algorithms and achieve 13.75--54.23x self-relative speedup. Compared to state-of-the-art algorithms, our algorithms require up to 237.3x less space. Our algorithms are able to scale to data set sizes with tens of millions of points, which previous algorithms are not able to handle."
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Shangdi is a PhD student at MIT Department of Electrical Engineering and Computer Science, advised by professor Julian Shun. Her research focuses on parallel algorithms for graph and metric data clustering. Shangdi received her BSc in Computer Science and Operations Research from Cornell University and MSc in Computer Science from MIT."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-12-06-data-science-lecture-series-speaker.toml b/_data/talks/2023-12-06-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..255bb3d
--- /dev/null
+++ b/_data/talks/2023-12-06-data-science-lecture-series-speaker.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Parallel Batch-Dynamic Graph Algorithms"
+date = 2023-12-06
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = ""
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+There has been significant interest in graph analytics due to their applications in many domains, including social network and Web analytics, machine learning, biology, and physical simulations. Real-world graphs today are massive and also dynamic. As many real-world graphs change rapidly, it is crucial to design dynamic algorithms that efficiently maintain graph statistics upon updates, since the cost of re-computation from scratch can be prohibitive. Furthermore, due to the high frequency of updates, we can improve performance by using parallelism to process batches of updates at a time. This talk presents new graph algorithms in this parallel batch-dynamic setting.
+
+Specifically, we present the first parallel batch-dynamic algorithm for approximate k-core decomposition that is efficient in both theory and practice. Our algorithm is based on our novel parallel level data structure, inspired by the sequential level data structures of Bhattacharya et al. and Henzinger et al. Given a graph with n vertices and a batch of B updates, our algorithm maintains a (2 + epsilon)-approximation of the coreness values of all vertices (for any constant epsilon > 0) in O(B log^2(n)) amortized work and O(log^2(n) loglog(n)) span (parallel time) with high probability. We implement and experimentally evaluate our algorithm, and demonstrate significant speedups over state-of-the-art serial and parallel implementations for dynamic k-core decomposition.
+
+We have also designed new parallel batch-dynamic algorithms for low out-degree orientation, maximal matching, clique counting, graph coloring, minimum spanning forest, single-linkage clustering, some of which use our parallel level data structure.
+"""
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Julian Shun is an Associate Professor of Electrical Engineering and Computer Science at MIT and a lead investigator in MIT Computer Science and Artificial Intelligence Laboratory (CSAIL). His research focuses on the theory and practice of parallel algorithms and programming, with particular emphasis on designing algorithms and frameworks for large-scale graph processing and spatial data analysis. Prior to joining MIT, he was a postdoctoral Miller Research Fellow at UC Berkeley. His honors include the NSF CAREER award, DOE Early Career Award, ACM Doctoral Dissertation Award, CMU School of Computer Science Doctoral Dissertation Award, Google Faculty Research Award, Google Research Scholar Award, SoE Ruth and Joel Spira Award for Excellence in Teaching, Allen Newell Award for Research Excellence, Facebook Graduate Fellowship, and best paper awards at PLDI, SPAA, CGO, and DCC."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2023-12-13-data-science-lecture-series-speaker.toml b/_data/talks/2023-12-13-data-science-lecture-series-speaker.toml
new file mode 100644
index 0000000..f6784af
--- /dev/null
+++ b/_data/talks/2023-12-13-data-science-lecture-series-speaker.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Felix Reidl, Birkbeck University"
+date = 2023-12-13
+start_time = "10:30"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "FASB 295"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Data Science and AI have an every increasing presence in our social, political and economic life. Given the immense influence the technologies of these field have and will have, I argue that researchers and academic institutions should reflect on the implications their work has in the world at large.
+
+In this talk I would like take stock of the larger context: a world lacking futures, fragmented academic disciplines, and the dystopian use of technology. While we cannot hope to solve any of these problems, I argue that we can and should resist the underlying trends. To that end, I propose that our institutions should be constructed first and foremost around creativity and participation and what that could mean in practice.
+"""
+
+[[speakers]]
+name = "Data Science Lecture Series. Speaker"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Felix is a senior lecturer at Birkbeck College (University of London) and the director of the Birkbeck Institute for Data Analytics."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "00m7j5ijbqpjopd1erulkvqqjj_R20230830T163000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-01-10-data-science-lecture-series.toml b/_data/talks/2024-01-10-data-science-lecture-series.toml
new file mode 100644
index 0000000..d7e6ba8
--- /dev/null
+++ b/_data/talks/2024-01-10-data-science-lecture-series.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Orientation"
+date = 2024-01-10
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = ""
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Data Science Lecture Series"
+affiliation = "Spring 24"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4635tcck8huo0hjfo9fn1jp7ol@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-01-31-pratik-soni.toml b/_data/talks/2024-01-31-pratik-soni.toml
new file mode 100644
index 0000000..c3c948f
--- /dev/null
+++ b/_data/talks/2024-01-31-pratik-soni.toml
@@ -0,0 +1,30 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Cryptography for Fairness"
+date = 2024-01-31
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science Seminar"
+location = ""
+zoom = "https://utah.zoom.us/j/96005100565?pwd=WmFGN25RazZwV2NoMGE2dVFGMngyZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Location: In person (GC 2560). Will also be streamed at:
+https://utah.zoom.us/j/96005100565?pwd=WmFGN25RazZwV2NoMGE2dVFGMngyZz09 (https://utah.zoom.us/j/96005100565?pwd=WmFGN25RazZwV2NoMGE2dVFGMngyZz09)
+"""
+
+[[speakers]]
+name = "Pratik Soni"
+affiliation = "Utah"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "2353s9ratq28anb7hj6h1v89up@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-03-13-swabha-swayamdipta.toml b/_data/talks/2024-03-13-swabha-swayamdipta.toml
new file mode 100644
index 0000000..f7f085b
--- /dev/null
+++ b/_data/talks/2024-03-13-swabha-swayamdipta.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Understanding LLMs through their Generative Behavior, Successes and Shortcomings"
+date = 2024-03-13
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science Seminar"
+location = ""
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Generative capabilities of large language models have grown beyond the wildest imagination of the broader AI research community, leading many to speculate whether these successes may be attributed to the training data or model design. I will present some work from my group which sheds light on understanding LLMs by studying their generative behavior, successes and shortcomings. First, I will show that standard inference algorithms work well because of the particular design behind LLMs. Next, I will discuss recently found successes and failures of LLMs on a combination of tasks, requiring world and domain-specific knowledge, linguistic capabilities and awareness of human and social utility. Overall, these findings paint a partial yet complex picture of our understanding of LLMs and provide a guide to the next steps forward."
+
+[[speakers]]
+name = "Swabha Swayamdipta"
+affiliation = "USC"
+website = ""
+photo = ""
+bio = "Swabha Swayamdipta is an Assistant Professor of Computer Science and a Gabilan Assistant Professor at the University of Southern California. Her research interests are in natural language processing and machine learning, with a primary interest in the estimation of dataset quality, understanding and evaluation of generative models of language, and using language technologies to understand social behavior. At USC, Swabha leads the Data, Interpretability, Language and Learning (DILL) Lab. She received her PhD from Carnegie Mellon University, followed by a postdoc at the Allen Institute for AI. Her work has received outstanding paper awards at ICML 2022, NeurIPS 2021 and an honorable mention for the best paper at ACL 2020. Her research is supported by awards from the Allen Institute for AI and Intel Labs."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "2g56aj3sae8vcjlcj8on851db8@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-04-10-ucds-lecture-series.toml b/_data/talks/2024-04-10-ucds-lecture-series.toml
new file mode 100644
index 0000000..5c715e6
--- /dev/null
+++ b/_data/talks/2024-04-10-ucds-lecture-series.toml
@@ -0,0 +1,37 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Service Operations for Justice-On-Time: A Data-Driven Queueing Approach"
+date = 2024-04-10
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = ""
+zoom = "https://utah.zoom.us/j/96005100565?pwd=WmFGN25RazZwV2NoMGE2dVFGMngyZz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Limited resources in the judicial system can lead to costly delays, stunted economic development, and even failure to deliver justice. Using the Supreme Court of India as an exemplar for such resource-constrained settings, we apply ideas from service operations to study delay. Specifically, court dynamics constitute a case-management queue, whereby each case may experience multiple service encounters spread across time, but all are necessarily with the same server. Our goal is to elucidate the drivers of congestion, focusing on metrics such as the expected case-disposition time (delay) and expected number of cases awaiting adjudication (pendency), and leverage this understanding to recommend operational interventions.
+
+We employ data-driven calibrated simulations to model the analytically intractable case-management queue. The life cycle of a case comprises two stages: pre-admission (before determining its merit for detailed hearings) and post-admission. Our methodology allows us to capture the queueing dynamics in which the judges are shared resources across the two stages. It also permits modeling of holiday capacity, which is flexibly tailored to address any surplus work that spills over from the regular year. We find that the second stage of this judicial queue is overloaded, but holiday capacity creates a perception of stability by steadying performance metrics.
+
+The sources of inefficiency that drive congestion include a misalignment between scheduling guidelines and judicial capacity, coupled with the requirement to schedule hearings in advance. Together, these factors inhibit utilization of shared capacity across the two-stage judicial queue. We demonstrate how interventions that account for these inefficiencies can successfully tackle judicial delay. In particular, scheduling to improve the allocation of time across pre- and post-admission cases can cut down the expected delay by as much as 65%.
+"""
+
+[[speakers]]
+name = "UCDS Lecture Series"
+affiliation = ""
+website = ""
+photo = ""
+bio = """
+Dr Nitin Bakshi is department chair and Professor of Operations and Information Systems at the David Eccles School of Business, University of Utah. He focuses his research on the management of disruption risk in operations and supply-chain management, with an emphasis on “low-probability high-consequence” events. He is currently investigating how to manage reporting of accident precursors to enhance safety in dangerous operations, and exploring new frontiers related to efficiency in judicial operations.
+
+He holds a B. Tech. in Electrical Engineering from IIT Bombay; an M.S. in Management Science from Stanford University; and a Ph.D. in Applied Economics from the Wharton School, University of Pennsylvania. Dr Bakshi has previously worked as a manager for Unilever and as an Algorithm Design Engineer for SmartOps Inc. Before joining the University of Utah he served on the faculty at the London Business School.
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "2k3lnrqsa9aki2hp8d1d9lqdir@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-08-27-jeff-phillips.toml b/_data/talks/2024-08-27-jeff-phillips.toml
new file mode 100644
index 0000000..fe50a1c
--- /dev/null
+++ b/_data/talks/2024-08-27-jeff-phillips.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "== Sketching and Classifying Spatial Trajectories =="
+date = 2024-08-27
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Spatial trajectories, often represented as a sequence of spatial positions, are a standard way to represent human mobility patterns. They also are used to represent motion patterns including for animals, drones, or last-mile rentals (e-scooters). However, these trajectories are notoriously difficult to work with as they overlap and can stretch long distances.
+In this talk we discuss a sketch (the minDist Sketch) that makes just about any data analysis on trajectories tasks simple and efficient. This first considers spatial trajectories as an abstract shape, and then maps them to a high-dimensional Euclidean space as a vector. We can show recovery, pseudo-metric, and metric properties of this representation. Variants can include direction information, or traits like velocity and acceleration.
+Moreover, once represented as this vector, the trajectory data is extremely easy to work with. Allowing for out-of-the-box use of software for nearest-neighbor search, clustering, and classification.
+In particular, we conduct the first formal study of classifying spatial trajectories: given trajectories from two different distributions (e.g., generated by car or bus) given a new trajectory that is unlabeled, how well can we predict which class it was from?
+Over several data sets we have assembled that demand this task, we conduct a large study, and show that the minDist sketch and its variants are consistently the easiest and most accurate method (or at the least among the best in each instance).
+"""
+
+[[speakers]]
+name = "Jeff Phillips"
+affiliation = "Utah KSoC"
+website = "https://users.cs.utah.edu/~jeffp/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-09-03-esha-datta.toml b/_data/talks/2024-09-03-esha-datta.toml
new file mode 100644
index 0000000..b5cb2ff
--- /dev/null
+++ b/_data/talks/2024-09-03-esha-datta.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Topological Signatures of Out-of-Distribution Examples"
+date = 2024-09-03
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Machine learning (ML) models employed for real-world tasks will invariably encounter inference data that is distributionally shifted from their training datasets. Such out-of-distribution (OOD) examples can have adverse effects on model performance and can pose significant problems in high-consequence application areas like healthcare or autonomous vehicles. We develop a topological characterization of OOD examples and present a computationally feasible methodology for detecting such data in a deployed pipeline. The approach leverages the known property that well-trained ML models induce a topological “simplification” on its training dataset. By computing the persistent homology of the hidden layer embeddings of training and test data, we demonstrate empirically our ability to identify the presence of OOD examples for a given model."
+
+[[speakers]]
+name = "Esha Datta"
+affiliation = "Sandia NL"
+website = "https://scholar.google.com/citations?user=FB5NpNQAAAAJ&hl=en"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-09-10-guanhong-tao.toml b/_data/talks/2024-09-10-guanhong-tao.toml
new file mode 100644
index 0000000..21b92aa
--- /dev/null
+++ b/_data/talks/2024-09-10-guanhong-tao.toml
@@ -0,0 +1,30 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Are AI-enabled Systems Safe and Secure?"
+date = 2024-09-10
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Abstract
+Artificial Intelligence (AI) has been integrated into various sectors, such as facial recognition and autonomous driving. But are the security and safety of these AI-enabled systems fully ensured? In this talk, I will present various vulnerabilities in these systems. My presentation will cover novel optimization techniques for identifying and mitigating backdoor vulnerabilities in both white-box and black-box settings, achieving substantial improvements in performance. I will share insights into the nature of backdoors and their presence in pre-trained models. Finally, I will conclude with an outlook on our recent exploration of the security of emerging AI techniques, such as generative AI.
+"""
+
+[[speakers]]
+name = "Guanhong Tao"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-09-17-aurora-clark.toml b/_data/talks/2024-09-17-aurora-clark.toml
new file mode 100644
index 0000000..da11cb8
--- /dev/null
+++ b/_data/talks/2024-09-17-aurora-clark.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "The Importance of Shape in Chemistry Data"
+date = 2024-09-17
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Data in the field of Chemistry has heavily leveraged graph theory representations within data science applications. However, there is a rich geometric and topological structure of many chemical systems (and their data) that has been less employed for feature optimization, dimensionality reduction, and predictive models. Within this discussion I will highlight some recent work and interests that seek to employ computational topology and geometry within chemistry data sets from molecular dynamics simulations – both in the context of ensemble average and temporally evolving data sets."
+
+[[speakers]]
+name = "Aurora Clark"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-09-24-rebecca-barter.toml b/_data/talks/2024-09-24-rebecca-barter.toml
new file mode 100644
index 0000000..87a0a06
--- /dev/null
+++ b/_data/talks/2024-09-24-rebecca-barter.toml
@@ -0,0 +1,30 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Veridical Data Science: the Practice of Responsible Data Analysis and Decision Making"
+date = 2024-09-24
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Data science is often presented as a straightforward, linear process involving statistical and computational techniques, without addressing the complexities inherent in real-world applications. In contrast, our new book,
+"Veridical Data Science: The Practice of Responsible Data Analysis and Decision Making", teaches data scientists to navigate the reality that most projects involve answering ambiguous domain questions with messy data, all while managing a complex web of human judgment calls. We emphasize that datasets are merely approximations of reality, and analyses are shaped by human interpretation. Using the Predictability, Computability, and Stability (PCS) framework to assess the trustworthiness and relevance of data-driven results, "Veridical Data Science" provides an actionable guide for conducting responsible and trustworthy data science.
+"""
+
+[[speakers]]
+name = "Rebecca Barter"
+affiliation = ""
+website = "http://www.rebeccabarter.com/"
+photo = ""
+bio = "Dr. Rebecca Barter is a Research Assistant Professor in the Division of Epidemiology at the University of Utah. As a statistician, data scientist, and educator, Dr Barter specializes in data science education and the analysis of complex healthcare data. Originally from Australia, Dr. Barter earned her PhD in Statistics from the University of California, Berkeley, in 2019, where she co-authored the book Veridical Data Science: The Practice of Responsible Data Analysis and Decision Making with her advisor, Professor Bin Yu. In addition to her academic work, Dr. Barter shares data science resources and insights on her blog, www.rebeccabarter.com (http://www.rebeccabarter.com/)."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-10-01-simon-brewer.toml b/_data/talks/2024-10-01-simon-brewer.toml
new file mode 100644
index 0000000..a7df4b2
--- /dev/null
+++ b/_data/talks/2024-10-01-simon-brewer.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Exploring long-term ecosystem change with self-organizing maps"
+date = 2024-10-01
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Ongoing climate change has the potential to impact a variety of physical, biological and social systems, and there is increasing concern that these changes may be sufficient to result in these systems crossing tipping points, effectively undergoing irreversible changes in state. For slow turnover systems, such as forest ecosystems, understanding the likelihood and ramifications of these state changes is challenging due to the relative short observational record. Sedimentary records of ecosystem change offer an alternative data source with a wide temporal and spatial scope, but are inherently noisy and high dimensional. Self-organizing maps provide a data-driven way to visualize nonlinear patterns in these data, and to identify past ecosystem states and state transitions. The results are used to build a simple Markov model illustrating the probability and directionality of these transitions."
+
+[[speakers]]
+name = "Simon Brewer"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-10-15-raghav-venkatraman.toml b/_data/talks/2024-10-15-raghav-venkatraman.toml
new file mode 100644
index 0000000..f7b2be6
--- /dev/null
+++ b/_data/talks/2024-10-15-raghav-venkatraman.toml
@@ -0,0 +1,35 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Minmax estimation rates for manifold learning"
+date = 2024-10-15
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+This talk is focused on obtaining minmax estimation rates for the "manifold learning" problem. Given N data points hypothesized to be i.i.d (independent and identically distributed) samples of a nice density (from within a reasonable class of densities) on a nice manifold (from within some class of nice manifolds of known intrinsic dimension d), the manifold learning problem boils down to estimating certain statistics of this "ground truth" manifold, such as the first few eigenmodes of the Laplace Beltrami operator on the manifold. The minmax estimation problem further asks: given N such data points, among all estimators of the desired statistics (say, a particular eigenvalue and associated eigenfunctions in a suitable norm), which one achieves the smallest maximum expected risk, and how does this minmax risk scale in N and the intrinsic dimension d of the manifold?
+
+An intuitive but impractical estimator consists in estimating the density from the given samples through a ``kernel density estimation'', and then solving the resulting continuum eigenproblem using a numerical method such as finite elements: this estimator turns out to be minmax optimal in scaling-- namely, the associated expected risk scales like N^{-2/d+4}, and we can show a matching lower bound for the minmax risk, demonstrating that no estimator can do better, in scaling, than this estimator.
+
+Next, we ask: do there exist *practical* estimators that are agnostic to knowledge of the manifold (so we don't have to discretize them in order to compute with finite elements!) that achieve, at least nearly, this minmax scaling of the expected risk? We affirmatively answer this question by showing that, the spectrum of a carefully constructed graph laplacian on a random geometric graph constructed from the given N samples provides an estimator for the eigenvalue and eigenvectors of the Laplace Beltrami operator on the manifold that achieves this minmax rate upto a log factor (to a small power). Both the lower and upper bound estimates in the talk bring in new PDE tools to this statistical question, and that we believe will be more broadly applicable in similar applications.
+
+This talk is based on joint work with Nicolas Garcia Trillos and his PhD student Chenghui Li (U. W. Madison), and builds on prior joint work with Scott N. Armstrong (Courant Institute).
+"""
+
+[[speakers]]
+name = "Raghav Venkatraman"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-10-29-vivek-gupta.toml b/_data/talks/2024-10-29-vivek-gupta.toml
new file mode 100644
index 0000000..e72c50f
--- /dev/null
+++ b/_data/talks/2024-10-29-vivek-gupta.toml
@@ -0,0 +1,35 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Reasoning on Tabular and Multimodal Data"
+date = 2024-10-29
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+In this talk, I’ll walk through some of the latest AI advancements that address the challenges of working with complex data, with a focus on improving reasoning for both tabular and multimodal data.
+
+I’ll begin by introducing H-STAR, a hybrid algorithm that combines symbolic and semantic reasoning to enhance question answering for tabular data. By leveraging multi-view table extraction and adaptive reasoning, H-STAR has shown great potential in improving reasoning across tabular datasets.
+
+Then, I’ll introduce MMTabQA, a dataset we developed to evaluate how AI systems manage multimodal tables that integrate structured text and images. Our research reveals where current models struggle to process these diverse data types, highlighting key areas for improvement.
+
+To wrap up, I’ll discuss open challenges and future directions, including expanding reasoning capabilities to other complex data types—such as charts, maps, and flowcharts—and enhancing AI systems' robustness in handling numerical, temporal, and visual reasoning, particularly with large (vision) language models.
+"""
+
+[[speakers]]
+name = "Vivek Gupta"
+affiliation = "ASU & UCDS alumni"
+website = "https://vgupta123.github.io"
+photo = ""
+bio = "Vivek Gupta is an Assistant Professor of Computer Science at Arizona State University (ASU), where he works on AI systems that help computers reason with complex data like tables, charts, diagrams, and maps etc. Before ASU, he was a postdoctoral researcher at the University of Pennsylvania in the Cognitive Computation Group. He earned his Ph.D. in Computer Science from the University of Utah and has received several awards, including the Bloomberg Data Science Fellowship and NLP Best Paper Awards. You can learn more about his work at vgupta123.github.io and his group CoRAL page at coral-lab-asu.github.io."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-11-12-amir-abdullah.toml b/_data/talks/2024-11-12-amir-abdullah.toml
new file mode 100644
index 0000000..c9acd73
--- /dev/null
+++ b/_data/talks/2024-11-12-amir-abdullah.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Interpreting Learned Feedback Patterns in Large Language Models"
+date = 2024-11-12
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Amir is an active researcher in mechanistic interpretability, opening the blackbox of large language models to reverse engineer the inner workings and analyze internal representations.. In this talk, he will discuss his paper in NeuIPS 2024 on interpreting reward models in language models using sparse autoencoders on internal representations. Further, Amir will introduce followup work studying whether internal representations can be transferred between large language models."
+
+[[speakers]]
+name = "Amir Abdullah"
+affiliation = ""
+website = "https://scholar.google.com/citations?user=jPEbq5wAAAAJ&hl=en"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-11-19-hoaning-xue.toml b/_data/talks/2024-11-19-hoaning-xue.toml
new file mode 100644
index 0000000..4c96ec3
--- /dev/null
+++ b/_data/talks/2024-11-19-hoaning-xue.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Computational and experimental approaches to examining short videos' persuasive effects"
+date = 2024-11-19
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "There’s a gap in understanding how people process multimodal information collectively, despite extensive research on the effects of individual multimodal features and the rapid advances in computer vision. This gap is increasingly relevant as short video platforms emerge as major information sources and influence public opinion. In this talk, I present findings from a project that combines a data-driven approach with social scientific theories to investigate how multimodal features in short videos impact audience engagement and attitude change. This research is grounded in the theoretical framework of Message Sensation Value (MSV) to theorize and quantify how multimodal features in short videos capture attention and affect information processing. This project includes two studies: (1) a computational model of MSV that predicts video engagement from 11 multimodal features across a dataset of 15,000 short videos from three popular short video platforms; second, an online experiment examining the attentional mechanism underlying the persuasive effects of MSV in short videos on message credibility and attitude change. This project provides a useful framework and computational tool for short video research."
+
+[[speakers]]
+name = "Hoaning Xue"
+affiliation = "Utah Communications"
+website = "https://faculty.utah.edu/u6059240-HAONING_XUE/hm/index.hml"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2024-11-26-zhichao-xu.toml b/_data/talks/2024-11-26-zhichao-xu.toml
new file mode 100644
index 0000000..900ae55
--- /dev/null
+++ b/_data/talks/2024-11-26-zhichao-xu.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Representation Learning for IR and Role of Retrieval in LLM Era"
+date = 2024-11-26
+start_time = "12:30"
+end_time = "13:30"
+series = "Data Science Seminar"
+location = "WEB 1230"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Retrieval is the critical way of accessing information in people’s daily lives. Retrieval applications include search engines, conversational shopping assistants, or when asked about 2+3=?, human brains do retrieval instead of reasoning. In this talk, I’ll briefly go through the history of representation learning in retrieval, from Bag-of-Words representations to the latest dense and learned sparse retrieval algorithms. Increasingly, people go to ChatGPT or other large language models for information seeking instead of search engines. With this existential crisis in mind, I will talk about the role of retrieval in LLM era, specifically, retrieval-augmented generation, strengths, weaknesses and open problems."
+
+[[speakers]]
+name = "Zhichao Xu"
+affiliation = "Utah KSoC"
+website = "https://zhichaoxu-utah.github.io/"
+photo = ""
+bio = "Zhichao Xu is a final year Ph.D. student in Kahlert School of Computing, University of Utah. He is affiliated with UtahNLP lab and TDAVIS lab, advised by Prof. Bei Wang Philips and Prof. Vivek Srikumar. His main research interests include efficient NLP methods, web search & information retrieval. He has published in major IR venues such as TheWebConf, SIGIR, WSDM, CIKM, ICTIR and NLP venues such as NAACL and EMNLP."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4u5bj5j0jouhse93h9s044844t@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-01-17-fengjiao-wang.toml b/_data/talks/2025-01-17-fengjiao-wang.toml
new file mode 100644
index 0000000..ea52b7f
--- /dev/null
+++ b/_data/talks/2025-01-17-fengjiao-wang.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Supervised Learning on Tabular Data"
+date = 2025-01-17
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Self-supervised and Semi-supervised learning (SSL) on tabular data is an understudied topic. Despite some attempts, there are two major challenges: 1. Imbalanced nature in the tabular dataset; 2. The one-hot encoding used in these methods becomes less efficient for high-cardinality categorical features. To cope with the challenges, we propose SAWTab which uses a target encoding method, Conditional Probability Representation (CPR), for efficient representation in the input space of categorical features. We improve this representation by incorporating the unlabeled samples through pseudo-labels. Furthermore, we propose a Smooth Adaptive Weighting mechanism in the target encoding to mitigate the issue of noisy and biased pseudo-labels. Experimental results on various datasets and comparisons with existing frameworks show that SAWTab yields best test accuracy on all datasets. We find that pseudo-labels can help improve the input space representation in the SSL setting, which enhances the generalization of the learning algorithm."
+
+[[speakers]]
+name = "Fengjiao Wang"
+affiliation = "Utah SoC"
+website = "https://fengjiaowang7.github.io"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "B14DC487-2CFD-4C18-B554-F5AF4C41C62F"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-01-24-data-science-ai-day.toml b/_data/talks/2025-01-24-data-science-ai-day.toml
new file mode 100644
index 0000000..41c96c7
--- /dev/null
+++ b/_data/talks/2025-01-24-data-science-ai-day.toml
@@ -0,0 +1,34 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Dieter Fox - \"Where is RobotGPT?\""
+date = 2025-01-24
+start_time = "14:00"
+end_time = "15:00"
+series = "Data Science & AI Lecture Series"
+location = "Union Ballroom"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "The last years have seen astonishing progress in the capabilities of generative AI techniques, particularly in the areas of language and visual understanding and generation. Key to the success of these models are the use of image and text data sets of unprecedented scale along with models that are able to digest such large datasets. We are now seeing the first examples of leveraging such models to equip robots with open-world visual understanding and reasoning capabilities. Unfortunately, however, we have not achieved the RobotGPT moment; these models still struggle with reasoning about geometry and physical interactions in the real world, resulting in brittle performance on seemingly simple tasks such as manipulating objects in the open world. A crucial reason for this problem is the lack of data suitable to train powerful, general models for robot decision making and control. In this talk, I will discuss approaches to generating large datasets for training robot manipulation capabilities, with a focus on the role simulation can play in this context. I will show some of our prior work, where we demonstrated robust sim-to-real transfer of manipulation skills trained in simulation, and then discuss a promising direction toward training a model architecture that combines high-level, semantic, open-world reasoning, with low-level 3D robot policies."
+
+[[speakers]]
+name = "Data Science"
+affiliation = ""
+website = "https://datascience.utah.edu/events/2025/data-science-day/"
+photo = ""
+bio = "Dieter Fox is Senior Director of Robotics Research at NVIDIA and Professor in the Allen School of Computer Science & Engineering at the University of Washington, where he heads the UW Robotics and State Estimation Lab. Dieter’s research is in robotics and artificial intelligence, with a focus on learning and perception applied to problems such as robot manipulation, mapping, and object detection and tracking. He has published more than 200 technical papers and is the co-author of the textbook “Probabilistic Robotics”. He is a Fellow of the IEEE, AAAI, and ACM, and recipient of the 2020 IEEE Pioneer in Robotics and Automation Award and the 2023 IJCAI John McCarthy Award. He was an editor of the IEEE Transactions on Robotics, program co-chair of the 2008 AAAI Conference on Artificial Intelligence, and program chair of the 2013 Robotics: Science and Systems conference."
+
+[[speakers]]
+name = "AI Day"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Dieter Fox is Senior Director of Robotics Research at NVIDIA and Professor in the Allen School of Computer Science & Engineering at the University of Washington, where he heads the UW Robotics and State Estimation Lab. Dieter’s research is in robotics and artificial intelligence, with a focus on learning and perception applied to problems such as robot manipulation, mapping, and object detection and tracking. He has published more than 200 technical papers and is the co-author of the textbook “Probabilistic Robotics”. He is a Fellow of the IEEE, AAAI, and ACM, and recipient of the 2020 IEEE Pioneer in Robotics and Automation Award and the 2023 IJCAI John McCarthy Award. He was an editor of the IEEE Transactions on Robotics, program co-chair of the 2008 AAAI Conference on Artificial Intelligence, and program chair of the 2013 Robotics: Science and Systems conference."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-02-07-omkar-bhalerao.toml b/_data/talks/2025-02-07-omkar-bhalerao.toml
new file mode 100644
index 0000000..f414520
--- /dev/null
+++ b/_data/talks/2025-02-07-omkar-bhalerao.toml
@@ -0,0 +1,32 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Triadic First-Order Logic Queries in Temporal Networks"
+date = 2025-02-07
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Motif counting is a fundamental problem in network analysis, and there is a rich literature of theoretical and applied algorithms for this problem. Given a large input network G, a motif H is a small “pattern" graph indicative of special local structure. Motif/pattern mining involves finding all matches of this pattern in the input G. The simplest, yet challenging, case of motif counting is when H has three vertices, often called a triadic query. Recent work has focused on temporal graph mining, where the network G has edges with timestamps (and directions) and H has time constraints. Such networks are common representations for communication networks, citation networks, financial transactions, etc.
+Inspired by concepts in logic and database theory, we introduce the study of Thresholded First Order Logic (FOL) Motif Analysis for massive temporal networks. A typical triadic motif query asks for the existence of three vertices that form a desired temporal pattern. An FOL motif query is obtained by having both existence and universal quantifiers with thresholds. This allows for query semantics that can mine richer information from networks. A typical triadic query would be "find all triples of vertices u,v,w such that they form a triangle within one hour". A thresholded FOL query can express "find all pairs u,v such that for half of w where (u,w) formed an edge, (v,w) also formed an edge within an hour".
+
+We design the first algorithm, FOLTY, for mining thresholded triadic FOL queries, whose theoretical running time matches the best known running time for sparse graphs. Specifically, our algorithms run in time 𝑂 (m $\\alpha \\log \\sigma_{\\max}$). Here, $m$ is the number of temporal edges in the input graph, $\\alpha$ is its degeneracy (maximum core number), and the $\\sigma_{\\max}$ is the maximum edge multiplicity. Our procedures can be efficiently implemented, and FOLTY has good empirical behavior. For example, we can answer triadic FOL queries on graphs with nearly 70M edges in less than an hour on commodity hardware. We believe that our work could start a new research direction in the classic well-studied problem of motif analysis.
+"""
+
+[[speakers]]
+name = "Omkar Bhalerao"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-02-14-bao-wang.toml b/_data/talks/2025-02-14-bao-wang.toml
new file mode 100644
index 0000000..4bfb155
--- /dev/null
+++ b/_data/talks/2025-02-14-bao-wang.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Conditional Flow Divergence Matching"
+date = 2025-02-14
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Conditional flow matching (CFM) stands out as an efficient simulation-free approach for training flow-based generative models, achieving remarkable performance for data generation. However, CFM is insufficient to ensure accuracy in learning probability paths, and the learned vector field significantly violates the continuity equation governing probability flows. In response, we establish a new total-variation bound between the learned and ground-truth probability paths, showing that the gap between probability paths is bounded above by a combination of CFM loss and an associated divergence loss. This theoretical bound informs us to design a new objective to match both flow and divergence accompanied by an efficient implementation. Our new training approach improves the performance of the flow-based generative model by a noticeable margin without significantly raising the computational cost.
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
+"""
+
+[[speakers]]
+name = "Bao Wang"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-02-21-chenglu-li.toml b/_data/talks/2025-02-21-chenglu-li.toml
new file mode 100644
index 0000000..b119f3e
--- /dev/null
+++ b/_data/talks/2025-02-21-chenglu-li.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Teaching and Learning at the Human-Technology Frontier: The Power of Artificial Intelligence and Big Data"
+date = 2025-02-21
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "In this talk, Chenglu will examine the rapidly evolving field of artificial intelligence in education (AIED) by focusing on three key research gaps: agentic AI needs, FAccT (fairness, accountability, and transparency) challenges, and issues related to computing supremacy. He will illustrate these challenges through his current project, ALTER-Math (AI-augmented Learning by Teaching to Enhance and Renovate Math Learning), a $10M initiative designed to accelerate middle school math learning using generative AI-powered solutions. Additionally, Chenglu will discuss his contributions to advancing learning and teaching through the development, evaluation, and dissemination of FAccT AI cyberinfrastructure. Finally, he will outline promising future directions for research and collaboration in this dynamic field."
+
+[[speakers]]
+name = "Chenglu Li"
+affiliation = "Utah Edu Psych"
+website = "https://www.chengluli.com"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-02-28-bernardo-modenesi.toml b/_data/talks/2025-02-28-bernardo-modenesi.toml
new file mode 100644
index 0000000..fbc1d5d
--- /dev/null
+++ b/_data/talks/2025-02-28-bernardo-modenesi.toml
@@ -0,0 +1,32 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Unveiling Hidden Patterns in Agent Behavior with Discrete-Choice and Network Theory"
+date = 2025-02-28
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Many datasets in data science stem from agents repeatedly making choices over time, with each choice leading to an observable outcome. In this talk, I introduce a novel approach to uncover latent agent heterogeneity, enhancing both our understanding of agent behavior and causal inference estimation. By combining discrete choice models with network theory, we develop a method to measure agent similarity based on their choice patterns. This results in a network-based unsupervised clustering technique that groups agents with similar behaviors—offering an interpretable alternative to black-box clustering models while maintaining explicit estimation assumptions. I will illustrate our approach using labor market data, where workers (agents) and jobs (choices) form a bipartite network, with worker-job matches represented as edges. By clustering workers based on their job choices, we can infer unobserved worker skills—a crucial factor in economic analysis. Through Bayesian estimation, we reveal latent worker groups, improving predictions of labor market outcomes and measuring labor market discrimination more effectively than models relying only on observable characteristics. This seminar will detail our methodological framework, estimation strategy, and practical applications for understanding and predicting agent-choice dynamics.
+
+Bonus project:
+In the final portion of the talk, I will pivot to a more informal discussion of a preliminary 'Model Ensemble Approach to Assessing Discrimination in Machine Learning Models' in the space of algorithmic fairness.
+"""
+
+[[speakers]]
+name = "Bernardo Modenesi"
+affiliation = "UU BioStats"
+website = "https://sites.google.com/view/bmodenesi"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-03-21-jeff-phillips.toml b/_data/talks/2025-03-21-jeff-phillips.toml
new file mode 100644
index 0000000..38acb04
--- /dev/null
+++ b/_data/talks/2025-03-21-jeff-phillips.toml
@@ -0,0 +1,37 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Data Science & AI Lecture Series"
+date = 2025-03-21
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Robust statistics aims to compute quantities to represent data where a fraction of it may be arbitrarily corrupted. The most essential statistic is the mean, and in recent years,
+there has been a flurry of theoretical advancement for efficiently estimating the mean in high dimensions on corrupted data. While several algorithms have been proposed that
+achieve near-optimal error, they all rely on large data size requirements as a function of dimension.
+
+In this talk, we perform an extensive experimentation over various mean estimation techniques where data size might not meet this requirement due to the high-dimensional setting.
+For data with inliers generated from a Gaussian with known covariance, we find experimentally that several robust mean estimation techniques can practically improve upon the sample mean, with the quantum entropy scaling approach from Dong et.al. (NeurIPS 2019) performing consistently the best. However, this consistent improvement is conditioned on a couple of simple modifications to how the steps to prune outliers work in the high-dimension
+low-data setting, and when the inliers deviate significantly from Gaussianity. In fact, with these modifications, they are typically able to achieve roughly the same error as taking the sample mean of the uncorrupted inlier data, even with very low data size. In addition to
+controlled experiments on synthetic data, we also explore these methods on large language models, deep pretrained image models, and non-contextual word embedding models that do not necessarily have an inherent Gaussian distribution. Yet, in these settings, a mean point of a set of embedded objects is a desirable quantity to learn, and the data exhibits the high-dimension low-data setting studied in this paper. We show both the challenges of achieving this goal, and that our updated robust mean estimation methods can provide
+significant improvement over using just the sample mean.
+"""
+
+[[speakers]]
+name = "Jeff Phillips"
+affiliation = "UU KSoC"
+website = "https://users.cs.utah.edu/~jeffp/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-03-28-seth-pettie.toml b/_data/talks/2025-03-28-seth-pettie.toml
new file mode 100644
index 0000000..13a8c93
--- /dev/null
+++ b/_data/talks/2025-03-28-seth-pettie.toml
@@ -0,0 +1,51 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Everything you always wanted to know about Cardinality"
+date = 2025-03-28
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+The Cardinality Estimation/Distinct Elements problem is to approximate
+the number of distinct elements in a data stream using a small
+probabilistic data structure called a "sketch". This problem has been
+studied for 40 years, has many industrial applications, and is
+featured prominently in most courses on Big Data algorithmics. It is
+therefore a real puzzle to explain why research on this popular and
+fundamental problem has been unusually slow.
+
+This talk presents a complete history of the Cardinality Estimation
+problem from Flajolet and Martin's seminal 1983 paper to the present,
+and includes an account of how the research community became
+fractured, delaying many natural developments by decades. I will
+present our recent efforts to achieve information-theoretically
+optimal cardinality sketches, which draws on two notions of
+"information" developed in the 20th century: Fisher information
+(governing optimal point estimation) and Shannon entropy (governing
+optimal space/communication).
+
+Joint work with Dingyu Wang.
+
+========
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
+"""
+
+[[speakers]]
+name = "Seth Pettie"
+affiliation = "U Michigan CS"
+website = "https://web.eecs.umich.edu/~pettie/"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-04-04-sabyasachi-basu.toml b/_data/talks/2025-04-04-sabyasachi-basu.toml
new file mode 100644
index 0000000..1f3fd9e
--- /dev/null
+++ b/_data/talks/2025-04-04-sabyasachi-basu.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "\"Triangles, Communities, and Dense Subgraphs\""
+date = 2025-04-04
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+In this talk, we will go over a few recent results on dense subgraph discovery. We aim to discover 'many' dense subgraphs of 'reasonable size' in real-world networks. We show that by leveraging triadic structure in graphs, one can do this efficiently without complicated distributional assumptions on the input. Our techniques bridge an important gap: most existing theory considers the setting where the number of pieces is a constant, whereas techniques that produce `satisfactory' decompositions rarely have density guarantees (and indeed, often give sparse, poorly connected subgraphs). We offer a community detection flavor to our results: we provide a new metric for the 'goodness' of communities in terms of density and show that the spectrum of graph matrices implies the existence of communities.
+
+A key goal of this talk is to unpack the several phrases in quotes in the preceding paragraph and offer some perspectives on why these are important (and sometimes difficult!). We also provide an algorithm that (provably) decomposes large social networks into dense subgraphs in minutes on regular laptops, and, time permitting, discuss the setting of overlapping subgraph detection using similar techniques.
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
+"""
+
+[[speakers]]
+name = "Sabyasachi Basu"
+affiliation = "UCSC"
+website = "https://sites.google.com/view/sabyaucsc/home"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-04-11-peter-jacobs.toml b/_data/talks/2025-04-11-peter-jacobs.toml
new file mode 100644
index 0000000..474d0b0
--- /dev/null
+++ b/_data/talks/2025-04-11-peter-jacobs.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Data Science & AI Lecture Series"
+date = 2025-04-11
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+We study estimation of large discrete distributions under the structural assumption that they follow a Zipfian distribution, in which the ranking of alphabet items and/or level of decay need to be estimated from data. Empirical evidence for near Zipfian distributions has been found in diverse applications such as word and n-gram probability distributions in natural language text and chord probabilities in musical pieces. We introduce the Sort and Snap estimator for when the level of decay is known but the ranking function needs to be estimated, and show it is minimax in several high dimensional regimes. When both the ranking and decay level are unknown, we introduce an adaptive variant of Sort and Snap and show via Monte Carlo simulation that it outperforms state of the art discrete distribution estimators in these same high dimensional regimes. Our results motivate assessment of whether linguistically motivated marginal distributions for generating natural language that are claimed to be Zipfian in quantitative linguistics communities are truly Zipfian. Through Monte Carlo experiments on one such well-regarded distribution, Sort and Snap procedures lag behind even the simplest non-parametric estimator (empirical proportions), which brings into focus that this distribution thought to be Zipfian actually departs meaningfully from the Zipfian pattern.
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
+"""
+
+[[speakers]]
+name = "Peter Jacobs"
+affiliation = "UU KSoC, Sandia"
+website = "https://jacobs269.github.io"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-04-18-tucker-hermans.toml b/_data/talks/2025-04-18-tucker-hermans.toml
new file mode 100644
index 0000000..c2a66c8
--- /dev/null
+++ b/_data/talks/2025-04-18-tucker-hermans.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Stein Variational Inference for Robotic Learning and Control"
+date = 2025-04-18
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Probabilistic inference, the problem of estimating a distribution given data, has been a central pillar of robotic algorithms for more than two decades. Inference techniques have defined the de facto standard for robotic localization, mapping, system calibration, and online error estimation for mobile robots. Academic work has shown how these same inference problem formulations and algorithms can be used to solve problems of planning and control.
+
+In this talk I will discuss how probabilistic inference techniques can be extended for use in robotic manipulation where models may come either from engineering first-principles or in the form of large neural networks. I will then give a brief dedication of Stein variational inference, a recent nonparametric technique for probabilistic inference that is easily parallelized on modern GPUs providing much faster inference times compared to more traditional Markov chain Monte Carlo methods. I will then show a few different applications of using Stein variational inference from my lab, including planning to goal distributions, adaptive control of magnetic manipulation, and generating diverse data for training from real-world robot failures.
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
+"""
+
+[[speakers]]
+name = "Tucker Hermans"
+affiliation = "UU KSoC, NVIDIA"
+website = "https://robot-learning.cs.utah.edu/thermans"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4E8CAE38-B3D7-4A6F-988F-B75BA6C93098"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-08-20-varun-shankar.toml b/_data/talks/2025-08-20-varun-shankar.toml
new file mode 100644
index 0000000..565a10e
--- /dev/null
+++ b/_data/talks/2025-08-20-varun-shankar.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Kernel Methods for Operator Learning"
+date = 2025-08-20
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "http://utah.zoom.us/my/vsutah"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Varun Shankar"
+affiliation = "Title: Kernel Methods for Operator Learning"
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "7B3802E7-14CE-46F9-9F98-759FAD9F938A"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-08-27-konstantin-genin.toml b/_data/talks/2025-08-27-konstantin-genin.toml
new file mode 100644
index 0000000..85b9cee
--- /dev/null
+++ b/_data/talks/2025-08-27-konstantin-genin.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Predictions as Public Reasons"
+date = 2025-08-27
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/7824755969?pwd=SUplL1EwUmU0TE5JRkhyQ2dtNmsvdz09"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Konstantin Genin"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250827T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-09-03-daniel-brown.toml b/_data/talks/2025-09-03-daniel-brown.toml
new file mode 100644
index 0000000..8f13353
--- /dev/null
+++ b/_data/talks/2025-09-03-daniel-brown.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Swarms, Emergent Behaviors, and Multi-Agent Systems"
+date = 2025-09-03
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Daniel Brown"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-09-10-bei-wang-phillips.toml b/_data/talks/2025-09-10-bei-wang-phillips.toml
new file mode 100644
index 0000000..57ec542
--- /dev/null
+++ b/_data/talks/2025-09-10-bei-wang-phillips.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Talks will be held in LNCO 1100, and also streamed via the following Zoom link"
+date = 2025-09-10
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Bei Wang Phillips"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-09-17-aditya-bhaskara.toml b/_data/talks/2025-09-17-aditya-bhaskara.toml
new file mode 100644
index 0000000..6150be0
--- /dev/null
+++ b/_data/talks/2025-09-17-aditya-bhaskara.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Descent with Misaligned Gradients and Applications to Hidden Convexity"
+date = 2025-09-17
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Aditya Bhaskara"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-09-24-kenneth-blake-vernon.toml b/_data/talks/2025-09-24-kenneth-blake-vernon.toml
new file mode 100644
index 0000000..6d0b1bf
--- /dev/null
+++ b/_data/talks/2025-09-24-kenneth-blake-vernon.toml
@@ -0,0 +1,35 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Indirect dating with mixture density networks"
+date = 2025-09-24
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+It is an astonishing fact about the world today that no one can say precisely how many people actually live on our planet, even though it would presumably be useful to have such information to plan for climate change and disaster risk management, among other things. Luckily for us, demographers and spatial data scientists are keenly aware of this problem and have devised sophisticated methods for interpolating population in these regions based on their built area, typically measured using remote sensing technology.
+
+As it happens, this is exactly the reasoning applied by archaeologists seeking to reconstruct population sizes in the past. Unfortunately, archaeology faces an additional challenge here, since the archaeological record is a palimpsest of built area, representing continuous human settlement over decades, centuries, and sometimes even millennia. So, reconstructing population sizes across a region of interest requires that archaeologists also develop a chronology for that region at the same time.
+
+A region’s chronology can be represented by a probability density function, p(t), with well-dated archaeological materials – like tree-rings and radiocarbon samples - assumed to be random draws from that distribution. Here, we propose to estimate p using a deep-learning extension to the mixture model known as a Mixture Density Network. With this model, we condition the chronology on diagnostic data X, treating mixture parameters as unknown functions of X that can be estimated using a simple multilayer perceptron. Specifically, we use the density of X in the area around each sampled date to estimate the mixture parameters. This allows us to interpolate dates at under-sampled sites and to build a better representation of the regional chronology, one that is based, in theory, on the totality of the archaeological record.
+
+As an example, we fit an MDN to the distribution of tree-rings in the Mesa Verde region of southwestern Colorado using the spatial distribution of ceramics to estimate mixture parameters. An important ancillary goal of this research is to develop software tools scientists can use to train MDNs on their own data without also having to learn the intricacies of AI development and testing.
+"""
+
+[[speakers]]
+name = "Kenneth Blake Vernon"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-10-01-luis-garcia.toml b/_data/talks/2025-10-01-luis-garcia.toml
new file mode 100644
index 0000000..9fc300a
--- /dev/null
+++ b/_data/talks/2025-10-01-luis-garcia.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "A Trip to the Neural Frontier: Neurosymbolic Sensor Fusion for Trustworthy AI-Enabled Neural Interventions"
+date = 2025-10-01
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Advances in computing and neuroscience are converging to enable new forms of recording and stimulation in naturalistic environments, or “neuroscience in the wild.” At the core of this effort is the ability to capture the human sensory experience, synchronized with intracranial recordings, to understand how brain activity relates to behavior in real-world contexts. I will share recent progress in building end-to-end pipelines for trustworthy brain–behavior research. I will begin with multimodal datasets we have collected during spatial navigation tasks, showing how hippocampal activity reflects context shifts such as doorways and landmarks. I will then discuss our work on event-driven synchronization, where large language models and human-in-the-loop oversight transform raw multimodal recordings into aligned, analyzable events. Building on this foundation, we are developing new platforms that combine neural signals, mobile sensing, and immersive environments to capture context in real time. I will also discuss our initial explorations on how extended reality can help manage the inherent noisiness of natural settings, and how privacy risks emerge when enabling sensors in sensitive environments. Together, these efforts outline a pathway toward reproducible, explainable, and privacy-preserving neural interventions in everyday life. Talks will be held in LNCO 1100, and also streamed via the following Zoom link: https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+
+[[speakers]]
+name = "Luis Garcia"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-10-15-vineet-pandey.toml b/_data/talks/2025-10-15-vineet-pandey.toml
new file mode 100644
index 0000000..b6f8b10
--- /dev/null
+++ b/_data/talks/2025-10-15-vineet-pandey.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Designing human-centered systems that yield data that is minimal, relevant, and actionable"
+date = 2025-10-15
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Vineet Pandey"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-10-22-kenneth-marino.toml b/_data/talks/2025-10-22-kenneth-marino.toml
new file mode 100644
index 0000000..50300d8
--- /dev/null
+++ b/_data/talks/2025-10-22-kenneth-marino.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "VLM Agents"
+date = 2025-10-22
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = true
+abstract = ""
+
+[[speakers]]
+name = "Kenneth Marino"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-10-29-anna-fariha.toml b/_data/talks/2025-10-29-anna-fariha.toml
new file mode 100644
index 0000000..d87f48b
--- /dev/null
+++ b/_data/talks/2025-10-29-anna-fariha.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Understanding Data through Change Summarization and Causal Disparity Explanations"
+date = 2025-10-29
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Anna Fariha"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-11-05-jenny-lin.toml b/_data/talks/2025-11-05-jenny-lin.toml
new file mode 100644
index 0000000..a8e8733
--- /dev/null
+++ b/_data/talks/2025-11-05-jenny-lin.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Talks will be held in LNCO 1100, and also streamed via the following Zoom link"
+date = 2025-11-05
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Jenny Lin"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-11-12-kyle-dawson-tyler-hagen.toml b/_data/talks/2025-11-12-kyle-dawson-tyler-hagen.toml
new file mode 100644
index 0000000..7eab99b
--- /dev/null
+++ b/_data/talks/2025-11-12-kyle-dawson-tyler-hagen.toml
@@ -0,0 +1,39 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "DESI: Disentangling Cosmology from Observational Artifacts"
+date = 2025-11-12
+start_time = "10:00"
+end_time = "11:00"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = "https://utexas.zoom.us/j/87159746528?pwd=ouQu8lN9ARbb6aRFpvdf6Ddb1Oqa8B.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+The Dark Energy Spectroscopic Instrument (DESI) has concluded three years of observation, leading to the largest spectroscopic galaxy sample ever produced. In combination with other cosmological probes, these measurements reveal hints of new physics beyond the standard cosmological model. In this talk, we will first present the observations and key measurements that led to these new constraints. We will then describe the role that neural networks and random forests play in this analysis and our tests of these machine learning algorithms against more physically-motivated, linear models.
+
+Where:
+The location and time is moved for this talk so it can coincide with the CosmicAI seminar. It will take place in WEB 3780 (the Evans Conference room in SCI) and at 10am. It will also be on Zoom at this *new* link:
+"""
+
+[[speakers]]
+name = "Kyle Dawson"
+affiliation = ""
+website = "https://profiles.faculty.utah.edu/u0634757"
+photo = ""
+bio = ""
+
+[[speakers]]
+name = "Tyler Hagen"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-11-19-vivek-srikumar.toml b/_data/talks/2025-11-19-vivek-srikumar.toml
new file mode 100644
index 0000000..ef3dcd9
--- /dev/null
+++ b/_data/talks/2025-11-19-vivek-srikumar.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Talks will be held in LNCO 1100, and also streamed via the following Zoom link"
+date = 2025-11-19
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Vivek Srikumar"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2025-12-03-data-visualization-101.toml b/_data/talks/2025-12-03-data-visualization-101.toml
new file mode 100644
index 0000000..5eba488
--- /dev/null
+++ b/_data/talks/2025-12-03-data-visualization-101.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Madison Golden and Kaylee Alexander"
+date = 2025-12-03
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "LNCO 1100"
+zoom = "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Data Visualization 101"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "_6t136e1g692jeb9h6h1kab9k6p33ib9p8osjgb9n6kskcga47533icpo84_R20250903T170000@google.com"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-01-09-marina-kogan.toml b/_data/talks/2026-01-09-marina-kogan.toml
new file mode 100644
index 0000000..12d7775
--- /dev/null
+++ b/_data/talks/2026-01-09-marina-kogan.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Human-Centered Data Science for Crisis Informatics"
+date = 2026-01-09
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Social media platforms have been increasingly used by the public in crisis situations, partly because they upend the traditional top-down broadcasting model of risk communication. Instead, social media platforms facilitate a two-way information exchange between the official response channels and the general public, enabling more participatory crisis communication, as well as coordination and self-organization among the public. In this more complex information ecosystem, understanding the flow of information is crucial to supporting those affected and preventing malicious actors from capitalizing on the uncertainty. However, the study of such information flows is challenging, as the high-tempo, high-volume convergent nature of crisis events produces vast amounts of social media data, necessitating the use of the data science methods. On the other hand, to glean meaningful insight from the crisis-related social media activity, it is necessary to use methods that account for the complex social context of the user activity. In this talk, Kogan will show how the Human-Centered Data Science (HCDS) provides methodological approaches that both harness the power of computational methods and account for the highly situated nature of social media activity in disruption. She will focus on sequence-based approaches as examples of HCDS methods in two empirical studies: analysis of attention-garnering information during a natural disaster and investigation of behavioral signatures in coordinated information operations."
+
+[[speakers]]
+name = "Marina Kogan"
+affiliation = "UU KSoC, RAI Faculty Fellow"
+website = ""
+photo = ""
+bio = "Marina Kogan is an Assistant Professor at Kahlert School of Computing at University of Utah. She works in the areas of social computing, crisis informatics, and human-centered data science. Her background in Sociology and Computer Science informs her focus on studying online coordination, collective problem-solving, and information flows during disruption events such as disasters arising from natural hazards and political crises. Her most recent work engages with issues of mis/disinformation, conspiracy theories, and state-sponsored information operations."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "CFD8C4B5-05ED-4DE5-8DD1-899F8B88D3E7"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-01-23-andrew-mcnutt.toml b/_data/talks/2026-01-23-andrew-mcnutt.toml
new file mode 100644
index 0000000..e992222
--- /dev/null
+++ b/_data/talks/2026-01-23-andrew-mcnutt.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Linters as Socio Technical Systems"
+date = 2026-01-23
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Interfaces—whether they are for data analysis, programming, or any other activities—exist within specific communities of practice. The norms and standards of those groups inscribe themselves in the form of those tools; implicitly driving what is and is not possible. In this talk I will explore how linters (a spell checker-like tool used in programming) can be used to interrogate this arrangement. In doing so I will describe recent and on-going efforts relating to application of linters to a variety of domains, including visualization, color palettes, and social media posts. Through this discussion, I will argue that using linters as a critical lens allows us to examine the technological world around us in a new light (offering new opportunities for research and design), and therein explore the values that we manifest in our tool design and selection."
+
+[[speakers]]
+name = "Andrew McNutt"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Andrew McNutt is an assistant professor at University of Utah’s Kahlert School of Computing and Scientific Computing and Imaging Institute. His research lives in the union of human computer interaction, visualization, and programming interfaces. It considers topics like creative coding, domain-specific languages, theories of visualization, and critical theory. He completed his PhD at University of Chicago, and a Post Doc at University of Washington. His work is funded by the NSF and is a Seibel Foundation Scholar. His work has won awards at top visualization and HCI conferences."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "01ED0903-3B52-4C91-B0C0-C93EEBDB55D1"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-02-06-makoto-kelp.toml b/_data/talks/2026-02-06-makoto-kelp.toml
new file mode 100644
index 0000000..d1ac57a
--- /dev/null
+++ b/_data/talks/2026-02-06-makoto-kelp.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Navigating Advances and Inflections in Machine Learning for Atmospheric Chemistry Modeling"
+date = 2026-02-06
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Global climate and Earth system models rarely include comprehensive atmospheric chemistry because of its high computational cost. A bottleneck is the chemical solver that integrates the large-dimensional coupled systems of kinetic equations describing the chemical mechanism. In recent years, machine learning (ML) methods have been proposed as a potentially transformative approach to reducing this cost by replacing traditional solvers with fast emulators. However, early efforts showed that ML-based chemical solvers often suffer from rapid error growth and instability. In this talk, I will review the evolving landscape of ML for atmospheric chemistry modeling over the past decade and how its trajectory mirrors broader developments in climate and weather AI. I begin with detailing how to achieve stable emulation in 0-D box models and then show how these principles translate to complex global atmospheric models. I conclude by discussing the current state of ML for modeling atmospheric chemistry and outlining how mechanistic interpretability in geospatial foundation models may offer a promising future line of inquiry."
+
+[[speakers]]
+name = "Makoto Kelp"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Makoto Kelp is an Assistant Professor in the Department of Atmospheric Sciences and a Fellow in the Wilkes Center for Climate Science & Policy at the University of Utah. His research focuses on the intersections of atmospheric chemistry, fires, and human-environmental systems, with an emphasis on using data-driven methods and machine learning techniques. He earned his PhD from Harvard University and conducted his Post Doc at Stanford University as a NOAA Climate & Global Change Fellow."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "B1C723AE-9C16-4580-A062-83F2FBCB6C78"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-02-17-juliana-freire.toml b/_data/talks/2026-02-17-juliana-freire.toml
new file mode 100644
index 0000000..506916e
--- /dev/null
+++ b/_data/talks/2026-02-17-juliana-freire.toml
@@ -0,0 +1,39 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Dataset Discovery and Integration in the Era of Large Language Models"
+date = 2026-02-17
+start_time = "11:00"
+end_time = "12:00"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780 (Evans)"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+**Abstract:**
+The proliferation of structured data across open-data portals, the web, and enterprise data lakes presents unprecedented opportunities for scientific discovery and data-driven decision-making. However, realizing this potential requires solving fundamental challenges in data discovery and integration: How do we find relevant datasets among millions of candidates? How do we understand their semantics to integrate them? And how do we build systems that are scalable, accurate, and cost-effective?
+
+In this talk, I will present our recent work addressing these challenges by combining techniques from data management, visualization, HCI, and modern language models. Our work is motivated by real-world problems across domains: supporting biomedical researchers in integrating heterogeneous datasets, enabling dataset search over urban data for policy analysis and planning, and augmenting training data to improve machine learning model performance.
+
+First, I will describe how we are reimagining dataset discovery by designing specialized search engines, developing methods to automatically derive metadata, and introducing novel data-driven queries that go beyond keywords to support complex information needs. Then, I will turn to data integration, showing how we can leverage the semantic power of Large Language Models (LLMs) to match schemas with state-of-the-art accuracy. I will also argue for the critical importance of human-AI collaboration, demonstrating how visual analytics can effectively place the user in the loop to guide and verify these complex processes.
+
+Finally, I will reflect on how LLMs are fundamentally changing computer science research. They make previously intractable problems like semantic data discovery and integration tractable, yet they behave more like natural phenomena, exhibiting variability and non-determinism. This shift forces us to move beyond deterministic algorithmic thinking and to practice computer science as a science, adopting empirical research methods to understand and leverage these powerful but unpredictable tools.
+
+**Bio:**
+Juliana Freire is an Institute Professor at the Tandon School of Engineering and Professor of Computer Science and Data Science at New York University, where she co-directs the Visualization Imaging and Data Analysis (VIDA) Center. Her research develops methods and systems that enable a wide range of users to obtain trustworthy insights from data. It spans topics in large-scale data analysis and integration, visualization, machine learning, provenance management, and web information discovery, addressing application areas including urban analytics, predictive modeling, computational reproducibility, and biomedical data harmonization. She has co-authored over 250 papers, including 12 award winners and a test-of-time award. She served as elected chair of ACM SIGMOD and as a council member of the Computing Community Consortium (CCC), and was the NYU lead investigator for the Moore-Sloan Data Science Environment. She is a Fellow of the ACM and AAAS, and a winner of the ACM SIGMOD Contributions Award. Her work has been supported by funding agencies and industry partners, including the National Science Foundation, DARPA, ARPA-H, the Department of Energy, the National Institutes of Health, and technology companies such as Google, Amazon, Microsoft Research, and IBM. Freire received her Ph.D. and M.Sc. degrees in computer science from the State University of New York at Stony Brook and her B.S. degree in computer science from the Federal University of Ceará in Brazil.
+"""
+
+[[speakers]]
+name = "Juliana Freire"
+affiliation = ""
+website = "https://engineering.nyu.edu/faculty/juliana-freire"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "DEAB3E9E-4337-460A-91AE-A8E46EE225EB"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-02-19-dinesh-manocha.toml b/_data/talks/2026-02-19-dinesh-manocha.toml
new file mode 100644
index 0000000..a84794a
--- /dev/null
+++ b/_data/talks/2026-02-19-dinesh-manocha.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Robot Navigation in the Wild"
+date = 2026-02-19
+start_time = "15:30"
+end_time = "16:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "In the last few decades, most robotics success stories have been limited to structured or controlled environments. A major challenge is to develop robot systems that can operate in complex or unstructured environments corresponding to homes, dense traffic, outdoor terrains, public places, etc. In this talk, we give an overview of our ongoing work on developing robust planning and navigation technologies that use recent advances in computer vision, sensor technologies, machine learning, and motion planning algorithms. We present new methods that utilize multi-modal observations from an RGB camera, 3D LiDAR, and robot odometry for scene perception, along with deep reinforcement learning for reliable planning. The latter is also used to compute dynamically feasible and spatial aware velocities for a robot navigating among mobile obstacles and uneven terrains. We have integrated these methods with wheeled robots, home robots, and legged platforms and highlight their performance in crowded indoor scenes, home environments, and dense outdoor terrains."
+
+[[speakers]]
+name = "Dinesh Manocha"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Prof. Dinesh Manocha is Paul Chrisman-Iribe Chair in Computer Science & ECE and Distinguished University Professor at University of Maryland College Park. His research interests include virtual environments, physics-based modeling, and robotics. His group has developed several software packages that are standard and licensed to 60+ commercial vendors. He has published more than 850 papers & supervised 63 PhD dissertations. He is a Fellow of AAAI, AAAS, ACM, IEEE, and NAI and member of ACM SIGGRAPH and IEEE VR Academies, and Bézier Award from Solid Modeling Association. He received the Distinguished Alumni Award from IIT Delhi the Distinguished Career in Computer Science Award from Washington Academy of Sciences. He was a co-founder of Impulsonic, a developer of physics-based audio simulation technologies, which was acquired by Valve Inc in November of 2016."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "FBA98777-F211-4B3A-AD5D-96DBA7018758"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-02-24-paul-parsons.toml b/_data/talks/2026-02-24-paul-parsons.toml
new file mode 100644
index 0000000..23d37dc
--- /dev/null
+++ b/_data/talks/2026-02-24-paul-parsons.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Visualization and Judgment: Human-Centered Computing in Data-Rich Practice"
+date = 2026-02-24
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Visualization and computational systems are increasingly powerful, but their impact depends on a persistent, often under-specified factor—human judgment. Designers frame problems and negotiate constraints; users interpret, challenge, and coordinate action around system outputs over time, under uncertainty and constraint. In this talk, I present a research agenda on visualization and judgment in data-rich work, grounded in empirical studies and design-oriented analyses across three contexts. First, I study data visualization design practice, showing how professional designers frame problems and co-evolve problem and solution spaces—work that is often invisible in pipeline-oriented accounts of visualization. Second, I examine expert judgment in complex sociotechnical settings, where visualization, procedures, and automation reshape decision spaces and can either support adaptive performance or encourage brittle reliance. Third, I extend these insights to scientific cyberinfrastructure, where platforms exposing advanced computation and data services succeed or fail based on whether diverse communities can understand, adopt, and sustain capabilities in practice. This perspective complements advances in modeling, simulation, and visual analytics by surfacing design constraints, evaluation targets, and failure modes that matter as systems intersect with the constraints and contingencies of practice. I close with directions for judgment-aware visualization and human–AI support that preserve interpretability, accountability, and coordination in consequential domains."
+
+[[speakers]]
+name = "Paul Parsons"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Paul Parsons is an Associate Professor in the School of Applied and Creative Computing at Purdue University. His research focuses on interactive visualization systems and interfaces and how they are designed and used in complex sociotechnical settings. His work at the intersection of design practice and data visualization has been recognized with an NSF CAREER award and a recent IEEE VIS best-paper recognition. His broader research program has also been supported by NSF and NASA funding. His work appears in leading venues such as IEEE TVCG and ACM CHI. He leads the Design, Visualization, & Cognition (DVC) Lab, where his group studies how practitioners think, create, and collaborate in data-rich domains shaped by uncertainty, complexity, and ambiguity, drawing on perspectives such as design cognition, judgment and decision-making, and joint cognitive systems. He collaborates widely across disciplines at Purdue and beyond, including through the NSF-funded Cyberinfrastructure Center of Excellence SGX3 and the NASA-funded Resilient Extra Terrestrial Habitats Institute (RETHi)."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4473080D-7DC7-4341-BEBE-F4676A4C7328"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-02-26-zezhong-wang.toml b/_data/talks/2026-02-26-zezhong-wang.toml
new file mode 100644
index 0000000..edede88
--- /dev/null
+++ b/_data/talks/2026-02-26-zezhong-wang.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Designing Narrative-Driven Data Experiences"
+date = 2026-02-26
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "Evans Conference Room (WEB 3780)"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "In today’s data-saturated world, people are facing an “infodemic” of information overload. As public demand to interpret and use data grows, so does the urgency to rethink how we help broader audiences engage with data in meaningful ways. This talk explores how we can reconnect data with its storytelling roots to support understanding and agency. Dr. Wang will present research at the intersection of visual design, data visualization, and human-computer interaction, showing how interdisciplinary collaboration can open up new modes of communication. He will share empirical findings on how visual data narratives, such as data comics, can improve comprehension and engagement, as well as methods for crafting visual data stories that connect data to context and lived experience. Examples will draw from cross-disciplinary projects in environmental and healthcare data storytelling. The talk will conclude with future directions toward embedding narrative-driven data experiences into everyday tasks and decision-making."
+
+[[speakers]]
+name = "Zezhong Wang"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Dr. Zezhong Wang is currently a Postdoctoral Fellow at the Interactive Experiences Lab (ixLab) at Simon Fraser University, Canada, working with Dr. Sheelagh Carpendale. His research integrates visual design, data visualization, and HCI to foster public engagement with data. He holds a PhD from the University of Edinburgh (2022), where his dissertation, Creating Data Comics for Data-Driven Storytelling, received an IEEE VGTC Best Visualization Dissertation Award Honorable Mention in 2023. Zezhong actively engages in interdisciplinary collaboration, working alongside artists, computer scientists, healthcare professionals, and environmental scientists to bridge the gap between complex data and human experience."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "094A4DE9-23CD-4251-859B-45E291465447"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-02-27-kenny-marino.toml b/_data/talks/2026-02-27-kenny-marino.toml
new file mode 100644
index 0000000..dc71c71
--- /dev/null
+++ b/_data/talks/2026-02-27-kenny-marino.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Agents: Hype or Opportunity"
+date = 2026-02-27
+start_time = "13:40"
+end_time = "14:40"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Are so-called \"AI Agents\" a fad or a potentially impactful research area enabled by the rapid progress in large language models? In this talk I will try to strip away the marketing copy and look at what an agent actually is, returning to the classical understanding of the word, and investigate how powerful new language models can present new opportunities for research in embodied decision making. We begin by recounting the places where language models have been useful in agent-like problems, then looking at the emerging environments for investigating VLM/LLM agents including computer use and robotics, and finally discussing the frontier research challenges of LLM agents."
+
+[[speakers]]
+name = "Kenny Marino"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Kenneth Marino joined the Kahlert School of Computing at the University of Utah as an Assistant Professor in Fall 2025. His research focuses on integrating multimodal language models into embodied agent problems, including computer use, games, and robotics. Previously, he was a Research Scientist at Google DeepMind in NYC, where he worked on retrieval and embodied reasoning with language. He earned his PhD in 2021 from Carnegie Mellon University, advised by Abhinav Gupta, with a thesis on incorporating semantic knowledge into embodied systems. He received his undergraduate degree from the Georgia Institute of Technology."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "05077C5C-C3DB-43A8-B2E3-232CE8E1547D"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-02-bogdan-raita.toml b/_data/talks/2026-03-02-bogdan-raita.toml
new file mode 100644
index 0000000..190623e
--- /dev/null
+++ b/_data/talks/2026-03-02-bogdan-raita.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Solving Linear PDE by Machine Learning and Commutative Algebra"
+date = 2026-03-02
+start_time = "16:00"
+end_time = "17:00"
+series = "Data Science & AI Lecture Series"
+location = "LCB 222"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "We use the theory of linear pde systems with constant coefficients (Malgrange, Palamodov, Pommaret, Sturmfels) to implement a machine learning algorithm which generates solutions to arbitrary linear pdes. Since we preprocess the equations with computer algebra, our methods are applicable to arbitrary pde systems, irrespective of type (elliptic, hyperbolic, etc.) or order. We test our method for classical equations (wave, heat, Laplace) and discuss future applications to equations describing wave-related phenomena, for example direct and inverse problems involving Maxwell’s system and the elasticity equations."
+
+[[speakers]]
+name = "Bogdan Raita"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "5EAB83C5-371B-40F6-BA6A-B1D707AF721C"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-03-grace-guo.toml b/_data/talks/2026-03-03-grace-guo.toml
new file mode 100644
index 0000000..04f13bc
--- /dev/null
+++ b/_data/talks/2026-03-03-grace-guo.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Concepts and Counterfactuals: Human-Centered Interpretability in the Age of Foundation Models"
+date = 2026-03-03
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "Evans Conference Room (WEB 3780)"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Foundation models are increasingly deployed in high-stakes domains, yet their scale and opacity challenge traditional notions of AI interpretability. In this talk, I present two complementary strategies for human-centered interpretability: reasoning through concepts and probing through counterfactuals. I first present MiMICRI, a visualization tool developed with doctors at Cleveland Clinic that enables them to interactively create counterfactual medical images to examine how anatomical changes influence model predictions. By grounding explanations in domain-relevant visual features, this tool helps experts reason about model behavior using their established medical knowledge. Next, I will introduce Concept2Concept, a framework for auditing text-to-image models by characterizing their outputs as distributions over named, interpretable concepts. By analyzing the metrics of concept frequency, stability, and co-occurrence, we uncover hidden and sometimes harmful associations in image generation models and real-world training datasets. Finally, I conclude with my research agenda for developing new visualization tools and theoretical foundations that address the ongoing challenges of auditing and aligning the foundation models of today."
+
+[[speakers]]
+name = "Grace Guo"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Grace Guo is a Postdoctoral Fellow at Harvard University’s School of Engineering and Applied Sciences (SEAS). She received her Ph.D. in Human-Centered Computing from the Georgia Institute of Technology, where she was advised by Alex Endert. Her research sits at the intersection of visualization, explainable AI, and human-centered machine learning, with a focus on how AI interpretability tools can be designed for domain experts. To this end, Grace has collaborated with experts across healthcare, education, immunobiology, astrophysics, and causal analytics. She has previously worked at the Pacific Northwest National Laboratory and IBM Research, where she was awarded the IBM PhD Fellowship for her work on developing CausalVis. In her free time, she enjoys reading science fiction and mystery novels."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "7DC50148-CB01-4044-8C4C-44CB4C52ADB3"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-05-josh-levine.toml b/_data/talks/2026-03-05-josh-levine.toml
new file mode 100644
index 0000000..d4d2513
--- /dev/null
+++ b/_data/talks/2026-03-05-josh-levine.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Extracting, Visualizing, and Analyzing Topological Features with Discrete Representations"
+date = 2026-03-05
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "Evans Conference Room (WEB 3780)"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Topological features provide multi-scale summaries of the behavior of continuous data from diverse applications ranging from astrophysics to medicine. Nevertheless, computing them robustly is challenging due to numerical precision issues. A promising strategy is to first convert the input to a discrete representation that satisfies criteria introduced by Forman's discrete Morse theory. While numerous approaches exist to discretize the restricted case of gradient fields from scalar data, state-of-the-art algorithms for the general case of vector fields require expensive optimization procedures. In this talk, I will present recent work that uses local evaluation to create discrete vector fields in linear time from two-dimensional, triangulated vector fields. I will also frame this work within my contributions to the Topological ToolKit (TTK), an open source software platform for topological data analysis, led by collaborators at UPMC Sorbonne."
+
+[[speakers]]
+name = "Josh Levine"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Joshua A. Levine is an associate professor in the Department of Computer Science at University of Arizona. Prior to starting at Arizona in 2016, he was an assistant professor at Clemson University from 2012 to 2016, and he is a postdoctoral alumnus of the University of Utah’s SCI Institute, 2009 to 2012. He is a recipient of the 2018 DOE Early Career award. He received his PhD in Computer Science from The Ohio State University in 2009 after completing BS degrees in Computer Engineering and Mathematics in 2003 and an MS in Computer Science in 2004 from Case Western Reserve University. His research and teaching interests include visualization, topological analysis, geometric modeling, and computer graphics."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "839C55DE-F665-4466-92B6-AA27619F1E89"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-16-tenghao-huang.toml b/_data/talks/2026-03-16-tenghao-huang.toml
new file mode 100644
index 0000000..44f24c7
--- /dev/null
+++ b/_data/talks/2026-03-16-tenghao-huang.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Generalizable, Proactive, and Agentic Learning for Open-Ended Tasks"
+date = 2026-03-16
+start_time = "10:00"
+end_time = "11:00"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780 (Evans)"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Open-ended, human-like intelligence requires flexibility, proactivity, and social intelligence: we must learn subjective goals and adapt to novel, complex scenarios. . Current AI systems struggle to learn these behaviors because reward signals are unclear, task context information is incomplete, and the environment lacks observability. I outline a research agenda focused on building agentic learning systems that operate under such uncertainty. (1) Instead of enumerating task-specific heuristics, I propose training AI systems through self-play in an adversarial setting, where models learn by interacting, critiquing, and improving against dynamically evolving counterparts. (2) I equip agents with the ability to proactively gather missing information when task context is incomplete and 3) I reconstruct environment representations through memory to support long-horizon agentic reasoning. Together, I will show, these techniques enable AI systems to learn abstract objectives in tasks as varied as creative writing, multi-agent coordination and long-horizon problem solving, enhancing the creativity, usefulness, and strategic helpfulness of agents in open-ended environments."
+
+[[speakers]]
+name = "Tenghao Huang"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Tenghao Huang is a Ph.D. candidate in Computer Science at the University of Southern California. His research focuses on proactive and agentic AI systems for open-ended tasks, including social world models that simulate how conversations unfold and how decisions emerge from social interaction. His work is recognized by an EMNLP Outstanding Paper Award, an ISI Viterbi Fellowship, and has received media coverage from leading technology presses such as MIT Technology and Microsoft’s Future of Work. He has served as Area Chair roles for ACL, EMNLP, NAACL conferences and has organized a tutorial on Creative Planning with LLMs at NAACL 2025."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "55F9E8E6-3D93-488A-BDF4-24C6AFA70A6F"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-20-kate-isaacs.toml b/_data/talks/2026-03-20-kate-isaacs.toml
new file mode 100644
index 0000000..71b127f
--- /dev/null
+++ b/_data/talks/2026-03-20-kate-isaacs.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "A Matter of Audiences: Capturing and Reporting Reasoning and Results Around Data Visualizations"
+date = 2026-03-20
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Data visualizations are used throughout the data science process to facilitate the exploratory analysis and to report results. Frequently, these uses are neither separate nor solitary, acting as a medium for data science teams to collaboratively reason about data. This team-based data science work is often fast-paced and involves several forms of non-digital communication. I will discuss how we leverage gesture, sketch, and speech to create support for common meetings around data, specifically collaborative remote meetings and informal presentations, to aid this aspect of data science work. Then, focusing on the wider dissemination of results, I will discuss findings regarding the public's views on the use of data and AI in science videos."
+
+[[speakers]]
+name = "Kate Isaacs"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Kate Isaacs is an Associate Professor in the Kahlert School of Computing and the Scientific Computing an Imaging Institute at the University of Utah. She received her Ph.D. in computer science from the University of California, Davis and has undergraduate degrees in computer science, mathematics, and physics. She publishes in data visualization, high performance computing, and human-centered computing venues, with interests in complex analysis scenarios, such as those arising from research and data science teams. She has received an NSF CAREER award, a Department of Energy Early Career Research Program award, and a Presidential Early Career Award for Scientists and Engineers (PECASE)."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "371219E2-FBE5-4F33-9ECE-420FDACCB4AD"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-20-xueguang-ma.toml b/_data/talks/2026-03-20-xueguang-ma.toml
new file mode 100644
index 0000000..0b54de6
--- /dev/null
+++ b/_data/talks/2026-03-20-xueguang-ma.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Breaking Information Silos: Advancing Search Systems for Unified Information Seeking"
+date = 2026-03-20
+start_time = "10:00"
+end_time = "11:00"
+series = "Data Science & AI Lecture Series"
+location = "MEB 3147 (LCR)"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Information seeking has been fundamental to human advancement, enabling knowledge acquisition, decision-making, and innovation across disciplines. However, traditional information retrieval systems often rely on specialized pipelines optimized for specific retrieval tasks, causing information silos that hinder unified information seeking. In this talk, I will present our work in building unified document retrieval systems that break these information silos across three dimensions: (1) domain and language silos, where I demonstrate how LLM-based dense retrievers achieve strong generalizability across retrieval tasks and present frameworks for training small, generalizable retrievers through diverse LLM augmentation; (2) modality silos, where I introduce a paradigm shift from text-based retrieval that relies on content extraction to directly encoding document screenshots, preserving all information including text, images, and layout in unified dense representations; and (3) space silos, where we show the importance of LLM-powered search agents in seeking and gathering information across disparate sources, and present fair and transparent evaluation benchmarks for assessing deep-search systems. I will conclude by discussing future directions that further pave the way toward building truly unified retrieval systems for seamless information seeking across world knowledge."
+
+[[speakers]]
+name = "Xueguang Ma"
+affiliation = ""
+website = "https://mxueguang.github.io/"
+photo = ""
+bio = "Xueguang Ma is currently a last-year PhD at the David R. Cheriton School of Computer Science at University of Waterloo, advised by Prof. Jimmy Lin. His research focuses on Information Retrieval (IR) and Natural Language Processing (NLP), with an overarching goal to make it easy for people and intelligent systems to access, understand, and interact with world information. His work has been published in top IR and NLP venues such as SIGIR, ACL, EMNLP, and NeurIPS. More details on his website: https://mxueguang.github.io/."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "B9FC8798-95E6-4D3B-AE4F-AB065B24ED0A"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-23-benjie-wang.toml b/_data/talks/2026-03-23-benjie-wang.toml
new file mode 100644
index 0000000..6e967e8
--- /dev/null
+++ b/_data/talks/2026-03-23-benjie-wang.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Bridging the Formalization Gap for Generative AI"
+date = 2026-03-23
+start_time = "10:00"
+end_time = "11:00"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Generative models, such as large language models and diffusion models, have tremendously increased the scope of problems that AI can address. As such, there is a significant trend toward incorporating generative AI to automate tasks across computing and more broadly, from controlling robotics systems, to software generation and testing, to searching over scientific knowledge. However, there remains a significant formalization gap between the domain knowledge, theories, and logical and semantic constraints that are vital to applications, and the statistical patterns over natural data represented by large generative models. In this talk, I will demonstrate how we can systematically bridge this formalization gap towards more trustworthy AI. First, drawing from examples and applications in my research, I will show how we can utilize suitable intermediate representations of probability distributions to bridge between formal language and generative models at scale. Then, I will discuss how these practical methods are underpinned by my work advancing the mathematical and computational foundations underlying these tractable representations of probability distributions."
+
+[[speakers]]
+name = "Benjie Wang"
+affiliation = ""
+website = ""
+photo = ""
+bio = """
+Benjie Wang is a postdoctoral researcher in the Statistical and Relational Artificial Intelligence (StarAI) lab in the Computer Science Department at UCLA. Dr. Wang's research interests are in artificial intelligence, including deep generative models, probabilistic machine learning, sequence and language modeling, and formal reasoning. His work develops theory-driven and scalable methods for understanding and controlling generative models, by studying the mathematical foundations, architecture, and manipulation of representations of high-dimensional probability distributions.
+
+Previously, he was a research fellow at the Simons Institute for the Theory of Computing at UC Berkeley in Fall 2023. Dr. Wang obtained his DPhil in Computer Science from the University of Oxford advised by Prof. Marta Kwiatkowska in 2023, his MSc in Statistical Science from the University of Oxford in 2019, and his BA in Mathematics from the University of Cambridge in 2018.
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "0903B995-B210-4B5E-9219-3897231C5A84"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-25-dick-sadler.toml b/_data/talks/2026-03-25-dick-sadler.toml
new file mode 100644
index 0000000..379ed6a
--- /dev/null
+++ b/_data/talks/2026-03-25-dick-sadler.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "What Happens in the 10 Years Following a State Government-Caused Environmental Injustice? Flint's Progress Since Its Water Crisis"
+date = 2026-03-25
+start_time = "16:00"
+end_time = "17:00"
+series = "Data Science & AI Lecture Series"
+location = "MLIB 1110 | Zoom: 890 9876 9672, Passcode: 156565"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "The Flint Water Crisis was the result of decades of deliberate disinvestment and state government ineptitude. In this talk, Dr. Sadler will discuss how the crisis unfolded, and how his research - examining environmental exposures, neighborhood conditions, and blood lead levels - revealed its scale. He will also address what has changed in Flint since then, including the massive new investments in the city that have brought hope in the wake of catastrophe."
+
+[[speakers]]
+name = "Dick Sadler"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "94D08A7C-78DD-4581-8F09-ECDF6C333DE4"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-26-bailing-lyu.toml b/_data/talks/2026-03-26-bailing-lyu.toml
new file mode 100644
index 0000000..14d7e6e
--- /dev/null
+++ b/_data/talks/2026-03-26-bailing-lyu.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Navigating the Human-AI Nexus in Education: Bridging Cognitive Science, Learning Analytics, and Intelligent Systems"
+date = 2026-03-26
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "Evans Conference room (WEB 3780)"
+zoom = "https://utah.zoom.us/j/87171666093"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Artificial intelligence is increasingly transforming education, reshaping how students engage with learning and how instructors design and deliver instruction. However, critical gaps remain in the field of AI in Education (AIED): (1) a frequent lack of grounding in the cognitive and learning sciences when developing AI pedagogical tools, (2) a \"black box\" regarding how students process and interact with AI-supported environments, and (3) a limited emphasis on meaningful human involvement in how AI is applied in practice. This talk presents a series of research projects on AI-augmented learning and teaching that address these gaps by integrating cognitive science into AI-powered educational technologies, examining human-AI interaction, and centering human agency in their application. Specifically, using teachable agents as an example, it will discuss (1) how pedagogical AI can be designed using theoretically grounded learning principles, (2) how students' learning processes unfold in these environments, and (3) how educators and learners can actively leverage AI to co-create and navigate engaging experiences that enhance learning outcomes. Employing experimental research, learning analytics, and educational data mining, these studies examine the intersection of human cognition and AI-driven learning. By aligning AI with evidence-based pedagogical strategies, this work advances our understanding of how intelligent technologies can foster deeper learning, positive learning experiences, personalized instruction, and adaptive support across diverse educational contexts."
+
+[[speakers]]
+name = "Bailing Lyu"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Dr. Bailing Lyu is an Assistant Professor at Auburn University, specializing in AI-driven educational technologies, learning analytics, and cognitive learning sciences. She earned her Ph.D. in Educational Psychology from Pennsylvania State University and subsequently served as a postdoctoral researcher at the University of Utah, focusing on conversational AI. Her research explores how AI can be designed and applied across both student-facing and instructor-facing contexts. For students, she focuses on designing AI systems that foster cognitive engagement while enhancing interest, motivation, and emotional experiences. In parallel, she investigates how AI can empower instructors to develop instructional and AI-embedded materials, such as generative interactive visuals, that make abstract concepts more concrete, accessible, and engaging. Methodologically, she applies learning analytics, educational data mining, and experimental research to study interactions with these multimodal resources. Her impactful work has contributed to large-scale funded projects, including the $10 million ALTER-Math initiative (supported by the Schmidt, Gates, and Walton Family Foundations) as well as several NSF-funded projects. A highly productive scholar, Dr. Lyu has an extensive publication record with 24 submitted or published journal articles and over 40 conference presentations advancing the future of AI in education."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "B321A7B0-A0C4-4F09-9053-9F72C727820D"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-27-erdogan-kaya.toml b/_data/talks/2026-03-27-erdogan-kaya.toml
new file mode 100644
index 0000000..1dda4c3
--- /dev/null
+++ b/_data/talks/2026-03-27-erdogan-kaya.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "From Self-Efficacy to Systemic Change: Building an Equitable Computing Education Research Program"
+date = 2026-03-27
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "Evans Conference room (WEB 3780)"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "This talk examines a foundational question in computational STEM education: to what extent can targeted interventions improve pre-service elementary teachers’ computational thinking teaching efficacy beliefs? I present a study of pre-service elementary teachers who participated in a three-week CT intervention integrating EV3 robotics, Code.org, and Zoombinis. Using the CTTEBI in a pre-post design, results show significant gains in personal CT teaching efficacy, pointing to important directions for future work. I then present my current AI education research program, including the “Educate AI” project developing AI literacy curriculum through linguistically inclusive elementary robotics, the Rural AI project integrating AI concepts for rural upper elementary students, and the Compose with AI platform guiding grades 4-8 students in critically evaluating AI-generated content. I will also discuss my emerging research agenda, including several NSF proposals under review focused on AI literacy across K-16 settings. I close with a vision for how University of Utah’s collaborative structure across the Scientific Computing and Imaging (SCI) Institute, Department of Educational Psychology, College of Education, and partner departments represents an ideal ecosystem to advance this agenda."
+
+[[speakers]]
+name = "Erdogan Kaya"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Dr. Kaya holds a joint appointment with the College of Education and the Division of Data Science at the University of Texas at Arlington. He is an assistant professor of computing education with a Ph.D. in Curriculum and Instruction, a B.S. in Chemical Engineering, an M.S. in Computer Science and Engineering, and an M.S. in Machine Learning Engineering from George Mason University. His research focuses on AI literacy, computational thinking, and computing education through equity lenses. He has secured more than $5 million in external research funding from NSF, Amazon, and Google, and has developed projects including the Educate AI curriculum, the Rural AI initiative, and the Compose with AI platform. Dr. Kaya advocates for research, teaching, and learning as a unified enterprise that benefits students and society. He has received numerous teaching awards, including the NCWIT Aspirations in Computing Educator Award and the ASEE Southeastern Section Outstanding New Teacher Award, and has contributed to educational outreach programs including Code.org and FIRST Robotics competitions."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "4248F7E7-849D-442E-8343-4D066EDE90E7"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-30-xiaoling-hu.toml b/_data/talks/2026-03-30-xiaoling-hu.toml
new file mode 100644
index 0000000..c842006
--- /dev/null
+++ b/_data/talks/2026-03-30-xiaoling-hu.toml
@@ -0,0 +1,33 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Principled Learning for Medical AI: Structure, Reliability, and Interpretability"
+date = 2026-03-30
+start_time = "10:00"
+end_time = "11:00"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+The widespread deployment of AI in medicine demands not only predictive accuracy but also structural awareness, reliability under uncertainty, and interpretability for clinical trust. In this talk, I will present a unified research agenda toward principled learning for medical AI, grounded in these core pillars.
+
+First, I will discuss how incorporating explicit structure, such as topology and spatial priors, into neural networks enhances the model's ability to reason about fine-grained anatomical and pathological features, which are critical for tasks like brain and tumor segmentation. Second, I will focus on reliability, exploring how we can quantify and mitigate uncertainty arising from imperfect labels, limited data, and domain shifts, using methods such as distributional modeling, hyperparameter learning, and probabilistic inference. Third, I will show how these approaches naturally support interpretability, enabling AI systems to communicate meaningful representations that align with human clinical understanding.
+
+Through applications in radiology, pathology, neuroimaging, and large-scale population datasets, I will demonstrate how these principles facilitate scalable annotation, robust generalization, and scientific discovery. I will conclude with future directions aimed at generalizing these principles to multimodal learning, real-world deployment, and next-generation AI systems in medicine.
+"""
+
+[[speakers]]
+name = "Xiaoling Hu"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Xiaoling Hu is a postdoctoral research fellow at Harvard Medical School. He received his Ph.D. in Computer Science from Stony Brook University. His research focuses on Machine Learning for Healthcare, with an emphasis on developing core AI/ML algorithms for healthcare applications. His work has been published in leading venues across machine learning, computer vision, and medical imaging, including NeurIPS, ICLR, AISTATS, CVPR, ICCV, ECCV, Medical Image Analysis, and MICCAI. Several of his papers have been selected for oral or spotlight presentations. Xiaoling has organized multiple tutorials and workshops at top-tier conferences and served as Area Chairs for venues such as NeurIPS, CVPR, AISTATS, and MICCAI. He is also a recipient of the prestigious Catacosinos Fellowship, awarded to SBU graduate students with exceptional research achievements."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "639767DF-6829-4B92-86CD-02EB8F5DBC79"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-03-31-he-yin.toml b/_data/talks/2026-03-31-he-yin.toml
new file mode 100644
index 0000000..78da4de
--- /dev/null
+++ b/_data/talks/2026-03-31-he-yin.toml
@@ -0,0 +1,35 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Advancing GeoAI and Earth observation for environmental monitoring"
+date = 2026-03-31
+start_time = "10:45"
+end_time = "11:45"
+series = "Data Science & AI Lecture Series"
+location = "Evans Conference room (WEB 3780)"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Landscapes around the world are changing rapidly, with important consequences for sustainability, climate resilience, and society. Yet monitoring these changes across regions and scales remains difficult. In this talk, I present a research program that combines multi-sensor Earth observation, geospatial artificial intelligence (GeoAI), and land system science to better understand how land systems are changing, what drives those changes, and why they matter.
+
+I begin by presenting my studies using satellite image time series to map land use change and, in collaboration with environmental scientists and ecologists, to examine its implications for carbon sequestration and biodiversity. These studies also reveal key limitations of conventional remote sensing approaches, including sensor constraints, limited transferability, and scarce training data. I then show how these challenges motivate my more recent work in sensor fusion, physics-informed machine learning, and deep learning with very-high-resolution imagery — applied to problems ranging from irrigation water use and wildfire-invasive species interactions to conflict-induced environmental damage. I conclude by discussing the broader goal of building GeoAI models for environmental monitoring that are informed by physical processes, transferable across contexts, and useful for real-world decision-making.
+"""
+
+[[speakers]]
+name = "He Yin"
+affiliation = ""
+website = ""
+photo = ""
+bio = """
+Dr. He Yin is an Associate Professor in the Department of Geography at Kent State University and Director of the Remote Sensing and Land Science Lab. He earned his PhD in Geography from Humboldt University of Berlin and completed postdoctoral training at the University of Wisconsin–Madison.
+
+His research combines geospatial artificial intelligence (GeoAI), Earth observation, and interdisciplinary methods to monitor landscape change and assess its environmental and societal impacts. He serves as principal investigator on projects supported by NASA, the National Science Foundation, Lawrence Livermore National Laboratory, and the Center for International Forestry Research, and currently advises the United Nations Environment Programme (UNEP) and the United Nations Office for Project Services (UNOPS). His work has received the European Space Agency Earth Observation Excellence Team Award (2025) and the American Association of Geographers Media Achievement Award (2026).
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "C62FB35B-E0E0-4D66-9ED2-9306D3E58CBB"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-01-md-mostafijur-rahman.toml b/_data/talks/2026-04-01-md-mostafijur-rahman.toml
new file mode 100644
index 0000000..32d50e5
--- /dev/null
+++ b/_data/talks/2026-04-01-md-mostafijur-rahman.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Efficient and Reliable AI for Real-World Healthcare Deployment"
+date = 2026-04-01
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Healthcare is one of the highest-impact domains for AI, yet reliable deployment at scale remains difficult. To truly improve patient care and clinical workflows, AI must operate under real clinical constraints, not just in ideal lab settings. In practice, deployment is limited by high compute and memory costs, scarce labeled data, and distribution shifts across sites and time. Many clinically important findings are also rare and long-tailed, which makes generalization especially challenging. My research makes deployability a design objective by developing methods that stay accurate under strict resource and data constraints. In this talk, I will first discuss high-performance lightweight deep learning architectures built by redesigning core building blocks. I will then present training-time generative supervision strategies that improve data efficiency and generalization to rare and long-tailed cases with no inference overhead. I will conclude with a forward-looking direction toward real-time perception for surgical assistance, where reliable performance under strict constraints is non-negotiable."
+
+[[speakers]]
+name = "Md Mostafijur Rahman"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Md Mostafijur Rahman is a Ph.D. candidate at The University of Texas at Austin, advised by Radu Marculescu. His research sits at the intersection of AI, biomedical imaging, and computer vision, with a focus on building efficient, reliable, and scalable AI systems for deployment in healthcare under real-world constraints. His work has been translated to practice through research internships at GE Healthcare, the National Institutes of Health (NIH), and Bosch Research. He has published over 20 peer-reviewed papers in venues including CVPR, NeurIPS, MICCAI, and ICCV, with several works selected for Spotlight and Oral presentations. His research contributions have been recognized by the NIH Summer IRTA Fellowship, the Texas Health Catalyst Award, and the Discovery to Impact Award."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EFFFD488-93B9-4050-9268-5ADF6E6F5E70"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-06-qiang-ji.toml b/_data/talks/2026-04-06-qiang-ji.toml
new file mode 100644
index 0000000..ee4cb72
--- /dev/null
+++ b/_data/talks/2026-04-06-qiang-ji.toml
@@ -0,0 +1,39 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Towards Data-Efficient, Trustworthy, and Generalizable AI for Visual Understanding"
+date = 2026-04-06
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Artificial Intelligence (AI) has achieved remarkable progress and is increasingly integrated across a wide range of fields, fueling what many describe as the fourth industrial revolution. However, behind this widespread enthusiasm lie fundamental limitations. Today’s AI systems face three major challenges: (1) an insatiable demand for large-scale labeled data, (2) limited trustworthiness due to inadequate uncertainty quantification, and (3) poor generalization across domains. These challenges cannot be addressed simply by scaling data and computation; instead, they require foundational advances in theory and methodology.
+
+In this talk, I will present recent research from my lab that addresses these challenges in a variety of computer vision tasks. To improve data efficiency and generalization, I will introduce our work on knowledge-augmented deep learning, where prior knowledge from diverse sources is systematically identified, encoded, and integrated with data-driven neural networks. This approach leads to hybrid neural-symbolic models that are both more data-efficient and more generalizable. To enhance model trustworthiness and explainability, I will discuss our advances in Bayesian deep learning. First, I will present our work on a Bayesian Transformer framework for accurate and robust human activity recognition. I will then introduce our work on uncertainty attribution, which identifies the sources of uncertainty in deep models and leverages this information for uncertainty mitigation and improved model performance. Finally, I will highlight our recent work on causal deep learning for addressing domain generalization. I will introduce a neural causal model that learns domain-invariant representations by identifying and eliminating spurious correlations arising from data biases.
+
+Together, these efforts aim to advance a new generation of AI systems that are more data-efficient, trustworthy, and robust, enabling reliable deployment across a range of domains including human behavior understanding, medical imaging, scientific discovery, and human–robot interaction.
+"""
+
+[[speakers]]
+name = "Qiang Ji"
+affiliation = ""
+website = ""
+photo = ""
+bio = """
+Dr. Qiang Ji received his Ph.D. in Electrical Engineering from the University of Washington. He is currently a Professor in the Department of Electrical, Computer, and Systems Engineering at Rensselaer Polytechnic Institute (RPI). From 2009 to 2010, he served as a Program Director at the National Science Foundation (NSF), where he managed NSF’s research programs in computer vision and machine learning. He has also held teaching and research positions at the University of Illinois at Urbana–Champaign, Carnegie Mellon University, the University of Nevada, and the Air Force Research Laboratory.
+
+Prof. Ji’s research focuses on computer vision, probabilistic graphical models, Bayesian deep learning, and causal machine learning, with applications across a wide range of domains including human behavior analysis, medical imaging, intelligent transportation, and human–robot interaction. He has published over 400 papers in leading journals and conferences and has received multiple awards recognizing his contributions to the field.
+
+Prof. Ji has served the research community extensively as an editor for several IEEE and international journals and as a General Chair, Program Chair, Area Chair, and Program Committee member for numerous major international conferences and workshops. He is a Fellow of both the IEEE and the International Association for Pattern Recognition (IAPR).
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "DF62302A-7BD5-4372-9590-757DB648B1A9"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-07-si-chen.toml b/_data/talks/2026-04-07-si-chen.toml
new file mode 100644
index 0000000..43af129
--- /dev/null
+++ b/_data/talks/2026-04-07-si-chen.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Advancing AI Literacy and Human-Centered AI for Teaching and Learning"
+date = 2026-04-07
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Artificial intelligence (AI) is rapidly transforming education and how people learn, teach, and prepare for the future. Yet many systems are still built around technical capabilities rather than the real needs of students, educators, and families. In this talk, I present a human-centered design research agenda that advances both AI literacy and AI for teaching and learning across students, families, and educators. First, I define and conceptualize generative AI literacy across children and parents by co-designing measurement frameworks and interactive tools. This work enables families to build a shared understanding of AI while supporting its critical and responsible use in everyday self-directed learning contexts. Second, I present AI Academy, an faculty-facing professional development program and badge system that expands institutional capacity for AI. Through curriculum design and lightweight tools, this work supports faculty in integrating generative AI into teaching while strengthening their AI literacy and maintaining pedagogical and disciplinary goals. Lastly, I present AI-powered tutoring systems and other AI interactions I have developed with college learners with disabilities, including LLM-based chatbots, particularly for Deaf and Hard of Hearing students.The talk is intended for a broad, interdisciplinary audience."
+
+[[speakers]]
+name = "Si Chen"
+affiliation = ""
+website = ""
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EE591036-3BA0-478C-A395-B66A283B0E4C"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-08-fahim-faisal.toml b/_data/talks/2026-04-08-fahim-faisal.toml
new file mode 100644
index 0000000..1865a71
--- /dev/null
+++ b/_data/talks/2026-04-08-fahim-faisal.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Multilingual Model Adaptation for Under-Served Languages"
+date = 2026-04-08
+start_time = "10:00"
+end_time = "11:00"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Language models with multilingual capabilities serve as crucial touchpoints for improving the inclusion of underrepresented languages in Natural Language Processing (NLP). This research investigates the structural sources of linguistic underrepresentation and explores strategies for improving the adaptation of low-resource language varieties through the development of linguistically grounded resources. We first examine the extent of multilingual and geographic representation gaps across three key dimensions of language modeling: datasets, model architecture, and model-generated text. Next, we introduce DialectBench, an initiative designed to evaluate language variation in the form of dialects and language varieties—an aspect often overlooked in NLP benchmarks, which primarily focus on standardized language forms. To address the challenges revealed by these analyses, we propose two adaptation frameworks. The first introduces phylogenetic adapter hierarchies that exploit language-family structure to enable zero-shot transfer across related languages. The second presents a pivot-based reinforcement learning approach that leverages high-resource expert models to transfer reasoning alignment without requiring target-language annotations. Together, these linguistically motivated adaptation strategies aim to improve the performance of language models on underrepresented languages and dialects, ultimately contributing to more equitable and accessible NLP systems for diverse linguistic communities."
+
+[[speakers]]
+name = "Fahim Faisal"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Fahim Faisal is a final-year Ph.D. candidate in the Department of Computer Science at George Mason University, where he is a member of the GMU NLP Lab under the supervision of Dr. Antonios Anastasopoulos. His research focuses on adapting language models for low-resource and underrepresented languages, spanning both the creation of linguistic resources and the systematic evaluation of state-of-the-art large language models (LLMs) across language varieties and dialects. His work also examines how modern LLMs handle domain- and policy-specific safety alignment, multilingual reasoning, and cultural disparities embedded in language modeling. His research is driven by the overarching goal of democratizing AI and NLP, ensuring that users from all linguistic, cultural, and demographic backgrounds receive equitable utility from advances in machine intelligence. His work DialectBench received the Best Social Impact Award at ACL 2024, recognizing its contribution to inclusive and socially responsible NLP research."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "26096E39-EACA-451B-949F-982C77600A3B"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-10-chase-neumann.toml b/_data/talks/2026-04-10-chase-neumann.toml
new file mode 100644
index 0000000..81417d7
--- /dev/null
+++ b/_data/talks/2026-04-10-chase-neumann.toml
@@ -0,0 +1,40 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Building and Utilizing Foundation Models for Drug Discovery and Clinical Development"
+date = 2026-04-10
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+The journey toward decoding biology at scale began with a focus on high-dimensional cellular morphology. At Recursion, our differentiation centered on deep learning models designed to learn biological representations directly from imaging, enabling predictive inference at a massive scale. By leading multiple cross-functional teams from early-stage discovery through to early clinical development, we demonstrated the power of this "inference-first" philosophy. A primary highlight of these efforts was the RBM39 program, a novel molecular glue degrader discovered entirely through computational inference rather than traditional screening. This success proved that models could identify complex biological mechanisms; however, moving from cellular discovery to comprehensive patient care requires a leap into even higher-dimensional, clinical data.
+
+Valinor represents the next evolution of this mission. We build multimodal clinical foundation models, co-designing data collection and model architecture to maximize signal within and across complex modalities. We have proprietary access to patient cohorts and collaborate with biobanks, clinical trial sites, and academic partners, giving us unique data advantages at scale. Our approach is to first build the best unimodal patient representations across modalities—including DNA, transcriptomics, proteomics, cfDNA, histopathology, and patient reports—and then fuse them.
+
+We have shown that attention-based fusion consistently outperforms unimodal approaches while making the contributions of different modalities interpretable. This enables genuine clinical reasoning: the model can chain evidence across modalities, explain which features drive a prediction, and engage with clinicians in natural language. Ultimately, we envision a virtual patient that reasons over the full spectrum of a patient's biology the way an expert clinician would, but at a scale and resolution no human can match.
+"""
+
+[[speakers]]
+name = "Chase Neumann"
+affiliation = "PhD"
+website = ""
+photo = ""
+bio = """
+I believe the next generation of life-saving medicines won't just be discovered; they will be engineered at the intersection of biology and machine learning.
+During my time at Recursion, I operated at the frontier of "AI for Drug Discovery," translating high-dimensional data into actionable therapeutic programs. I led cross-functional teams through the critical transition from late-stage discovery into early clinical development, ensuring that AI-driven insights were successfully translated into clinical-ready drug candidates.
+
+With a PhD in Translational Medicine from CCLCM at Case Western Reserve University, I bridge the gap between translational oncology and computational innovation. My work has focused on deconstructing complex disease mechanisms and scaling them through automated, industrialized platforms.
+
+I am a passionate advocate for the "TechBio" shift—moving away from serendipity and toward a predictable model of drug discovery to get better medicines to patients, faster. In late 2025, I co-founded Valinor Discovery to bridge the gap in clinical translation using foundation models trained on real-world multi-modal clinical data.
+"""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "C6B74209-73DA-4893-9AFD-17260B143711"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-13-jihyun-rho.toml b/_data/talks/2026-04-13-jihyun-rho.toml
new file mode 100644
index 0000000..2b311da
--- /dev/null
+++ b/_data/talks/2026-04-13-jihyun-rho.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Supporting responsible use of AI in visual-based teaching and learning"
+date = 2026-04-13
+start_time = "13:00"
+end_time = "14:00"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Artificial intelligence (AI) is increasingly integrated into educational contexts, reshaping how teachers and students generate and interpret visual representations such as diagrams, illustrations, and data visualizations. While AI offers efficiency and flexibility, it also introduces inaccuracies and misleading interpretations that can undermine learning. In this talk, I present a research program centered on supporting the responsible use of AI in visual-based teaching and learning by promoting visual literacy. Grounded in design-based research, I design and evaluate AI-augmented learning environment where educators and students engage with AI-generated outputs as critical inquirers. This work shows that these approaches improved interpretive accuracy, deepened reasoning about visual representations, and reduce over-reliance on AI. Together, this work contributes design principles for fostering responsible AI use in visual-based educational contexts."
+
+[[speakers]]
+name = "Jihyun Rho"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Jihyun Rho is a Postdoctoral Researcher at the University of Florida. She earned her Ph.D. in the Learning Sciences program in the Department of Educational Psychology at the University of Wisconsin–Madison. Her research uses design-based research and learning analytics to design and evaluate AI-augmented learning environments for visual-based learning. She studies how learners and educators interact with AI-generated representations, combining qualitative analysis of interaction processes with experimental approaches to examine learning outcomes, reasoning, and reliance on AI. Her scholarship appears in leading journals and conferences in learning sciences and AI in education."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "E9585003-22AD-4747-8A94-B9E79CA677A9"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-14-sameer-honwad.toml b/_data/talks/2026-04-14-sameer-honwad.toml
new file mode 100644
index 0000000..29040e2
--- /dev/null
+++ b/_data/talks/2026-04-14-sameer-honwad.toml
@@ -0,0 +1,31 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Move Slow and Build Community"
+date = 2026-04-14
+start_time = "10:30"
+end_time = "11:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Socio-emotional learning (SEL) and reflection are foundational components of quality education, yet they remain underrepresented in many school curricula. Reflection supports scientific thinking and inquiry, while SEL equips students with the emotional resilience needed to take risks, embrace failure, and navigate the uncertainties inherent in learning and innovation. Together, these competencies are essential not only for academic and professional growth, but for everyday wellbeing.Implementing SEL and reflective practices in rural India presents distinct challenges. Limited access to trained professionals and consistent resources is compounded by cultural stigma around openly discussing emotions, barriers that make it difficult for students to develop these critical skills in traditional school settings.
+
+To address this gap, we designed a conversational chatbot tailored for students in rural India, providing a low-barrier, accessible space for daily reflection and emotional expression. Importantly, the chatbot is also designed to serve as a springboard for teachers to introduce concepts of AI literacy, enabling meaningful classroom conversations about how AI systems work, their limitations, and the ethical considerations surrounding their use. This presentation presents the co-design process behind the chatbot, highlighting the collaborative contributions of an interdisciplinary team comprising teachers, therapists, computer scientists, and learning scientists. The presentation discusses how this cross-disciplinary approach shaped a tool intended to help students engage with their socio-emotional selves and build a habit of reflective thinking in their daily lives.
+"""
+
+[[speakers]]
+name = "Sameer Honwad"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Sameer Honwad, Ph.D., is an Assistant Professor of Learning Sciences in the Department of Learning and Instruction at SUNY Buffalo. His research explores how technology can support learning in culturally diverse communities worldwide through participatory design approaches. At the heart of his work are long-term, trust-based partnerships with Indigenous communities in Idaho, rural and Indigenous communities in Bhutan and India, and urban communities in New Orleans. Rather than parachuting in with ready-made tools, Dr. Honwad co-designs technology-enhanced learning environments alongside teachers, parents, local leaders, and students, positioning community members as essential partners, not subjects of study. His current research examines how rural communities in South Asia make sense of and use artificial intelligence in everyday life. Grounded in participatory design-based research, Dr. Honwad centers communities that have historically faced systemic barriers to technology access due to class, race, gender, and caste. He has served as PI and Co-PI on multiple National Science Foundation (NSF) grants focused on designing technologies that help learners grapple with complex, real-world problems rooted in their own communities. Throughout his career he has taught courses that focus on how people learn within their cultures and the design of technology-enhanced learning environments that honor local knowledge, needs and aspirations. He is currently in the process of building a Global Design Studio that would be an interdisciplinary hub connecting scholars, students, and community members worldwide to think critically about AI and its role in everyday life."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "C9891AD1-60F7-4382-8F93-BF3F29069EF6"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-15-amirali-abdullah.toml b/_data/talks/2026-04-15-amirali-abdullah.toml
new file mode 100644
index 0000000..fc4b34e
--- /dev/null
+++ b/_data/talks/2026-04-15-amirali-abdullah.toml
@@ -0,0 +1,30 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Controlling LLM's via Activation Geometry"
+date = 2026-04-15
+start_time = "10:30"
+end_time = "11:00"
+series = "Data Science & AI Lecture Series"
+location = "Dinosaur Room (CSC 206)"
+zoom = "https://utexas.zoom.us/j/84742203545?pwd=lUosaf3T6bkIS1QAaIQYHYiiClE2ZP.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = """
+Controlling the behavior of large language models at inference time is an increasingly important problem. In this talk, I present a simple and unified approach to steering model behavior based on activation geometry. By learning a single classifier over hidden representations, we can derive directions that control multiple attributes such as helpfulness, style, or safety, and compose them dynamically without retraining.
+This framework enables flexible, low cost control of model outputs and highlights a geometric view of representation space beyond fixed linear directions. I will discuss empirical results showing how this approach supports multi attribute control in practice, and briefly outline how such steering mechanisms can be useful in scientific settings where reliable and interpretable model behavior is critical. Our recent followup work suggests that similar activation level interventions can extend across modalities, enabling systematic analysis and control in text to image models through composable operations.
+"""
+
+[[speakers]]
+name = "Amirali Abdullah"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Amirali Abdullah is a Lead AI Researcher at Thoughtworks Inc and a Research Advisor at Martian Learning. His research focuses on the interpretability and control of large language models, with particular emphasis on activation-level steering, representation geometry, and the structure of learned features."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "8464956E-1112-4B71-B0CB-5D76A25D4B25"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-15-chengbin-deng.toml b/_data/talks/2026-04-15-chengbin-deng.toml
new file mode 100644
index 0000000..4731f4c
--- /dev/null
+++ b/_data/talks/2026-04-15-chengbin-deng.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "AI-Driven Environmental Intelligence: Scalable and System-Level Approaches for Environmental Decision Making"
+date = 2026-04-15
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 3780"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = "Environmental systems are becoming more complex, dynamic, and tightly connected to human activities, yet much of our current work remains focused on isolated models or static mapping. In this talk, a framework will be present and discussed that uses AI to move beyond observation toward a more integrated understanding of environmental systems. The central idea is to link geospatial data, models, and real-world decisions so that environmental information can better reflect changing conditions across space and time while remaining meaningful for researchers and stakeholders. By drawing on some recent federally supported projects, this talk will show how this framework can capture large scale environmental dynamics, account for social and local context, and support more informed decision processes in various settings, ranging from urban systems to extreme events. Instead of using AI just as a tool, this work views it as part of a broader system that shapes how environmental problems are understood and acted, with the goal of advancing a more adaptive and practical form of environmental intelligence."
+
+[[speakers]]
+name = "Chengbin Deng"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Dr. Chengbin Deng is an Associate Professor in the Department of Geography and Environmental Sustainability and Director of the Center for Spatial Analysis at the University of Oklahoma (OU). His research focuses on advancing AI for environmental monitoring, with an emphasis on integrating Earth observation, geospatial data, and decision systems to address challenges in climate, hazards, and urban environments. His work has been supported by major federal agencies including NASA, the U.S. National Science Foundation, and USGS, where he serves as Principal Investigator on multiple externally funded projects. He has been recognized among the World’s Top 2% Scientists by Stanford University and Elsevier and received the Outstanding Research Award from the College of Atmospheric and Geographic Sciences at OU. Currently, Dr. Deng serves on NASA Land Cover and Land Use Change Science Team, and is Advisor of NASA FINESST Future Investigator and an NSF EPSCoR RII Research Fellow."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "5EA5C977-9267-4F10-B521-AC063406ADC6"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-15-shiqi-yu.toml b/_data/talks/2026-04-15-shiqi-yu.toml
new file mode 100644
index 0000000..a68a7bb
--- /dev/null
+++ b/_data/talks/2026-04-15-shiqi-yu.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "AI for Multi-Wavelength X-ray Analysis"
+date = 2026-04-15
+start_time = "10:00"
+end_time = "10:30"
+series = "Data Science & AI Lecture Series"
+location = "Dinosaur Room (CSC 206)"
+zoom = "https://utexas.zoom.us/j/84742203545?pwd=lUosaf3T6bkIS1QAaIQYHYiiClE2ZP.1"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Accurate parameter estimation in X-ray astronomy typically relies on traditional methods, such as likelihood-based spectral fitting, which can be computationally prohibitive as model complexity and data dimensionality increase. In this talk, I present a neural network-based framework designed to bypass iterative fitting by directly mapping spectral observations to physical parameters. Using the Circinus galaxy as a benchmark, we demonstrate how architectures trained on synthetic data from theoretical models can recover intrinsic properties, such as column density and torus geometry, with both high speed and high precision. This approach maintains the physical rigor required for broadband analysis with multiple telescopes while significantly reducing inference time. I will discuss the challenges and resolutions associated with training and predicting on multi-instrument data, as well as the potential for these AI-driven methods to enable large-scale systematic studies across various astrophysical sources and other scientific applications."
+
+[[speakers]]
+name = "Shiqi Yu"
+affiliation = ""
+website = ""
+photo = ""
+bio = "Prof. Yu is a Research Assistant Professor in the Department of Physics and Astronomy at the University of Utah. Their research focuses on the intersection of high-energy astrophysics and advanced computational methods. As a member of the IceCube Collaboration and former co-lead of the Reconstruction and Machine Learning working group, they have extensive experience applying machine learning techniques to diverse astrophysical data. Their current work focuses on developing neural network architectures for multi-wavelength analysis."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "EB37141A-E657-44A4-9F13-5B6DF6636D1D"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-04-17-daniel-sieta.toml b/_data/talks/2026-04-17-daniel-sieta.toml
new file mode 100644
index 0000000..9e82b3b
--- /dev/null
+++ b/_data/talks/2026-04-17-daniel-sieta.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Multimodal Data Augmentation for Data-Efficient Robot Manipulation"
+date = 2026-04-17
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = ""
+recording = ""
+canceled = false
+abstract = "Despite recent advances, learning-based robot manipulation systems often require large demonstration datasets and degrade in cluttered or deformable environments. This talk presents diffusion-based multimodal data augmentation methods that synthesize consistent observations and action labels. By augmenting limited demonstrations, these approaches substantially reduce data requirements and enable robust manipulation in complex, real-world settings."
+
+[[speakers]]
+name = "Daniel Sieta"
+affiliation = "USC"
+website = ""
+photo = ""
+bio = "Daniel Seita is an Assistant Professor in the Computer Science department at the University of Southern California and the director of the Sensing, Learning, and Understanding for Robotic Manipulation (SLURM) Lab. His research interests are in computer vision, machine learning, and foundation models for robot manipulation, focusing on improving performance in visually and geometrically challenging settings. Daniel was a postdoc at Carnegie Mellon University's Robotics Institute and holds a PhD in computer science from the University of California, Berkeley. Daniel has been honored with the AAAI 2026 New Faculty Highlights program. He presents his work at premier robotics conferences such as ICRA, IROS, RSS, and CoRL."
+
+[meta]
+source = "google-calendar"
+calendar_uid = "E962F911-1AB5-4C37-A9D0-C596E52EC06B"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-08-28-warren-pettine.toml b/_data/talks/2026-08-28-warren-pettine.toml
new file mode 100644
index 0000000..63705dd
--- /dev/null
+++ b/_data/talks/2026-08-28-warren-pettine.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "About Start-UP MTN: which involves Comptutational Neuroscience and AI"
+date = 2026-08-28
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 2250"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "Warren Pettine"
+affiliation = "UU Psychiatry & MTN"
+website = "https://medicine.utah.edu/faculty/warren-pettine"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "D2FDD8AC-86E1-4961-A286-9596FA9D0FAE"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/2026-09-04-george-vega-yon.toml b/_data/talks/2026-09-04-george-vega-yon.toml
new file mode 100644
index 0000000..e7e1cda
--- /dev/null
+++ b/_data/talks/2026-09-04-george-vega-yon.toml
@@ -0,0 +1,27 @@
+# Imported from the seminar Google Calendar -- please review and complete.
+
+[talk]
+title = "Data Science of Tracking Measles in Utah"
+date = 2026-09-04
+start_time = "13:30"
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB 2250"
+zoom = ""
+slides = ""
+recording = ""
+canceled = false
+abstract = ""
+
+[[speakers]]
+name = "George Vega Yon"
+affiliation = "UU Epidemiology"
+website = "https://medicine.utah.edu/faculty/george-g-vega-yon"
+photo = ""
+bio = ""
+
+[meta]
+source = "google-calendar"
+calendar_uid = "C76F38E7-9637-4FBE-A45E-F1673A54287A"
+imported_on = 2026-08-18
+needs_review = true
diff --git a/_data/talks/_TEMPLATE.toml b/_data/talks/_TEMPLATE.toml
new file mode 100644
index 0000000..229fb30
--- /dev/null
+++ b/_data/talks/_TEMPLATE.toml
@@ -0,0 +1,40 @@
+# Copy this file to _data/talks/YYYY-MM-DD-speaker-name.toml and fill it in.
+# Files whose name starts with "_" are ignored by the generator.
+#
+# After editing, run `make talks` to (re)generate the page under _talks/,
+# and commit both the TOML file and the generated markdown.
+
+[talk]
+title = "Title of the talk"
+date = 2026-09-04 # required, YYYY-MM-DD (Mountain Time)
+start_time = "13:30" # 24h clock; defaults to 13:30 if omitted
+end_time = "14:30"
+series = "Data Science & AI Lecture Series"
+location = "WEB L112"
+zoom = "https://utah.zoom.us/j/85983626630"
+slides = "" # link to slides, once available
+recording = "" # link to the YouTube recording, once available
+paper = "" # optional link to a related paper
+tags = [] # optional, e.g. ["machine learning", "visualization"]
+canceled = false
+# slug = "custom-url-slug" # optional; defaults to this file's name
+abstract = """
+The abstract, in markdown. Triple-quoted strings can span several
+paragraphs.
+"""
+
+# Repeat the [[speakers]] block for talks with more than one speaker.
+[[speakers]]
+name = "Speaker Name"
+affiliation = "Department, University"
+role = "" # e.g. "Assistant Professor"
+website = "https://example.edu/~speaker"
+photo = "" # /assets/img/talk_photos/... or an external URL
+email = ""
+bio = """
+A short bio, in markdown.
+"""
+
+# Optional bookkeeping; ignored when building the page.
+[meta]
+source = "manual"
diff --git a/_includes/next_talks.html b/_includes/next_talks.html
index 696a975..895458e 100644
--- a/_includes/next_talks.html
+++ b/_includes/next_talks.html
@@ -1,35 +1,46 @@
-
+{%- comment -%}
+Renders the next upcoming talk(s) from the `talks` collection (built from the
+TOML records in `_data/talks/`). Pass `include.count` to show more than one.
+{%- endcomment -%}
-
-
-
-
-
+{%- assign count = include.count | default: 1 -%}
+{%- assign now = site.time | date: "%s" | plus: 0 -%}
+{%- assign shown = 0 -%}
+{%- assign talks = site.talks | sort: "date" -%}
-
+{% if shown == 0 %}
+
+ {% endif %}
+ {%- comment -%}
+ Zoom links are only useful (and only worth publishing) while a talk is
+ still ahead of us; past talks keep the link in their TOML record.
+ {%- endcomment -%}
+ {% if page.zoom and status == "upcoming" %}
+
+
diff --git a/_talks/2020-01-09-chris-musco.md b/_talks/2020-01-09-chris-musco.md
new file mode 100644
index 0000000..26fcf84
--- /dev/null
+++ b/_talks/2020-01-09-chris-musco.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Randomized FunctionalAnalysis"
+date: "2020-01-09 12:15:00 -0700"
+permalink: "/talks/2020-01-09-chris-musco/"
+slug: "2020-01-09-chris-musco"
+start_time: "12:15 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Chris Musco"
+ affiliation: "NYU"
+ website: "https://www.chrismusco.com"
+speaker_names: "Chris Musco"
+source_file: "_data/talks/2020-01-09-chris-musco.toml"
+generated: true
+---
+
+
+
+Sketching and subsampling are central algorithmic tools in scaling statisticalmethods to very large datasets. These techniques seek to quickly compress datadown to a compact set of informative features or examples, which can then beprocessed in place of the original data, at much lower computational cost. Thecentral question of this talk is what sketching methods can teach us abouteffective machine learning and data analysis in the small data regime. Inapplications where high quality data examples remain a rare luxury, can ourknowledge of data sketching guide more efficient initial data collection?
+
+We study this problem by focusing specifically on techniques for large matrixcomputations. In the field of randomized numerical linear algebra, importancesampling has emerged as an important tool for dataset compression. Statisticalleverage scores and related measures are used to judge the importance of rowsor columns in a matrix, which are then non-uniformly subsampled, leading tofaster algorithms for regression, low-rank approximation, kernel methods, andmany other data problems.
+
+I will introduce a simple generalization of leverage score sampling to infinitedimensional linear operators and show the potential of this generalization indeveloping sample efficient algorithms for small data applications.Specifically, I will survey a number of recent results on robust polynomialcurve fitting, bandlimited function interpolation, off-grid sparse Fouriertransforms, and sample efficient covariance estimation. I will illustrateconnections between these new results and classical tools in approximationtheory and signal processing, and will discuss several open researchdirections.
diff --git a/_talks/2020-01-16-alexander-lex.md b/_talks/2020-01-16-alexander-lex.md
new file mode 100644
index 0000000..ea8411c
--- /dev/null
+++ b/_talks/2020-01-16-alexander-lex.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Literate Visualization: Making Visual Analysis Sessions Reproducible and Reusable"
+date: "2020-01-16 12:15:00 -0700"
+permalink: "/talks/2020-01-16-alexander-lex/"
+slug: "2020-01-16-alexander-lex"
+start_time: "12:15 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Alexander Lex"
+ website: "https://vdl.sci.utah.edu/team/lex/"
+speaker_names: "Alexander Lex"
+source_file: "_data/talks/2020-01-16-alexander-lex.toml"
+generated: true
+---
+
+
+
+Interactive visualization is an important part of the data science process. It enables analysts to directly interact with the data, exploring it with minimal effort. Unlike code, however, an interactive visualization session is ephemeral and can’t be easily shared, revisited, or reused. Computational notebooks, such as Jupyter Notebooks, R Markdown, or Observable are a perfect match for many data science applications. They are also the most popular embodiment of Knuth’s “Literate Programming”, where the logic of a program is explained in natural language, figures, and equations. In this talk, I will sketch approaches to “Literate Visualization”. I will show how we can leverage provenance data of an analysis session to create well-documented and annotated visualization stories that enable reproducibility and sharing. I will also introduce early work on semi-automatically inferring mid-level analysis goals, which allows us to understand the analysis process at a higher level. Understanding analysis goals enables us to speed up interactions and even re-used visual analysis processes.
diff --git a/_talks/2020-01-23-harish-maringanti.md b/_talks/2020-01-23-harish-maringanti.md
new file mode 100644
index 0000000..de32c7a
--- /dev/null
+++ b/_talks/2020-01-23-harish-maringanti.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Data Science projects in Marriott Library"
+date: "2020-01-23 12:15:00 -0700"
+permalink: "/talks/2020-01-23-harish-maringanti/"
+slug: "2020-01-23-harish-maringanti"
+start_time: "12:15 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Harish Maringanti"
+ affiliation: "Associate Dean for IT & Digital Library Services, Marriott Library"
+ website: "https://collectionsasdata.github.io/"
+speaker_names: "Harish Maringanti"
+source_file: "_data/talks/2020-01-23-harish-maringanti.toml"
+generated: true
+---
+
+
+
+At research intensive universities, libraries have traditionally supported data science activities in various ways including acquiring datasets that researchers need, hosting workshops and training sessions on data science tools, and offering data support tools for creation of persistent identifiers(dois), etc. At Marriott Library, in addition to supporting data science programs on campus, we have embarked on a suite of data science projects to add value to our culturally-rich collections. Our efforts are focused on enriching our collection data, and making this collection data available for computational use (collections as data [1]) so that developers, scientists, and digital humanists can programmatically interact with the data in myriad ways and undertake projects related to data mining & text analysis, advanced visualizations, and geospatial analysis. In this presentation, I will talk about two specific projects - Utah Digital Newspapers [2] and machine learning meets archives [3] - to highlight these efforts.
+
+Utah Digital Newspapers (UDN): Marriott Library was an early pioneer in digitizing newspapers and making the content available to historians, researchers, and lifelong learners. UDN program has been operating since 2002 and is recognized as one of the leaders in newspaper digitization in the United States. We have continued to partner with universities, colleges, state agencies, county and city libraries, and other agencies to digitize, deliver, and archive historical newspaper collections; As of 2019, UDN has well over 22.5 million newspaper articles and 3.5 million pages in the repository platform. In this presentation, we will talk about the importance of looking at collections as data, our API work with UDN and demonstrate the usefulness of this approach with specific examples.
+
+Machine learning meets archives: Metadata is the bedrock of library archives and Digital Library systems, as it helps in users discovering the unique content in various collections housed in digital libraries. But creating metadata is a time-intensive process. We are working with machine learning algorithms to generate descriptive metadata for digital images. I will share the results of our work, and also lessons learned from working with digital library data.
diff --git a/_talks/2020-01-30-qingyao-ai.md b/_talks/2020-01-30-qingyao-ai.md
new file mode 100644
index 0000000..7cf2bce
--- /dev/null
+++ b/_talks/2020-01-30-qingyao-ai.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Unbiased Learning to Rank: Theory and Practice"
+date: "2020-01-30 12:15:00 -0700"
+permalink: "/talks/2020-01-30-qingyao-ai/"
+slug: "2020-01-30-qingyao-ai"
+start_time: "12:15 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Qingyao Ai"
+ affiliation: "Utah SoC"
+ website: "http://ir.aiqingyao.org/"
+speaker_names: "Qingyao Ai"
+source_file: "_data/talks/2020-01-30-qingyao-ai.toml"
+generated: true
+---
+
+
+
+Implicit feedback (e.g., user clicks) is an important source of data for modern search engines. While heavily biased, it is cheap to collect and particularly useful for user-centric retrieval applications such as search ranking. Therefore, a learning-to-rank algorithm that can effectively learn from implicit user feedback without affected by its inherent biases could fundamentally change the design of ranking systems and significantly improve the quality of modern search engines. To develop an unbiased learning-to-rank system with biased feedback, previous studies have focused on constructing probabilistic graphical models (e.g., click models) with user behavior hypothesis to extract and train ranking systems with unbiased relevance signals. Recently, a novel counterfactual learning framework that estimates and adopts examination propensity for unbiased learning to rank has attracted much attention. In this talk, we aim to provide an overview of the fundamental mechanism for unbiased learning to rank. We describe the theory behind existing frameworks, and give instructions on how to conduct unbiased learning to rank in practice.
diff --git a/_talks/2020-02-06-gail-zasowski.md b/_talks/2020-02-06-gail-zasowski.md
new file mode 100644
index 0000000..ec1ee59
--- /dev/null
+++ b/_talks/2020-02-06-gail-zasowski.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Big Data, Big Universe: Data-Driven Discoveries in Astrophysics"
+date: "2020-02-06 12:15:00 -0700"
+permalink: "/talks/2020-02-06-gail-zasowski/"
+slug: "2020-02-06-gail-zasowski"
+start_time: "12:15 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Gail Zasowski"
+ affiliation: "Utah Physics & Astronomy"
+ website: "http://www.physics.utah.edu/~zasowski/"
+speaker_names: "Gail Zasowski"
+source_file: "_data/talks/2020-02-06-gail-zasowski.toml"
+generated: true
+---
+
+
+
+The stars in the night sky have inspired questions about our place in the Universe throughout history. The development of telescopes showed us that the stars visible to the naked eye are but a tiny fraction of their vast numbers within our own Galaxy, and revealed energy signatures invisible to the human senses. We now know that there are billions of stars in our galaxy, billions of galaxies in our Universe, and nearly 14 billion years of cosmic evolution that have led to where and what we are today. As the volume of astronomical data grows at an ever quickening rate, new discoveries increasingly come from careful mining and analysis of existing data, often used in unforeseen ways. This talk will describe some of the major unanswered questions in astrophysics, and how new data-driven analysis techniques are uncovering new insights into solving them.
diff --git a/_talks/2020-02-20-bei-wang.md b/_talks/2020-02-20-bei-wang.md
new file mode 100644
index 0000000..586d3a4
--- /dev/null
+++ b/_talks/2020-02-20-bei-wang.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "TopoAct: Exploring the Shape of Activations in Deep Learning"
+date: "2020-02-20 12:15:00 -0700"
+permalink: "/talks/2020-02-20-bei-wang/"
+slug: "2020-02-20-bei-wang"
+start_time: "12:15 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Bei Wang"
+ affiliation: "Utah SoC, SCI"
+ website: "http://www.sci.utah.edu/~beiwang/"
+speaker_names: "Bei Wang"
+source_file: "_data/talks/2020-02-20-bei-wang.toml"
+generated: true
+---
+
+
+
+Deep neural networks such as GoogLeNet and ResNet have achieved superhuman performance in tasks like image classification. To understand how such superior performance is achieved, we can probe a trained deep neural network by studying neuron activations, that is, combinations of neuron firings, at any layer of the network in response to a particular input. With a large set of input images, we aim to obtain a global view of what neurons detect by studying their activations. We ask the following questions: What is the shape of the space of activations? That is, what is the organizational principle behind neuron activations, and how are the activations related within a layer and across layers? Applying tools from topological data analysis, we present TopoAct, a visual exploration system used to study topological summaries of activation vectors for a single layer as well as the evolution of such summaries across multiple layers. We present visual exploration scenarios using TopoAct that provide valuable insights towards learned representations of an image classifier.
diff --git a/_talks/2020-02-27-taylor-sparks.md b/_talks/2020-02-27-taylor-sparks.md
new file mode 100644
index 0000000..04a39a7
--- /dev/null
+++ b/_talks/2020-02-27-taylor-sparks.md
@@ -0,0 +1,28 @@
+---
+layout: "talk"
+title: "New Algorithms, Descriptors, and Machine Learning Techniques Tailored to the Challenges of Materials Informatics"
+date: "2020-02-27 12:15:00 -0700"
+permalink: "/talks/2020-02-27-taylor-sparks/"
+slug: "2020-02-27-taylor-sparks"
+start_time: "12:15 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Taylor Sparks"
+ affiliation: "Utah Materials Science & Engineering"
+ website: "https://my.eng.utah.edu/~sparks/"
+ bio: "Dr. Sparks is an Associate Professor and Associate Chair of the Materials Science and Engineering Department at the University of Utah. He is originally from Utah and an alumni of the department he now teaches in. Before graduate school he worked at Ceramatec Inc. He did his MS in Materials at UCSB and his PhD in Applied Physics at Harvard University in David Clarke’s laboratory and then did a postdoc with Ram Seshadri in the Materials Research Laboratory at UCSB. His current research centers on the discovery, synthesis, characterization, and properties of new materials for energy applications. He is a pioneer in the emerging field of materials informatics whereby big data, data mining, and machine learning are leveraged to solve challenges in materials science. He also hosts a podcast entitled “Materialism” where he discusses the past, present, and future of Materials Science."
+speaker_names: "Taylor Sparks"
+source_file: "_data/talks/2020-02-27-taylor-sparks.toml"
+generated: true
+---
+
+
+
+New materials are required to address many of the energy, environmental, and technological needs of the present and future. Materials Informatics, or the application of data science techniques to solve materials research challenges, is expected to play a key role in materials development and discovery given the infinite palette available for new materials. Interestingly, the requirements and tasks of Materials Informatics do not always overlap with general machine learning. Therefore, adopting existing data science tools including visualization, algorithms, featurization schemas etc may not provide the ideal outcomes for the specific needs of Materials Informatics.
+
+In this talk I will focus on some of our recent work to bring tailored data science approaches to actual materials research problems. Specifically, I will introduce how we have created an attention-based neural network architecture for the prediction of materials properties. We show that this novel algorithm outperforms other methods in the absence of chemical information, even when the statistical and ensemble learning techniques are given domain-specific chemical knowledge about the materials.
+
+Dr. Sparks is an Associate Professor and Associate Chair of the Materials Science and Engineering Department at the University of Utah. He is originally from Utah and an alumni of the department he now teaches in. Before graduate school he worked at Ceramatec Inc. He did his MS in Materials at UCSB and his PhD in Applied Physics at Harvard University in David Clarke’s laboratory and then did a postdoc with Ram Seshadri in the Materials Research Laboratory at UCSB. His current research centers on the discovery, synthesis, characterization, and properties of new materials for energy applications. He is a pioneer in the emerging field of materials informatics whereby big data, data mining, and machine learning are leveraged to solve challenges in materials science. He also hosts a podcast entitled “Materialism” where he discusses the past, present, and future of Materials Science.
diff --git a/_talks/2020-03-05-john-horel.md b/_talks/2020-03-05-john-horel.md
new file mode 100644
index 0000000..ec46d29
--- /dev/null
+++ b/_talks/2020-03-05-john-horel.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "The Promise and Perils of Big Data in the Cloud - Examples from the Atmospheric Sciences"
+date: "2020-03-05 12:15:00 -0700"
+permalink: "/talks/2020-03-05-john-horel/"
+slug: "2020-03-05-john-horel"
+start_time: "12:15 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "John Horel"
+ affiliation: "Professor, Chair, Department of Atmospheric Sciences"
+speaker_names: "John Horel"
+source_file: "_data/talks/2020-03-05-john-horel.toml"
+generated: true
+---
+
+
+
+From the inception of numerical weather prediction in the 1950’s, atmospheric scientists have stretched the envelope on the hardware and procedures available to write, store, and use data on mass storage systems. The opportunities now to rely on cloud resources to process, access, and disseminate environmental data offer improved capabilities for data science applications in the atmospheric sciences but also introduce complexities for university researchers.
+
+The Big Data Project of the National Oceanographic and Atmospheric Administration is assessing the potential benefits of storing in the cloud observations and weather and climate model output that are generating petabytes of data daily. Retrieving, archiving, analyzing, and disseminating only a fraction of this environmental information has required us to move beyond computational approaches traditionally used within the atmospheric science community. For example, hundreds of users rely on a 140+ Tbyte archive we maintain of High Resolution Rapid Refresh (HRRR) model output on the Pando system of the University’s Center for High Performance Computing. Computing resources available nationwide as part of the Open Science Grid- a high-throughput computing resource- have been used to analyze wildland fire events. Data analytical methods are being explored that rely on Zarr data compression to provide classes and functions for working with N-dimensional arrays.
diff --git a/_talks/2020-08-28-vivek-gupta.md b/_talks/2020-08-28-vivek-gupta.md
new file mode 100644
index 0000000..2a41188
--- /dev/null
+++ b/_talks/2020-08-28-vivek-gupta.md
@@ -0,0 +1,28 @@
+---
+layout: "talk"
+title: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09 (https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+date: "2020-08-28 11:50:00 -0600"
+permalink: "/talks/2020-08-28-vivek-gupta/"
+slug: "2020-08-28-vivek-gupta"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+recording: "https://www.youtube.com/redirect?q=https%3A%2F%2Fvgupta123.github.io%2F&v=YhfU1BON8EI&event=video_description&redir_token=QUFFLUhqbFVhUUxOS2RDSHBPYWJPMjNqaFpoalpzY2E3d3xBQ3Jtc0tuWlZRQ0tfSTdPNkxsTzFuWDN2Y3lvdXdwRXJrOTJ6bGI3NXhrc2x4QVYxalNTTUVXWE10V3VncklsZ2Q1RnltN1VaWkpsek9OY0EtdWwtQWZNVHR0Q2k4Rng3eVE4N3FWYTk5dUtETV9aZXRSaGlEcw%3D%3D"
+canceled: false
+speakers:
+ - name: "Vivek Gupta"
+ affiliation: "UoU"
+ website: "https://vgupta123.github.io/"
+ bio: "Vivek is a Ph.D. student at the School of Computing, University of Utah. Previously, he was working as a Research Fellow in Microsoft Research Lab, India, in Machine Learning and Natural Language Processing group. He graduated as a dual degree student in the Department of Computer Science and Engineering at IIT Kanpur in 2016. He is broadly interested in research in the field of Machine Learning and Natural Language Processing. To know more about his current research interest, you can visit https://vgupta123.github.io/ (https://www.youtube.com/redirect?q=https%3A%2F%2Fvgupta123.github.io%2F&v=YhfU1BON8EI&event=video_description&redir_token=QUFFLUhqbFVhUUxOS2RDSHBPYWJPMjNqaFpoalpzY2E3d3xBQ3Jtc0tuWlZRQ0tfSTdPNkxsTzFuWDN2Y3lvdXdwRXJrOTJ6bGI3NXhrc2x4QVYxalNTTUVXWE10V3VncklsZ2Q1RnltN1VaWkpsek9OY0EtdWwtQWZNVHR0Q2k4Rng3eVE4N3FWYTk5dUtETV9aZXRSaGlEcw%3D%3D)"
+speaker_names: "Vivek Gupta"
+source_file: "_data/talks/2020-08-28-vivek-gupta.toml"
+generated: true
+---
+
+
+
+Vivek Gupta (Utah School of Computing)
+(https://vgupta123.github.io/ (https://www.youtube.com/redirect?q=https%3A%2F%2Fvgupta123.github.io%2F&v=YhfU1BON8EI&event=video_description&redir_token=QUFFLUhqbFVhUUxOS2RDSHBPYWJPMjNqaFpoalpzY2E3d3xBQ3Jtc0tuWlZRQ0tfSTdPNkxsTzFuWDN2Y3lvdXdwRXJrOTJ6bGI3NXhrc2x4QVYxalNTTUVXWE10V3VncklsZ2Q1RnltN1VaWkpsek9OY0EtdWwtQWZNVHR0Q2k4Rng3eVE4N3FWYTk5dUtETV9aZXRSaGlEcw%3D%3D))
+
+Experience of the everyday language indicates the use of complicated reasonings both for people and the AI systems. Natural Language Inference (NLI) is the process of reasoning about inferential relationships, meaning to establish whether a hypothesis is a true (entailment), false (contradiction), or undetermined (neutral) given a premise. Previous works have generated inference corpora, such as the SNLI and the MNLI, which comprise only unstructured representations of text in the form of sentences in which relationships between words are explicitly expressed, and often need information extraction. However, text can also occur universally in other structured forms like tables, graphs, and databases.Building upon previous work on large-scale datasets for inference, we introduce a new dataset called INFOTABS, comprising of human-written textual hypotheses based on premises that are tables extracted from Wikipedia info-boxes. Our analysis shows that the semi-structured, multi-domain, and heterogeneous nature of the premises admits complex, multi-faceted reasoning. Experiments reveal that, while human annotators agree on the relationships between a table-hypothesis pair, several standard modeling strategies are unsuccessful at the task, suggesting that reasoning about tables can pose a new modeling challenge. For more details on InfoTabS visit http://infotabs.github.io (http://infotabs.github.io/)
diff --git a/_talks/2020-09-04-dheeraj-mekala.md b/_talks/2020-09-04-dheeraj-mekala.md
new file mode 100644
index 0000000..644774d
--- /dev/null
+++ b/_talks/2020-09-04-dheeraj-mekala.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Contextualized Weak Supervision for Text Classification"
+date: "2020-09-04 11:50:00 -0600"
+permalink: "/talks/2020-09-04-dheeraj-mekala/"
+slug: "2020-09-04-dheeraj-mekala"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Dheeraj Mekala"
+ affiliation: "UCSD"
+ bio: "I am a Master’s student in the Computer Science department at the University of California, San Diego working with Prof. Jingbo Shang. I am broadly interested in Natural Language Processing, Text Mining, and Graph Mining. My current research revolves around developing principled data-driven approaches with minimal human effort. Specifically, there are huge amounts of unlabeled data available on the Internet and I try to leverage them to minimize the human effort. I completed my Bachelor of Technology in Computer Science And Engineering from the Indian Institute of Technology, Kanpur in 2017, where I worked with Prof. Harish Karnick, Prof. Purushottam Kar on Hierarchical Classification and Text Document Representation. I worked as a Data Scientist and Product Engineer at Sprinklr for 2 years and I interned at Microsoft India in the summer of 2016."
+speaker_names: "Dheeraj Mekala"
+source_file: "_data/talks/2020-09-04-dheeraj-mekala.toml"
+generated: true
+---
+
+
+
+Weakly supervised text classification based on a few user-provided seed words has recently attracted much attention from researchers. Existing methods mainly generate pseudo-labels in a context-free manner (e.g., string matching), therefore, the ambiguous, context-dependent nature of human language has been long overlooked. In this paper, we propose a novel framework ConWea, providing contextualized weak supervision for text classification. Specifically, we leverage contextualized representations of word occurrences and seed word information to automatically differentiate multiple interpretations of the same word, and thus create a contextualized corpus. This contextualized corpus is further utilized to train the classifier and expand seed words in an iterative manner. This process not only adds new contextualized, highly label-indicative keywords but also disambiguates initial seed words, making our weak supervision fully contextualized. Extensive experiments and case studies on real-world datasets demonstrate the necessity and significant advantages of using contextualized weak supervision, especially when the class labels are fine-grained.
diff --git a/_talks/2020-09-11-parthe-pandit.md b/_talks/2020-09-11-parthe-pandit.md
new file mode 100644
index 0000000..302a958
--- /dev/null
+++ b/_talks/2020-09-11-parthe-pandit.md
@@ -0,0 +1,29 @@
+---
+layout: "talk"
+title: "Characterizing the asymptotic performance of inverse problems over Deep Networks"
+date: "2020-09-11 11:50:00 -0600"
+permalink: "/talks/2020-09-11-parthe-pandit/"
+slug: "2020-09-11-parthe-pandit"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Parthe Pandit"
+ affiliation: "UCLA"
+ website: "https://parthe.github.io/"
+ bio: "Parthe is a 5th year Ph.D. candidate at UCLA in ECE and an MS candidate in Statistics, where he is advised by Allie K. Fletcher, Sundeep Rangan, and Arash A. Amini. He is currently a research intern with Amazon AWS and Amazon Search, working on problems in conditional text generation. He is interested in analyzing optimization problems arising out of Machine Learning, Statistics, and Information Theory. His past research includes theoretical results in Network Economics, Mechanism Design, and Approximating NP-hard problems in graph theory.\n\nParthe is the winner of the 2019 Jack K. Wolf Best Paper award, the Gurukrupa Foundation Fellowship, and scholarships from the J. N. Tata Endowment, and K. C. Mahindra foundation. In 2015, he received a B.Tech. and M.Tech. in Electrical Engineering at IIT Bombay, where he worked on Speech and Music Processing.\n"
+speaker_names: "Parthe Pandit"
+source_file: "_data/talks/2020-09-11-parthe-pandit.toml"
+generated: true
+---
+
+
+
+At the heart of Machine Learning lies the question of generalizability of learned models over previously unseen data. While over-parameterized models based on neural networks are now ubiquitous in machine learning applications, our understanding of their generalization capabilities remains incomplete. This task is made harder by the non-convexity of the underlying estimation/learning problems.
+
+The solutions to some of these estimation problems can be analyzed using a class of algorithms called Approximate Message Passing, even in the presence of the non-convexity. The dynamics of this algorithm follow a simplified macroscopic description often called the State Evolution. This analytical tool allows us to provide some insights into two broad classes of problems related to Neural Networks:
+
+1. Generalization error in 1 and 2-layer Networks
+2. Image Reconstruction error with Deep Image Priors
diff --git a/_talks/2020-09-18-vishnu-lokhande.md b/_talks/2020-09-18-vishnu-lokhande.md
new file mode 100644
index 0000000..7f65b02
--- /dev/null
+++ b/_talks/2020-09-18-vishnu-lokhande.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Optimization methods for imposing Fairness in Computer Vision Models"
+date: "2020-09-18 11:50:00 -0600"
+permalink: "/talks/2020-09-18-vishnu-lokhande/"
+slug: "2020-09-18-vishnu-lokhande"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Vishnu Lokhande"
+ affiliation: "University of Wisconsin-Madison"
+ website: "https://lokhande-vishnu.github.io"
+ bio: "Vishnu Lokhande is a fourth year PhD student in Computer Sciences at the University of Wisconsin-Madison. He is currently completing his research internship at Microsoft Research in the Interactive Media Group. His research interests include Algorithmic Fairness, Semi-Supervised Learning, Constrained and Stochastic Optimization problems. Prior to his graduate studies, he received his bachelor’s in Electrical Engineering at the Indian Institute of Technology Kanpur."
+speaker_names: "Vishnu Lokhande"
+source_file: "_data/talks/2020-09-18-vishnu-lokhande.toml"
+generated: true
+---
+
+
+
+In this talk, we will study a mechanism to impose fairness in computer vision models concurrently while training the model and informed by standard fairness measures. While existing fairness based approaches in vision have largely relied on training adversarial modules together with the primary classification/regression task, in an effort to remove the influence of the protected attribute or variable, we will discuss how ideas based on well-known optimization concepts can provide a simpler alternative. In our proposed scheme, imposing fairness just requires specifying the protected attribute and utilizing our optimization routine. We will discuss experiments, that are interpretable, demonstrating that several fairness measures from the literature can be reliably imposed on standard vision tasks. We will also discuss technical analysis on the convergence guarantees of the said optimization routine.
diff --git a/_talks/2020-10-02-ellen-riloff.md b/_talks/2020-10-02-ellen-riloff.md
new file mode 100644
index 0000000..c2271b5
--- /dev/null
+++ b/_talks/2020-10-02-ellen-riloff.md
@@ -0,0 +1,40 @@
+---
+layout: "talk"
+title: "Identifying Affective Events and the Reasons for their Polarity"
+date: "2020-10-02 11:50:00 -0600"
+permalink: "/talks/2020-10-02-ellen-riloff/"
+slug: "2020-10-02-ellen-riloff"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Ellen Riloff"
+ affiliation: "Utah Computer Science"
+ website: "http://www.cs.utah.edu/~riloff/"
+ bio: "Ellen Riloff is a Professor in the School of Computing at the\nUniversity of Utah. Her primary research area is natural language\nprocessing, with an emphasis on information extraction, affective text\nanalysis, semantic class induction, and bootstrapping methods that\nlearn from unannotated texts. Prof. Riloff has served as the General\nChair for the EMNLP 2018 conference, Program Co-Chair for the NAACL\nHLT 2012 and CoNLL 2004 conferences, on the NAACL Executive Board for\n2004-2005 and 2017-2018, the Computational Linguistics Editorial\nBoard, and the Transactions of the Association for Computational\nLinguistics (TACL) Editorial Board. In 2018, Prof. Riloff was named a\nFellow of the Association for Computational Linguistics (ACL).\n"
+speaker_names: "Ellen Riloff"
+source_file: "_data/talks/2020-10-02-ellen-riloff.toml"
+generated: true
+---
+
+
+
+Recognizing affective states is essential for narrative text
+understanding and for applications such as conversational dialogue,
+summarization, and sarcasm recognition. Many tools have been developed
+to recognize explicit expressions of sentiment, but affective states
+can also be inferred from events. This talk will focus on "affective
+events", which are generally desirable or undesirable experiences that
+implicitly suggest an affective state for the experiencer. For
+example, buying a home is usually desirable and associated with a
+positive affective state, but being laid off is undesirable and
+associated with a negative state. First, we will describe a weakly
+supervised learning method to induce affective events from a text
+corpus by optimizing for semantic consistency. Second, we aim to
+characterize affective events based on Human Needs Categories, which
+often explain people's motivations, goals, and desires. We will
+present a co-training model for Human Needs categorization that uses
+an event expression classifier and an event context classifier to
+learn from both labeled and unlabeled texts.
diff --git a/_talks/2020-10-09-swaroop-mishra.md b/_talks/2020-10-09-swaroop-mishra.md
new file mode 100644
index 0000000..df93189
--- /dev/null
+++ b/_talks/2020-10-09-swaroop-mishra.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "DQI: Measuring Data Quality in NLP"
+date: "2020-10-09 11:50:00 -0600"
+permalink: "/talks/2020-10-09-swaroop-mishra/"
+slug: "2020-10-09-swaroop-mishra"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Swaroop Mishra"
+ affiliation: "Arizona State University"
+ website: "https://scholar.google.com/citations?user=-7LK2SwAAAAJ&hl=en"
+ bio: "Swaroop is currently a 2nd year PhD student at ASU. Prior to this, he received an M.S. degree from IIT Kanpur in 2016, post which he worked as a software engineer at MathWorks for 2 years, and as a Technical Consultant in the Ministry of Electronics and Information Technology, Govt. of India for a year."
+speaker_names: "Swaroop Mishra"
+source_file: "_data/talks/2020-10-09-swaroop-mishra.toml"
+generated: true
+---
+
+
+
+Neural language models have achieved human-level performance across several NLP datasets. However, recent studies have shown that these models are not truly learning the desired task; rather, their high performance is attributed to overfitting using spurious biases, which suggests that the capabilities of AI systems have been over-estimated. We introduce a generic formula for Data Quality Index (DQI) to help dataset creators create datasets with minimal unwanted biases. We propose a new data creation paradigm using DQI to create higher quality data. The data creation paradigm consists of several data visualizations to help data creators (i) understand the quality of data and (ii) visualize the impact of the created data instance on the overall quality. It also has a couple of automation methods to (i) assist data creators and (ii) make the model more robust to adversarial attacks. We use DQI along with these automation methods to renovate biased examples in SNLI. We show that models trained on the renovated SNLI dataset generalize better to out of distribution tasks. Renovation results in reduced model performance, exposing a large gap with respect to human performance. DQI systematically helps in creating harder benchmarks using active learning. Our work takes the process of dynamic dataset creation forward, wherein datasets evolve together with the evolving state of the art, therefore serving as a means of benchmarking the true progress of AI. Finally, we also show that DQI helps in pruning a dataset without compromising IID and OOD performance significantly.
diff --git a/_talks/2020-10-16-varun-gangal.md b/_talks/2020-10-16-varun-gangal.md
new file mode 100644
index 0000000..6984fbc
--- /dev/null
+++ b/_talks/2020-10-16-varun-gangal.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Examining Extra Sentential Abilities of Contextual Embeddings"
+date: "2020-10-16 11:50:00 -0600"
+permalink: "/talks/2020-10-16-varun-gangal/"
+slug: "2020-10-16-varun-gangal"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Varun Gangal"
+ affiliation: "LTI, CMU"
+ website: "https://scholar.google.com/citations?user=rWZq2nQAAAAJ&hl=en"
+ bio: "Varun is a PhD student at CMU LTI, advised by Eduard Hovy. His research is primarily on language generation, with specific interests in style transfer, data-to-text generation and document/story-level generation tasks. He has recently also been exploring probing and data augmentation questions motivated by his primary interests. His research has appeared at ACL, EMNLP and AAAI."
+speaker_names: "Varun Gangal"
+source_file: "_data/talks/2020-10-16-varun-gangal.toml"
+generated: true
+---
+
+
+
+In the first third of our talk, we try to understand what and how much does BERT already know about event arguments (including cross-sentence ones)?. We observe that BERT's attention heads have modest but well above-chance ability to spot event arguments sans any training. Furthermore, we investigate how our methods do for cross-sentence event arguments, proposing a procedure to isolate "best heads" for cross-sentence argument detection separately of those for intra-sentence arguments. In the second third, we take a closer look at the infilling abilities of BERT. We know BERT is good at Word-level Infilling (obviously!) . How good is it though at Sentence-level Infilling a.k.a Cloze ? We introduce a human-created sentence cloze dataset, collected from public school English examinations. Our task requires a model to fill up multiple blanks in a passage from a shared candidate set with distractors designed by English teachers. Our experiments show a significant performance gap between BERT (72%) and humans (87%), encouraging future models to bridge this gap.We conclude our talk by going through a somewhat unrelated, recent foray into data augmentation for finetuning pretrained generators on low resource domains.
diff --git a/_talks/2020-10-23-nancy-wang.md b/_talks/2020-10-23-nancy-wang.md
new file mode 100644
index 0000000..9ff296f
--- /dev/null
+++ b/_talks/2020-10-23-nancy-wang.md
@@ -0,0 +1,29 @@
+---
+layout: "talk"
+title: "Global Table Extractor (GTE): A Framework for Joint Table Identification and Cell Structure Recognition Using Visual Context"
+date: "2020-10-23 11:50:00 -0600"
+permalink: "/talks/2020-10-23-nancy-wang/"
+slug: "2020-10-23-nancy-wang"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Nancy Wang"
+ affiliation: "IBM"
+ website: "https://researcher.watson.ibm.com/researcher/view.php?person=ibm-wangnxr"
+ bio: "Nancy Wang is a Researcher with IBM Research - Almaden who is currently working on applying deep learning and computer vision methods for table extraction and table understanding from documents. She graduated from the University of Washington with her PhD in Computer Science in 2018 in the area of computer vision for computational neuroscience. Her new table extraction work is under review at top-level AI conferences and is in the process of being incorporated into Watson Discovery. She was also one of the presenting tutors for the Table Extraction and Understanding Tutorial at ICDM 2019 and VLDB 2020."
+speaker_names: "Nancy Wang"
+source_file: "_data/talks/2020-10-23-nancy-wang.toml"
+generated: true
+---
+
+
+
+Documents are often the format of choice for knowledge sharing and preservation in business and science, within which are tables that capture most of the critical data. Unfortunately, most documents are stored and distributed as PDF or scanned images, which fail to preserve table formatting.
+Recent vision-based deep learning approaches have been proposed to address this gap, but most still cannot achieve state-of-the-art results.
+
+ We present Global Table Extractor (GTE), a vision-guided systematic framework for joint table detection and cell structured recognition, which could be built on top of any object detection model. With GTE-Table, we invent a new penalty based on the natural cell containment constraint of tables to train our table network aided by cell location predictions. GTE-Cell is a new hierarchical cell detection network that leverages table styles. Further, we design a method to automatically label table and cell structure in existing documents to cheaply create a large corpus of training and test data. We use this to enhance PubTabNet with cell labels and create FinTabNet, real-world and complex scientific and financial datasets with detailed table structure annotations to help train and test structure recognition.
+
+ Our deep learning framework surpasses previous state-of-the-art results on the ICDAR 2013 and ICDAR 2019 table competition test dataset in both table detection and cell structure recognition. Further experiments demonstrate a greater than 45% improvement in cell structure recognition when compared to a vanilla RetinaNet object detection model in our new financial dataset (FinTabNet).
diff --git a/_talks/2020-10-30-alberto-cairo.md b/_talks/2020-10-30-alberto-cairo.md
new file mode 100644
index 0000000..31a7a31
--- /dev/null
+++ b/_talks/2020-10-30-alberto-cairo.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "Data Visualization: How to Make Good Decisions"
+date: "2020-10-30 15:30:00 -0600"
+permalink: "/talks/2020-10-30-alberto-cairo/"
+slug: "2020-10-30-alberto-cairo"
+start_time: "3:30 PM"
+end_time: "4:30 PM"
+series: "Data Science Seminar"
+canceled: false
+speakers:
+ - name: "Alberto Cairo"
+ bio: "Alberto Cairo is a journalist and designer with many years of experience leading graphics and visualization teams in several countries. He is the Knight Chair at the School of Communication of the University of Miami, where he teaches courses on infographics and data visualization. He is also director of the Center for Visualization at UM’s Institute for Data Science and Computing, and a Faculty Fellow at the Abess Center for Ecosystem Science and Policy.In the past decade, Cairo has taught and consulted in nearly thirty countries, working for Microsoft, Google, the U.S. National Guard, and many other companies and institutions. Cairo has also written for The New York Times and Scientific American magazine and he is the author of numerous books, the latest one being 'How Charts Lie: Getting Smarter About Visual Information' (W.W. Norton, 2019)"
+speaker_names: "Alberto Cairo"
+source_file: "_data/talks/2020-10-30-alberto-cairo.toml"
+generated: true
+---
+
+
+
+Data visualization, the display of data through graphs, charts, maps, and diagrams, is a skill in great demand in many disciplines, from the sciences to communication or business analytics. However, visualization is often misunderstood. For instance, it's often taught as the application of a series of strict rules. This talk argues that visualization is more akin to writing: yes, we do need to understand visualization's grammar but, beyond that, visualization design is flexible, and can't be based on rules that are set in stone. Instead, designers need to develop a good decision-making framework based on asking themselves a series of questions.
diff --git a/_talks/2020-10-30-daniel-scharfstein.md b/_talks/2020-10-30-daniel-scharfstein.md
new file mode 100644
index 0000000..ad7cd0b
--- /dev/null
+++ b/_talks/2020-10-30-daniel-scharfstein.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Semiparametrics: A Biostatistician’s Toolbox"
+date: "2020-10-30 11:50:00 -0600"
+permalink: "/talks/2020-10-30-daniel-scharfstein/"
+slug: "2020-10-30-daniel-scharfstein"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Daniel Scharfstein"
+ affiliation: "Utah, Population Health Sciences"
+ website: "http://www.biostat.jhsph.edu/~dscharf/about.html"
+ bio: "Daniel Scharfstein is a Professor of Biostatistics in the Department of Population Health Sciences, at the University of Utah School of Medicine.\nHe joined the U in August 2020 after spending 23 years on the faculty in the Department of Biostatistics at the Johns Hopkins Bloomberg School of Public Health.\n"
+speaker_names: "Daniel Scharfstein"
+source_file: "_data/talks/2020-10-30-daniel-scharfstein.toml"
+generated: true
+---
+
+
+
+In this talk, I will discuss the theory of semiparametrics that I use to estimate causal effects at root-n rates. Estimators of these effects depend on estimators of nuisance parameters that can be estimated at rates slower than root-n; I provide sufficient conditions for these rates. I will seek advice on the machine learning estimation techniques that satisfy these conditions. I will illustrate the theory in the context of estimating the causal contrast of two competing treatments based on data from a comprehensive cohort study in which clinically eligible individuals are first asked to enroll in a randomized trial and, if they refuse, are then asked to enroll in a parallel observational study in which they can choose treatment according to their own preference.
diff --git a/_talks/2020-11-06-bhargavi-paranjape.md b/_talks/2020-11-06-bhargavi-paranjape.md
new file mode 100644
index 0000000..35f9994
--- /dev/null
+++ b/_talks/2020-11-06-bhargavi-paranjape.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09 (https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+date: "2020-11-06 11:50:00 -0700"
+permalink: "/talks/2020-11-06-bhargavi-paranjape/"
+slug: "2020-11-06-bhargavi-paranjape"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Bhargavi Paranjape"
+ affiliation: "University of Washington"
+ website: "https://bhargaviparanjape.github.io"
+ bio: "Bhargavi is a second-year Ph.D. student in the Paul G. Allen School of Computer Science & Engineering, University of Washington, where she is advised by Hannaneh Hajishirzi and Luke Zettlemoyer. Her research interests include interpretability and explainability of neural models, managing model and dataset bias, and pre-training techniques for NLP. She earned an MS in Language Technology from LTI, Carnegie Mellon University, and a B.Tech in Computer Science and Engineering from IIT Kharagpur."
+speaker_names: "Bhargavi Paranjape"
+source_file: "_data/talks/2020-11-06-bhargavi-paranjape.toml"
+generated: true
+---
+
+
+
+Decisions of complex models for language understanding can be explained by limiting the inputs they are provided to a relevant sub-sequence of the original text — a rationale. Models that condition predictions on a concise rationale, while being more interpretable, tend to be less accurate than models that are able to use the entire context. In this paper, we show that it is possible to better manage the trade-off between concise explanations and high task accuracy by optimizing abound on the Information Bottleneck (IB) objective. Our approach jointly learns an explainer that predicts sparse binary masks over input sentences without explicit supervision and an end-task predictor that considers only the residual sentences. Using IB, we derive a learning objective that allows direct control of mask sparsity levels through a tunable sparse prior. Experiments on the ERASER benchmark demonstrate significant gains over previous work for both task performance and agreement with human rationales
diff --git a/_talks/2020-11-13-akanksha-atrey.md b/_talks/2020-11-13-akanksha-atrey.md
new file mode 100644
index 0000000..5c71fa3
--- /dev/null
+++ b/_talks/2020-11-13-akanksha-atrey.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Towards High-Performance Machine Learning on the Edge"
+date: "2020-11-13 11:50:00 -0700"
+permalink: "/talks/2020-11-13-akanksha-atrey/"
+slug: "2020-11-13-akanksha-atrey"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Akanksha Atrey"
+ affiliation: "UMass"
+ website: "https://akanksha-atrey.github.io"
+ bio: "Akanksha Atrey is a Ph.D. student in the College of Information and Computer Sciences at the University of Massachusetts Amherst. She is a member of the Laboratory of Advanced Software Systems, where she is advised by Prof. Prashant Shenoy. Her research interests lie at the intersection of machine learning, edge computing, and privacy with a focus on applications in IoT and mobile computing. Her work seeks to make machine learning more available, usable, and scalable for applications in edge systems. Prior to joining UMass, she was a software engineer at IBM where she worked on the IBM z/OS Mainframe. In 2016, she received her Bachelor of Science degree in Mathematics and Computer Science from the State University of New York at Albany. Among her achievements, she has been named a CRA-W Research Scholar, received an Honorable Mention for the NSF Graduate Research Fellowship Program, and received the Lori A. Clarke Scholarship in Computer Science."
+speaker_names: "Akanksha Atrey"
+source_file: "_data/talks/2020-11-13-akanksha-atrey.toml"
+generated: true
+---
+
+
+
+Modern day distributed technologies, such as mobile systems and the Internet of Things (IoT), enable the global integration of heterogeneous smart devices via wireless networks. A common characteristic across these technologies is their ability to collect and communicate continuously streaming data. The generation of high bandwidth data makes machine learning (ML) and artificial intelligence (AI) appealing for processing, reasoning, and predicting about the environment, but low network latency requirements make offloading intelligence to the cloud undesirable. This raises an important question: how can we design, develop and evaluate ML algorithms that perform beyond predictive accuracy (e.g., generalizable, explainable, privacy-aware, and efficient) while being accessible and scalable in resource-constrained edge environments? In this talk, I will cover two aspects, explainability and privacy, when deploying such ML models in modern distributed technologies. The focus of the talk will be two-fold: (1) counterfactual evaluation of the explanations generated using saliency maps in deep reinforcement learning for applications such as autonomous vehicles, and (2) privacy implications of personalized ML models in context-aware mobility applications. The talk will be concluded with a discussion on where the future of ML lies in evolving distributed technologies.
diff --git a/_talks/2020-11-20-akhil-arora.md b/_talks/2020-11-20-akhil-arora.md
new file mode 100644
index 0000000..335ccbe
--- /dev/null
+++ b/_talks/2020-11-20-akhil-arora.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Low-rank Subspaces for Unsupervised Entity Linking"
+date: "2020-11-20 11:50:00 -0700"
+permalink: "/talks/2020-11-20-akhil-arora/"
+slug: "2020-11-20-akhil-arora"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Akhil Arora"
+ affiliation: "EPFL"
+ bio: "Akhil Arora is a PhD student affiliated with the Data Science Lab (dlab) at EPFL. Prior to this, Akhil spent close to five years in industry working with the research labs of American Express and Xerox. Akhil’s research interests include large scale data management, graph mining, and machine learning. He is a recipient of the prestigious “EDIC Doctoral Fellowship” for the academic year 2018-19, and the “Most Reproducible Paper” award at SIGMOD 2018. He has published his research in prestigious data mining and database conferences, served as a reviewer, and co-organized workshops in these conferences."
+speaker_names: "Akhil Arora"
+source_file: "_data/talks/2020-11-20-akhil-arora.toml"
+generated: true
+---
+
+
+
+Entity linking is an important problem with many applications. Most previous solutions were designed for settings where annotated training data is available, which is, however, not the case in numerous domains. We propose a light-weight and scalable entity linking method, Eigenthemes, that relies solely on the availability of entity names and a referent knowledge base. Eigenthemes exploits the fact that the entities that are truly mentioned in a document (the ``gold entities'') tend to form a semantically dense subset of the set of all candidate entities in the document. Geometrically speaking, when representing entities as vectors via some given embedding, the gold entities tend to lie in a low-rank subspace of the full embedding space. Eigenthemes identifies this subspace using the singular value decomposition and scores candidate entities according to their proximity to the subspace. Extensive experiments on benchmark datasets from a variety of real-world domains showcase the effectiveness of our approach.
diff --git a/_talks/2020-12-04-danish-pruthi.md b/_talks/2020-12-04-danish-pruthi.md
new file mode 100644
index 0000000..dc92d70
--- /dev/null
+++ b/_talks/2020-12-04-danish-pruthi.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "A Tale of Evidence and Explanations"
+date: "2020-12-04 11:50:00 -0700"
+permalink: "/talks/2020-12-04-danish-pruthi/"
+slug: "2020-12-04-danish-pruthi"
+start_time: "11:50 AM"
+end_time: "1:10 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/83503994251?pwd=Wm01bk40UWxva2pSV0dTbFZHbGtLQT09"
+canceled: false
+speakers:
+ - name: "Danish Pruthi"
+ affiliation: "LTI, CMU"
+ website: "https://www.cs.cmu.edu/~ddanish/"
+ bio: "Danish Pruthi is a senior Ph.D. student at School of Computer Science in Carnegie Mellon University. Broadly, his research aims to enable machines to understand and explain natural language phenomena. He completed his bachelors degree in computer science from BITS Pilani, Pilani in 2015. He has also spent time doing research at Google AI, Facebook AI Research, Microsoft Research, and Indian Institute of Science. He is a recipient of the Siebel Scholarship, and CMU Presidential Fellowship."
+speaker_names: "Danish Pruthi"
+source_file: "_data/talks/2020-12-04-danish-pruthi.toml"
+generated: true
+---
+
+
+
+I would present a brief overview of the state of research in explainability and its evaluation (or lack thereof). Then, I would offer a new lens into explanations, viewing them as a communication channel between a teacher and a student. This view enables us to quantitatively evaluate different attribution methods in a principled way at scale. Shifting gears, in the second part of the talk, I would introduce new techniques to supplement predictions with evidence to enable stakeholders to verify the outcomes readily.
diff --git a/_talks/2021-01-22-grad-student-spotlights.md b/_talks/2021-01-22-grad-student-spotlights.md
new file mode 100644
index 0000000..86fdf44
--- /dev/null
+++ b/_talks/2021-01-22-grad-student-spotlights.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "1. Introduction and Logistics"
+date: "2021-01-22 14:00:00 -0700"
+permalink: "/talks/2021-01-22-grad-student-spotlights/"
+slug: "2021-01-22-grad-student-spotlights"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+canceled: false
+speakers:
+ - name: "Grad Student Spotlights"
+ affiliation: "10-minute talks by 4 current graduate students"
+speaker_names: "Grad Student Spotlights"
+source_file: "_data/talks/2021-01-22-grad-student-spotlights.toml"
+generated: true
+---
+
+
+
+Archit Rathore: Exploring the Shape of Activations - A BERT case studyBenwei Shi: At-the-time and Back-in-time Persistent Sketches
+Joe Vinu: What all it takes for Performant Deep Nets to be Reliable and Fair?
+Brian Lavallee: Rounding Out Structural Rounding
+Vivek Gupta: Inference on Tables as Semi-Structured Data
diff --git a/_talks/2021-01-29-sanghamitra-dutta.md b/_talks/2021-01-29-sanghamitra-dutta.md
new file mode 100644
index 0000000..7349546
--- /dev/null
+++ b/_talks/2021-01-29-sanghamitra-dutta.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "A Systematic Understanding of Exempt and Non-Exempt Algorithmic Biases"
+date: "2021-01-29 14:00:00 -0700"
+permalink: "/talks/2021-01-29-sanghamitra-dutta/"
+slug: "2021-01-29-sanghamitra-dutta"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+canceled: false
+speakers:
+ - name: "Sanghamitra Dutta"
+ affiliation: "CMU"
+ bio: "Sanghamitra Dutta (B. Tech. IIT Kharagpur) is a doctoral candidate in the Department of Electrical and Computer Engineering at Carnegie Mellon University, PA, USA. Her research interests revolve around machine learning and information theory. She is currently focussed on addressing the emerging trust issues in machine learning concerning fairness, privacy and reliability. Her work bridges the fields of information theory, causality, reliability and machine learning. In her prior work, she has also examined problems in reliable computing, proposing novel algorithmic solutions for large-scale machine-learning in the presence of faults and failures, using tools from coding theory (an emerging area called “coded computing”). Her results on coded computing address problems that have been open for several decades and have received substantial attention from across communities. She is a recipient of the 2019 K&L Gates Presidential Fellowship, 2019 Axel Berny Presidential Graduate Fellowship, 2017 Tan Endowed Graduate Fellowship, 2016 Prabhu and Poonam Goel Graduate Fellowship, and the 2014 HONDA Young Engineer and Scientist Award."
+speaker_names: "Sanghamitra Dutta"
+source_file: "_data/talks/2021-01-29-sanghamitra-dutta.toml"
+generated: true
+---
+
+
+
+With the growing use of machine learning algorithms in highly consequential domains, the quantification and removal of bias with respect to gender, race, etc., is becoming increasingly important. While quantifying bias is essential, sometimes the needs of a business (e.g., hiring) may require the use of certain features that are critical in a way that any bias that can be explained by them might need to be exempted (inspired from the business necessity defense of Title VII of Civil Rights Act). For instance, in hiring a software engineer, a standardized coding-test score may be a critical feature that is weighed strongly in the decision even if it introduces bias, whereas other features, such as name, zip code, or reference letters may be used to improve decision-making, but only to the extent that they do not introduce bias. In this work, we propose a novel information-theoretic measure of non-exempt bias, which quantifies the part of the bias that cannot be accounted for by the critical features. This measure can be applied for (i) Auditing trained models to check if the bias arose purely due to the critical features; and also for (ii) Training with selective removal of the non-exempt bias if desired. We arrive at this decomposition through canonical examples that lead to a set of desirable properties (axioms) that any measure of non-exempt bias should satisfy. We then propose a causal measure of non-exempt bias that satisfies all of them. We also propose observational measures that only satisfy some of these properties (including an impossibility result on observational measures being able to satisfy all properties). Then, we perform case studies using them to show how one can train models while reducing non-exempt bias. Our quantification bridges ideas of causality, Simpson's paradox, and a body of work from information theory called Partial Information Decomposition (PID). The talk will be fairly accessible, no knowledge of information theory is required.
diff --git a/_talks/2021-02-05-michal-moshkovitz.md b/_talks/2021-02-05-michal-moshkovitz.md
new file mode 100644
index 0000000..4fdfcd3
--- /dev/null
+++ b/_talks/2021-02-05-michal-moshkovitz.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Unexpected Effects of Online no-Substitution k-means Clustering"
+date: "2021-02-05 14:00:00 -0700"
+permalink: "/talks/2021-02-05-michal-moshkovitz/"
+slug: "2021-02-05-michal-moshkovitz"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+canceled: false
+speakers:
+ - name: "Michal Moshkovitz"
+ affiliation: "UCSD"
+ bio: "Michal is a postdoctoral fellow at the Qualcomm Institute of the University of California, San Diego. Her interests lie in the foundations of AI, exploring how different constraints affect learning. She works on explainable machine learning, bounded memory learning, and online no-substitution clustering. Michal received her PhD from the Hebrew University and an MSc from Tel-Aviv University. She was the recipient of an Anita Borg scholarship from Google and a Hoffman scholarship from the Hebrew University."
+speaker_names: "Michal Moshkovitz"
+source_file: "_data/talks/2021-02-05-michal-moshkovitz.toml"
+generated: true
+---
+
+
+
+Offline k-means clustering was studied extensively and algorithms with a constant approximation are available. However, online clustering is still uncharted. New factors come into play: the ordering of the dataset and whether the number of points, n, is known in advance or not. Their exact effects are unknown. In this work, we focus on the online setting where the decisions are irreversible: after a point arrives the algorithm needs to decide whether to take the point as a center or not, and this decision is final. How many centers are needed and sufficient to achieve constant approximation in this setting? We show upper and lower bounds for all the different cases. These bounds are exactly the same up to a constant, thus achieving optimal bounds. For example, for k-means cost with constant k>1 and random order, Θ(logn) centers are enough to achieve a constant approximation, while the mere a priori knowledge of n reduces the number of centers to a constant. These bounds hold for any distance function that obeys a triangle-type inequality.
diff --git a/_talks/2021-02-12-fritz-lekschas.md b/_talks/2021-02-12-fritz-lekschas.md
new file mode 100644
index 0000000..3cb5221
--- /dev/null
+++ b/_talks/2021-02-12-fritz-lekschas.md
@@ -0,0 +1,37 @@
+---
+layout: "talk"
+title: "Visual Pattern Exploration At and Across Scales"
+date: "2021-02-12 14:00:00 -0700"
+permalink: "/talks/2021-02-12-fritz-lekschas/"
+slug: "2021-02-12-fritz-lekschas"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+canceled: false
+speakers:
+ - name: "Fritz Lekschas"
+ affiliation: "Harvard"
+speaker_names: "Fritz Lekschas"
+source_file: "_data/talks/2021-02-12-fritz-lekschas.toml"
+generated: true
+---
+
+
+
+Visually exploring data is a powerful approach to discover,
+understand, and interpret novel or not-well defined patterns. It
+allows us to gain insights and generate hypotheses for subsequent
+analyses. However, visual exploration can become challenging when the
+patterns of interest are sparsely-distributed, several orders of
+magnitude smaller than the entire dataset, or detected with high
+uncertainty. In this talk, I will discuss challenges in visually
+exploring multi-modal and multi-scale data, and present new
+visualization systems for efficiently browsing, comparing, and finding
+patterns in the context of genomic, geospatial, and time-series data.
+Specifically, I will describe a web platform for browsing multi-modal
+and multi-scale datasets, as well as their guided navigation. I will
+present a generalized framework and toolkit for interactively
+arranging, grouping, and aggregating thousands of pattern instances.
+And I will demonstrate how interactive visual machine learning can
+enhance our ability to find patterns effectively.
diff --git a/_talks/2021-02-19-samson-zhou.md b/_talks/2021-02-19-samson-zhou.md
new file mode 100644
index 0000000..f6a718e
--- /dev/null
+++ b/_talks/2021-02-19-samson-zhou.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Tight Bounds for Adversarially Robust Streams and Sliding Windows via Difference Estimators"
+date: "2021-02-19 14:00:00 -0700"
+permalink: "/talks/2021-02-19-samson-zhou/"
+slug: "2021-02-19-samson-zhou"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+canceled: false
+speakers:
+ - name: "Samson Zhou"
+ affiliation: "CMU"
+ bio: "Samson is a postdoctoral researcher at Carnegie Mellon University, hosted by David P. Woodruff. He received his PhD from Purdue, where he was advised by Greg Frederickson and Elena Grigorescu. He spent a year as a postdoctoral researcher at Indiana University, hosted by Grigory Yaroslavtsev. His research focuses on the theoretical foundations of data science, including sublinear algorithms with an emphasis on streaming algorithms, machine learning, and numerical linear algebra."
+speaker_names: "Samson Zhou"
+source_file: "_data/talks/2021-02-19-samson-zhou.toml"
+generated: true
+---
+
+
+
+We introduce difference estimators for data stream computation, which provide approximations to F(v)-F(u) for frequency vectors v,u and a given function F. We show how to use such estimators to carefully trade error for memory in an iterative manner. The function F is generally non-linear, and we give the first difference estimators for the frequency moments F_p for p between 0 and 2, as well as for integers p>2. Using these, we resolve a number of central open questions in adversarial robust streaming and sliding window models.
+
+For both models, we obtain algorithms for norm estimation whose dependence on epsilon is 1/epsilon^2, which shows, up to logarithmic factors, that there is no overhead over the standard insertion-only data stream model for these problems.
diff --git a/_talks/2021-02-26-rajesh-jayaram.md b/_talks/2021-02-26-rajesh-jayaram.md
new file mode 100644
index 0000000..9551b7e
--- /dev/null
+++ b/_talks/2021-02-26-rajesh-jayaram.md
@@ -0,0 +1,32 @@
+---
+layout: "talk"
+title: "An Improved Analysis of the Quadtree for High Dimensional EMD"
+date: "2021-02-26 14:00:00 -0700"
+permalink: "/talks/2021-02-26-rajesh-jayaram/"
+slug: "2021-02-26-rajesh-jayaram"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides: "https://rajeshjayaram.com/EarthMoverCJLW.pdf"
+canceled: false
+speakers:
+ - name: "Rajesh Jayaram"
+ affiliation: "CMU"
+ website: "https://rajeshjayaram.com/EarthMoverCJLW.pdf"
+ bio: "Rajesh Jayaram is a PhD student at Carnegie Mellon University, advised by David Woodruff. His research focuses on the design of sketching algorithms, especially streaming and distributed algorithms, for problems in big-data. A central theme of his work is the usage of sketching techniques to speed up algorithmic tasks across various applications, such as in machine learning, optimization, databases, and numerical linear algebra. Additionally, he is interested in aspects of robustness in machine learning and streaming. His work in sketching has received two Best Paper Awards at the Symposium on Principles of Database Systems (PODS) in 2019 and 2020."
+speaker_names: "Rajesh Jayaram"
+source_file: "_data/talks/2021-02-26-rajesh-jayaram.toml"
+generated: true
+---
+
+
+
+The Earth Mover Distance (EMD) between two multi-sets A,B in R^d of size s is the min-cost of bipartite matchings between points in A and B, where cost is measured by distance between points. In this talk, we discuss a classic divide-and-conquer algorithm known as Quadtree for approximating EMD.
+We give a new analysis of the Quadtree, showing that it gives a Õ(log s) approximation. This improves on the previous known O(min{log s , log d} * log s)-approximation of Andoni, Indyk, and Krauthgamer [SODA 08], and Backurs, Dong, Indyk, Razenshteyn, and Wagner (ICML 20).
+
+We also give new space efficient sketching and streaming algorithms for estimating EMD with the improved approximation factor. The main conceptual contribution is an analytical framework for studying the Quadtree which goes beyond worst-case distortion of randomized tree embeddings.
+
+Based on a joint work with Xi Chen, Amit Levi, and Erik Waingarten.
+
+Paper: https://rajeshjayaram.com/EarthMoverCJLW.pdf (https://rajeshjayaram.com/EarthMoverCJLW.pdf)
diff --git a/_talks/2021-03-12-yaoqing-yang.md b/_talks/2021-03-12-yaoqing-yang.md
new file mode 100644
index 0000000..76c47d5
--- /dev/null
+++ b/_talks/2021-03-12-yaoqing-yang.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Boundary thickness and robustness in learning models"
+date: "2021-03-12 14:00:00 -0700"
+permalink: "/talks/2021-03-12-yaoqing-yang/"
+slug: "2021-03-12-yaoqing-yang"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+canceled: false
+speakers:
+ - name: "Yaoqing Yang"
+ affiliation: "UC Berkeley"
+ bio: "Yaoqing Yang obtained the PhD degree from Carnegie Mellon University. Now he is a postdoctoral researcher in RISE Lab, UC Berkeley. His primary research interest lies in applying theoretical approaches to robustness issues in large-scale distributed machine learning, as well as designing learning algorithms on structured data such as point clouds and graphs."
+speaker_names: "Yaoqing Yang"
+source_file: "_data/talks/2021-03-12-yaoqing-yang.toml"
+generated: true
+---
+
+
+
+Robustness of machine learning models to various adversarial and non-adversarial corruptions continues to be of interest. In this talk, we present the notion of the "boundary thickness" of a classifier, and we describe its connection with and usefulness for model robustness. Thick decision boundaries lead to improved performance, while thin decision boundaries lead to overfitting (e.g., measured by the robust generalization gap between training and testing) and lower robustness. We show that a thicker boundary helps improve robustness against adversarial examples (e.g., improving the robust test accuracy of adversarial training) as well as so-called out-of-distribution (OOD) transforms, and we show that many commonly-used regularization and data augmentation procedures can increase boundary thickness. On the theoretical side, we establish that maximizing boundary thickness during training is akin to the so-called mixup training procedure. Using these observations, we show that noise-augmentation on mixup training further increases boundary thickness, thereby combating vulnerability to various forms of adversarial attacks and OOD transforms. We can also show that the performance improvement in several lines of recent work happens in conjunction with a thicker boundary.
diff --git a/_talks/2021-03-19-vivek-gupta.md b/_talks/2021-03-19-vivek-gupta.md
new file mode 100644
index 0000000..27e84a6
--- /dev/null
+++ b/_talks/2021-03-19-vivek-gupta.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Logic based classification for Low Resource Setting"
+date: "2021-03-19 14:00:00 -0600"
+permalink: "/talks/2021-03-19-vivek-gupta/"
+slug: "2021-03-19-vivek-gupta"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+slides: "https://www.aclweb.org/anthology/2020.aacl-main.71.pdf"
+canceled: false
+speakers:
+ - name: "Vivek Gupta"
+ affiliation: "University of Utah"
+ website: "https://www.aclweb.org/anthology/2020.aacl-main.71.pdf"
+speaker_names: "Vivek Gupta"
+source_file: "_data/talks/2021-03-19-vivek-gupta.toml"
+generated: true
+---
+
+
+
+An NLP model’s ability to reason should be independent of language. Previous works utilize Natural Language Inference(NLI) to understand the reasoning ability of models, mostly focusing on high resource languages like English. To address scarcity of data in low-resource languages such as Hindi, we use data recasting to create four NLI datasets from existing four text classification datasets in Hindi language. Through experiments, we show that our recasted dataset is devoid of statistical irregularities and spurious patterns. We study the consistency in predictions of the textual entailment models and propose a consistency regulariser to remove pairwise-inconsistencies in predictions. Furthermore, we propose a novel two-step classification method which uses textual-entailment predictions for classification tasks. We further improve the classification performance by jointly training the classification and textual entailment tasks together. We therefore highlight the benefits of data recasting and our approach with supporting experimental results. You can access the dataset and paper here: https://www.aclweb.org/anthology/2020.aacl-main.71.pdf (https://www.aclweb.org/anthology/2020.aacl-main.71.pdf). Joint work with BloomBerg AI.
diff --git a/_talks/2021-04-09-emily-beth-wall.md b/_talks/2021-04-09-emily-beth-wall.md
new file mode 100644
index 0000000..e4a3926
--- /dev/null
+++ b/_talks/2021-04-09-emily-beth-wall.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "As We Are: Detecting and Mitigating Human Bias in Visual Analytics"
+date: "2021-04-09 14:00:00 -0600"
+permalink: "/talks/2021-04-09-emily-beth-wall/"
+slug: "2021-04-09-emily-beth-wall"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "Virtual"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+canceled: false
+speakers:
+ - name: "Emily Beth Wall"
+ affiliation: "Emory"
+ bio: "Dr. Emily Wall is an Assistant Professor in the Computer Science department at Emory University (beginning Summer 2021). She completed her PhD in the School of Interactive Computing at Georgia Tech in 2020 and is currently a Postdoctoral Scholar at Northwestern University. Her research interests lie at the intersection of cognitive science and data visualization. Particularly, her research has focused on increasing awareness of unconscious and implicit human biases through the design and evaluation of (1) computational approaches to quantify bias from user interaction and (2) interfaces to support visual data analysis. Her research has been supported by NSF, Pacific Northwest National Laboratory, and Siemens, among others."
+speaker_names: "Emily Beth Wall"
+source_file: "_data/talks/2021-04-09-emily-beth-wall.toml"
+generated: true
+---
+
+
+
+Visual Analytics combines the complementary strengths of humans (perception and sensemaking capabilities) and machines (fast and accurate information processing). However, people are susceptible to inherent limitations and biases, including cognitive biases (e.g., anchoring bias), social biases borne of cultural stereotypes and prejudices (e.g., gender bias), and perceptual biases (e.g., illusions). These biases can impact data analysis and decision making in critical ways, leading to inaccurate or inefficient choices, or even propagating long-standing institutional and systemic biases.
+
+Given our knowledge of these biases and the increased use of data visualization to support decision making in data science, the goal of this research is to detect and mitigate human biases in visual data analysis. In this talk, I describe (1) which types of bias are particularly relevant in the process of visual data analysis, (2) how user interactions with data can be used to approximate human biases, and (3) how visualization systems can be designed to increase user awareness of potentially unconscious or implicit biases. By creating systems that promote real-time awareness of bias, people can reflect on their behavior and decision making and ultimately engage in a less-biased analysis and decision making process.
diff --git a/_talks/2021-04-16-arun-sai-suggala.md b/_talks/2021-04-16-arun-sai-suggala.md
new file mode 100644
index 0000000..9120211
--- /dev/null
+++ b/_talks/2021-04-16-arun-sai-suggala.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "Game Theoretic Statistics"
+date: "2021-04-16 14:00:00 -0600"
+permalink: "/talks/2021-04-16-arun-sai-suggala/"
+slug: "2021-04-16-arun-sai-suggala"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "Virtual"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+canceled: false
+speakers:
+ - name: "Arun Sai Suggala"
+ affiliation: "CMU"
+ bio: "Arun Sai Suggala is a final year Machine Learning PhD student at Carnegie Mellon University (CMU), advised by Pradeep Ravikumar. Arun is broadly interested in online learning, game theory, and their applications to machine learning and statistics. He is particularly interested in designing new algorithmic and analytic tools in game theory for solving statistical and machine learning problems. His work has received the best student paper award at ALT'20. Prior to CMU, Arun completed his undergraduate studies in Computer Science and Engineering in the Indian Institute of Technology, Bombay."
+speaker_names: "Arun Sai Suggala"
+source_file: "_data/talks/2021-04-16-arun-sai-suggala.toml"
+generated: true
+---
+
+
+
+Game theory and statistics are often regarded as disparate research areas. This is because typical statistical estimation settings are non-adversarial, and the samples are assumed to be generated by some stationary non-reactive source. However, there is a great degree of commonality between the two fields. Classically, the mathematical philosophy of statistics, particularly frequentist statistics, posits that the source of samples is potentially adversarial. This resulted in the rich theory of minimax statistical games and estimation. Boosting algorithms, which are often regarded as best off-the-shelf classifiers, can be viewed as playing a zero-sum game against a weak learner. To allow for various departures of ``test environment'' from ``train environments'', the emerging field of robust machine learning allows for adversarial manipulation of the train or test environments. Finally, an emerging class of density estimators (GANs) in modern machine learning use an adversarial ``critic'' of the density estimator to improve the final density estimation. The common theme among these classical and modern developments is an interplay between statistical estimation and two player games.
+
+In this talk, I will present some of my recent work at the intersection of statistics and game theory and show how game theory can help statistics. In particular, I will present my work on minimax statistical estimation, where we develop algorithmic techniques for constructing minimax estimators. Our algorithms rely on tools from online nonconvex learning and help us construct minimax estimators for fundamental estimation problems such as covariance estimation and entropy estimation.
diff --git a/_talks/2021-04-23-lizzie-kumar.md b/_talks/2021-04-23-lizzie-kumar.md
new file mode 100644
index 0000000..f04717e
--- /dev/null
+++ b/_talks/2021-04-23-lizzie-kumar.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Epistemic values in feature importance methods: Lessons from feminist epistemology"
+date: "2021-04-23 14:00:00 -0600"
+permalink: "/talks/2021-04-23-lizzie-kumar/"
+slug: "2021-04-23-lizzie-kumar"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "Virtual"
+zoom: "https://us02web.zoom.us/j/87538638627?pwd=WGZrYmIyNjFsVEtwWk5pemRuM0JsZz09"
+canceled: false
+speakers:
+ - name: "Lizzie Kumar"
+ affiliation: "Utah"
+ bio: "Lizzie Kumar is a second-year Computing Ph.D. student advised by Suresh Venkatasubramanian at the University of Utah where her work has previously been supported by the ARCS Foundation. She is interested in the practice of analyzing the social impact of machine learning systems and developing responsible AI law and policy. Previously, she developed risk models on the Data Science team at MassMutual while completing her M.S. in Computer Science at the University of Massachusetts, and also holds a B.A. in Mathematics from Scripps College."
+speaker_names: "Lizzie Kumar"
+source_file: "_data/talks/2021-04-23-lizzie-kumar.toml"
+generated: true
+---
+
+
+
+As the public seeks greater accountability and transparency from machine learning algorithms, the research literature on methods to explain algorithms and their outputs has rapidly expanded. Feature importance, or the practice of assigning quantitative importance values to the input features of a machine learning model, form a popular class of such methods. Much of the research on feature importance rests on formalizations that attempt to capture universally desirable properties. We investigate the ways in which epistemic values are implicitly embedded in these methods and analyze the ways in which they conflict with ideas from feminist philosophy. We offer some suggestions on how to conduct research on explanations that respects feminist epistemic values, taking into account the importance of social context, the epistemic privileges of subjugated knowers, and adopting more interactional ways of knowing.
diff --git a/_talks/2021-08-27-jeff-phillips.md b/_talks/2021-08-27-jeff-phillips.md
new file mode 100644
index 0000000..b41f8eb
--- /dev/null
+++ b/_talks/2021-08-27-jeff-phillips.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "A Visual tour of Bias Mitigation"
+date: "2021-08-27 14:00:00 -0600"
+permalink: "/talks/2021-08-27-jeff-phillips/"
+slug: "2021-08-27-jeff-phillips"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Jeff Phillips"
+ affiliation: "University of Utah"
+ website: "https://www.cs.utah.edu/~jeffp/"
+speaker_names: "Jeff Phillips"
+source_file: "_data/talks/2021-08-27-jeff-phillips.toml"
+generated: true
+---
+
+
+
+Word vector embeddings have been shown to contain and amplify biases in data they are extracted from. Consequently, many techniques have been proposed to identify, mitigate, and attenuate these biases in word representations. In this talk, I will review a collection of state-of-the-art debiasing techniques. To aid this, we provide an open source web-based visualization tool VERB (Visualization of Embedding Representations for deBiasing) and offer hands-on experience in exploring the effects of these debiasing techniques on the geometry of high-dimensional word vectors. To help understand how various debiasing techniques change the underlying geometry, I will show how to decompose each technique into interpretable sequences of primitive operations and study their effect on the word vectors using dimensionality reduction and interactive visual exploration.
diff --git a/_talks/2021-09-03-shandian-zhe.md b/_talks/2021-09-03-shandian-zhe.md
new file mode 100644
index 0000000..1baa6e8
--- /dev/null
+++ b/_talks/2021-09-03-shandian-zhe.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Multi-fidelity Learning and Optimization for Physical Simulation and AutoML"
+date: "2021-09-03 14:00:00 -0600"
+permalink: "/talks/2021-09-03-shandian-zhe/"
+slug: "2021-09-03-shandian-zhe"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Shandian Zhe"
+ affiliation: "Utah SoC"
+ website: "https://www.cs.utah.edu/~zhe/"
+speaker_names: "Shandian Zhe"
+source_file: "_data/talks/2021-09-03-shandian-zhe.toml"
+generated: true
+---
+
+
+
+Multi-fidelity learning involves using training examples at different fidelities or resolutions. High-fidelity examples are of high-quality but often are much more costly to collect than inaccurate, low-fidelity examples. How to retrieve and leverage examples at multiple fidelities is the key to reduce the learning cost while maximizing efficiency.
+
+This talk will introduce our recent work in multi-fidelity learning and optimization. First, I will introduce our deep auto-regressive models that can capture complex correlations across the fidelities to integrate examples of high-dimensional outputs. These are common in applications of physical simulation. Second, I will introduce our work of deep multi-fidelity active learning and Bayesian optimization that can improve the learning and optimization efficiency while reducing the cost of generating training examples, namely maximizing the benefit-cost ratio. Finally, a batch version of the active learning and optimization technique will be presented, which can reduce the query redundancy, improve diversity, and further boost the benefit-cost ratio. I will showcase the advantage of our methods in standard benchmarks of physical simulation, topology structure optimization, and typical tasks in hyper-parameter tuning/AutoML.
diff --git a/_talks/2021-09-10-pierre-lermusiaux.md b/_talks/2021-09-10-pierre-lermusiaux.md
new file mode 100644
index 0000000..1d61785
--- /dev/null
+++ b/_talks/2021-09-10-pierre-lermusiaux.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Neural Closure Models for Dynamical Systems"
+date: "2021-09-10 14:00:00 -0600"
+permalink: "/talks/2021-09-10-pierre-lermusiaux/"
+slug: "2021-09-10-pierre-lermusiaux"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Pierre Lermusiaux"
+ website: "http://meche.mit.edu/people/faculty/pierrel@mit.edu"
+ bio: "Abhinav is a 5th year Ph.D. candidate in Mechanical Engineering and Computation at MIT. He received his Bachelor's degree and Master's degree in Mechanical Engineering from the Indian Institute of Technology, Kanpur. At MIT he was a fellow of the MIT-Tata Center for Technology & Design from 2018-20, and recipient of the 2020-21 MathWorks Mechanical Engineering Fellowship.Abhinav is currently developing state-of-the-art scientific machine learning algorithms, with applications to predictive ocean modeling. Apart from his present work, he has specifically worked on uncertainty quantification, data assimilation, Bayesian model learning, and optimal sampling for high-dimensional systems. The algorithms he develops are problem agnostic and can be widely applied. He believes that his unique background in mechanical engineering, applied mathematics, machine learning, and computing position him to identify and implement cross-disciplinary solutions to problems.\n\nhttps://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09 (https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09)\n"
+speaker_names: "Pierre Lermusiaux"
+source_file: "_data/talks/2021-09-10-pierre-lermusiaux.toml"
+generated: true
+---
+
+
+
+Complex dynamical systems are used for predictions in many domains. Because of computational costs, models are truncated, coarsened or aggregated. As the neglected and unresolved terms become important, the utility of model predictions diminishes. We develop a novel, versatile and rigorous methodology to learn non-Markovian closure parametrizations for known-physics/low-fidelity models using data from high-fidelity simulations. The new neural closure models augment low-fidelity models with neural delay differential equations (nDDEs), motivated by the Mori–Zwanzig formulation and the inherent delays in complex dynamical systems. We demonstrate that neural closures efficiently account for truncated modes in reduced-order models, capture the effects of subgrid-scale processes in coarse models and augment the simplification of complex biological and physical-biogeochemical models. We find that using non-Markovian over Markovian closures improves long-term prediction accuracy and requires smaller networks. We derive adjoint equations and network architectures needed to efficiently implement the new discrete and distributed nDDEs, for any time-integration schemes and allowing non-uniformly spaced temporal training data. The performance of discrete over-distributed delays in closure models is explained using information theory, and we find an optimal amount of past information for a specified architecture. Finally, we analyze computational complexity and explain the limited additional cost due to neural closure models.
+
+Paper: https://royalsocietypublishing.org/doi/10.1098/rspa.2020.1004 (https://royalsocietypublishing.org/doi/10.1098/rspa.2020.1004?fbclid=IwAR1DZy5lHuf68lYwkk5Z_kUYzbHXdgRi0zUhoy5bohGy-w0BjZuwjqPhytQ)
diff --git a/_talks/2021-09-17-ross-whitaker.md b/_talks/2021-09-17-ross-whitaker.md
new file mode 100644
index 0000000..cf75b74
--- /dev/null
+++ b/_talks/2021-09-17-ross-whitaker.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Air Quality Mapping Using Sensor Networks and Statistical Regression"
+date: "2021-09-17 14:00:00 -0600"
+permalink: "/talks/2021-09-17-ross-whitaker/"
+slug: "2021-09-17-ross-whitaker"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+canceled: false
+speakers:
+ - name: "Ross Whitaker"
+ affiliation: "Utah SoC & SCI"
+ website: "http://www.cs.utah.edu/~whitaker/"
+speaker_names: "Ross Whitaker"
+source_file: "_data/talks/2021-09-17-ross-whitaker.toml"
+generated: true
+---
+
+
+
+This talk describes an air quality mapping system that is currently deployed in the Salt Lake Valley. We begin with the motivations for the system and then briefly describe the cyberphysical infrastructure. We then review the Gaussian process modeling approach we are using and discuss several important practical considerations around challenges of data wrangling and numerical implementations. We present some examples of AQ estimates associated with specific events of bad air quality. Finally we talk about current directions in research associated with this approach.
+
+COI Disclaimer: Ross Whitaker has a financial interest in the company Tetrad, which has business interests related to the technologies in this talk.
+
+MEB 3147
diff --git a/_talks/2021-09-24-c-seshadhri.md b/_talks/2021-09-24-c-seshadhri.md
new file mode 100644
index 0000000..7c04ae1
--- /dev/null
+++ b/_talks/2021-09-24-c-seshadhri.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "Studying the (in)effectiveness of low dimensional graph embeddings"
+date: "2021-09-24 14:00:00 -0600"
+permalink: "/talks/2021-09-24-c-seshadhri/"
+slug: "2021-09-24-c-seshadhri"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "C. Seshadhri"
+ affiliation: "UC Santa Cruz"
+ website: "https://users.soe.ucsc.edu/~sesh/"
+ bio: "C. Seshadhri (Sesh) is a professor of Computer Science at the University of California, Santa Cruz. Prior to joining UCSC, he was a researcher at Sandia National Labs, Livermore. His primary interests are in theoretical computer science and the mathematical foundations of big data algorithms. His work spans many areas: sublinear algorithms, graph algorithms, graph modeling, scalable computation, and data mining. In the theory world, his work has resolved numerous open problems in property testing and sublinear algorithms. A number of his papers in the interface of TCS and applied algorithms have received paper awards at KDD, WWW, ICDM, SDM, and WSDM. He received the 2019 SDM/IBM Early Career Award for Excellence in Data Analytics.\n\nhttps://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09 (https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09)\n"
+speaker_names: "C. Seshadhri"
+source_file: "_data/talks/2021-09-24-c-seshadhri.toml"
+generated: true
+---
+
+
+
+Low dimensional graph embeddings are a fundamental and popular tool used for machine learning on graphs. Given a graph, the basic idea is to produce a low-dimensional vector for each vertex, such that "similarity" in geometric space corresponds to "proximity" in the graph. These vectors can then be used as features in a plethora of machine learning tasks, such as link prediction, community labeling, recommendations, etc. Despite many results emerging in this area over the past few years, there is less study on the core premise of these embeddings. Can such low-dimensional embeddings effectively capture the structure of real-world (such as social) networks? Contrary to common wisdom, we mathematically prove and empirically demonstrate that popular low-dimensional graph embeddings do not capture salient properties of real-world networks. We mathematically prove that common low-dimensional embeddings cannot generate graphs with both low average degree and large clustering coefficients, which have been widely established to be empirically true for real-world networks. Empirically, we observe that the embeddings generated by popular methods fail to recreate the triangle structure of real-world networks, and do not perform well on certain community labeling tasks.
+
+Joint work with Ashish Goel, Caleb Levy, Aneesh Sharma, and Andrew Stolman
diff --git a/_talks/2021-10-01-tony-h-grubesic.md b/_talks/2021-10-01-tony-h-grubesic.md
new file mode 100644
index 0000000..39e75f9
--- /dev/null
+++ b/_talks/2021-10-01-tony-h-grubesic.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Estimating Potential Oil Spill Trajectories and Coastal Impacts from Near-Shore Storage Facilities: A Case Study of FSO Nabarima and the Gulf of Paria"
+date: "2021-10-01 14:00:00 -0600"
+permalink: "/talks/2021-10-01-tony-h-grubesic/"
+slug: "2021-10-01-tony-h-grubesic"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Tony H. Grubesic"
+ affiliation: "UT Austin"
+ website: "http://tonygrubesic.net"
+speaker_names: "Tony H. Grubesic"
+source_file: "_data/talks/2021-10-01-tony-h-grubesic.toml"
+generated: true
+---
+
+
+
+The FSO Nabarima is a floating storage facility and offloading vessel in the Gulf of Paria, between Venezuela and the island of Trinidad. During the latter half of 2020, the Nabarima was disabled, holding approximately 1.3 million barrels (55 million gallons) of crude oil on board. In October of 2020, the vessel was tilting and potentially at risk of spilling its payload into open water. Although all of the oil on the Nabarima was successfully offloaded by April 2021, the threat of large crude oil releases is ubiquitous and persistent in many coastal regions, threatening local ecosystems and livelihoods in coastal communities. The purpose of this presentation is to highlight a geocomputational framework for evaluating the potential spatial vulnerability of coastlines should oil be released from near-shore storage facilities. We use the Nabarima as a broadly representative case study, discuss potential spill cleanup and mitigation strategies, and highlight the challenges of coordinating cross-national responses to these types of spill scenarios.
diff --git a/_talks/2021-10-08-anna-little.md b/_talks/2021-10-08-anna-little.md
new file mode 100644
index 0000000..b16cb9a
--- /dev/null
+++ b/_talks/2021-10-08-anna-little.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "The Mathematics of the Signal-to-Noise Ratio and Insights for Data Science"
+date: "2021-10-08 14:00:00 -0600"
+permalink: "/talks/2021-10-08-anna-little/"
+slug: "2021-10-08-anna-little"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Anna Little"
+ affiliation: "Utah Math"
+ website: "https://www.anna-little.com"
+speaker_names: "Anna Little"
+source_file: "_data/talks/2021-10-08-anna-little.toml"
+generated: true
+---
+
+
+
+Despite the huge variety of data types and goals in data science, there is usually an underlying signal-to-noise ratio that governs the difficulty of the data science task. Analysis of this signal-to-noise ratio leads to important insights about sample size requirements and the impact of the data dimension, which are consistent across various data models and tasks. This talk will illustrate these universal insights in three specific contexts: (1) density-based clustering via graph embeddings, (2) clustering mixture models via multidimensional scaling, and (3) signal recovery from noisy data. The underlying data models are motivated by applications such as imaging, particle physics, single-cell RNA sequencing, and cryo-electron microscopy.
diff --git a/_talks/2021-10-22-erin-wolf-chambers.md b/_talks/2021-10-22-erin-wolf-chambers.md
new file mode 100644
index 0000000..1671051
--- /dev/null
+++ b/_talks/2021-10-22-erin-wolf-chambers.md
@@ -0,0 +1,44 @@
+---
+layout: "talk"
+title: "Applications of topology and geometry to root analysis"
+date: "2021-10-22 14:00:00 -0600"
+permalink: "/talks/2021-10-22-erin-wolf-chambers/"
+slug: "2021-10-22-erin-wolf-chambers"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Erin Wolf Chambers"
+ affiliation: "St. Loius University"
+ website: "https://cs.slu.edu/~chambers/"
+speaker_names: "Erin Wolf Chambers"
+source_file: "_data/talks/2021-10-22-erin-wolf-chambers.toml"
+generated: true
+---
+
+
+
+Analysis of 3d shapes is a core problem in many fields, and there are
+many tools from topology and geometry that can provide insight and
+understanding. In this talk, we focus on developing significance
+measures for 3d plant structures, primarily root systems of plants.
+Our measures are based on the medial axis transform, which plays a
+fundamental role in shape matching and analysis, but is widely known
+to be unstable to even small boundary perturbations. Methods for
+pruning the medial axis are usually guided by some measure of
+significance, with considerable work done for both 2- and
+3-dimensional shapes. Such significance measures can be used for
+identifying salient features, and hence are useful for simplification,
+comparison, and alignment. In this talk, we will present theoretical
+insights and properties of commonly used significance measures,
+focusing on those in 2D and 3D that are both shape-revealing and
+topology-preserving, as well as being robust to noise on the boundary.
+We'll then discuss several methods that de-noise a shape and identify
+topologically and geometrically prominent features, using both the
+medial axis and other measures commonly used in topological data
+analysis. Our methods are quite successful compared to the state of
+the art, and are available in the package TopoRoot, an automatic
+pipeline for plant architectural analysis from 3D Imaging.
diff --git a/_talks/2021-10-29-bao-wang.md b/_talks/2021-10-29-bao-wang.md
new file mode 100644
index 0000000..806e524
--- /dev/null
+++ b/_talks/2021-10-29-bao-wang.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "How Differential Equations and Random Graph Insights Benefit Deep Learning"
+date: "2021-10-29 14:00:00 -0600"
+permalink: "/talks/2021-10-29-bao-wang/"
+slug: "2021-10-29-bao-wang"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Bao Wang"
+ affiliation: "Utah Math, SCI"
+ website: "http://www.math.utah.edu/~bwang/index.html"
+speaker_names: "Bao Wang"
+source_file: "_data/talks/2021-10-29-bao-wang.toml"
+generated: true
+---
+
+
+
+We will present recent results on developing new deep learning algorithms leveraging differential equations and random graph insights.
+First, we will present a new class of continuous-depth deep neural networks that were motivated by the ODE limit of the classical momentum method, named heavy-ball neural ODEs (HBNODEs). HBNODEs enjoy two properties that imply practical advantages over NODEs: (i) The adjoint state of an HBNODE also satisfies an HBNODE, accelerating both forward and backward ODE solvers, thus significantly accelerate learning and improve the utility of the trained models. (ii) The spectrum of HBNODEs is well structured, enabling effective learning of long-term dependencies from complex sequential data.
+Second, we will extend HBNODE to graph learning leveraging diffusion on graphs, resulting in new algorithms for deep graph learning. The new algorithms are more accurate than existing deep graph learning algorithms and more scalable to deep architectures, and also suitable for learning at low labeling rate regimes. Moreover, we will present a fast multipole method-based efficient attention mechanism for modeling graph nodes interactions.
+Third, if time permits, we will discuss building an efficient and reliable overlay network for decentralized federated learning based on the random graph theory.
diff --git a/_talks/2021-11-05-marina-kogan.md b/_talks/2021-11-05-marina-kogan.md
new file mode 100644
index 0000000..6700e6b
--- /dev/null
+++ b/_talks/2021-11-05-marina-kogan.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Sequence-based approaches as human-centered data science methods for crisis informatics"
+date: "2021-11-05 14:00:00 -0600"
+permalink: "/talks/2021-11-05-marina-kogan/"
+slug: "2021-11-05-marina-kogan"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Marina Kogan"
+ affiliation: "Utah SoC"
+ website: "http://www.mkoganresearch.com"
+speaker_names: "Marina Kogan"
+source_file: "_data/talks/2021-11-05-marina-kogan.toml"
+generated: true
+---
+
+
+
+Social media platforms have been increasingly used by the public in crisis situations, partly because they upend the traditional top-down broadcasting model of risk communication. Instead, social media platforms facilitate a two-way information exchange between the official response channels and the general public, enabling more participatory crisis communication, as well as coordination and self-organization among the public. In this more complex information ecosystem, understanding the flow of information is crucial to supporting those affected and preventing malicious actors from capitalizing on the uncertainty. However, the study of such information flows is challenging, because the high-tempo, high-volume convergent nature of crisis events produces vast amounts of social media data, necessitating the use of the data science methods. On the other hand, to glean meaningful insight from the crisis-related social media activity, it is necessary to use methods that account for the complex social context of the user activity. In this talk I will show how the Human-Centered Data Science (HCDS) provides methodological approaches that both harness the power of computational methods and account for the highly situated nature of social media activity in disruption. I will focus on sequence-based approaches as examples of HCDS methods in two empirical studies: analysis of attention-garnering information during a natural disaster and investigation of behavioral signatures in coordinated information operations.
diff --git a/_talks/2021-11-12-mikhail-belkin.md b/_talks/2021-11-12-mikhail-belkin.md
new file mode 100644
index 0000000..dc38a6e
--- /dev/null
+++ b/_talks/2021-11-12-mikhail-belkin.md
@@ -0,0 +1,37 @@
+---
+layout: "talk"
+title: "From classical statistics to modern deep learning"
+date: "2021-11-12 14:00:00 -0700"
+permalink: "/talks/2021-11-12-mikhail-belkin/"
+slug: "2021-11-12-mikhail-belkin"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Mikhail Belkin"
+ website: "http://misha.belkin-wang.org"
+speaker_names: "Mikhail Belkin"
+source_file: "_data/talks/2021-11-12-mikhail-belkin.toml"
+generated: true
+---
+
+
+
+Recent empirical successes of deep learning have exposed significant gaps in our
+fundamental understanding of learning and optimization mechanisms.
+Modern best practices for model selection are in direct contradiction to the methodologies
+suggested by classical analyses. Similarly, the efficiency of SGD-based local methods
+used in training modern models, appeared at odds with the standard intuitions on optimization.
+
+First, I will present evidence, empirical and mathematical, that necessitates
+revisiting classical statistical notions, such as over-fitting. I will continue to discuss the emerging
+understanding of generalization, and, in particular, the "double descent" risk curve, which extends
+the classical U-shaped generalization curve beyond the point of interpolation.
+
+Second, I will discuss why the landscapes of over-parameterized neural networks are
+generically never convex, even locally. Instead they satisfy the Polyak-Lojasiewicz (PL)
+condition across most of the parameter space instead, presents an powerful framework for optimization in general over-parameterized models and allows SGD-type methods to converge to a global minimum.
+
+While our understanding has significantly grown in the last few years, a key piece of the puzzle remains -- how does optimization align with statistics to form the complete mathematical picture of modern ML?
diff --git a/_talks/2021-11-19-michael-yeh.md b/_talks/2021-11-19-michael-yeh.md
new file mode 100644
index 0000000..8050f6a
--- /dev/null
+++ b/_talks/2021-11-19-michael-yeh.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Towards a Near Universal Time Series Data Mining Tool: Introducing the Matrix Profile"
+date: "2021-11-19 14:00:00 -0700"
+permalink: "/talks/2021-11-19-michael-yeh/"
+slug: "2021-11-19-michael-yeh"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Michael Yeh"
+ affiliation: "Visa Research"
+ website: "https://mcyeh.github.io/"
+speaker_names: "Michael Yeh"
+source_file: "_data/talks/2021-11-19-michael-yeh.toml"
+generated: true
+---
+
+
+
+Matrix profile is a data structure that annotates a time series by recording the location of and the distance to the nearest neighbors of each subsequences in the time series. The matrix profile stores such information in an efficient and easy-to-access fashion and can be used in a variety of data mining tasks like motif/discord discovery, semantic segmentation, and clustering. In this talk, I will 1) introduce what matrix profile is, 2) discuss the computational challenge associated with matrix profile, and 3) show how it can be used in different time series data mining tasks.
diff --git a/_talks/2021-12-03-julia-silge.md b/_talks/2021-12-03-julia-silge.md
new file mode 100644
index 0000000..50ffb08
--- /dev/null
+++ b/_talks/2021-12-03-julia-silge.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Data visualization for machine learning practitioners"
+date: "2021-12-03 14:00:00 -0700"
+permalink: "/talks/2021-12-03-julia-silge/"
+slug: "2021-12-03-julia-silge"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147 and Zoom"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Julia Silge"
+ affiliation: "RStudio"
+ website: "https://juliasilge.com"
+ bio: "Julia Silge is a data scientist and software engineer at RStudio PBC where she works on open source modeling tools. She is an author, an international keynote speaker, and a real-world practitioner focusing on data analysis and machine learning practice. Julia loves text analysis, making beautiful charts, and communicating about technical topics with diverse audiences.\n\nMEB 3147 (coffee and snacks provided)\n"
+speaker_names: "Julia Silge"
+source_file: "_data/talks/2021-12-03-julia-silge.toml"
+generated: true
+---
+
+
+
+Visual representations of data inform how machine learning practitioners think, understand, and decide. Before charts are ever used for outward communication about a ML system, they are used by the system designers and operators themselves as a tool to make better modeling choices. Practitioners use visualization, from very familiar statistical graphics to creative and less standard plots, at the points of most important human decisions when other ways to validate those decisions can be difficult. Visualization approaches are used to understand both the data that serves as input for machine learning and the models that practitioners create. In this talk, learn about the process of building a ML model in the real world, how and when practitioners use visualization to make more effective choices, and considerations for ML visualization tooling.
diff --git a/_talks/2021-12-10-sameer-singh.md b/_talks/2021-12-10-sameer-singh.md
new file mode 100644
index 0000000..9f4e7ac
--- /dev/null
+++ b/_talks/2021-12-10-sameer-singh.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Evaluating and Testing Natural Language Processing Models"
+date: "2021-12-10 14:00:00 -0700"
+permalink: "/talks/2021-12-10-sameer-singh/"
+slug: "2021-12-10-sameer-singh"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/93778940103?pwd=TStQRWhWVjRxd0hGV1hTK05SUFZwUT09"
+canceled: false
+speakers:
+ - name: "Sameer Singh"
+ affiliation: "UC Irvine"
+ website: "http://sameersingh.org"
+ bio: "Dr. Sameer Singh is an Associate Professor of Computer Science at the University of California, Irvine (UCI) and an Allen AI Fellow at Allen Institute for AI. He is working primarily on robustness and interpretability of machine learning algorithms, along with models that reason with text and structure for natural language processing. Sameer was a postdoctoral researcher at the University of Washington and received his PhD from the University of Massachusetts, Amherst. He has received the NSF CAREER award, selected as a DARPA Riser, UCI Distinguished Early Career Faculty award, and the Hellman Faculty Fellowship. His group has received funding from Allen Institute for AI, Amazon, NSF, DARPA, Adobe Research, Hasso Plattner Institute, NEC, Base 11, and FICO. Sameer has published extensively at machine learning and natural language processing venues and received conference paper awards at KDD 2016, ACL 2018, EMNLP 2019, AKBC 2020, and ACL 2020. (https://sameersingh.org/)"
+speaker_names: "Sameer Singh"
+source_file: "_data/talks/2021-12-10-sameer-singh.toml"
+generated: true
+---
+
+
+
+Current evaluation of the generalization of natural language processing (NLP) systems, and much of machine learning, primarily consists of measuring the accuracy on held-out instances of the dataset. Since the held-out instances are often gathered using similar annotation process as the training data, they include the same biases that act as shortcuts for machine learning models, allowing them to achieve accurate results without requiring actual natural language understanding. Thus held-out accuracy is often a poor proxy for measuring generalization. Further, aggregate metrics have little to say about where the problems may lie, and how to address them.
+In this talk, I will introduce a number of approaches we are investigating to perform a more thorough evaluation of NLP systems. I will first provide a quick overview of automated techniques for perturbing instances in the dataset that identify loopholes and shortcuts in NLP models, including semantic adversaries and universal triggers. I will then describe recent work on creating comprehensive and thorough tests and evaluation benchmarks for NLP using CheckList, that aim to directly evaluate comprehension and understanding capabilities. The talk will include a number of NLP tasks, such as sentiment analysis, textual entailment, paraphrase detection, and question answering.
+
+Dr. Sameer Singh is an Associate Professor of Computer Science at the University of California, Irvine (UCI) and an Allen AI Fellow at Allen Institute for AI. He is working primarily on robustness and interpretability of machine learning algorithms, along with models that reason with text and structure for natural language processing. Sameer was a postdoctoral researcher at the University of Washington and received his PhD from the University of Massachusetts, Amherst. He has received the NSF CAREER award, selected as a DARPA Riser, UCI Distinguished Early Career Faculty award, and the Hellman Faculty Fellowship. His group has received funding from Allen Institute for AI, Amazon, NSF, DARPA, Adobe Research, Hasso Plattner Institute, NEC, Base 11, and FICO. Sameer has published extensively at machine learning and natural language processing venues and received conference paper awards at KDD 2016, ACL 2018, EMNLP 2019, AKBC 2020, and ACL 2020. (https://sameersingh.org/)
diff --git a/_talks/2022-01-14-yi-zhou.md b/_talks/2022-01-14-yi-zhou.md
new file mode 100644
index 0000000..d613783
--- /dev/null
+++ b/_talks/2022-01-14-yi-zhou.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Understanding the Convergence of Optimization Algorithms for Minimax Machine Learning"
+date: "2022-01-14 15:00:00 -0700"
+permalink: "/talks/2022-01-14-yi-zhou/"
+slug: "2022-01-14-yi-zhou"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "WEB 1250"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Yi Zhou"
+ affiliation: "Utah ECE"
+ website: "https://sites.google.com/site/yizhouhomepage/home"
+ bio: "Yi Zhou is an Assistant Professor affiliated with the Dept. of ECE at The University of Utah. Before joining the University of Utah, he received a Ph.D. in ECE in 2018 from The Ohio State University and worked as a post-doctoral fellow at Information Initiative at Duke University. His research interests include statistical machine learning, nonconvex & distributed optimization, deep learning, reinforcement learning and statistical signal processing."
+speaker_names: "Yi Zhou"
+source_file: "_data/talks/2022-01-14-yi-zhou.toml"
+generated: true
+---
+
+
+
+The past decade has witnessed the great success of deep learning in broad societal and commercial applications. However, conventional deep learning relies on wildly fitting data with neural networks, which is known to produce models that lack resilience. For instance, models used in facial recognition and healthcare are known to be biased toward people of a certain race or gender. Models used in autonomous driving are vulnerable to malicious attacks, i.e., putting an art sticker on a stop sign may force the model to classify it as a speed limit sign. Therefore, the next-generation deep learning paradigm aims to deliver resilient models that promote robustness to malicious attacks, fairness among users, and privacy preservation, and this can be realized by leveraging the emerging minimax machine learning framework. In this talk, I will present three gradient-descent-ascent (GDA) type of optimization algorithms for solving different classes of nonconvex minimax machine learning problems. Then, I will present a principled nonconvex minimax optimization theory that establishes the global convergence and convergence rates of these algorithms.
diff --git a/_talks/2022-01-21-chinmay-hedge.md b/_talks/2022-01-21-chinmay-hedge.md
new file mode 100644
index 0000000..9bb9e9c
--- /dev/null
+++ b/_talks/2022-01-21-chinmay-hedge.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Designing Neural Networks for Efficient Encrypted Inference"
+date: "2022-01-21 15:00:00 -0700"
+permalink: "/talks/2022-01-21-chinmay-hedge/"
+slug: "2022-01-21-chinmay-hedge"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Chinmay Hedge"
+ affiliation: "NYU"
+ bio: "Chinmay is a faculty member in the CSE and ECE Departments at the NYU Tandon School of Engineering. His research focuses on developing principled, fast, and robust algorithms for diverse problems in machine learning, with applications to imaging and computer vision, materials design, and transportation. Prior to NYU, he was an assistant professor in the Electrical and Computer Engineering Department at Iowa State University, and before that, a post-doctoral associate in the Theory of Computation (TOC) group at MIT, working with Piotr Indyk. He received his Ph.D. at Rice University under the supervision of Rich Baraniuk."
+speaker_names: "Chinmay Hedge"
+source_file: "_data/talks/2022-01-21-chinmay-hedge.toml"
+generated: true
+---
+
+
+
+As deep neural networks become ever more pervasive, so too are concerns surrounding users' data privacy. Curiously, standard cryptographic encryption approaches for guaranteeing data privacy do not interact well with traditional neural network models. In this talk, I will (a) outline why standard networks are not encryption-efficient, (b) suggest two new approaches for designing deep networks that do support efficient and secure inference, and (c) show results instantiating these approaches on real-world use cases.
diff --git a/_talks/2022-01-28-swaroop-mishra.md b/_talks/2022-01-28-swaroop-mishra.md
new file mode 100644
index 0000000..4fa0196
--- /dev/null
+++ b/_talks/2022-01-28-swaroop-mishra.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Towards the Development of Models that Learn New Tasks from Instructions"
+date: "2022-01-28 15:00:00 -0700"
+permalink: "/talks/2022-01-28-swaroop-mishra/"
+slug: "2022-01-28-swaroop-mishra"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147 (LCR)"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Swaroop Mishra"
+ affiliation: "ASU, Ph.D. Student"
+ bio: "Swaroop Mishra is a 3rd year Ph.D. student at Arizona State University. He finished his Masters from IIT Kanpur in 2016. He worked as a Software Engineer for 2 years at MathWorks and as a Technical Consultant for 1 year at the Information Technology Research Academy, Ministry of Electronics and Information Technology, Govt. of India before starting his Ph.D. He did a research internship with Allen AI and has been collaborating with them for the past year."
+speaker_names: "Swaroop Mishra"
+source_file: "_data/talks/2022-01-28-swaroop-mishra.toml"
+generated: true
+---
+
+
diff --git a/_talks/2022-02-04-tuhin-chakrabarty.md b/_talks/2022-02-04-tuhin-chakrabarty.md
new file mode 100644
index 0000000..5ebd053
--- /dev/null
+++ b/_talks/2022-02-04-tuhin-chakrabarty.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "The Curious Case of Figurative Language"
+date: "2022-02-04 15:00:00 -0700"
+permalink: "/talks/2022-02-04-tuhin-chakrabarty/"
+slug: "2022-02-04-tuhin-chakrabarty"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147 (LCR)"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Tuhin Chakrabarty"
+ affiliation: "Columbia University, Ph.D Student"
+ bio: "Tuhin is a Ph.D. student at Columbia University (based in NYC) advised by Smaranda Muresan and an Amazon Ph.D. fellow. Tuhin's interests are language understanding and generation especially non-literal languages, which require world knowledge or commonsense reasoning"
+speaker_names: "Tuhin Chakrabarty"
+source_file: "_data/talks/2022-02-04-tuhin-chakrabarty.toml"
+generated: true
+---
+
+
+
+Despite the ubiquity of figurative language across various forms of speech and writing, the vast majority of NLP research focuses primarily on literal language. Figurative language is challenging because of its implicit nature. In this talk, I will specifically focus on the following questions: 1) Can large language models understand/interpret them? 2) Can computers generate figurative language?
diff --git a/_talks/2022-02-11-vaggos-chatziafratis.md b/_talks/2022-02-11-vaggos-chatziafratis.md
new file mode 100644
index 0000000..6339094
--- /dev/null
+++ b/_talks/2022-02-11-vaggos-chatziafratis.md
@@ -0,0 +1,31 @@
+---
+layout: "talk"
+title: "Neural Networks Expressivity through the lens of Dynamical Systems"
+date: "2022-02-11 15:00:00 -0700"
+permalink: "/talks/2022-02-11-vaggos-chatziafratis/"
+slug: "2022-02-11-vaggos-chatziafratis"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Vaggos Chatziafratis"
+ affiliation: "UC Santa Cruz"
+ website: "https://cs.stanford.edu/~vaggos/"
+speaker_names: "Vaggos Chatziafratis"
+source_file: "_data/talks/2022-02-11-vaggos-chatziafratis.toml"
+generated: true
+---
+
+
+
+Given a target function f, how large must a neural network be in order to approximate f?
+Understanding the representational power of Deep Neural Networks (DNNs) and how their structural properties (e.g., depth, width, type of activation unit) affect the functions they can compute, has been an important yet challenging question in approximation theory and deep learning even in the early days of AI.
+
+In this talk, I want to tell you about some recent progress on this topic that uses ideas from dynamical systems. The main results are exponential depth-width trade-offs for DNNs representing certain families of functions. Our techniques rely on a generalized notion of fixed points, called periodic points that have played a major role in chaos theory (Li-Yorke chaos and Sharkovsky's theorem).
+
+Based on three recent works:
+- with Ioannis Panageas, Sai Ganesh Nagarajan and Xiao Wang from ICLR'20 (spotlight): https://arxiv.org/abs/1912.04378 (https://arxiv.org/abs/1912.04378)
+- with Ioannis Panageas and Sai Ganesh Nagarajan from ICML'20: https://arxiv.org/abs/2003.00777 (https://arxiv.org/abs/2003.00777)
+- with Clayton Sanford from AISTATS'22: https://arxiv.org/abs/2110.10295 (https://arxiv.org/abs/2110.10295)
diff --git a/_talks/2022-02-18-jeff-phillips.md b/_talks/2022-02-18-jeff-phillips.md
new file mode 100644
index 0000000..e88ddb0
--- /dev/null
+++ b/_talks/2022-02-18-jeff-phillips.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Some Very Basic Theory of Classification (in low dimensions"
+date: "2022-02-18 15:00:00 -0700"
+permalink: "/talks/2022-02-18-jeff-phillips/"
+slug: "2022-02-18-jeff-phillips"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "WEB 1250"
+canceled: false
+speakers:
+ - name: "Jeff Phillips"
+ affiliation: "Utah SoC"
+ website: "https://www.cs.utah.edu/~jeffp/"
+speaker_names: "Jeff Phillips"
+source_file: "_data/talks/2022-02-18-jeff-phillips.toml"
+generated: true
+---
+
+
+
+I plan to talk about some new results on some very fundamental (but overlooked) theory questions in classification. First, how fast can you find a linear classifier that eps-approximates the optimal one in terms of miss-classification? Second, if you want to preserve a Euclidean margin between perfectly classified points for a polynomial classifier, how many samples do you need?
diff --git a/_talks/2022-02-25-sunipa-dev.md b/_talks/2022-02-25-sunipa-dev.md
new file mode 100644
index 0000000..7643708
--- /dev/null
+++ b/_talks/2022-02-25-sunipa-dev.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Towards Inclusive and Socially Aware Language Technologies"
+date: "2022-02-25 15:00:00 -0700"
+permalink: "/talks/2022-02-25-sunipa-dev/"
+slug: "2022-02-25-sunipa-dev"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Sunipa Dev"
+ affiliation: "Google Research"
+ website: "https://sunipa.github.io"
+ bio: "Sunipa Dev is a Research Scientist on the Ethical AI team at Google RAI. Previously, she was an NSF Computing Innovation Fellow at UCLA, before which she completed her PhD at the University of Utah. Her ongoing research focuses on various facets of fairness and interpretability in NLP, including robust measurements of bias, cross-cultural understanding of concepts in NLP, and inclusive language representations."
+speaker_names: "Sunipa Dev"
+source_file: "_data/talks/2022-02-25-sunipa-dev.toml"
+generated: true
+---
+
+
+
+Large language models are commonly used in different paradigms of natural language processing and machine learning, and are known for their efficiency as well as their overall lack of interpretability. Their data driven approach for emulating human language often results in human biases being encoded and even amplified, potentially leading to cyclic propagation of representational and allocational harm. We discuss in this talk some aspects of detecting, evaluating, and mitigating biases and associated harms in a holistic, inclusive, and culturally-aware manner. In particular, we discuss the disparate impact on society of common language tools that are not inclusive of all gender identities.
diff --git a/_talks/2022-03-04-anirudh-goyal-of-montreal.md b/_talks/2022-03-04-anirudh-goyal-of-montreal.md
new file mode 100644
index 0000000..b4377bf
--- /dev/null
+++ b/_talks/2022-03-04-anirudh-goyal-of-montreal.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "From Specialists to Generalists: Inductive Biases of Deep Learning for Higher Level Cognition"
+date: "2022-03-04 15:00:00 -0700"
+permalink: "/talks/2022-03-04-anirudh-goyal-of-montreal/"
+slug: "2022-03-04-anirudh-goyal-of-montreal"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147 (LCR)"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Anirudh Goyal of Montreal"
+ website: "https://anirudh9119.github.io/"
+ bio: "Anirudh Goyal is a student of science advised by Prof. Yoshua Bengio. His current research interests center around understanding how neural learners can compose and abstract their own representations in a way that can be used to better generalize to out of distribution samples. More concretely, his work focuses on designing such models by incorporating in them strong but general assumptions (inductive biases) that enable high-level reasoning about the structure of the world. During his PhD, he has spent time as a visiting researcher at UC Berkeley, MPI Tuebingen and DeepMind. He was also one of the recipients of the 2021 Google PhD Fellowship in Machine Learning."
+speaker_names: "Anirudh Goyal of Montreal"
+source_file: "_data/talks/2022-03-04-anirudh-goyal-of-montreal.toml"
+generated: true
+---
+
+
+
+A fascinating hypothesis is that human and animal intelligence could be explained by a few principles (rather than an encyclopedic list of heuristics). If that hypothesis was correct, we could more easily both understand our own intelligence and build intelligent machines. Just like in physics, the principles themselves would not be sufficient to predict the behavior of complex systems like brains, and substantial computation might be needed to simulate human-like intelligence. This hypothesis would suggest that studying the kind of inductive biases that humans and animals exploit could help both clarify these principles and provide inspiration for AI research and neuroscience theories. Deep learning already exploits several key inductive biases, and my work considers a larger list, focusing on those which concern mostly higher-level and sequential conscious processing. The objective of clarifying these particular principles is that they could potentially help us build AI systems benefiting from humans' abilities in terms of flexible out-of-distribution and systematic generalization, which is currently an area where a large gap exists between state-of-the-art machine learning and human intelligence.
diff --git a/_talks/2022-03-25-chad-topaz-jude-higdon.md b/_talks/2022-03-25-chad-topaz-jude-higdon.md
new file mode 100644
index 0000000..8c1b533
--- /dev/null
+++ b/_talks/2022-03-25-chad-topaz-jude-higdon.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "Quantitative Approaches to Social Justice"
+date: "2022-03-25 15:00:00 -0600"
+permalink: "/talks/2022-03-25-chad-topaz-jude-higdon/"
+slug: "2022-03-25-chad-topaz-jude-higdon"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "WEB 1250"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Chad Topaz"
+ affiliation: "QSIDE"
+ website: "https://qsideinstitute.org/#"
+ - name: "Jude Higdon"
+ affiliation: "QSIDE"
+speaker_names: "Chad Topaz, Jude Higdon"
+source_file: "_data/talks/2022-03-25-chad-topaz-jude-higdon.toml"
+generated: true
+---
+
+
+
+Civil rights leader, educator, and investigative journalist Ida B. Wells said that "the way to right wrongs is to shine the light of truth upon them." This talk will demonstrate how mathematical, statistical, and computational approaches can shine a light on social injustices and help build solutions to remedy them. We will present research-to-action projects on diversity in art museums, inclusion in STEM, equity in criminal sentencing, and other topics. The tools engaged include crowdsourcing, data cleaning, clustering, hypothesis testing, statistical modeling, Markov chains, data visualization, and much more. Overall, we hope that this talk leaves you informed about the breadth of social justice applications that one can tackle using quantitative tools in careful collaboration with other scholars and activists.
diff --git a/_talks/2022-04-01-debanjan-mahata.md b/_talks/2022-04-01-debanjan-mahata.md
new file mode 100644
index 0000000..eee1d23
--- /dev/null
+++ b/_talks/2022-04-01-debanjan-mahata.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Identifying Keyphrases from Text Documents - From Heuristics to Language Models"
+date: "2022-04-01 15:00:00 -0600"
+permalink: "/talks/2022-04-01-debanjan-mahata/"
+slug: "2022-04-01-debanjan-mahata"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "\""
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Debanjan Mahata"
+ affiliation: "Moody Analytics"
+ bio: "Debanjan Mahata is a Director of AI at Moody's Analytics, New York, and leads the KYC machine learning team. He is also an Adjunct Faculty at the Department of Computer Science and Engineering, Indraprastha Institute of Information Technology (IIIT-Delhi). He closely collaborates with the Multimodal Digital Media Analysis lab (MIDAS@IIITD). His work lies at the intersection of natural language processing and information retrieval. He is currently interested in keyphrase extraction and generation, understanding code-switched text from social media, and computational social science problems solved using NLP. Before joining Moody's in August 2021, he was a Senior Research Scientist at Bloomberg AI (Nov 2017 - Jul 2021) and Senior Research Associate at Infosys Limited, Palo Alto, California (Aug 2015 - Oct 2017). He holds a Ph.D. in Integrated Computing from Donaghey College of Engineering and Information Technology, University of Arkansas at Little Rock."
+speaker_names: "Debanjan Mahata"
+source_file: "_data/talks/2022-04-01-debanjan-mahata.toml"
+generated: true
+---
+
+
+
+Automatic identification of keyphrases from text documents is an extreme summarization problem that lies at the intersection of the areas of natural language processing (NLP) and information retrieval (IR). Keyphrases aid in capturing the most salient topics from the input text and are useful in multiple downstream tasks such as classification, clustering, summarization, document recommendation, query expansion, interactive document retrieval, semantic and faceted search. Despite the ground-breaking advancements triggered by deep neural networks, automatically identifying keyphrases from text using machine learning techniques is still a challenging problem that hasn't been explored as much as other related and popular tasks such as named entity extraction, question answering, summarization. This talk will provide an overview of the advances made in keyphrase extraction and generation from text documents and present the approaches, datasets, evaluation strategies, and the associated challenges. It will also dive into the topic of how language models have been effective in pushing state-of-the-art performances in this domain. Lastly, the speaker will present the current trends and future directions for research in this domain.
diff --git a/_talks/2022-04-08-tao-li.md b/_talks/2022-04-08-tao-li.md
new file mode 100644
index 0000000..8940cf3
--- /dev/null
+++ b/_talks/2022-04-08-tao-li.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Improving Data Efficiency of Neural Models using Logic"
+date: "2022-04-08 15:00:00 -0600"
+permalink: "/talks/2022-04-08-tao-li/"
+slug: "2022-04-08-tao-li"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147 (LCR)"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Tao Li"
+ affiliation: "Google Research"
+ bio: "Tao Li is interested in on Natural Language Processing and Machine Learning. He is currently a Research Engineer at Google Research working. Earlier this year, He graduated as a PhD at the the U’s School of Computing where he was advised by Prof. Vivek Srikumar. He was also a MS graduate at the U back in 2014. Besides school, he had research internships at AI2, Amazon A9, and Philips Research."
+speaker_names: "Tao Li"
+source_file: "_data/talks/2022-04-08-tao-li.toml"
+generated: true
+---
+
+
+
+In this talk, we will focus on a simple approach that uses logic to improve neural model performance for natural language processing (NLP) tasks. Many downstream NLP tasks involve domain knowledge that can be easily stated in logical forms. We argue that we can use such knowledge to improve model learning. This results in better data efficiency, i.e., a model that performs better with less annotation. To this end, we propose frameworks that integrate domain knowledge, expressed as declarative constraints, with neural models. We show that such integration substantially improves state-of-the-art neural models in a variety of NLP tasks. To facilitate using our frameworks, we will also propose a PyTorch library that unifies differentiable tensor operations and logical operations.
diff --git a/_talks/2022-04-15-abhinav-kumar.md b/_talks/2022-04-15-abhinav-kumar.md
new file mode 100644
index 0000000..b65f4d5
--- /dev/null
+++ b/_talks/2022-04-15-abhinav-kumar.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Mathematical Modeling for Landmark and 3D Object Detection"
+date: "2022-04-15 15:00:00 -0600"
+permalink: "/talks/2022-04-15-abhinav-kumar/"
+slug: "2022-04-15-abhinav-kumar"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147 (LCR)"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Abhinav Kumar"
+ website: "https://sites.google.com/view/abhinavkumar/"
+ bio: "Abhinav is a Ph.D. student in computer science at Michigan State University working with Prof. Xiaoming Liu. His current research focus is mathematical modeling applied to 3D object detection for autonomous driving. Before joining the graduate program at MSU, he was at the University of Utah and worked at Xerox Research Center India, Bangalore. He holds a master's and bachelor's in electrical engineering from the Indian Institute of Technology (IIT) Bombay and IIT Patna, respectively."
+speaker_names: "Abhinav Kumar"
+source_file: "_data/talks/2022-04-15-abhinav-kumar.toml"
+generated: true
+---
+
+
+
+Modern computer vision models have excelled on several tasks and beaten several benchmarks. However, many of these models have avoided the mathematical and principled approaches to computer vision, resulting in sub-optimal performance and issues with interpretability. This talk will re-introduce mathematical modeling for computer vision tasks such as facial landmark localization and monocular 3D object detection for autonomous driving. In particular, we discuss the joint estimation of location, uncertainty, and visibility for facial landmark detection and introduce a mathematically differentiable NMS for monocular 3D detection. The mathematical modeling enables end-to-end learning in these tasks, resulting in improved performance.
diff --git a/_talks/2022-04-22-khyati-chandu.md b/_talks/2022-04-22-khyati-chandu.md
new file mode 100644
index 0000000..35ca993
--- /dev/null
+++ b/_talks/2022-04-22-khyati-chandu.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Anchoring Multimodal Narrative Generation"
+date: "2022-04-22 15:00:00 -0600"
+permalink: "/talks/2022-04-22-khyati-chandu/"
+slug: "2022-04-22-khyati-chandu"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "WEB 1250 and Zoom"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "Khyati Chandu"
+ affiliation: "Meta AI"
+ website: "https://www.cs.cmu.edu/~kchandu/"
+ bio: "Dr. Khyathi Chandu is a Research Scientist at Meta AI. Prior to this, she completed her Ph.D. at Carnegie Mellon University, working on controllable generation and multimodality. The avid goal of her research is to enable seamless communication between humans and machines with multiple modalities and languages. Her research focuses on improving vision-and-language generation, by adapting to appropriate content, structure, and persona, particularly in long-form generation. She has also done an array of work in code-switching and biomedical text summarization. She was selected as Rising Stars EECS 2020, won the sixth edition of the BioAsq challenge, organized several workshops at *CL conferences, and co-chaired D&I initiatives at NLP conferences and WiNLP workshops. She has consistently been on the Dean’s Merit list and was awarded the Best All-Rounder Student in her undergraduate. She is also an editor for the Machine Learning Blog at CMU and loves designing creative content in her free time."
+speaker_names: "Khyati Chandu"
+source_file: "_data/talks/2022-04-22-khyati-chandu.toml"
+generated: true
+---
+
+
+
+Humans inherently learn from and interact with multiple views of information, be it various modalities or languages. So, the expectations from contemporary and 21st-century technology are a testimony to the increasing need to model these multiview contexts better. Natural language generation plays a pivotal role in communicating these contexts in human-understandable languages. This talk brings together both of these transformative technologies to make strides toward a longstanding dream of human-like multiview narrative generation. The critical challenge is identifying the natural-sounding properties of long-form texts and modeling them in tandem with visual contexts. I present anchors for grounding three such properties including content (relevance), structure (coherence), and surface form realization (expression), and anchors them with relevant visual contexts. These anchors also provide us with human interpretable handles for controlling these properties.
+
+Details: To illustrate the effectiveness of the anchors for each of the three properties, I present: Starting with content: In situated multimodal contexts, relevance is the concept of the elements in one modality being connected to the other modality that makes this context informative and complementary. I present visual infilling with curriculum learning as a global objective for content and hierarchically attending over entity skeletons as a local objective for content, to generate visual stories and procedures. To improve the controllability and transferability in English and five other languages, I also introduce a dual-stage model with weakly supervised skeletons and a text-as-side attention mechanism to denoise the content in an image caption. Moving onto structure: The alignment of descriptions in language to the corresponding visual inputs is crucial to generating a logical and coherent narrative. I present a scaffolding technique as a local objective for structure by extracting a layout from vast amounts of unsupervised text to incorporate structure into cooking recipes generated from images. Finally, surface form: The crux of naturalness to automatic generation comes by incorporating individualized and personalized ways of expressing the same content. I present a locally guided, weakly supervised model for generating persona-based visual stories and a dual-staged adversarial technique to generate mixed views from non-parallel data. All of the above work mainly focuses on static multimodal narratives, and finally, I present a case to highlight the significance of transitioning to dynamic grounding. I conclude by presenting the shortcomings of the current approaches in the NLP domain to the grounding problem and offer recommendations along with executable actions for course correction to bridge this gap and enable grounding for machines to resemble human communication.
diff --git a/_talks/2022-04-29-james-brundage.md b/_talks/2022-04-29-james-brundage.md
new file mode 100644
index 0000000..4e0b88f
--- /dev/null
+++ b/_talks/2022-04-29-james-brundage.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Leveraging Unlabeled Data for Machine Learning in the Electrocardiogram"
+date: "2022-04-29 15:00:00 -0600"
+permalink: "/talks/2022-04-29-james-brundage/"
+slug: "2022-04-29-james-brundage"
+start_time: "3:00 PM"
+end_time: "4:00 PM"
+series: "Data Science Seminar"
+location: "MEB 3147 (LCR)"
+zoom: "https://utah.zoom.us/j/94615915833?pwd=TW50Mk5GeFpQN0lBMEZQQ2Z1ZUdFUT09"
+canceled: false
+speakers:
+ - name: "James Brundage"
+ affiliation: "UU HSC"
+ bio: "James Brundage is a second year medical student (MSII) at the University of Utah. He completed a BS and MS in neuroscience at BYU with an emphasis in cellular neuro-electrophysiology. Since starting at the University of Utah, he has worked in the MacLeod lab, coordinating and carrying out a collaborative effort between the Scientific Computing and Imaging Institute (SCII), Nora Eccles Cardiovascular Research and Training Institute (CVRTI) and the cardiology team at University of Utah Hospital focused on analysis of the electrocardiogram (ECG) using machine learning. Following medical school, he hopes to pursue a career as a physician scientist with a focus on applied machine learning in the clinical setting."
+speaker_names: "James Brundage"
+source_file: "_data/talks/2022-04-29-james-brundage.toml"
+generated: true
+---
+
+
+
+Supervised deep learning (DL) has become an increasingly common tool for advanced analysis of the electrocardiogram (ECG). These methods rely heavily on labeled datasets, in which there is a clinical annotation for each ECG. However, real world ECG datasets may not contain enough labeled recordings to facilitate robust feature extraction, preventing DL analysis for clinical problems with small datasets. Self-supervised learning (SSL) seeks to utilize cheaply labeled or unlabeled data to improve performance in a supervised learning task. This process consists of first training a model on a primary task with cheap data labels, followed by a second training process which attempts to learn the downstream task by initializing with weights learned from the first. While SSL has become a popular tool in many machine learning domains, it is only starting to be used in ECG based machine learning. Here, we demonstrate the progress we have made in applying SSL approaches to detect low left ventricular ejection fraction, a complex ECG detection task, using data extracted from the University of Utah.
diff --git a/_talks/2022-08-24-bei-wang-phillips.md b/_talks/2022-08-24-bei-wang-phillips.md
new file mode 100644
index 0000000..dd5d15b
--- /dev/null
+++ b/_talks/2022-08-24-bei-wang-phillips.md
@@ -0,0 +1,34 @@
+---
+layout: "talk"
+title: "On Hypergraph Analysis and Visualization"
+date: "2022-08-24 10:30:00 -0600"
+permalink: "/talks/2022-08-24-bei-wang-phillips/"
+slug: "2022-08-24-bei-wang-phillips"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Bei Wang Phillips"
+ affiliation: "Utah SoC/SCI"
+ website: "http://www.sci.utah.edu/~beiwang/"
+speaker_names: "Bei Wang Phillips"
+source_file: "_data/talks/2022-08-24-bei-wang-phillips.toml"
+generated: true
+---
+
+
+
+Hypergraphs capture multi-way relationships in data, and they have
+consequently seen a number of applications in higher-order network
+analysis, computer vision, geometry processing, and machine learning.
+In this talk, I will discuss hypergraph analysis and visualization.
+In particular, I will focus on developing the theoretical foundations
+in studying the space
+of hypergraphs using ingredients from optimal transport.
+
+This talk is
+based on joint works
+with Youjia Zhou, Archit Rathore, Emilie Purvine, Samir Chowdhury, Tom
+Needham, and Ethan Semrad.
diff --git a/_talks/2022-08-31-casey-greene.md b/_talks/2022-08-31-casey-greene.md
new file mode 100644
index 0000000..88fb7cc
--- /dev/null
+++ b/_talks/2022-08-31-casey-greene.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "TBA"
+date: "2022-08-31 10:30:00 -0600"
+permalink: "/talks/2022-08-31-casey-greene/"
+slug: "2022-08-31-casey-greene"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Casey Greene"
+ affiliation: "CU Anschutz"
+speaker_names: "Casey Greene"
+source_file: "_data/talks/2022-08-31-casey-greene.toml"
+generated: true
+---
+
+
diff --git a/_talks/2022-09-07-kevin-moon.md b/_talks/2022-09-07-kevin-moon.md
new file mode 100644
index 0000000..6601731
--- /dev/null
+++ b/_talks/2022-09-07-kevin-moon.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Scalable supervised manifold learning with random forests and neural networks"
+date: "2022-09-07 10:30:00 -0600"
+permalink: "/talks/2022-09-07-kevin-moon/"
+slug: "2022-09-07-kevin-moon"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Kevin Moon"
+ affiliation: "USU"
+ website: "https://sites.google.com/a/umich.edu/kevin-r-moon/home"
+ bio: "Kevin Moon is an assistant professor at Utah State University in the department of mathematics and statistics. He received his B.S. and M.S. in Electrical Engineering from BYU while focusing on signal processing with minors in economics and math. He then received an M.S. in Mathematics and a PhD in Electrical Engineering from the University of Michigan where he worked with Dr. Alfred Hero on the problem of nonparametric estimation of distributional functionals. Prior to joining USU in 2018, he worked with Dr. Smita Krishnaswamy and Dr. Ronald Coifman as a postdoc at Yale University in the Genetics department and the Applied Math program where he developed methods for data visualization and exploratory data analysis with a focus in biomedical applications. His current research focuses on the development of theory and applications in machine learning, big data, information theory, deep learning, and data science in general. Applications of interest include biology (including medical), finance, ecology, engineering, and navigation."
+speaker_names: "Kevin Moon"
+source_file: "_data/talks/2022-09-07-kevin-moon.toml"
+generated: true
+---
+
+
+
+The manifold assumption has been used in many machine learning applications to combat the curse of dimensionality. Most manifold learning methods are unsupervised and typically focus on preserving the dominant structure and variation in the data. In many cases, we wish to analyze the data in a supervised setting with respect to expert-provided data labels. Most supervised manifold learning methods exaggerate the separation between data points of different classes, distorting the true structure of the data. In this talk, I will present RF-PHATE, a supervised dimensionality reduction method that preserves the true structure of the variables that are relevant for the supervised task. RF-PHATE is based upon a diffusion process applied to random forest proximities and is well-suited for data visualization. I will then show how to improve the scalability of RF-PHATE and any other manifold learning algorithm and perform out of sample extension using geometry regularized autoencoders (GRAE).
diff --git a/_talks/2022-09-14-elliot-smith.md b/_talks/2022-09-14-elliot-smith.md
new file mode 100644
index 0000000..d6de9a8
--- /dev/null
+++ b/_talks/2022-09-14-elliot-smith.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Human neuronal population encoding of temporal difference learning variables during risky choices"
+date: "2022-09-14 10:30:00 -0600"
+permalink: "/talks/2022-09-14-elliot-smith/"
+slug: "2022-09-14-elliot-smith"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Elliot Smith"
+ affiliation: "Utah Neurology"
+ website: "http://neurosmiths.org"
+speaker_names: "Elliot Smith"
+source_file: "_data/talks/2022-09-14-elliot-smith.toml"
+generated: true
+---
+
+
+
+Recent research in AI showed that agents designed to predict the full distribution of potential rewards, rather than a central estimate of that distribution, generate richer learning distributions that allow them to perform better, especially on risky tasks. Such distributional reinforcement learning (distRL) was also discovered in dopamine neurons in the rodent ventral tegmental area. In this nanosymposium presentation, I will discuss recent work from direct brain recordings in neurosurgical patients undergoing monitoring for treatment of medically refractory epilepsy who performed a risky decision making task called the Balloon Analog Risk Task. Results from two studies will be presented: In the first study, we examined neuronal population recordings (157 neurons) from microelectrodes implanted in the anterior cingulate, orbitofrontal and temporal cortices (15 participants), finding that human prefrontal and mesial temporal neurons exhibited signatures of distRL: correlated diverse optimism in reward coding and diverse asymmetric scaling of reward prediction error. In the second study, we examined correlations between broadband high frequency local field potentials (an established correlate of population neuronal firing) and variables from temporal difference learning models for reward and risk while 37 participants made risky choices during BART. We found differences in which brain areas (3199 stereoelectroencephalography or electrocorticography contacts sampling frontal, temporal, and parietal lobes) encoded temporal difference learning model variables between participants who were more or less risk averse in their choices during BART. These areas included the left dorsolateral prefrontal, anterior cingulate, and orbitofrontal cortices. The results from these studies shed light on the neural underpinnings of human value learning in uncertain environments.
diff --git a/_talks/2022-09-21-jes-ford.md b/_talks/2022-09-21-jes-ford.md
new file mode 100644
index 0000000..efe0a49
--- /dev/null
+++ b/_talks/2022-09-21-jes-ford.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Model Review: Improving Transparency, Reproducibility, & Knowledge Sharing using MLflow"
+date: "2022-09-21 10:30:00 -0600"
+permalink: "/talks/2022-09-21-jes-ford/"
+slug: "2022-09-21-jes-ford"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Jes Ford"
+ affiliation: "Cash App"
+ website: "http://jesford.github.io"
+ bio: "Jes Ford is a sponsored snowboarder turned astrophysicist turned data scientist. She enjoys applying Python data science tools to a wide variety of problems, and teaching skills and best practices to others. Jes completed her PhD in Physics at UBC Vancouver in 2015, and did a Postdoc in Data Science at the University of Washington under Jake VanderPlas. Currently based in Salt Lake City, she works remotely for Cash App (Block) as a Senior Machine Learning Engineer, and previously held local data science positions at Recursion and Backcountry. Jes spends her free time exploring the Wasatch mountains on snowboard, mountain bike, and foot. She has been involved in organizing the Salt Lake PyLadies chapter and the local Women in Data Science Conference, and is always looking for fun ways to be a part of her local tech community."
+speaker_names: "Jes Ford"
+source_file: "_data/talks/2022-09-21-jes-ford.toml"
+generated: true
+---
+
+
+
+Code Review is an integral part of software development, but many teams don’t have similar processes in place for the development and deployment of Machine Learning (ML) models. I will motivate the decision to create a Model Review process, starting from the principles of transparency, reproducibility, and knowledge sharing. MLflow is a useful Python package to help simplify and automate much of the tracking necessary to create detailed records of machine learning experiments. Much of this talk will be spent introducing this tool, and demonstrating the core MLflow Tracking functionality. I’ll discuss how my team is currently running a Model Review process for any ML models that we push to production, and how we use MLflow to streamline this work and learn from each other.
diff --git a/_talks/2022-09-28-prashant-pandey.md b/_talks/2022-09-28-prashant-pandey.md
new file mode 100644
index 0000000..5bbe221
--- /dev/null
+++ b/_talks/2022-09-28-prashant-pandey.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Scalability Challenges in Large-Scale Sequence Search"
+date: "2022-09-28 10:30:00 -0600"
+permalink: "/talks/2022-09-28-prashant-pandey/"
+slug: "2022-09-28-prashant-pandey"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Prashant Pandey"
+ affiliation: "Utah SoC"
+ website: "https://prashantpandey.github.io"
+speaker_names: "Prashant Pandey"
+source_file: "_data/talks/2022-09-28-prashant-pandey.toml"
+generated: true
+---
+
+
+
+Sequence-level searches on large collections of RNA sequencing experiments, such as the NCBI Sequence Read Archive (SRA), would enable one to ask many questions about the expression or variation of a given transcript in a population. Building an efficient and scalable sequence search index at the scale of SRA data is a challenging task and requires fundamental innovations in compression and scalable indexing. Recently, several tools have been proposed to index and search through SRA data but they offer various trade-offs in terms of space, speed, updatability, and accuracy. In this talk, I will present Mantis, a fast, exact, and updatable sequence search index. Mantis uses recent advancements in fast and compact hash tables, domain-specific data compression techniques, and scalable indexing to build a scalable and updatable index and supports fast sequence searches on ~40K experiments (>100TB) in size from SRA.
diff --git a/_talks/2022-10-05-jessica-shi.md b/_talks/2022-10-05-jessica-shi.md
new file mode 100644
index 0000000..ca5cfe8
--- /dev/null
+++ b/_talks/2022-10-05-jessica-shi.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Theoretically and Practically Efficient Parallel Nucleus Decomposition"
+date: "2022-10-05 10:30:00 -0600"
+permalink: "/talks/2022-10-05-jessica-shi/"
+slug: "2022-10-05-jessica-shi"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Jessica Shi"
+ affiliation: "MIT"
+ website: "https://jeshi96.github.io"
+ bio: "Jessica Shi is a 4th-year PhD student at MIT in the EECS department, where she is advised by Julian Shun. She is also a Student Researcher at Google on the Graph Mining team, where she is mentored by Jakub Łącki. Jessica’s current research interests include developing shared-memory parallel graph algorithms with provable theoretical guarantees and efficient scalable implementations, with a focus on subgraph decomposition and clustering algorithms. She is supported by a 2018 NSF Graduate Research Fellowship, and she has previously received her S.M. in computer science from MIT, and her A.B. in mathematics from Princeton University."
+speaker_names: "Jessica Shi"
+source_file: "_data/talks/2022-10-05-jessica-shi.toml"
+generated: true
+---
+
+
+
+We study the nucleus decomposition problem, which has been shown to be useful in finding dense substructures in graphs. We present a novel parallel algorithm that is efficient both in theory and in practice. Our algorithm achieves a work complexity matching the best sequential algorithm while also having low depth (parallel running time), which significantly improves upon the only existing parallel nucleus decomposition algorithm (Sariyuce et al., PVLDB 2018). The key to the theoretical efficiency of our algorithm is a new lemma that bounds the amount of work done when peeling cliques from the graph, combined with the use of theoretically-efficient parallel algorithms for clique listing and bucketing.
+
+We introduce several new practical optimizations, including a new multi-level hash table structure to store information on cliques space-efficiently and a technique for traversing this structure cache-efficiently. On a 30-core machine with two-way hyper-threading on real-world graphs, we achieve up to a 55x speedup over the state-of-the-art parallel nucleus decomposition algorithm by Sariyuce et al., and up to a 40x self-relative parallel speedup. We are able to efficiently compute larger nucleus decompositions than prior work on several million-scale graphs for the first time.
diff --git a/_talks/2022-10-19-jie-zhang.md b/_talks/2022-10-19-jie-zhang.md
new file mode 100644
index 0000000..1a85b5e
--- /dev/null
+++ b/_talks/2022-10-19-jie-zhang.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Active Sampling for Min-Max Fairness"
+date: "2022-10-19 10:30:00 -0600"
+permalink: "/talks/2022-10-19-jie-zhang/"
+slug: "2022-10-19-jie-zhang"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Jie Zhang"
+ affiliation: "U Washington"
+ bio: "Claire Zhang is a PhD student at the Paul G. Allen School of Computer Science and Engineering at the University of Washington. She is fortunate to be advised by Professor Jamie Morgenstern. She completed her undergraduate studies at the University of Utah School of Computing, and was very fortunate to be advised by professor Suresh Venkatasubramanian while at the U. Her research interests include fairness in machine learning and learning theory."
+speaker_names: "Jie Zhang"
+source_file: "_data/talks/2022-10-19-jie-zhang.toml"
+generated: true
+---
+
+
+
+Models satisfying Min-max fairness minimizes maximum group specific losses, so that the model has a more equitable performance over all groups. Benefit of min-max fair models over other fairness notions and models include it levels up: meaning it only degrades performance of a group if the degradation improves performance on the worst off group.
+
+In this talk, I will briefly mention the prior works that defined min-max fairness. I will mainly focus on our paper “Active Sampling for Min-Max Fairness” to introduce two algorithms that find min-max fair models with convergence guarantees.
diff --git a/_talks/2022-11-02-alex-chin.md b/_talks/2022-11-02-alex-chin.md
new file mode 100644
index 0000000..c0afd0b
--- /dev/null
+++ b/_talks/2022-11-02-alex-chin.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Quantifying supply-demand imbalance in ridesharing systems"
+date: "2022-11-02 10:30:00 -0600"
+permalink: "/talks/2022-11-02-alex-chin/"
+slug: "2022-11-02-alex-chin"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Alex Chin"
+ affiliation: "Lyft"
+ website: "http://alexchin.com"
+speaker_names: "Alex Chin"
+source_file: "_data/talks/2022-11-02-alex-chin.toml"
+generated: true
+---
+
+
+
+The status of the rider and driver distributions and how they interact in a two-sided marketplace has implications for market efficiency and policy-making. As such, accurately characterizing the supply-demand state of Lyft's marketplace is a crucial task. I will describe how this task can be framed in terms of an asymmetric optimal transport problem that yields a multi-resolution view of the supply-demand state and how it varies both spatially and temporally. I will then discuss how this approach can be incorporated into applications such as policy optimization and machine learning prediction problems. Time permitting I will also discuss other science efforts we have at Lyft.
diff --git a/_talks/2022-11-09-shireen-elhabian.md b/_talks/2022-11-09-shireen-elhabian.md
new file mode 100644
index 0000000..fc3a811
--- /dev/null
+++ b/_talks/2022-11-09-shireen-elhabian.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Data-driven Shape Analysis: Methods, Applications, and Future"
+date: "2022-11-09 10:30:00 -0700"
+permalink: "/talks/2022-11-09-shireen-elhabian/"
+slug: "2022-11-09-shireen-elhabian"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Shireen Elhabian"
+ affiliation: "Utah CS/SCI"
+ website: "http://www.sci.utah.edu/~shireen/"
+speaker_names: "Shireen Elhabian"
+source_file: "_data/talks/2022-11-09-shireen-elhabian.toml"
+generated: true
+---
+
+
+
+Quantitative analysis of shapes is contingent upon defining a metric in the space of shapes to compare shapes and perform shape statistics. A growing consensus in the field that such a metric should be adapted to the specific population under investigation, begging for learning such a metric in a data-driven manner. This talk will cover a state-of-art data-driven approach for statistical shape modeling that provides unbiased, objective, and intuitive evaluation of geometric shapes, emphasizing anatomical structures reconstructed from volumetric images. I will talk about how we applied shape analysis to clinical and scientific questions in various ways. I will also highlight the role of machine learning in mitigating critical bottlenecks and significant barriers to making shape modeling a robust tool for on-demand clinical diagnostics and streamlining its adoption in research and practice. I will end with a future outlook for shape modeling to enable more complex and diverse modeling scenarios.
diff --git a/_talks/2022-11-16-bernadette-stolz.md b/_talks/2022-11-16-bernadette-stolz.md
new file mode 100644
index 0000000..ad42e93
--- /dev/null
+++ b/_talks/2022-11-16-bernadette-stolz.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Applications of global and local persistent homology for the shape of biological data"
+date: "2022-11-16 10:30:00 -0700"
+permalink: "/talks/2022-11-16-bernadette-stolz/"
+slug: "2022-11-16-bernadette-stolz"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Bernadette Stolz"
+ affiliation: "Oxford"
+ website: "https://www.maths.ox.ac.uk/people/bernadette.stolz"
+ bio: "Bernadette obtained her DPhil in 2020 from the Mathematical Institute at the University of Oxford. She is now a Postdoctoral Researcher at the Laboratory for Topology and Neuroscience at EPFL and a Visiting Research Fellow at the Mathematical Institute, University of Oxford. Previously she was a Postdoctoral Research Assistant at the Centre for Topological Data Analysis at the University of Oxford. In her research, she develops techniques in topological data analysis (TDA) to study biological data, in particular dynamical networks and spatial data. Her research can be broadly categorised into three main groups: 1) Developing TDA techniques to answer biological questions arising from experimental data. 2) Developing novel data science methods based on TDA. 3) Using TDA in combination with mechanistic models to link form and function in biological systems. Her expertise in TDA is complemented by an MSc in Mathematical Modelling and Scientific Computing (University of Oxford, 2014) and undergraduate degrees in Mathematics (University of Bern, 2012) and Molecular Medicine (University of Göttingen, 2009). Her research has been recognised with the L'Oréal-Unesco For Women in Science Rising Talent Award 2022, the Anile-ECMI Prize for best PhD thesis 2021, and the Mathematical Institute (University of Oxford) 2020 DPhil Thesis Prize."
+speaker_names: "Bernadette Stolz"
+source_file: "_data/talks/2022-11-16-bernadette-stolz.toml"
+generated: true
+---
+
+
+
+In the first part of this talk, I will showcase how persistent homology can be used to spatially characterise structural abnormality in tumour blood vessel networks. More specifically, I will show that the number of vessel loops and their distribution in these networks change over time when tumours undergo treatment with vascular targeting agents and radiation therapy. In the second part of the talk, I will speak about applications of local persistent homology. I will show how local persistent homology can be used to select landmarks from large and noisy data sets. In contrast to existing methods, this subsampling process is robust to outliers and is developed specifically for persistent homology. I will further introduce a novel method that can detect geometric anomalies, such as intersections or boundaries, in point cloud data sampled from intersecting surfaces. This detection is based on the computation of persistent homology in local annular neighbourhoods around points and is less sensitive to the size of the local neighbourhood and surface curvature than an existing method.
diff --git a/_talks/2022-11-30-aaron-quinlan.md b/_talks/2022-11-30-aaron-quinlan.md
new file mode 100644
index 0000000..2f7fcfa
--- /dev/null
+++ b/_talks/2022-11-30-aaron-quinlan.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Finding signals of genome mutation in the noise of DNA sequencing error"
+date: "2022-11-30 10:30:00 -0700"
+permalink: "/talks/2022-11-30-aaron-quinlan/"
+slug: "2022-11-30-aaron-quinlan"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Aaron Quinlan"
+ affiliation: "Utah Human Genetics"
+ website: "http://quinlanlab.org"
+speaker_names: "Aaron Quinlan"
+source_file: "_data/talks/2022-11-30-aaron-quinlan.toml"
+generated: true
+---
+
+
+
+The research in our laboratory is focused on the application of computational methods to develop a deeper understanding of genetic variation in diverse contexts. Modern experimental methods allow us to examine entire genomes with exquisite detail. Perhaps not surprisingly, staggering complexity is revealed as we look more closely at how genetic variation (both inherited and somatic) contributes to phenotypes. Modern genomic technologies necessitate efficient approaches for exploring, manipulating and comparing large genomic datasets. We develop such methods so that we and others may apply them to experiments investigating the impact of genetic variation on human disease, evolution, and somatic differentiation. Genome research is difficult - we strive to develop computational means that make it easier.
diff --git a/_talks/2022-12-07-tao-yang.md b/_talks/2022-12-07-tao-yang.md
new file mode 100644
index 0000000..eebd9cc
--- /dev/null
+++ b/_talks/2022-12-07-tao-yang.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "Optimizing Ranking Effectiveness and Fairness"
+date: "2022-12-07 10:30:00 -0700"
+permalink: "/talks/2022-12-07-tao-yang/"
+slug: "2022-12-07-tao-yang"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Tao Yang"
+ affiliation: "Utah SoC"
+ website: "https://www.cs.utah.edu/~taoyang/"
+ bio: "Tao Yang is fourth year Ph.D. student from the University of Utah, supervised by Prof. Qingyao Ai and Prof. Jeff M Phillips. He mainly focuses on Information Retrieval (IR) and Machine Learning related topics. Especially, he mainly focuses on how to construct and optimize ranking systems while considering ranking effectiveness and fairness. His works have been published on top-tier IR conferences and journals, like SIGIR, WWW, CIKM,WSDM,TOIS,...."
+speaker_names: "Tao Yang"
+source_file: "_data/talks/2022-12-07-tao-yang.toml"
+generated: true
+---
+
+
+
+Advanced ranking techniques have led to improvements in AI-powered information services that significantly changed people's lives. For example, search engines that rank information according to their utilities to use's queries have helped billions of people better finish their tasks in daily work; recommendation systems that rank products/movies/news according to the user's interests have completely changed the way people discover information everyday. Therefore, how to construct and optimize ranking systems is one of the most important research problems in the field of Information Retrieval (IR). When optimizing ranking systems, there are two important criteria to measure the quality of result rankings in IR systems. The first criterion is ranking effectiveness, which refers to the ability of a ranking system to effectively present results based on their relevance to the users' needs. The second criterion is ranking fairness, which refers to the ability of a ranking system to present results fairly. For example, in job recommendation, if a ranking system only considers ranking effectiveness and ranks items solely according to relevance, a small number of top candidates will always be exposed to users and dominate users' attention as users usually only examine the top ranks. In such case, other candidates will rarely have the chance to be hired even when they are highly qualified for the job. Therefore, it is important to balance the effectiveness of ranked lists with the fairness in ranking optimization.
+In this talk, I will present my recent works on ranking effectiveness and fairness optimization. The talk will be two parts. In the first part of this talk, I will introduce works on sole-effectiveness optimization where I propose uncertainty-aware rank systems based on Bayes modelling. In the second part of this talk, I will introduce works on fairness-effectiveness joint optimization.
diff --git a/_talks/2023-01-11-george-vega-yon.md b/_talks/2023-01-11-george-vega-yon.md
new file mode 100644
index 0000000..14f5936
--- /dev/null
+++ b/_talks/2023-01-11-george-vega-yon.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Prediction of Gene Functions by Leveraging Biological Insights with Mechanistic Machine Learning"
+date: "2023-01-11 10:45:00 -0700"
+permalink: "/talks/2023-01-11-george-vega-yon/"
+slug: "2023-01-11-george-vega-yon"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "George Vega Yon"
+ affiliation: "Utah Epidemiology"
+ website: "https://ggvy.cl/"
+ bio: "Dr. George G. Vega Yon is a research assistant professor of epidemiology at the University of Utah. Dr. Vega Yon is a methodologist that uses statistical computing tools to study complex systems. His research includes statistical models for social network analysis, agent-based models, and phylogenetics. Dr. Vega Yon has over ten years of experience working in data science creating scientific software, including multiple R packages in data visualization, network science, high-performance computing, and bayesian statistics. George is originally from Chile and obtained his Ph.D. in biostatistics from USC, an M.Sc. in Social Science from Caltech, and an M.A. in economics and public policy from Universidad Adolfo Ibáñez. (https://ggv.cl)"
+speaker_names: "George Vega Yon"
+source_file: "_data/talks/2023-01-11-george-vega-yon.toml"
+generated: true
+---
+
+
+
+Biomedical sciences, in particular, bioinformaticians and computational biologists, are in a race to annotate the immense number of genes and gene products we are still learning from. In this talk, I present a new method for predicting gene functions using mechanistic machine learning. Mechanistic machine learning is a relatively new area of research where predictive ML-based algorithms are improved by incorporating domain knowledge via mechanistic models. Here, we use a theoretically-funded function evolution model that relies solely on phylogenetic trees from PantherDB and annotations from the Gene Ontology (GO) to make high-quality predictions. I will illustrate how combining the mentioned model with a large gene expression database (Bgee) in an ML model significantly improves prediction quality.
diff --git a/_talks/2023-01-18-echo-warner.md b/_talks/2023-01-18-echo-warner.md
new file mode 100644
index 0000000..f89a064
--- /dev/null
+++ b/_talks/2023-01-18-echo-warner.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "College of Nursing, Division of Acute and Chronic Care"
+date: "2023-01-18 10:45:00 -0700"
+permalink: "/talks/2023-01-18-echo-warner/"
+slug: "2023-01-18-echo-warner"
+start_time: "10:45 AM"
+end_time: "12:00 PM"
+series: "Data Science Seminar"
+location: "Virtual:"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Echo Warner"
+ affiliation: "Utah Nursing, HCI"
+ website: "https://faculty.utah.edu/u0600488-ECHO_LYN_WARNER/hm/index.hml"
+speaker_names: "Echo Warner"
+source_file: "_data/talks/2023-01-18-echo-warner.toml"
+generated: true
+---
+
+
+
+Unproven health claims on the internet may have substantial influence on patient behaviors and decision making. For example, cancer patients with curable disease who pursue unproven cancer treatment in lieu of evidence-based approaches demonstrate 2-4 times higher mortality than patients who avoid unproven cancer treatment. Interventions to mitigate the impact of online health misinformation are desperately needed. The few interventions that have been tested do not consider the extent of exposure to misinformation online, primarily because individual estimates of exposure are based on self-report and are considered highly unreliable. Web-monitoring software is typically used by businesses to monitor remote employee productivity. Our paradigm shifting approach applies web-monitoring to quantify online cancer misinformation exposure. We will discuss the feasibility of using web-monitoring software to quantify exposure to online health information with a special focus on cancer symptom management and unproven cancer treatment misinformation. Specifically, we will review 1) characteristics of online cancer health misinformation and the impacts this exposure may have on cancer patient health outcomes, relationships, and finances 2) Ethical considerations of web-monitoring, and 3) methodological rigor and reproducibility of web-monitoring approaches for studying exposure to other types of health misinformation online.
diff --git a/_talks/2023-01-25-shweta-jain.md b/_talks/2023-01-25-shweta-jain.md
new file mode 100644
index 0000000..e01e74f
--- /dev/null
+++ b/_talks/2023-01-25-shweta-jain.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Putting Parameterization into Practice"
+date: "2023-01-25 10:45:00 -0700"
+permalink: "/talks/2023-01-25-shweta-jain/"
+slug: "2023-01-25-shweta-jain"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Shweta Jain"
+ affiliation: "Utah SoC"
+ website: "https://sjain12.github.io"
+ bio: "Shweta Jain is a Computing Innovation Fellow (CIFellow) at the University of Utah working with Prof. Blair D. Sullivan. She was previously a postdoc at the University of Illinois, Urbana-Champaign and prior to that she completed her Ph.D. in Computer Science at the University of California, Santa Cruz (UCSC), advised by Prof. Seshadhri Comandur. Her research interests are in graph mining, parameterized algorithms, randomized and approximation algorithms, and algorithms for massive data and the goal of her research is to use these tools to design algorithms that work well in practice and have provable guarantees. Her work has been recognized by two Best Paper Awards, the SIGKDD Best Dissertation Runner-Up Award 2021, and the Best Dissertation Award 2020 of the Computer Science Department at UCSC. She was also chosen as a Rising Star of EECS 2020."
+speaker_names: "Shweta Jain"
+source_file: "_data/talks/2023-01-25-shweta-jain.toml"
+generated: true
+---
+
+
+
+Graphs are everywhere: social networks, protein interaction networks, citation networks, epidemic spread networks. Applications that use these graphs have progressively become more sophisticated, relying on getting fast and accurate solutions to graph-theoretic problems, many of which are NP-Hard. However, the sizes of today's graphs easily run into millions of vertices and edges, if not more. As a result, many classic algorithms are infeasible for such graphs.
+
+Fortunately, real-world graphs across different domains show a lot of common characteristics such as an abundance of triangles, low average distance between vertices (small-world property), low degeneracy (a measure of the sparsity of edges) etc. which we can leverage to design algorithms that are provably efficient given those parameters. I will demonstrate this in the context of clique counting and decomposition of graphs, which has applications in community detection, spam detection, fraud detection, gene module detection etc. The new algorithms not only massively improve the performance in practice but also improve our theoretical understanding of real-world graphs and help to bridge the gap between theory and practice.
diff --git a/_talks/2023-02-01-ana-marsovic.md b/_talks/2023-02-01-ana-marsovic.md
new file mode 100644
index 0000000..4cb99f9
--- /dev/null
+++ b/_talks/2023-02-01-ana-marsovic.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "“AI” That Masters Language Could Reason about Negation"
+date: "2023-02-01 10:45:00 -0700"
+permalink: "/talks/2023-02-01-ana-marsovic/"
+slug: "2023-02-01-ana-marsovic"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Ana Marsovic"
+ affiliation: "Utah SoC"
+ website: "https://www.anamarasovic.com"
+ bio: "Ana Marasović is an Assistant Professor in the Kahlert School of Computing at the University of Utah. Her primary research interests are at the confluence of NLP, explainable AI, and multimodality. She aims to rigorously validate AI technologies and make human interaction with AI more intuitive. She was a Young Investigator at the Allen Institute for AI from 2019–2022. During that time, she also had a courtesy appointment in the Paul G. Allen School of Computer Science & Engineering at the University of Washington. She obtained her PhD in 2019 from Heidelberg University.\n\nName pronunciation: Ah-nah Mara-so-veetch, with “Mara” as the actress “Mara Wilson”\n"
+speaker_names: "Ana Marsovic"
+source_file: "_data/talks/2023-02-01-ana-marsovic.toml"
+generated: true
+---
+
+
+
+We have experienced firsthand the growing impact that AI technologies like ChatGPT have. It is agreed upon that risks involving such technologies must be managed. This is at a glance akin to how people handle safety-critical systems in, e.g., aviation. However, the rigorous principles of safety engineering are not easily applicable to AI technologies. AI-backed solutions are obscured in AI’s internals, and the exact requirements for AI safety are unverifiable since they are neither defined nor regulated. One technique that has emerged as a step to measure an aspect of AI safety is to construct data that represents a specific phenomenon that a safe model must handle well, and that has no spurious correlations. Low performance on the dataset is undesired.In this talk, I’ll focus on one common linguistic phenomenon, negation, without which the full power of human language-based communication cannot be realized. I will show how we carefully constructed a question-answering dataset, CondaQA, to study how well current models (described by the New York Time Magazine as “mastering language”) reason about negated statements. An InstructGPT model (the latest GPT model before Nov 28, 2022) combined with chain-of-thought prompting achieves 66.28% accuracy and 27.28% consistency on CondaQA, way behind human accuracy of 91.94% and consistency of 81.58%.
diff --git a/_talks/2023-02-08-aaron-clauset.md b/_talks/2023-02-08-aaron-clauset.md
new file mode 100644
index 0000000..dcf8a9e
--- /dev/null
+++ b/_talks/2023-02-08-aaron-clauset.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Meritocracy or systemic bias? Untangling the drivers the productivity and prominence among scientists"
+date: "2023-02-08 10:45:00 -0700"
+permalink: "/talks/2023-02-08-aaron-clauset/"
+slug: "2023-02-08-aaron-clauset"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Aaron Clauset"
+ affiliation: "UC Boulder"
+ website: "https://www.colorado.edu/cs/aaron-clauset"
+ bio: "Aaron Clauset is a Professor in the Department of Computer Science and the BioFrontiers Institute at the University of Colorado Boulder, and is External Faculty at the Santa Fe Institute. He received a PhD in Computer Science, with distinction, from the University of New Mexico, a BS in Physics, with honors, from Haverford College, and was an Omidyar Fellow at the prestigious Santa Fe Institute. In 2016, he was awarded the Erdos-Renyi Prize in Network Science, and since 2017, he has been a Deputy Editor responsible for the Social, Computing, and Interdisciplinary Sciences at Science Advances.\n\nClauset is an internationally recognized expert on network science, data science, and machine learning for complex systems. His research program is around two general themes: identifying fundamental principles of the organization and behavior of complex social and biological systems, and developing approaches for using data and computation to illuminate those ideas. A recent major focus of this work has been on the \"science of science,\" where he studies the shape, origins, and consequences of social and epistemic inequalities on scientific careers, productivity, the spread of ideas, and the composition of the scientific workforce. His research results have appeared in many prestigious scientific venues, including Nature, Science, PNAS, SIAM Review, Science Advances, Nature Communications, AAAI, and ICDM. His work has been covered in the popular press by Quanta Magazine, the Wall Street Journal, The Economist, Discover Magazine, Wired, the Boston Globe and The Guardian.\n"
+speaker_names: "Aaron Clauset"
+source_file: "_data/talks/2023-02-08-aaron-clauset.toml"
+generated: true
+---
+
+
+
+Simple measures of scholarly productivity and prominence vary enormously across both individual scientists and institutions -- but to what degree do these inequalities represent genuine meritocratic differences vs. systemic biases that limit scientific progress?
+
+In this talk, I'll describe a sequence of results that substantially untangle the underlying systemic drivers of productivity and prominence among scientists. First, I'll show that productivity and prominence are, to a significant degree, environmental variables such that the prestige of a scientist's working environment drives their individual productivity, largely by providing larger research groups to elite scientists. Second, I'll describe a network-based generative model of individual productivity and prominence that untangles these measures from their underlying collaboration networks. These models corroborate the labor-advantage hypothesis of elite institutions, and also reveal both that gendered differences in the productivity and prominence of mid-career researchers can be largely explained by gendered differences in coauthorship networks, and that these networks are partially transferable from senior to junior collaborators. Hence, collaboration networks, and the systemic factors that shape them, play a critical role in driving scholarly inequalities in science, and suggest that these networks are an important form of unequally distributed social capital that shapes who makes what scientific discoveries. I'll close with a discussion of policies that could potentially mitigate the unequal distribution of this social capital and help both diversify the academy and broaden its contributions to society.
diff --git a/_talks/2023-02-15-orly-alter.md b/_talks/2023-02-15-orly-alter.md
new file mode 100644
index 0000000..30f6f1c
--- /dev/null
+++ b/_talks/2023-02-15-orly-alter.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Solving Cancer with Data: Mathematical Discovery and Computational and Experimental Validation of Whole-Genome Genotype–Survival and Response to Treatment Phenotype Relationships in Cancer"
+date: "2023-02-15 10:45:00 -0700"
+permalink: "/talks/2023-02-15-orly-alter/"
+slug: "2023-02-15-orly-alter"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "WEB 3780"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Orly Alter"
+ affiliation: "Utah BME, SCI"
+ website: "https://alterlab.org/"
+speaker_names: "Orly Alter"
+source_file: "_data/talks/2023-02-15-orly-alter.toml"
+generated: true
+---
+
+
+
+1/2 of men and 1/3 of women will face cancer, a disease of the whole 3B-nucleotide genome. But, despite the availability of open-source data and the $100/1-hour genome, genetic tests remain limited to one to a few hundred genes. Therefore, the prognosis, diagnosis, and treatment of cancer remain unchanged. This is due to the lack of suitable AI/ML. I will describe work in my lab inventing AI/ML that connects the whole genome with a patient’s survival and response to treatment. Our algorithms discover accurate, precise, and interpretable predictors, applicable to the general population, from as few as 50–100 patients. Our predictors outperform all other indicators, where they exist. All other methods miss them. I will describe my international retrospective clinical trial, which validated a genome-wide pattern in tumors from glioblastoma brain cancer patients as the best predictor of life expectancy and response to standard of care. We discovered this, and predictors in, e.g., adult lung, ovarian, and uterine adenocarcinoma tumors and pediatric nerve neuroblastoma tumors, in public data, proving that the algorithms and predictors are uniquely suited to personalized medicine. I will also describe work translating the algorithms and predictors to the clinic.
diff --git a/_talks/2023-02-22-vivek-gupta.md b/_talks/2023-02-22-vivek-gupta.md
new file mode 100644
index 0000000..28572bf
--- /dev/null
+++ b/_talks/2023-02-22-vivek-gupta.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Inference and Reasoning for Semi-structured Tables"
+date: "2023-02-22 10:45:00 -0700"
+permalink: "/talks/2023-02-22-vivek-gupta/"
+slug: "2023-02-22-vivek-gupta"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Vivek Gupta"
+ affiliation: "Utah SoC"
+ website: "https://vgupta123.github.io"
+ bio: "Vivek is a fifth-year doctorate candidate at the Utah NLP Group's at Kahlert School of Computing, University of Utah. He is fortunate to be advised by Prof. Vivek Srikumar. He is broadly interested in NLP research in semi-structured data and low-resource languages. He is awarded Bloomberg Data Science Fellowship 2021-23, the Best paper award at the DeeLIO 2022 workshop, and the Outstanding paper award at the NLP4ConvAI 2022 workshop . He currently also serves as the Utah Data Science Club's coordinator. He used to be a Research Fellow (Microsoft Research Fellowship 2016–18) at the Microsoft Research Lab, India, where he worked with the Machine Learning and Natural Language Processing group. In 2016, he graduated from IIT Kanpur as a dual degree (BS-MS) student in the Department of Computer Science and Engineering. He was the inaugural coordinator of IIT Kanpur's Special Interest Group in Machine Learning (SIGML)"
+speaker_names: "Vivek Gupta"
+source_file: "_data/talks/2023-02-22-vivek-gupta.toml"
+generated: true
+---
+
+
+
+Understanding semi-structured tabular data, which is ubiquitous in the real world, requires an understanding of the meaning of text fragments and the implicit connections between them. We believe such data could be used to investigate how individuals and machines reason about semi-structured data. First, we present the InfoTabS dataset, which consists of human-written textual predictions based on tables collected from Wikipedia's infoboxes. Our research demonstrates that the semi-structured, multi-domain, and heterogeneous nature of the premises prompts complicated, multi-faceted reasoning, offering a modeling challenge for traditional modeling techniques. Second, we analyzed these challenges in-depth and developed simple, effective preprocessing strategies to overcome them. Thirdly, despite accurate NLI prediction, we demonstrate through rigorous probing that the existing model does not reason with the provided tabular facts. To address this, we suggest a two-stage evidence extraction and tabular inference technique for enhancing model reasoning and interpretability. We also investigate efficient methods for enhancing tabular inference datasets with semi-automatic data augmentation and pattern-based pre-training. Lastly, to ensure that tabular reasoning models work in more than one language, we introduce XInfoTabS, a unique problem of bilingual tabular inference, and a cost-effective pipeline for translating tables. In the near future, we plan to test the tabular reasoning model for temporal changes, especially for dynamic tables where information changes over time.
diff --git a/_talks/2023-03-01-emily-hadley.md b/_talks/2023-03-01-emily-hadley.md
new file mode 100644
index 0000000..e529f3b
--- /dev/null
+++ b/_talks/2023-03-01-emily-hadley.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Applied Strategies for Advancing Racial Equity and Addressing Bias in Big Data Research"
+date: "2023-03-01 10:45:00 -0700"
+permalink: "/talks/2023-03-01-emily-hadley/"
+slug: "2023-03-01-emily-hadley"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Emily Hadley"
+ affiliation: "RTI International"
+ bio: "Emily Hadley is a Research Data Scientist with the RTI International Center for Data Science. Her work spans several practice areas including health, education, social policy, and criminal justice. She has experience with machine learning, natural language processing, and predictive analytics, and a passion for antiracism, bias, and equity in data science."
+speaker_names: "Emily Hadley"
+source_file: "_data/talks/2023-03-01-emily-hadley.toml"
+generated: true
+---
+
+
+
+In the last decade, big data research studies have proliferated and, in some cases, offered considerable promise for humanity. Yet, numerous incidents have documented that without safeguards, this same research can reproduce and amplify existing societal biases and disparities. Given the increased public awareness of structural racism, bias based on race and ethnicity in big data research is particularly concerning. Big data researchers can mitigate the risk of perpetuating bias based on race and ethnicity by intentionally incorporating best practices that reduce or eliminate racial biases. We synthesize key findings and applied recommendations from over 140 sources for addressing race and ethnicity bias in big data research. We discuss considerations when planning a big data project, including identifying study motivations, centering participatory involvement, and addressing key concerns regarding race and ethnicity in big data collection and quality. We detail issues related to proxy discrimination, algorithmic audits, data completeness, and the Big Data Paradox. We provide real-world examples of advancing racial equity and addressing bias and share recommendations for additional resources and opportunities for further investigation.
diff --git a/_talks/2023-03-15-nate-veldt.md b/_talks/2023-03-15-nate-veldt.md
new file mode 100644
index 0000000..edd7e83
--- /dev/null
+++ b/_talks/2023-03-15-nate-veldt.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Measuring homophily in group interactions: hypergraph models and combinatorial impossibilities"
+date: "2023-03-15 10:45:00 -0600"
+permalink: "/talks/2023-03-15-nate-veldt/"
+slug: "2023-03-15-nate-veldt"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Nate Veldt"
+ affiliation: "Texas A&M"
+ website: "https://veldt.engr.tamu.edu"
+speaker_names: "Nate Veldt"
+source_file: "_data/talks/2023-03-15-nate-veldt.toml"
+generated: true
+---
+
+
+
+Homophily is the well-known sociological principle that people tend to connect with others who are similar to them, or more informally: "birds of a feather flock together." Although many social interactions occur in groups, homophily is typically measured using a graph, which only accounts for interactions involving two individuals. This talk will present a new hypergraph framework for more directly measuring homophily in group settings. Our measures highlight natural patterns in group homophily that appear with gender in scientific collaboration and political affiliation in legislative bill cosponsorship, and also reveal distinctive gender distributions in group photographs, all of which cannot be fully captured by graph-based measures. We will also discuss subtle combinatorial limits and impossibilities that arise when measuring homophily in hypergraphs, which are completely independent of human behavior and must be properly accounted for in order to understand how homophily can (and cannot) be manifested in group interactions.
diff --git a/_talks/2023-03-22-bailey-fosdick.md b/_talks/2023-03-22-bailey-fosdick.md
new file mode 100644
index 0000000..17970df
--- /dev/null
+++ b/_talks/2023-03-22-bailey-fosdick.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Modeling Infection Fatality Rates to Assess the Burden of COVID-19 in Developing Countries"
+date: "2023-03-22 10:45:00 -0600"
+permalink: "/talks/2023-03-22-bailey-fosdick/"
+slug: "2023-03-22-bailey-fosdick"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Bailey Fosdick"
+ affiliation: "CU Anschutz"
+ website: "https://www.baileyfosdick.com"
+speaker_names: "Bailey Fosdick"
+source_file: "_data/talks/2023-03-22-bailey-fosdick.toml"
+generated: true
+---
+
+
+
+COVID-19 spread quickly around the world after first being discovered in China in late 2019. It has had devastating impacts, however its impacts, both in terms of infection prevalence and fatalities, have been non-uniformly distributed worldwide. While early studies focused on COVID-19 infection and fatality rates in high-income countries, much less attention has been given to the impacts of COVID-19 in developing countries. In this work, we systematically reviewed the literature to identify all COVID-19 serology studies conducted by early 2021 using population representative samples. We developed a Bayesian hierarchical model for simultaneously modeling serology and death data to make inference on age-specific seroprevalence and age-specific infection fatality rates. This model directly accounts for conventional sampling uncertainty, as well as uncertainty about the serological test assay sensitivity and specificity. Through a careful analysis of data from over thirty developing countries, we found seroprevalence in many developing country locations was markedly higher than in high-income countries early in the pandemic and age-specific infection fatality rates were roughly twice as high as that in high-income countries.
diff --git a/_talks/2023-03-29-justin-baker.md b/_talks/2023-03-29-justin-baker.md
new file mode 100644
index 0000000..2000fa0
--- /dev/null
+++ b/_talks/2023-03-29-justin-baker.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Monotone Implicit Graph Neural Networks for Long-Range Dependency Learning"
+date: "2023-03-29 10:45:00 -0600"
+permalink: "/talks/2023-03-29-justin-baker/"
+slug: "2023-03-29-justin-baker"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Justin Baker"
+ affiliation: "Utah Math & SCI"
+speaker_names: "Justin Baker"
+source_file: "_data/talks/2023-03-29-justin-baker.toml"
+generated: true
+---
+
+
+
+From social networks to chemical engineering, deep graph neural networks play an instrumental role in advancing our industrial and scientific frontier. Of particular interest are networks which can learn long range dependencies in a scalable and expressive manner. In this talk, we will delve into the power of deep learning on graphs, with a focus on implicit graph neural networks (IGNNs) and their scalability. We will also discuss how monotone operator theory enhances the expressivity of IGNNs, overcoming a crucial obstacle to learning long-range dependencies. By doing so, monotone IGNNs can significantly improve graph learning and have the potential to make breakthroughs in various fields.
diff --git a/_talks/2023-04-05-yao-yaun-mao.md b/_talks/2023-04-05-yao-yaun-mao.md
new file mode 100644
index 0000000..a5cc167
--- /dev/null
+++ b/_talks/2023-04-05-yao-yaun-mao.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Searching for dwarf (small) galaxies in astronomical surveys"
+date: "2023-04-05 10:45:00 -0600"
+permalink: "/talks/2023-04-05-yao-yaun-mao/"
+slug: "2023-04-05-yao-yaun-mao"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Yao-Yaun Mao"
+ affiliation: "Utah Astro"
+ website: "https://yymao.github.io"
+speaker_names: "Yao-Yaun Mao"
+source_file: "_data/talks/2023-04-05-yao-yaun-mao.toml"
+generated: true
+---
+
+
+
+Dwarf galaxies are small fuzzy galaxies that contain much fewer stars than Milky Way-mass galaxies. Observations of these little galaxies can enhance our understanding of galaxy formation and the nature of dark matter. Finding these dwarf galaxies is, however, not an easy task because they are faint and dim from our perspective. I will describe the challenges and recent efforts on the search for nearby dwarf galaxies in astronomical surveys. One particular challenge is to identify potential dwarf galaxies with only image data that do not contain distance information. I will discuss a few traditional methods that are used to identify dwarf galaxies and obtain their astronomical distances, and why these methods are mostly used to find dwarf galaxies in specific patches of the sky. I will then discuss how the distance information can be used as training data in machine learning algorithms, such as convolutional neural networks, to derive distance information from just astronomical images. Finally, I will discuss the remaining challenges we have, and how we may improve the methods in preparation for future observations from the Rubin Observatory Legacy Survey of Space and Time (LSST) and Roman Space Telescope.
diff --git a/_talks/2023-04-12-titus-brown.md b/_talks/2023-04-12-titus-brown.md
new file mode 100644
index 0000000..897711c
--- /dev/null
+++ b/_talks/2023-04-12-titus-brown.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Is everything everywhere all at once? Asking questions of all public microbiome shotgun data"
+date: "2023-04-12 10:45:00 -0600"
+permalink: "/talks/2023-04-12-titus-brown/"
+slug: "2023-04-12-titus-brown"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+location: "MEB 3147"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Titus Brown"
+ affiliation: "UC Davis"
+ website: "http://ivory.idyll.org/lab/"
+ bio: "C. Titus Brown is a Professor at the School of Veterinary Medicine at UC Davis, where he works on effective large scale sequencing data analysis, methods development, training, and open science. His most recent work focuses enabling on petabase-scale search of all available public microbiome data, with the goal of better hypothesis generation and refinement. He tweets at @ctitusbrown and blogs at http://ivory.idyll.org/blog/. His google scholar profile is reasonably up to date but is nevertheless highly misleading."
+speaker_names: "Titus Brown"
+source_file: "_data/talks/2023-04-12-titus-brown.toml"
+generated: true
+---
+
+
+
+Public sequence data offers many opportunities for reuse, exploration, and discovery. What happens if you make it really, really easy to search the content of all the public microbiome data sets? It turns out you can enable some interesting science, but you also need to address many technical, social, and policy issues. In this talk I’ll showcase some of our results from being able to search everything, everywhere, all at once; describe our current efforts; and discuss the possible opportunities and challenges of petabyte-scale sequence search.
diff --git a/_talks/2023-04-19-sumana-basu.md b/_talks/2023-04-19-sumana-basu.md
new file mode 100644
index 0000000..a0cc2ed
--- /dev/null
+++ b/_talks/2023-04-19-sumana-basu.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Towards Reinforcement Learning for Precision Drug Dosing"
+date: "2023-04-19 10:45:00 -0600"
+permalink: "/talks/2023-04-19-sumana-basu/"
+slug: "2023-04-19-sumana-basu"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/92180148411?pwd=dG1OaXlZTTQ1d0M4R0RiSUpsb3kvdz09"
+canceled: false
+speakers:
+ - name: "Sumana Basu"
+ affiliation: "McGill University"
+ website: "https://scholar.google.com/citations?view_op=view_org&hl=en&org=13784427342582529234"
+ bio: "Sumana is a PhD student at McGill University (Mila), researching Deep Reinforcement Learning in Healthcare. Her work focuses on applying Reinforcement Learning to Autonomous Drug Dosing. She completed her Masters at Mila, where she studied deep learning for predicting Alzheimer's disease progression. She has also interned in the past at Meta AI (FAIR) on the fastMRI Active Acquisition project, using Reinforcement Learning to accelerate MRI acquisition, and will be interning at Microsoft Research Labs coming summer, on exploring Reinforcement Learning for optimizing genetic perturbation in Cancer Treatment."
+speaker_names: "Sumana Basu"
+source_file: "_data/talks/2023-04-19-sumana-basu.toml"
+generated: true
+---
+
+
+
+Drug dosing is an important application of AI, which can be formulated as a Reinforcement Learning (RL) problem, since every individual’s drug dosing requirement is different. In this talk, we will talk about two major challenges of using RL for drug dosing: delayed and prolonged effects of medications, which break the Markov assumption of the RL framework. We will talk about an approach to solve this problem in a model free reinforcement learning setting, talk further about the challenges of deploying it in real life and sketch the outline of a more realistic Model Based Reinforcement Learning (MBRL) solution to it.
diff --git a/_talks/2023-05-30-bodhisattwa-majumder.md b/_talks/2023-05-30-bodhisattwa-majumder.md
new file mode 100644
index 0000000..b8e6586
--- /dev/null
+++ b/_talks/2023-05-30-bodhisattwa-majumder.md
@@ -0,0 +1,29 @@
+---
+layout: "talk"
+title: "User-centric Natural Language Processing"
+date: "2023-05-30 15:00:00 -0600"
+permalink: "/talks/2023-05-30-bodhisattwa-majumder/"
+slug: "2023-05-30-bodhisattwa-majumder"
+start_time: "3:00 PM"
+end_time: "4:30 PM"
+series: "Data Science Seminar"
+location: "Where: MEB 3147 Large Conference Room Simcast: (Meeting ID: 965 3805 0936, Passcode: 404653)"
+zoom: "https://utah.zoom.us/j/96538050936"
+canceled: false
+speakers:
+ - name: "Bodhisattwa Majumder"
+ affiliation: "UCSD"
+ website: "http://nlp.cs.utah.edu/"
+ bio: "Bodhisattwa Prasad Majumder (https://www.majumderb.com/) recently received his Ph.D. in Computer Science from UC San Diego and was advised by Prof. Julian McAuley. His research goal is to build safe, trustworthy, and user-centric interactive systems. He previously spent time at the Allen Institute of AI, Google AI, Microsoft Research, and FAIR (Meta AI), along with collaborations from U of Oxford, U of British Columbia, and the Alan Turing Institute. His work has been recognized by the UCSD CSE Doctoral Award for Research, Adobe Research Fellowship, Qualcomm Innovation Fellowship, and Highlights of ACM RecSys, among many awards and several media coverages. In 2019, Bodhi led UCSD in the finals of the Amazon Alexa Prize. He also co-authored a best-selling NLP book with O’Reilly Media that is being adopted in universities internationally."
+speaker_names: "Bodhisattwa Majumder"
+source_file: "_data/talks/2023-05-30-bodhisattwa-majumder.toml"
+generated: true
+---
+
+
+
+Artificial intelligence (AI) has shown remarkable effectiveness in knowledge-seeking applications (e.g., for recommendations and explanations). However, the increasing expectation of more trust, accessibility, and anthropomorphism in these AI systems requires the underlying components (dialog models, LLMs, classifiers) to be adaptive and adequately knowledge grounded. In reality, the outputs of the constituent models often lack commonsense, explanations, and subjectivity, which motivates us to ask the question: what can we achieve by redesigning AI systems to start with individual needs?
+
+Ideally, an assistive AI system must be aware of the surrounding world, produce faithful explanations, and align with the user's preferences. In this talk, I will discuss a post-hoc knowledge-injection technique that enriches the dialog responses at the decoding time and promotes achieving conversational goals. Then, I will explore how to elevate existing AI systems using a unified framework to map low-level and abstractive explanations by background knowledge. Finally, I will hint at a user-centric interventionist approach that can help users obtain more equitable predictions backed by faithful explanations as compared to a black-box counterpart. I will conclude with the future possibilities and societal impacts of next-generation user-centric systems.
+
+bodhi_flyer.png
diff --git a/_talks/2023-08-23-data-science-lecture-series.md b/_talks/2023-08-23-data-science-lecture-series.md
new file mode 100644
index 0000000..fdd4907
--- /dev/null
+++ b/_talks/2023-08-23-data-science-lecture-series.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Zoom link: https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+date: "2023-08-23 10:30:00 -0600"
+permalink: "/talks/2023-08-23-data-science-lecture-series/"
+slug: "2023-08-23-data-science-lecture-series"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "Warnock Engineering Building (Room: 3780), 72 Central Campus Dr, Salt Lake City, UT 84112, USA"
+zoom: "https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series"
+speaker_names: "Data Science Lecture Series"
+source_file: "_data/talks/2023-08-23-data-science-lecture-series.toml"
+generated: true
+---
+
+
diff --git a/_talks/2023-08-30-data-science-lecture-series-speaker.md b/_talks/2023-08-30-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..ebdc396
--- /dev/null
+++ b/_talks/2023-08-30-data-science-lecture-series-speaker.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Improving Fairness of Information Access in Networks"
+date: "2023-08-30 10:30:00 -0600"
+permalink: "/talks/2023-08-30-data-science-lecture-series-speaker/"
+slug: "2023-08-30-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ bio: "Blair D. Sullivan is a Professor in the School of Computing at the University of Utah. Prior to joining Utah, Dr. Sullivan was an Associate Professor at NC State University, and before that a Research Scientist at Oak Ridge National Laboratory. She received her Ph.D. in Mathematics from Princeton University in 2008 as a Department of Homeland Security Graduate Fellow, and B.S. degrees in Applied Mathematics and Computer Science from Georgia Tech in 2003. Sullivan’s research cross-cuts the fields of data-driven science, parameterized graph algorithms, network science, and algorithm engineering with a recent focus on problems arising in computational genomics. In 2014, Sullivan was named one of 14 Moore Investigators in Data-Driven Discovery. She currently serves on the Steering Committee for SODA, and was recently elected Chair of the SIAM SIAG on Applied & Computational Discrete Algorithms."
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-08-30-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+In social networks, node position is a form of social capital which enables faster and more reliable access to diverse information. Structural biases often arise from network formation and can lead to significant disparities in information access based on position. We discuss ways to quantify this social capital through the lens of information flow in the network, focusing on the setting where each node may be a source of distinct desirable information. We define several measures of access advantage, and consider the problem of improving equity by making interventions in the network, focusing on the case of edge augmentation. We describe several heuristic strategies for budgeted intervention, and present the results of an empirical evaluation on a corpus of real-world networks.
diff --git a/_talks/2023-08-30-data-science-lecture-series.md b/_talks/2023-08-30-data-science-lecture-series.md
new file mode 100644
index 0000000..cbf8000
--- /dev/null
+++ b/_talks/2023-08-30-data-science-lecture-series.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Zoom link: https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+date: "2023-08-30 10:30:00 -0600"
+permalink: "/talks/2023-08-30-data-science-lecture-series/"
+slug: "2023-08-30-data-science-lecture-series"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+zoom: "https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series"
+speaker_names: "Data Science Lecture Series"
+source_file: "_data/talks/2023-08-30-data-science-lecture-series.toml"
+generated: true
+---
+
+
diff --git a/_talks/2023-09-06-data-science-lecture-series-speaker.md b/_talks/2023-09-06-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..2ece7c0
--- /dev/null
+++ b/_talks/2023-09-06-data-science-lecture-series-speaker.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "Blame the data, not the system: how data constraints can help in trustworthy machine learning and explain causes of data-system malfunction"
+date: "2023-09-06 10:30:00 -0600"
+permalink: "/talks/2023-09-06-data-science-lecture-series-speaker/"
+slug: "2023-09-06-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-09-06-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+The core of modern data-driven systems comprises models learned from large datasets, and they are usually optimized to target particular data and workloads. While these data-driven systems have seen wide adoption and success, their reliability and proper function hinge on the data's continued conformance to the systems initial settings and assumptions. My research focuses on designing mechanisms to assess the trustworthiness of a system's inferences and explain causes of system malfunction due to data nonconformance. The key idea here is that since data is central to data-driven systems, it can guide us to determine whether predictions made by an ML model can be trusted, and to expose the cause of a system's unexpected behavior. In this talk, I will talk about mechanisms and explanation frameworks to facilitate trusting and understanding outcomes involving data and data systems.
diff --git a/_talks/2023-09-13-data-science-lecture-series-speaker.md b/_talks/2023-09-13-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..0ec58f2
--- /dev/null
+++ b/_talks/2023-09-13-data-science-lecture-series-speaker.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Dynamic Graph Sketching: To Infinity And Beyond"
+date: "2023-09-13 10:30:00 -0600"
+permalink: "/talks/2023-09-13-data-science-lecture-series-speaker/"
+slug: "2023-09-13-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ bio: "David is the 2023 Grace Hopper Postdoctoral Fellow at Lawrence Berkeley Lab and his research focuses on compact, dynamic, and memory-hierarchy-aware algorithms for large-scale data science. Prior to that, he was a CRA Computing Innovation Postdoctoral Fellow working with Martin Farach-Colton at Rutgers University. He earned his PhD at UMass Amherst working with Andrew McGregor."
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-09-13-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+Existing graph stream processing systems must store the graph explicitly in RAM which limits the scale of graphs they can process. The graph semi-streaming literature offers algorithms which avoid this limitation via linear sketching data structures that use small (sublinear) space, but these algorithms have not seen use in practice to date. In this talk I will explore what is needed to make graph sketching algorithms practically useful, and as a case study present a sketching algorithm for connected components and a corresponding high-performance implementation. Finally, I will give an overview of the many open problems in this area, focusing on improving query performance of graph sketching algorithms.
diff --git a/_talks/2023-09-20-data-science-lecture-series-speaker.md b/_talks/2023-09-20-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..afae98d
--- /dev/null
+++ b/_talks/2023-09-20-data-science-lecture-series-speaker.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "Data Preparation: The Biggest Roadblock in Data Science"
+date: "2023-09-20 10:30:00 -0600"
+permalink: "/talks/2023-09-20-data-science-lecture-series-speaker/"
+slug: "2023-09-20-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-09-20-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+When building Machine learning (ML) models, data scientists face a significant hurdle: data preparation. ML models are exactly as good as the data we train them on. Unfortunately, data preparation is tedious and laborious because it often requires human judgment on how to proceed. In fact, data scientists spend at least 80% of their time locating the datasets they want to analyze, integrating them together, and cleaning the result.In this talk, I will present my key contributions in data preparation for data science, which address the following problems: (1) data discovery: how to discover data of interest from a large collection of heterogeneous tables (e.g., data lakes); (2) error detection: how to find errors in the input and intermediate data in complex data workflows; and (3) data repairing: how to repair data errors with minimal human intervention. The developed systems are specifically designed to support data science development which poses particular requirements such as interactivity and modularity.
diff --git a/_talks/2023-09-25-sandia-information-session.md b/_talks/2023-09-25-sandia-information-session.md
new file mode 100644
index 0000000..2dbc00b
--- /dev/null
+++ b/_talks/2023-09-25-sandia-information-session.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "TBA"
+date: "2023-09-25 11:00:00 -0600"
+permalink: "/talks/2023-09-25-sandia-information-session/"
+slug: "2023-09-25-sandia-information-session"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science Seminar"
+location: "WEB 2460"
+canceled: false
+speakers:
+ - name: "Sandia Information Session"
+ affiliation: "pizza"
+ website: "https://tinyurl.com/2244r966"
+speaker_names: "Sandia Information Session"
+source_file: "_data/talks/2023-09-25-sandia-information-session.toml"
+generated: true
+---
+
+
diff --git a/_talks/2023-09-27-data-science-lecture-series-speaker.md b/_talks/2023-09-27-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..3e7165a
--- /dev/null
+++ b/_talks/2023-09-27-data-science-lecture-series-speaker.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Loss Minimization and Multi-group Fairness"
+date: "2023-09-27 10:30:00 -0600"
+permalink: "/talks/2023-09-27-data-science-lecture-series-speaker/"
+slug: "2023-09-27-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ website: "https://parikg.github.io/"
+ bio: "Parikshit Gopalan (https://parikg.github.io/) is a machine learning researcher at Apple (https://machinelearning.apple.com/). His current interests are in fairness in machine learning, unsupervised learning and algorithms/systems for big data. In the past, he has made important contributions to erasure coding for distributed storage, coding theory and computational complexity. His work has been awarded the 2014 Joint IEEE Communication Society & Information Theory Society Paper Prize, 2013 Microsoft TCN Storage Technical Award and the best paper award for the 2012 USENIX Advanced Technology Conference. In the past, he has been a researcher at VMware, Microsoft Research (Silicon valley and Redmond), a postdoc at the University of Washington and UT Austin, a graduate student at Georgia Tech and an undergraduate at IIT Bombay."
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-09-27-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+Training a predictor to minimize a loss function fixed in advance is the dominant paradigm in machine learning. However, loss minimization by itself might not guarantee desiderata like fairness and accuracy that one could reasonably expect from a predictor. In contrast, various group-fairness notions have been propsoed that constrain the predictor to share certain statistical properties of the data, even when conditioned on a rich family of subgroups. There is no explicit attempt at loss minimization.In this talk, we will explore some recently discovered connections between loss minimization and notions of multi-group fairness. We will see settings where one can lead to the other, and other settings where this is unlikely.
diff --git a/_talks/2023-10-04-data-science-lecture-series-speaker.md b/_talks/2023-10-04-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..3451ee2
--- /dev/null
+++ b/_talks/2023-10-04-data-science-lecture-series-speaker.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Leveraging the Structure of Data"
+date: "2023-10-04 10:30:00 -0600"
+permalink: "/talks/2023-10-04-data-science-lecture-series-speaker/"
+slug: "2023-10-04-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ affiliation: "such as NeurIPS, ICML, ICLR, KDD, and WWW"
+ website: "https://ai.google/research/teams/algorithms-optimization/"
+ bio: "Bryan Perozzi is a Research Scientist in Google Research’s Algorithms and Optimization (https://ai.google/research/teams/algorithms-optimization/) group, where he routinely analyzes some of the world’s largest (and perhaps most interesting) graphs. Bryan’s research (https://scholar.google.com/citations?hl=en&user=rZgbMs4AAAAJ&view_op=list_works) focuses on developing techniques for learning expressive representations of relational data with neural networks. These scalable algorithms are useful for prediction tasks (classification/regression), pattern discovery, and anomaly detection in large networked data sets.\n\nBryan is an author of 40+ peer-reviewed papers at leading conferences in machine learning and data mining (such as NeurIPS, ICML, ICLR, KDD, and WWW). His doctoral work on learning network representations was awarded the prestigious SIGKDD Dissertation Award. Bryan received his Ph.D. in Computer Science from Stony Brook University in 2016, and his M.S. from the Johns Hopkins University in 2011.\n"
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-10-04-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+Although predictions from machine learning models influence more and more of our lives, the standard way of posing a ML problem has remained relatively unchanged for decades. In the search for better models, a new and popular family of techniques (sometimes called Graph Machine Learning) has emerged. These techniques rely on expanding beyond the features of an individual entity and instead look to pull information from its relationships. The methods offer a tantalizing way of improving task performance by leveraging previously unused information. However, it is not a free lunch, as these models can be more complex, difficult to train, and may have challenges in interpretability. This talk will discuss the fundamentals of graph machine learning, a few models, and some insights from years of real-world applications.
diff --git a/_talks/2023-10-18-data-science-lecture-series-speaker.md b/_talks/2023-10-18-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..42b4b92
--- /dev/null
+++ b/_talks/2023-10-18-data-science-lecture-series-speaker.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "DBSP: A formal model for streaming computation and its applications to incremental computations and databases"
+date: "2023-10-18 10:30:00 -0600"
+permalink: "/talks/2023-10-18-data-science-lecture-series-speaker/"
+slug: "2023-10-18-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+zoom: "https://utah.zoom.us/j/91737198805pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ bio: "Mihai Budiu is chief scientist at Feldera. He has a Ph.D. in CS from Carnegie Mellon University. He was previously employed at VMware Research, Barefoot Networks, and Microsoft Research. Mihai has worked on reconfigurable hardware, computer architecture, compilers, security, distributed systems, big data platforms, large-scale machine learning, programmable networks and P4, data visualization, and databases; four of his papers have received “test of time” awards. He has also received two technology transfer awards.\n\nZoom link: https://utah.zoom.us/j/91737198805pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09)\n"
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-10-18-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+DBSP is a simple streaming programming language inspired by Digital Signal Processing [DSP]. DBSP can be used to give a precise definition of incremental computations -- operating on changes (deltas, diffs). Moreover, given a DBSP program, a simple algorithm can convert it to a DBSP program that computes on changes. All practical database query operators (the relational algebra, group-by, aggregations, fixed-points, etc) can be expressed in DBSP. As a consequence we obtain an algorithm which can incrementalize essentially any database query. The DBSP theory has been formally verified using a theorem prover, making it the first verified theory of incremental view maintenance.
+
+The DBSP paper has received the best paper award at the 2023 conference on Very Large Databases [VLDB].
diff --git a/_talks/2023-10-25-data-science-lecture-series-speaker.md b/_talks/2023-10-25-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..3b45703
--- /dev/null
+++ b/_talks/2023-10-25-data-science-lecture-series-speaker.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Computational journeys in a sparse universe"
+date: "2023-10-25 10:30:00 -0600"
+permalink: "/talks/2023-10-25-data-science-lecture-series-speaker/"
+slug: "2023-10-25-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+zoom: "https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ bio: "Aydın Buluç is a Senior Scientist at the Applied Math and Computational Research Division of the Lawrence Berkeley National Laboratory (LBNL) and an Adjunct Faculty at EECS department of UC Berkeley. His research interests include parallel computing, combinatorial scientific computing, high performance graph analysis and machine learning, sparse linear algebra, and computational genomics. He received his Ph.D. in Computer Science from the University of California, Santa Barbara in 2010. After that, he was a Luis W. Alvarez postdoctoral fellow at LBNL. Dr. Buluç is a recipient of the DOE Early Career Award in 2013 and the IEEE TCSC Award for Excellence for Early Career Researchers in 2015. He recently led a team that\nwas chosen as a finalist for the 2022 ACM Gordon Bell Prize. He was a founding associate editor of the ACM Transactions on Parallel Computing. He is currently leading a DOE Mathematical Multifaceted Integrated Capabilities Center named Sparsitute.\n\nZoom: https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09)\n"
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-10-25-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+Sparsity is a fundamental assumption that allows us to compute efficiently on and find parsimonious solutions to science and engineering problems. Sparsity exists in all basic sciences such as physics, biology, and chemistry. I am going to give a sampling of recent work we have done on sparse computations. My talk will travel across diverse problem domains including randomized linear algebra, graph neural networks, protein family and structure discovery from metagenomic data, and tensor computations. The underlying theme will be the challenges posed by sparsity and the computational techniques we employ to overcome these challenges.
diff --git a/_talks/2023-11-01-data-science-lecture-series-speaker.md b/_talks/2023-11-01-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..a1849d2
--- /dev/null
+++ b/_talks/2023-11-01-data-science-lecture-series-speaker.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Linear Probing Revisited: How to Get Rid of Clustering"
+date: "2023-11-01 10:30:00 -0600"
+permalink: "/talks/2023-11-01-data-science-lecture-series-speaker/"
+slug: "2023-11-01-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+zoom: "https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ bio: "William Kuszmaul's research focuses on the design and analysis of randomized algorithms and data structures. He is currently the Rabin Postdoctoral Fellow in Theoretical Computer Science at Harvard University, and after that, he will begin as an Assistant Professor in the CS Department at CMU. His research has won numerous awards at both theory and systems conferences, including Distinguished Paper at ASPLOS'23, Best Student Paper at ESA'22, Best Paper Finalist at SPAA'22, Best Paper at FUN'20, and Best Paper Finalist at APOCS'20. Prior to his postdoc, William completed a PhD at MIT, where he was funded by the John and Fannie Hertz Fellowship.\n\nZoom: https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09 (https://utah.zoom.us/j/91737198805?pwd=Z1o2SzE4OVRodVhDWWExOTdVcUs5Zz09)\n"
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-11-01-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+The linear-probing hash table is one of the oldest and most widely used data structures in computer science. However, linear probing also famously comes with a major drawback: as soon as the hash table reaches a high memory utilization, elements within the hash table begin to cluster together, causing insertions to become slow. This clustering phenomenon, which was first discovered by Donald Knuth in 1962, increases the expected time per insertion to $\Theta(x^2)$ (rather than the more desirable $\Theta(x)$) in a hash table that is a $1 - 1/x$ fraction full.
+A natural question is whether one can somehow reduce clustering. In this talk, we establish an even stronger statement: the classical linear-probing hash table (even as it was first implemented in the 1950s) already has less clustering than the classical results would seem to suggest. As insertions and deletions are performed over time, the tombstones left behind by deletions cause the combinatorial structure of the hash table to stabilize in a way that eliminates clustering. This means that, for some versions of linear probing, the amortized expected time per operation is actually $\tilde{O}(x)$. We also present a new version of linear probing that avoids clustering entirely, achieving $O(x)$ expected time per operation.
diff --git a/_talks/2023-11-22-data-science-lecture-series-speaker.md b/_talks/2023-11-22-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..b2f9a64
--- /dev/null
+++ b/_talks/2023-11-22-data-science-lecture-series-speaker.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "“Leveraging the Structure of Data“"
+date: "2023-11-22 10:30:00 -0700"
+permalink: "/talks/2023-11-22-data-science-lecture-series-speaker/"
+slug: "2023-11-22-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ affiliation: "such as NeurIPS, ICML, ICLR, KDD, and WWW"
+ website: "https://ai.google/research/teams/algorithms-optimization/"
+ bio: "Bryan Perozzi is a Research Scientist in Google Research’s Algorithms and Optimization (https://ai.google/research/teams/algorithms-optimization/) group, where he routinely analyzes some of the world’s largest (and perhaps most interesting) graphs. Bryan’s research (https://scholar.google.com/citations?hl=en&user=rZgbMs4AAAAJ&view_op=list_works) focuses on developing techniques for learning expressive representations of relational data with neural networks. These scalable algorithms are useful for prediction tasks (classification/regression), pattern discovery, and anomaly detection in large networked data sets.\n\nBryan is an author of 40+ peer-reviewed papers at leading conferences in machine learning and data mining (such as NeurIPS, ICML, ICLR, KDD, and WWW). His doctoral work on learning network representations was awarded the prestigious SIGKDD Dissertation Award. Bryan received his Ph.D. in Computer Science from Stony Brook University in 2016, and his M.S. from the Johns Hopkins University in 2011.\n"
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-11-22-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+Although predictions from machine learning models influence more and more of our lives, the standard way of posing a ML problem has remained relatively unchanged for decades. In the search for better models, a new and popular family of techniques (sometimes called Graph Machine Learning) has emerged. These techniques rely on expanding beyond the features of an individual entity and instead look to pull information from its relationships. The methods offer a tantalizing way of improving task performance by leveraging previously unused information. However, it is not a free lunch, as these models can be more complex, difficult to train, and may have challenges in interpretability. This talk will discuss the fundamentals of graph machine learning, a few models, and some insights from years of real-world applications.
diff --git a/_talks/2023-11-29-data-science-lecture-series-speaker.md b/_talks/2023-11-29-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..25ac261
--- /dev/null
+++ b/_talks/2023-11-29-data-science-lecture-series-speaker.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Framework for Parallel Hierarchical Agglomerative Clustering"
+date: "2023-11-29 10:30:00 -0700"
+permalink: "/talks/2023-11-29-data-science-lecture-series-speaker/"
+slug: "2023-11-29-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ bio: "Shangdi is a PhD student at MIT Department of Electrical Engineering and Computer Science, advised by professor Julian Shun. Her research focuses on parallel algorithms for graph and metric data clustering. Shangdi received her BSc in Computer Science and Operations Research from Cornell University and MSc in Computer Science from MIT."
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-11-29-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+We study the hierarchical clustering problem, where the goal is to produce a dendrogram that represents clusters at varying scales of a data set. We propose the ParChain framework for designing parallel hierarchical agglomerative clustering (HAC) algorithms, and using the framework we obtain novel parallel algorithms for the complete linkage, average linkage, and Ward's linkage criteria. Compared to most previous parallel HAC algorithms, which require quadratic memory, our new algorithms require only linear memory, and are scalable to large data sets. ParChain is based on our parallelization of the nearest-neighbor chain algorithm, and enables multiple clusters to be merged on every round. We introduce two key optimizations that are critical for efficiency: a range query optimization that reduces the number of distance computations required when finding nearest neighbors of clusters, and a caching optimization that stores a subset of previously computed distances, which are likely to be reused. Experimentally, we show that our highly-optimized implementations using 48 cores with two-way hyper-threading achieve 5.8--110.1x speedup over state-of-the-art parallel HAC algorithms and achieve 13.75--54.23x self-relative speedup. Compared to state-of-the-art algorithms, our algorithms require up to 237.3x less space. Our algorithms are able to scale to data set sizes with tens of millions of points, which previous algorithms are not able to handle.
diff --git a/_talks/2023-12-06-data-science-lecture-series-speaker.md b/_talks/2023-12-06-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..8696163
--- /dev/null
+++ b/_talks/2023-12-06-data-science-lecture-series-speaker.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Parallel Batch-Dynamic Graph Algorithms"
+date: "2023-12-06 10:30:00 -0700"
+permalink: "/talks/2023-12-06-data-science-lecture-series-speaker/"
+slug: "2023-12-06-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ bio: "Julian Shun is an Associate Professor of Electrical Engineering and Computer Science at MIT and a lead investigator in MIT Computer Science and Artificial Intelligence Laboratory (CSAIL). His research focuses on the theory and practice of parallel algorithms and programming, with particular emphasis on designing algorithms and frameworks for large-scale graph processing and spatial data analysis. Prior to joining MIT, he was a postdoctoral Miller Research Fellow at UC Berkeley. His honors include the NSF CAREER award, DOE Early Career Award, ACM Doctoral Dissertation Award, CMU School of Computer Science Doctoral Dissertation Award, Google Faculty Research Award, Google Research Scholar Award, SoE Ruth and Joel Spira Award for Excellence in Teaching, Allen Newell Award for Research Excellence, Facebook Graduate Fellowship, and best paper awards at PLDI, SPAA, CGO, and DCC."
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-12-06-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+There has been significant interest in graph analytics due to their applications in many domains, including social network and Web analytics, machine learning, biology, and physical simulations. Real-world graphs today are massive and also dynamic. As many real-world graphs change rapidly, it is crucial to design dynamic algorithms that efficiently maintain graph statistics upon updates, since the cost of re-computation from scratch can be prohibitive. Furthermore, due to the high frequency of updates, we can improve performance by using parallelism to process batches of updates at a time. This talk presents new graph algorithms in this parallel batch-dynamic setting.
+
+Specifically, we present the first parallel batch-dynamic algorithm for approximate k-core decomposition that is efficient in both theory and practice. Our algorithm is based on our novel parallel level data structure, inspired by the sequential level data structures of Bhattacharya et al. and Henzinger et al. Given a graph with n vertices and a batch of B updates, our algorithm maintains a (2 + epsilon)-approximation of the coreness values of all vertices (for any constant epsilon > 0) in O(B log^2(n)) amortized work and O(log^2(n) loglog(n)) span (parallel time) with high probability. We implement and experimentally evaluate our algorithm, and demonstrate significant speedups over state-of-the-art serial and parallel implementations for dynamic k-core decomposition.
+
+We have also designed new parallel batch-dynamic algorithms for low out-degree orientation, maximal matching, clique counting, graph coloring, minimum spanning forest, single-linkage clustering, some of which use our parallel level data structure.
diff --git a/_talks/2023-12-13-data-science-lecture-series-speaker.md b/_talks/2023-12-13-data-science-lecture-series-speaker.md
new file mode 100644
index 0000000..6351164
--- /dev/null
+++ b/_talks/2023-12-13-data-science-lecture-series-speaker.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Felix Reidl, Birkbeck University"
+date: "2023-12-13 10:30:00 -0700"
+permalink: "/talks/2023-12-13-data-science-lecture-series-speaker/"
+slug: "2023-12-13-data-science-lecture-series-speaker"
+start_time: "10:30 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "FASB 295"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series. Speaker"
+ bio: "Felix is a senior lecturer at Birkbeck College (University of London) and the director of the Birkbeck Institute for Data Analytics."
+speaker_names: "Data Science Lecture Series. Speaker"
+source_file: "_data/talks/2023-12-13-data-science-lecture-series-speaker.toml"
+generated: true
+---
+
+
+
+Data Science and AI have an every increasing presence in our social, political and economic life. Given the immense influence the technologies of these field have and will have, I argue that researchers and academic institutions should reflect on the implications their work has in the world at large.
+
+In this talk I would like take stock of the larger context: a world lacking futures, fragmented academic disciplines, and the dystopian use of technology. While we cannot hope to solve any of these problems, I argue that we can and should resist the underlying trends. To that end, I propose that our institutions should be constructed first and foremost around creativity and participation and what that could mean in practice.
diff --git a/_talks/2024-01-10-data-science-lecture-series.md b/_talks/2024-01-10-data-science-lecture-series.md
new file mode 100644
index 0000000..bcc7eac
--- /dev/null
+++ b/_talks/2024-01-10-data-science-lecture-series.md
@@ -0,0 +1,19 @@
+---
+layout: "talk"
+title: "Orientation"
+date: "2024-01-10 13:30:00 -0700"
+permalink: "/talks/2024-01-10-data-science-lecture-series/"
+slug: "2024-01-10-data-science-lecture-series"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+canceled: false
+speakers:
+ - name: "Data Science Lecture Series"
+ affiliation: "Spring 24"
+speaker_names: "Data Science Lecture Series"
+source_file: "_data/talks/2024-01-10-data-science-lecture-series.toml"
+generated: true
+---
+
+
diff --git a/_talks/2024-01-31-pratik-soni.md b/_talks/2024-01-31-pratik-soni.md
new file mode 100644
index 0000000..8868a64
--- /dev/null
+++ b/_talks/2024-01-31-pratik-soni.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Cryptography for Fairness"
+date: "2024-01-31 13:30:00 -0700"
+permalink: "/talks/2024-01-31-pratik-soni/"
+slug: "2024-01-31-pratik-soni"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science Seminar"
+zoom: "https://utah.zoom.us/j/96005100565?pwd=WmFGN25RazZwV2NoMGE2dVFGMngyZz09"
+canceled: false
+speakers:
+ - name: "Pratik Soni"
+ affiliation: "Utah"
+speaker_names: "Pratik Soni"
+source_file: "_data/talks/2024-01-31-pratik-soni.toml"
+generated: true
+---
+
+
+
+Location: In person (GC 2560). Will also be streamed at:
+https://utah.zoom.us/j/96005100565?pwd=WmFGN25RazZwV2NoMGE2dVFGMngyZz09 (https://utah.zoom.us/j/96005100565?pwd=WmFGN25RazZwV2NoMGE2dVFGMngyZz09)
diff --git a/_talks/2024-03-13-swabha-swayamdipta.md b/_talks/2024-03-13-swabha-swayamdipta.md
new file mode 100644
index 0000000..d87bdce
--- /dev/null
+++ b/_talks/2024-03-13-swabha-swayamdipta.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Understanding LLMs through their Generative Behavior, Successes and Shortcomings"
+date: "2024-03-13 13:30:00 -0600"
+permalink: "/talks/2024-03-13-swabha-swayamdipta/"
+slug: "2024-03-13-swabha-swayamdipta"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science Seminar"
+canceled: false
+speakers:
+ - name: "Swabha Swayamdipta"
+ affiliation: "USC"
+ bio: "Swabha Swayamdipta is an Assistant Professor of Computer Science and a Gabilan Assistant Professor at the University of Southern California. Her research interests are in natural language processing and machine learning, with a primary interest in the estimation of dataset quality, understanding and evaluation of generative models of language, and using language technologies to understand social behavior. At USC, Swabha leads the Data, Interpretability, Language and Learning (DILL) Lab. She received her PhD from Carnegie Mellon University, followed by a postdoc at the Allen Institute for AI. Her work has received outstanding paper awards at ICML 2022, NeurIPS 2021 and an honorable mention for the best paper at ACL 2020. Her research is supported by awards from the Allen Institute for AI and Intel Labs."
+speaker_names: "Swabha Swayamdipta"
+source_file: "_data/talks/2024-03-13-swabha-swayamdipta.toml"
+generated: true
+---
+
+
+
+Generative capabilities of large language models have grown beyond the wildest imagination of the broader AI research community, leading many to speculate whether these successes may be attributed to the training data or model design. I will present some work from my group which sheds light on understanding LLMs by studying their generative behavior, successes and shortcomings. First, I will show that standard inference algorithms work well because of the particular design behind LLMs. Next, I will discuss recently found successes and failures of LLMs on a combination of tasks, requiring world and domain-specific knowledge, linguistic capabilities and awareness of human and social utility. Overall, these findings paint a partial yet complex picture of our understanding of LLMs and provide a guide to the next steps forward.
diff --git a/_talks/2024-04-10-ucds-lecture-series.md b/_talks/2024-04-10-ucds-lecture-series.md
new file mode 100644
index 0000000..9a6f275
--- /dev/null
+++ b/_talks/2024-04-10-ucds-lecture-series.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "Service Operations for Justice-On-Time: A Data-Driven Queueing Approach"
+date: "2024-04-10 13:30:00 -0600"
+permalink: "/talks/2024-04-10-ucds-lecture-series/"
+slug: "2024-04-10-ucds-lecture-series"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+zoom: "https://utah.zoom.us/j/96005100565?pwd=WmFGN25RazZwV2NoMGE2dVFGMngyZz09"
+canceled: false
+speakers:
+ - name: "UCDS Lecture Series"
+ bio: "Dr Nitin Bakshi is department chair and Professor of Operations and Information Systems at the David Eccles School of Business, University of Utah. He focuses his research on the management of disruption risk in operations and supply-chain management, with an emphasis on “low-probability high-consequence” events. He is currently investigating how to manage reporting of accident precursors to enhance safety in dangerous operations, and exploring new frontiers related to efficiency in judicial operations.\n\nHe holds a B. Tech. in Electrical Engineering from IIT Bombay; an M.S. in Management Science from Stanford University; and a Ph.D. in Applied Economics from the Wharton School, University of Pennsylvania. Dr Bakshi has previously worked as a manager for Unilever and as an Algorithm Design Engineer for SmartOps Inc. Before joining the University of Utah he served on the faculty at the London Business School.\n"
+speaker_names: "UCDS Lecture Series"
+source_file: "_data/talks/2024-04-10-ucds-lecture-series.toml"
+generated: true
+---
+
+
+
+Limited resources in the judicial system can lead to costly delays, stunted economic development, and even failure to deliver justice. Using the Supreme Court of India as an exemplar for such resource-constrained settings, we apply ideas from service operations to study delay. Specifically, court dynamics constitute a case-management queue, whereby each case may experience multiple service encounters spread across time, but all are necessarily with the same server. Our goal is to elucidate the drivers of congestion, focusing on metrics such as the expected case-disposition time (delay) and expected number of cases awaiting adjudication (pendency), and leverage this understanding to recommend operational interventions.
+
+We employ data-driven calibrated simulations to model the analytically intractable case-management queue. The life cycle of a case comprises two stages: pre-admission (before determining its merit for detailed hearings) and post-admission. Our methodology allows us to capture the queueing dynamics in which the judges are shared resources across the two stages. It also permits modeling of holiday capacity, which is flexibly tailored to address any surplus work that spills over from the regular year. We find that the second stage of this judicial queue is overloaded, but holiday capacity creates a perception of stability by steadying performance metrics.
+
+The sources of inefficiency that drive congestion include a misalignment between scheduling guidelines and judicial capacity, coupled with the requirement to schedule hearings in advance. Together, these factors inhibit utilization of shared capacity across the two-stage judicial queue. We demonstrate how interventions that account for these inefficiencies can successfully tackle judicial delay. In particular, scheduling to improve the allocation of time across pre- and post-admission cases can cut down the expected delay by as much as 65%.
diff --git a/_talks/2024-08-27-jeff-phillips.md b/_talks/2024-08-27-jeff-phillips.md
new file mode 100644
index 0000000..5af9819
--- /dev/null
+++ b/_talks/2024-08-27-jeff-phillips.md
@@ -0,0 +1,28 @@
+---
+layout: "talk"
+title: "== Sketching and Classifying Spatial Trajectories =="
+date: "2024-08-27 12:30:00 -0600"
+permalink: "/talks/2024-08-27-jeff-phillips/"
+slug: "2024-08-27-jeff-phillips"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Jeff Phillips"
+ affiliation: "Utah KSoC"
+ website: "https://users.cs.utah.edu/~jeffp/"
+speaker_names: "Jeff Phillips"
+source_file: "_data/talks/2024-08-27-jeff-phillips.toml"
+generated: true
+---
+
+
+
+Spatial trajectories, often represented as a sequence of spatial positions, are a standard way to represent human mobility patterns. They also are used to represent motion patterns including for animals, drones, or last-mile rentals (e-scooters). However, these trajectories are notoriously difficult to work with as they overlap and can stretch long distances.
+In this talk we discuss a sketch (the minDist Sketch) that makes just about any data analysis on trajectories tasks simple and efficient. This first considers spatial trajectories as an abstract shape, and then maps them to a high-dimensional Euclidean space as a vector. We can show recovery, pseudo-metric, and metric properties of this representation. Variants can include direction information, or traits like velocity and acceleration.
+Moreover, once represented as this vector, the trajectory data is extremely easy to work with. Allowing for out-of-the-box use of software for nearest-neighbor search, clustering, and classification.
+In particular, we conduct the first formal study of classifying spatial trajectories: given trajectories from two different distributions (e.g., generated by car or bus) given a new trajectory that is unlabeled, how well can we predict which class it was from?
+Over several data sets we have assembled that demand this task, we conduct a large study, and show that the minDist sketch and its variants are consistently the easiest and most accurate method (or at the least among the best in each instance).
diff --git a/_talks/2024-09-03-esha-datta.md b/_talks/2024-09-03-esha-datta.md
new file mode 100644
index 0000000..883ebff
--- /dev/null
+++ b/_talks/2024-09-03-esha-datta.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Topological Signatures of Out-of-Distribution Examples"
+date: "2024-09-03 12:30:00 -0600"
+permalink: "/talks/2024-09-03-esha-datta/"
+slug: "2024-09-03-esha-datta"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Esha Datta"
+ affiliation: "Sandia NL"
+ website: "https://scholar.google.com/citations?user=FB5NpNQAAAAJ&hl=en"
+speaker_names: "Esha Datta"
+source_file: "_data/talks/2024-09-03-esha-datta.toml"
+generated: true
+---
+
+
+
+Machine learning (ML) models employed for real-world tasks will invariably encounter inference data that is distributionally shifted from their training datasets. Such out-of-distribution (OOD) examples can have adverse effects on model performance and can pose significant problems in high-consequence application areas like healthcare or autonomous vehicles. We develop a topological characterization of OOD examples and present a computationally feasible methodology for detecting such data in a deployed pipeline. The approach leverages the known property that well-trained ML models induce a topological “simplification” on its training dataset. By computing the persistent homology of the hidden layer embeddings of training and test data, we demonstrate empirically our ability to identify the presence of OOD examples for a given model.
diff --git a/_talks/2024-09-10-guanhong-tao.md b/_talks/2024-09-10-guanhong-tao.md
new file mode 100644
index 0000000..95899ff
--- /dev/null
+++ b/_talks/2024-09-10-guanhong-tao.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Are AI-enabled Systems Safe and Secure?"
+date: "2024-09-10 12:30:00 -0600"
+permalink: "/talks/2024-09-10-guanhong-tao/"
+slug: "2024-09-10-guanhong-tao"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Guanhong Tao"
+speaker_names: "Guanhong Tao"
+source_file: "_data/talks/2024-09-10-guanhong-tao.toml"
+generated: true
+---
+
+
+
+Abstract
+Artificial Intelligence (AI) has been integrated into various sectors, such as facial recognition and autonomous driving. But are the security and safety of these AI-enabled systems fully ensured? In this talk, I will present various vulnerabilities in these systems. My presentation will cover novel optimization techniques for identifying and mitigating backdoor vulnerabilities in both white-box and black-box settings, achieving substantial improvements in performance. I will share insights into the nature of backdoors and their presence in pre-trained models. Finally, I will conclude with an outlook on our recent exploration of the security of emerging AI techniques, such as generative AI.
diff --git a/_talks/2024-09-17-aurora-clark.md b/_talks/2024-09-17-aurora-clark.md
new file mode 100644
index 0000000..2f1de11
--- /dev/null
+++ b/_talks/2024-09-17-aurora-clark.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "The Importance of Shape in Chemistry Data"
+date: "2024-09-17 12:30:00 -0600"
+permalink: "/talks/2024-09-17-aurora-clark/"
+slug: "2024-09-17-aurora-clark"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Aurora Clark"
+speaker_names: "Aurora Clark"
+source_file: "_data/talks/2024-09-17-aurora-clark.toml"
+generated: true
+---
+
+
+
+Data in the field of Chemistry has heavily leveraged graph theory representations within data science applications. However, there is a rich geometric and topological structure of many chemical systems (and their data) that has been less employed for feature optimization, dimensionality reduction, and predictive models. Within this discussion I will highlight some recent work and interests that seek to employ computational topology and geometry within chemistry data sets from molecular dynamics simulations – both in the context of ensemble average and temporally evolving data sets.
diff --git a/_talks/2024-09-24-rebecca-barter.md b/_talks/2024-09-24-rebecca-barter.md
new file mode 100644
index 0000000..675c04b
--- /dev/null
+++ b/_talks/2024-09-24-rebecca-barter.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Veridical Data Science: the Practice of Responsible Data Analysis and Decision Making"
+date: "2024-09-24 12:30:00 -0600"
+permalink: "/talks/2024-09-24-rebecca-barter/"
+slug: "2024-09-24-rebecca-barter"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Rebecca Barter"
+ website: "http://www.rebeccabarter.com/"
+ bio: "Dr. Rebecca Barter is a Research Assistant Professor in the Division of Epidemiology at the University of Utah. As a statistician, data scientist, and educator, Dr Barter specializes in data science education and the analysis of complex healthcare data. Originally from Australia, Dr. Barter earned her PhD in Statistics from the University of California, Berkeley, in 2019, where she co-authored the book Veridical Data Science: The Practice of Responsible Data Analysis and Decision Making with her advisor, Professor Bin Yu. In addition to her academic work, Dr. Barter shares data science resources and insights on her blog, www.rebeccabarter.com (http://www.rebeccabarter.com/)."
+speaker_names: "Rebecca Barter"
+source_file: "_data/talks/2024-09-24-rebecca-barter.toml"
+generated: true
+---
+
+
+
+Data science is often presented as a straightforward, linear process involving statistical and computational techniques, without addressing the complexities inherent in real-world applications. In contrast, our new book,
+"Veridical Data Science: The Practice of Responsible Data Analysis and Decision Making", teaches data scientists to navigate the reality that most projects involve answering ambiguous domain questions with messy data, all while managing a complex web of human judgment calls. We emphasize that datasets are merely approximations of reality, and analyses are shaped by human interpretation. Using the Predictability, Computability, and Stability (PCS) framework to assess the trustworthiness and relevance of data-driven results, "Veridical Data Science" provides an actionable guide for conducting responsible and trustworthy data science.
diff --git a/_talks/2024-10-01-simon-brewer.md b/_talks/2024-10-01-simon-brewer.md
new file mode 100644
index 0000000..bd6b893
--- /dev/null
+++ b/_talks/2024-10-01-simon-brewer.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Exploring long-term ecosystem change with self-organizing maps"
+date: "2024-10-01 12:30:00 -0600"
+permalink: "/talks/2024-10-01-simon-brewer/"
+slug: "2024-10-01-simon-brewer"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Simon Brewer"
+speaker_names: "Simon Brewer"
+source_file: "_data/talks/2024-10-01-simon-brewer.toml"
+generated: true
+---
+
+
+
+Ongoing climate change has the potential to impact a variety of physical, biological and social systems, and there is increasing concern that these changes may be sufficient to result in these systems crossing tipping points, effectively undergoing irreversible changes in state. For slow turnover systems, such as forest ecosystems, understanding the likelihood and ramifications of these state changes is challenging due to the relative short observational record. Sedimentary records of ecosystem change offer an alternative data source with a wide temporal and spatial scope, but are inherently noisy and high dimensional. Self-organizing maps provide a data-driven way to visualize nonlinear patterns in these data, and to identify past ecosystem states and state transitions. The results are used to build a simple Markov model illustrating the probability and directionality of these transitions.
diff --git a/_talks/2024-10-15-raghav-venkatraman.md b/_talks/2024-10-15-raghav-venkatraman.md
new file mode 100644
index 0000000..a3e309d
--- /dev/null
+++ b/_talks/2024-10-15-raghav-venkatraman.md
@@ -0,0 +1,28 @@
+---
+layout: "talk"
+title: "Minmax estimation rates for manifold learning"
+date: "2024-10-15 12:30:00 -0600"
+permalink: "/talks/2024-10-15-raghav-venkatraman/"
+slug: "2024-10-15-raghav-venkatraman"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Raghav Venkatraman"
+speaker_names: "Raghav Venkatraman"
+source_file: "_data/talks/2024-10-15-raghav-venkatraman.toml"
+generated: true
+---
+
+
+
+This talk is focused on obtaining minmax estimation rates for the "manifold learning" problem. Given N data points hypothesized to be i.i.d (independent and identically distributed) samples of a nice density (from within a reasonable class of densities) on a nice manifold (from within some class of nice manifolds of known intrinsic dimension d), the manifold learning problem boils down to estimating certain statistics of this "ground truth" manifold, such as the first few eigenmodes of the Laplace Beltrami operator on the manifold. The minmax estimation problem further asks: given N such data points, among all estimators of the desired statistics (say, a particular eigenvalue and associated eigenfunctions in a suitable norm), which one achieves the smallest maximum expected risk, and how does this minmax risk scale in N and the intrinsic dimension d of the manifold?
+
+An intuitive but impractical estimator consists in estimating the density from the given samples through a ``kernel density estimation'', and then solving the resulting continuum eigenproblem using a numerical method such as finite elements: this estimator turns out to be minmax optimal in scaling-- namely, the associated expected risk scales like N^{-2/d+4}, and we can show a matching lower bound for the minmax risk, demonstrating that no estimator can do better, in scaling, than this estimator.
+
+Next, we ask: do there exist *practical* estimators that are agnostic to knowledge of the manifold (so we don't have to discretize them in order to compute with finite elements!) that achieve, at least nearly, this minmax scaling of the expected risk? We affirmatively answer this question by showing that, the spectrum of a carefully constructed graph laplacian on a random geometric graph constructed from the given N samples provides an estimator for the eigenvalue and eigenvectors of the Laplace Beltrami operator on the manifold that achieves this minmax rate upto a log factor (to a small power). Both the lower and upper bound estimates in the talk bring in new PDE tools to this statistical question, and that we believe will be more broadly applicable in similar applications.
+
+This talk is based on joint work with Nicolas Garcia Trillos and his PhD student Chenghui Li (U. W. Madison), and builds on prior joint work with Scott N. Armstrong (Courant Institute).
diff --git a/_talks/2024-10-29-vivek-gupta.md b/_talks/2024-10-29-vivek-gupta.md
new file mode 100644
index 0000000..4f9af53
--- /dev/null
+++ b/_talks/2024-10-29-vivek-gupta.md
@@ -0,0 +1,31 @@
+---
+layout: "talk"
+title: "Reasoning on Tabular and Multimodal Data"
+date: "2024-10-29 12:30:00 -0600"
+permalink: "/talks/2024-10-29-vivek-gupta/"
+slug: "2024-10-29-vivek-gupta"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Vivek Gupta"
+ affiliation: "ASU & UCDS alumni"
+ website: "https://vgupta123.github.io"
+ bio: "Vivek Gupta is an Assistant Professor of Computer Science at Arizona State University (ASU), where he works on AI systems that help computers reason with complex data like tables, charts, diagrams, and maps etc. Before ASU, he was a postdoctoral researcher at the University of Pennsylvania in the Cognitive Computation Group. He earned his Ph.D. in Computer Science from the University of Utah and has received several awards, including the Bloomberg Data Science Fellowship and NLP Best Paper Awards. You can learn more about his work at vgupta123.github.io and his group CoRAL page at coral-lab-asu.github.io."
+speaker_names: "Vivek Gupta"
+source_file: "_data/talks/2024-10-29-vivek-gupta.toml"
+generated: true
+---
+
+
+
+In this talk, I’ll walk through some of the latest AI advancements that address the challenges of working with complex data, with a focus on improving reasoning for both tabular and multimodal data.
+
+I’ll begin by introducing H-STAR, a hybrid algorithm that combines symbolic and semantic reasoning to enhance question answering for tabular data. By leveraging multi-view table extraction and adaptive reasoning, H-STAR has shown great potential in improving reasoning across tabular datasets.
+
+Then, I’ll introduce MMTabQA, a dataset we developed to evaluate how AI systems manage multimodal tables that integrate structured text and images. Our research reveals where current models struggle to process these diverse data types, highlighting key areas for improvement.
+
+To wrap up, I’ll discuss open challenges and future directions, including expanding reasoning capabilities to other complex data types—such as charts, maps, and flowcharts—and enhancing AI systems' robustness in handling numerical, temporal, and visual reasoning, particularly with large (vision) language models.
diff --git a/_talks/2024-11-12-amir-abdullah.md b/_talks/2024-11-12-amir-abdullah.md
new file mode 100644
index 0000000..6cc8048
--- /dev/null
+++ b/_talks/2024-11-12-amir-abdullah.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Interpreting Learned Feedback Patterns in Large Language Models"
+date: "2024-11-12 12:30:00 -0700"
+permalink: "/talks/2024-11-12-amir-abdullah/"
+slug: "2024-11-12-amir-abdullah"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Amir Abdullah"
+ website: "https://scholar.google.com/citations?user=jPEbq5wAAAAJ&hl=en"
+speaker_names: "Amir Abdullah"
+source_file: "_data/talks/2024-11-12-amir-abdullah.toml"
+generated: true
+---
+
+
+
+Amir is an active researcher in mechanistic interpretability, opening the blackbox of large language models to reverse engineer the inner workings and analyze internal representations.. In this talk, he will discuss his paper in NeuIPS 2024 on interpreting reward models in language models using sparse autoencoders on internal representations. Further, Amir will introduce followup work studying whether internal representations can be transferred between large language models.
diff --git a/_talks/2024-11-19-hoaning-xue.md b/_talks/2024-11-19-hoaning-xue.md
new file mode 100644
index 0000000..37ddd7a
--- /dev/null
+++ b/_talks/2024-11-19-hoaning-xue.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Computational and experimental approaches to examining short videos' persuasive effects"
+date: "2024-11-19 12:30:00 -0700"
+permalink: "/talks/2024-11-19-hoaning-xue/"
+slug: "2024-11-19-hoaning-xue"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Hoaning Xue"
+ affiliation: "Utah Communications"
+ website: "https://faculty.utah.edu/u6059240-HAONING_XUE/hm/index.hml"
+speaker_names: "Hoaning Xue"
+source_file: "_data/talks/2024-11-19-hoaning-xue.toml"
+generated: true
+---
+
+
+
+There’s a gap in understanding how people process multimodal information collectively, despite extensive research on the effects of individual multimodal features and the rapid advances in computer vision. This gap is increasingly relevant as short video platforms emerge as major information sources and influence public opinion. In this talk, I present findings from a project that combines a data-driven approach with social scientific theories to investigate how multimodal features in short videos impact audience engagement and attitude change. This research is grounded in the theoretical framework of Message Sensation Value (MSV) to theorize and quantify how multimodal features in short videos capture attention and affect information processing. This project includes two studies: (1) a computational model of MSV that predicts video engagement from 11 multimodal features across a dataset of 15,000 short videos from three popular short video platforms; second, an online experiment examining the attentional mechanism underlying the persuasive effects of MSV in short videos on message credibility and attitude change. This project provides a useful framework and computational tool for short video research.
diff --git a/_talks/2024-11-26-zhichao-xu.md b/_talks/2024-11-26-zhichao-xu.md
new file mode 100644
index 0000000..3aa75b8
--- /dev/null
+++ b/_talks/2024-11-26-zhichao-xu.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Representation Learning for IR and Role of Retrieval in LLM Era"
+date: "2024-11-26 12:30:00 -0700"
+permalink: "/talks/2024-11-26-zhichao-xu/"
+slug: "2024-11-26-zhichao-xu"
+start_time: "12:30 PM"
+end_time: "1:30 PM"
+series: "Data Science Seminar"
+location: "WEB 1230"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Zhichao Xu"
+ affiliation: "Utah KSoC"
+ website: "https://zhichaoxu-utah.github.io/"
+ bio: "Zhichao Xu is a final year Ph.D. student in Kahlert School of Computing, University of Utah. He is affiliated with UtahNLP lab and TDAVIS lab, advised by Prof. Bei Wang Philips and Prof. Vivek Srikumar. His main research interests include efficient NLP methods, web search & information retrieval. He has published in major IR venues such as TheWebConf, SIGIR, WSDM, CIKM, ICTIR and NLP venues such as NAACL and EMNLP."
+speaker_names: "Zhichao Xu"
+source_file: "_data/talks/2024-11-26-zhichao-xu.toml"
+generated: true
+---
+
+
+
+Retrieval is the critical way of accessing information in people’s daily lives. Retrieval applications include search engines, conversational shopping assistants, or when asked about 2+3=?, human brains do retrieval instead of reasoning. In this talk, I’ll briefly go through the history of representation learning in retrieval, from Bag-of-Words representations to the latest dense and learned sparse retrieval algorithms. Increasingly, people go to ChatGPT or other large language models for information seeking instead of search engines. With this existential crisis in mind, I will talk about the role of retrieval in LLM era, specifically, retrieval-augmented generation, strengths, weaknesses and open problems.
diff --git a/_talks/2025-01-17-fengjiao-wang.md b/_talks/2025-01-17-fengjiao-wang.md
new file mode 100644
index 0000000..9a36fb0
--- /dev/null
+++ b/_talks/2025-01-17-fengjiao-wang.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Supervised Learning on Tabular Data"
+date: "2025-01-17 13:30:00 -0700"
+permalink: "/talks/2025-01-17-fengjiao-wang/"
+slug: "2025-01-17-fengjiao-wang"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Fengjiao Wang"
+ affiliation: "Utah SoC"
+ website: "https://fengjiaowang7.github.io"
+speaker_names: "Fengjiao Wang"
+source_file: "_data/talks/2025-01-17-fengjiao-wang.toml"
+generated: true
+---
+
+
+
+Self-supervised and Semi-supervised learning (SSL) on tabular data is an understudied topic. Despite some attempts, there are two major challenges: 1. Imbalanced nature in the tabular dataset; 2. The one-hot encoding used in these methods becomes less efficient for high-cardinality categorical features. To cope with the challenges, we propose SAWTab which uses a target encoding method, Conditional Probability Representation (CPR), for efficient representation in the input space of categorical features. We improve this representation by incorporating the unlabeled samples through pseudo-labels. Furthermore, we propose a Smooth Adaptive Weighting mechanism in the target encoding to mitigate the issue of noisy and biased pseudo-labels. Experimental results on various datasets and comparisons with existing frameworks show that SAWTab yields best test accuracy on all datasets. We find that pseudo-labels can help improve the input space representation in the SSL setting, which enhances the generalization of the learning algorithm.
diff --git a/_talks/2025-01-24-data-science-ai-day.md b/_talks/2025-01-24-data-science-ai-day.md
new file mode 100644
index 0000000..f4d1bf8
--- /dev/null
+++ b/_talks/2025-01-24-data-science-ai-day.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Dieter Fox - \"Where is RobotGPT?\""
+date: "2025-01-24 14:00:00 -0700"
+permalink: "/talks/2025-01-24-data-science-ai-day/"
+slug: "2025-01-24-data-science-ai-day"
+start_time: "2:00 PM"
+end_time: "3:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "Union Ballroom"
+canceled: false
+speakers:
+ - name: "Data Science"
+ website: "https://datascience.utah.edu/events/2025/data-science-day/"
+ bio: "Dieter Fox is Senior Director of Robotics Research at NVIDIA and Professor in the Allen School of Computer Science & Engineering at the University of Washington, where he heads the UW Robotics and State Estimation Lab. Dieter’s research is in robotics and artificial intelligence, with a focus on learning and perception applied to problems such as robot manipulation, mapping, and object detection and tracking. He has published more than 200 technical papers and is the co-author of the textbook “Probabilistic Robotics”. He is a Fellow of the IEEE, AAAI, and ACM, and recipient of the 2020 IEEE Pioneer in Robotics and Automation Award and the 2023 IJCAI John McCarthy Award. He was an editor of the IEEE Transactions on Robotics, program co-chair of the 2008 AAAI Conference on Artificial Intelligence, and program chair of the 2013 Robotics: Science and Systems conference."
+ - name: "AI Day"
+ bio: "Dieter Fox is Senior Director of Robotics Research at NVIDIA and Professor in the Allen School of Computer Science & Engineering at the University of Washington, where he heads the UW Robotics and State Estimation Lab. Dieter’s research is in robotics and artificial intelligence, with a focus on learning and perception applied to problems such as robot manipulation, mapping, and object detection and tracking. He has published more than 200 technical papers and is the co-author of the textbook “Probabilistic Robotics”. He is a Fellow of the IEEE, AAAI, and ACM, and recipient of the 2020 IEEE Pioneer in Robotics and Automation Award and the 2023 IJCAI John McCarthy Award. He was an editor of the IEEE Transactions on Robotics, program co-chair of the 2008 AAAI Conference on Artificial Intelligence, and program chair of the 2013 Robotics: Science and Systems conference."
+speaker_names: "Data Science, AI Day"
+source_file: "_data/talks/2025-01-24-data-science-ai-day.toml"
+generated: true
+---
+
+
+
+The last years have seen astonishing progress in the capabilities of generative AI techniques, particularly in the areas of language and visual understanding and generation. Key to the success of these models are the use of image and text data sets of unprecedented scale along with models that are able to digest such large datasets. We are now seeing the first examples of leveraging such models to equip robots with open-world visual understanding and reasoning capabilities. Unfortunately, however, we have not achieved the RobotGPT moment; these models still struggle with reasoning about geometry and physical interactions in the real world, resulting in brittle performance on seemingly simple tasks such as manipulating objects in the open world. A crucial reason for this problem is the lack of data suitable to train powerful, general models for robot decision making and control. In this talk, I will discuss approaches to generating large datasets for training robot manipulation capabilities, with a focus on the role simulation can play in this context. I will show some of our prior work, where we demonstrated robust sim-to-real transfer of manipulation skills trained in simulation, and then discuss a promising direction toward training a model architecture that combines high-level, semantic, open-world reasoning, with low-level 3D robot policies.
diff --git a/_talks/2025-02-07-omkar-bhalerao.md b/_talks/2025-02-07-omkar-bhalerao.md
new file mode 100644
index 0000000..f74a13a
--- /dev/null
+++ b/_talks/2025-02-07-omkar-bhalerao.md
@@ -0,0 +1,25 @@
+---
+layout: "talk"
+title: "Triadic First-Order Logic Queries in Temporal Networks"
+date: "2025-02-07 13:30:00 -0700"
+permalink: "/talks/2025-02-07-omkar-bhalerao/"
+slug: "2025-02-07-omkar-bhalerao"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Omkar Bhalerao"
+speaker_names: "Omkar Bhalerao"
+source_file: "_data/talks/2025-02-07-omkar-bhalerao.toml"
+generated: true
+---
+
+
+
+Motif counting is a fundamental problem in network analysis, and there is a rich literature of theoretical and applied algorithms for this problem. Given a large input network G, a motif H is a small “pattern" graph indicative of special local structure. Motif/pattern mining involves finding all matches of this pattern in the input G. The simplest, yet challenging, case of motif counting is when H has three vertices, often called a triadic query. Recent work has focused on temporal graph mining, where the network G has edges with timestamps (and directions) and H has time constraints. Such networks are common representations for communication networks, citation networks, financial transactions, etc.
+Inspired by concepts in logic and database theory, we introduce the study of Thresholded First Order Logic (FOL) Motif Analysis for massive temporal networks. A typical triadic motif query asks for the existence of three vertices that form a desired temporal pattern. An FOL motif query is obtained by having both existence and universal quantifiers with thresholds. This allows for query semantics that can mine richer information from networks. A typical triadic query would be "find all triples of vertices u,v,w such that they form a triangle within one hour". A thresholded FOL query can express "find all pairs u,v such that for half of w where (u,w) formed an edge, (v,w) also formed an edge within an hour".
+
+We design the first algorithm, FOLTY, for mining thresholded triadic FOL queries, whose theoretical running time matches the best known running time for sparse graphs. Specifically, our algorithms run in time 𝑂 (m $\alpha \log \sigma_{\max}$). Here, $m$ is the number of temporal edges in the input graph, $\alpha$ is its degeneracy (maximum core number), and the $\sigma_{\max}$ is the maximum edge multiplicity. Our procedures can be efficiently implemented, and FOLTY has good empirical behavior. For example, we can answer triadic FOL queries on graphs with nearly 70M edges in less than an hour on commodity hardware. We believe that our work could start a new research direction in the classic well-studied problem of motif analysis.
diff --git a/_talks/2025-02-14-bao-wang.md b/_talks/2025-02-14-bao-wang.md
new file mode 100644
index 0000000..d76508f
--- /dev/null
+++ b/_talks/2025-02-14-bao-wang.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Conditional Flow Divergence Matching"
+date: "2025-02-14 13:30:00 -0700"
+permalink: "/talks/2025-02-14-bao-wang/"
+slug: "2025-02-14-bao-wang"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Bao Wang"
+speaker_names: "Bao Wang"
+source_file: "_data/talks/2025-02-14-bao-wang.toml"
+generated: true
+---
+
+
+
+Conditional flow matching (CFM) stands out as an efficient simulation-free approach for training flow-based generative models, achieving remarkable performance for data generation. However, CFM is insufficient to ensure accuracy in learning probability paths, and the learned vector field significantly violates the continuity equation governing probability flows. In response, we establish a new total-variation bound between the learned and ground-truth probability paths, showing that the gap between probability paths is bounded above by a combination of CFM loss and an associated divergence loss. This theoretical bound informs us to design a new objective to match both flow and divergence accompanied by an efficient implementation. Our new training approach improves the performance of the flow-based generative model by a noticeable margin without significantly raising the computational cost.
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
diff --git a/_talks/2025-02-21-chenglu-li.md b/_talks/2025-02-21-chenglu-li.md
new file mode 100644
index 0000000..28cb879
--- /dev/null
+++ b/_talks/2025-02-21-chenglu-li.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Teaching and Learning at the Human-Technology Frontier: The Power of Artificial Intelligence and Big Data"
+date: "2025-02-21 13:30:00 -0700"
+permalink: "/talks/2025-02-21-chenglu-li/"
+slug: "2025-02-21-chenglu-li"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Chenglu Li"
+ affiliation: "Utah Edu Psych"
+ website: "https://www.chengluli.com"
+speaker_names: "Chenglu Li"
+source_file: "_data/talks/2025-02-21-chenglu-li.toml"
+generated: true
+---
+
+
+
+In this talk, Chenglu will examine the rapidly evolving field of artificial intelligence in education (AIED) by focusing on three key research gaps: agentic AI needs, FAccT (fairness, accountability, and transparency) challenges, and issues related to computing supremacy. He will illustrate these challenges through his current project, ALTER-Math (AI-augmented Learning by Teaching to Enhance and Renovate Math Learning), a $10M initiative designed to accelerate middle school math learning using generative AI-powered solutions. Additionally, Chenglu will discuss his contributions to advancing learning and teaching through the development, evaluation, and dissemination of FAccT AI cyberinfrastructure. Finally, he will outline promising future directions for research and collaboration in this dynamic field.
diff --git a/_talks/2025-02-28-bernardo-modenesi.md b/_talks/2025-02-28-bernardo-modenesi.md
new file mode 100644
index 0000000..6fabc47
--- /dev/null
+++ b/_talks/2025-02-28-bernardo-modenesi.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "Unveiling Hidden Patterns in Agent Behavior with Discrete-Choice and Network Theory"
+date: "2025-02-28 13:30:00 -0700"
+permalink: "/talks/2025-02-28-bernardo-modenesi/"
+slug: "2025-02-28-bernardo-modenesi"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Bernardo Modenesi"
+ affiliation: "UU BioStats"
+ website: "https://sites.google.com/view/bmodenesi"
+speaker_names: "Bernardo Modenesi"
+source_file: "_data/talks/2025-02-28-bernardo-modenesi.toml"
+generated: true
+---
+
+
+
+Many datasets in data science stem from agents repeatedly making choices over time, with each choice leading to an observable outcome. In this talk, I introduce a novel approach to uncover latent agent heterogeneity, enhancing both our understanding of agent behavior and causal inference estimation. By combining discrete choice models with network theory, we develop a method to measure agent similarity based on their choice patterns. This results in a network-based unsupervised clustering technique that groups agents with similar behaviors—offering an interpretable alternative to black-box clustering models while maintaining explicit estimation assumptions. I will illustrate our approach using labor market data, where workers (agents) and jobs (choices) form a bipartite network, with worker-job matches represented as edges. By clustering workers based on their job choices, we can infer unobserved worker skills—a crucial factor in economic analysis. Through Bayesian estimation, we reveal latent worker groups, improving predictions of labor market outcomes and measuring labor market discrimination more effectively than models relying only on observable characteristics. This seminar will detail our methodological framework, estimation strategy, and practical applications for understanding and predicting agent-choice dynamics.
+
+Bonus project:
+In the final portion of the talk, I will pivot to a more informal discussion of a preliminary 'Model Ensemble Approach to Assessing Discrimination in Machine Learning Models' in the space of algorithmic fairness.
diff --git a/_talks/2025-03-21-jeff-phillips.md b/_talks/2025-03-21-jeff-phillips.md
new file mode 100644
index 0000000..d533a51
--- /dev/null
+++ b/_talks/2025-03-21-jeff-phillips.md
@@ -0,0 +1,32 @@
+---
+layout: "talk"
+title: "Data Science & AI Lecture Series"
+date: "2025-03-21 13:30:00 -0600"
+permalink: "/talks/2025-03-21-jeff-phillips/"
+slug: "2025-03-21-jeff-phillips"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Jeff Phillips"
+ affiliation: "UU KSoC"
+ website: "https://users.cs.utah.edu/~jeffp/"
+speaker_names: "Jeff Phillips"
+source_file: "_data/talks/2025-03-21-jeff-phillips.toml"
+generated: true
+---
+
+
+
+Robust statistics aims to compute quantities to represent data where a fraction of it may be arbitrarily corrupted. The most essential statistic is the mean, and in recent years,
+there has been a flurry of theoretical advancement for efficiently estimating the mean in high dimensions on corrupted data. While several algorithms have been proposed that
+achieve near-optimal error, they all rely on large data size requirements as a function of dimension.
+
+In this talk, we perform an extensive experimentation over various mean estimation techniques where data size might not meet this requirement due to the high-dimensional setting.
+For data with inliers generated from a Gaussian with known covariance, we find experimentally that several robust mean estimation techniques can practically improve upon the sample mean, with the quantum entropy scaling approach from Dong et.al. (NeurIPS 2019) performing consistently the best. However, this consistent improvement is conditioned on a couple of simple modifications to how the steps to prune outliers work in the high-dimension
+low-data setting, and when the inliers deviate significantly from Gaussianity. In fact, with these modifications, they are typically able to achieve roughly the same error as taking the sample mean of the uncorrupted inlier data, even with very low data size. In addition to
+controlled experiments on synthetic data, we also explore these methods on large language models, deep pretrained image models, and non-contextual word embedding models that do not necessarily have an inherent Gaussian distribution. Yet, in these settings, a mean point of a set of embedded objects is a desirable quantity to learn, and the data exhibits the high-dimension low-data setting studied in this paper. We show both the challenges of achieving this goal, and that our updated robust mean estimation methods can provide
+significant improvement over using just the sample mean.
diff --git a/_talks/2025-03-28-seth-pettie.md b/_talks/2025-03-28-seth-pettie.md
new file mode 100644
index 0000000..72fbb25
--- /dev/null
+++ b/_talks/2025-03-28-seth-pettie.md
@@ -0,0 +1,46 @@
+---
+layout: "talk"
+title: "Everything you always wanted to know about Cardinality"
+date: "2025-03-28 13:30:00 -0600"
+permalink: "/talks/2025-03-28-seth-pettie/"
+slug: "2025-03-28-seth-pettie"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Seth Pettie"
+ affiliation: "U Michigan CS"
+ website: "https://web.eecs.umich.edu/~pettie/"
+speaker_names: "Seth Pettie"
+source_file: "_data/talks/2025-03-28-seth-pettie.toml"
+generated: true
+---
+
+
+
+The Cardinality Estimation/Distinct Elements problem is to approximate
+the number of distinct elements in a data stream using a small
+probabilistic data structure called a "sketch". This problem has been
+studied for 40 years, has many industrial applications, and is
+featured prominently in most courses on Big Data algorithmics. It is
+therefore a real puzzle to explain why research on this popular and
+fundamental problem has been unusually slow.
+
+This talk presents a complete history of the Cardinality Estimation
+problem from Flajolet and Martin's seminal 1983 paper to the present,
+and includes an account of how the research community became
+fractured, delaying many natural developments by decades. I will
+present our recent efforts to achieve information-theoretically
+optimal cardinality sketches, which draws on two notions of
+"information" developed in the 20th century: Fisher information
+(governing optimal point estimation) and Shannon entropy (governing
+optimal space/communication).
+
+Joint work with Dingyu Wang.
+
+========
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
diff --git a/_talks/2025-04-04-sabyasachi-basu.md b/_talks/2025-04-04-sabyasachi-basu.md
new file mode 100644
index 0000000..41c44c0
--- /dev/null
+++ b/_talks/2025-04-04-sabyasachi-basu.md
@@ -0,0 +1,28 @@
+---
+layout: "talk"
+title: "\"Triangles, Communities, and Dense Subgraphs\""
+date: "2025-04-04 13:30:00 -0600"
+permalink: "/talks/2025-04-04-sabyasachi-basu/"
+slug: "2025-04-04-sabyasachi-basu"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Sabyasachi Basu"
+ affiliation: "UCSC"
+ website: "https://sites.google.com/view/sabyaucsc/home"
+speaker_names: "Sabyasachi Basu"
+source_file: "_data/talks/2025-04-04-sabyasachi-basu.toml"
+generated: true
+---
+
+
+
+In this talk, we will go over a few recent results on dense subgraph discovery. We aim to discover 'many' dense subgraphs of 'reasonable size' in real-world networks. We show that by leveraging triadic structure in graphs, one can do this efficiently without complicated distributional assumptions on the input. Our techniques bridge an important gap: most existing theory considers the setting where the number of pieces is a constant, whereas techniques that produce `satisfactory' decompositions rarely have density guarantees (and indeed, often give sparse, poorly connected subgraphs). We offer a community detection flavor to our results: we provide a new metric for the 'goodness' of communities in terms of density and show that the spectrum of graph matrices implies the existence of communities.
+
+A key goal of this talk is to unpack the several phrases in quotes in the preceding paragraph and offer some perspectives on why these are important (and sometimes difficult!). We also provide an algorithm that (provably) decomposes large social networks into dense subgraphs in minutes on regular laptops, and, time permitting, discuss the setting of overlapping subgraph detection using similar techniques.
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
diff --git a/_talks/2025-04-11-peter-jacobs.md b/_talks/2025-04-11-peter-jacobs.md
new file mode 100644
index 0000000..a69e602
--- /dev/null
+++ b/_talks/2025-04-11-peter-jacobs.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "Data Science & AI Lecture Series"
+date: "2025-04-11 13:30:00 -0600"
+permalink: "/talks/2025-04-11-peter-jacobs/"
+slug: "2025-04-11-peter-jacobs"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Peter Jacobs"
+ affiliation: "UU KSoC, Sandia"
+ website: "https://jacobs269.github.io"
+speaker_names: "Peter Jacobs"
+source_file: "_data/talks/2025-04-11-peter-jacobs.toml"
+generated: true
+---
+
+
+
+We study estimation of large discrete distributions under the structural assumption that they follow a Zipfian distribution, in which the ranking of alphabet items and/or level of decay need to be estimated from data. Empirical evidence for near Zipfian distributions has been found in diverse applications such as word and n-gram probability distributions in natural language text and chord probabilities in musical pieces. We introduce the Sort and Snap estimator for when the level of decay is known but the ranking function needs to be estimated, and show it is minimax in several high dimensional regimes. When both the ranking and decay level are unknown, we introduce an adaptive variant of Sort and Snap and show via Monte Carlo simulation that it outperforms state of the art discrete distribution estimators in these same high dimensional regimes. Our results motivate assessment of whether linguistically motivated marginal distributions for generating natural language that are claimed to be Zipfian in quantitative linguistics communities are truly Zipfian. Through Monte Carlo experiments on one such well-regarded distribution, Sort and Snap procedures lag behind even the simplest non-parametric estimator (empirical proportions), which brings into focus that this distribution thought to be Zipfian actually departs meaningfully from the Zipfian pattern.
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
diff --git a/_talks/2025-04-18-tucker-hermans.md b/_talks/2025-04-18-tucker-hermans.md
new file mode 100644
index 0000000..c3463ef
--- /dev/null
+++ b/_talks/2025-04-18-tucker-hermans.md
@@ -0,0 +1,28 @@
+---
+layout: "talk"
+title: "Stein Variational Inference for Robotic Learning and Control"
+date: "2025-04-18 13:30:00 -0600"
+permalink: "/talks/2025-04-18-tucker-hermans/"
+slug: "2025-04-18-tucker-hermans"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1"
+canceled: false
+speakers:
+ - name: "Tucker Hermans"
+ affiliation: "UU KSoC, NVIDIA"
+ website: "https://robot-learning.cs.utah.edu/thermans"
+speaker_names: "Tucker Hermans"
+source_file: "_data/talks/2025-04-18-tucker-hermans.toml"
+generated: true
+---
+
+
+
+Probabilistic inference, the problem of estimating a distribution given data, has been a central pillar of robotic algorithms for more than two decades. Inference techniques have defined the de facto standard for robotic localization, mapping, system calibration, and online error estimation for mobile robots. Academic work has shown how these same inference problem formulations and algorithms can be used to solve problems of planning and control.
+
+In this talk I will discuss how probabilistic inference techniques can be extended for use in robotic manipulation where models may come either from engineering first-principles or in the form of large neural networks. I will then give a brief dedication of Stein variational inference, a recent nonparametric technique for probabilistic inference that is easily parallelized on modern GPUs providing much faster inference times compared to more traditional Markov chain Monte Carlo methods. I will then show a few different applications of using Stein variational inference from my lab, including planning to goal distributions, adaptive control of magnetic manipulation, and generating diverse data for training from real-world robot failures.
+
+Talks will be held in WEB L112, and also streamed via the following Zoom link: https://utah.zoom.us/j/93909986581?pwd=d90LoHKoVAkagz1aCpJH1vTuaME9gG.1
diff --git a/_talks/2025-08-20-varun-shankar.md b/_talks/2025-08-20-varun-shankar.md
new file mode 100644
index 0000000..002cfbc
--- /dev/null
+++ b/_talks/2025-08-20-varun-shankar.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "Kernel Methods for Operator Learning"
+date: "2025-08-20 11:00:00 -0600"
+permalink: "/talks/2025-08-20-varun-shankar/"
+slug: "2025-08-20-varun-shankar"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "http://utah.zoom.us/my/vsutah"
+canceled: false
+speakers:
+ - name: "Varun Shankar"
+ affiliation: "Title: Kernel Methods for Operator Learning"
+speaker_names: "Varun Shankar"
+source_file: "_data/talks/2025-08-20-varun-shankar.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-08-27-konstantin-genin.md b/_talks/2025-08-27-konstantin-genin.md
new file mode 100644
index 0000000..348e626
--- /dev/null
+++ b/_talks/2025-08-27-konstantin-genin.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Predictions as Public Reasons"
+date: "2025-08-27 11:00:00 -0600"
+permalink: "/talks/2025-08-27-konstantin-genin/"
+slug: "2025-08-27-konstantin-genin"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/7824755969?pwd=SUplL1EwUmU0TE5JRkhyQ2dtNmsvdz09"
+canceled: false
+speakers:
+ - name: "Konstantin Genin"
+speaker_names: "Konstantin Genin"
+source_file: "_data/talks/2025-08-27-konstantin-genin.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-09-03-daniel-brown.md b/_talks/2025-09-03-daniel-brown.md
new file mode 100644
index 0000000..ef7c28f
--- /dev/null
+++ b/_talks/2025-09-03-daniel-brown.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Swarms, Emergent Behaviors, and Multi-Agent Systems"
+date: "2025-09-03 11:00:00 -0600"
+permalink: "/talks/2025-09-03-daniel-brown/"
+slug: "2025-09-03-daniel-brown"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Daniel Brown"
+speaker_names: "Daniel Brown"
+source_file: "_data/talks/2025-09-03-daniel-brown.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-09-10-bei-wang-phillips.md b/_talks/2025-09-10-bei-wang-phillips.md
new file mode 100644
index 0000000..a69fb00
--- /dev/null
+++ b/_talks/2025-09-10-bei-wang-phillips.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Talks will be held in LNCO 1100, and also streamed via the following Zoom link"
+date: "2025-09-10 11:00:00 -0600"
+permalink: "/talks/2025-09-10-bei-wang-phillips/"
+slug: "2025-09-10-bei-wang-phillips"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Bei Wang Phillips"
+speaker_names: "Bei Wang Phillips"
+source_file: "_data/talks/2025-09-10-bei-wang-phillips.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-09-17-aditya-bhaskara.md b/_talks/2025-09-17-aditya-bhaskara.md
new file mode 100644
index 0000000..d64608a
--- /dev/null
+++ b/_talks/2025-09-17-aditya-bhaskara.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Descent with Misaligned Gradients and Applications to Hidden Convexity"
+date: "2025-09-17 11:00:00 -0600"
+permalink: "/talks/2025-09-17-aditya-bhaskara/"
+slug: "2025-09-17-aditya-bhaskara"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Aditya Bhaskara"
+speaker_names: "Aditya Bhaskara"
+source_file: "_data/talks/2025-09-17-aditya-bhaskara.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-09-24-kenneth-blake-vernon.md b/_talks/2025-09-24-kenneth-blake-vernon.md
new file mode 100644
index 0000000..85f8605
--- /dev/null
+++ b/_talks/2025-09-24-kenneth-blake-vernon.md
@@ -0,0 +1,28 @@
+---
+layout: "talk"
+title: "Indirect dating with mixture density networks"
+date: "2025-09-24 11:00:00 -0600"
+permalink: "/talks/2025-09-24-kenneth-blake-vernon/"
+slug: "2025-09-24-kenneth-blake-vernon"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Kenneth Blake Vernon"
+speaker_names: "Kenneth Blake Vernon"
+source_file: "_data/talks/2025-09-24-kenneth-blake-vernon.toml"
+generated: true
+---
+
+
+
+It is an astonishing fact about the world today that no one can say precisely how many people actually live on our planet, even though it would presumably be useful to have such information to plan for climate change and disaster risk management, among other things. Luckily for us, demographers and spatial data scientists are keenly aware of this problem and have devised sophisticated methods for interpolating population in these regions based on their built area, typically measured using remote sensing technology.
+
+As it happens, this is exactly the reasoning applied by archaeologists seeking to reconstruct population sizes in the past. Unfortunately, archaeology faces an additional challenge here, since the archaeological record is a palimpsest of built area, representing continuous human settlement over decades, centuries, and sometimes even millennia. So, reconstructing population sizes across a region of interest requires that archaeologists also develop a chronology for that region at the same time.
+
+A region’s chronology can be represented by a probability density function, p(t), with well-dated archaeological materials – like tree-rings and radiocarbon samples - assumed to be random draws from that distribution. Here, we propose to estimate p using a deep-learning extension to the mixture model known as a Mixture Density Network. With this model, we condition the chronology on diagnostic data X, treating mixture parameters as unknown functions of X that can be estimated using a simple multilayer perceptron. Specifically, we use the density of X in the area around each sampled date to estimate the mixture parameters. This allows us to interpolate dates at under-sampled sites and to build a better representation of the regional chronology, one that is based, in theory, on the totality of the archaeological record.
+
+As an example, we fit an MDN to the distribution of tree-rings in the Mesa Verde region of southwestern Colorado using the spatial distribution of ceramics to estimate mixture parameters. An important ancillary goal of this research is to develop software tools scientists can use to train MDNs on their own data without also having to learn the intricacies of AI development and testing.
diff --git a/_talks/2025-10-01-luis-garcia.md b/_talks/2025-10-01-luis-garcia.md
new file mode 100644
index 0000000..38a71d3
--- /dev/null
+++ b/_talks/2025-10-01-luis-garcia.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "A Trip to the Neural Frontier: Neurosymbolic Sensor Fusion for Trustworthy AI-Enabled Neural Interventions"
+date: "2025-10-01 11:00:00 -0600"
+permalink: "/talks/2025-10-01-luis-garcia/"
+slug: "2025-10-01-luis-garcia"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Luis Garcia"
+speaker_names: "Luis Garcia"
+source_file: "_data/talks/2025-10-01-luis-garcia.toml"
+generated: true
+---
+
+
+
+Advances in computing and neuroscience are converging to enable new forms of recording and stimulation in naturalistic environments, or “neuroscience in the wild.” At the core of this effort is the ability to capture the human sensory experience, synchronized with intracranial recordings, to understand how brain activity relates to behavior in real-world contexts. I will share recent progress in building end-to-end pipelines for trustworthy brain–behavior research. I will begin with multimodal datasets we have collected during spatial navigation tasks, showing how hippocampal activity reflects context shifts such as doorways and landmarks. I will then discuss our work on event-driven synchronization, where large language models and human-in-the-loop oversight transform raw multimodal recordings into aligned, analyzable events. Building on this foundation, we are developing new platforms that combine neural signals, mobile sensing, and immersive environments to capture context in real time. I will also discuss our initial explorations on how extended reality can help manage the inherent noisiness of natural settings, and how privacy risks emerge when enabling sensors in sensitive environments. Together, these efforts outline a pathway toward reproducible, explainable, and privacy-preserving neural interventions in everyday life. Talks will be held in LNCO 1100, and also streamed via the following Zoom link: https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1
diff --git a/_talks/2025-10-15-vineet-pandey.md b/_talks/2025-10-15-vineet-pandey.md
new file mode 100644
index 0000000..39bc808
--- /dev/null
+++ b/_talks/2025-10-15-vineet-pandey.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Designing human-centered systems that yield data that is minimal, relevant, and actionable"
+date: "2025-10-15 11:00:00 -0600"
+permalink: "/talks/2025-10-15-vineet-pandey/"
+slug: "2025-10-15-vineet-pandey"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Vineet Pandey"
+speaker_names: "Vineet Pandey"
+source_file: "_data/talks/2025-10-15-vineet-pandey.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-10-22-kenneth-marino.md b/_talks/2025-10-22-kenneth-marino.md
new file mode 100644
index 0000000..b5ab9c6
--- /dev/null
+++ b/_talks/2025-10-22-kenneth-marino.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "VLM Agents"
+date: "2025-10-22 11:00:00 -0600"
+permalink: "/talks/2025-10-22-kenneth-marino/"
+slug: "2025-10-22-kenneth-marino"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: true
+speakers:
+ - name: "Kenneth Marino"
+speaker_names: "Kenneth Marino"
+source_file: "_data/talks/2025-10-22-kenneth-marino.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-10-29-anna-fariha.md b/_talks/2025-10-29-anna-fariha.md
new file mode 100644
index 0000000..df1c872
--- /dev/null
+++ b/_talks/2025-10-29-anna-fariha.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Understanding Data through Change Summarization and Causal Disparity Explanations"
+date: "2025-10-29 11:00:00 -0600"
+permalink: "/talks/2025-10-29-anna-fariha/"
+slug: "2025-10-29-anna-fariha"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Anna Fariha"
+speaker_names: "Anna Fariha"
+source_file: "_data/talks/2025-10-29-anna-fariha.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-11-05-jenny-lin.md b/_talks/2025-11-05-jenny-lin.md
new file mode 100644
index 0000000..8ca9b1b
--- /dev/null
+++ b/_talks/2025-11-05-jenny-lin.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Talks will be held in LNCO 1100, and also streamed via the following Zoom link"
+date: "2025-11-05 11:00:00 -0700"
+permalink: "/talks/2025-11-05-jenny-lin/"
+slug: "2025-11-05-jenny-lin"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Jenny Lin"
+speaker_names: "Jenny Lin"
+source_file: "_data/talks/2025-11-05-jenny-lin.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-11-12-kyle-dawson-tyler-hagen.md b/_talks/2025-11-12-kyle-dawson-tyler-hagen.md
new file mode 100644
index 0000000..93b428a
--- /dev/null
+++ b/_talks/2025-11-12-kyle-dawson-tyler-hagen.md
@@ -0,0 +1,27 @@
+---
+layout: "talk"
+title: "DESI: Disentangling Cosmology from Observational Artifacts"
+date: "2025-11-12 10:00:00 -0700"
+permalink: "/talks/2025-11-12-kyle-dawson-tyler-hagen/"
+slug: "2025-11-12-kyle-dawson-tyler-hagen"
+start_time: "10:00 AM"
+end_time: "11:00 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+zoom: "https://utexas.zoom.us/j/87159746528?pwd=ouQu8lN9ARbb6aRFpvdf6Ddb1Oqa8B.1"
+canceled: false
+speakers:
+ - name: "Kyle Dawson"
+ website: "https://profiles.faculty.utah.edu/u0634757"
+ - name: "Tyler Hagen"
+speaker_names: "Kyle Dawson, Tyler Hagen"
+source_file: "_data/talks/2025-11-12-kyle-dawson-tyler-hagen.toml"
+generated: true
+---
+
+
+
+The Dark Energy Spectroscopic Instrument (DESI) has concluded three years of observation, leading to the largest spectroscopic galaxy sample ever produced. In combination with other cosmological probes, these measurements reveal hints of new physics beyond the standard cosmological model. In this talk, we will first present the observations and key measurements that led to these new constraints. We will then describe the role that neural networks and random forests play in this analysis and our tests of these machine learning algorithms against more physically-motivated, linear models.
+
+Where:
+The location and time is moved for this talk so it can coincide with the CosmicAI seminar. It will take place in WEB 3780 (the Evans Conference room in SCI) and at 10am. It will also be on Zoom at this *new* link:
diff --git a/_talks/2025-11-19-vivek-srikumar.md b/_talks/2025-11-19-vivek-srikumar.md
new file mode 100644
index 0000000..5011241
--- /dev/null
+++ b/_talks/2025-11-19-vivek-srikumar.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Talks will be held in LNCO 1100, and also streamed via the following Zoom link"
+date: "2025-11-19 11:00:00 -0700"
+permalink: "/talks/2025-11-19-vivek-srikumar/"
+slug: "2025-11-19-vivek-srikumar"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Vivek Srikumar"
+speaker_names: "Vivek Srikumar"
+source_file: "_data/talks/2025-11-19-vivek-srikumar.toml"
+generated: true
+---
+
+
diff --git a/_talks/2025-12-03-data-visualization-101.md b/_talks/2025-12-03-data-visualization-101.md
new file mode 100644
index 0000000..91444d2
--- /dev/null
+++ b/_talks/2025-12-03-data-visualization-101.md
@@ -0,0 +1,20 @@
+---
+layout: "talk"
+title: "Madison Golden and Kaylee Alexander"
+date: "2025-12-03 11:00:00 -0700"
+permalink: "/talks/2025-12-03-data-visualization-101/"
+slug: "2025-12-03-data-visualization-101"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LNCO 1100"
+zoom: "https://utah.zoom.us/j/81370106930?pwd=5rAKB2C2SrgkOprGOGuAtzeF5xxbbT.1"
+canceled: false
+speakers:
+ - name: "Data Visualization 101"
+speaker_names: "Data Visualization 101"
+source_file: "_data/talks/2025-12-03-data-visualization-101.toml"
+generated: true
+---
+
+
diff --git a/_talks/2026-01-09-marina-kogan.md b/_talks/2026-01-09-marina-kogan.md
new file mode 100644
index 0000000..0bbed5d
--- /dev/null
+++ b/_talks/2026-01-09-marina-kogan.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Human-Centered Data Science for Crisis Informatics"
+date: "2026-01-09 13:30:00 -0700"
+permalink: "/talks/2026-01-09-marina-kogan/"
+slug: "2026-01-09-marina-kogan"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/85983626630"
+canceled: false
+speakers:
+ - name: "Marina Kogan"
+ affiliation: "UU KSoC, RAI Faculty Fellow"
+ bio: "Marina Kogan is an Assistant Professor at Kahlert School of Computing at University of Utah. She works in the areas of social computing, crisis informatics, and human-centered data science. Her background in Sociology and Computer Science informs her focus on studying online coordination, collective problem-solving, and information flows during disruption events such as disasters arising from natural hazards and political crises. Her most recent work engages with issues of mis/disinformation, conspiracy theories, and state-sponsored information operations."
+speaker_names: "Marina Kogan"
+source_file: "_data/talks/2026-01-09-marina-kogan.toml"
+generated: true
+---
+
+
+
+Social media platforms have been increasingly used by the public in crisis situations, partly because they upend the traditional top-down broadcasting model of risk communication. Instead, social media platforms facilitate a two-way information exchange between the official response channels and the general public, enabling more participatory crisis communication, as well as coordination and self-organization among the public. In this more complex information ecosystem, understanding the flow of information is crucial to supporting those affected and preventing malicious actors from capitalizing on the uncertainty. However, the study of such information flows is challenging, as the high-tempo, high-volume convergent nature of crisis events produces vast amounts of social media data, necessitating the use of the data science methods. On the other hand, to glean meaningful insight from the crisis-related social media activity, it is necessary to use methods that account for the complex social context of the user activity. In this talk, Kogan will show how the Human-Centered Data Science (HCDS) provides methodological approaches that both harness the power of computational methods and account for the highly situated nature of social media activity in disruption. She will focus on sequence-based approaches as examples of HCDS methods in two empirical studies: analysis of attention-garnering information during a natural disaster and investigation of behavioral signatures in coordinated information operations.
diff --git a/_talks/2026-01-23-andrew-mcnutt.md b/_talks/2026-01-23-andrew-mcnutt.md
new file mode 100644
index 0000000..031cb69
--- /dev/null
+++ b/_talks/2026-01-23-andrew-mcnutt.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Linters as Socio Technical Systems"
+date: "2026-01-23 13:30:00 -0700"
+permalink: "/talks/2026-01-23-andrew-mcnutt/"
+slug: "2026-01-23-andrew-mcnutt"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/85983626630"
+canceled: false
+speakers:
+ - name: "Andrew McNutt"
+ bio: "Andrew McNutt is an assistant professor at University of Utah’s Kahlert School of Computing and Scientific Computing and Imaging Institute. His research lives in the union of human computer interaction, visualization, and programming interfaces. It considers topics like creative coding, domain-specific languages, theories of visualization, and critical theory. He completed his PhD at University of Chicago, and a Post Doc at University of Washington. His work is funded by the NSF and is a Seibel Foundation Scholar. His work has won awards at top visualization and HCI conferences."
+speaker_names: "Andrew McNutt"
+source_file: "_data/talks/2026-01-23-andrew-mcnutt.toml"
+generated: true
+---
+
+
+
+Interfaces—whether they are for data analysis, programming, or any other activities—exist within specific communities of practice. The norms and standards of those groups inscribe themselves in the form of those tools; implicitly driving what is and is not possible. In this talk I will explore how linters (a spell checker-like tool used in programming) can be used to interrogate this arrangement. In doing so I will describe recent and on-going efforts relating to application of linters to a variety of domains, including visualization, color palettes, and social media posts. Through this discussion, I will argue that using linters as a critical lens allows us to examine the technological world around us in a new light (offering new opportunities for research and design), and therein explore the values that we manifest in our tool design and selection.
diff --git a/_talks/2026-02-06-makoto-kelp.md b/_talks/2026-02-06-makoto-kelp.md
new file mode 100644
index 0000000..d95c105
--- /dev/null
+++ b/_talks/2026-02-06-makoto-kelp.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Navigating Advances and Inflections in Machine Learning for Atmospheric Chemistry Modeling"
+date: "2026-02-06 13:30:00 -0700"
+permalink: "/talks/2026-02-06-makoto-kelp/"
+slug: "2026-02-06-makoto-kelp"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/85983626630"
+canceled: false
+speakers:
+ - name: "Makoto Kelp"
+ bio: "Makoto Kelp is an Assistant Professor in the Department of Atmospheric Sciences and a Fellow in the Wilkes Center for Climate Science & Policy at the University of Utah. His research focuses on the intersections of atmospheric chemistry, fires, and human-environmental systems, with an emphasis on using data-driven methods and machine learning techniques. He earned his PhD from Harvard University and conducted his Post Doc at Stanford University as a NOAA Climate & Global Change Fellow."
+speaker_names: "Makoto Kelp"
+source_file: "_data/talks/2026-02-06-makoto-kelp.toml"
+generated: true
+---
+
+
+
+Global climate and Earth system models rarely include comprehensive atmospheric chemistry because of its high computational cost. A bottleneck is the chemical solver that integrates the large-dimensional coupled systems of kinetic equations describing the chemical mechanism. In recent years, machine learning (ML) methods have been proposed as a potentially transformative approach to reducing this cost by replacing traditional solvers with fast emulators. However, early efforts showed that ML-based chemical solvers often suffer from rapid error growth and instability. In this talk, I will review the evolving landscape of ML for atmospheric chemistry modeling over the past decade and how its trajectory mirrors broader developments in climate and weather AI. I begin with detailing how to achieve stable emulation in 0-D box models and then show how these principles translate to complex global atmospheric models. I conclude by discussing the current state of ML for modeling atmospheric chemistry and outlining how mechanistic interpretability in geospatial foundation models may offer a promising future line of inquiry.
diff --git a/_talks/2026-02-17-juliana-freire.md b/_talks/2026-02-17-juliana-freire.md
new file mode 100644
index 0000000..de8b858
--- /dev/null
+++ b/_talks/2026-02-17-juliana-freire.md
@@ -0,0 +1,33 @@
+---
+layout: "talk"
+title: "Dataset Discovery and Integration in the Era of Large Language Models"
+date: "2026-02-17 11:00:00 -0700"
+permalink: "/talks/2026-02-17-juliana-freire/"
+slug: "2026-02-17-juliana-freire"
+start_time: "11:00 AM"
+end_time: "12:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780 (Evans)"
+zoom: "https://utah.zoom.us/j/85983626630"
+canceled: false
+speakers:
+ - name: "Juliana Freire"
+ website: "https://engineering.nyu.edu/faculty/juliana-freire"
+speaker_names: "Juliana Freire"
+source_file: "_data/talks/2026-02-17-juliana-freire.toml"
+generated: true
+---
+
+
+
+**Abstract:**
+The proliferation of structured data across open-data portals, the web, and enterprise data lakes presents unprecedented opportunities for scientific discovery and data-driven decision-making. However, realizing this potential requires solving fundamental challenges in data discovery and integration: How do we find relevant datasets among millions of candidates? How do we understand their semantics to integrate them? And how do we build systems that are scalable, accurate, and cost-effective?
+
+In this talk, I will present our recent work addressing these challenges by combining techniques from data management, visualization, HCI, and modern language models. Our work is motivated by real-world problems across domains: supporting biomedical researchers in integrating heterogeneous datasets, enabling dataset search over urban data for policy analysis and planning, and augmenting training data to improve machine learning model performance.
+
+First, I will describe how we are reimagining dataset discovery by designing specialized search engines, developing methods to automatically derive metadata, and introducing novel data-driven queries that go beyond keywords to support complex information needs. Then, I will turn to data integration, showing how we can leverage the semantic power of Large Language Models (LLMs) to match schemas with state-of-the-art accuracy. I will also argue for the critical importance of human-AI collaboration, demonstrating how visual analytics can effectively place the user in the loop to guide and verify these complex processes.
+
+Finally, I will reflect on how LLMs are fundamentally changing computer science research. They make previously intractable problems like semantic data discovery and integration tractable, yet they behave more like natural phenomena, exhibiting variability and non-determinism. This shift forces us to move beyond deterministic algorithmic thinking and to practice computer science as a science, adopting empirical research methods to understand and leverage these powerful but unpredictable tools.
+
+**Bio:**
+Juliana Freire is an Institute Professor at the Tandon School of Engineering and Professor of Computer Science and Data Science at New York University, where she co-directs the Visualization Imaging and Data Analysis (VIDA) Center. Her research develops methods and systems that enable a wide range of users to obtain trustworthy insights from data. It spans topics in large-scale data analysis and integration, visualization, machine learning, provenance management, and web information discovery, addressing application areas including urban analytics, predictive modeling, computational reproducibility, and biomedical data harmonization. She has co-authored over 250 papers, including 12 award winners and a test-of-time award. She served as elected chair of ACM SIGMOD and as a council member of the Computing Community Consortium (CCC), and was the NYU lead investigator for the Moore-Sloan Data Science Environment. She is a Fellow of the ACM and AAAS, and a winner of the ACM SIGMOD Contributions Award. Her work has been supported by funding agencies and industry partners, including the National Science Foundation, DARPA, ARPA-H, the Department of Energy, the National Institutes of Health, and technology companies such as Google, Amazon, Microsoft Research, and IBM. Freire received her Ph.D. and M.Sc. degrees in computer science from the State University of New York at Stony Brook and her B.S. degree in computer science from the Federal University of Ceará in Brazil.
diff --git a/_talks/2026-02-19-dinesh-manocha.md b/_talks/2026-02-19-dinesh-manocha.md
new file mode 100644
index 0000000..05bbd54
--- /dev/null
+++ b/_talks/2026-02-19-dinesh-manocha.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Robot Navigation in the Wild"
+date: "2026-02-19 15:30:00 -0700"
+permalink: "/talks/2026-02-19-dinesh-manocha/"
+slug: "2026-02-19-dinesh-manocha"
+start_time: "3:30 PM"
+end_time: "4:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Dinesh Manocha"
+ bio: "Prof. Dinesh Manocha is Paul Chrisman-Iribe Chair in Computer Science & ECE and Distinguished University Professor at University of Maryland College Park. His research interests include virtual environments, physics-based modeling, and robotics. His group has developed several software packages that are standard and licensed to 60+ commercial vendors. He has published more than 850 papers & supervised 63 PhD dissertations. He is a Fellow of AAAI, AAAS, ACM, IEEE, and NAI and member of ACM SIGGRAPH and IEEE VR Academies, and Bézier Award from Solid Modeling Association. He received the Distinguished Alumni Award from IIT Delhi the Distinguished Career in Computer Science Award from Washington Academy of Sciences. He was a co-founder of Impulsonic, a developer of physics-based audio simulation technologies, which was acquired by Valve Inc in November of 2016."
+speaker_names: "Dinesh Manocha"
+source_file: "_data/talks/2026-02-19-dinesh-manocha.toml"
+generated: true
+---
+
+
+
+In the last few decades, most robotics success stories have been limited to structured or controlled environments. A major challenge is to develop robot systems that can operate in complex or unstructured environments corresponding to homes, dense traffic, outdoor terrains, public places, etc. In this talk, we give an overview of our ongoing work on developing robust planning and navigation technologies that use recent advances in computer vision, sensor technologies, machine learning, and motion planning algorithms. We present new methods that utilize multi-modal observations from an RGB camera, 3D LiDAR, and robot odometry for scene perception, along with deep reinforcement learning for reliable planning. The latter is also used to compute dynamically feasible and spatial aware velocities for a robot navigating among mobile obstacles and uneven terrains. We have integrated these methods with wheeled robots, home robots, and legged platforms and highlight their performance in crowded indoor scenes, home environments, and dense outdoor terrains.
diff --git a/_talks/2026-02-24-paul-parsons.md b/_talks/2026-02-24-paul-parsons.md
new file mode 100644
index 0000000..6d69677
--- /dev/null
+++ b/_talks/2026-02-24-paul-parsons.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Visualization and Judgment: Human-Centered Computing in Data-Rich Practice"
+date: "2026-02-24 10:30:00 -0700"
+permalink: "/talks/2026-02-24-paul-parsons/"
+slug: "2026-02-24-paul-parsons"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Paul Parsons"
+ bio: "Paul Parsons is an Associate Professor in the School of Applied and Creative Computing at Purdue University. His research focuses on interactive visualization systems and interfaces and how they are designed and used in complex sociotechnical settings. His work at the intersection of design practice and data visualization has been recognized with an NSF CAREER award and a recent IEEE VIS best-paper recognition. His broader research program has also been supported by NSF and NASA funding. His work appears in leading venues such as IEEE TVCG and ACM CHI. He leads the Design, Visualization, & Cognition (DVC) Lab, where his group studies how practitioners think, create, and collaborate in data-rich domains shaped by uncertainty, complexity, and ambiguity, drawing on perspectives such as design cognition, judgment and decision-making, and joint cognitive systems. He collaborates widely across disciplines at Purdue and beyond, including through the NSF-funded Cyberinfrastructure Center of Excellence SGX3 and the NASA-funded Resilient Extra Terrestrial Habitats Institute (RETHi)."
+speaker_names: "Paul Parsons"
+source_file: "_data/talks/2026-02-24-paul-parsons.toml"
+generated: true
+---
+
+
+
+Visualization and computational systems are increasingly powerful, but their impact depends on a persistent, often under-specified factor—human judgment. Designers frame problems and negotiate constraints; users interpret, challenge, and coordinate action around system outputs over time, under uncertainty and constraint. In this talk, I present a research agenda on visualization and judgment in data-rich work, grounded in empirical studies and design-oriented analyses across three contexts. First, I study data visualization design practice, showing how professional designers frame problems and co-evolve problem and solution spaces—work that is often invisible in pipeline-oriented accounts of visualization. Second, I examine expert judgment in complex sociotechnical settings, where visualization, procedures, and automation reshape decision spaces and can either support adaptive performance or encourage brittle reliance. Third, I extend these insights to scientific cyberinfrastructure, where platforms exposing advanced computation and data services succeed or fail based on whether diverse communities can understand, adopt, and sustain capabilities in practice. This perspective complements advances in modeling, simulation, and visual analytics by surfacing design constraints, evaluation targets, and failure modes that matter as systems intersect with the constraints and contingencies of practice. I close with directions for judgment-aware visualization and human–AI support that preserve interpretability, accountability, and coordination in consequential domains.
diff --git a/_talks/2026-02-26-zezhong-wang.md b/_talks/2026-02-26-zezhong-wang.md
new file mode 100644
index 0000000..d023a46
--- /dev/null
+++ b/_talks/2026-02-26-zezhong-wang.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Designing Narrative-Driven Data Experiences"
+date: "2026-02-26 10:30:00 -0700"
+permalink: "/talks/2026-02-26-zezhong-wang/"
+slug: "2026-02-26-zezhong-wang"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "Evans Conference Room (WEB 3780)"
+canceled: false
+speakers:
+ - name: "Zezhong Wang"
+ bio: "Dr. Zezhong Wang is currently a Postdoctoral Fellow at the Interactive Experiences Lab (ixLab) at Simon Fraser University, Canada, working with Dr. Sheelagh Carpendale. His research integrates visual design, data visualization, and HCI to foster public engagement with data. He holds a PhD from the University of Edinburgh (2022), where his dissertation, Creating Data Comics for Data-Driven Storytelling, received an IEEE VGTC Best Visualization Dissertation Award Honorable Mention in 2023. Zezhong actively engages in interdisciplinary collaboration, working alongside artists, computer scientists, healthcare professionals, and environmental scientists to bridge the gap between complex data and human experience."
+speaker_names: "Zezhong Wang"
+source_file: "_data/talks/2026-02-26-zezhong-wang.toml"
+generated: true
+---
+
+
+
+In today’s data-saturated world, people are facing an “infodemic” of information overload. As public demand to interpret and use data grows, so does the urgency to rethink how we help broader audiences engage with data in meaningful ways. This talk explores how we can reconnect data with its storytelling roots to support understanding and agency. Dr. Wang will present research at the intersection of visual design, data visualization, and human-computer interaction, showing how interdisciplinary collaboration can open up new modes of communication. He will share empirical findings on how visual data narratives, such as data comics, can improve comprehension and engagement, as well as methods for crafting visual data stories that connect data to context and lived experience. Examples will draw from cross-disciplinary projects in environmental and healthcare data storytelling. The talk will conclude with future directions toward embedding narrative-driven data experiences into everyday tasks and decision-making.
diff --git a/_talks/2026-02-27-kenny-marino.md b/_talks/2026-02-27-kenny-marino.md
new file mode 100644
index 0000000..9eb7932
--- /dev/null
+++ b/_talks/2026-02-27-kenny-marino.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Agents: Hype or Opportunity"
+date: "2026-02-27 13:40:00 -0700"
+permalink: "/talks/2026-02-27-kenny-marino/"
+slug: "2026-02-27-kenny-marino"
+start_time: "1:40 PM"
+end_time: "2:40 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/85983626630"
+canceled: false
+speakers:
+ - name: "Kenny Marino"
+ bio: "Kenneth Marino joined the Kahlert School of Computing at the University of Utah as an Assistant Professor in Fall 2025. His research focuses on integrating multimodal language models into embodied agent problems, including computer use, games, and robotics. Previously, he was a Research Scientist at Google DeepMind in NYC, where he worked on retrieval and embodied reasoning with language. He earned his PhD in 2021 from Carnegie Mellon University, advised by Abhinav Gupta, with a thesis on incorporating semantic knowledge into embodied systems. He received his undergraduate degree from the Georgia Institute of Technology."
+speaker_names: "Kenny Marino"
+source_file: "_data/talks/2026-02-27-kenny-marino.toml"
+generated: true
+---
+
+
+
+Are so-called "AI Agents" a fad or a potentially impactful research area enabled by the rapid progress in large language models? In this talk I will try to strip away the marketing copy and look at what an agent actually is, returning to the classical understanding of the word, and investigate how powerful new language models can present new opportunities for research in embodied decision making. We begin by recounting the places where language models have been useful in agent-like problems, then looking at the emerging environments for investigating VLM/LLM agents including computer use and robotics, and finally discussing the frontier research challenges of LLM agents.
diff --git a/_talks/2026-03-02-bogdan-raita.md b/_talks/2026-03-02-bogdan-raita.md
new file mode 100644
index 0000000..1a8725f
--- /dev/null
+++ b/_talks/2026-03-02-bogdan-raita.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "Solving Linear PDE by Machine Learning and Commutative Algebra"
+date: "2026-03-02 16:00:00 -0700"
+permalink: "/talks/2026-03-02-bogdan-raita/"
+slug: "2026-03-02-bogdan-raita"
+start_time: "4:00 PM"
+end_time: "5:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "LCB 222"
+canceled: false
+speakers:
+ - name: "Bogdan Raita"
+speaker_names: "Bogdan Raita"
+source_file: "_data/talks/2026-03-02-bogdan-raita.toml"
+generated: true
+---
+
+
+
+We use the theory of linear pde systems with constant coefficients (Malgrange, Palamodov, Pommaret, Sturmfels) to implement a machine learning algorithm which generates solutions to arbitrary linear pdes. Since we preprocess the equations with computer algebra, our methods are applicable to arbitrary pde systems, irrespective of type (elliptic, hyperbolic, etc.) or order. We test our method for classical equations (wave, heat, Laplace) and discuss future applications to equations describing wave-related phenomena, for example direct and inverse problems involving Maxwell’s system and the elasticity equations.
diff --git a/_talks/2026-03-03-grace-guo.md b/_talks/2026-03-03-grace-guo.md
new file mode 100644
index 0000000..ef1068b
--- /dev/null
+++ b/_talks/2026-03-03-grace-guo.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Concepts and Counterfactuals: Human-Centered Interpretability in the Age of Foundation Models"
+date: "2026-03-03 10:30:00 -0700"
+permalink: "/talks/2026-03-03-grace-guo/"
+slug: "2026-03-03-grace-guo"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "Evans Conference Room (WEB 3780)"
+canceled: false
+speakers:
+ - name: "Grace Guo"
+ bio: "Grace Guo is a Postdoctoral Fellow at Harvard University’s School of Engineering and Applied Sciences (SEAS). She received her Ph.D. in Human-Centered Computing from the Georgia Institute of Technology, where she was advised by Alex Endert. Her research sits at the intersection of visualization, explainable AI, and human-centered machine learning, with a focus on how AI interpretability tools can be designed for domain experts. To this end, Grace has collaborated with experts across healthcare, education, immunobiology, astrophysics, and causal analytics. She has previously worked at the Pacific Northwest National Laboratory and IBM Research, where she was awarded the IBM PhD Fellowship for her work on developing CausalVis. In her free time, she enjoys reading science fiction and mystery novels."
+speaker_names: "Grace Guo"
+source_file: "_data/talks/2026-03-03-grace-guo.toml"
+generated: true
+---
+
+
+
+Foundation models are increasingly deployed in high-stakes domains, yet their scale and opacity challenge traditional notions of AI interpretability. In this talk, I present two complementary strategies for human-centered interpretability: reasoning through concepts and probing through counterfactuals. I first present MiMICRI, a visualization tool developed with doctors at Cleveland Clinic that enables them to interactively create counterfactual medical images to examine how anatomical changes influence model predictions. By grounding explanations in domain-relevant visual features, this tool helps experts reason about model behavior using their established medical knowledge. Next, I will introduce Concept2Concept, a framework for auditing text-to-image models by characterizing their outputs as distributions over named, interpretable concepts. By analyzing the metrics of concept frequency, stability, and co-occurrence, we uncover hidden and sometimes harmful associations in image generation models and real-world training datasets. Finally, I conclude with my research agenda for developing new visualization tools and theoretical foundations that address the ongoing challenges of auditing and aligning the foundation models of today.
diff --git a/_talks/2026-03-05-josh-levine.md b/_talks/2026-03-05-josh-levine.md
new file mode 100644
index 0000000..e1539f3
--- /dev/null
+++ b/_talks/2026-03-05-josh-levine.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Extracting, Visualizing, and Analyzing Topological Features with Discrete Representations"
+date: "2026-03-05 10:30:00 -0700"
+permalink: "/talks/2026-03-05-josh-levine/"
+slug: "2026-03-05-josh-levine"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "Evans Conference Room (WEB 3780)"
+canceled: false
+speakers:
+ - name: "Josh Levine"
+ bio: "Joshua A. Levine is an associate professor in the Department of Computer Science at University of Arizona. Prior to starting at Arizona in 2016, he was an assistant professor at Clemson University from 2012 to 2016, and he is a postdoctoral alumnus of the University of Utah’s SCI Institute, 2009 to 2012. He is a recipient of the 2018 DOE Early Career award. He received his PhD in Computer Science from The Ohio State University in 2009 after completing BS degrees in Computer Engineering and Mathematics in 2003 and an MS in Computer Science in 2004 from Case Western Reserve University. His research and teaching interests include visualization, topological analysis, geometric modeling, and computer graphics."
+speaker_names: "Josh Levine"
+source_file: "_data/talks/2026-03-05-josh-levine.toml"
+generated: true
+---
+
+
+
+Topological features provide multi-scale summaries of the behavior of continuous data from diverse applications ranging from astrophysics to medicine. Nevertheless, computing them robustly is challenging due to numerical precision issues. A promising strategy is to first convert the input to a discrete representation that satisfies criteria introduced by Forman's discrete Morse theory. While numerous approaches exist to discretize the restricted case of gradient fields from scalar data, state-of-the-art algorithms for the general case of vector fields require expensive optimization procedures. In this talk, I will present recent work that uses local evaluation to create discrete vector fields in linear time from two-dimensional, triangulated vector fields. I will also frame this work within my contributions to the Topological ToolKit (TTK), an open source software platform for topological data analysis, led by collaborators at UPMC Sorbonne.
diff --git a/_talks/2026-03-16-tenghao-huang.md b/_talks/2026-03-16-tenghao-huang.md
new file mode 100644
index 0000000..574a199
--- /dev/null
+++ b/_talks/2026-03-16-tenghao-huang.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Generalizable, Proactive, and Agentic Learning for Open-Ended Tasks"
+date: "2026-03-16 10:00:00 -0600"
+permalink: "/talks/2026-03-16-tenghao-huang/"
+slug: "2026-03-16-tenghao-huang"
+start_time: "10:00 AM"
+end_time: "11:00 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780 (Evans)"
+canceled: false
+speakers:
+ - name: "Tenghao Huang"
+ bio: "Tenghao Huang is a Ph.D. candidate in Computer Science at the University of Southern California. His research focuses on proactive and agentic AI systems for open-ended tasks, including social world models that simulate how conversations unfold and how decisions emerge from social interaction. His work is recognized by an EMNLP Outstanding Paper Award, an ISI Viterbi Fellowship, and has received media coverage from leading technology presses such as MIT Technology and Microsoft’s Future of Work. He has served as Area Chair roles for ACL, EMNLP, NAACL conferences and has organized a tutorial on Creative Planning with LLMs at NAACL 2025."
+speaker_names: "Tenghao Huang"
+source_file: "_data/talks/2026-03-16-tenghao-huang.toml"
+generated: true
+---
+
+
+
+Open-ended, human-like intelligence requires flexibility, proactivity, and social intelligence: we must learn subjective goals and adapt to novel, complex scenarios. . Current AI systems struggle to learn these behaviors because reward signals are unclear, task context information is incomplete, and the environment lacks observability. I outline a research agenda focused on building agentic learning systems that operate under such uncertainty. (1) Instead of enumerating task-specific heuristics, I propose training AI systems through self-play in an adversarial setting, where models learn by interacting, critiquing, and improving against dynamically evolving counterparts. (2) I equip agents with the ability to proactively gather missing information when task context is incomplete and 3) I reconstruct environment representations through memory to support long-horizon agentic reasoning. Together, I will show, these techniques enable AI systems to learn abstract objectives in tasks as varied as creative writing, multi-agent coordination and long-horizon problem solving, enhancing the creativity, usefulness, and strategic helpfulness of agents in open-ended environments.
diff --git a/_talks/2026-03-20-kate-isaacs.md b/_talks/2026-03-20-kate-isaacs.md
new file mode 100644
index 0000000..a2b6c2d
--- /dev/null
+++ b/_talks/2026-03-20-kate-isaacs.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "A Matter of Audiences: Capturing and Reporting Reasoning and Results Around Data Visualizations"
+date: "2026-03-20 13:30:00 -0600"
+permalink: "/talks/2026-03-20-kate-isaacs/"
+slug: "2026-03-20-kate-isaacs"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/85983626630"
+canceled: false
+speakers:
+ - name: "Kate Isaacs"
+ bio: "Kate Isaacs is an Associate Professor in the Kahlert School of Computing and the Scientific Computing an Imaging Institute at the University of Utah. She received her Ph.D. in computer science from the University of California, Davis and has undergraduate degrees in computer science, mathematics, and physics. She publishes in data visualization, high performance computing, and human-centered computing venues, with interests in complex analysis scenarios, such as those arising from research and data science teams. She has received an NSF CAREER award, a Department of Energy Early Career Research Program award, and a Presidential Early Career Award for Scientists and Engineers (PECASE)."
+speaker_names: "Kate Isaacs"
+source_file: "_data/talks/2026-03-20-kate-isaacs.toml"
+generated: true
+---
+
+
+
+Data visualizations are used throughout the data science process to facilitate the exploratory analysis and to report results. Frequently, these uses are neither separate nor solitary, acting as a medium for data science teams to collaboratively reason about data. This team-based data science work is often fast-paced and involves several forms of non-digital communication. I will discuss how we leverage gesture, sketch, and speech to create support for common meetings around data, specifically collaborative remote meetings and informal presentations, to aid this aspect of data science work. Then, focusing on the wider dissemination of results, I will discuss findings regarding the public's views on the use of data and AI in science videos.
diff --git a/_talks/2026-03-20-xueguang-ma.md b/_talks/2026-03-20-xueguang-ma.md
new file mode 100644
index 0000000..b5e44eb
--- /dev/null
+++ b/_talks/2026-03-20-xueguang-ma.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Breaking Information Silos: Advancing Search Systems for Unified Information Seeking"
+date: "2026-03-20 10:00:00 -0600"
+permalink: "/talks/2026-03-20-xueguang-ma/"
+slug: "2026-03-20-xueguang-ma"
+start_time: "10:00 AM"
+end_time: "11:00 AM"
+series: "Data Science & AI Lecture Series"
+location: "MEB 3147 (LCR)"
+canceled: false
+speakers:
+ - name: "Xueguang Ma"
+ website: "https://mxueguang.github.io/"
+ bio: "Xueguang Ma is currently a last-year PhD at the David R. Cheriton School of Computer Science at University of Waterloo, advised by Prof. Jimmy Lin. His research focuses on Information Retrieval (IR) and Natural Language Processing (NLP), with an overarching goal to make it easy for people and intelligent systems to access, understand, and interact with world information. His work has been published in top IR and NLP venues such as SIGIR, ACL, EMNLP, and NeurIPS. More details on his website: https://mxueguang.github.io/."
+speaker_names: "Xueguang Ma"
+source_file: "_data/talks/2026-03-20-xueguang-ma.toml"
+generated: true
+---
+
+
+
+Information seeking has been fundamental to human advancement, enabling knowledge acquisition, decision-making, and innovation across disciplines. However, traditional information retrieval systems often rely on specialized pipelines optimized for specific retrieval tasks, causing information silos that hinder unified information seeking. In this talk, I will present our work in building unified document retrieval systems that break these information silos across three dimensions: (1) domain and language silos, where I demonstrate how LLM-based dense retrievers achieve strong generalizability across retrieval tasks and present frameworks for training small, generalizable retrievers through diverse LLM augmentation; (2) modality silos, where I introduce a paradigm shift from text-based retrieval that relies on content extraction to directly encoding document screenshots, preserving all information including text, images, and layout in unified dense representations; and (3) space silos, where we show the importance of LLM-powered search agents in seeking and gathering information across disparate sources, and present fair and transparent evaluation benchmarks for assessing deep-search systems. I will conclude by discussing future directions that further pave the way toward building truly unified retrieval systems for seamless information seeking across world knowledge.
diff --git a/_talks/2026-03-23-benjie-wang.md b/_talks/2026-03-23-benjie-wang.md
new file mode 100644
index 0000000..6e0c143
--- /dev/null
+++ b/_talks/2026-03-23-benjie-wang.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Bridging the Formalization Gap for Generative AI"
+date: "2026-03-23 10:00:00 -0600"
+permalink: "/talks/2026-03-23-benjie-wang/"
+slug: "2026-03-23-benjie-wang"
+start_time: "10:00 AM"
+end_time: "11:00 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Benjie Wang"
+ bio: "Benjie Wang is a postdoctoral researcher in the Statistical and Relational Artificial Intelligence (StarAI) lab in the Computer Science Department at UCLA. Dr. Wang's research interests are in artificial intelligence, including deep generative models, probabilistic machine learning, sequence and language modeling, and formal reasoning. His work develops theory-driven and scalable methods for understanding and controlling generative models, by studying the mathematical foundations, architecture, and manipulation of representations of high-dimensional probability distributions.\n\nPreviously, he was a research fellow at the Simons Institute for the Theory of Computing at UC Berkeley in Fall 2023. Dr. Wang obtained his DPhil in Computer Science from the University of Oxford advised by Prof. Marta Kwiatkowska in 2023, his MSc in Statistical Science from the University of Oxford in 2019, and his BA in Mathematics from the University of Cambridge in 2018.\n"
+speaker_names: "Benjie Wang"
+source_file: "_data/talks/2026-03-23-benjie-wang.toml"
+generated: true
+---
+
+
+
+Generative models, such as large language models and diffusion models, have tremendously increased the scope of problems that AI can address. As such, there is a significant trend toward incorporating generative AI to automate tasks across computing and more broadly, from controlling robotics systems, to software generation and testing, to searching over scientific knowledge. However, there remains a significant formalization gap between the domain knowledge, theories, and logical and semantic constraints that are vital to applications, and the statistical patterns over natural data represented by large generative models. In this talk, I will demonstrate how we can systematically bridge this formalization gap towards more trustworthy AI. First, drawing from examples and applications in my research, I will show how we can utilize suitable intermediate representations of probability distributions to bridge between formal language and generative models at scale. Then, I will discuss how these practical methods are underpinned by my work advancing the mathematical and computational foundations underlying these tractable representations of probability distributions.
diff --git a/_talks/2026-03-25-dick-sadler.md b/_talks/2026-03-25-dick-sadler.md
new file mode 100644
index 0000000..d4d3315
--- /dev/null
+++ b/_talks/2026-03-25-dick-sadler.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "What Happens in the 10 Years Following a State Government-Caused Environmental Injustice? Flint's Progress Since Its Water Crisis"
+date: "2026-03-25 16:00:00 -0600"
+permalink: "/talks/2026-03-25-dick-sadler/"
+slug: "2026-03-25-dick-sadler"
+start_time: "4:00 PM"
+end_time: "5:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "MLIB 1110 | Zoom: 890 9876 9672, Passcode: 156565"
+canceled: false
+speakers:
+ - name: "Dick Sadler"
+speaker_names: "Dick Sadler"
+source_file: "_data/talks/2026-03-25-dick-sadler.toml"
+generated: true
+---
+
+
+
+The Flint Water Crisis was the result of decades of deliberate disinvestment and state government ineptitude. In this talk, Dr. Sadler will discuss how the crisis unfolded, and how his research - examining environmental exposures, neighborhood conditions, and blood lead levels - revealed its scale. He will also address what has changed in Flint since then, including the massive new investments in the city that have brought hope in the wake of catastrophe.
diff --git a/_talks/2026-03-26-bailing-lyu.md b/_talks/2026-03-26-bailing-lyu.md
new file mode 100644
index 0000000..22216fb
--- /dev/null
+++ b/_talks/2026-03-26-bailing-lyu.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "Navigating the Human-AI Nexus in Education: Bridging Cognitive Science, Learning Analytics, and Intelligent Systems"
+date: "2026-03-26 10:30:00 -0600"
+permalink: "/talks/2026-03-26-bailing-lyu/"
+slug: "2026-03-26-bailing-lyu"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "Evans Conference room (WEB 3780)"
+zoom: "https://utah.zoom.us/j/87171666093"
+canceled: false
+speakers:
+ - name: "Bailing Lyu"
+ bio: "Dr. Bailing Lyu is an Assistant Professor at Auburn University, specializing in AI-driven educational technologies, learning analytics, and cognitive learning sciences. She earned her Ph.D. in Educational Psychology from Pennsylvania State University and subsequently served as a postdoctoral researcher at the University of Utah, focusing on conversational AI. Her research explores how AI can be designed and applied across both student-facing and instructor-facing contexts. For students, she focuses on designing AI systems that foster cognitive engagement while enhancing interest, motivation, and emotional experiences. In parallel, she investigates how AI can empower instructors to develop instructional and AI-embedded materials, such as generative interactive visuals, that make abstract concepts more concrete, accessible, and engaging. Methodologically, she applies learning analytics, educational data mining, and experimental research to study interactions with these multimodal resources. Her impactful work has contributed to large-scale funded projects, including the $10 million ALTER-Math initiative (supported by the Schmidt, Gates, and Walton Family Foundations) as well as several NSF-funded projects. A highly productive scholar, Dr. Lyu has an extensive publication record with 24 submitted or published journal articles and over 40 conference presentations advancing the future of AI in education."
+speaker_names: "Bailing Lyu"
+source_file: "_data/talks/2026-03-26-bailing-lyu.toml"
+generated: true
+---
+
+
+
+Artificial intelligence is increasingly transforming education, reshaping how students engage with learning and how instructors design and deliver instruction. However, critical gaps remain in the field of AI in Education (AIED): (1) a frequent lack of grounding in the cognitive and learning sciences when developing AI pedagogical tools, (2) a "black box" regarding how students process and interact with AI-supported environments, and (3) a limited emphasis on meaningful human involvement in how AI is applied in practice. This talk presents a series of research projects on AI-augmented learning and teaching that address these gaps by integrating cognitive science into AI-powered educational technologies, examining human-AI interaction, and centering human agency in their application. Specifically, using teachable agents as an example, it will discuss (1) how pedagogical AI can be designed using theoretically grounded learning principles, (2) how students' learning processes unfold in these environments, and (3) how educators and learners can actively leverage AI to co-create and navigate engaging experiences that enhance learning outcomes. Employing experimental research, learning analytics, and educational data mining, these studies examine the intersection of human cognition and AI-driven learning. By aligning AI with evidence-based pedagogical strategies, this work advances our understanding of how intelligent technologies can foster deeper learning, positive learning experiences, personalized instruction, and adaptive support across diverse educational contexts.
diff --git a/_talks/2026-03-27-erdogan-kaya.md b/_talks/2026-03-27-erdogan-kaya.md
new file mode 100644
index 0000000..0d9955b
--- /dev/null
+++ b/_talks/2026-03-27-erdogan-kaya.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "From Self-Efficacy to Systemic Change: Building an Equitable Computing Education Research Program"
+date: "2026-03-27 10:30:00 -0600"
+permalink: "/talks/2026-03-27-erdogan-kaya/"
+slug: "2026-03-27-erdogan-kaya"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "Evans Conference room (WEB 3780)"
+canceled: false
+speakers:
+ - name: "Erdogan Kaya"
+ bio: "Dr. Kaya holds a joint appointment with the College of Education and the Division of Data Science at the University of Texas at Arlington. He is an assistant professor of computing education with a Ph.D. in Curriculum and Instruction, a B.S. in Chemical Engineering, an M.S. in Computer Science and Engineering, and an M.S. in Machine Learning Engineering from George Mason University. His research focuses on AI literacy, computational thinking, and computing education through equity lenses. He has secured more than $5 million in external research funding from NSF, Amazon, and Google, and has developed projects including the Educate AI curriculum, the Rural AI initiative, and the Compose with AI platform. Dr. Kaya advocates for research, teaching, and learning as a unified enterprise that benefits students and society. He has received numerous teaching awards, including the NCWIT Aspirations in Computing Educator Award and the ASEE Southeastern Section Outstanding New Teacher Award, and has contributed to educational outreach programs including Code.org and FIRST Robotics competitions."
+speaker_names: "Erdogan Kaya"
+source_file: "_data/talks/2026-03-27-erdogan-kaya.toml"
+generated: true
+---
+
+
+
+This talk examines a foundational question in computational STEM education: to what extent can targeted interventions improve pre-service elementary teachers’ computational thinking teaching efficacy beliefs? I present a study of pre-service elementary teachers who participated in a three-week CT intervention integrating EV3 robotics, Code.org, and Zoombinis. Using the CTTEBI in a pre-post design, results show significant gains in personal CT teaching efficacy, pointing to important directions for future work. I then present my current AI education research program, including the “Educate AI” project developing AI literacy curriculum through linguistically inclusive elementary robotics, the Rural AI project integrating AI concepts for rural upper elementary students, and the Compose with AI platform guiding grades 4-8 students in critically evaluating AI-generated content. I will also discuss my emerging research agenda, including several NSF proposals under review focused on AI literacy across K-16 settings. I close with a vision for how University of Utah’s collaborative structure across the Scientific Computing and Imaging (SCI) Institute, Department of Educational Psychology, College of Education, and partner departments represents an ideal ecosystem to advance this agenda.
diff --git a/_talks/2026-03-30-xiaoling-hu.md b/_talks/2026-03-30-xiaoling-hu.md
new file mode 100644
index 0000000..54f1953
--- /dev/null
+++ b/_talks/2026-03-30-xiaoling-hu.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "Principled Learning for Medical AI: Structure, Reliability, and Interpretability"
+date: "2026-03-30 10:00:00 -0600"
+permalink: "/talks/2026-03-30-xiaoling-hu/"
+slug: "2026-03-30-xiaoling-hu"
+start_time: "10:00 AM"
+end_time: "11:00 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Xiaoling Hu"
+ bio: "Xiaoling Hu is a postdoctoral research fellow at Harvard Medical School. He received his Ph.D. in Computer Science from Stony Brook University. His research focuses on Machine Learning for Healthcare, with an emphasis on developing core AI/ML algorithms for healthcare applications. His work has been published in leading venues across machine learning, computer vision, and medical imaging, including NeurIPS, ICLR, AISTATS, CVPR, ICCV, ECCV, Medical Image Analysis, and MICCAI. Several of his papers have been selected for oral or spotlight presentations. Xiaoling has organized multiple tutorials and workshops at top-tier conferences and served as Area Chairs for venues such as NeurIPS, CVPR, AISTATS, and MICCAI. He is also a recipient of the prestigious Catacosinos Fellowship, awarded to SBU graduate students with exceptional research achievements."
+speaker_names: "Xiaoling Hu"
+source_file: "_data/talks/2026-03-30-xiaoling-hu.toml"
+generated: true
+---
+
+
+
+The widespread deployment of AI in medicine demands not only predictive accuracy but also structural awareness, reliability under uncertainty, and interpretability for clinical trust. In this talk, I will present a unified research agenda toward principled learning for medical AI, grounded in these core pillars.
+
+First, I will discuss how incorporating explicit structure, such as topology and spatial priors, into neural networks enhances the model's ability to reason about fine-grained anatomical and pathological features, which are critical for tasks like brain and tumor segmentation. Second, I will focus on reliability, exploring how we can quantify and mitigate uncertainty arising from imperfect labels, limited data, and domain shifts, using methods such as distributional modeling, hyperparameter learning, and probabilistic inference. Third, I will show how these approaches naturally support interpretability, enabling AI systems to communicate meaningful representations that align with human clinical understanding.
+
+Through applications in radiology, pathology, neuroimaging, and large-scale population datasets, I will demonstrate how these principles facilitate scalable annotation, robust generalization, and scientific discovery. I will conclude with future directions aimed at generalizing these principles to multimodal learning, real-world deployment, and next-generation AI systems in medicine.
diff --git a/_talks/2026-03-31-he-yin.md b/_talks/2026-03-31-he-yin.md
new file mode 100644
index 0000000..f03dc4d
--- /dev/null
+++ b/_talks/2026-03-31-he-yin.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Advancing GeoAI and Earth observation for environmental monitoring"
+date: "2026-03-31 10:45:00 -0600"
+permalink: "/talks/2026-03-31-he-yin/"
+slug: "2026-03-31-he-yin"
+start_time: "10:45 AM"
+end_time: "11:45 AM"
+series: "Data Science & AI Lecture Series"
+location: "Evans Conference room (WEB 3780)"
+canceled: false
+speakers:
+ - name: "He Yin"
+ bio: "Dr. He Yin is an Associate Professor in the Department of Geography at Kent State University and Director of the Remote Sensing and Land Science Lab. He earned his PhD in Geography from Humboldt University of Berlin and completed postdoctoral training at the University of Wisconsin–Madison.\n\nHis research combines geospatial artificial intelligence (GeoAI), Earth observation, and interdisciplinary methods to monitor landscape change and assess its environmental and societal impacts. He serves as principal investigator on projects supported by NASA, the National Science Foundation, Lawrence Livermore National Laboratory, and the Center for International Forestry Research, and currently advises the United Nations Environment Programme (UNEP) and the United Nations Office for Project Services (UNOPS). His work has received the European Space Agency Earth Observation Excellence Team Award (2025) and the American Association of Geographers Media Achievement Award (2026).\n"
+speaker_names: "He Yin"
+source_file: "_data/talks/2026-03-31-he-yin.toml"
+generated: true
+---
+
+
+
+Landscapes around the world are changing rapidly, with important consequences for sustainability, climate resilience, and society. Yet monitoring these changes across regions and scales remains difficult. In this talk, I present a research program that combines multi-sensor Earth observation, geospatial artificial intelligence (GeoAI), and land system science to better understand how land systems are changing, what drives those changes, and why they matter.
+
+I begin by presenting my studies using satellite image time series to map land use change and, in collaboration with environmental scientists and ecologists, to examine its implications for carbon sequestration and biodiversity. These studies also reveal key limitations of conventional remote sensing approaches, including sensor constraints, limited transferability, and scarce training data. I then show how these challenges motivate my more recent work in sensor fusion, physics-informed machine learning, and deep learning with very-high-resolution imagery — applied to problems ranging from irrigation water use and wildfire-invasive species interactions to conflict-induced environmental damage. I conclude by discussing the broader goal of building GeoAI models for environmental monitoring that are informed by physical processes, transferable across contexts, and useful for real-world decision-making.
diff --git a/_talks/2026-04-01-md-mostafijur-rahman.md b/_talks/2026-04-01-md-mostafijur-rahman.md
new file mode 100644
index 0000000..ca6200f
--- /dev/null
+++ b/_talks/2026-04-01-md-mostafijur-rahman.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Efficient and Reliable AI for Real-World Healthcare Deployment"
+date: "2026-04-01 10:30:00 -0600"
+permalink: "/talks/2026-04-01-md-mostafijur-rahman/"
+slug: "2026-04-01-md-mostafijur-rahman"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Md Mostafijur Rahman"
+ bio: "Md Mostafijur Rahman is a Ph.D. candidate at The University of Texas at Austin, advised by Radu Marculescu. His research sits at the intersection of AI, biomedical imaging, and computer vision, with a focus on building efficient, reliable, and scalable AI systems for deployment in healthcare under real-world constraints. His work has been translated to practice through research internships at GE Healthcare, the National Institutes of Health (NIH), and Bosch Research. He has published over 20 peer-reviewed papers in venues including CVPR, NeurIPS, MICCAI, and ICCV, with several works selected for Spotlight and Oral presentations. His research contributions have been recognized by the NIH Summer IRTA Fellowship, the Texas Health Catalyst Award, and the Discovery to Impact Award."
+speaker_names: "Md Mostafijur Rahman"
+source_file: "_data/talks/2026-04-01-md-mostafijur-rahman.toml"
+generated: true
+---
+
+
+
+Healthcare is one of the highest-impact domains for AI, yet reliable deployment at scale remains difficult. To truly improve patient care and clinical workflows, AI must operate under real clinical constraints, not just in ideal lab settings. In practice, deployment is limited by high compute and memory costs, scarce labeled data, and distribution shifts across sites and time. Many clinically important findings are also rare and long-tailed, which makes generalization especially challenging. My research makes deployability a design objective by developing methods that stay accurate under strict resource and data constraints. In this talk, I will first discuss high-performance lightweight deep learning architectures built by redesigning core building blocks. I will then present training-time generative supervision strategies that improve data efficiency and generalization to rare and long-tailed cases with no inference overhead. I will conclude with a forward-looking direction toward real-time perception for surgical assistance, where reliable performance under strict constraints is non-negotiable.
diff --git a/_talks/2026-04-06-qiang-ji.md b/_talks/2026-04-06-qiang-ji.md
new file mode 100644
index 0000000..0e71465
--- /dev/null
+++ b/_talks/2026-04-06-qiang-ji.md
@@ -0,0 +1,26 @@
+---
+layout: "talk"
+title: "Towards Data-Efficient, Trustworthy, and Generalizable AI for Visual Understanding"
+date: "2026-04-06 10:30:00 -0600"
+permalink: "/talks/2026-04-06-qiang-ji/"
+slug: "2026-04-06-qiang-ji"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Qiang Ji"
+ bio: "Dr. Qiang Ji received his Ph.D. in Electrical Engineering from the University of Washington. He is currently a Professor in the Department of Electrical, Computer, and Systems Engineering at Rensselaer Polytechnic Institute (RPI). From 2009 to 2010, he served as a Program Director at the National Science Foundation (NSF), where he managed NSF’s research programs in computer vision and machine learning. He has also held teaching and research positions at the University of Illinois at Urbana–Champaign, Carnegie Mellon University, the University of Nevada, and the Air Force Research Laboratory.\n\nProf. Ji’s research focuses on computer vision, probabilistic graphical models, Bayesian deep learning, and causal machine learning, with applications across a wide range of domains including human behavior analysis, medical imaging, intelligent transportation, and human–robot interaction. He has published over 400 papers in leading journals and conferences and has received multiple awards recognizing his contributions to the field.\n\nProf. Ji has served the research community extensively as an editor for several IEEE and international journals and as a General Chair, Program Chair, Area Chair, and Program Committee member for numerous major international conferences and workshops. He is a Fellow of both the IEEE and the International Association for Pattern Recognition (IAPR).\n"
+speaker_names: "Qiang Ji"
+source_file: "_data/talks/2026-04-06-qiang-ji.toml"
+generated: true
+---
+
+
+
+Artificial Intelligence (AI) has achieved remarkable progress and is increasingly integrated across a wide range of fields, fueling what many describe as the fourth industrial revolution. However, behind this widespread enthusiasm lie fundamental limitations. Today’s AI systems face three major challenges: (1) an insatiable demand for large-scale labeled data, (2) limited trustworthiness due to inadequate uncertainty quantification, and (3) poor generalization across domains. These challenges cannot be addressed simply by scaling data and computation; instead, they require foundational advances in theory and methodology.
+
+In this talk, I will present recent research from my lab that addresses these challenges in a variety of computer vision tasks. To improve data efficiency and generalization, I will introduce our work on knowledge-augmented deep learning, where prior knowledge from diverse sources is systematically identified, encoded, and integrated with data-driven neural networks. This approach leads to hybrid neural-symbolic models that are both more data-efficient and more generalizable. To enhance model trustworthiness and explainability, I will discuss our advances in Bayesian deep learning. First, I will present our work on a Bayesian Transformer framework for accurate and robust human activity recognition. I will then introduce our work on uncertainty attribution, which identifies the sources of uncertainty in deep models and leverages this information for uncertainty mitigation and improved model performance. Finally, I will highlight our recent work on causal deep learning for addressing domain generalization. I will introduce a neural causal model that learns domain-invariant representations by identifying and eliminating spurious correlations arising from data biases.
+
+Together, these efforts aim to advance a new generation of AI systems that are more data-efficient, trustworthy, and robust, enabling reliable deployment across a range of domains including human behavior understanding, medical imaging, scientific discovery, and human–robot interaction.
diff --git a/_talks/2026-04-07-si-chen.md b/_talks/2026-04-07-si-chen.md
new file mode 100644
index 0000000..e6301a2
--- /dev/null
+++ b/_talks/2026-04-07-si-chen.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "Advancing AI Literacy and Human-Centered AI for Teaching and Learning"
+date: "2026-04-07 10:30:00 -0600"
+permalink: "/talks/2026-04-07-si-chen/"
+slug: "2026-04-07-si-chen"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Si Chen"
+speaker_names: "Si Chen"
+source_file: "_data/talks/2026-04-07-si-chen.toml"
+generated: true
+---
+
+
+
+Artificial intelligence (AI) is rapidly transforming education and how people learn, teach, and prepare for the future. Yet many systems are still built around technical capabilities rather than the real needs of students, educators, and families. In this talk, I present a human-centered design research agenda that advances both AI literacy and AI for teaching and learning across students, families, and educators. First, I define and conceptualize generative AI literacy across children and parents by co-designing measurement frameworks and interactive tools. This work enables families to build a shared understanding of AI while supporting its critical and responsible use in everyday self-directed learning contexts. Second, I present AI Academy, an faculty-facing professional development program and badge system that expands institutional capacity for AI. Through curriculum design and lightweight tools, this work supports faculty in integrating generative AI into teaching while strengthening their AI literacy and maintaining pedagogical and disciplinary goals. Lastly, I present AI-powered tutoring systems and other AI interactions I have developed with college learners with disabilities, including LLM-based chatbots, particularly for Deaf and Hard of Hearing students.The talk is intended for a broad, interdisciplinary audience.
diff --git a/_talks/2026-04-08-fahim-faisal.md b/_talks/2026-04-08-fahim-faisal.md
new file mode 100644
index 0000000..aabcd74
--- /dev/null
+++ b/_talks/2026-04-08-fahim-faisal.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Multilingual Model Adaptation for Under-Served Languages"
+date: "2026-04-08 10:00:00 -0600"
+permalink: "/talks/2026-04-08-fahim-faisal/"
+slug: "2026-04-08-fahim-faisal"
+start_time: "10:00 AM"
+end_time: "11:00 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Fahim Faisal"
+ bio: "Fahim Faisal is a final-year Ph.D. candidate in the Department of Computer Science at George Mason University, where he is a member of the GMU NLP Lab under the supervision of Dr. Antonios Anastasopoulos. His research focuses on adapting language models for low-resource and underrepresented languages, spanning both the creation of linguistic resources and the systematic evaluation of state-of-the-art large language models (LLMs) across language varieties and dialects. His work also examines how modern LLMs handle domain- and policy-specific safety alignment, multilingual reasoning, and cultural disparities embedded in language modeling. His research is driven by the overarching goal of democratizing AI and NLP, ensuring that users from all linguistic, cultural, and demographic backgrounds receive equitable utility from advances in machine intelligence. His work DialectBench received the Best Social Impact Award at ACL 2024, recognizing its contribution to inclusive and socially responsible NLP research."
+speaker_names: "Fahim Faisal"
+source_file: "_data/talks/2026-04-08-fahim-faisal.toml"
+generated: true
+---
+
+
+
+Language models with multilingual capabilities serve as crucial touchpoints for improving the inclusion of underrepresented languages in Natural Language Processing (NLP). This research investigates the structural sources of linguistic underrepresentation and explores strategies for improving the adaptation of low-resource language varieties through the development of linguistically grounded resources. We first examine the extent of multilingual and geographic representation gaps across three key dimensions of language modeling: datasets, model architecture, and model-generated text. Next, we introduce DialectBench, an initiative designed to evaluate language variation in the form of dialects and language varieties—an aspect often overlooked in NLP benchmarks, which primarily focus on standardized language forms. To address the challenges revealed by these analyses, we propose two adaptation frameworks. The first introduces phylogenetic adapter hierarchies that exploit language-family structure to enable zero-shot transfer across related languages. The second presents a pivot-based reinforcement learning approach that leverages high-resource expert models to transfer reasoning alignment without requiring target-language annotations. Together, these linguistically motivated adaptation strategies aim to improve the performance of language models on underrepresented languages and dialects, ultimately contributing to more equitable and accessible NLP systems for diverse linguistic communities.
diff --git a/_talks/2026-04-10-chase-neumann.md b/_talks/2026-04-10-chase-neumann.md
new file mode 100644
index 0000000..766b50e
--- /dev/null
+++ b/_talks/2026-04-10-chase-neumann.md
@@ -0,0 +1,28 @@
+---
+layout: "talk"
+title: "Building and Utilizing Foundation Models for Drug Discovery and Clinical Development"
+date: "2026-04-10 13:30:00 -0600"
+permalink: "/talks/2026-04-10-chase-neumann/"
+slug: "2026-04-10-chase-neumann"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/85983626630"
+canceled: false
+speakers:
+ - name: "Chase Neumann"
+ affiliation: "PhD"
+ bio: "I believe the next generation of life-saving medicines won't just be discovered; they will be engineered at the intersection of biology and machine learning.\nDuring my time at Recursion, I operated at the frontier of \"AI for Drug Discovery,\" translating high-dimensional data into actionable therapeutic programs. I led cross-functional teams through the critical transition from late-stage discovery into early clinical development, ensuring that AI-driven insights were successfully translated into clinical-ready drug candidates.\n\nWith a PhD in Translational Medicine from CCLCM at Case Western Reserve University, I bridge the gap between translational oncology and computational innovation. My work has focused on deconstructing complex disease mechanisms and scaling them through automated, industrialized platforms.\n\nI am a passionate advocate for the \"TechBio\" shift—moving away from serendipity and toward a predictable model of drug discovery to get better medicines to patients, faster. In late 2025, I co-founded Valinor Discovery to bridge the gap in clinical translation using foundation models trained on real-world multi-modal clinical data.\n"
+speaker_names: "Chase Neumann"
+source_file: "_data/talks/2026-04-10-chase-neumann.toml"
+generated: true
+---
+
+
+
+The journey toward decoding biology at scale began with a focus on high-dimensional cellular morphology. At Recursion, our differentiation centered on deep learning models designed to learn biological representations directly from imaging, enabling predictive inference at a massive scale. By leading multiple cross-functional teams from early-stage discovery through to early clinical development, we demonstrated the power of this "inference-first" philosophy. A primary highlight of these efforts was the RBM39 program, a novel molecular glue degrader discovered entirely through computational inference rather than traditional screening. This success proved that models could identify complex biological mechanisms; however, moving from cellular discovery to comprehensive patient care requires a leap into even higher-dimensional, clinical data.
+
+Valinor represents the next evolution of this mission. We build multimodal clinical foundation models, co-designing data collection and model architecture to maximize signal within and across complex modalities. We have proprietary access to patient cohorts and collaborate with biobanks, clinical trial sites, and academic partners, giving us unique data advantages at scale. Our approach is to first build the best unimodal patient representations across modalities—including DNA, transcriptomics, proteomics, cfDNA, histopathology, and patient reports—and then fuse them.
+
+We have shown that attention-based fusion consistently outperforms unimodal approaches while making the contributions of different modalities interpretable. This enables genuine clinical reasoning: the model can chain evidence across modalities, explain which features drive a prediction, and engage with clinicians in natural language. Ultimately, we envision a virtual patient that reasons over the full spectrum of a patient's biology the way an expert clinician would, but at a scale and resolution no human can match.
diff --git a/_talks/2026-04-13-jihyun-rho.md b/_talks/2026-04-13-jihyun-rho.md
new file mode 100644
index 0000000..35aa54f
--- /dev/null
+++ b/_talks/2026-04-13-jihyun-rho.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "Supporting responsible use of AI in visual-based teaching and learning"
+date: "2026-04-13 13:00:00 -0600"
+permalink: "/talks/2026-04-13-jihyun-rho/"
+slug: "2026-04-13-jihyun-rho"
+start_time: "1:00 PM"
+end_time: "2:00 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Jihyun Rho"
+ bio: "Jihyun Rho is a Postdoctoral Researcher at the University of Florida. She earned her Ph.D. in the Learning Sciences program in the Department of Educational Psychology at the University of Wisconsin–Madison. Her research uses design-based research and learning analytics to design and evaluate AI-augmented learning environments for visual-based learning. She studies how learners and educators interact with AI-generated representations, combining qualitative analysis of interaction processes with experimental approaches to examine learning outcomes, reasoning, and reliance on AI. Her scholarship appears in leading journals and conferences in learning sciences and AI in education."
+speaker_names: "Jihyun Rho"
+source_file: "_data/talks/2026-04-13-jihyun-rho.toml"
+generated: true
+---
+
+
+
+Artificial intelligence (AI) is increasingly integrated into educational contexts, reshaping how teachers and students generate and interpret visual representations such as diagrams, illustrations, and data visualizations. While AI offers efficiency and flexibility, it also introduces inaccuracies and misleading interpretations that can undermine learning. In this talk, I present a research program centered on supporting the responsible use of AI in visual-based teaching and learning by promoting visual literacy. Grounded in design-based research, I design and evaluate AI-augmented learning environment where educators and students engage with AI-generated outputs as critical inquirers. This work shows that these approaches improved interpretive accuracy, deepened reasoning about visual representations, and reduce over-reliance on AI. Together, this work contributes design principles for fostering responsible AI use in visual-based educational contexts.
diff --git a/_talks/2026-04-14-sameer-honwad.md b/_talks/2026-04-14-sameer-honwad.md
new file mode 100644
index 0000000..1c88a21
--- /dev/null
+++ b/_talks/2026-04-14-sameer-honwad.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Move Slow and Build Community"
+date: "2026-04-14 10:30:00 -0600"
+permalink: "/talks/2026-04-14-sameer-honwad/"
+slug: "2026-04-14-sameer-honwad"
+start_time: "10:30 AM"
+end_time: "11:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Sameer Honwad"
+ bio: "Sameer Honwad, Ph.D., is an Assistant Professor of Learning Sciences in the Department of Learning and Instruction at SUNY Buffalo. His research explores how technology can support learning in culturally diverse communities worldwide through participatory design approaches. At the heart of his work are long-term, trust-based partnerships with Indigenous communities in Idaho, rural and Indigenous communities in Bhutan and India, and urban communities in New Orleans. Rather than parachuting in with ready-made tools, Dr. Honwad co-designs technology-enhanced learning environments alongside teachers, parents, local leaders, and students, positioning community members as essential partners, not subjects of study. His current research examines how rural communities in South Asia make sense of and use artificial intelligence in everyday life. Grounded in participatory design-based research, Dr. Honwad centers communities that have historically faced systemic barriers to technology access due to class, race, gender, and caste. He has served as PI and Co-PI on multiple National Science Foundation (NSF) grants focused on designing technologies that help learners grapple with complex, real-world problems rooted in their own communities. Throughout his career he has taught courses that focus on how people learn within their cultures and the design of technology-enhanced learning environments that honor local knowledge, needs and aspirations. He is currently in the process of building a Global Design Studio that would be an interdisciplinary hub connecting scholars, students, and community members worldwide to think critically about AI and its role in everyday life."
+speaker_names: "Sameer Honwad"
+source_file: "_data/talks/2026-04-14-sameer-honwad.toml"
+generated: true
+---
+
+
+
+Socio-emotional learning (SEL) and reflection are foundational components of quality education, yet they remain underrepresented in many school curricula. Reflection supports scientific thinking and inquiry, while SEL equips students with the emotional resilience needed to take risks, embrace failure, and navigate the uncertainties inherent in learning and innovation. Together, these competencies are essential not only for academic and professional growth, but for everyday wellbeing.Implementing SEL and reflective practices in rural India presents distinct challenges. Limited access to trained professionals and consistent resources is compounded by cultural stigma around openly discussing emotions, barriers that make it difficult for students to develop these critical skills in traditional school settings.
+
+To address this gap, we designed a conversational chatbot tailored for students in rural India, providing a low-barrier, accessible space for daily reflection and emotional expression. Importantly, the chatbot is also designed to serve as a springboard for teachers to introduce concepts of AI literacy, enabling meaningful classroom conversations about how AI systems work, their limitations, and the ethical considerations surrounding their use. This presentation presents the co-design process behind the chatbot, highlighting the collaborative contributions of an interdisciplinary team comprising teachers, therapists, computer scientists, and learning scientists. The presentation discusses how this cross-disciplinary approach shaped a tool intended to help students engage with their socio-emotional selves and build a habit of reflective thinking in their daily lives.
diff --git a/_talks/2026-04-15-amirali-abdullah.md b/_talks/2026-04-15-amirali-abdullah.md
new file mode 100644
index 0000000..151adc9
--- /dev/null
+++ b/_talks/2026-04-15-amirali-abdullah.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Controlling LLM's via Activation Geometry"
+date: "2026-04-15 10:30:00 -0600"
+permalink: "/talks/2026-04-15-amirali-abdullah/"
+slug: "2026-04-15-amirali-abdullah"
+start_time: "10:30 AM"
+end_time: "11:00 AM"
+series: "Data Science & AI Lecture Series"
+location: "Dinosaur Room (CSC 206)"
+zoom: "https://utexas.zoom.us/j/84742203545?pwd=lUosaf3T6bkIS1QAaIQYHYiiClE2ZP.1"
+canceled: false
+speakers:
+ - name: "Amirali Abdullah"
+ bio: "Amirali Abdullah is a Lead AI Researcher at Thoughtworks Inc and a Research Advisor at Martian Learning. His research focuses on the interpretability and control of large language models, with particular emphasis on activation-level steering, representation geometry, and the structure of learned features."
+speaker_names: "Amirali Abdullah"
+source_file: "_data/talks/2026-04-15-amirali-abdullah.toml"
+generated: true
+---
+
+
+
+Controlling the behavior of large language models at inference time is an increasingly important problem. In this talk, I present a simple and unified approach to steering model behavior based on activation geometry. By learning a single classifier over hidden representations, we can derive directions that control multiple attributes such as helpfulness, style, or safety, and compose them dynamically without retraining.
+This framework enables flexible, low cost control of model outputs and highlights a geometric view of representation space beyond fixed linear directions. I will discuss empirical results showing how this approach supports multi attribute control in practice, and briefly outline how such steering mechanisms can be useful in scientific settings where reliable and interpretable model behavior is critical. Our recent followup work suggests that similar activation level interventions can extend across modalities, enabling systematic analysis and control in text to image models through composable operations.
diff --git a/_talks/2026-04-15-chengbin-deng.md b/_talks/2026-04-15-chengbin-deng.md
new file mode 100644
index 0000000..d0de674
--- /dev/null
+++ b/_talks/2026-04-15-chengbin-deng.md
@@ -0,0 +1,22 @@
+---
+layout: "talk"
+title: "AI-Driven Environmental Intelligence: Scalable and System-Level Approaches for Environmental Decision Making"
+date: "2026-04-15 13:30:00 -0600"
+permalink: "/talks/2026-04-15-chengbin-deng/"
+slug: "2026-04-15-chengbin-deng"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 3780"
+canceled: false
+speakers:
+ - name: "Chengbin Deng"
+ bio: "Dr. Chengbin Deng is an Associate Professor in the Department of Geography and Environmental Sustainability and Director of the Center for Spatial Analysis at the University of Oklahoma (OU). His research focuses on advancing AI for environmental monitoring, with an emphasis on integrating Earth observation, geospatial data, and decision systems to address challenges in climate, hazards, and urban environments. His work has been supported by major federal agencies including NASA, the U.S. National Science Foundation, and USGS, where he serves as Principal Investigator on multiple externally funded projects. He has been recognized among the World’s Top 2% Scientists by Stanford University and Elsevier and received the Outstanding Research Award from the College of Atmospheric and Geographic Sciences at OU. Currently, Dr. Deng serves on NASA Land Cover and Land Use Change Science Team, and is Advisor of NASA FINESST Future Investigator and an NSF EPSCoR RII Research Fellow."
+speaker_names: "Chengbin Deng"
+source_file: "_data/talks/2026-04-15-chengbin-deng.toml"
+generated: true
+---
+
+
+
+Environmental systems are becoming more complex, dynamic, and tightly connected to human activities, yet much of our current work remains focused on isolated models or static mapping. In this talk, a framework will be present and discussed that uses AI to move beyond observation toward a more integrated understanding of environmental systems. The central idea is to link geospatial data, models, and real-world decisions so that environmental information can better reflect changing conditions across space and time while remaining meaningful for researchers and stakeholders. By drawing on some recent federally supported projects, this talk will show how this framework can capture large scale environmental dynamics, account for social and local context, and support more informed decision processes in various settings, ranging from urban systems to extreme events. Instead of using AI just as a tool, this work views it as part of a broader system that shapes how environmental problems are understood and acted, with the goal of advancing a more adaptive and practical form of environmental intelligence.
diff --git a/_talks/2026-04-15-shiqi-yu.md b/_talks/2026-04-15-shiqi-yu.md
new file mode 100644
index 0000000..2f0a050
--- /dev/null
+++ b/_talks/2026-04-15-shiqi-yu.md
@@ -0,0 +1,23 @@
+---
+layout: "talk"
+title: "AI for Multi-Wavelength X-ray Analysis"
+date: "2026-04-15 10:00:00 -0600"
+permalink: "/talks/2026-04-15-shiqi-yu/"
+slug: "2026-04-15-shiqi-yu"
+start_time: "10:00 AM"
+end_time: "10:30 AM"
+series: "Data Science & AI Lecture Series"
+location: "Dinosaur Room (CSC 206)"
+zoom: "https://utexas.zoom.us/j/84742203545?pwd=lUosaf3T6bkIS1QAaIQYHYiiClE2ZP.1"
+canceled: false
+speakers:
+ - name: "Shiqi Yu"
+ bio: "Prof. Yu is a Research Assistant Professor in the Department of Physics and Astronomy at the University of Utah. Their research focuses on the intersection of high-energy astrophysics and advanced computational methods. As a member of the IceCube Collaboration and former co-lead of the Reconstruction and Machine Learning working group, they have extensive experience applying machine learning techniques to diverse astrophysical data. Their current work focuses on developing neural network architectures for multi-wavelength analysis."
+speaker_names: "Shiqi Yu"
+source_file: "_data/talks/2026-04-15-shiqi-yu.toml"
+generated: true
+---
+
+
+
+Accurate parameter estimation in X-ray astronomy typically relies on traditional methods, such as likelihood-based spectral fitting, which can be computationally prohibitive as model complexity and data dimensionality increase. In this talk, I present a neural network-based framework designed to bypass iterative fitting by directly mapping spectral observations to physical parameters. Using the Circinus galaxy as a benchmark, we demonstrate how architectures trained on synthetic data from theoretical models can recover intrinsic properties, such as column density and torus geometry, with both high speed and high precision. This approach maintains the physical rigor required for broadband analysis with multiple telescopes while significantly reducing inference time. I will discuss the challenges and resolutions associated with training and predicting on multi-instrument data, as well as the potential for these AI-driven methods to enable large-scale systematic studies across various astrophysical sources and other scientific applications.
diff --git a/_talks/2026-04-17-daniel-sieta.md b/_talks/2026-04-17-daniel-sieta.md
new file mode 100644
index 0000000..145bc6c
--- /dev/null
+++ b/_talks/2026-04-17-daniel-sieta.md
@@ -0,0 +1,24 @@
+---
+layout: "talk"
+title: "Multimodal Data Augmentation for Data-Efficient Robot Manipulation"
+date: "2026-04-17 13:30:00 -0600"
+permalink: "/talks/2026-04-17-daniel-sieta/"
+slug: "2026-04-17-daniel-sieta"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB L112"
+zoom: "https://utah.zoom.us/j/85983626630"
+canceled: false
+speakers:
+ - name: "Daniel Sieta"
+ affiliation: "USC"
+ bio: "Daniel Seita is an Assistant Professor in the Computer Science department at the University of Southern California and the director of the Sensing, Learning, and Understanding for Robotic Manipulation (SLURM) Lab. His research interests are in computer vision, machine learning, and foundation models for robot manipulation, focusing on improving performance in visually and geometrically challenging settings. Daniel was a postdoc at Carnegie Mellon University's Robotics Institute and holds a PhD in computer science from the University of California, Berkeley. Daniel has been honored with the AAAI 2026 New Faculty Highlights program. He presents his work at premier robotics conferences such as ICRA, IROS, RSS, and CoRL."
+speaker_names: "Daniel Sieta"
+source_file: "_data/talks/2026-04-17-daniel-sieta.toml"
+generated: true
+---
+
+
+
+Despite recent advances, learning-based robot manipulation systems often require large demonstration datasets and degrade in cluttered or deformable environments. This talk presents diffusion-based multimodal data augmentation methods that synthesize consistent observations and action labels. By augmenting limited demonstrations, these approaches substantially reduce data requirements and enable robust manipulation in complex, real-world settings.
diff --git a/_talks/2026-08-28-warren-pettine.md b/_talks/2026-08-28-warren-pettine.md
new file mode 100644
index 0000000..dc341e0
--- /dev/null
+++ b/_talks/2026-08-28-warren-pettine.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "About Start-UP MTN: which involves Comptutational Neuroscience and AI"
+date: "2026-08-28 13:30:00 -0600"
+permalink: "/talks/2026-08-28-warren-pettine/"
+slug: "2026-08-28-warren-pettine"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 2250"
+canceled: false
+speakers:
+ - name: "Warren Pettine"
+ affiliation: "UU Psychiatry & MTN"
+ website: "https://medicine.utah.edu/faculty/warren-pettine"
+speaker_names: "Warren Pettine"
+source_file: "_data/talks/2026-08-28-warren-pettine.toml"
+generated: true
+---
+
+
diff --git a/_talks/2026-09-04-george-vega-yon.md b/_talks/2026-09-04-george-vega-yon.md
new file mode 100644
index 0000000..baad087
--- /dev/null
+++ b/_talks/2026-09-04-george-vega-yon.md
@@ -0,0 +1,21 @@
+---
+layout: "talk"
+title: "Data Science of Tracking Measles in Utah"
+date: "2026-09-04 13:30:00 -0600"
+permalink: "/talks/2026-09-04-george-vega-yon/"
+slug: "2026-09-04-george-vega-yon"
+start_time: "1:30 PM"
+end_time: "2:30 PM"
+series: "Data Science & AI Lecture Series"
+location: "WEB 2250"
+canceled: false
+speakers:
+ - name: "George Vega Yon"
+ affiliation: "UU Epidemiology"
+ website: "https://medicine.utah.edu/faculty/george-g-vega-yon"
+speaker_names: "George Vega Yon"
+source_file: "_data/talks/2026-09-04-george-vega-yon.toml"
+generated: true
+---
+
+
diff --git a/scripts/__pycache__/import_calendar_talks.cpython-313.pyc b/scripts/__pycache__/import_calendar_talks.cpython-313.pyc
new file mode 100644
index 0000000000000000000000000000000000000000..52f1b95756177e6053611367ce35854a958f68e2
GIT binary patch
literal 23077
zcmb_^d2~}(p61iOJS|@26^w;#jO7)KS!|4rV>>Ly2K+2SVuMSzWFxR8=RKK4aSA8t
zp4g!s(T*z>At(Kzx(}`d%rZBjSQZa|NA><_3aGvPvk>>bdp2y<*$?s^A5u?
zJi{qC<$&U_l2_V^TQ#8KRrFWQYv`|**V11ducN7dES77s_LnbtLV4SB@3gOFI7|wiL!IzKehj*=yjrLm*vS)pW
z++Tdf6H{8Bk#ZAP&E=IdXVrWqSHtB~mz~?f6;Rh^u9hpLt}1RTXQQrau8u3Bt{Tq4
z6;sz1uAVEQu3D~vE2XZjTq9RTU3FX&S593Hu9@3FUG-cGw~@LUP=gBUYD5h-QCAbU
zjjN=tW^OxYr>+*RmD@~R+s3wYRUIc8##PENV}@b(`ktv`#4J5vGjSlO9r)wIg7
z!=*siADW%>^8tInHFM6+dnWz7Td+^@{#kqAjK?l`X1zWazg4i$%}>mDC+!FP{^=Qy
zz1KD4@wr{Reaho;Hye#3XS{+P|6TTa0X|;ew7tbX>Gz%Y_`Dw9q^I5<@cU;v>|PW!
z&-(=Xq@01r7vQ}fB=!aT_K~B5M~reM=6!AtZ~w-)+ZAxNkm?C7-)OX7IOCl>V|Vc$
zQa_(mN&h@Q>9PB#>;Zm0aK=96=Ouq{z|(9@SJmwiCVB5%!0Si(B!em5<7q@cHxYFe#VWM@Hp@FTtG;j
z%jc%qO?f>t=uI?wl1B3R1NJHJ%nTZfBIhU17+r#0nDL@!cGOvFI4OU|>pLej(zwXa
z>$6W>vY$b&4s^9sa7}wUjCMS8mjY+}z8!X{T|$dgw{f{|$7!cE&s`#MS3T7jw*b1a
z!7e4S`#cxu3=n?1e4Cp5=#&dQI$FxV+U}Y`U-BNw&vVf$1W1F>YR_+-w|A1xlV8I{hXHcY`7wlV{Y!qI
z=a6s8A8hlS^Uv&_pLuz9dz!W$nKO3w+
z69~)+9W5<60@HpiUWB%Yiv2;=mbPud{1$1QwM@7K?<8qs(9$#O;SsB)&x7T}iyFEh
zL=7EWhdx;B?Cs+kdJfr-cqXy7Ja!JzJwnhb`*YG74(1aN3^}68UK{$bw(Q&_U88m~
zZLwJEDogn3~Kw7vkBelx05SVb`&%JQQj|Jnpv>?p!WM)n99@n{rDHm3i
zZ{iZ$Oz~%!m
z;ADn1*}jsSQ*ml}3z9dp8pfd+3L3}w?lIq}eJpTFWP(bwQB)WmDpARMM3pctZns1A581_V++yLh_Ai8cr=^Sid~Xy3l4t)2`;5yUf#_$Q!
z=rvD!183$Zn!Wy(XORO@JC9jCiu6L4zsEQ)$pJl_=j-3!=Q7ryO%`AL%L1woSfmRXN0Rc%~R8BIo(sk
z^_)Qt$DkO8g32);&!T2sHP*GSNH&n=ORot*q
za@#^e+^|7%7l&
zT=U2iw~#q_i(!1qZH!OlR1TBU*E|@n)HoyV%ZRJa%9-Np)d{e4RL=AQ@oAizx>N)=
z7TRDn!#Fj==GAG)7Zz2joF{~9o)NB-sWIPLGmdrH@lD$8Sr|SUWQ)
zFbvDs8K+{H49%Lysglzm=I_ffG0${E!OG|Q4;&imIMIKie`sXPSJzzM?HKd*5A~(I
zq)AwDx^A?m>7}Mq4UW3H?vA~qr#nv7JGvbmi0`QRg7m-_gdTl?Ie_WQoA4v*ro2F3
zGc$AriwaKE2y-*ufT)>uVOtk9bW;+If+yhiOu6P~0(7Q|TG!khV4K4r>M?uA3B?h$
zfHfd2qGkg7>N!zEKSebVh(}b+imG#-OQIII$TchQVCzMl9Ku1+9@r8<#sCBEcER+9ikPx_9{yV}0o9^%4vake_r6Ju+m
z>e}@cC8bgtu?soA!<@|9i+SV1-!yYlK?3LDKUa-4K<3k$2h~sc%Inhg5^9m|h+)zV
zYo6g0x18O{PSvjPsh#RvD|a+cjx5){h8ZOr8hwUsvMcio
zI;3`{S9UfXa$#bRmdG(dZOUYK<`=tAs+Z*7qnb<|d|D0^W-p?AfAGRXmx}qGBXyUX>uIl|CJ2Q4dNZTmUE$mF|G32UPI{yt5urF9dinVF0yc
zlNI&E83Ya#Rr7&K5gbw4Lpk!2MDDbL$Hh;c;YnxmHF!C6nMK4m6JJ7_Km`M?**QR|
zpw}%b13`hzQcx-KNR&!1Q7U(U??uo!{)GRI7gn0i7}u2~t%aaM*ABlnxcGchXL|FS
zuYEH#nb4KTbmfE|vDer{UQH~oCgO}XcEP_Yh}z@`F&;FS{>C_6NuG*r9E1-C2~H_HZS&lR#W$_(ooHf`fK&!
z<8L?JH`qUzSbPq(S@OKG@W7gX^_AsU64vUNwK~$j(jB+%O;~$l*4~d>qt@Q2bttMI
zB3bTT-kV@6V{Bza6S){?wAWt80$WqYIQ-u18|O)}~!$;jHm(^Cu`lX_(I
zD}#iu0YQL?or>Qgz6J~Ew}}ri*{b;Ig*Kc6+*D!vmX~d=MZgT1&P2YqX68C-B?U
z)%3)^$Uz9v*Jk9Nstso$1GTD7xzkwx6>0L;rvc2#0mWni5_Y!yp`f|0+w16@;K8r8
zbazQGMo;}(=DLDPqcI31gBY&S(_N?P&+wk9eN~_vRUjKx)T@hPq0SapS5PVRM>-h$~`rV`GVRp-zz&BUk#eZd`%r4-^9O;=6asQ43>`h
z4*MY{A>>9DQ4M-LfsWvYG2amo51)P357yW<;hzuKNr>GQ=l~TVn+{JJEUM20W@kjb
z4DTs%;;Tpqs0zeJ4yCB#T^D$=?PTxUqKPCN5BMpZ;s;2EBuWymG)omNsEYKhRMQnN
zb6E}H^N~QDZCvb2`9ur$-3@-+7H5y8y+ZZjiAYVH-Aa6{SNC1n2TpXP9`izewd?MxZD^Q6&SN4RC((je+F}6vH2G6Li=i7k|@7px#O+QsJdi5H6d$`0+0nbz78)9wj?#feF~?dkU60glr893
zuoQsV{;HgOcAD`xrW`Diz_y_DDQl3^jk+B*6}7xHxGFlXqJ|3IDEd7opgl}_<1iA$Jx+u9V+eQ+e*!^k7-N&=
z>ftMgqb1w#w8c$(7W@CcaeJI?{Xu`E@9MzvKv?^!sq%JzjBSmoTYvKj+bR%Jb*sEx
zy^76C)8n&%8Bjj}RnmXTW;J?kRcukKe?IHY=z$-5W5UtXgb}hbow0W^fVnN`_lmlXqf>lQfj2-vRzVrj5+c(
z(tIy`jeOZ0`J$Y`cW7Hp*?GH&|9iB(CYt~IS&W)_h*BJfT2u%9{#ntGmcoi!Ni-|!
zE_fych&Y2f5~F!u2(q&vAt_lOKa4I6vX?+v<|jN<@~5XC%@ZD;BVjFdQbr47&2#Rl
z8uBw2xY%6-5f)@u9^UPlJhvc4IRm-0gC8NOPSugp#!%b>DQzrO+(N1#>30l;EXeZ3
zbQEEJj>z=W-?J|7%&g0|Kx9T^%_Lg~??f38MQLXI7d*VQB8Vu+q2`D1BdX9G{wQ@J
z2A#4Q6lIy5aS4KV>QZV!yoh-J4u8U*;sq>KSQL5gPSw4Fn6YoM_ZN9ZOX{Sh;Og1s
zvv18TsgkCmaADlEF=D<`7`M0GoxE53vvbMvO(E@D14+w>;Dr
z_?#ly`~0s}ih_fRFB$Sq9^;}TrxM3b$Bvy&R%{L#-WqyT#N-!0EM`pB#l!1qnhuSj
zH0@6?f9Q%-5=kGYAuIBkcnM;uTr>>(E6I1wBd=c>qY&nh)01~rIr=q;R)^+51x+YP
zUVA>}XnJ1<4ap%2$q0(%TKH5D0;#hEBxzv>sL|h~n937QjijO+O^Zh|LXi$IWUeBt
zyq!`UvMPu1z4npYf7FC^^{S>>8Qe9GbIi{4aU{4i^@?CVohF(*IDijiYHWxOm?N-7|O;uI`!_4@oy
zuE?WnXY9=2&n2oh__rzQrUqJLgU=nChiH7n8-UiX#ojZ4{mwNRfb%)8-%e#d&^-YZ
zL3`%PXvFMU%ChEQJ_J)l7(XU7IJyHQt6-ihmD7SSKRfHRbs&AaB#KstYp
zD3~1G3lij_xCsDg@`^lexOVl92tRNbu(7{yh#&3bQ!z4io
zEvgHSlWb6}lG>%6A0}Pc0Q9_CeFHeS*KS#1tPx#%J_-|0B&jm(V?YTI|3k0DO
z$}2w-A8VoVzHvDx{ybv+2mA>oKnIHqIM0IGq-|@`+Mg`k`pBTMXcwP!gH*AaHwf?OSsFd*f}UWghAMY2D_w{o)tHRwB(wQ
zUtlox8h-T|HRUv`DZTiQ>xBqeR`l(l18(+mYSWTf}KmO+F
zl1hbkEbKR|!r`O`caS+jEKsUJSBcygV~icspTCMOvjl0LXWr3(_M!zuNifaqR!u1!
zilN&K_(f&&Hc>IRidi&tPJ4W0U3ZE0$r+E!H=ZJ&&7D-OEp#=fWBds!AP_m`7g*8J
z!us#bhd2LFaiiy2&v)j3Bak`ty0J_VGL$JG&}VylhI;#t^!JSq_e%_|iR&Le)X$9%
z9X-VLOO#M6sjhf8(o)`5B54!&bZ9_eB}$%Rg?xT5l#qm<6E%>U1q6ZY-jobUG|0TF
z0sQYG^h5j!5xfZJTYR;9xjWn%zCiicU*=aVX~@bwf93r3i5uQ;dn1~7aqWs~W#p#?
zKeK&gi&?v_s2{1Af=z^1g&s0mP`6UO5{R>H8HTn#JQ-<=vvuG{TXuXHydAt-{lVq9
zsrv_)S2kaLVfls7_^0MAw=XaDM@`*H)^c^v@}5{>{muD!VLce*sJi|)|Eyxn`xOFN
z0)J*N@3pPw!iYDe2nBe9B7iPPPX-+g=cEnt`~)YfNUsGTu3U2l^B=Q`%om>$yqz*V
zt5DjsGD~%rp$IB!=7Yg-DmgV72cIfy)#s!qoF>B|5h1!PiC#k
z?MkPvqG%+ICFOhw2Pww|TFDizlf&Dp9ZQ=t&>Oy
zTF`!7mKqU@BrCskIji#1`80!@kgqluZE~j5U@Zf0QH7s2ogRYfEEKHG(0Oqm&WFq@sjbYQ~$J3()TIpiUJhZ6zqBqHc?HIO_ZIY5D7ZUOepM{C?1py
zy6RG(t~ouir3Y93w8P^cz}@0w570nVkOb`cFI1bw@E^-XX?ebE8^C?P`$*sZp~JwxPSO#&$NvrqQlIg-+NB9#4u+|ruUzKi&Ovg)`6Qymj(zZL?{nB<2vrkIfL)(9H{Oa*oX*;E4pRsvK
zYtaKJ1$Mp>3>|;ta?)xG9bdkYszwq9Ll?(B$yQ77>qo&;tDwIu??-s;_@p8J~!o2IhXoV
zQw8u=e(}E;nd+_I>J6!HRexgIwDR0y|7UDLvTRfM<(s8HoxHc@Cucw2{uAH#cSPn_
zj;{@JOhL=sl+2_~!3>l4bUHD{fXqgq5CnS@W`GNta{`uI^jj_f~hf
z|J{L`1K%BtvW}?Q@tc2cqV$}Ar2lAqUakJ27Oz!1Iw6->CyFnd(dVmB9#+TN2WUwF
z^ux%USS}u3UO7Z^DX&KtDEercvxi(AWs!$db&%DLuW9H*a_Jt@xhXkKRt{;}FDUmZ
zDIbm8Ic-M=!?-~oHz7aZjc1!)=rS_?bX{p}GV%aU53B{cD%V7=JNQS#5MUFL){i;p
zTKfV}mwBEp-7iSL2Gye-drz(EZbFhMfFi3W3qS!+`SUr3qhk-Oz#CiQrO7
zlY`XB6_y*;z@*^Vo)jEYP&fwrZ4^2LM5gl+(HPh~I9xMxXI#+6zk)DPaUmO~@vjoU
znso#t2T1z|#1shmp&-rD`(x)_MJRtcaCKpM;jM2btUDsdBQL*4t*O3u`u^q}_pLjA
z?EDRdQm=RQFp6v)goMT@*{VX0JdwwVDwugdWKGl#q8iq@9#M0F*yyejPxH$pK6%~8
zpKuAUwC*9TLrAbqkow2j7HAz=>-8O>z}xL%{_QRhlO&e#;pM|`9a-#8vZikdp`C99
z|Bl@hK7OCwgkWejO_lh4pfkLA>GeyYl5kBzR~ge$Kh_en^nT=wTYBTqQZ-5zOoVVt
zPi{tuFj`Mrb{QrtYx#i{$}L*UuqjW=^9gCw(Bx@pDLhd-nZ0O~>o~}OL2e(~kp^0_
z9~N=iu1Q{KoQA?XteR(~pc{HM1bAc_t$F0t04tBDgp%4my?;3^i9;-yYUsX9gb~=3
z>!7|HgJSPkr)vs!xA1YmPziJ)fO?1;>d6!mhJF`QZ*?+U7C+QguVKbT6(&kFC
z;lGR58ewy>w3m`vjMGBA)I^uRg{YsC=){594#9=~iawgZ=Wfl<>OZRgNn>1pV6o>j
zojz&JyLw^yLg@TkmlM{Sn6)NutzA)ksJpGZql?$=Nz`@4>bl}}`=i$VQT=|h1-ajN
z<+>1Vd$;3eM`ZH5-SLu!m6zis+pyc+ID74Er0{M3G7C|KweV_iIr!E>*!k}0&C&0k
zidwN@)~6L^;esf;A*$Z6UhY7KhweyPP~xw%BV{BHxrUYldXyK%)1=EjW>k-o=`)-|
zsz44$l(2W;mV-u|X6V@InUV)CZIl=CXu{-Zf--VgQ
z!329W#vYBc$I}cdQS|nvBDTd?Kssnp6Kr>kg%xd2I$EeM+#5l58EY;~g!Gf19)AQc
zqIgKrmU>DbvSX%_teOD1!n15>eF|b|L3S)oN8Xgh%DE#yGuJH3TACkKSYRlPL)0dK
zCpK(6k=qAVRaI5c-&c`;K{Y;%LA4P-qM{0>w^e)@B4H~gMm)TLSN3KM3n4!g?2udY
zw-NLu2|k4vvDsml=HGY~8`ITSmtPH!M0Ulk^$BZh%-VWq=iP?5^?9;LVO+cX&4?~$
z-5S+zg-jq+bfe;01ytoH;`xn<{I*y=M(OUBc>ewvyZ_#f82fxw{rvhh3r&W!X0HQG
zWVAuq5&}uXDid`JvqOolMmLl3q+fb`8Kf`HxI+cbj^U`L(t@L1_%^S~jaq)J%FVhe
zJdfnGDm;Hxc)_~x!d2n6b>T&;!i(31m!Q6-#8(DfM6=43fdek5?c1;_{l<0al`U4H
z%jPOb2>AE`A~yxV+r(97rBrhEbtyNm3a?rhUcD;3W?lFeu69+7t?ObaTXapJ5s<15
zrucsa4tP1Z`mA&e*RU>K)%HX3
z`J}aHxMaJSUisuzQV_!8B`5wyVWjt_vvoh6=CFg#9w^D7wYpb6@eOfazwq+dY;B#h+n7jlA)A3e6h
z5;-VQj}TA2{2{5{;U#I?4XHy@_#UD8r8J^gPmz4&57-9s^MKGeFCut?
ztgK-osm72Bx*3UO7lNkIQQ3X!R8U9Y2Z=>cO3;#pLrC^))_2gDvKAn*og}t#?u;L2
z+t4wffzV`;07AgfxwD>t>r}8@kPfgRhID+WNfxG51+B7keBMi&qQY;`A|2cExW{Qf
z1q*#RbSI3b#&I7x2j`(Q7zm6IO&$yp!~YO3DA)WrJtG>tf}b3Ha|QSbc*7DvO-7np
zz<2~I=~%^Kh%Ap$^YAcqnw{Z)j3i&-PxwphHssvHNK$W1>hqGug6n(2eUXxzgOIpJ
z3mT&O#-uSnR1-RXtueGIs<$VNrD0omCRWxEEp1%siRzn^#^O*Qd^lFJHCkM^q96#$
zSa|&)$a&q(({Wp4w6KZBD!P6KMEh{8vL#-;En2ias&7Ts*XzP<5$(;+ctKq>-|@u6
z#n&%J)REDceS5s5HCntQs&7jgi{SNA!oKa=)Sc;QQCC#o4Zj?v=#396^UT`C{)bhJ
ztu$=BUsw%!gx;Jql_xDE!G*~Ao0sFoEzzQFQPcLMW#g@rk?ED$SZ!y#VjpbYi7?ty
zdaE?Db;S{@YKxccjF#?#;3a7)Oo}0dSQB%~`95uBhO_qmNeSz^2qc<3@
zXdn}})jYH@8rsA9nxwJpR$ZiRMf2X?czH{-Y+F>n9o>4X;N6m&B}8iS-O5B+Ypkp_
zUe*>Z-5J&IO6n^xJ`c@0i}n$t(~y=xB_^4cJL69PsRWr*kopy9-fB@emo`r%A`rs;
zPp=Y{gXhA=K%Oja9zcL_^G_$2fS@q3Xs7rDDz%J21!tz^JfB83ab`u8-kbm@M$<4i
zmqB2Z7yM|A-;7&thO_ta<%?bLn8N-KB!bH)>sg{gD
zoEn0!oYoy**;vp
z&p5hS&cJDqTS5E$Z=f8pI8AbV7gk$V$s6TVnO~G#;55l4A62vvK-Q0OwK}a&>t_-!
zmqN``khjw^T%E~Tb~&w1vs1?vl3snnNZ2xF*IZF^dNn`^b07u6Ea|c^6F_A$GrmSH
zC-aLo6=#%Q>6FymFxPDfTuH}IA+tembEN>WWxhP85t1NBCPN&$8hL*(dWWhan|d
zrfVLqTplM#ZJr*U=hUOWHqidkaTGq|n}X$0sJx_SberrN-Xgm)znPVoE;aXB+?45g
zr)3ottCZ&iXKywGR&E{=b$uh;NYC(ysOcNQ`86v3CHi(8edl4}!Z`;Adxl>W*}i_R
zclgk;kwZs^ME#MYy*vntVR}@%K#s`6O>q~+SdcJLxzdkF
zYA1P7v4VQnE)sPh2mO*D>;KH6gBdH3Qu}+XNnX}Y?b^o{AGAL@}-kFXGyAamr0}{@*+(aO(~Quna>m1uxJ2UcjJsV%%(DfM|`oD12Zn2<;4SjT@^H
z#{S6s2mFT%KUlc8Io8q_+X@TT2bTQMmT+0zQk}3kViwpwH7p*0O6$5mQQQzKZdf^x
zC~k`tw;`ur6>PdCM0Tz;#Vhy33)&Y49uyXb=EG-UF_kE6i50fQ3%4&Gd0?#wpM$Jo
z@!+pa<+nN_=kD~yYWLjjjMW~9Z#Wn?9a`*9SxSWb5q_mJZruYrE6H{uTHFrZ+{L){
zV8S{SvkrZHA#QyE8I^Cpvo&7c5mT2e9!ct~Z!WyH5IU65ZHnnOB}*zJhFD2`qGVgF
zWE)KVONgm|NqeHCJ5~beW>2Ez_`R_ulbq~8Lboxd+nBU%48IVwZAsV~Vz!2rzPN4s
zoy{>@Tf){6vvtI6olAz#Slg}PcgManw$gmpk*Mg4RrDoo6-&_f<-w3^N%O1H=9Ptb
zX=hAb1d9!8{^G&Uc6D4=N(Va|Rz~jUwcb;HlJ{JeFm*}&OV+y7mNeO}9$h~A`9^!P
zx;a@{6L}H9Ke?g$-OihxE1DHPzM&OnZTSWNqSBcyk916aui{@|$ZNwGz*r4hQF_X%
zWWB
zE4{TLQPu$7X5|2|cigrUacR=>mEpK;8@?V`$xPc4x7r_Vu$Z(D8H;@68?!Y17~sl=
z-sync3X?3W0R;|B67ha6_(iRk+m-ewh2sPkxqPTy^~4x}lK|&W&Hgl)(rftI&yS2N%Xp-}Y69hb9mdF@K2bS5GV*yS`C`&^%jW5_>wTeCAq2QmjY
zhlyNo&6CSQIRiZ8jTI$7;b^Md%3Lka6G&
zxxx%2l*^qvPi`AD|-CKGVs+q6$FL1~{C$;T`Ks0pP|?L>@bnHQH^GU-mm!9W&$aG@;~;JJM{n&-{$s
zc4y};xt;6Tu|m{aF-&AdYaV&PDB%CWsltW*XD;+Kthi!aA(Pc-ol*ilKkpr>bP86iSjKm->2dTX1pLxV8mNw2H6+1XiphhBhs+hJTX-ZesmCdU*$KC+
zODx7IDqofbNXq{;OaY2U7n!gNl1T4L{z~WV%2-7^)gke3B3J&sT$KZTP<&aBkSrq^
z?3uEWOv;pnmDa$tY|26q@eohzD%-9I+3GUMcBN!ZWo4`5H%M)5$by=yRQ7$cqj$v>
zsr^F(9KWH8_t*0a6%yMKLbRZ!nF@U0A#Q^#b+!jBbh6Y54#~_
zFP!tj+Q!{s$E{8&nBNGUL;6~#Msn*Jl(s|=jeDRRRXFp(Ptzm%qNRtQrgt74BX^6y
zVrP8Z?VlVU7fmj=dmPbdplGDA$+sY?utY?)A6J|ZBR!WJ_Z;yvGhTYl(#yEUCm`zJ
zLBn;2dbx4u@DctG&@fTy$El229K~~c!O!aZdPaK2`wk7`jwVllUS~p?2%H5WSBHrj
zI{PV~B&r3Rv!gm=Dn^I3j@wOa199ttmR^2_^C9Ug(qOqmZc3xlbJ$el`YP0j$Fd~R
z)x$@RjyNpzt`*W=HEoiX-1;`-;kS}%>S1|^8?yutP9!2D+ye!hZK;b<5s$~m6WAy5
zV|PgcHBoBwO~msNUV;U$Y<@|;6+umA3(sI~Op*EVGWIYqO6H=3x#kmd&5ufA=62kN
zpf)DdCEzc@x@heO7gEZ=Ayw5f=BAme9VnThtCroQ7zHb22`
zjIkSWsClvPa~=EU#WycLFhgN*qwZQ=l(n6u-4_yq^U6U^74_T
z1Ah=8V6}B%Pj;TASG35hmaY(pv@!G{$W#65)yGAU+cx7L_q%Usm
zzc+c0k6I6-68VJL*ch^ayr{$Hqn276pJ6R;?WN_IuIiWdp~7#O33~`D7h-pI>)=Wu
zcqC_JHg0W$jVR3NcQ5b0+O^yTyDfT6#ESYu%ZCIqZicEiEZoX5SG!eCwX2s?^AQrSHH#pHhpSt{OZ@O
zDB<;DSr?&pohfezeVG(Jw|
z>-c!edbCDtB|+K_bUUf0vd1%DX3IL
zlP;DNm1<+A(zF8beSK9va_*~rzp@7Aq0Jpd_PB3SDr0r|{Kv>J
zMAA(P>7kPoZV5oG9|4ZT`EF!%?OWF&l
z$gP+74UiWVsZrq?6|?;=@%ugUS|%?lQX@7+k{vs|i+Dd&DHMuFYO_N9e>O3S>VIOazhH`f!LYwz4D`S8
z7fb!^*KT>FQfO*F*OoofC^hA85yl5
zYS|RmR4yt%hn?qz*Dkzy`L)X-{>FuC7j9g>b~(Zmv+?&Yzjyi0#rW3!5F5pH&n<#R
zO4OwjWKj)8CQkdbEb9MRYsgbBnsBxnWC=$Oqh+0UCw}Js$PF{mMK)PryHWq``iDw|
zrsi|C_RYT6`o7eu3zdu3$2NtcNAXy%Q)nOCSVh6(jS9uq$9l7(Pw}|cq$qv7NvYWJ
z*l1Q*AKQx+w#ThHMZ;rTsiOYzE(3{Es#9!yY*Qgkfm(4u@z_#>?-4~evg=beD9jJX
i6pY%m6pX9Nl4{+X2VXz9Jo46wzf%(vS9Lk<`u_)Vtr+|O
literal 0
HcmV?d00001
diff --git a/scripts/generate_talks.py b/scripts/generate_talks.py
new file mode 100644
index 0000000..5828406
--- /dev/null
+++ b/scripts/generate_talks.py
@@ -0,0 +1,235 @@
+#!/usr/bin/env python3
+"""Generate one Jekyll page per talk from the TOML records in `_data/talks/`.
+
+The TOML files are the source of truth; the markdown pages under `_talks/` are
+generated and should never be edited by hand (GitHub Pages builds Jekyll in
+safe mode, so the pages have to be generated ahead of time and committed).
+
+Usage:
+ python3 scripts/generate_talks.py # (re)generate _talks/*.md
+ python3 scripts/generate_talks.py --check # fail if pages are out of date
+"""
+
+from __future__ import annotations
+
+import argparse
+import datetime as dt
+import glob
+import json
+import os
+import sys
+import tomllib
+from zoneinfo import ZoneInfo
+
+ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+DATA_DIR = os.path.join(ROOT, "_data", "talks")
+OUT_DIR = os.path.join(ROOT, "_talks")
+TZ = ZoneInfo("America/Denver")
+BANNER = ""
+
+SPEAKER_FIELDS = ("name", "affiliation", "role", "website", "photo", "email", "bio")
+TALK_FIELDS = (
+ "series",
+ "location",
+ "room",
+ "zoom",
+ "slides",
+ "recording",
+ "paper",
+ "canceled",
+ "tags",
+)
+
+
+class TalkError(Exception):
+ pass
+
+
+def yaml_scalar(value) -> str:
+ if isinstance(value, bool):
+ return "true" if value else "false"
+ if isinstance(value, (int, float)):
+ return str(value)
+ return json.dumps(str(value), ensure_ascii=False) # JSON strings are valid YAML
+
+
+def yaml_block(data: dict, indent: int = 0) -> list[str]:
+ lines = []
+ pad = " " * indent
+ for key, value in data.items():
+ if value in ("", None, [], {}):
+ continue
+ if isinstance(value, list):
+ if all(isinstance(item, dict) for item in value):
+ lines.append(f"{pad}{key}:")
+ for item in value:
+ inner = yaml_block(item, indent + 4)
+ if not inner:
+ continue
+ lines.append(f"{pad} - " + inner[0].strip())
+ lines.extend(inner[1:])
+ else:
+ lines.append(f"{pad}{key}: [" + ", ".join(yaml_scalar(v) for v in value) + "]")
+ else:
+ lines.append(f"{pad}{key}: {yaml_scalar(value)}")
+ return lines
+
+
+def parse_time(value, field: str, source: str) -> dt.time | None:
+ if value in (None, ""):
+ return None
+ if isinstance(value, dt.time):
+ return value
+ for fmt in ("%H:%M", "%H:%M:%S", "%I:%M %p", "%I%p"):
+ try:
+ return dt.datetime.strptime(str(value).strip().upper(), fmt.upper()).time()
+ except ValueError:
+ continue
+ raise TalkError(f"{source}: could not parse {field} {value!r} (use e.g. \"13:30\")")
+
+
+def load_talk(path: str) -> dict:
+ source = os.path.basename(path)
+ with open(path, "rb") as handle:
+ try:
+ raw = tomllib.load(handle)
+ except tomllib.TOMLDecodeError as error:
+ raise TalkError(f"{source}: invalid TOML ({error})") from error
+
+ talk = raw.get("talk")
+ if not isinstance(talk, dict):
+ raise TalkError(f"{source}: missing a [talk] table")
+ if not talk.get("title"):
+ raise TalkError(f"{source}: [talk] needs a title")
+
+ date = talk.get("date")
+ if isinstance(date, dt.datetime):
+ date = date.date()
+ if not isinstance(date, dt.date):
+ raise TalkError(f"{source}: [talk] needs a date (e.g. date = 2026-09-04)")
+
+ speakers = raw.get("speakers") or []
+ if not isinstance(speakers, list) or not speakers:
+ raise TalkError(f"{source}: needs at least one [[speakers]] entry")
+ for speaker in speakers:
+ if not speaker.get("name"):
+ raise TalkError(f"{source}: every [[speakers]] entry needs a name")
+
+ start = parse_time(talk.get("start_time"), "start_time", source)
+ end = parse_time(talk.get("end_time"), "end_time", source)
+ starts_at = dt.datetime.combine(date, start or dt.time(13, 30), tzinfo=TZ)
+
+ slug = talk.get("slug") or os.path.splitext(source)[0]
+ front = {
+ "layout": "talk",
+ "title": talk["title"],
+ "date": starts_at.strftime("%Y-%m-%d %H:%M:%S %z"),
+ "permalink": f"/talks/{slug}/",
+ "slug": slug,
+ }
+ if start:
+ front["start_time"] = starts_at.strftime("%-I:%M %p")
+ if end:
+ front["end_time"] = dt.datetime.combine(date, end, tzinfo=TZ).strftime("%-I:%M %p")
+ for field in TALK_FIELDS:
+ if field in talk:
+ front[field] = talk[field]
+ front["speakers"] = [
+ {field: speaker[field] for field in SPEAKER_FIELDS if speaker.get(field)}
+ for speaker in speakers
+ ]
+ front["speaker_names"] = ", ".join(speaker["name"] for speaker in speakers)
+ front["source_file"] = f"_data/talks/{source}"
+ front["generated"] = True
+
+ body = str(talk.get("abstract", "")).strip()
+ return {"source": source, "slug": slug, "front": front, "body": body}
+
+
+def render(talk: dict) -> str:
+ lines = ["---"]
+ lines += yaml_block(talk["front"])
+ lines += ["---", "", BANNER % talk["source"], ""]
+ if talk["body"]:
+ # abstracts are prose, so protect any accidental Liquid from the build
+ if "{{" in talk["body"] or "{%" in talk["body"]:
+ lines += ["{% raw %}", talk["body"], "{% endraw %}", ""]
+ else:
+ lines += [talk["body"], ""]
+ return "\n".join(lines)
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument(
+ "--check",
+ action="store_true",
+ help="do not write anything; exit non-zero if pages are out of date",
+ )
+ args = parser.parse_args()
+
+ os.makedirs(OUT_DIR, exist_ok=True)
+ sources = sorted(
+ path
+ for path in glob.glob(os.path.join(DATA_DIR, "*.toml"))
+ if not os.path.basename(path).startswith("_")
+ )
+
+ pages, errors, seen = {}, [], {}
+ for path in sources:
+ try:
+ talk = load_talk(path)
+ except TalkError as error:
+ errors.append(str(error))
+ continue
+ if talk["slug"] in seen:
+ errors.append(
+ f'{talk["source"]}: duplicate talk slug "{talk["slug"]}" '
+ f'(also used by {seen[talk["slug"]]})'
+ )
+ continue
+ seen[talk["slug"]] = talk["source"]
+ pages[os.path.join(OUT_DIR, f'{os.path.splitext(talk["source"])[0]}.md')] = render(talk)
+
+ if errors:
+ for error in errors:
+ print(f"error: {error}", file=sys.stderr)
+ return 1
+
+ existing = set(glob.glob(os.path.join(OUT_DIR, "*.md")))
+ stale = sorted(existing - set(pages))
+ changed = sorted(
+ path
+ for path, content in pages.items()
+ if not os.path.exists(path) or open(path, encoding="utf-8").read() != content
+ )
+
+ if args.check:
+ if changed or stale:
+ for path in changed:
+ print(f"out of date: {os.path.relpath(path, ROOT)}", file=sys.stderr)
+ for path in stale:
+ print(f"orphaned: {os.path.relpath(path, ROOT)}", file=sys.stderr)
+ print(
+ "\nRun `make talks` (or python3 scripts/generate_talks.py) and commit the result.",
+ file=sys.stderr,
+ )
+ return 1
+ print(f"{len(pages)} talk page(s) up to date")
+ return 0
+
+ for path in changed:
+ with open(path, "w", encoding="utf-8") as handle:
+ handle.write(pages[path])
+ for path in stale:
+ os.remove(path)
+
+ print(
+ f"{len(pages)} talk page(s) in {os.path.relpath(OUT_DIR, ROOT)}/ "
+ f"({len(changed)} written, {len(stale)} removed)"
+ )
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/scripts/import_calendar_talks.py b/scripts/import_calendar_talks.py
new file mode 100644
index 0000000..373a09a
--- /dev/null
+++ b/scripts/import_calendar_talks.py
@@ -0,0 +1,551 @@
+#!/usr/bin/env python3
+"""Import talk records from the seminar's public Google Calendar feed.
+
+This is a *seeding / convenience* tool: it turns calendar entries into TOML
+records under `_data/talks/`, which are then the source of truth for the site.
+Calendar descriptions are free-form, so the parsing here is best effort; every
+imported record should be reviewed (and the fields it could not fill in, such
+as slides or recording links, filled in by hand).
+
+Usage:
+ python3 scripts/import_calendar_talks.py # fetch + import new talks
+ python3 scripts/import_calendar_talks.py --overwrite # also rewrite existing files
+ python3 scripts/import_calendar_talks.py --ics cal.ics # use a local .ics file
+ python3 scripts/import_calendar_talks.py --since 2025-01-01
+"""
+
+from __future__ import annotations
+
+import argparse
+import datetime as dt
+import html
+import os
+import re
+import sys
+import unicodedata
+import urllib.request
+from zoneinfo import ZoneInfo
+
+CALENDAR_ID = "ekol7ulqm14nv155angut2rlfo@group.calendar.google.com"
+ICS_URL = (
+ "https://calendar.google.com/calendar/ical/"
+ + CALENDAR_ID.replace("@", "%40")
+ + "/public/basic.ics"
+)
+TZ = ZoneInfo("America/Denver")
+
+ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+DATA_DIR = os.path.join(ROOT, "_data", "talks")
+
+# Boilerplate that shows up in the calendar summaries and is not part of a title.
+SERIES_NOISE = [
+ "UCDS+AI Lecture Series",
+ "UCDS+AI Seminar",
+ "UCDS + AI Seminar",
+ "Data Science and AI Seminar",
+ "Data Science & AI Seminar",
+ "Data Science Seminar",
+ "Data Seminar",
+ "UCDS Seminar",
+]
+
+CANCELED_RE = re.compile(r"\[?\b(cancell?ed|postponed)\b\]?", re.I)
+SKIP_SUMMARY_RE = re.compile(
+ r"^\s*(no (seminar|talk|lecture)|tba|tbd|holiday|spring break|fall break|"
+ r"reserved|placeholder|hold\b|organizational|planning meeting|"
+ r"(ucds\+?a?i? ?)?(seminar |lecture series )?(kick ?off|welcome|social|lunch|"
+ r"open (house|discussion)))",
+ re.I,
+)
+
+
+# --------------------------------------------------------------------------- #
+# ICS parsing
+# --------------------------------------------------------------------------- #
+def unfold(text: str) -> str:
+ return re.sub(r"\r?\n[ \t]", "", text.replace("\r\n", "\n"))
+
+
+def unescape_ics(value: str) -> str:
+ return (
+ value.replace("\\n", "\n")
+ .replace("\\N", "\n")
+ .replace("\\,", ",")
+ .replace("\\;", ";")
+ .replace("\\\\", "\\")
+ )
+
+
+def parse_events(ics_text: str) -> list[dict]:
+ events = []
+ for block in re.findall(r"BEGIN:VEVENT\n(.*?)\nEND:VEVENT", unfold(ics_text), re.S):
+ event = {}
+ for line in block.split("\n"):
+ m = re.match(r"^([A-Z-]+)((?:;[^:]*)?):(.*)$", line)
+ if not m:
+ continue
+ key, params, value = m.group(1), m.group(2), m.group(3)
+ event.setdefault(key, (params, unescape_ics(value)))
+ events.append(event)
+ return events
+
+
+def get(event: dict, key: str) -> str:
+ return event.get(key, ("", ""))[1]
+
+
+def parse_dt(event: dict, key: str) -> dt.datetime | None:
+ if key not in event:
+ return None
+ params, value = event[key]
+ if value.endswith("Z"):
+ stamp = dt.datetime.strptime(value, "%Y%m%dT%H%M%SZ").replace(
+ tzinfo=dt.timezone.utc
+ )
+ return stamp.astimezone(TZ)
+ tzid = re.search(r"TZID=([^;:]+)", params)
+ tz = ZoneInfo(tzid.group(1)) if tzid else TZ
+ if "T" in value:
+ return dt.datetime.strptime(value, "%Y%m%dT%H%M%S").replace(tzinfo=tz)
+ return dt.datetime.strptime(value, "%Y%m%d").replace(tzinfo=tz)
+
+
+# --------------------------------------------------------------------------- #
+# Field extraction
+# --------------------------------------------------------------------------- #
+def html_to_text(raw: str) -> str:
+ text = re.sub(r"(?i) ", "\n", raw)
+ text = re.sub(r"(?i)
", "\n\n", text)
+ text = re.sub(r'(?i)]*href="([^"]+)"[^>]*>(.*?)', r"\2 (\1)", text)
+ text = re.sub(r"<[^>]+>", "", text)
+ text = html.unescape(text).replace("\xa0", " ")
+ text = re.sub(r"[ \t]+\n", "\n", text)
+ text = re.sub(r"\n{3,}", "\n\n", text)
+ # Google Meet boilerplate appended by Calendar
+ text = re.split(r"\n-::~:~::.*", text)[0]
+ text = re.split(r"\nJoin with Google Meet:", text)[0]
+ text = re.split(r"\nLearn more about Meet at:", text)[0]
+ return text.strip()
+
+
+def find_links(text: str) -> list[str]:
+ links = re.findall(r"https?://[^\s<>\"')\]]+", text)
+ return [link.rstrip(".,;)") for link in links]
+
+
+def classify_links(links: list[str]) -> dict:
+ out = {"zoom": "", "recording": "", "slides": "", "website": ""}
+ for link in links:
+ low = link.lower()
+ if ("zoom.us" in low or "meet.google" in low) and not out["zoom"]:
+ out["zoom"] = link
+ elif ("youtube.com" in low or "youtu.be" in low) and not out["recording"]:
+ out["recording"] = link
+ elif re.search(r"(slides|\.pdf$|\.pptx?$|speakerdeck|slideshare)", low) and not out["slides"]:
+ out["slides"] = link
+ elif not re.search(r"(zoom\.us|meet\.google|youtube|youtu\.be|calendar\.google|"
+ r"support\.google|mailman|utah\.zoom|map\.utah\.edu|"
+ r"maps\.google|goo\.gl/maps|/map)", low) and not out["website"]:
+ out["website"] = link
+ return out
+
+
+BOILERPLATE_LINE_RE = re.compile(
+ r"^(?:the\s+)?(?:utah\s+center\s+for\s+)?data\s+science(?:\s*(?:&|and)\s*ai)?"
+ r"\s*(?:seminar|lecture series)?\s*$|"
+ r"^(?:ucds|ucds\+ai)\b.*$|"
+ r"^join zoom meeting$|^meeting id\b|^passcode\b|^one tap mobile$|"
+ r"^dial by your location$|^\+?\d[\d\s().,*#+-]{6,}$|^find your local number|"
+ r"^join by (?:sip|h\.?323)$|^\d{3} \d{4} \d{4}$|^[\s.*#-]+$|"
+ r"^time:\s.*(?:mountain time|am|pm)\b.*$|"
+ r"^in person\b.*$|^zoom\b\s*:?\s*$|^https?://\S+$|^[\s.*#-]*$",
+ re.I,
+)
+
+
+def strip_boilerplate(text: str) -> str:
+ # blank lines are kept: they are what separates title / speaker / abstract blocks
+ kept = [
+ line
+ for line in text.split("\n")
+ if not line.strip() or not BOILERPLATE_LINE_RE.match(line.strip())
+ ]
+ return re.sub(r"\n{3,}", "\n\n", "\n".join(kept)).strip()
+
+
+def paragraphs(text: str) -> list[str]:
+ return [p.strip() for p in re.split(r"\n\s*\n", text) if p.strip()]
+
+
+def looks_like_title(text: str) -> bool:
+ if not 8 < len(text) <= 250:
+ return False
+ if looks_like_person(text.split("\n")[0]):
+ return False
+ return not re.match(r"(?i)^(abstract|bio|speaker|zoom|title)\b\s*:?$", text)
+
+
+def split_sections(text: str) -> dict:
+ """Pull Title / Abstract / Bio blocks out of a free-form description."""
+ labels = {
+ "title": r"(?:talk\s+)?title",
+ "abstract": r"abstract|summary",
+ "bio": r"bio(?:graphy|sketch)?|about the speaker|speaker bio",
+ "speaker": r"speaker|presenter",
+ }
+ pattern = re.compile(
+ r"^\s*(?P