diff --git a/.github/workflows/ci-core.yml b/.github/workflows/ci-core.yml index 673563f121..83ca41221c 100644 --- a/.github/workflows/ci-core.yml +++ b/.github/workflows/ci-core.yml @@ -57,16 +57,13 @@ jobs: - name: 🐍 Setup Pixi uses: prefix-dev/setup-pixi@v0.10.2 + with: + environments: tests + # Linting is covered by the pre-commit job (ruff, black, isort on all files) - name: 🧪 Run Python tests run: | - pixi run test-fast || echo "Tests completed" - - - name: 🔍 Check Python code quality - run: | - # Run ruff for linting - ruff check utils/ || true - ruff format utils/ --check || true + pixi run -e tests test-fast pre-commit: name: Pre-commit Checks diff --git a/_data/archive.yml b/_data/archive.yml index 96856727f1..5e076f5825 100644 --- a/_data/archive.yml +++ b/_data/archive.yml @@ -64,6 +64,90 @@ latitude: 51.3406321 longitude: 12.3747329 +- conference: PyOhio + year: 2026 + link: https://www.pyohio.org/2026/ + cfp_link: https://pretalx.com/pyohio-2026/cfp + cfp: '2026-04-20 23:59:00' + place: Cleveland, USA + start: 2026-07-25 + end: 2026-07-26 + sponsor: https://www.pyohio.org/2026/PyOhio-2026-Sponsorship-Prospectus.pdf + sub: PY + location: + - title: PyOhio 2026 + latitude: 41.4996574 + longitude: -81.6936772 + +- conference: Black Python Devs Leadership Summit + year: 2026 + link: https://blackpythondevs.com/bpd-events/black-python-devs-leadership-summit-2026-ohio.html + cfp_link: https://pretalx.com/pyohio-2026/submit/?track=6985-black-python-devs-leadership-summit + cfp: '2026-04-20 23:59:00' + place: Cleveland, USA + start: 2026-07-24 + end: 2026-07-24 + sub: PY + location: + - title: Black Python Devs Leadership Summit 2026 + latitude: 41.4996574 + longitude: -81.6936772 + +- conference: PyCon Armenia + year: 2026 + link: https://pycon.am/ + cfp: TBA + place: Yerevan, Armenia + start: 2026-07-24 + end: 2026-07-26 + sub: PY + location: + - title: PyCon Armenia 2026 + latitude: 40.1777112 + longitude: 44.5126233 + +- conference: PyCon Colombia + year: 2026 + link: https://2026.pycon.co/ + cfp_link: https://2026.pycon.co/#/call-for-proposals + cfp: '2026-05-12 23:59:00' + place: Medellín, Colombia + start: 2026-07-24 + end: 2026-07-26 + sponsor: https://2026.pycon.co/#/sponsors + sub: PY + location: + - title: PyCon Colombia 2026 + latitude: 6.2443382 + longitude: -75.573553 + +- conference: Python Sudeste + year: 2026 + link: https://ingressos.python.org.br/sudeste/ingressos + cfp: TBA + place: Rio de Janeiro, RJ + start: 2026-07-24 + end: 2026-07-26 + sub: PY + location: + - title: Python Sudeste 2026 + latitude: -22.9110137 + longitude: -43.2093727 + +- conference: EuroSciPy + year: 2026 + link: https://euroscipy.org/ + cfp_link: https://euroscipy.org/call-for-proposals/ + cfp: '2026-02-22 23:59:00' + place: Kraków, Poland + start: 2026-07-18 + end: 2026-07-23 + sub: SCIPY + location: + - title: EuroSciPy 2026 + latitude: 50.0478 + longitude: 19.9314 + - conference: EuroPython year: 2026 link: https://ep2026.europython.eu/ @@ -179,19 +263,6 @@ latitude: 47.3615882 longitude: 7.354445 -- conference: PyDay Valparaiso - year: 2026 - link: https://www.pyday.cl/valparaiso.html - cfp: TBA - place: Valparaiso, Chile - start: 2026-06-05 - end: 2026-06-05 - sub: DAY - location: - - title: PyDay Valparaiso 2026 - latitude: -32.5976089 - longitude: -70.8529753 - - conference: PyData London year: 2026 link: https://pydata.org/london2026 @@ -208,6 +279,34 @@ latitude: 51.5074456 longitude: -0.1277653 +- conference: PyDay Valparaiso + alt_name: PyDay Chile Valparaiso + year: 2026 + link: https://www.pyday.cl/valparaiso.html + cfp: TBA + place: Valparaiso, Chile + start: 2026-06-05 + end: 2026-06-05 + sponsor: https://pyday.cl/sponsors + sub: DAY + location: + - title: PyDay Chile Valparaiso 2026 + latitude: -32.5976089 + longitude: -70.8529753 + +- conference: Python Leiden User Group + year: 2026 + link: https://pythonleiden.nl/meeting-2026-05-28.html?py + cfp: TBA + place: Leiden, The Netherlands + start: 2026-05-28 + end: 2026-05-28 + sub: PY + location: + - title: Python Leiden User Group 2026 + latitude: 52.1594747 + longitude: 4.4908843 + - conference: PyCon Italy alt_name: PyCon Italia year: 2026 @@ -511,8 +610,9 @@ longitude: 77.590082 - conference: FOSDEM + alt_name: 'FOSDEM: Python Dev Room' year: 2026 - link: https://fosdem.org/2026/ + link: https://fosdem.org/2026/schedule/track/python cfp: TBA place: Brussels, Belgium start: 2026-01-31 @@ -520,7 +620,7 @@ sponsor: https://fosdem.org/2026/about/sponsors/ sub: PY location: - - title: FOSDEM 2026 + - title: 'FOSDEM: Python Dev Room 2026' latitude: 50.8465573 longitude: 4.351697 @@ -6111,7 +6211,7 @@ link: https://pycon.hk/2021/ cfp_link: https://forms.gle/roP6WDYGYK61eKX16 cfp: '2021-06-14 23:59:00' - cfp_ext: '2021-06-30' + cfp_ext: '2021-06-30 23:59:00' place: Hong Kong, Hong Kong start: 2021-10-08 end: 2021-10-09 @@ -6147,7 +6247,7 @@ link: https://2021.es.pycon.org/ cfp_link: https://docs.google.com/forms/d/e/1FAIpQLSeNa71Fi9w8tppJlYoACnBVf5GB2ywU9lfJiHO0Aj0lmamwlQ/closedform cfp: '2021-05-02 23:59:00' - cfp_ext: '2021-05-16' + cfp_ext: '2021-05-16 23:59:00' place: Online start: 2021-10-02 end: 2021-10-03 @@ -8287,7 +8387,7 @@ link: https://web.archive.org/web/20190130011628/https://cz.pycon.org/2019/ cfp_link: https://web.archive.org/web/20210518170623/https://cz.pycon.org/2019/proposals/ cfp: '2019-02-17 23:59:00' - cfp_ext: '2019-03-07' + cfp_ext: '2019-03-07 23:59:00' place: Ostrava, Czechia start: 2019-06-14 end: 2019-06-16 diff --git a/_data/conferences.yml b/_data/conferences.yml index f50279f603..5afbcb49dc 100644 --- a/_data/conferences.yml +++ b/_data/conferences.yml @@ -69,6 +69,22 @@ latitude: -22.5597 longitude: 17.0832 +- conference: PyCon Panamá + year: 2026 + link: https://pycon.pa/2026/ + cfp_link: https://pycon.pa/2026/ponentes + cfp: '2026-09-18 23:59:00' + timezone: America/Panama + place: Panama City, Panamá + start: 2026-10-22 + end: 2026-10-23 + sponsor: https://pycon.pa/2026/patrocinadores + sub: PY + location: + - title: PyCon Panamá 2026 + latitude: 8.9714493 + longitude: -79.5341802 + - conference: Wagtail Space year: 2026 link: https://wagtail.org/wagtail-space-2026/ @@ -304,6 +320,7 @@ place: Santa Monica, USA start: 2026-10-24 end: 2026-10-24 + sponsor: https://2026.pybeach.org/prospectus mastodon: https://fosstodon.org/@pybeach bluesky: https://bsky.app/profile/pybeach.bsky.social sub: PY @@ -331,6 +348,7 @@ - conference: PyCon Ghana year: 2026 link: https://gh.pycon.org/2026/ + cfp_link: https://gh.pycon.org/2026/talks/submit_talk cfp: '2026-06-06 23:59:00' timezone: Africa/Accra place: Accra, Ghana @@ -361,7 +379,7 @@ - conference: PyCon Taiwan alt_name: PyCon TW year: 2026 - link: https://tw.pycon.org/2026/ + link: https://tw.pycon.org/2026/en-us cfp_link: https://tw.pycon.org/2026/en-us/speaking/cfp cfp: '2026-06-01 23:59:59' place: Taipei, Taiwan @@ -446,12 +464,14 @@ - conference: PyCon Greece year: 2026 link: https://pycon.gr/ + cfp_link: https://2026.pycon.gr/en/program/cfp-faq cfp: '2026-05-10 23:59:00' cfp_ext: '2026-05-17 23:59:00' timezone: Europe/Athens place: Athens, Greece start: 2026-10-12 end: 2026-10-13 + sponsor: https://2026.pycon.gr/en/sponsors youtube: https://www.youtube.com/@pycongreece sub: PY location: @@ -501,6 +521,7 @@ place: Yaoundé, Cameroon start: 2026-09-17 end: 2026-09-19 + sponsor: https://cm.pycon.org/en/sponsor sub: PY location: - title: PyCon Cameroon 2026 @@ -616,7 +637,8 @@ - conference: PyCon Germany year: 2028 link: https://pycon.de/ - cfp: nan + cfp: TBA + timezone: Europe/Berlin place: Heidelberg, Germany start: 2028-04-24 end: 2028-04-29 @@ -644,6 +666,21 @@ latitude: 50.294113 longitude: 18.6657306 +- conference: PyOhio + year: 2027 + link: https://www.pyohio.org/ + cfp: TBA + place: Cleveland, USA + start: 2027-07-24 + end: 2027-07-25 + sponsor: https://www.pyohio.org/2026/sponsors/ + mastodon: https://fosstodon.org/@pyohio + sub: PY + location: + - title: PyOhio 2027 + latitude: 41.4996574 + longitude: -81.6936772 + - conference: PyCon Thailand year: 2027 link: https://th.pycon.org/ @@ -658,6 +695,37 @@ latitude: 13.7563 longitude: 100.5018 +- conference: PyCon Angola + year: 2027 + link: https://ao.pycon.org/ + cfp: TBA + place: Luanda, Angola + start: 2027-05-20 + end: 2027-05-26 + sub: PY + location: + - title: PyCon Angola 2027 + latitude: -8.8272699 + longitude: 13.2439512 + +- conference: PyCon US + year: 2027 + link: https://us.pycon.org/ + cfp: TBA + place: Long Beach, USA + start: 2027-05-12 + end: 2027-05-18 + sponsor: https://www.python.org/sponsors/application + finaid: https://us.pycon.org/2026/about/diversity/ + mastodon: https://fosstodon.org/@pycon + bluesky: https://bsky.app/profile/pycon.us + youtube: https://www.youtube.com/c/pyconus + sub: PY + location: + - title: PyCon US 2027 + latitude: 33.7690164 + longitude: -118.191604 + - conference: PyCon Austria year: 2027 link: https://pycon.at/ @@ -677,6 +745,7 @@ alt_name: PyCon Germany year: 2027 link: https://2027.pycon.de/ + cfp_link: https://pycon.de/ cfp: TBA timezone: Europe/Berlin place: Heidelberg, Germany @@ -690,6 +759,35 @@ latitude: 49.4093582 longitude: 8.694724 +- conference: PyLadiesCon + year: 2026 + link: https://2026.conference.pyladies.com/en/ + cfp: TBA + place: Online + start: 2026-12-05 + end: 2026-12-07 + sponsor: https://2026.conference.pyladies.com/en/sponsors/ + mastodon: https://fosstodon.org/@pyladiescon + bluesky: https://bsky.app/profile/pyladiescon.bsky.social + youtube: https://www.youtube.com/@PyLadiesGlobal + sub: PY + +- conference: PyCon Senegambia + year: 2026 + link: https://www.pyconsenegambia.org/ + cfp: TBA + timezone: Africa/Dakar + place: Dakar, Senegal + start: 2026-11-27 + end: 2026-11-28 + sponsor: https://www.pyconsenegambia.org/en/sponsorship + finaid: https://www.pyconsenegambia.org/en/#diversity + sub: PY + location: + - title: PyCon Senegambia 2026 + latitude: 14.693425 + longitude: -17.447938 + - conference: Python E-Commerce Forum year: 2026 link: https://codehut.co.uk/2026/ @@ -743,18 +841,19 @@ latitude: 52.3730796 longitude: 4.8924534 -- conference: PyCon Panamá +- conference: PyDay Mexico + alt_name: 'PyDay México: CDMX' year: 2026 - link: https://pycon.pa/2026/ + link: https://www.eventbrite.com.mx/e/pyday-mexico-2026-cdmx-tickets-1989262280029 cfp: TBA - place: Panama City, Panamá - start: 2026-10-22 - end: 2026-10-23 - sub: PY + place: Ciudad de México, México + start: 2026-10-10 + end: 2026-10-10 + sub: DAY location: - - title: PyCon Panamá 2026 - latitude: 8.9714493 - longitude: -79.5341802 + - title: PyDay Mexico 2026 + latitude: 19.3207722 + longitude: -99.1514678 - conference: PyCon Estonia alt_name: PyCon EE @@ -789,6 +888,21 @@ latitude: 50.8214626 longitude: -0.1400561 +- conference: PyCaxias + year: 2026 + link: https://pycaxias.com.br/ + cfp: TBA + timezone: America/Sao_Paulo + place: Caxias do Sul, Brasil + start: 2026-09-26 + end: 2026-09-26 + sponsor: https://pycaxias.com.br/#patrocinio + sub: PY + location: + - title: PyCaxias 2026 + latitude: -29.1685045 + longitude: -51.1796385 + - conference: Pythoncamp Rügen year: 2026 link: https://barcamps.eu/pythoncamp-ruegen-2026/ @@ -804,6 +918,20 @@ latitude: 54.4529015 longitude: 13.3882344 +- conference: PyDay Boyacá + year: 2026 + link: https://www.pyday.co/ + cfp: TBA + place: Sogamoso, Colombia + start: 2026-09-12 + end: 2026-09-13 + sponsor: https://www.pyday.co/#sponsors + sub: DAY + location: + - title: PyDay Boyacá 2026 + latitude: 5.7148307 + longitude: -72.9279328 + - conference: PyHEP year: 2026 link: https://indico.cern.ch/category/7971/ diff --git a/_data/legacy.yml b/_data/legacy.yml index d60861dd90..68baa10524 100644 --- a/_data/legacy.yml +++ b/_data/legacy.yml @@ -1206,7 +1206,7 @@ year: 2018 link: https://web.archive.org/web/20181107132342/http://www.pyparis.org/ cfp: '2018-09-30 23:59:00' - cfp_ext: '2018-10-07' + cfp_ext: '2018-10-07 23:59:00' place: Paris, France start: 2018-11-14 end: 2018-11-15 @@ -1232,8 +1232,8 @@ - conference: PyOhio year: 2017 link: https://web.archive.org/web/20180129225635/http://www.pyohio.org/2017/ + cfp_link: https://www.pyohio.org/2017/call-for-proposals/ cfp: '2017-05-25 23:59:00' - cfp_ext: https://www.pyohio.org/2017/call-for-proposals/ place: Columbus, USA start: 2017-07-29 end: 2017-07-30 @@ -2035,7 +2035,7 @@ year: 2016 link: https://web.archive.org/web/20160602152545/http://pydata.org/paris2016/ cfp: '2016-05-15 23:59:00' - cfp_ext: '2016-05-20' + cfp_ext: '2016-05-20 23:59:00' place: Paris, France start: 2016-06-14 end: 2016-06-15 @@ -3944,7 +3944,7 @@ year: 2016 link: https://web.archive.org/web/20160722184438/https://pycon.my/ cfp: '2016-05-28 23:59:00' - cfp_ext: '2016-06-10' + cfp_ext: '2016-06-10 23:59:00' place: Kuala Lumpur, Malaysia start: 2016-08-26 end: 2016-08-28 @@ -3960,7 +3960,7 @@ link: https://web.archive.org/web/20150812151551/http://www.pycon.my/pycon-my-2015 cfp_link: https://web.archive.org/web/20150812142938/http://www.pycon.my/call-for-proposal cfp: '2015-07-03 23:59:00' - cfp_ext: '2015-07-10' + cfp_ext: '2015-07-10 23:59:00' timezone: Asia/Kuala_Lumpur place: Kuala Lumpur, Malaysia start: 2015-08-21 @@ -5110,7 +5110,7 @@ link: https://web.archive.org/web/20180202200818/https://cz.pycon.org/2018/ cfp_link: https://web.archive.org/web/20180203120450/https://cz.pycon.org/2018/proposals/ cfp: '2018-02-28 23:59:00' - cfp_ext: '2018-03-10' + cfp_ext: '2018-03-10 23:59:00' place: Prague, Czechia start: 2018-06-01 end: 2018-06-03 @@ -6290,7 +6290,7 @@ year: 2013 link: https://web.archive.org/web/20130208160717/http://www.ploneconf.org/ cfp: '2012-08-15 23:59:00' - cfp_ext: '2013-08-19' + cfp_ext: '2013-08-19 23:59:00' place: Brasilia, Brazil start: 2013-09-30 end: 2013-10-06 diff --git a/pixi.lock b/pixi.lock index b2b77436a4..ffeff127af 100644 --- a/pixi.lock +++ b/pixi.lock @@ -5,6 +5,8 @@ environments: - url: https://conda.anaconda.org/conda-forge/ indexes: - https://pypi.org/simple + options: + pypi-prerelease-mode: if-necessary-or-explicit packages: linux-64: - conda: https://conda.anaconda.org/conda-forge/linux-64/_libgcc_mutex-0.1-conda_forge.tar.bz2 @@ -245,6 +247,8 @@ environments: - url: https://conda.anaconda.org/conda-forge/ indexes: - https://pypi.org/simple + options: + pypi-prerelease-mode: if-necessary-or-explicit packages: linux-64: - conda: https://conda.anaconda.org/conda-forge/linux-64/_libgcc_mutex-0.1-conda_forge.tar.bz2 @@ -469,10 +473,13 @@ environments: - url: https://conda.anaconda.org/conda-forge/ indexes: - https://pypi.org/simple + options: + pypi-prerelease-mode: if-necessary-or-explicit packages: linux-64: - conda: https://conda.anaconda.org/conda-forge/linux-64/_libgcc_mutex-0.1-conda_forge.tar.bz2 - conda: https://conda.anaconda.org/conda-forge/linux-64/_openmp_mutex-4.5-2_gnu.tar.bz2 + - conda: https://conda.anaconda.org/conda-forge/noarch/_python_abi3_support-1.0-hd8ed1ab_3.conda - conda: https://conda.anaconda.org/conda-forge/noarch/annotated-types-0.7.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/brotli-python-1.1.0-py311h1ddb823_4.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/bzip2-1.0.8-hda65f42_8.conda @@ -483,12 +490,15 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/charset-normalizer-3.4.3-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/coverage-7.10.6-py311h3778330_1.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/cpython-3.11.16-py311hd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/distlib-0.4.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/exceptiongroup-1.3.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/filelock-3.19.1-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/freezegun-1.5.5-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/h2-4.3.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hpack-4.1.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hyperframe-6.1.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/linux-64/hypothesis-6.168.1-py311hf77984d_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/icalendar-5.0.13-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/icu-75.1-he02047a_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/identify-2.6.14-pyhd8ed1ab_0.conda @@ -536,6 +546,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pytest-mock-3.15.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/python-3.11.13-h9e4cc4f_0_cpython.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/python-gil-3.11.16-hd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-levenshtein-0.27.1-pyhff2d567_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2025.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.11-8_cp311.conda @@ -544,8 +555,10 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/rapidfuzz-3.14.1-py311h1ddb823_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/readline-8.2-h8c095d6_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.31.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/responses-0.26.3-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/setuptools-80.9.0-pyhff2d567_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/sortedcontainers-2.4.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/thefuzz-0.22.1-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/tk-8.6.13-noxft_hd72426e_102.conda - conda: https://conda.anaconda.org/conda-forge/noarch/toml-0.10.2-pyhd8ed1ab_1.conda @@ -563,6 +576,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/zstd-1.5.7-hb8e6e7a_2.conda - pypi: https://files.pythonhosted.org/packages/aa/50/fc98c8754a5963399da16edbb82ea74d524d7fd6e732315cfcce6008e138/vladiate-0.0.26-py3-none-any.whl osx-arm64: + - conda: https://conda.anaconda.org/conda-forge/noarch/_python_abi3_support-1.0-hd8ed1ab_3.conda - conda: https://conda.anaconda.org/conda-forge/noarch/annotated-types-0.7.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/brotli-python-1.1.0-py311hf719da1_4.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/bzip2-1.0.8-hd037594_8.conda @@ -573,12 +587,15 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/charset-normalizer-3.4.3-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/coverage-7.10.6-py311ha9b3269_1.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/cpython-3.11.16-py311hd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/distlib-0.4.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/exceptiongroup-1.3.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/filelock-3.19.1-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/freezegun-1.5.5-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/h2-4.3.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hpack-4.1.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hyperframe-6.1.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/osx-arm64/hypothesis-6.168.1-py311hfcb3ee1_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/icalendar-5.0.13-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-75.1-hfee45f7_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/identify-2.6.14-pyhd8ed1ab_0.conda @@ -619,6 +636,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pytest-mock-3.15.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/python-3.11.13-hc22306f_0_cpython.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/python-gil-3.11.16-hd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-levenshtein-0.27.1-pyhff2d567_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2025.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.11-8_cp311.conda @@ -627,8 +645,10 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/rapidfuzz-3.14.1-py311h251fd82_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/readline-8.2-h1d1bf99_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.31.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/responses-0.26.3-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/setuptools-80.9.0-pyhff2d567_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/sortedcontainers-2.4.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/noarch/thefuzz-0.22.1-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/tk-8.6.13-h892fb3f_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/toml-0.10.2-pyhd8ed1ab_1.conda @@ -646,6 +666,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/zstd-1.5.7-h6491c7d_2.conda - pypi: https://files.pythonhosted.org/packages/aa/50/fc98c8754a5963399da16edbb82ea74d524d7fd6e732315cfcce6008e138/vladiate-0.0.26-py3-none-any.whl win-64: + - conda: https://conda.anaconda.org/conda-forge/noarch/_python_abi3_support-1.0-hd8ed1ab_3.conda - conda: https://conda.anaconda.org/conda-forge/noarch/annotated-types-0.7.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/brotli-python-1.1.0-py311h3e6a449_4.conda - conda: https://conda.anaconda.org/conda-forge/win-64/bzip2-1.0.8-h0ad9c76_8.conda @@ -656,12 +677,15 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/charset-normalizer-3.4.3-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/colorama-0.4.6-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/coverage-7.10.6-py311h3f79411_1.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/cpython-3.11.16-py311hd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/distlib-0.4.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/exceptiongroup-1.3.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/filelock-3.19.1-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/freezegun-1.5.5-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/h2-4.3.0-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hpack-4.1.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/hyperframe-6.1.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/win-64/hypothesis-6.168.1-py311h66afae6_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/icalendar-5.0.13-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/icu-75.1-he0c23c2_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/identify-2.6.14-pyhd8ed1ab_0.conda @@ -702,6 +726,7 @@ environments: - conda: https://conda.anaconda.org/conda-forge/noarch/pytest-mock-3.15.0-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/python-3.11.13-h3f84c4b_0_cpython.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-dateutil-2.9.0.post0-pyhe01879c_2.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/python-gil-3.11.16-hd8ed1ab_2.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-levenshtein-0.27.1-pyhff2d567_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python-tzdata-2025.2-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/python_abi-3.11-8_cp311.conda @@ -709,8 +734,10 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/pyyaml-6.0.2-py311h5082efb_2.conda - conda: https://conda.anaconda.org/conda-forge/win-64/rapidfuzz-3.14.1-py311h3e6a449_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/requests-2.31.0-pyhd8ed1ab_0.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/responses-0.26.3-pyhcf101f3_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/setuptools-80.9.0-pyhff2d567_0.conda - conda: https://conda.anaconda.org/conda-forge/noarch/six-1.17.0-pyhe01879c_1.conda + - conda: https://conda.anaconda.org/conda-forge/noarch/sortedcontainers-2.4.0-pyhd8ed1ab_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/tbb-2021.13.0-h18a62a1_3.conda - conda: https://conda.anaconda.org/conda-forge/noarch/thefuzz-0.22.1-pyhd8ed1ab_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/tk-8.6.13-h2c6b04d_2.conda @@ -755,6 +782,17 @@ packages: purls: [] size: 23621 timestamp: 1650670423406 +- conda: https://conda.anaconda.org/conda-forge/noarch/_python_abi3_support-1.0-hd8ed1ab_3.conda + sha256: 2a7204314663eeda5dec482a956f0e2eaf289bd5b9953eaaaad0e81aa64638f2 + md5: 3845f3d75991bae0fb90884662f4327c + depends: + - cpython + - python-gil + license: MIT + license_family: MIT + purls: [] + size: 8144 + timestamp: 1784221492234 - conda: https://conda.anaconda.org/conda-forge/noarch/annotated-types-0.7.0-pyhd8ed1ab_0.conda sha256: 668f0825b6c18e4012ca24a0070562b6ec801ebc7008228a428eb52b4038873f md5: 7e9f4612544c8edbfd6afad17f1bd045 @@ -1172,6 +1210,17 @@ packages: - pkg:pypi/coverage?source=hash-mapping size: 417327 timestamp: 1756930890538 +- conda: https://conda.anaconda.org/conda-forge/noarch/cpython-3.11.16-py311hd8ed1ab_2.conda + noarch: generic + sha256: be18e37b4ab19a72b2ceac449e4ad11e5e18756aeb1c2a5b7f8811bff0c2e030 + md5: 18b8c7456c7a5052ca5d0002957a6f29 + depends: + - python >=3.11,<3.12.0a0 + - python_abi * *_cp311 + license: Python-2.0 + purls: [] + size: 48577 + timestamp: 1788392044498 - conda: https://conda.anaconda.org/conda-forge/noarch/distlib-0.3.8-pyhd8ed1ab_0.conda sha256: 3ff11acdd5cc2f80227682966916e878e45ced94f59c402efb94911a5774e84e md5: db16c66b759a64dc5183d69cc3745a52 @@ -1246,6 +1295,18 @@ packages: - pkg:pypi/filelock?source=compressed-mapping size: 18003 timestamp: 1755216353218 +- conda: https://conda.anaconda.org/conda-forge/noarch/freezegun-1.5.5-pyhd8ed1ab_0.conda + sha256: 66340acff8a4015c41544797f99049713076986a24b86b349ec13c8fbf8c210e + md5: 0fdb8ebc8dbdb62530ff73181ae63554 + depends: + - python >=3.9 + - python-dateutil >=2.7 + license: Apache-2.0 + license_family: APACHE + purls: + - pkg:pypi/freezegun?source=hash-mapping + size: 23901 + timestamp: 1754754556193 - conda: https://conda.anaconda.org/conda-forge/noarch/h2-4.2.0-pyhd8ed1ab_0.conda sha256: 0aa1cdc67a9fe75ea95b5644b734a756200d6ec9d0dff66530aec3d1c1e9df75 md5: b4754fb1bdcb70c8fd54f918301582c6 @@ -1295,6 +1356,64 @@ packages: - pkg:pypi/hyperframe?source=hash-mapping size: 17397 timestamp: 1737618427549 +- conda: https://conda.anaconda.org/conda-forge/linux-64/hypothesis-6.168.1-py311hf77984d_0.conda + noarch: python + sha256: 110115201ef3e76fa211f94683faa06acb3106c6a4ce857cd688a3732de916bf + md5: 6ad04eb646cd12a40dd8befe6499640c + depends: + - python + - sortedcontainers >=2.1.0,<3.0.0 + - exceptiongroup >=1.0.0 + - libgcc >=15 + - __glibc >=2.17,<3.0.a0 + - _python_abi3_support 1.* + - cpython >=3.11 + constrains: + - __glibc >=2.17 + license: MPL-2.0 + license_family: MOZILLA + purls: + - pkg:pypi/hypothesis?source=compressed-mapping + size: 640901 + timestamp: 1790163360436 +- conda: https://conda.anaconda.org/conda-forge/osx-arm64/hypothesis-6.168.1-py311hfcb3ee1_0.conda + noarch: python + sha256: 09e23172a972f610b5a81c2db60ad3f21e7418af6abd46ba9427986b5c4ba9fc + md5: 5a6ce3afa4fc334f6df6b3340bc67512 + depends: + - python + - sortedcontainers >=2.1.0,<3.0.0 + - exceptiongroup >=1.0.0 + - __osx >=11.0 + - _python_abi3_support 1.* + - cpython >=3.11 + constrains: + - __osx >=11.0 + license: MPL-2.0 + license_family: MOZILLA + purls: + - pkg:pypi/hypothesis?source=compressed-mapping + size: 627644 + timestamp: 1790163362928 +- conda: https://conda.anaconda.org/conda-forge/win-64/hypothesis-6.168.1-py311h66afae6_0.conda + noarch: python + sha256: 21b7d868f14909fc048757ca39eeb790709a3b52d47d9895c275cc3758e08425 + md5: fdf8ce7f28ab206b19c63c3bc15c6adc + depends: + - python + - sortedcontainers >=2.1.0,<3.0.0 + - exceptiongroup >=1.0.0 + - vc >=14.3,<15 + - vc14_runtime >=14.44.35208 + - ucrt >=10.0.20348.0 + - _python_abi3_support 1.* + - cpython >=3.11 + license: MPL-2.0 + license_family: MOZILLA + purls: + - pkg:pypi/hypothesis?source=compressed-mapping + size: 553855 + timestamp: 1790163352804 - conda: https://conda.anaconda.org/conda-forge/noarch/icalendar-5.0.12-pyhd8ed1ab_0.conda sha256: c0a5017090710e25c3735b8d0661c27d8c1827b2ec909a80cfddb7940b01a733 md5: 27cd8ec0c95ba9405f912037b3b8b99c @@ -3674,6 +3793,16 @@ packages: - pkg:pypi/python-dateutil?source=hash-mapping size: 222505 timestamp: 1733215763718 +- conda: https://conda.anaconda.org/conda-forge/noarch/python-gil-3.11.16-hd8ed1ab_2.conda + sha256: caec48ea90cf78622e923366d405ec29a419d916cb1690c2319ef76f984b9233 + md5: fab671748c957da704c76efb0686dcdb + depends: + - cpython 3.11.16.* + - python_abi * *_cp311 + license: Python-2.0 + purls: [] + size: 48553 + timestamp: 1788392053631 - conda: https://conda.anaconda.org/conda-forge/noarch/python-levenshtein-0.25.1-pyhd8ed1ab_0.conda sha256: cebd04148756aed1ed1fabb6de217c6f017a34a47b4bcbd4a9e614b987b549d9 md5: ed8e671cb3bfbd38d4da6c45e87a2e97 @@ -3929,6 +4058,21 @@ packages: - pkg:pypi/requests?source=hash-mapping size: 56690 timestamp: 1684774408600 +- conda: https://conda.anaconda.org/conda-forge/noarch/responses-0.26.3-pyhcf101f3_0.conda + sha256: 0a48b4b7d7760d63a3f20d7526a8788c992a0b09eb5625f4e8dbadd3a3c3af77 + md5: 7a3630f737fa87aec89b8a56db496200 + depends: + - pyyaml + - python >=3.10 + - requests >=2.30.0,<3.0 + - urllib3 >=1.25.10,<3.0 + - python + license: Apache-2.0 + license_family: APACHE + purls: + - pkg:pypi/responses?source=hash-mapping + size: 42882 + timestamp: 1787783809696 - conda: https://conda.anaconda.org/conda-forge/noarch/setuptools-69.5.1-pyhd8ed1ab_0.conda sha256: 72d143408507043628b32bed089730b6d5f5445eccc44b59911ec9f262e365e7 md5: 7462280d81f639363e6e63c81276bd9e @@ -4007,6 +4151,17 @@ packages: - pkg:pypi/six?source=compressed-mapping size: 18455 timestamp: 1753199211006 +- conda: https://conda.anaconda.org/conda-forge/noarch/sortedcontainers-2.4.0-pyhd8ed1ab_1.conda + sha256: d1e3e06b5cf26093047e63c8cc77b70d970411c5cbc0cb1fad461a8a8df599f7 + md5: 0401a17ae845fa72c7210e206ec5647d + depends: + - python >=3.9 + license: Apache-2.0 + license_family: APACHE + purls: + - pkg:pypi/sortedcontainers?source=hash-mapping + size: 28657 + timestamp: 1738440459037 - conda: https://conda.anaconda.org/conda-forge/win-64/tbb-2021.12.0-h91493d7_0.conda sha256: 621926aae93513408bdca3dd21c97e2aa8ba7dcd2c400dab804fb0ce7da1387b md5: 21745fdd12f01b41178596143cbecffd diff --git a/pixi.toml b/pixi.toml index b9f9c121e4..eb8a82a5bc 100644 --- a/pixi.toml +++ b/pixi.toml @@ -47,6 +47,9 @@ nodejs = "*" pytest = ">=7.4.0,<8.0" pytest-cov = ">=4.1.0,<5.0" pytest-mock = ">=3.11.0,<4.0" +freezegun = ">=1.5.5,<2" +hypothesis = ">=6.168.1,<7" +responses = ">=0.26.3,<0.27" [feature.all-contrib.tasks] diff --git a/tests/test_accent_folding.py b/tests/test_accent_folding.py new file mode 100644 index 0000000000..cbef779d55 --- /dev/null +++ b/tests/test_accent_folding.py @@ -0,0 +1,187 @@ +"""Tests for accent-insensitive conference name matching. + +Names from different sources disagree on diacritics ("PyCon Panamá" vs +"PyCon Panama", "PyDay México" vs "PyDay Mexico"). Matching should treat +these as the same conference while stored names keep their original spelling. +""" + +import sys +from pathlib import Path +from unittest.mock import patch + +import pandas as pd +import pytest + +sys.path.append(str(Path(__file__).parent.parent / "utils")) + +from tidy_conf.interactive_merge import conference_scorer +from tidy_conf.interactive_merge import fuzzy_match +from tidy_conf.interactive_merge import is_identical_name +from tidy_conf.interactive_merge import merge_conferences +from tidy_conf.titles import tidy_df_names +from tidy_conf.titles import tidy_titles +from tidy_conf.utils import fold_name +from tidy_conf.utils import strip_accents +from tidy_conf.yaml import load_title_mappings + +ACCENTED_PAIRS = [ + ("PyCon Panamá", "PyCon Panama"), + ("PyDay México", "PyDay Mexico"), + ("PyCon España", "PyCon Espana"), + ("PyCon Medellín", "PyCon Medellin"), + ("Pythoncamp Rügen", "Pythoncamp Rugen"), + ("PyDay Boyacá", "PyDay Boyaca"), +] + + +class TestStripAccents: + """strip_accents removes diacritics but keeps case and other characters.""" + + @pytest.mark.parametrize(("accented", "plain"), ACCENTED_PAIRS) + def test_removes_diacritics(self, accented, plain): + assert strip_accents(accented) == plain + + def test_preserves_case_and_punctuation(self): + assert strip_accents("PyDay México: CDMX") == "PyDay Mexico: CDMX" + + def test_plain_ascii_unchanged(self): + assert strip_accents("PyCon US") == "PyCon US" + + def test_decomposed_input_matches_composed(self): + """NFD input ("e" + combining acute) folds the same as NFC ("é").""" + decomposed = "PyDay México" + assert strip_accents(decomposed) == "PyDay Mexico" + + def test_non_latin_scripts_preserved(self): + """Scripts without combining marks must survive untouched.""" + assert strip_accents("PyCon 中国") == "PyCon 中国" + + +class TestFoldName: + """fold_name produces a comparison key: no accents, casefolded, single spaces.""" + + @pytest.mark.parametrize(("accented", "plain"), ACCENTED_PAIRS) + def test_accented_and_plain_fold_equal(self, accented, plain): + assert fold_name(accented) == fold_name(plain) + + def test_collapses_whitespace_and_case(self): + assert fold_name(" PyDay MÉXICO ") == "pyday mexico" + + +class TestMergeMatching: + """Merge helpers treat accent-only differences as identical names.""" + + @pytest.mark.parametrize(("accented", "plain"), ACCENTED_PAIRS) + def test_is_identical_name_ignores_accents(self, accented, plain): + assert is_identical_name(accented, plain) + + @pytest.mark.parametrize(("accented", "plain"), ACCENTED_PAIRS) + def test_scorer_gives_full_score(self, accented, plain): + assert conference_scorer(accented, plain) == 100 + + def test_different_conferences_still_distinct(self): + """Folding must not make genuinely different names identical.""" + assert not is_identical_name("PyCon Panamá", "PyCon Paraguay") + assert not is_identical_name("PyCon Africa", "PyCon South Africa") + + +class TestTitleMappings: + """titles.yml variations match regardless of accents.""" + + def test_reverse_mapping_contains_unaccented_variation(self): + """The real titles.yml lists "PyDay México"; the plain form must map too.""" + _, reverse = load_title_mappings(reverse=True) + assert reverse.get("PyDay México") == "PyDay Mexico" + assert reverse.get("PyDay Mexico: CDMX") == "PyDay Mexico" + + def test_tidy_df_names_maps_unaccented_variant(self): + mapping = {"PyDay México": "PyDay Mexico", "PyDay Mexico": "PyDay Mexico"} + with patch("tidy_conf.titles.load_title_mappings", return_value=([], mapping)): + result = tidy_df_names(pd.DataFrame({"conference": ["PyDay México"]})) + assert result["conference"].iloc[0] == "PyDay Mexico" + + def test_tidy_titles_matches_accent_variant(self): + """A variation listed without accents matches an accented input, and vice versa.""" + alt_names = { + "PyCon Panama": {"global": None, "variations": ["PyCon Panama City"], "regexes": []}, + "PyDay Mexico": {"global": None, "variations": ["PyDay México"], "regexes": []}, + } + data = [{"conference": "PyCon Panamá City"}, {"conference": "PyDay Mexico"}] + with patch("tidy_conf.titles.load_title_mappings", return_value=([], alt_names)): + result = tidy_titles(data) + assert result[0]["conference"] == "PyCon Panama" + assert result[0]["alt_name"] == "PyCon Panamá City" + assert result[1]["conference"] == "PyDay Mexico" + assert "alt_name" not in result[1] + + def test_tidy_titles_matches_variation_without_conference(self): + """The "Conference"-stripped comparison works (it was dead code before).""" + alt_names = {"PyCon Foo": {"global": None, "variations": ["Foo Conference"], "regexes": []}} + with patch("tidy_conf.titles.load_title_mappings", return_value=([], alt_names)): + result = tidy_titles([{"conference": "Foo"}]) + assert result[0]["conference"] == "PyCon Foo" + assert result[0]["alt_name"] == "Foo" + + def test_tidy_df_names_falls_back_to_accent_free_lookup(self): + """A mapping keyed by the accent-free spelling must still catch the accented input.""" + mapping = {"PyDay Mexico": "PyDay Mexico"} + with patch("tidy_conf.titles.load_title_mappings", return_value=([], mapping)): + result = tidy_df_names(pd.DataFrame({"conference": ["PyDay México"]})) + assert result["conference"].iloc[0] == "PyDay Mexico" + + +class TestMergeKeepsYamlSpelling: + """An accent-only match merges without prompting but must keep the YAML spelling. + + The importers drop the YAML "conference" column and take the name from the + merged index, so keying the remote row under its own spelling would rename + the conference and trip the bot's data-loss guard. + """ + + def _frames(self): + base = { + "year": [2026], + "cfp": ["2026-09-18 23:59:00"], + "link": ["https://pycon.pa/2026/"], + "start": ["2026-10-22"], + "end": ["2026-10-23"], + } + df_yml = pd.DataFrame({"conference": ["PyCon Panamá"], "place": ["Panama City, Panamá"], **base}) + df_remote = pd.DataFrame({"conference": ["PyCon Panama"], "place": ["Panama City, Panama"], **base}) + return df_yml, df_remote + + def test_fuzzy_match_keys_remote_row_by_yaml_name(self): + df_yml, df_remote = self._frames() + with ( + patch("tidy_conf.interactive_merge.load_title_mappings", return_value=([], {})), + patch("tidy_conf.titles.load_title_mappings", return_value=([], {})), + patch("tidy_conf.interactive_merge.update_title_mappings") as mock_update, + patch("tidy_conf.interactive_merge.query_yes_no", side_effect=AssertionError("must not prompt")), + ): + matched, remote, report = fuzzy_match(df_yml, df_remote) + + assert matched.index.tolist() == ["PyCon Panamá"] + assert remote.index.tolist() == ["PyCon Panamá"] + assert report.records[0].action == "merged" + # The remote spelling is recorded as a variation of the YAML name + mock_update.assert_any_call({"PyCon Panamá": ["PyCon Panama"]}) + + def test_merged_conference_keeps_yaml_name(self): + df_yml, df_remote = self._frames() + with ( + patch("tidy_conf.interactive_merge.load_title_mappings", return_value=([], {})), + patch("tidy_conf.titles.load_title_mappings", return_value=([], {})), + patch("tidy_conf.interactive_merge.update_title_mappings"), + ): + matched, remote, _report = fuzzy_match(df_yml, df_remote) + + # Both importers drop the conference column and rely on the index + matched = matched.drop(columns=["conference"]) + schema = pd.DataFrame(columns=["conference", "year", "cfp", "link", "place", "start", "end", "sub"]) + with ( + patch("tidy_conf.interactive_merge.get_schema", return_value=schema), + patch("tidy_conf.interactive_merge.query_yes_no", return_value=False), + ): + result = merge_conferences(matched, remote) + + assert result["conference"].tolist() == ["PyCon Panamá"] diff --git a/tests/test_link_checking.py b/tests/test_link_checking.py index 99a9faf990..4e3e91df8f 100644 --- a/tests/test_link_checking.py +++ b/tests/test_link_checking.py @@ -86,9 +86,12 @@ def test_404_triggers_archive_lookup(self): test_start = date(2025, 6, 1) - with patch("tidy_conf.links.get_cache") as mock_cache, patch( - "tidy_conf.links.get_cache_location", - ) as mock_cache_location: + with ( + patch("tidy_conf.links.get_cache") as mock_cache, + patch( + "tidy_conf.links.get_cache_location", + ) as mock_cache_location, + ): mock_cache.return_value = (set(), set()) mock_cache_file = Mock() mock_file_handle = Mock() @@ -292,13 +295,18 @@ def test_link_check_404_error(self, mock_get): test_url = "https://example.com/not-found" test_start = date(2025, 6, 1) - with patch("tidy_conf.links.tqdm.write"), patch("tidy_conf.links.attempt_archive_url"), patch( - "tidy_conf.links.get_cache", - ) as mock_get_cache, patch("tidy_conf.links.get_cache_location") as mock_cache_location, patch( - "builtins.open", - create=True, + with ( + patch("tidy_conf.links.tqdm.write"), + patch("tidy_conf.links.attempt_archive_url"), + patch( + "tidy_conf.links.get_cache", + ) as mock_get_cache, + patch("tidy_conf.links.get_cache_location") as mock_cache_location, + patch( + "builtins.open", + create=True, + ), ): - # Mock cache returns empty sets mock_get_cache.return_value = (set(), set()) # Mock cache file paths with proper context manager support @@ -382,6 +390,17 @@ def test_get_cache_location(self): assert cache_file.name == "no_archive.txt" assert cache_file_archived.name == "archived_links.txt" + def test_get_cache_creates_missing_directory(self, tmp_path, monkeypatch): + """The gitignored .tmp directory is absent in a fresh checkout; get_cache must not crash.""" + monkeypatch.chdir(tmp_path) + + cache, cache_archived = links.get_cache() + + assert cache == set() + assert cache_archived == set() + assert (tmp_path / "utils" / "tidy_conf" / "data" / ".tmp" / "no_archive.txt").is_file() + assert (tmp_path / "utils" / "tidy_conf" / "data" / ".tmp" / "archived_links.txt").is_file() + @patch("tidy_conf.links.Path.read_text") @patch("tidy_conf.links.Path.touch") def test_get_cache(self, mock_touch, mock_read_text): @@ -423,10 +442,13 @@ def test_successful_archive_attempt(self, mock_get): test_url = "https://example.com" - with patch("tidy_conf.links.get_cache_location") as mock_cache_location, patch( - "tidy_conf.links.get_cache", - ) as mock_get_cache, patch("builtins.open", create=True) as mock_open: - + with ( + patch("tidy_conf.links.get_cache_location") as mock_cache_location, + patch( + "tidy_conf.links.get_cache", + ) as mock_get_cache, + patch("builtins.open", create=True) as mock_open, + ): # Mock cache file paths with proper context manager support mock_cache_file = Mock() mock_cache_archived = Mock() @@ -492,13 +514,17 @@ def test_old_conference_archive_check(self, mock_get): old_start = date(2019, 6, 1) test_url = "https://example.com" - with patch("tidy_conf.links.attempt_archive_url") as mock_archive, patch( - "tidy_conf.links.get_cache", - ) as mock_get_cache, patch("tidy_conf.links.get_cache_location") as mock_cache_location, patch( - "builtins.open", - create=True, + with ( + patch("tidy_conf.links.attempt_archive_url") as mock_archive, + patch( + "tidy_conf.links.get_cache", + ) as mock_get_cache, + patch("tidy_conf.links.get_cache_location") as mock_cache_location, + patch( + "builtins.open", + create=True, + ), ): - # Mock cache returns to avoid cache hits mock_get_cache.return_value = (set(), set()) # Mock cache file paths with proper context manager support diff --git a/tests/test_mastodon_migration.py b/tests/test_mastodon_migration.py index 9546b8f109..f0111717fc 100644 --- a/tests/test_mastodon_migration.py +++ b/tests/test_mastodon_migration.py @@ -1,6 +1,7 @@ """Tests for Mastodon account migration detection functionality.""" import sys +from datetime import date from pathlib import Path from unittest.mock import patch @@ -208,8 +209,10 @@ def test_circular_migration_detection(self): with patch("tidy_conf.links.time.sleep"), patch("tidy_conf.links.tqdm.write"): result = links.check_mastodon_migration(url_a) - # Should return B (last valid before cycle detected) - assert result == url_b + # A cycle has no canonical destination: leave the stored URL untouched. + # Returning B would make the next links run (starting from B) return A, + # rewriting the field on every run. + assert result is None @responses.activate def test_max_depth_limit(self): @@ -403,7 +406,8 @@ def test_check_links_updates_mastodon(self): "year": 2025, "link": "https://example.com", "mastodon": old_url, - "start": "2025-06-01", + # sort_data runs tidy_dates before check_links, so start is a date + "start": date(2025, 6, 1), }, ] @@ -445,7 +449,8 @@ def test_check_links_preserves_non_migrated(self): "year": 2025, "link": "https://example.com", "mastodon": url, - "start": "2025-06-01", + # sort_data runs tidy_dates before check_links, so start is a date + "start": date(2025, 6, 1), }, ] diff --git a/tests/test_merge_no_data_loss.py b/tests/test_merge_no_data_loss.py index ebe9a0f75a..8be6fed310 100644 --- a/tests/test_merge_no_data_loss.py +++ b/tests/test_merge_no_data_loss.py @@ -13,6 +13,7 @@ import pandas as pd import pytest +import yaml from hypothesis import given from hypothesis import settings from hypothesis import strategies as st @@ -228,28 +229,24 @@ def test_update_title_mappings_handles_missing_variations(self): """, ) - # Patch the path resolution to use our temp file - with patch("tidy_conf.yaml.Path") as mock_path: - # Make the module-relative path point to our temp file - mock_path.return_value = titles_path - mock_path.__truediv__ = lambda self, other: titles_path + import tidy_conf.yaml - # This should not raise KeyError - try: - # Import fresh to use patched path - import importlib + real_titles = Path(tidy_conf.yaml.__file__).parent / "data" / "titles.yml" + real_before = real_titles.read_bytes() - import tidy_conf.yaml + # This should not raise KeyError + try: + tidy_conf.yaml.update_title_mappings( + {"ExistingConf": ["New Variation"]}, + path=str(titles_path), + ) + except KeyError as e: + pytest.fail(f"KeyError raised for missing 'variations' key: {e}") - importlib.reload(tidy_conf.yaml) - - # Try to update with new mapping - tidy_conf.yaml.update_title_mappings( - {"ExistingConf": ["New Variation"]}, - path=str(titles_path), - ) - except KeyError as e: - pytest.fail(f"KeyError raised for missing 'variations' key: {e}") + written = yaml.safe_load(titles_path.read_text(encoding="utf-8")) + assert written["alt_name"]["ExistingConf"]["variations"] == ["New Variation"] + # The real mapping file must not be touched by tests + assert real_titles.read_bytes() == real_before def test_load_title_mappings_handles_missing_variations(self): """load_title_mappings should handle entries without 'variations' key.""" diff --git a/tests/test_nan_deadlines.py b/tests/test_nan_deadlines.py new file mode 100644 index 0000000000..e88d04fb1c --- /dev/null +++ b/tests/test_nan_deadlines.py @@ -0,0 +1,80 @@ +"""Tests guarding against pandas NaN leaking into deadline fields as "nan". + +A missing CFP in a DataFrame becomes the string "nan" after astype(str) and was +written to conferences.yml as `cfp: nan`. The writer and the schema both map it +to TBA (cfp) or drop it (optional deadlines) instead. +""" + +import sys +from pathlib import Path + +import pandas as pd +import pytest +import yaml +from pydantic import ValidationError + +sys.path.append(str(Path(__file__).parent.parent / "utils")) + +from tidy_conf.date import clean_dates +from tidy_conf.schema import Conference +from tidy_conf.yaml import write_df_yaml + + +class TestSchemaNanDeadlines: + """The schema replaces NaN deadlines instead of rejecting the conference.""" + + @pytest.mark.parametrize("missing", [float("nan"), "nan", "NaN", " nan ", ""]) + def test_cfp_nan_becomes_tba(self, sample_conference, missing): + conf = Conference(**{**sample_conference, "cfp": missing}) + assert conf.cfp == "TBA" + + @pytest.mark.parametrize("field", ["cfp_ext", "workshop_deadline", "tutorial_deadline"]) + @pytest.mark.parametrize("missing", [float("nan"), "nan"]) + def test_optional_deadline_nan_becomes_none(self, sample_conference, field, missing): + conf = Conference(**{**sample_conference, field: missing}) + assert getattr(conf, field) is None + assert field not in conf.model_dump(exclude_none=True) + + @pytest.mark.parametrize("value", ["TBA", "tbd", "None", "Cancelled", "n/a", "2025-02-15", "2025-02-15 23:59:00"]) + def test_legitimate_cfp_values_unchanged(self, sample_conference, value): + assert Conference(**{**sample_conference, "cfp": value}).cfp == value + + +class TestWriteDfYamlNan: + """write_df_yaml never writes `cfp: nan`.""" + + def test_missing_cfp_written_as_tba(self, tmp_path, sample_conference): + rows = [ + {**sample_conference, "cfp": float("nan")}, + {**sample_conference, "conference": "PyCon Other", "cfp": None}, + {**sample_conference, "conference": "PyCon Dated"}, + ] + out = tmp_path / "conferences.yml" + + write_df_yaml(pd.DataFrame(rows), out) + + written = yaml.safe_load(out.read_text(encoding="utf-8")) + assert [c["cfp"] for c in written] == ["TBA", "TBA", "2025-02-15 23:59:00"] + assert "nan" not in out.read_text(encoding="utf-8") + + +class TestDeadlineSanity: + """Deadlines must be real calendar dates, and blank optional deadlines must not crash the sort.""" + + @pytest.mark.parametrize("value", ["2026-02-30", "2026-13-01 23:59:00", "2026-02-15 25:00:00"]) + def test_impossible_deadline_rejected(self, sample_conference, value): + with pytest.raises(ValidationError): + Conference(**{**sample_conference, "cfp": value}) + + @pytest.mark.parametrize("value", ["2026-02-28", "2024-02-29 23:59:00"]) + def test_real_deadline_accepted(self, sample_conference, value): + assert Conference(**{**sample_conference, "cfp": value}).cfp == value + + def test_clean_dates_skips_blank_optional_deadline(self): + """`cfp_ext:` with no value loads as None and previously raised AttributeError.""" + data = {"start": "2026-06-01", "end": "2026-06-03", "cfp": "2026-02-15", "cfp_ext": None} + + cleaned = clean_dates(data) + + assert cleaned["cfp"] == "2026-02-15 23:59:00" + assert cleaned["cfp_ext"] is None diff --git a/tests/test_sort_yaml_enhanced.py b/tests/test_sort_yaml_enhanced.py index 3a579fe882..f7bb5bcae4 100644 --- a/tests/test_sort_yaml_enhanced.py +++ b/tests/test_sort_yaml_enhanced.py @@ -8,6 +8,7 @@ import pytest import pytz +from freezegun import freeze_time sys.path.append(str(Path(__file__).parent.parent / "utils")) @@ -34,7 +35,8 @@ def test_sort_by_cfp_tba_words(self): sub="PY", ) result = sort_yaml.sort_by_cfp(conf) - assert result == word + # The schema normalises the pandas artefact "nan" to "TBA" + assert result == ("TBA" if word == "nan" else word) def test_sort_by_cfp_without_time(self): """Test CFP sorting when no time is specified.""" @@ -172,6 +174,8 @@ def test_sort_by_date_different_formats(self): class TestSortByDatePassed: """Test date passed sorting functionality.""" + # Freeze "today" so the 2026 CFP stays in the future regardless of when the suite runs + @freeze_time("2026-01-15") def test_sort_by_date_passed_future(self): """Test date passed sorting for future conferences.""" conf = Conference( @@ -390,9 +394,12 @@ def mock_clean_side_effect(x): {"conference": "Error Conference", "cfp": "invalid-date"}, ] - with patch("tqdm.tqdm", side_effect=lambda x, total=None: x), pytest.raises( - ValueError, - match="Date parsing error", + with ( + patch("tqdm.tqdm", side_effect=lambda x, total=None: x), + pytest.raises( + ValueError, + match="Date parsing error", + ), ): # Error should propagate from clean_dates sort_yaml.tidy_dates(data) @@ -401,6 +408,8 @@ def mock_clean_side_effect(x): class TestSplitData: """Test data splitting functionality.""" + # Freeze "today" so the 2026 conference is still upcoming regardless of when the suite runs + @freeze_time("2026-01-15") def test_split_data_basic_categories(self): """Test basic data splitting into categories.""" # Use fixed dates to avoid year boundary issues @@ -463,7 +472,11 @@ def test_split_data_basic_categories(self): assert "Active Conference" in conf_names assert "TBA Conference" in tba_names + assert [c.conference for c in expired] == ["Expired Conference"] + assert [c.conference for c in legacy] == ["Legacy Conference"] + # Freeze "today" so the 2026 conference is still upcoming regardless of when the suite runs + @freeze_time("2026-01-15") def test_split_data_cfp_ext_handling(self): """Test handling of extended CFP deadlines.""" # Use fixed dates in same year to avoid validation issues @@ -482,13 +495,11 @@ def test_split_data_cfp_ext_handling(self): with patch("tqdm.tqdm", side_effect=lambda x: x): result_conf, _, _, _ = sort_yaml.split_data([conf]) - # Should have added time to cfp + # Should have added the default time to both deadlines assert len(result_conf) == 1 processed = result_conf[0] - assert "23:59:00" in processed.cfp - # cfp_ext time handling depends on Conference object attribute check - # Just verify the conference was processed correctly - assert processed.cfp_ext is not None + assert processed.cfp == "2026-02-15 23:59:00" + assert processed.cfp_ext == "2026-03-01 23:59:00" def test_split_data_boundary_dates(self): """Test splitting with boundary date conditions.""" @@ -596,9 +607,30 @@ def test_sort_data_basic_flow(self): def test_sort_data_no_files_exist(self): """Test sort_data when no data files exist.""" - @pytest.mark.skip(reason="Test requires complex Path mock with context manager - covered by real integration tests") - def test_sort_data_validation_errors(self): - """Test sort_data with validation errors.""" + def test_sort_data_refuses_to_write_on_validation_errors(self, tmp_path): + """An entry that fails the schema must abort the run, not be silently dropped. + + The automated sort runs commit whatever is written, so dropping the entry + would delete it from the data files unnoticed. + """ + data_dir = tmp_path / "_data" + data_dir.mkdir() + conferences = data_dir / "conferences.yml" + entry = ( + "- conference: {name}\n year: {year}\n link: https://example.com/\n cfp: '2026-02-15 23:59:00'\n" + " place: Online\n start: 2026-06-01\n end: 2026-06-03\n sub: PY\n" + ) + conferences.write_text( + entry.format(name="Valid Conference", year=2026) + entry.format(name="Invalid Conference", year=1988), + encoding="utf-8", + ) + before = conferences.read_bytes() + + with pytest.raises(ValueError, match="1 conferences failed validation"): + sort_yaml.sort_data(base=str(tmp_path), skip_links=True) + + assert conferences.read_bytes() == before + assert not (data_dir / "archive.yml").exists() class TestCommandLineInterface: @@ -667,9 +699,12 @@ def test_tidy_dates_empty_data(self): def test_check_links_empty_data(self): """Test check links with empty data.""" - with patch("sort_yaml.get_cache", return_value=(set(), set())), patch( - "tqdm.tqdm", - side_effect=lambda x, total=None: x, + with ( + patch("sort_yaml.get_cache", return_value=(set(), set())), + patch( + "tqdm.tqdm", + side_effect=lambda x, total=None: x, + ), ): result = sort_yaml.check_links([]) diff --git a/tests/test_sync_integration.py b/tests/test_sync_integration.py index 8f703f2cad..341703978a 100644 --- a/tests/test_sync_integration.py +++ b/tests/test_sync_integration.py @@ -132,7 +132,7 @@ def test_full_pipeline_produces_valid_output(self, mock_title_mappings, minimal_ # Step 1: Fuzzy match with patch("builtins.input", return_value="y"): # Accept matches - matched, remote = fuzzy_match(df_yml, df_csv) + matched, remote, _report = fuzzy_match(df_yml, df_csv) # Verify fuzzy match output assert not matched.empty, "Fuzzy match should produce output" @@ -184,7 +184,7 @@ def test_pipeline_with_conflicts_logs_resolution(self, mock_title_mappings, capl ) with patch("builtins.input", return_value="y"): - matched, remote = fuzzy_match(df_yml, df_csv) + matched, remote, _report = fuzzy_match(df_yml, df_csv) with patch("tidy_conf.interactive_merge.get_schema") as mock_schema: mock_schema.return_value = pd.DataFrame( @@ -259,7 +259,7 @@ def test_no_data_loss_through_pipeline(self, mock_title_mappings): # Run through pipeline with patch("builtins.input", return_value="n"): - result, _ = fuzzy_match(df_yml, df_csv) + result, _, _report = fuzzy_match(df_yml, df_csv) # All conferences should be present result_names = result["conference"].tolist() @@ -290,7 +290,7 @@ def test_field_preservation_through_pipeline(self, mock_title_mappings): df_csv = pd.DataFrame(columns=["conference", "year", "cfp", "link", "place", "start", "end"]) with patch("builtins.input", return_value="n"): - result, _ = fuzzy_match(df_yml, df_csv) + result, _, _report = fuzzy_match(df_yml, df_csv) # Optional fields should be preserved if "mastodon" in result.columns: @@ -319,7 +319,7 @@ def test_pipeline_handles_unicode(self, mock_title_mappings): df_csv = pd.DataFrame(columns=["conference", "year", "cfp", "link", "place", "start", "end"]) with patch("builtins.input", return_value="n"): - result, _ = fuzzy_match(df_yml, df_csv) + result, _, _report = fuzzy_match(df_yml, df_csv) # Unicode names should be preserved result_names = " ".join(result["conference"].tolist()) @@ -349,7 +349,7 @@ def test_pipeline_handles_very_long_names(self, mock_title_mappings): df_csv = pd.DataFrame(columns=["conference", "year", "cfp", "link", "place", "start", "end"]) with patch("builtins.input", return_value="n"): - result, _ = fuzzy_match(df_yml, df_csv) + result, _, _report = fuzzy_match(df_yml, df_csv) # Long name should be preserved (possibly without year) assert len(result) == 1 diff --git a/tests/test_yaml_integrity.py b/tests/test_yaml_integrity.py index 3aacc63aad..a4379d5e63 100644 --- a/tests/test_yaml_integrity.py +++ b/tests/test_yaml_integrity.py @@ -221,6 +221,19 @@ def test_date_format_consistency(self, all_conference_data): if date_errors: pytest.fail("Date format errors:\n" + "\n".join(date_errors[:10])) + def test_no_nan_deadlines(self, all_conference_data): + """No deadline may be the pandas artefact "nan" (use TBA or None instead).""" + nan_errors = [ + f"{file_name}: {conf.get('conference')} {conf.get('year')} has {field}: {conf[field]}" + for file_name, file_data in all_conference_data.items() + for conf in file_data or [] + for field in ["cfp", "cfp_ext", "workshop_deadline", "tutorial_deadline"] + if field in conf and str(conf[field]).strip().lower() in {"nan", ""} + ] + + if nan_errors: + pytest.fail("NaN deadlines found:\n" + "\n".join(nan_errors)) + def test_geographic_data_consistency(self, all_conference_data): """Test geographic data consistency.""" all_conferences = [] diff --git a/utils/main.py b/utils/main.py index aafa03374d..9594f515f7 100644 --- a/utils/main.py +++ b/utils/main.py @@ -50,6 +50,10 @@ def main() -> None: total_time = time.time() - start_time logger.info(f"🎉 Data processing pipeline completed successfully in {total_time:.2f}s") + except KeyboardInterrupt: + # Not an Exception subclass, so it needs its own handler to be logged + logger.error("❌ Pipeline interrupted by user") + sys.exit(1) except Exception as e: logger.error(f"❌ Pipeline failed with error: {e}", exc_info=True) sys.exit(1) diff --git a/utils/sort_yaml.py b/utils/sort_yaml.py index c627ee9513..a6254b7bc7 100644 --- a/utils/sort_yaml.py +++ b/utils/sort_yaml.py @@ -160,7 +160,7 @@ def merge_duplicates(data): filtered = [] filtered_reduced = [] for q in tqdm(data): - q_reduced = f'{q.get("conference", None)} {q.get("year", None)} {q.get("place", None)}' + q_reduced = f"{q.get('conference', None)} {q.get('year', None)} {q.get('place', None)}" if q_reduced not in filtered_reduced: filtered.append(q) filtered_reduced.append(q_reduced) @@ -203,7 +203,8 @@ def split_data(data: list[Conference]) -> tuple[list, list, list, list]: for q in tqdm(data): if q.cfp.lower() not in TBA_WORDS and " " not in q.cfp: q.cfp += DEFAULT_CFP_TIME - if "cfp_ext" in q and " " not in q.cfp_ext: + # `"cfp_ext" in q` is always False on a pydantic model, so check the attribute itself + if q.cfp_ext and " " not in q.cfp_ext: q.cfp_ext += DEFAULT_CFP_TIME date_today = datetime.datetime.now(tz=timezone.utc).replace(microsecond=0).date() # if the conference is older than CFP_WARNING_DAYS, it moves off the main page @@ -332,7 +333,11 @@ def validate_conference(q: dict) -> Conference | None: validation_errors = len(validated) - len(new_data) if validation_errors > 0: - logger.warning(f"⚠️ {validation_errors} conferences failed validation and were skipped") + # Writing the remaining entries would silently delete the invalid ones from + # the data files, and the automated sort runs commit the result unattended. + msg = f"{validation_errors} conferences failed validation; refusing to write data files" + logger.error(f"❌ {msg}") + raise ValueError(msg) data = new_data logger.info(f"✅ {len(data)} conferences passed validation") diff --git a/utils/tidy_conf/data/titles.yml b/utils/tidy_conf/data/titles.yml index 20a5d6ad87..2d6a391df2 100644 --- a/utils/tidy_conf/data/titles.yml +++ b/utils/tidy_conf/data/titles.yml @@ -8,13 +8,12 @@ alt_name: variations: - Euro Python - EuroPython Conference - ExistingConf: - variations: - - New Variation FOSDEM: variations: - Python Devroom @ FOSDEM - Python devroom @ FOSDEM + - 'FOSDEM: Python Dev Room' + - 'FOSDEM: Python Devroom' Kiwi PyCon: global: PyCon NZ regexes: @@ -43,6 +42,12 @@ alt_name: - Plone Conference & PyCon Finland PyCamp Czechia: global: PyCamp CZ + PyCon Armenia: + global: PyData PyCon Yerevan + variations: + - PyCon Yerevan + - PyData PyCon Armenia + - PyData Yerevan PyCon Asia Pacific: global: PyCon APAC PyCon Australia: @@ -126,12 +131,6 @@ alt_name: - Python Tanzania Conference PyCon Ukraine: global: PyCon UA - PyCon Armenia: - global: PyData PyCon Yerevan - variations: - - PyCon Yerevan - - PyData PyCon Armenia - - PyData Yerevan PyCon Web: variations: - PyCon+Web @@ -140,6 +139,17 @@ alt_name: variations: - PyConf HYD - Python Conference Hyderabad + PyDay Mexico: + variations: + - PyDay México + - 'PyDay México: CDMX' + - 'PyDay Mexico: CDMX' + - PyDay CDMX + - Python Day Mexico + PyDay Valparaiso: + variations: + - PyDay Chile Valparaiso + - PyDay Valparaíso PyLadiesCon: variations: - PyLadies Conference @@ -153,6 +163,10 @@ alt_name: Python Niger: variations: - Python Niger (Niger) + Python Nordeste: + variations: + - PyNE + - Python Nordeste (PyNE) Python Pizza: global: null regexes: diff --git a/utils/tidy_conf/date.py b/utils/tidy_conf/date.py index 38a185c325..7e60d2f711 100644 --- a/utils/tidy_conf/date.py +++ b/utils/tidy_conf/date.py @@ -20,10 +20,13 @@ def clean_dates(data): ) # Make deadlines - for datetimes in ["cfp", "workshop_deadline", "tutorial_deadline"]: + for datetimes in ["cfp", "cfp_ext", "workshop_deadline", "tutorial_deadline"]: if datetimes not in data: # Check if we have this key continue + if data[datetimes] is None: + # A blank YAML value (`cfp_ext:`) loads as None; leave it to the schema + continue if isinstance(data[datetimes], datetime.datetime): # If it's a datetime make it a string, because of timezones data[datetimes] = data[datetimes].strftime(dateformat) diff --git a/utils/tidy_conf/interactive_merge.py b/utils/tidy_conf/interactive_merge.py index c26225249a..e26f3562de 100644 --- a/utils/tidy_conf/interactive_merge.py +++ b/utils/tidy_conf/interactive_merge.py @@ -20,6 +20,7 @@ from tidy_conf.countries import normalize_place from tidy_conf.schema import get_schema from tidy_conf.titles import tidy_df_names + from tidy_conf.utils import fold_name from tidy_conf.utils import query_yes_no from tidy_conf.validation import MergeRecord from tidy_conf.validation import MergeReport @@ -33,6 +34,7 @@ from .countries import normalize_place from .schema import get_schema from .titles import tidy_df_names + from .utils import fold_name from .utils import query_yes_no from .validation import MergeRecord from .validation import MergeReport @@ -60,7 +62,7 @@ def is_identical_name(s1: str, s2: str) -> bool: A fuzzy score of 100 does NOT imply identity: token_set_ratio returns 100 whenever one name's tokens are a subset of the other's (e.g. "PyCon Africa" - vs "PyCon South Africa"). Only names that are equal after case and + vs "PyCon South Africa"). Only names that are equal after case, accent and whitespace normalization may be auto-merged without confirmation. Parameters @@ -75,7 +77,7 @@ def is_identical_name(s1: str, s2: str) -> bool: bool True if the names are identical after normalization """ - return " ".join(s1.lower().split()) == " ".join(s2.lower().split()) + return fold_name(s1) == fold_name(s2) def is_placeholder_value(value) -> bool: @@ -182,9 +184,9 @@ def conference_scorer(s1: str, s2: str) -> int: int Maximum similarity score from all strategies (0-100) """ - # Normalize case for comparison - s1_lower = s1.lower().strip() - s2_lower = s2.lower().strip() + # Normalize case, accents, and whitespace for comparison + s1_lower = fold_name(s1) + s2_lower = fold_name(s2) # Calculate different similarity scores scores = [ @@ -370,7 +372,13 @@ def best_match(title_match): logger.debug( f"Exact match: '{conference_name}' -> '{title}' (score: {prob})", ) - df.at[i, "title_match"] = title + if title != conference_name: + # Identical after accent/case folding but spelled differently. + # YAML is the source of truth, so key the remote row by the YAML + # spelling and remember the remote spelling as a variation. + df_remote = df_remote.rename(index={title: conference_name}) + new_mappings[conference_name].append(title) + df.at[i, "title_match"] = conference_name claimed_titles[title] = conference_name record.match_type = "exact" record.action = "merged" @@ -678,8 +686,11 @@ def merge_conferences( else: # Check if it's an extension of the deadline and update both if query_yes_no("Is this an extension?"): - rrx, rry = int(rx.replace("-", "").split(" ")[0]), int( - ry.replace("-", "").split(" ")[0], + rrx, rry = ( + int(rx.replace("-", "").split(" ")[0]), + int( + ry.replace("-", "").split(" ")[0], + ), ) if rrx < rry: df_new.loc[i, "cfp"] = rx + cfp_time_x diff --git a/utils/tidy_conf/links.py b/utils/tidy_conf/links.py index a3adec7241..efa88d9405 100644 --- a/utils/tidy_conf/links.py +++ b/utils/tidy_conf/links.py @@ -79,6 +79,8 @@ def get_cache_location(): # Check if the URL is cached cache_file = Path("utils", "tidy_conf", "data", ".tmp", "no_archive.txt") cache_file_archived = Path("utils", "tidy_conf", "data", ".tmp", "archived_links.txt") + # The cache directory is gitignored, so it does not exist in a fresh checkout + cache_file.parent.mkdir(parents=True, exist_ok=True) return cache_file, cache_file_archived @@ -236,6 +238,10 @@ def check_mastodon_migration(mastodon_url: str, max_depth: int = 5) -> str | Non """Check if a Mastodon account has migrated and return the new URL. Follows migration chains (A→B→C) until finding the final destination. + A chain that loops back to an account already visited (A→B→A) has no + canonical destination, so it returns None and the stored URL is left + untouched; anything else would flip-flop between the accounts on every + link check. Args: mastodon_url: Full Mastodon profile URL (e.g., https://fosstodon.org/@pycon) @@ -251,18 +257,11 @@ def check_mastodon_migration(mastodon_url: str, max_depth: int = 5) -> str | Non tqdm.write(f"Warning: Could not parse Mastodon URL: {mastodon_url}") return None - visited = set() current_url = mastodon_url current_instance, current_username = parsed + visited = {f"{current_username}@{current_instance}"} for _ in range(max_depth): - # Detect circular migrations - account_key = f"{current_username}@{current_instance}" - if account_key in visited: - tqdm.write(f"Warning: Circular migration detected for {mastodon_url}") - break - visited.add(account_key) - # Query the Mastodon API api_url = f"https://{current_instance}/api/v1/accounts/lookup?acct={current_username}" headers = {"User-Agent": "Pythondeadlin.es Link Checker/0.1 (https://pythondeadlin.es)"} @@ -291,15 +290,23 @@ def check_mastodon_migration(mastodon_url: str, max_depth: int = 5) -> str | Non if not new_url: break + # Parse the new URL for the next iteration + new_parsed = parse_mastodon_url(new_url) + + # A circular migration has no canonical destination: keep the + # stored URL rather than flip-flopping between the accounts. + if new_parsed and f"{new_parsed[1]}@{new_parsed[0]}" in visited: + tqdm.write(f"Warning: Circular migration detected for {mastodon_url}") + return None + tqdm.write(f"Mastodon migration detected: {current_url} → {new_url}") current_url = new_url - # Parse the new URL for the next iteration - new_parsed = parse_mastodon_url(new_url) if not new_parsed: # If we can't parse the new URL, return it anyway return new_url current_instance, current_username = new_parsed + visited.add(f"{current_username}@{current_instance}") # Rate limit: wait 1 second between API calls time.sleep(1) diff --git a/utils/tidy_conf/schema.py b/utils/tidy_conf/schema.py index ea8e391b78..a5b8fcd7ff 100644 --- a/utils/tidy_conf/schema.py +++ b/utils/tidy_conf/schema.py @@ -1,25 +1,34 @@ import re from datetime import date +from datetime import time from pathlib import Path from typing import Annotated import pandas as pd import yaml from pydantic import BaseModel +from pydantic import Field from pydantic import HttpUrl -from pydantic import condate -from pydantic import confloat -from pydantic import conint -from pydantic import constr +from pydantic import StringConstraints +from pydantic import ValidationInfo from pydantic import field_serializer from pydantic import field_validator from pydantic import model_validator -DatetimeString = Annotated[str, constr(pattern=r"^\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}$")] -PythonYear = Annotated[int, conint(ge=1989, le=3000)] -PythonDate = Annotated[date, condate(gt=date.fromisoformat("1989-01-01"))] -LatitudeFloat = Annotated[float, confloat(ge=-90, le=90)] -LongitudeFloat = Annotated[float, confloat(ge=-180, le=180)] +# Constraints must be Annotated metadata (StringConstraints/Field); nesting constr()/conint() +# inside Annotated is silently ignored by pydantic v2. +# Date-only deadlines are allowed; sort_yaml appends DEFAULT_CFP_TIME to them. +DatetimeString = Annotated[str, StringConstraints(pattern=r"^\d{4}-\d{2}-\d{2}( \d{2}:\d{2}:\d{2})?$")] +# The main CFP also accepts the TBA words sort_yaml.TBA_WORDS understands, in any case. +# "nan" is excluded on purpose: replace_missing_deadline turns it into "TBA" first. +CfpString = Annotated[ + str, + StringConstraints(pattern=r"^(\d{4}-\d{2}-\d{2}( \d{2}:\d{2}:\d{2})?|(?i:tba|tbd|cancelled|none|na|n/a|n\.a\.))$"), +] +PythonYear = Annotated[int, Field(ge=1989, le=3000)] +PythonDate = Annotated[date, Field(gt=date.fromisoformat("1989-01-01"))] +LatitudeFloat = Annotated[float, Field(ge=-90, le=90)] +LongitudeFloat = Annotated[float, Field(ge=-180, le=180)] class Location(BaseModel): @@ -58,7 +67,7 @@ class Conference(BaseModel): year: PythonYear link: HttpUrl cfp_link: HttpUrl | None = None - cfp: DatetimeString + cfp: CfpString cfp_ext: DatetimeString | None = None workshop_deadline: DatetimeString | None = None tutorial_deadline: DatetimeString | None = None @@ -107,6 +116,29 @@ def validate_sub(cls, v): raise ValueError("Invalid submission type") return v + @field_validator("cfp", "cfp_ext", "workshop_deadline", "tutorial_deadline", mode="before") + @classmethod + def replace_missing_deadline(cls, v: object, info: ValidationInfo) -> object: + """Turn pandas missing values (NaN or the string "nan") into TBA for cfp, None otherwise. + + Rejecting them would make sort_yaml drop the whole conference. + """ + if v is None: + return v + if (isinstance(v, float) and pd.isna(v)) or (isinstance(v, str) and v.strip().lower() in {"nan", ""}): + return "TBA" if info.field_name == "cfp" else None + return v + + @field_validator("cfp", "cfp_ext", "workshop_deadline", "tutorial_deadline") + @classmethod + def validate_deadline_is_real(cls, v: str | None) -> str | None: + """The pattern only checks the digit layout; reject impossible values such as 2026-02-30.""" + if v and v[:4].isdigit(): + date.fromisoformat(v[:10]) + if len(v) > 10: + time.fromisoformat(v[11:]) + return v + @field_validator("twitter") @classmethod def validate_twitter(cls, v): diff --git a/utils/tidy_conf/titles.py b/utils/tidy_conf/titles.py index b5137611e4..0258d573fd 100644 --- a/utils/tidy_conf/titles.py +++ b/utils/tidy_conf/titles.py @@ -2,6 +2,8 @@ # Import centralized country mappings - this is the SINGLE SOURCE OF TRUTH from tidy_conf.countries import COUNTRY_CODE_TO_NAME +from tidy_conf.utils import fold_name +from tidy_conf.utils import strip_accents from tidy_conf.yaml import load_title_mappings from tqdm import tqdm @@ -19,23 +21,27 @@ def tidy_titles(data): index = low_conf.index(spelling.lower()) q["conference"] = q["conference"][:index] + spelling + q["conference"][index + len(spelling) :] + # Accent-insensitive form for matching, so "PyDay Mexico" matches "PyDay México" + folded_conf = fold_name(q["conference"]) + for key, values in alt_names.items(): global_name = values.get("global") variations = values.get("variations", []) regexes = values.get("regexes", []) # Match global name - if global_name and global_name.lower().strip() == low_conf: + if global_name and fold_name(global_name) == folded_conf: if "alt_name" not in q: q["alt_name"] = global_name.strip() continue # Match variations for variation in variations: + folded_variation = fold_name(variation) if ( - (variation.lower().strip() == low_conf) - or (variation.lower().strip().replace(" ", "") == low_conf) - or (variation.lower().strip().replace("Conference", "") == low_conf) + (folded_variation == folded_conf) + or (folded_variation.replace(" ", "") == folded_conf) + or (" ".join(folded_variation.replace("conference", "").split()) == folded_conf) ): if "alt_name" not in q and q["conference"].strip() != key: q["alt_name"] = q["conference"].strip() @@ -130,9 +136,13 @@ def normalize_conference_name(name: str, known_mappings: dict | None = None) -> # Remove leading and trailing whitespace result = result.strip() - # Apply known mappings FIRST (mappings may contain unexpanded country codes) + # Apply known mappings FIRST (mappings may contain unexpanded country codes). + # The reverse mapping also holds accent-free variants, so fall back to the + # accent-free form: "PyDay México" finds the "PyDay Mexico" entry. if result in known_mappings: result = known_mappings[result] + elif strip_accents(result) in known_mappings: + result = known_mappings[strip_accents(result)] # Expand country codes to full names AFTER mappings # This ensures idempotency: normalize(normalize(x)) == normalize(x) diff --git a/utils/tidy_conf/utils.py b/utils/tidy_conf/utils.py index 0ee7eafa4d..bfa7dfcbb8 100644 --- a/utils/tidy_conf/utils.py +++ b/utils/tidy_conf/utils.py @@ -1,4 +1,5 @@ import sys +import unicodedata import pandas as pd import yaml @@ -39,6 +40,23 @@ def _dict_representer(dumper, data): return yaml.dump(data, stream, OrderedDumper, **kwds) +def strip_accents(text: str) -> str: + """Remove diacritics while preserving case, e.g. "PyCon Panamá" -> "PyCon Panama". + + Only use this for matching. Stored conference names keep their original spelling. + """ + decomposed = unicodedata.normalize("NFKD", text) + return unicodedata.normalize("NFC", "".join(c for c in decomposed if not unicodedata.combining(c))) + + +def fold_name(text: str) -> str: + """Fold a name for comparison: strip accents, casefold, and collapse whitespace. + + "PyDay México" and "pyday mexico" both fold to "pyday mexico". + """ + return " ".join(strip_accents(text).casefold().split()) + + def pretty_print(header, conf, tba=None, expired=None) -> None: """Print order of conferences. diff --git a/utils/tidy_conf/yaml.py b/utils/tidy_conf/yaml.py index 6dcbd39353..186b9f4c4e 100644 --- a/utils/tidy_conf/yaml.py +++ b/utils/tidy_conf/yaml.py @@ -9,10 +9,12 @@ from tidy_conf.schema import Conference from tidy_conf.schema import get_schema from tidy_conf.utils import ordered_dump + from tidy_conf.utils import strip_accents except ImportError: from .schema import Conference from .schema import get_schema from .utils import ordered_dump + from .utils import strip_accents def write_conference_yaml(data: list[dict] | pd.DataFrame, url: str) -> None: @@ -36,6 +38,7 @@ def write_conference_yaml(data: list[dict] | pd.DataFrame, url: str) -> None: with Path(url).open( "w", encoding="utf-8", + newline="\n", ) as outfile: for line in ordered_dump( data, @@ -138,6 +141,8 @@ def load_title_mappings(reverse=False, path="utils/tidy_conf/data/titles.yml"): current_variations.update( re.sub(r"\b\s*(19|20)\d{2}\s*\b", "", variation).strip() for variation in current_variations.copy() ) + # Add variations without accents, so "PyDay Mexico" matches "PyDay México" + current_variations.update(strip_accents(variation) for variation in current_variations.copy()) # Filter out empty strings variations.extend(v for v in current_variations if v) @@ -163,21 +168,24 @@ def load_title_mappings(reverse=False, path="utils/tidy_conf/data/titles.yml"): def update_title_mappings(data, path="utils/tidy_conf/data/titles.yml"): - """Update the title mappings in the YAML file.""" - original_path = Path(path) - module_dir = Path(__file__).parent - - # Determine filename based on what was requested - filename = "rejections.yml" if "rejection" in str(original_path).lower() else "titles.yml" + """Update the title mappings in the YAML file. - # Use module-relative path (most reliable) - path = module_dir / "data" / filename + Repo-relative paths under utils/tidy_conf/data/ (the defaults used by the merge + pipeline) resolve relative to this module so the working directory doesn't matter. + Any other path, e.g. a temporary file in tests, is written exactly as given. + """ + original_path = Path(path) + if original_path.as_posix().startswith("utils/tidy_conf/data/"): + path = Path(__file__).parent / "data" / original_path.name + else: + path = original_path if not path.exists() or path.stat().st_size == 0: path.parent.mkdir(parents=True, exist_ok=True) with path.open( "w", encoding="utf-8", + newline="\n", ) as file: yaml.dump({"spelling": [], "alt_name": data}, file, default_flow_style=False, allow_unicode=True) else: @@ -201,6 +209,7 @@ def update_title_mappings(data, path="utils/tidy_conf/data/titles.yml"): with path.open( "w", encoding="utf-8", + newline="\n", ) as file: yaml.dump(title_data, file, default_flow_style=False, allow_unicode=True) @@ -212,5 +221,6 @@ def write_df_yaml(df, out_url): df["end"] = pd.to_datetime(df["end"]).dt.date df["start"] = pd.to_datetime(df["start"]).dt.date df["year"] = df["year"].astype(int) - df["cfp"] = df["cfp"].astype(str) + # astype(str) would turn missing values into the literal string "nan" + df["cfp"] = df["cfp"].fillna("TBA").astype(str).replace({"nan": "TBA", "NaN": "TBA", "": "TBA"}) write_conference_yaml(df, out_url)