From a4ad82e1b7ec0e58fa5955c3ccddbc6909547558 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sun, 9 Aug 2026 16:04:12 +0000 Subject: [PATCH 01/10] build(deps-dev): bump ruff from 0.15.22 to 0.16.0 Bumps [ruff](https://github.com/astral-sh/ruff) from 0.15.22 to 0.16.0. - [Release notes](https://github.com/astral-sh/ruff/releases) - [Changelog](https://github.com/astral-sh/ruff/blob/main/CHANGELOG.md) - [Commits](https://github.com/astral-sh/ruff/compare/0.15.22...0.16.0) --- updated-dependencies: - dependency-name: ruff dependency-version: 0.16.0 dependency-type: direct:development update-type: version-update:semver-minor ... Signed-off-by: dependabot[bot] --- uv.lock | 42 +++++++++++++++++++++--------------------- 1 file changed, 21 insertions(+), 21 deletions(-) diff --git a/uv.lock b/uv.lock index 60bef7e..3131277 100644 --- a/uv.lock +++ b/uv.lock @@ -1900,27 +1900,27 @@ wheels = [ [[package]] name = "ruff" -version = "0.15.22" -source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/3a/06/ae069393fc66e8ff33036d4b368003833bf6e88ccf182e17e7a2f1c754fd/ruff-0.15.22.tar.gz", hash = "sha256:3f15175b1fb580126f58285a5dae6b2ea89000136d980c64499211f116b54809", size = 4785063, upload-time = "2026-07-16T15:14:13.244Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/23/18/ee54b7ae1e121be7a28ea6da4b67564ebb0530e183a54415ab7e3bcd2c4e/ruff-0.15.22-py3-none-linux_armv6l.whl", hash = "sha256:44423e73493737f5e7c5b41d475483898ff37afcdae38bc3da5085e29af1c2d8", size = 10781258, upload-time = "2026-07-16T15:13:19.452Z" }, - { url = "https://files.pythonhosted.org/packages/2f/d2/2520cb14761ddbeaf57642a76942fc36adcbdbe53b4532241995f6fc485c/ruff-0.15.22-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:b82c6482946e9eda7ff2e091d25b8bad3f718684e1916d41bd56873cee05b697", size = 10999477, upload-time = "2026-07-16T15:13:23.318Z" }, - { url = "https://files.pythonhosted.org/packages/c9/10/74e53572aa758dfaa678c2a2646b5c5515d884b7ca56be4d2ce03ca4b560/ruff-0.15.22-py3-none-macosx_11_0_arm64.whl", hash = "sha256:11c1c715af53a09f714e011106bffc419751ec8232fcb5da42173284ea3fec6f", size = 10466716, upload-time = "2026-07-16T15:13:26.162Z" }, - { url = "https://files.pythonhosted.org/packages/1e/cc/44eaaf0844e028182f2d0a8f2190d0f359159aed0a9e5ab861d892f1ae2a/ruff-0.15.22-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:742a29cf29bddb7c8327895d6a10e0e6c5b38a96dd407af9b5d0857f809c0576", size = 10892644, upload-time = "2026-07-16T15:13:29.229Z" }, - { url = "https://files.pythonhosted.org/packages/9f/21/8edf559014d2b0f82beea19cfb713993ad802ccda16868769979c6090a84/ruff-0.15.22-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:72af58b951b0ae395935ae79763dc349bc0eb706319d28f7a33ad2cfb3cfc178", size = 10576719, upload-time = "2026-07-16T15:13:32.35Z" }, - { url = "https://files.pythonhosted.org/packages/bf/1e/3a13abd392a3b50b62e5938a831f9ab6e588358cacad5c18545b716d2182/ruff-0.15.22-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:62d425005c1835eb24e2ee4161cb90e8db263415f4a71c8c72c33abaa6c0c224", size = 11376494, upload-time = "2026-07-16T15:13:35.958Z" }, - { url = "https://files.pythonhosted.org/packages/bf/3e/422d3d95bcf04dd78e1aeac22184d4f9a8fb2c01865d39d44618484a0317/ruff-0.15.22-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:e8b9b3f8779a4f08c969defc3c8c35abffaa757e601ed5ae66d6d1db6519969a", size = 12208370, upload-time = "2026-07-16T15:13:39.185Z" }, - { url = "https://files.pythonhosted.org/packages/1e/91/5d065a0e0a02bf4813f5119ad278462eed081d2b832eb7c021ade0ec9e65/ruff-0.15.22-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:1e0dd1b2e4d3d585f897a0d137cbf4eaf6223bef4e8ce34d6bb12556c5f9249e", size = 11581098, upload-time = "2026-07-16T15:13:42.132Z" }, - { url = "https://files.pythonhosted.org/packages/f6/f9/a0d4871d12fae702eb1f41b686caf05f1f8b124dc6db6f784f53d74918fa/ruff-0.15.22-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:365523eb91d9224e1bcb03b022fbf0facb8f9e23792a2c53d9d4b3924bdbdebb", size = 11399422, upload-time = "2026-07-16T15:13:45.2Z" }, - { url = "https://files.pythonhosted.org/packages/18/80/c843a5176cddbceb0b7e8dd41cf9993490796c1c469348d384f5a5c13c56/ruff-0.15.22-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:fabfd168afdf29fee5be98b831efa9683c94d7c5a3b58b9ce5a2e38444589a74", size = 11381683, upload-time = "2026-07-16T15:13:48.46Z" }, - { url = "https://files.pythonhosted.org/packages/d4/00/8485de0ae92239438a36cfc51350db9b9e85c9ebdfaea91b18e422706662/ruff-0.15.22-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:225dbf095a87f1d9f90f5fd7924d2613ee452a75a4308c63a8f50f761787aa7c", size = 10850295, upload-time = "2026-07-16T15:13:51.655Z" }, - { url = "https://files.pythonhosted.org/packages/fa/91/24977ec2ec72eaf15e4394ace2959fdff2dd1e14f03e005e838023407169/ruff-0.15.22-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:1877d63b9d24ed278744f1523fd11b85540566d54641f97c566d7d9dc5ca5296", size = 10579640, upload-time = "2026-07-16T15:13:54.79Z" }, - { url = "https://files.pythonhosted.org/packages/9c/47/9b51216951974df1f263ac19da550d34252e0ed7218c25f10c5ef9ed7517/ruff-0.15.22-py3-none-musllinux_1_2_i686.whl", hash = "sha256:a1606c510bd7215680d32efab38965f7cdec3ef69f5170a3f4791404ffdd5262", size = 11105077, upload-time = "2026-07-16T15:13:57.915Z" }, - { url = "https://files.pythonhosted.org/packages/c2/47/20e9d4a3b8016778acea5fc32bb50d35d207500a17ddb529ffa6996feef8/ruff-0.15.22-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:630479b18625f5ffc373f77603a22a9f8ac0acd7ff0501178b5db28ec71e9c64", size = 11490980, upload-time = "2026-07-16T15:14:01.032Z" }, - { url = "https://files.pythonhosted.org/packages/4d/76/3f72d8fc38c1cb77b38c56a70da9d0c17700cc1cc50f9649c9d3c8f5ba71/ruff-0.15.22-py3-none-win32.whl", hash = "sha256:e5ba0e4a13fd14abbed2a77b517a3911290c6c6c59ef67784328d1668fab76cf", size = 10789165, upload-time = "2026-07-16T15:14:04.16Z" }, - { url = "https://files.pythonhosted.org/packages/cb/46/4965251734c2b6fcdca1b1b187d20bcac3af0ee5b083b89c910bb961ce3a/ruff-0.15.22-py3-none-win_amd64.whl", hash = "sha256:9be63ba1eb936acd2d1342fb8337c356353706fce233b2a15a09a97037e6acde", size = 11938297, upload-time = "2026-07-16T15:14:07.316Z" }, - { url = "https://files.pythonhosted.org/packages/57/c9/e69b1ff4c8b69093ef08b8919ab767af0569666865b39c30a8795d88d3c6/ruff-0.15.22-py3-none-win_arm64.whl", hash = "sha256:e1168075b72158510839f250027659cdd78476f40507dd517892304c41318661", size = 11298172, upload-time = "2026-07-16T15:14:10.51Z" }, +version = "0.16.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/4d/94/1e5e4967626faf12fa56999cd6222dff6992ceb086ad7945756baf70c7a7/ruff-0.16.0.tar.gz", hash = "sha256:e460aafd5495ec89efaa6ced2e4a9a581116451e1c88b9d37ef497e0f8e93982", size = 4790557, upload-time = "2026-07-23T19:11:30.981Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/4b/81/1c8818fee7ce1a04cd7d1b3172e0a8f8e4f1dc4feb7fc390e16daa8af323/ruff-0.16.0-py3-none-linux_armv6l.whl", hash = "sha256:e5115729eb08c585e5121978ba5d5b60caeae394ce21b9fb5e6cd33a1c6c9b1e", size = 10754633, upload-time = "2026-07-23T19:10:46.415Z" }, + { url = "https://files.pythonhosted.org/packages/23/df/beaf59c09d68db84304d555f188b276a77132a5d5b0b67a5c762aa143628/ruff-0.16.0-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:3c954b1d580bfa035b41654f7858cc7e71d5fc3ac5b723dd62bd9133830ed522", size = 10969164, upload-time = "2026-07-23T19:10:50.271Z" }, + { url = "https://files.pythonhosted.org/packages/42/ce/741cd197496a1abbf51352710fd15ed995d2a2be87189c1da26a450d6e83/ruff-0.16.0-py3-none-macosx_11_0_arm64.whl", hash = "sha256:e01c21d10eb1b29f47b7454e1f4056db9a3f0260c646aa88457c610291db9f81", size = 10488846, upload-time = "2026-07-23T19:10:52.639Z" }, + { url = "https://files.pythonhosted.org/packages/52/2a/a2db8e88cade358f5cdcb05674a917751074109315d014eb6352d9a893f7/ruff-0.16.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:6e364e5ed22ed8dc05082fd78e35308618260907ac2d3c1d637b2e682415b6c9", size = 10889729, upload-time = "2026-07-23T19:10:54.89Z" }, + { url = "https://files.pythonhosted.org/packages/42/65/62a771694ebd63029dc953e27dbad40e1588bd4860ff9fe881018fddaa49/ruff-0.16.0-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:d327b8fc113a1d4421a04f3839d3752057c8dd1ee320223a6f3f52d04ada462a", size = 10568275, upload-time = "2026-07-23T19:10:56.993Z" }, + { url = "https://files.pythonhosted.org/packages/3f/e2/ced249fe8af5f086c5c58cc21cc3356d50f32f7401c5df87050c999620a7/ruff-0.16.0-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:a9b50c55e263103586b3dcf5f73d479eb8cb5fdb6098fec59a62891dab653717", size = 11385112, upload-time = "2026-07-23T19:10:59.615Z" }, + { url = "https://files.pythonhosted.org/packages/87/0b/05154977a8fd69eeb6c103271f55403bfd8711f5c0f8ed07489d95a504e7/ruff-0.16.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:0ff4a79ce3ec0172f3241943835de1c4cb4e2dcd07f0f8c2d02603dbbbee4b17", size = 12207008, upload-time = "2026-07-23T19:11:02.154Z" }, + { url = "https://files.pythonhosted.org/packages/fb/29/98225831a3a1eab0e02f4acc6ca6559a98611dcc68b6965ff4b7234627c1/ruff-0.16.0-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:e95c448fca1fb2a18372a9440926c5a6ee789639bb975c72e7ae6d0b04218ab4", size = 11650842, upload-time = "2026-07-23T19:11:04.557Z" }, + { url = "https://files.pythonhosted.org/packages/91/66/6bd3cf90500653d55dc0ffc8507aa8300bd49d0214b2e8cb4d3fef2943ba/ruff-0.16.0-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:4f11a8d11010301d0a398a2fdef67691feca7294da6aef55e2150e8fa2cd520b", size = 11400718, upload-time = "2026-07-23T19:11:09.233Z" }, + { url = "https://files.pythonhosted.org/packages/8e/a2/a54eb4eae05d66364050a5d3b8a9c5ef88196531b3cbe7109d873f87f819/ruff-0.16.0-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:48044c678e9cb8698246c99b14aaccfa6601dea7379eb48a6f8f73f7a6d86cd0", size = 11426177, upload-time = "2026-07-23T19:11:11.994Z" }, + { url = "https://files.pythonhosted.org/packages/1a/be/16e3eea4b2a478a496919f5e36f17c4559e54620bd3bbac5d6affa068006/ruff-0.16.0-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:7aa0959bad8eb8bef50340154fc9b58678dae31fa4293afa38b44b6e552c0213", size = 10856126, upload-time = "2026-07-23T19:11:14.221Z" }, + { url = "https://files.pythonhosted.org/packages/a2/84/252eb8b868a16eec7257c14f504f77537e734b2d69c762e639e588e304a3/ruff-0.16.0-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:28ea2b7df8ebf7f9da6b7d47b230ab48f387c0a29be3b474c4d0740e197bb9af", size = 10571208, upload-time = "2026-07-23T19:11:16.378Z" }, + { url = "https://files.pythonhosted.org/packages/21/09/817a482f542f7570cbb4554b26e896610c7114f539b1d9e2d2145bf6bef6/ruff-0.16.0-py3-none-musllinux_1_2_i686.whl", hash = "sha256:33a3dfac8c35f81498dea9181bccc2f4c4bc8f1521a1dd9406e77643e0f0fb09", size = 11063329, upload-time = "2026-07-23T19:11:19.173Z" }, + { url = "https://files.pythonhosted.org/packages/2e/23/9403c180ca1cb9b1f7335f5c3e5305c09d49ea5b345196682a36028bde4a/ruff-0.16.0-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:a5237a0bda500d30d81b8e07a6973a5cbc772864cbf746ae2f4e8a2e01c9f4ed", size = 11489751, upload-time = "2026-07-23T19:11:21.74Z" }, + { url = "https://files.pythonhosted.org/packages/b2/1d/1b2ef7bcde851c78d7f17f1cca13fd6dc695fc4b3d6197941e72cae5b132/ruff-0.16.0-py3-none-win32.whl", hash = "sha256:7fab76fa065c873f41ff744347c6e77bcc3dfec4bcc754dc26b63d23c0f7f5fb", size = 10785885, upload-time = "2026-07-23T19:11:23.947Z" }, + { url = "https://files.pythonhosted.org/packages/b2/a3/d5e4ef7a56be3f928ffb90b94c25ba7d3cb9c7fe0736aeaaedf361770712/ruff-0.16.0-py3-none-win_amd64.whl", hash = "sha256:429c117f022bf481fabd9d551e7a3952b24c65e6ef44337ea09d90bebef14472", size = 11923141, upload-time = "2026-07-23T19:11:26.409Z" }, + { url = "https://files.pythonhosted.org/packages/cb/9a/8415f2657cbe200f41a4531ccededf135505a92d4a012229121f885b26f9/ruff-0.16.0-py3-none-win_arm64.whl", hash = "sha256:14296fedcd2705c77ab8235439278bbb38f285cf7da5528b00b3e330c3d4872d", size = 11273407, upload-time = "2026-07-23T19:11:28.705Z" }, ] [[package]] From 8d1cf2776a34893d6afd9a09179fb7675d45d850 Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:33:32 +1000 Subject: [PATCH 02/10] chore(lint): apply ruff 0.16 autofixes ruff 0.16.0 expanded its default rule set from 59 rules to 413. This repo never pinned `select`, so the bump inherited the whole new set at once and `./doit.sh lint` reported 534 findings, failing CI. This commit is pure tool output, so it can be skimmed rather than read: uv run ruff check --fix --unsafe-fixes --unfixable FLY002,UP031 . uv run ruff format $(git ls-files '*.py') ./doit.sh gen-docs FLY002 and UP031 are held back because their fixes are worse than the source. FLY002 collapses the CSP list in web/app.py into one 300-character line and deletes the comments inside it. UP031 doubles every brace in the eolymp GraphQL query. The next commit suppresses both at the site. gen-docs runs here and not later: docs_test asserts docs/library.md matches the generator, and the generator reads the signatures this commit rewrote. 44 findings remain, cleared by the commits that follow. --- docs/library.md | 6 ++-- scripts/generate_library_docs.py | 21 +++++++----- src/ojhunt/__main__.py | 27 ++++++++------- src/ojhunt/cli/__init__.py | 2 +- src/ojhunt/cli/models.py | 3 +- src/ojhunt/cli/output.py | 17 +++++----- src/ojhunt/cli/parser.py | 28 ++++++++-------- src/ojhunt/cli/progress.py | 28 ++++++++-------- src/ojhunt/core/__init__.py | 2 +- src/ojhunt/core/credentials.py | 3 +- src/ojhunt/core/models.py | 21 ++++++------ src/ojhunt/core/runner.py | 2 +- src/ojhunt/core/session.py | 4 +-- src/ojhunt/core/stats.py | 8 ++--- src/ojhunt/crawlers/__init__.py | 12 +++---- src/ojhunt/crawlers/_help.py | 11 ++++--- src/ojhunt/crawlers/_utils.py | 16 ++++----- src/ojhunt/crawlers/aizu.py | 7 ++-- src/ojhunt/crawlers/atcoder.py | 9 ++--- src/ojhunt/crawlers/codechef.py | 7 ++-- src/ojhunt/crawlers/codeforces.py | 7 ++-- src/ojhunt/crawlers/codewars.py | 5 ++- src/ojhunt/crawlers/cses.py | 22 ++++++------- src/ojhunt/crawlers/csg.py | 5 ++- src/ojhunt/crawlers/csu.py | 10 +++--- src/ojhunt/crawlers/darkbzoj.py | 11 +++---- src/ojhunt/crawlers/eolymp.py | 6 ++-- src/ojhunt/crawlers/hdu.py | 6 ++-- src/ojhunt/crawlers/hust.py | 5 ++- src/ojhunt/crawlers/kilonova.py | 5 ++- src/ojhunt/crawlers/leetcode.py | 5 ++- src/ojhunt/crawlers/lightoj.py | 7 ++-- src/ojhunt/crawlers/loj.py | 12 +++---- src/ojhunt/crawlers/luogu.py | 8 ++--- src/ojhunt/crawlers/nbut.py | 7 ++-- src/ojhunt/crawlers/nit.py | 9 ++--- src/ojhunt/crawlers/nod.py | 5 ++- src/ojhunt/crawlers/nowcoder.py | 5 ++- src/ojhunt/crawlers/ojuz.py | 9 +++-- src/ojhunt/crawlers/poj.py | 18 +++++----- src/ojhunt/crawlers/sdutoj.py | 5 ++- src/ojhunt/crawlers/timus.py | 10 +++--- src/ojhunt/crawlers/tlx.py | 5 ++- src/ojhunt/crawlers/toph.py | 5 ++- src/ojhunt/crawlers/uoj.py | 13 ++++---- src/ojhunt/crawlers/uva.py | 11 +++---- src/ojhunt/crawlers/vjudge.py | 24 +++++++------- src/ojhunt/crawlers/vnoj.py | 6 ++-- src/ojhunt/crawlers/yosupo.py | 7 ++-- src/ojhunt/crawlers/yukicoder.py | 7 ++-- src/ojhunt/web/api.py | 55 ++++++++++++++----------------- src/ojhunt/web/app.py | 8 +++-- src/ojhunt/web/legacy_db.py | 13 ++++---- src/ojhunt/web/pages.py | 2 +- src/ojhunt/web/pdf.py | 19 +++++------ tests/cli/output_test.py | 4 +-- tests/cli/parser_test.py | 12 +++---- tests/core/runner_test.py | 2 +- tests/crawlers/_utils_test.py | 15 ++++----- tests/crawlers/aizu_test.py | 1 + tests/crawlers/atcoder_test.py | 1 + tests/crawlers/codechef_test.py | 1 + tests/crawlers/codeforces_test.py | 1 + tests/crawlers/codewars_test.py | 1 + tests/crawlers/cses_test.py | 4 ++- tests/crawlers/csg_test.py | 1 + tests/crawlers/csu_test.py | 1 + tests/crawlers/darkbzoj_test.py | 1 + tests/crawlers/eolymp_test.py | 1 + tests/crawlers/hdu_test.py | 1 + tests/crawlers/hust_test.py | 1 + tests/crawlers/kilonova_test.py | 1 + tests/crawlers/leetcode_test.py | 1 + tests/crawlers/lightoj_test.py | 1 + tests/crawlers/loj_test.py | 1 + tests/crawlers/luogu_test.py | 1 + tests/crawlers/nbut_test.py | 1 + tests/crawlers/nit_test.py | 3 +- tests/crawlers/nod_test.py | 3 +- tests/crawlers/nowcoder_test.py | 1 + tests/crawlers/ojuz_test.py | 1 + tests/crawlers/poj_test.py | 1 + tests/crawlers/registry_test.py | 12 ++++--- tests/crawlers/sdutoj_test.py | 1 + tests/crawlers/timus_test.py | 1 + tests/crawlers/tlx_test.py | 1 + tests/crawlers/toph_test.py | 1 + tests/crawlers/uoj_test.py | 1 + tests/crawlers/uva_test.py | 1 + tests/crawlers/vjudge_test.py | 6 ++-- tests/crawlers/vnoj_test.py | 1 + tests/crawlers/yosupo_test.py | 1 + tests/crawlers/yukicoder_test.py | 1 + tests/e2e/test_pdf_workflow.py | 3 +- tests/web/pdf_test.py | 16 ++++----- 95 files changed, 348 insertions(+), 352 deletions(-) diff --git a/docs/library.md b/docs/library.md index 8c4c1a6..4803453 100644 --- a/docs/library.md +++ b/docs/library.md @@ -149,7 +149,7 @@ should. ### `query_sync()` ```text -query_sync(crawler: Union[CrawlerInfo, Callable[..., Awaitable[Any]]], username: str, **kwargs: Any) -> CrawlerResult +query_sync(crawler: CrawlerInfo | collections.abc.Callable[..., collections.abc.Awaitable[Any]], username: str, **kwargs: Any) -> CrawlerResult Query a crawler synchronously, opening and closing a session for you. @@ -200,7 +200,7 @@ Attributes: Fields solved: int submissions: int - solved_list: Optional[List[str]] = None + solved_list: list[str] | None = None ``` ### `CrawlerInfo` @@ -226,7 +226,7 @@ Attributes: Fields name: str meta: CrawlerMeta - query: Callable[..., Awaitable[CrawlerResult]] + query: collections.abc.Callable[..., collections.abc.Awaitable[CrawlerResult]] ``` ### `CrawlerInfo.query_sync()` diff --git a/scripts/generate_library_docs.py b/scripts/generate_library_docs.py index 0424b88..ae1e77d 100644 --- a/scripts/generate_library_docs.py +++ b/scripts/generate_library_docs.py @@ -8,10 +8,11 @@ """ import inspect +from collections.abc import Callable from dataclasses import MISSING, Field, fields from enum import Enum from pathlib import Path -from typing import Any, Callable, List +from typing import Any import ojhunt.crawlers from ojhunt.core.models import ( @@ -138,7 +139,7 @@ def _crawler_row(crawler: CrawlerInfo) -> str: def render_library_docs() -> str: - crawlers: List[CrawlerInfo] = [c for _, c in sorted(crawler_registry.items())] + crawlers: list[CrawlerInfo] = [c for _, c in sorted(crawler_registry.items())] assert crawlers, "no crawlers discovered — the table and example would be empty" sections = [ @@ -151,9 +152,11 @@ def render_library_docs() -> str: "", "## API", "", - "Everything below is importable from `ojhunt.crawlers`. The registry itself " - "is the module attribute `crawlers`, a `CrawlerRegistry` built on first " - "access.", + ( + "Everything below is importable from `ojhunt.crawlers`. The registry itself " + "is the module attribute `crawlers`, a `CrawlerRegistry` built on first " + "access." + ), "", _class_entry(CrawlerRegistry), "", @@ -171,9 +174,11 @@ def render_library_docs() -> str: "", "## Supported crawlers", "", - f"{len(crawlers)} crawlers. Every one takes `(session, username)`; the " - 'arguments below are additional. Run `help(crawlers[""])` for one ' - "crawler's full entry.", + ( + f"{len(crawlers)} crawlers. Every one takes `(session, username)`; the " + 'arguments below are additional. Run `help(crawlers[""])` for one ' + "crawler's full entry." + ), "", "| Crawler | Platform | Login | Username / notes | Extra arguments |", "| --- | --- | --- | --- | --- |", diff --git a/src/ojhunt/__main__.py b/src/ojhunt/__main__.py index c88b23b..c2c8478 100755 --- a/src/ojhunt/__main__.py +++ b/src/ojhunt/__main__.py @@ -9,7 +9,6 @@ import inspect import sys from datetime import datetime -from typing import Dict, List, Optional, Tuple import aiohttp @@ -34,12 +33,12 @@ async def query_crawler( session: aiohttp.ClientSession, crawler_name: str, username: str, - crawlers: Dict[str, CrawlerInfo], - progress: Optional[ProgressManager] = None, - progress_key: Optional[str] = None, - password: Optional[str] = None, - login_user: Optional[str] = None, - login_password: Optional[str] = None, + crawlers: dict[str, CrawlerInfo], + progress: ProgressManager | None = None, + progress_key: str | None = None, + password: str | None = None, + login_user: str | None = None, + login_password: str | None = None, ) -> QueryResult: """ Query a single crawler for user statistics. @@ -72,7 +71,7 @@ async def query_crawler( if progress and progress_key: progress.start_task(progress_key) - kwargs: Dict[str, str] = {} + kwargs: dict[str, str] = {} if login_user is not None: kwargs["login_user"] = login_user if login_password is not None: @@ -105,21 +104,21 @@ async def query_crawler( async def run_queries( - queries: List[Query], - crawlers: Dict[str, CrawlerInfo], - crawler_logins: Dict[str, Tuple[str, str]], + queries: list[Query], + crawlers: dict[str, CrawlerInfo], + crawler_logins: dict[str, tuple[str, str]], no_progress: bool = False, -) -> List[QueryResult]: +) -> list[QueryResult]: """Execute all queries with live progress updates.""" progress = ProgressManager(is_tty=not no_progress and sys.stderr.isatty()) - keys: List[str] = [] + keys: list[str] = [] for q in queries: title = crawlers[q.crawler].meta.title key = progress.add_task(q.crawler, title, q.username) keys.append(key) - results: Dict[str, QueryResult] = {} + results: dict[str, QueryResult] = {} async with create_session() as session: with progress: diff --git a/src/ojhunt/cli/__init__.py b/src/ojhunt/cli/__init__.py index cac7c72..e59373c 100644 --- a/src/ojhunt/cli/__init__.py +++ b/src/ojhunt/cli/__init__.py @@ -21,8 +21,8 @@ from ojhunt.cli.progress import ProgressManager, TaskStatus __all__ = [ - "Query", "ProgressManager", + "Query", "TaskStatus", "build_all_queries", "check_duplicate_queries", diff --git a/src/ojhunt/cli/models.py b/src/ojhunt/cli/models.py index a23783d..83f2d6e 100644 --- a/src/ojhunt/cli/models.py +++ b/src/ojhunt/cli/models.py @@ -3,7 +3,6 @@ """ from dataclasses import dataclass -from typing import Optional @dataclass @@ -12,4 +11,4 @@ class Query: crawler: str username: str - password: Optional[str] = None + password: str | None = None diff --git a/src/ojhunt/cli/output.py b/src/ojhunt/cli/output.py index 86d79fc..170351f 100644 --- a/src/ojhunt/cli/output.py +++ b/src/ojhunt/cli/output.py @@ -5,7 +5,6 @@ import json import sys from collections import Counter -from typing import Dict, List, Tuple from rich.console import Console from rich.table import Table @@ -16,7 +15,7 @@ from ojhunt.crawlers import crawlers as crawler_registry -def check_duplicate_queries(queries: List[Query]) -> None: +def check_duplicate_queries(queries: list[Query]) -> None: """Check for duplicate queries and print warning to stderr.""" counter = Counter((q.crawler, q.username) for q in queries) for (crawler, username), count in counter.items(): @@ -27,9 +26,9 @@ def check_duplicate_queries(queries: List[Query]) -> None: ) -def validate_crawlers(queries: List[Query], crawlers: Dict[str, CrawlerInfo]) -> bool: +def validate_crawlers(queries: list[Query], crawlers: dict[str, CrawlerInfo]) -> bool: """Validate that all queried crawlers exist.""" - unknown = set(q.crawler for q in queries if q.crawler not in crawlers) + unknown = {q.crawler for q in queries if q.crawler not in crawlers} if unknown: print( f"Error: unknown crawler(s): {', '.join(sorted(unknown))}", @@ -41,9 +40,9 @@ def validate_crawlers(queries: List[Query], crawlers: Dict[str, CrawlerInfo]) -> def validate_credentials( - queries: List[Query], - crawlers: Dict[str, CrawlerInfo], - crawler_logins: Dict[str, Tuple[str, str]], + queries: list[Query], + crawlers: dict[str, CrawlerInfo], + crawler_logins: dict[str, tuple[str, str]], ) -> bool: """ Validate that credential requirements are met for each query. @@ -108,7 +107,7 @@ def validate_credentials( def print_report( - results: List[QueryResult], + results: list[QueryResult], show_problems: bool, total_duration: float, json_output: bool = False, @@ -123,7 +122,7 @@ def print_report( if json_output: result_list = [] for r in results: - entry: Dict = { + entry: dict = { "crawler": r.crawler.name, "title": r.crawler.meta.title, "username": r.username, diff --git a/src/ojhunt/cli/parser.py b/src/ojhunt/cli/parser.py index 0a484a5..3c5fd42 100644 --- a/src/ojhunt/cli/parser.py +++ b/src/ojhunt/cli/parser.py @@ -4,13 +4,12 @@ import argparse import sys -from typing import Dict, List, Optional, Tuple from ojhunt.cli.models import Query from ojhunt.core.models import CrawlerInfo -def parse_crawler_login(values: Optional[List[str]]) -> Dict[str, Tuple[str, str]]: +def parse_crawler_login(values: list[str] | None) -> dict[str, tuple[str, str]]: """ Parse -l user:pass@crawler arguments. @@ -23,7 +22,7 @@ def parse_crawler_login(values: Optional[List[str]]) -> Dict[str, Tuple[str, str Raises: ValueError: If format is invalid. """ - result: Dict[str, Tuple[str, str]] = {} + result: dict[str, tuple[str, str]] = {} if not values: return result @@ -63,10 +62,10 @@ def parse_crawler_login(values: Optional[List[str]]) -> Dict[str, Tuple[str, str def build_all_queries( - default_username: str, crawlers: Dict[str, CrawlerInfo] -) -> List[Query]: + default_username: str, crawlers: dict[str, CrawlerInfo] +) -> list[Query]: """Build queries for all crawlers using default username.""" - return [Query(crawler=name, username=default_username) for name in crawlers.keys()] + return [Query(crawler=name, username=default_username) for name in crawlers] def create_parser() -> argparse.ArgumentParser: @@ -188,7 +187,7 @@ def create_parser() -> argparse.ArgumentParser: return parser -def parse_positional(args: List[str], default_username: Optional[str]) -> List[Query]: +def parse_positional(args: list[str], default_username: str | None) -> list[Query]: """ Parse positional arguments like: tourist@codeforces user:pass@vjudge poj @@ -243,8 +242,8 @@ def parse_positional(args: List[str], default_username: Optional[str]) -> List[Q def parse_args( - argv: Optional[List[str]] = None, -) -> Tuple[argparse.Namespace, List[Query], Dict[str, Tuple[str, str]]]: + argv: list[str] | None = None, +) -> tuple[argparse.Namespace, list[Query], dict[str, tuple[str, str]]]: """ Parse command line arguments and build query list. @@ -260,16 +259,15 @@ def parse_args( parser = create_parser() args = parser.parse_args(argv) - queries: List[Query] = [] - crawler_logins: Dict[str, Tuple[str, str]] = {} + queries: list[Query] = [] + crawler_logins: dict[str, tuple[str, str]] = {} if args.list: return args, queries, crawler_logins - if args.all: - if not args.default_username: - print("Error: -a/--all requires -d/--default-username", file=sys.stderr) - sys.exit(1) + if args.all and not args.default_username: + print("Error: -a/--all requires -d/--default-username", file=sys.stderr) + sys.exit(1) if not args.all and not args.queries: parser.print_help() diff --git a/src/ojhunt/cli/progress.py b/src/ojhunt/cli/progress.py index 8cd18f0..e96bdb4 100644 --- a/src/ojhunt/cli/progress.py +++ b/src/ojhunt/cli/progress.py @@ -7,7 +7,7 @@ import sys from dataclasses import dataclass, field from enum import Enum -from typing import Any, Dict, List, Optional +from typing import Any from rich.console import Console from rich.live import Live @@ -30,18 +30,18 @@ class TaskInfo: title: str username: str status: TaskStatus = TaskStatus.PENDING - solved: Optional[int] = None - submissions: Optional[int] = None - duration: Optional[float] = None - error: Optional[str] = None + solved: int | None = None + submissions: int | None = None + duration: float | None = None + error: str | None = None @dataclass class ProgressManager: - tasks: Dict[str, TaskInfo] = field(default_factory=dict) - task_order: List[str] = field(default_factory=list) + tasks: dict[str, TaskInfo] = field(default_factory=dict) + task_order: list[str] = field(default_factory=list) console: Console = field(default_factory=lambda: Console(stderr=True)) - live: Optional[Live] = None + live: Live | None = None is_tty: bool = field(default_factory=lambda: sys.stderr.isatty()) @staticmethod @@ -71,10 +71,10 @@ def complete_task( self, key: str, success: bool, - solved: Optional[int] = None, - submissions: Optional[int] = None, - duration: Optional[float] = None, - error: Optional[str] = None, + solved: int | None = None, + submissions: int | None = None, + duration: float | None = None, + error: str | None = None, ) -> None: if key in self.tasks: task = self.tasks[key] @@ -138,11 +138,11 @@ def __exit__(self, exc_type, exc_val, exc_tb): self.live.__exit__(exc_type, exc_val, exc_tb) return False - def get_results(self) -> List[Dict[str, Any]]: + def get_results(self) -> list[dict[str, Any]]: results = [] for crawler in self.task_order: task = self.tasks[crawler] - result: Dict[str, Any] = { + result: dict[str, Any] = { "crawler": task.crawler, "title": task.title, "username": task.username, diff --git a/src/ojhunt/core/__init__.py b/src/ojhunt/core/__init__.py index 2efc63c..c157d44 100644 --- a/src/ojhunt/core/__init__.py +++ b/src/ojhunt/core/__init__.py @@ -24,6 +24,6 @@ "LoginType", "NullCrawler", "QueryResult", - "run_crawler", "collect_solved_problems", + "run_crawler", ] diff --git a/src/ojhunt/core/credentials.py b/src/ojhunt/core/credentials.py index 5bef3bd..de79a17 100644 --- a/src/ojhunt/core/credentials.py +++ b/src/ojhunt/core/credentials.py @@ -3,10 +3,9 @@ """ import os -from typing import Dict, Optional -def get_login_kwargs(crawler_name: str) -> Optional[Dict[str, str]]: +def get_login_kwargs(crawler_name: str) -> dict[str, str] | None: """Return login kwargs for a crawler, or None if credentials are not configured. Looks up LOGIN_USERNAME__ and LOGIN_PASSWORD__ env vars. diff --git a/src/ojhunt/core/models.py b/src/ojhunt/core/models.py index e2951b6..e87db75 100644 --- a/src/ojhunt/core/models.py +++ b/src/ojhunt/core/models.py @@ -4,9 +4,10 @@ These types are used across CLI, web, and crawler modules. """ +from collections.abc import Awaitable, Callable from dataclasses import dataclass from enum import Enum -from typing import Any, Awaitable, Callable, Dict, List, Optional +from typing import Any import aiohttp @@ -29,7 +30,7 @@ class LoginType(Enum): SHARED_ACCOUNT = "shared_account" # Any shared account can query any user @classmethod - def from_meta(cls, value: Optional[str]) -> "LoginType": + def from_meta(cls, value: str | None) -> "LoginType": if not value: return cls.NOT_REQUIRED return cls(value) @@ -56,10 +57,10 @@ class CrawlerResult: solved: int submissions: int - solved_list: Optional[List[str]] = None + solved_list: list[str] | None = None @classmethod - def from_dict(cls, d: Dict[str, Any]) -> "CrawlerResult": + def from_dict(cls, d: dict[str, Any]) -> "CrawlerResult": """Build a CrawlerResult from the raw dict a crawler's query returns. Args: @@ -190,7 +191,7 @@ def query_sync(self, username: str, **kwargs: Any) -> CrawlerResult: return query_sync(self.query, username, **kwargs) -class CrawlerRegistry(Dict[str, CrawlerInfo]): +class CrawlerRegistry(dict[str, CrawlerInfo]): """Every crawler in this build, keyed by name. This is a dict, so anything a dict does works — iteration, len(), @@ -219,16 +220,16 @@ def __getattr__(self, name: str) -> CrawlerInfo: except KeyError: raise AttributeError(f"no crawler named {name!r}") from None - def __dir__(self) -> List[str]: + def __dir__(self) -> list[str]: return [*super().__dir__(), *self] def copy(self) -> "CrawlerRegistry": return CrawlerRegistry(self) - def __or__(self, other: Dict[str, CrawlerInfo]) -> "CrawlerRegistry": + def __or__(self, other: dict[str, CrawlerInfo]) -> "CrawlerRegistry": return CrawlerRegistry({**self, **other}) - def __ror__(self, other: Dict[str, CrawlerInfo]) -> "CrawlerRegistry": + def __ror__(self, other: dict[str, CrawlerInfo]) -> "CrawlerRegistry": return CrawlerRegistry({**other, **self}) @@ -241,9 +242,9 @@ class QueryResult: success: bool solved: int = 0 submissions: int = 0 - solved_list: Optional[List[str]] = None + solved_list: list[str] | None = None duration: float = 0.0 - error: Optional[str] = None + error: str | None = None class NullCrawler(CrawlerInfo): diff --git a/src/ojhunt/core/runner.py b/src/ojhunt/core/runner.py index 6c4a8fa..1aabb18 100644 --- a/src/ojhunt/core/runner.py +++ b/src/ojhunt/core/runner.py @@ -50,5 +50,5 @@ async def run_crawler( crawler=crawler, username=username, success=False, - error=f"{type(e).__name__}: {str(e)}", + error=f"{type(e).__name__}: {e!s}", ) diff --git a/src/ojhunt/core/session.py b/src/ojhunt/core/session.py index 11dcc18..15ff1fc 100644 --- a/src/ojhunt/core/session.py +++ b/src/ojhunt/core/session.py @@ -9,7 +9,7 @@ """ import importlib.metadata -from typing import Any, Optional +from typing import Any import aiohttp @@ -36,7 +36,7 @@ def _version() -> str: def create_session( - *, headers: Optional[dict] = None, **kwargs: Any + *, headers: dict | None = None, **kwargs: Any ) -> aiohttp.ClientSession: """ Build an ``aiohttp.ClientSession`` pre-seeded with OJHunt identification headers. diff --git a/src/ojhunt/core/stats.py b/src/ojhunt/core/stats.py index 647f42e..3172d04 100644 --- a/src/ojhunt/core/stats.py +++ b/src/ojhunt/core/stats.py @@ -2,12 +2,10 @@ Statistics calculation functions. """ -from typing import List, Set - from ojhunt.core.models import QueryResult -def collect_solved_problems(results: List[QueryResult]) -> Set[str]: +def collect_solved_problems(results: list[QueryResult]) -> set[str]: """ Collect all solved problems with deduplication. @@ -20,7 +18,7 @@ def collect_solved_problems(results: List[QueryResult]) -> Set[str]: Returns: Set of unique problem identifiers """ - all_solved: Set[str] = set() + all_solved: set[str] = set() for result in results: if not result.success or not result.solved_list: continue @@ -33,7 +31,7 @@ def collect_solved_problems(results: List[QueryResult]) -> Set[str]: return all_solved -def get_unique_solved(results: List[QueryResult]) -> int: +def get_unique_solved(results: list[QueryResult]) -> int: """ Total unique solved across all crawlers. """ diff --git a/src/ojhunt/crawlers/__init__.py b/src/ojhunt/crawlers/__init__.py index 2d55c80..1ba381c 100644 --- a/src/ojhunt/crawlers/__init__.py +++ b/src/ojhunt/crawlers/__init__.py @@ -114,11 +114,11 @@ async def main(): import inspect import pkgutil import sys +from collections.abc import Awaitable, Callable from functools import cache from pathlib import Path -from typing import TYPE_CHECKING, Any, Awaitable, Callable, List, Union +from typing import TYPE_CHECKING, Any, List, Union -from ojhunt.core.session import create_session from ojhunt.core.models import ( CrawlerInfo, CrawlerMeta, @@ -126,6 +126,7 @@ async def main(): CrawlerResult, LoginType, ) +from ojhunt.core.session import create_session from ojhunt.crawlers._help import compose_query_doc, render_crawler_doc if TYPE_CHECKING: @@ -136,7 +137,7 @@ async def main(): def query_sync( - crawler: "Union[CrawlerInfo, Callable[..., Awaitable[Any]]]", + crawler: "CrawlerInfo | Callable[..., Awaitable[Any]]", username: str, **kwargs: Any, ) -> CrawlerResult: @@ -223,9 +224,8 @@ def _discover() -> CrawlerRegistry: for _, module_name, _ in pkgutil.iter_modules([str(package_dir)]): if ( - module_name.startswith("test_") + module_name.startswith(("test_", "_")) or module_name.endswith("_test") - or module_name.startswith("_") or module_name == "conftest" ): continue @@ -274,5 +274,5 @@ def __getattr__(name: str) -> Any: raise AttributeError(f"module {__name__!r} has no attribute {name!r}") -def __dir__() -> List[str]: +def __dir__() -> list[str]: return sorted([*globals(), "crawlers"]) diff --git a/src/ojhunt/crawlers/_help.py b/src/ojhunt/crawlers/_help.py index 7e67799..3239452 100644 --- a/src/ojhunt/crawlers/_help.py +++ b/src/ojhunt/crawlers/_help.py @@ -27,14 +27,15 @@ """ import inspect -from typing import Any, Callable, List, Optional +from collections.abc import Callable +from typing import Any from ojhunt.core.models import CrawlerMeta, LoginType _RULE = "-" * 56 -def compose_query_doc(raw_doc: Optional[str], crawler_doc: str) -> str: +def compose_query_doc(raw_doc: str | None, crawler_doc: str) -> str: """Append generated crawler documentation to a query function's own docstring. The crawler's docstring is dedented first so that help() renders both halves @@ -61,7 +62,7 @@ def compose_query_doc(raw_doc: Optional[str], crawler_doc: str) -> str: return f"{inspect.cleandoc(raw_doc)}\n\n{_RULE}\n\n{wrapped_note}\n\n{crawler_doc}" -def _login_paragraph(name: str, meta: CrawlerMeta, params: List[str]) -> str: +def _login_paragraph(name: str, meta: CrawlerMeta, params: list[str]) -> str: if meta.login_type is LoginType.NOT_REQUIRED: return "Login: not required." @@ -102,7 +103,7 @@ def _login_paragraph(name: str, meta: CrawlerMeta, params: List[str]) -> str: return "\n".join(lines) -def _usage_example(name: str, meta: CrawlerMeta, params: List[str]) -> str: +def _usage_example(name: str, meta: CrawlerMeta, params: list[str]) -> str: sample = meta.test_username or "username" creds = "" if meta.login_type is not LoginType.NOT_REQUIRED: @@ -144,7 +145,7 @@ def render_crawler_doc( Returns: Documentation text, suitable for assigning to __doc__. """ - params: List[str] = list(inspect.signature(query_fn).parameters) + params: list[str] = list(inspect.signature(query_fn).parameters) sections = [f"{meta.title} — {meta.url}" if meta.url else meta.title] diff --git a/src/ojhunt/crawlers/_utils.py b/src/ojhunt/crawlers/_utils.py index c663777..9083def 100644 --- a/src/ojhunt/crawlers/_utils.py +++ b/src/ojhunt/crawlers/_utils.py @@ -28,8 +28,8 @@ import asyncio import sqlite3 +from collections.abc import Awaitable, Callable from pathlib import Path -from typing import Awaitable, Callable, Dict, List, Optional import aiohttp @@ -61,9 +61,7 @@ def _init_db() -> None: conn.close() -def _get_cached_labels( - oj_name: str, problem_ids: List[int] -) -> Dict[int, Optional[str]]: +def _get_cached_labels(oj_name: str, problem_ids: list[int]) -> dict[int, str | None]: if not _DB_PATH.exists(): return {} @@ -80,7 +78,7 @@ def _get_cached_labels( conn.close() -def _cache_labels(oj_name: str, mappings: Dict[int, Optional[str]]) -> None: +def _cache_labels(oj_name: str, mappings: dict[int, str | None]) -> None: _init_db() conn = _get_connection() @@ -98,10 +96,10 @@ def _cache_labels(oj_name: str, mappings: Dict[int, Optional[str]]) -> None: async def resolve_labels( session: aiohttp.ClientSession, oj_name: str, - problem_ids: List[int], - resolver: Callable[[aiohttp.ClientSession, int], Awaitable[Optional[str]]], + problem_ids: list[int], + resolver: Callable[[aiohttp.ClientSession, int], Awaitable[str | None]], rate_limit_delay: float = 0.0, -) -> Dict[int, Optional[str]]: +) -> dict[int, str | None]: """ Resolve problem labels with caching. @@ -130,7 +128,7 @@ async def resolve_labels( semaphore = asyncio.Semaphore(_CONCURRENT_REQUESTS) - async def resolve_with_semaphore(pid: int) -> tuple[int, Optional[str]]: + async def resolve_with_semaphore(pid: int) -> tuple[int, str | None]: async with semaphore: if rate_limit_delay > 0: await asyncio.sleep(rate_limit_delay) diff --git a/src/ojhunt/crawlers/aizu.py b/src/ojhunt/crawlers/aizu.py index 1698325..bb37232 100644 --- a/src/ojhunt/crawlers/aizu.py +++ b/src/ojhunt/crawlers/aizu.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "AIZU", @@ -39,7 +38,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query AIZU for user statistics. @@ -85,7 +84,7 @@ async def query( list_data = await response.json() # Extract unique problem IDs - solved_set = set(item.get("problemId") for item in list_data) + solved_set = {item.get("problemId") for item in list_data} return { "solved": status_data.get("status", {}).get("solved", 0), @@ -96,7 +95,7 @@ async def query( except aiohttp.ClientError as e: if "404" in str(e): raise ValueError("The user does not exist") - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except ValueError: raise except Exception: diff --git a/src/ojhunt/crawlers/atcoder.py b/src/ojhunt/crawlers/atcoder.py index ead84e7..fbb940d 100644 --- a/src/ojhunt/crawlers/atcoder.py +++ b/src/ojhunt/crawlers/atcoder.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, Union __crawler_meta__ = { "title": "AtCoder", @@ -37,9 +36,7 @@ } -async def query( - session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, None]]: +async def query(session: aiohttp.ClientSession, username: str) -> dict[str, int | None]: """ Query AtCoder for user statistics using kenkoooo's API. @@ -71,7 +68,7 @@ async def query( except aiohttp.ClientError as e: if isinstance(e, aiohttp.ClientResponseError) and e.status == 404: raise ValueError("The user does not exist") - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") # Query kenkoooo's API for AC count # Thank @kenkoooo for the API @@ -85,7 +82,7 @@ async def query( response.raise_for_status() data = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") solved = data.get("count", 0) diff --git a/src/ojhunt/crawlers/codechef.py b/src/ojhunt/crawlers/codechef.py index 8800c3c..f6f8e8c 100644 --- a/src/ojhunt/crawlers/codechef.py +++ b/src/ojhunt/crawlers/codechef.py @@ -28,7 +28,6 @@ import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "CodeChef", @@ -40,7 +39,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query CodeChef for user statistics. @@ -99,12 +98,12 @@ async def query( return { "solved": len(solved_set), "submissions": submissions, - "solved_list": sorted(list(solved_set)), + "solved_list": sorted(solved_set), } except ValueError: raise except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except Exception: raise RuntimeError("Error while parsing") diff --git a/src/ojhunt/crawlers/codeforces.py b/src/ojhunt/crawlers/codeforces.py index ac17a04..d9583a8 100644 --- a/src/ojhunt/crawlers/codeforces.py +++ b/src/ojhunt/crawlers/codeforces.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "CodeForces", @@ -41,7 +40,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query CodeForces for user statistics. @@ -67,7 +66,7 @@ async def query( return { "solved": len(ac_set), "submissions": submissions, - "solved_list": sorted(list(ac_set)), + "solved_list": sorted(ac_set), } @@ -114,7 +113,7 @@ async def _query_recursively( raise RuntimeError(comment) except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") problem_array = data.get("result", []) diff --git a/src/ojhunt/crawlers/codewars.py b/src/ojhunt/crawlers/codewars.py index f20a220..03af4f6 100644 --- a/src/ojhunt/crawlers/codewars.py +++ b/src/ojhunt/crawlers/codewars.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "Codewars", @@ -39,7 +38,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query Codewars for user statistics. @@ -112,7 +111,7 @@ async def query( except aiohttp.ClientError as e: if "404" in str(e): raise ValueError("The user does not exist") - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except ValueError: raise except Exception: diff --git a/src/ojhunt/crawlers/cses.py b/src/ojhunt/crawlers/cses.py index d47e035..491a57a 100644 --- a/src/ojhunt/crawlers/cses.py +++ b/src/ojhunt/crawlers/cses.py @@ -27,9 +27,9 @@ """ import re + import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union, Optional __crawler_meta__ = { "title": "CSES", @@ -71,10 +71,10 @@ async def _login( ): raise RuntimeError("CSES login failed: invalid credentials") except aiohttp.ClientError as e: - raise RuntimeError(f"Login request failed: {str(e)}") + raise RuntimeError(f"Login request failed: {e!s}") -async def _get_login_user_id(session: aiohttp.ClientSession) -> Optional[str]: +async def _get_login_user_id(session: aiohttp.ClientSession) -> str | None: async with session.get( f"{BASE_URL}/", timeout=aiohttp.ClientTimeout(total=30), @@ -93,10 +93,10 @@ async def _get_login_user_id(session: aiohttp.ClientSession) -> Optional[str]: async def query( session: aiohttp.ClientSession, username: str, - password: Optional[str] = None, - login_user: Optional[str] = None, - login_password: Optional[str] = None, -) -> Dict[str, Union[int, List[str], None]]: + password: str | None = None, + login_user: str | None = None, + login_password: str | None = None, +) -> dict[str, int | list[str] | None]: """ Query CSES for user statistics. @@ -139,7 +139,7 @@ async def query( else: raise ValueError("CSES requires login credentials.") - user_id: Optional[str] = None + user_id: str | None = None if username != cred_user: if username.isdigit(): user_id = username @@ -147,8 +147,8 @@ async def query( raise ValueError("CSES requires a numeric user ID to query other users") MAX_LOGIN_RETRIES = 2 - user_text: Optional[str] = None - problemset_text: Optional[str] = None + user_text: str | None = None + problemset_text: str | None = None for attempt in range(MAX_LOGIN_RETRIES + 1): if attempt > 0: @@ -180,7 +180,7 @@ async def query( response.raise_for_status() problemset_text = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") session_expired = LexborHTMLParser(problemset_text).css_first( '.content:lexbor-contains("Please login")' diff --git a/src/ojhunt/crawlers/csg.py b/src/ojhunt/crawlers/csg.py index 3bf6067..d831149 100644 --- a/src/ojhunt/crawlers/csg.py +++ b/src/ojhunt/crawlers/csg.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "CSG", @@ -39,7 +38,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str], None]]: +) -> dict[str, int | list[str] | None]: """ Query CSGrandeur OJ for user statistics. @@ -73,7 +72,7 @@ async def query( response.raise_for_status() data = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") if data.get("total", 0) == 0 or not data.get("rows"): raise ValueError("The user does not exist") diff --git a/src/ojhunt/crawlers/csu.py b/src/ojhunt/crawlers/csu.py index e23b436..fb8700e 100644 --- a/src/ojhunt/crawlers/csu.py +++ b/src/ojhunt/crawlers/csu.py @@ -26,10 +26,10 @@ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. """ -import aiohttp import re + +import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "CSU", @@ -41,7 +41,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query CSU for user statistics. @@ -89,7 +89,7 @@ async def query( """ ac_list_script = doc.css_first('script:lexbor-contains("function p(id,c){")') if ac_list_script: - ac_list = re.findall(r"p\((\d+),\d+\);", ac_list_script.text(), re.A) + ac_list = re.findall(r"p\((\d+),\d+\);", ac_list_script.text(), re.ASCII) else: ac_list = [] @@ -100,7 +100,7 @@ async def query( } except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except ValueError: raise except Exception: diff --git a/src/ojhunt/crawlers/darkbzoj.py b/src/ojhunt/crawlers/darkbzoj.py index 2348458..7114c1e 100644 --- a/src/ojhunt/crawlers/darkbzoj.py +++ b/src/ojhunt/crawlers/darkbzoj.py @@ -27,9 +27,9 @@ """ import re + import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "DarkBZOJ", @@ -41,7 +41,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query DarkBZOJ (黑暗爆炸OJ) for user statistics. @@ -73,15 +73,14 @@ async def query( except aiohttp.ClientError as e: if "404" in str(e): raise ValueError("The user does not exist") - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") doc = LexborHTMLParser(html) for panel in doc.css("div.panel"): classes = panel.attributes.get("class") or "" - if "panel-danger" in classes: - if "不存在该用户" in panel.text(): - raise ValueError("The user does not exist") + if "panel-danger" in classes and "不存在该用户" in panel.text(): + raise ValueError("The user does not exist") try: solved = 0 diff --git a/src/ojhunt/crawlers/eolymp.py b/src/ojhunt/crawlers/eolymp.py index b99e607..bbaee9c 100644 --- a/src/ojhunt/crawlers/eolymp.py +++ b/src/ojhunt/crawlers/eolymp.py @@ -27,8 +27,8 @@ """ import json + import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "EOlymp", @@ -44,7 +44,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str], None]]: +) -> dict[str, int | list[str] | None]: """ Query EOlymp for user statistics. @@ -88,7 +88,7 @@ async def query( response.raise_for_status() data = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except json.JSONDecodeError: raise RuntimeError("Failed to parse response") diff --git a/src/ojhunt/crawlers/hdu.py b/src/ojhunt/crawlers/hdu.py index c849aaa..8f2b7bf 100644 --- a/src/ojhunt/crawlers/hdu.py +++ b/src/ojhunt/crawlers/hdu.py @@ -27,9 +27,9 @@ """ import re + import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "HDU", @@ -41,7 +41,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query Hangzhou Dianzi University OJ for user statistics. @@ -77,7 +77,7 @@ async def query( except aiohttp.ClientError as e: if attempt >= max_retries - 1: raise RuntimeError( - f"Request failed after {max_retries} attempts: {str(e)}" + f"Request failed after {max_retries} attempts: {e!s}" ) print(f"HDU connection error, retry {attempt + 1}/{max_retries}...") diff --git a/src/ojhunt/crawlers/hust.py b/src/ojhunt/crawlers/hust.py index fbac8db..d1e6c24 100644 --- a/src/ojhunt/crawlers/hust.py +++ b/src/ojhunt/crawlers/hust.py @@ -28,7 +28,6 @@ import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "HUST", @@ -40,7 +39,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str], None]]: +) -> dict[str, int | list[str] | None]: """ Query HUST Online Judge for user statistics. @@ -67,7 +66,7 @@ async def query( response.raise_for_status() html = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") try: doc = LexborHTMLParser(html) diff --git a/src/ojhunt/crawlers/kilonova.py b/src/ojhunt/crawlers/kilonova.py index b6d1fb8..83088a3 100644 --- a/src/ojhunt/crawlers/kilonova.py +++ b/src/ojhunt/crawlers/kilonova.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "Kilonova", @@ -41,7 +40,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query Kilonova for user statistics. @@ -89,4 +88,4 @@ async def query( } except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") diff --git a/src/ojhunt/crawlers/leetcode.py b/src/ojhunt/crawlers/leetcode.py index 372de4b..4b963b1 100644 --- a/src/ojhunt/crawlers/leetcode.py +++ b/src/ojhunt/crawlers/leetcode.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "LeetCode.com", @@ -54,7 +53,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query LeetCode for user statistics via the public GraphQL API. @@ -93,7 +92,7 @@ async def query( raise RuntimeError(f"LeetCode API returned HTTP {response.status}") data = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") if "errors" in data: errors = data["errors"] diff --git a/src/ojhunt/crawlers/lightoj.py b/src/ojhunt/crawlers/lightoj.py index b7b0d41..7dd8360 100644 --- a/src/ojhunt/crawlers/lightoj.py +++ b/src/ojhunt/crawlers/lightoj.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "LightOJ", @@ -38,7 +37,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query LightOJ for user statistics. @@ -73,7 +72,7 @@ async def query( data = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") data_obj = data.get("data") if isinstance(data, dict) else None if not isinstance(data_obj, dict) or "userStat" not in data_obj: @@ -85,7 +84,7 @@ async def query( solved = int(user_stat["isSolved"]) submissions = int(user_stat["numSubmissions"]) except (KeyError, ValueError, TypeError) as e: - raise RuntimeError(f"Failed to parse response: {str(e)}") + raise RuntimeError(f"Failed to parse response: {e!s}") return { "solved": solved, diff --git a/src/ojhunt/crawlers/loj.py b/src/ojhunt/crawlers/loj.py index 1f64f25..45c1345 100644 --- a/src/ojhunt/crawlers/loj.py +++ b/src/ojhunt/crawlers/loj.py @@ -26,9 +26,9 @@ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. """ +from datetime import UTC, datetime + import aiohttp -from typing import Dict, List, Union -from datetime import datetime, timezone __crawler_meta__ = { "title": "LibreOJ", @@ -40,7 +40,7 @@ async def _resolve_solved_list( session: aiohttp.ClientSession, username: str -) -> List[str]: +) -> list[str]: """ Resolve solved list by querying submission API. @@ -85,7 +85,7 @@ async def _resolve_solved_list( async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query LibreOJ for user statistics. @@ -109,7 +109,7 @@ async def query( async with session.post( "https://api.loj.ac/api/user/getUserDetail", json={ - "now": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.000Z"), + "now": datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%S.000Z"), "timezone": "UTC", "username": username, }, @@ -137,7 +137,7 @@ async def query( } except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except ValueError: raise except Exception: diff --git a/src/ojhunt/crawlers/luogu.py b/src/ojhunt/crawlers/luogu.py index 0236a97..bc95c0c 100644 --- a/src/ojhunt/crawlers/luogu.py +++ b/src/ojhunt/crawlers/luogu.py @@ -26,10 +26,10 @@ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. """ -import aiohttp import json import re -from typing import Dict, List, Union + +import aiohttp __crawler_meta__ = { "title": "洛谷", @@ -67,7 +67,7 @@ def _extract_lentille_context(html: str) -> dict: async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str], None]]: +) -> dict[str, int | list[str] | None]: """ Query Luogu (洛谷) for user statistics. @@ -120,6 +120,6 @@ async def query( except ValueError: raise except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except Exception: raise RuntimeError("Error while parsing") diff --git a/src/ojhunt/crawlers/nbut.py b/src/ojhunt/crawlers/nbut.py index ef49e70..3070c5c 100644 --- a/src/ojhunt/crawlers/nbut.py +++ b/src/ojhunt/crawlers/nbut.py @@ -30,7 +30,6 @@ import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "NBUT", @@ -42,7 +41,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query NBUT for user statistics. @@ -75,7 +74,7 @@ async def query( raise RuntimeError(f"Server Response Error: {response.status}") submission_text = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") doc_submission = LexborHTMLParser(submission_text) user_link = doc_submission.css_first('a[href^="/User/view_user.xhtml?id="]') @@ -99,7 +98,7 @@ async def query( raise RuntimeError(f"Server Response Error: {response.status}") profile_text = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") try: doc_profile = LexborHTMLParser(profile_text) diff --git a/src/ojhunt/crawlers/nit.py b/src/ojhunt/crawlers/nit.py index e704bdf..1c4a499 100644 --- a/src/ojhunt/crawlers/nit.py +++ b/src/ojhunt/crawlers/nit.py @@ -28,7 +28,6 @@ import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Optional, Union from ojhunt.crawlers._utils import resolve_labels @@ -79,9 +78,7 @@ def _oj_map(oj: str) -> str: return oj -async def _resolve_label( - session: aiohttp.ClientSession, problem_id: int -) -> Optional[str]: +async def _resolve_label(session: aiohttp.ClientSession, problem_id: int) -> str | None: """ Resolve NIT problem ID to OJ-specific label. @@ -150,7 +147,7 @@ def _extract_number_from_cell(cell) -> int: async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query NIT (OurOJ) for user statistics. @@ -179,7 +176,7 @@ async def query( raise RuntimeError(f"Server Response Error: {response.status}") text = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") if "No such user!" in text: raise ValueError("The user does not exist") diff --git a/src/ojhunt/crawlers/nod.py b/src/ojhunt/crawlers/nod.py index 1aff75e..8fc6cd0 100644 --- a/src/ojhunt/crawlers/nod.py +++ b/src/ojhunt/crawlers/nod.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "51Nod", @@ -41,7 +40,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str], None]]: +) -> dict[str, int | list[str] | None]: """ Query 51Nod for user statistics. @@ -76,7 +75,7 @@ async def query( response.raise_for_status() data = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") user_stat = data.get("UserStat") if not user_stat: diff --git a/src/ojhunt/crawlers/nowcoder.py b/src/ojhunt/crawlers/nowcoder.py index bc18373..3e23d36 100644 --- a/src/ojhunt/crawlers/nowcoder.py +++ b/src/ojhunt/crawlers/nowcoder.py @@ -28,7 +28,6 @@ import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "牛客OJ", @@ -40,7 +39,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query Nowcoder for user statistics. @@ -126,7 +125,7 @@ async def query( page += 1 except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except ValueError: raise except Exception: diff --git a/src/ojhunt/crawlers/ojuz.py b/src/ojhunt/crawlers/ojuz.py index cbc3537..f5d7283 100644 --- a/src/ojhunt/crawlers/ojuz.py +++ b/src/ojhunt/crawlers/ojuz.py @@ -28,7 +28,6 @@ import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "oj.uz", @@ -42,7 +41,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query oj.uz for user statistics. @@ -74,7 +73,7 @@ async def query( response.raise_for_status() html = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") try: doc = LexborHTMLParser(html) @@ -104,7 +103,7 @@ def _extract_solved_count(doc: LexborHTMLParser) -> int: return 0 -def _extract_solved_list(doc: LexborHTMLParser) -> List[str]: +def _extract_solved_list(doc: LexborHTMLParser) -> list[str]: """Extract list of solved problem IDs from profile page.""" solved_list = [] for panel in doc.css("div.panel"): @@ -139,7 +138,7 @@ async def _count_submissions(session: aiohttp.ClientSession, username: str) -> i response.raise_for_status() html = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") doc = LexborHTMLParser(html) diff --git a/src/ojhunt/crawlers/poj.py b/src/ojhunt/crawlers/poj.py index f1e68a1..be76940 100644 --- a/src/ojhunt/crawlers/poj.py +++ b/src/ojhunt/crawlers/poj.py @@ -27,9 +27,9 @@ """ import re + import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Set, Union __crawler_meta__ = { "title": "POJ", @@ -55,7 +55,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query PKU JudgeOnline for user statistics. @@ -91,7 +91,7 @@ async def query( response.raise_for_status() html = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") if "Error -- no user found" in html: raise ValueError("The user does not exist") @@ -143,7 +143,7 @@ async def query( async def _query_via_status( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Reconstruct user statistics from the public /status submission log. @@ -153,12 +153,12 @@ async def _query_via_status( distinct problem ids of Accepted submissions. """ submissions = 0 - solved: Set[str] = set() - cursor: Union[int, None] = None + solved: set[str] = set() + cursor: int | None = None try: while True: - params: Dict[str, Union[str, int]] = { + params: dict[str, str | int] = { "user_id": username, "size": _STATUS_PAGE_SIZE, } @@ -179,7 +179,7 @@ async def _query_via_status( if not rows: break - run_ids: List[int] = [] + run_ids: list[int] = [] for row in rows: cells = row.css("td") if len(cells) < 4: @@ -209,7 +209,7 @@ async def _query_via_status( break cursor = next_cursor except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") # /status cannot tell a missing user apart from one with no submissions; # treat an empty log as a non-existent user. diff --git a/src/ojhunt/crawlers/sdutoj.py b/src/ojhunt/crawlers/sdutoj.py index e259dab..38b9101 100644 --- a/src/ojhunt/crawlers/sdutoj.py +++ b/src/ojhunt/crawlers/sdutoj.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "SDUT OJ", @@ -71,7 +70,7 @@ async def _fetch_sdutoj(session: aiohttp.ClientSession, api: str, data: dict) -> async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query SDUT OJ for user statistics. @@ -131,7 +130,7 @@ async def query( } except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except ValueError: raise except Exception: diff --git a/src/ojhunt/crawlers/timus.py b/src/ojhunt/crawlers/timus.py index 07176df..b1d5e79 100644 --- a/src/ojhunt/crawlers/timus.py +++ b/src/ojhunt/crawlers/timus.py @@ -26,10 +26,10 @@ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. """ -import aiohttp import re + +import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "Timus (URAL)", @@ -73,7 +73,7 @@ async def _query_list(session: aiohttp.ClientSession, uri: str) -> int: async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query Timus for user statistics. @@ -101,7 +101,7 @@ async def query( ) as response: search_text = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") doc_search = LexborHTMLParser(search_text) @@ -128,7 +128,7 @@ async def query( ) as response: profile_text = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") try: doc_profile = LexborHTMLParser(profile_text) diff --git a/src/ojhunt/crawlers/tlx.py b/src/ojhunt/crawlers/tlx.py index 65bab78..f8e202a 100644 --- a/src/ojhunt/crawlers/tlx.py +++ b/src/ojhunt/crawlers/tlx.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "TLX (TOKI Learning Exchange)", @@ -41,7 +40,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str], None]]: +) -> dict[str, int | list[str] | None]: """ Query TLX for user statistics. @@ -73,7 +72,7 @@ async def query( data = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") total_problems_tried = data.get("totalProblemsTried", 0) verdicts_map = data.get("totalProblemVerdictsMap", {}) diff --git a/src/ojhunt/crawlers/toph.py b/src/ojhunt/crawlers/toph.py index 31bb616..030405f 100644 --- a/src/ojhunt/crawlers/toph.py +++ b/src/ojhunt/crawlers/toph.py @@ -28,7 +28,6 @@ import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "Toph", @@ -40,7 +39,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str], None]]: +) -> dict[str, int | list[str] | None]: """ Query Toph for user statistics. @@ -70,7 +69,7 @@ async def query( response.raise_for_status() html = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") try: doc = LexborHTMLParser(html) diff --git a/src/ojhunt/crawlers/uoj.py b/src/ojhunt/crawlers/uoj.py index 077f622..9d202f0 100644 --- a/src/ojhunt/crawlers/uoj.py +++ b/src/ojhunt/crawlers/uoj.py @@ -26,10 +26,10 @@ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. """ -import aiohttp import re + +import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "UOJ", @@ -41,7 +41,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query UOJ for user statistics. @@ -74,15 +74,14 @@ async def query( except aiohttp.ClientError as e: if "404" in str(e): raise ValueError("The user does not exist") - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") # Check for user not found:
containing "不存在该用户" doc_profile = LexborHTMLParser(profile_text) for panel in doc_profile.css("div.panel"): classes = panel.attributes.get("class") or "" - if "panel-danger" in classes: - if "不存在该用户" in panel.text(): - raise ValueError("The user does not exist") + if "panel-danger" in classes and "不存在该用户" in panel.text(): + raise ValueError("The user does not exist") try: # Extract solved count - "AC 过的题目:共 217 道题" diff --git a/src/ojhunt/crawlers/uva.py b/src/ojhunt/crawlers/uva.py index c5ddff3..df2e12e 100644 --- a/src/ojhunt/crawlers/uva.py +++ b/src/ojhunt/crawlers/uva.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Optional, Union from ojhunt.crawlers._utils import resolve_labels @@ -41,9 +40,7 @@ UHUNT_PREFIX = "https://uhunt.onlinejudge.org" -async def _resolve_label( - session: aiohttp.ClientSession, problem_id: int -) -> Optional[str]: +async def _resolve_label(session: aiohttp.ClientSession, problem_id: int) -> str | None: """ Resolve UVA problem ID to display number using uhunt API. @@ -69,7 +66,7 @@ async def _resolve_label( async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query UVA Online Judge for user statistics using UHunt API. @@ -103,7 +100,7 @@ async def query( if uid == 0: raise ValueError("The user does not exist") except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") try: async with session.get( @@ -113,7 +110,7 @@ async def query( response.raise_for_status() data = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") ac_set = set() problem_array = data.get("subs", []) diff --git a/src/ojhunt/crawlers/vjudge.py b/src/ojhunt/crawlers/vjudge.py index 28abdd0..b515cb2 100644 --- a/src/ojhunt/crawlers/vjudge.py +++ b/src/ojhunt/crawlers/vjudge.py @@ -28,8 +28,8 @@ import json import logging + import aiohttp -from typing import Dict, List, Union, Optional logger = logging.getLogger(__name__) @@ -112,9 +112,9 @@ def _error_text(err: object) -> str: async def _fetch_page( - session: aiohttp.ClientSession, username: str, max_id: Optional[int] -) -> List: - params: Dict[str, str] = {"username": username, "pageSize": str(MAX_PAGE_SIZE)} + session: aiohttp.ClientSession, username: str, max_id: int | None +) -> list: + params: dict[str, str] = {"username": username, "pageSize": str(MAX_PAGE_SIZE)} if max_id is not None: params["maxId"] = str(max_id) @@ -145,7 +145,7 @@ async def _fetch_page( async def _paginate(session: aiohttp.ClientSession, username: str) -> tuple: ac_set: set = set() total_submissions = 0 - max_id: Optional[int] = None + max_id: int | None = None while True: problem_array = await _fetch_page(session, username, max_id) if not problem_array: @@ -180,7 +180,7 @@ async def _try_login( ) as response: text = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"vjudge login failed: {str(e)}") + raise RuntimeError(f"vjudge login failed: {e!s}") # Login response used to be the literal string "success". It is now an # empty 200 body on success, or JSON like # {"i18nKey":"user.auth.error.invalid_credentials","trustable":false} on failure. @@ -200,10 +200,10 @@ async def _try_login( async def query( session: aiohttp.ClientSession, username: str, - password: Optional[str] = None, - login_user: Optional[str] = None, - login_password: Optional[str] = None, -) -> Dict[str, Union[int, List[str]]]: + password: str | None = None, + login_user: str | None = None, + login_password: str | None = None, +) -> dict[str, int | list[str]]: """ Query VJudge for user statistics. @@ -260,7 +260,7 @@ async def query( logger.debug("vjudge: _AuthRequired on attempt %d", attempt) continue except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") except ValueError: raise except Exception as e: @@ -273,5 +273,5 @@ async def query( return { "solved": len(ac_set), "submissions": total_submissions, - "solved_list": sorted(list(ac_set)), + "solved_list": sorted(ac_set), } diff --git a/src/ojhunt/crawlers/vnoj.py b/src/ojhunt/crawlers/vnoj.py index 445f0a5..0a5b148 100644 --- a/src/ojhunt/crawlers/vnoj.py +++ b/src/ojhunt/crawlers/vnoj.py @@ -27,9 +27,9 @@ """ import re + import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union __crawler_meta__ = { "title": "VNOJ", @@ -43,7 +43,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str], None]]: +) -> dict[str, int | list[str] | None]: """ Query VNOJ for user statistics. @@ -79,7 +79,7 @@ async def query( response.raise_for_status() html = await response.text() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") solved = _extract_solved_count(LexborHTMLParser(html)) diff --git a/src/ojhunt/crawlers/yosupo.py b/src/ojhunt/crawlers/yosupo.py index 9bbb59e..f348290 100644 --- a/src/ojhunt/crawlers/yosupo.py +++ b/src/ojhunt/crawlers/yosupo.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "Yosupo Judge", @@ -41,7 +40,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query Yosupo Judge for user statistics. @@ -72,7 +71,7 @@ async def query( raise ValueError("The user does not exist") raise RuntimeError(f"API error: {response.status} {text}") except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") try: async with session.get( @@ -84,7 +83,7 @@ async def query( raise RuntimeError(f"API error: {response.status} {text}") statistics = await response.json() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") solved_map = statistics.get("solved_map", {}) solved_list = [ diff --git a/src/ojhunt/crawlers/yukicoder.py b/src/ojhunt/crawlers/yukicoder.py index 17610f2..b4b88ed 100644 --- a/src/ojhunt/crawlers/yukicoder.py +++ b/src/ojhunt/crawlers/yukicoder.py @@ -27,7 +27,6 @@ """ import aiohttp -from typing import Dict, List, Union __crawler_meta__ = { "title": "yukicoder", @@ -41,7 +40,7 @@ async def query( session: aiohttp.ClientSession, username: str -) -> Dict[str, Union[int, List[str]]]: +) -> dict[str, int | list[str]]: """ Query yukicoder for user statistics. @@ -71,7 +70,7 @@ async def query( raise ValueError("The user does not exist") response.raise_for_status() except aiohttp.ClientError as e: - raise RuntimeError(f"Request failed: {str(e)}") + raise RuntimeError(f"Request failed: {e!s}") solved_list = await _fetch_solved_list(session, username) @@ -86,7 +85,7 @@ async def query( async def _fetch_solved_list( session: aiohttp.ClientSession, username: str -) -> List[str]: +) -> list[str]: """ Fetch the list of solved problems for a user. diff --git a/src/ojhunt/web/api.py b/src/ojhunt/web/api.py index 359f5de..d937f22 100644 --- a/src/ojhunt/web/api.py +++ b/src/ojhunt/web/api.py @@ -3,26 +3,16 @@ """ import base64 -from datetime import datetime, timezone -from typing import Dict, List, Literal, Optional +from datetime import UTC, datetime +from typing import Literal -from fastapi import APIRouter, HTTPException, Path as PathParam +from fastapi import APIRouter, HTTPException +from fastapi import Path as PathParam from fastapi.responses import JSONResponse from pydantic import BaseModel, Field -from ojhunt.web.pdf import ( - HistoryEntry, - PdfSettings, - PdfSnapshot, - compute_day_key, - extract_data, - generate_pdf, - merge_history, -) - from ojhunt.core.credentials import get_login_kwargs -from ojhunt.core.models import LoginType -from ojhunt.core.models import NullCrawler +from ojhunt.core.models import LoginType, NullCrawler from ojhunt.core.models import QueryResult as CoreQueryResult from ojhunt.core.runner import run_crawler from ojhunt.core.stats import get_unique_solved @@ -33,6 +23,15 @@ get_all_status, ) from ojhunt.web.http_client import HttpClientDep +from ojhunt.web.pdf import ( + HistoryEntry, + PdfSettings, + PdfSnapshot, + compute_day_key, + extract_data, + generate_pdf, + merge_history, +) class CrawlerInfo(BaseModel): @@ -54,7 +53,7 @@ class CrawlerInfo(BaseModel): "checked in the current pass (also the default when the checker hasn't run)." ), ) - statusError: Optional[str] = Field( + statusError: str | None = Field( None, description="Reason the crawler is offline (present only when status is 'offline').", ) @@ -62,15 +61,13 @@ class CrawlerInfo(BaseModel): class CrawlersListResponse(BaseModel): error: bool = Field(False, description="Always false for success") - data: Dict[str, CrawlerInfo] = Field(..., description="Map of crawler name to info") + data: dict[str, CrawlerInfo] = Field(..., description="Map of crawler name to info") class QueryResult(BaseModel): solved: int = Field(..., description="Number of accepted problems") submissions: int = Field(..., description="Total number of submissions") - solvedList: Optional[List[str]] = Field( - None, description="List of solved problem IDs" - ) + solvedList: list[str] | None = Field(None, description="List of solved problem IDs") duration: float = Field(0, description="Query duration in seconds") @@ -80,12 +77,10 @@ class CrawlerResult(BaseModel): crawler: str = Field(..., description="Crawler name (e.g. 'codeforces')") username: str = Field(..., description="Username that was queried") error: bool = Field(..., description="True if the query failed") - data: Optional[QueryResult] = Field( + data: QueryResult | None = Field( None, description="Query data (present on success)" ) - message: Optional[str] = Field( - None, description="Error message (present on failure)" - ) + message: str | None = Field(None, description="Error message (present on failure)") @classmethod def from_model(cls, result: CoreQueryResult) -> "CrawlerResult": @@ -175,7 +170,7 @@ async def query_crawler( crawler = crawler_registry[crawler_name] - kwargs: Dict[str, str] = {} + kwargs: dict[str, str] = {} if crawler.meta.login_type == LoginType.SHARED_ACCOUNT: login_kwargs = get_login_kwargs(crawler_name) @@ -225,7 +220,7 @@ class MergeResponse(BaseModel): "other crawlers to avoid double-counting." ), ) -async def merge_results(results: List[CrawlerResult]) -> MergeResponse: +async def merge_results(results: list[CrawlerResult]) -> MergeResponse: core_results = [item.to_model() for item in results] total_submissions = sum(r.submissions for r in core_results if r.success) return MergeResponse( @@ -245,7 +240,7 @@ class PdfExtractRequest(BaseModel): class PdfExtractResponse(BaseModel): settings: PdfSettings = Field(..., description="Saved query settings from the PDF") - report_date: Optional[str] = Field( + report_date: str | None = Field( None, description="YYYY-MM-DD of the most recent history entry (for UI display)" ) @@ -253,7 +248,7 @@ class PdfExtractResponse(BaseModel): class PdfGenerateRequest(BaseModel): snapshot: PdfSnapshot settings: PdfSettings - previous_pdf_b64: Optional[str] = Field( + previous_pdf_b64: str | None = Field( None, description="Base64-encoded previous PDF to merge history from" ) @@ -305,13 +300,13 @@ async def generate_pdf_report(body: PdfGenerateRequest) -> PdfGenerateResponse: day_key = compute_day_key(body.snapshot.timezone) new_entry = HistoryEntry( key=day_key, - date=datetime.now(timezone.utc).isoformat(), + date=datetime.now(UTC).isoformat(), totalSolved=body.snapshot.totalSolved, totalSubmissions=body.snapshot.totalSubmissions, username=body.snapshot.username, ) - existing_history: List[HistoryEntry] = [] + existing_history: list[HistoryEntry] = [] if body.previous_pdf_b64: try: prev_bytes = base64.b64decode(body.previous_pdf_b64) diff --git a/src/ojhunt/web/app.py b/src/ojhunt/web/app.py index 174cb29..cae0197 100644 --- a/src/ojhunt/web/app.py +++ b/src/ojhunt/web/app.py @@ -5,9 +5,9 @@ """ import random +from collections.abc import AsyncGenerator from contextlib import asynccontextmanager from pathlib import Path -from typing import AsyncGenerator from dotenv import load_dotenv from fastapi import FastAPI, Request @@ -21,10 +21,11 @@ from starlette.middleware.base import BaseHTTPMiddleware from uvicorn.middleware.proxy_headers import ProxyHeadersMiddleware -from ojhunt.web.http_client import close_http_client, get_http_client, init_http_client from ojhunt.web.api import router as api_router -from ojhunt.web.pages import router as pages_router, jinja_env from ojhunt.web.crawler_status import start_checker, stop_checker +from ojhunt.web.http_client import close_http_client, get_http_client, init_http_client +from ojhunt.web.pages import jinja_env +from ojhunt.web.pages import router as pages_router load_dotenv() @@ -128,6 +129,7 @@ async def redoc_html() -> HTMLResponse: with_google_fonts=False, ) + app.add_middleware(SecurityHeadersMiddleware) app.add_middleware(LLMsDiscoverabilityMiddleware) app.add_middleware(ProxyHeadersMiddleware, trusted_hosts="*") diff --git a/src/ojhunt/web/legacy_db.py b/src/ojhunt/web/legacy_db.py index 99e06fe..7c4890b 100644 --- a/src/ojhunt/web/legacy_db.py +++ b/src/ojhunt/web/legacy_db.py @@ -5,9 +5,8 @@ import json import sqlite3 -from datetime import datetime, timezone +from datetime import UTC, datetime from pathlib import Path -from typing import List from zoneinfo import ZoneInfo, ZoneInfoNotFoundError from ojhunt.web.pdf import ( @@ -69,7 +68,7 @@ def windows_to_iana(win_tz: str) -> str: def _day_key_from_utc(utc_dt_str: str, iana_tz: str) -> str: """Convert a UTC datetime string to a YYYY-MM-DD day key in the user's timezone.""" try: - dt = datetime.fromisoformat(utc_dt_str).replace(tzinfo=timezone.utc) + dt = datetime.fromisoformat(utc_dt_str).replace(tzinfo=UTC) except ValueError: return utc_dt_str[:10] try: @@ -79,7 +78,7 @@ def _day_key_from_utc(utc_dt_str: str, iana_tz: str) -> str: return dt.astimezone(tz).strftime("%Y-%m-%d") -def find_user(con: sqlite3.Connection, username: str) -> List[dict]: +def find_user(con: sqlite3.Connection, username: str) -> list[dict]: """Find matching users by ABP username (case-insensitive). Returns list of {user_id, abp_username} dicts. @@ -114,7 +113,7 @@ def build_settings( (user_id,), ).fetchone() - queries: List[PdfQueryItem] = [] + queries: list[PdfQueryItem] = [] if row: try: data: dict = json.loads(row[0]) @@ -150,7 +149,7 @@ def build_settings( def build_history( con: sqlite3.Connection, user_id: int, iana_tz: str, main_username: str -) -> List[HistoryEntry]: +) -> list[HistoryEntry]: """Build history entries from all query summaries, deduped per day.""" rows = con.execute( """ @@ -163,7 +162,7 @@ def build_history( (user_id,), ).fetchall() - history: List[HistoryEntry] = [] + history: list[HistoryEntry] = [] for generate_time, solved, submission in rows: day_key = _day_key_from_utc(generate_time, iana_tz) entry = HistoryEntry( diff --git a/src/ojhunt/web/pages.py b/src/ojhunt/web/pages.py index febaf3c..1b0ce8a 100644 --- a/src/ojhunt/web/pages.py +++ b/src/ojhunt/web/pages.py @@ -18,7 +18,7 @@ from jinja2 import Environment, FileSystemLoader, select_autoescape from ojhunt.crawlers import crawlers as crawler_registry -from ojhunt.web.crawler_status import get_all_status, CrawlerAvailability, CheckStatus +from ojhunt.web.crawler_status import CheckStatus, CrawlerAvailability, get_all_status from ojhunt.web.legacy_db import export_user_pdf from ojhunt.web.pdf import PdfSnapshot, extract_data, generate_pdf, merge_history diff --git a/src/ojhunt/web/pdf.py b/src/ojhunt/web/pdf.py index c7edd8c..d21b8b2 100644 --- a/src/ojhunt/web/pdf.py +++ b/src/ojhunt/web/pdf.py @@ -3,9 +3,8 @@ import io import json import re -from datetime import datetime, timedelta, timezone +from datetime import UTC, datetime, timedelta from pathlib import Path -from typing import List from zoneinfo import ZoneInfo, ZoneInfoNotFoundError import matplotlib @@ -80,7 +79,7 @@ class PdfQueryItem(BaseModel): class PdfSettings(BaseModel): username: str - queries: List[PdfQueryItem] + queries: list[PdfQueryItem] class PdfCrawlerResult(BaseModel): @@ -95,7 +94,7 @@ class PdfSnapshot(BaseModel): totalSubmissions: int username: str timezone: str = "UTC" # IANA name; only used by compute_day_key() at query time - results: List[PdfCrawlerResult] = [] + results: list[PdfCrawlerResult] = [] class HistoryEntry(BaseModel): @@ -110,7 +109,7 @@ class PdfEmbeddedData(BaseModel): version: int exportedAt: str settings: PdfSettings - history: List[HistoryEntry] + history: list[HistoryEntry] def compute_day_key(iana_timezone: str) -> str: @@ -153,9 +152,9 @@ def extract_data(pdf_bytes: bytes) -> PdfEmbeddedData: def merge_history( - existing: List[HistoryEntry], + existing: list[HistoryEntry], new_entry: HistoryEntry, -) -> List[HistoryEntry]: +) -> list[HistoryEntry]: """Upsert new_entry into existing history by key (YYYY-MM-DD day key). For the same day, keeps the entry with the higher totalSolved. @@ -222,7 +221,7 @@ def _section(pdf: _Report, title: str) -> None: pdf.ln(2) -def _render_chart_png(history: List[HistoryEntry]) -> bytes: +def _render_chart_png(history: list[HistoryEntry]) -> bytes: """Render a solved-over-time line chart for all history entries, return PNG bytes.""" dates = [datetime.strptime(e.key, "%Y-%m-%d") for e in history] solved = [e.totalSolved for e in history] @@ -250,7 +249,7 @@ def _render_chart_png(history: List[HistoryEntry]) -> bytes: def generate_pdf( settings: PdfSettings, - history: List[HistoryEntry], + history: list[HistoryEntry], snapshot: PdfSnapshot, ) -> bytes: """Build an OJHunt PDF report and return the bytes.""" @@ -382,7 +381,7 @@ def generate_pdf( # JSON extraction by pypdf (it injects visible text at page boundaries). embedded = PdfEmbeddedData( version=1, - exportedAt=datetime.now(timezone.utc).isoformat(), + exportedAt=datetime.now(UTC).isoformat(), settings=settings, history=history, ).model_dump_json() diff --git a/tests/cli/output_test.py b/tests/cli/output_test.py index c8eeaee..72ba851 100644 --- a/tests/cli/output_test.py +++ b/tests/cli/output_test.py @@ -41,7 +41,7 @@ def test_build_all_queries(self): result = build_all_queries("tourist", crawlers) assert len(result) == 3 assert all(q.username == "tourist" for q in result) - assert set(q.crawler for q in result) == {"codeforces", "poj", "hdu"} + assert {q.crawler for q in result} == {"codeforces", "poj", "hdu"} def test_build_all_queries_empty(self): """Test with empty crawlers dict.""" @@ -342,7 +342,7 @@ def make_full_result( submissions: int = 0, solved_list=None, duration: float = 1.0, - error: str = None, + error: str | None = None, **meta_kwargs, ) -> QueryResult: """Helper to create QueryResult with all fields set.""" diff --git a/tests/cli/parser_test.py b/tests/cli/parser_test.py index 9a1615f..de2ccbd 100644 --- a/tests/cli/parser_test.py +++ b/tests/cli/parser_test.py @@ -171,7 +171,7 @@ def test_all_flag_with_default(self): def test_positional_after_separator(self): """Test positional args after --.""" - args, queries, crawler_logins = parse_args(["--", "tourist@codeforces"]) + _args, queries, crawler_logins = parse_args(["--", "tourist@codeforces"]) assert queries == [ Query(crawler="codeforces", username="tourist", password=None) ] @@ -213,25 +213,25 @@ def test_a_short_flag(self): def test_multiple_crawlers_with_default(self): """Test multiple crawlers with default username.""" - args, queries, _ = parse_args(["-d", "user", "--", "codeforces", "poj", "hdu"]) + _args, queries, _ = parse_args(["-d", "user", "--", "codeforces", "poj", "hdu"]) assert len(queries) == 3 assert all(q.username == "user" for q in queries) def test_user_with_at_in_name(self): """Test user with @ in username using last @ as separator.""" - args, queries, _ = parse_args(["--", "user@domain@codeforces"]) + _args, queries, _ = parse_args(["--", "user@domain@codeforces"]) assert queries == [ Query(crawler="codeforces", username="user@domain", password=None) ] def test_password_in_query(self): """Test password in query string.""" - args, queries, _ = parse_args(["--", "user:pass@vjudge"]) + _args, queries, _ = parse_args(["--", "user:pass@vjudge"]) assert queries == [Query(crawler="vjudge", username="user", password="pass")] def test_crawler_login_flag(self): """Test -l flag parses login credentials.""" - args, queries, crawler_logins = parse_args( + _args, queries, crawler_logins = parse_args( ["-l", "user:pass@vjudge", "--", "target@vjudge"] ) assert queries == [Query(crawler="vjudge", username="target", password=None)] @@ -239,7 +239,7 @@ def test_crawler_login_flag(self): def test_multiple_crawler_logins(self): """Test multiple -l flags.""" - args, queries, crawler_logins = parse_args( + _args, queries, crawler_logins = parse_args( [ "-l", "user1:pass1@vjudge", diff --git a/tests/core/runner_test.py b/tests/core/runner_test.py index d782181..7f3e6a1 100644 --- a/tests/core/runner_test.py +++ b/tests/core/runner_test.py @@ -4,7 +4,7 @@ import pytest -from ojhunt.core.models import CrawlerMeta, CrawlerInfo, CrawlerResult +from ojhunt.core.models import CrawlerInfo, CrawlerMeta, CrawlerResult from ojhunt.core.runner import run_crawler diff --git a/tests/crawlers/_utils_test.py b/tests/crawlers/_utils_test.py index de57e71..d128dec 100644 --- a/tests/crawlers/_utils_test.py +++ b/tests/crawlers/_utils_test.py @@ -2,18 +2,15 @@ Tests for crawlers._utils module """ -from typing import Optional - -import pytest import aiohttp +import pytest + from ojhunt.crawlers import _utils pytestmark = pytest.mark.network -async def _mock_resolver( - session: aiohttp.ClientSession, problem_id: int -) -> Optional[str]: +async def _mock_resolver(session: aiohttp.ClientSession, problem_id: int) -> str | None: return f"label-{problem_id}" @@ -33,7 +30,7 @@ async def test_resolve_labels_empty_list(session): async def test_resolve_labels_caching(session): call_count = 0 - async def counting_resolver(sess: aiohttp.ClientSession, pid: int) -> Optional[str]: + async def counting_resolver(sess: aiohttp.ClientSession, pid: int) -> str | None: nonlocal call_count call_count += 1 return f"label-{pid}" @@ -51,7 +48,7 @@ async def counting_resolver(sess: aiohttp.ClientSession, pid: int) -> Optional[s async def test_resolve_labels_mixed_cache_and_fetch(session): call_count = 0 - async def counting_resolver(sess: aiohttp.ClientSession, pid: int) -> Optional[str]: + async def counting_resolver(sess: aiohttp.ClientSession, pid: int) -> str | None: nonlocal call_count call_count += 1 return f"label-{pid}" @@ -71,7 +68,7 @@ async def counting_resolver(sess: aiohttp.ClientSession, pid: int) -> Optional[s async def test_resolve_labels_rate_limit(session): import time - async def slow_resolver(sess: aiohttp.ClientSession, pid: int) -> Optional[str]: + async def slow_resolver(sess: aiohttp.ClientSession, pid: int) -> str | None: return f"label-{pid}" start = time.time() diff --git a/tests/crawlers/aizu_test.py b/tests/crawlers/aizu_test.py index 74d72cb..7afc167 100644 --- a/tests/crawlers/aizu_test.py +++ b/tests/crawlers/aizu_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.aizu import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/atcoder_test.py b/tests/crawlers/atcoder_test.py index e9ffdbd..5b2b1ee 100644 --- a/tests/crawlers/atcoder_test.py +++ b/tests/crawlers/atcoder_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.atcoder import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/codechef_test.py b/tests/crawlers/codechef_test.py index 3af3cb5..dd6ada2 100644 --- a/tests/crawlers/codechef_test.py +++ b/tests/crawlers/codechef_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.codechef import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/codeforces_test.py b/tests/crawlers/codeforces_test.py index b1f1f6f..be301a5 100644 --- a/tests/crawlers/codeforces_test.py +++ b/tests/crawlers/codeforces_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.codeforces import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/codewars_test.py b/tests/crawlers/codewars_test.py index 2dca799..627fbca 100644 --- a/tests/crawlers/codewars_test.py +++ b/tests/crawlers/codewars_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.codewars import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/cses_test.py b/tests/crawlers/cses_test.py index 269e0c6..4538c02 100644 --- a/tests/crawlers/cses_test.py +++ b/tests/crawlers/cses_test.py @@ -7,8 +7,10 @@ """ import os + import pytest -from ojhunt.crawlers.cses import query, __crawler_meta__ + +from ojhunt.crawlers.cses import __crawler_meta__, query TEST_USERNAME = __crawler_meta__["test_username"] NOT_EXIST_ID = "9999999999" diff --git a/tests/crawlers/csg_test.py b/tests/crawlers/csg_test.py index 34ee4ee..80b0a72 100644 --- a/tests/crawlers/csg_test.py +++ b/tests/crawlers/csg_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.csg import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/csu_test.py b/tests/crawlers/csu_test.py index 98a0aa9..d0f1607 100644 --- a/tests/crawlers/csu_test.py +++ b/tests/crawlers/csu_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.csu import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/darkbzoj_test.py b/tests/crawlers/darkbzoj_test.py index 7618185..1f4813c 100644 --- a/tests/crawlers/darkbzoj_test.py +++ b/tests/crawlers/darkbzoj_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.darkbzoj import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/eolymp_test.py b/tests/crawlers/eolymp_test.py index abdc215..54f83ec 100644 --- a/tests/crawlers/eolymp_test.py +++ b/tests/crawlers/eolymp_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.eolymp import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/hdu_test.py b/tests/crawlers/hdu_test.py index 7d7f0d1..57f6819 100644 --- a/tests/crawlers/hdu_test.py +++ b/tests/crawlers/hdu_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.hdu import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/hust_test.py b/tests/crawlers/hust_test.py index 7e116a4..233f049 100644 --- a/tests/crawlers/hust_test.py +++ b/tests/crawlers/hust_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.hust import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/kilonova_test.py b/tests/crawlers/kilonova_test.py index 50278cc..785e4cb 100644 --- a/tests/crawlers/kilonova_test.py +++ b/tests/crawlers/kilonova_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.kilonova import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/leetcode_test.py b/tests/crawlers/leetcode_test.py index fd4af7b..100dbef 100644 --- a/tests/crawlers/leetcode_test.py +++ b/tests/crawlers/leetcode_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.leetcode import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/lightoj_test.py b/tests/crawlers/lightoj_test.py index 554020c..1270650 100644 --- a/tests/crawlers/lightoj_test.py +++ b/tests/crawlers/lightoj_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.lightoj import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/loj_test.py b/tests/crawlers/loj_test.py index ef5332b..808c18e 100644 --- a/tests/crawlers/loj_test.py +++ b/tests/crawlers/loj_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.loj import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/luogu_test.py b/tests/crawlers/luogu_test.py index 5bf9e47..fd1cc0d 100644 --- a/tests/crawlers/luogu_test.py +++ b/tests/crawlers/luogu_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.luogu import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/nbut_test.py b/tests/crawlers/nbut_test.py index e8b5767..f8ae773 100644 --- a/tests/crawlers/nbut_test.py +++ b/tests/crawlers/nbut_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.nbut import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/nit_test.py b/tests/crawlers/nit_test.py index b2571e2..862fbe3 100644 --- a/tests/crawlers/nit_test.py +++ b/tests/crawlers/nit_test.py @@ -4,7 +4,8 @@ import pytest from selectolax.lexbor import LexborHTMLParser -from ojhunt.crawlers.nit import __crawler_meta__, query, _extract_number_from_cell + +from ojhunt.crawlers.nit import __crawler_meta__, _extract_number_from_cell, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/nod_test.py b/tests/crawlers/nod_test.py index 325dadb..4eaffa2 100644 --- a/tests/crawlers/nod_test.py +++ b/tests/crawlers/nod_test.py @@ -3,7 +3,8 @@ """ import pytest -from ojhunt.crawlers.nod import query, __crawler_meta__ + +from ojhunt.crawlers.nod import __crawler_meta__, query TEST_USERNAME = __crawler_meta__["test_username"] NOT_EXIST_ID = "9999999999" diff --git a/tests/crawlers/nowcoder_test.py b/tests/crawlers/nowcoder_test.py index 36de68d..672a3a4 100644 --- a/tests/crawlers/nowcoder_test.py +++ b/tests/crawlers/nowcoder_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.nowcoder import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/ojuz_test.py b/tests/crawlers/ojuz_test.py index ccfe9a1..2b4ef0e 100644 --- a/tests/crawlers/ojuz_test.py +++ b/tests/crawlers/ojuz_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.ojuz import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/poj_test.py b/tests/crawlers/poj_test.py index 7972888..a7bfb45 100644 --- a/tests/crawlers/poj_test.py +++ b/tests/crawlers/poj_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.poj import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/registry_test.py b/tests/crawlers/registry_test.py index 63fa022..465e4bc 100644 --- a/tests/crawlers/registry_test.py +++ b/tests/crawlers/registry_test.py @@ -20,11 +20,13 @@ def _crawlers_imported_by(import_line: str) -> int: [ sys.executable, "-c", - "import sys\n" - f"{import_line}\n" - "print(len([m for m in sys.modules" - " if m.startswith('ojhunt.crawlers.')" - " and not m.rpartition('.')[2].startswith('_')]))", + ( + "import sys\n" + f"{import_line}\n" + "print(len([m for m in sys.modules" + " if m.startswith('ojhunt.crawlers.')" + " and not m.rpartition('.')[2].startswith('_')]))" + ), ], capture_output=True, text=True, diff --git a/tests/crawlers/sdutoj_test.py b/tests/crawlers/sdutoj_test.py index 3bfd645..f7d829a 100644 --- a/tests/crawlers/sdutoj_test.py +++ b/tests/crawlers/sdutoj_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.sdutoj import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/timus_test.py b/tests/crawlers/timus_test.py index a6e50c3..e0ba5c2 100644 --- a/tests/crawlers/timus_test.py +++ b/tests/crawlers/timus_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.timus import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/tlx_test.py b/tests/crawlers/tlx_test.py index e22e726..c04e49e 100644 --- a/tests/crawlers/tlx_test.py +++ b/tests/crawlers/tlx_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.tlx import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/toph_test.py b/tests/crawlers/toph_test.py index 3c5aa73..86369f9 100644 --- a/tests/crawlers/toph_test.py +++ b/tests/crawlers/toph_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.toph import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/uoj_test.py b/tests/crawlers/uoj_test.py index d7396ab..9d73864 100644 --- a/tests/crawlers/uoj_test.py +++ b/tests/crawlers/uoj_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.uoj import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/uva_test.py b/tests/crawlers/uva_test.py index b99e716..ed0a63a 100644 --- a/tests/crawlers/uva_test.py +++ b/tests/crawlers/uva_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.uva import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/vjudge_test.py b/tests/crawlers/vjudge_test.py index 81e2d14..42b509e 100644 --- a/tests/crawlers/vjudge_test.py +++ b/tests/crawlers/vjudge_test.py @@ -6,11 +6,13 @@ Set LOGIN_USERNAME__VJUDGE and LOGIN_PASSWORD__VJUDGE environment variables to run these tests. """ +import os + import pytest import pytest_asyncio -import os -from ojhunt.crawlers.vjudge import __crawler_meta__, query + from ojhunt.core.session import create_session +from ojhunt.crawlers.vjudge import __crawler_meta__, query TEST_USERNAME = __crawler_meta__["test_username"] NOT_EXIST_USERNAME = "fmv84zcq3hwu" diff --git a/tests/crawlers/vnoj_test.py b/tests/crawlers/vnoj_test.py index 94f9f2b..e911e50 100644 --- a/tests/crawlers/vnoj_test.py +++ b/tests/crawlers/vnoj_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.vnoj import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/yosupo_test.py b/tests/crawlers/yosupo_test.py index daea7c8..11164d9 100644 --- a/tests/crawlers/yosupo_test.py +++ b/tests/crawlers/yosupo_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.yosupo import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/crawlers/yukicoder_test.py b/tests/crawlers/yukicoder_test.py index 0727780..6f0513d 100644 --- a/tests/crawlers/yukicoder_test.py +++ b/tests/crawlers/yukicoder_test.py @@ -3,6 +3,7 @@ """ import pytest + from ojhunt.crawlers.yukicoder import __crawler_meta__, query pytestmark = pytest.mark.network diff --git a/tests/e2e/test_pdf_workflow.py b/tests/e2e/test_pdf_workflow.py index e1e28ea..de45b40 100644 --- a/tests/e2e/test_pdf_workflow.py +++ b/tests/e2e/test_pdf_workflow.py @@ -8,6 +8,7 @@ import pytest from playwright.sync_api import BrowserContext, Page, Route, expect +from e2e.helpers import BASE_URL, _add_query, _clear_storage, _row from ojhunt.web.pdf import ( HistoryEntry, PdfQueryItem, @@ -17,8 +18,6 @@ generate_pdf, ) -from e2e.helpers import BASE_URL, _add_query, _clear_storage, _row - _TMPDIR = os.environ.get("TMPDIR", tempfile.gettempdir()) _MOCK_CODEFORCES_RESPONSE = json.dumps( { diff --git a/tests/web/pdf_test.py b/tests/web/pdf_test.py index bbcd2f8..11ffa1a 100644 --- a/tests/web/pdf_test.py +++ b/tests/web/pdf_test.py @@ -1,8 +1,8 @@ """Tests for PDF generation and data extraction (web/pdf.py and /api/pdf/* routes).""" import base64 +from datetime import UTC from datetime import datetime as _datetime -from datetime import timezone as _timezone from unittest.mock import patch import pytest @@ -85,28 +85,28 @@ def _make_blank_pdf(text: str = "no data here") -> bytes: def test_compute_day_key_after_4am(): - fixed = _datetime(2026, 3, 29, 10, 0, tzinfo=_timezone.utc) + fixed = _datetime(2026, 3, 29, 10, 0, tzinfo=UTC) with patch("ojhunt.web.pdf.datetime") as mock_dt: mock_dt.now.side_effect = _mock_now(fixed) assert compute_day_key("UTC") == "2026-03-29" def test_compute_day_key_before_4am_returns_previous_day(): - fixed = _datetime(2026, 3, 29, 2, 30, tzinfo=_timezone.utc) + fixed = _datetime(2026, 3, 29, 2, 30, tzinfo=UTC) with patch("ojhunt.web.pdf.datetime") as mock_dt: mock_dt.now.side_effect = _mock_now(fixed) assert compute_day_key("UTC") == "2026-03-28" def test_compute_day_key_exactly_4am_is_current_day(): - fixed = _datetime(2026, 3, 29, 4, 0, tzinfo=_timezone.utc) + fixed = _datetime(2026, 3, 29, 4, 0, tzinfo=UTC) with patch("ojhunt.web.pdf.datetime") as mock_dt: mock_dt.now.side_effect = _mock_now(fixed) assert compute_day_key("UTC") == "2026-03-29" def test_compute_day_key_invalid_timezone_falls_back_to_utc(): - fixed = _datetime(2026, 3, 29, 10, 0, tzinfo=_timezone.utc) + fixed = _datetime(2026, 3, 29, 10, 0, tzinfo=UTC) with patch("ojhunt.web.pdf.datetime") as mock_dt: mock_dt.now.side_effect = _mock_now(fixed) assert compute_day_key("Not/A/Zone") == "2026-03-29" @@ -114,7 +114,7 @@ def test_compute_day_key_invalid_timezone_falls_back_to_utc(): def test_compute_day_key_timezone_before_4am_local(): # 03:00 America/New_York (EDT, UTC-4) = 07:00 UTC — before 4am locally - fixed = _datetime(2026, 3, 29, 7, 0, tzinfo=_timezone.utc) + fixed = _datetime(2026, 3, 29, 7, 0, tzinfo=UTC) with patch("ojhunt.web.pdf.datetime") as mock_dt: mock_dt.now.side_effect = _mock_now(fixed) assert compute_day_key("America/New_York") == "2026-03-28" @@ -124,7 +124,7 @@ def test_compute_day_key_dst_spring_forward(): # 2024-03-10 is US "spring forward" day (clocks jump 2am EST → 3am EDT at 07:00 UTC). # 07:30 UTC = 03:30 EDT — inside the spring-forward gap, still before 4am local. # Day key should roll back to the previous day, not produce an incorrect date. - fixed = _datetime(2024, 3, 10, 7, 30, tzinfo=_timezone.utc) + fixed = _datetime(2024, 3, 10, 7, 30, tzinfo=UTC) with patch("ojhunt.web.pdf.datetime") as mock_dt: mock_dt.now.side_effect = _mock_now(fixed) assert compute_day_key("America/New_York") == "2024-03-09" @@ -132,7 +132,7 @@ def test_compute_day_key_dst_spring_forward(): def test_compute_day_key_dst_spring_forward_after_4am(): # 08:00 UTC on spring-forward day = 04:00 EDT — exactly 4am, so current day. - fixed = _datetime(2024, 3, 10, 8, 0, tzinfo=_timezone.utc) + fixed = _datetime(2024, 3, 10, 8, 0, tzinfo=UTC) with patch("ojhunt.web.pdf.datetime") as mock_dt: mock_dt.now.side_effect = _mock_now(fixed) assert compute_day_key("America/New_York") == "2024-03-10" From 14a0bfd8764ac0e7015a9960937e4391efc9eb9d Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:33:51 +1000 Subject: [PATCH 03/10] fix(docs): keep Callable unqualified in generated library docs Moving `Callable`/`Awaitable` from `typing` to `collections.abc` (UP035, in the previous commit) changed how they stringify: `typing.Callable` rendered bare, but `collections.abc.Callable` carries its module prefix. That leaked into docs/library.md, where `query_sync()` and `CrawlerInfo.query` grew `collections.abc.` prefixes that tell a reader nothing. _readable() already exists to drop such prefixes; give it the new module. --- docs/library.md | 4 ++-- scripts/generate_library_docs.py | 6 +++++- 2 files changed, 7 insertions(+), 3 deletions(-) diff --git a/docs/library.md b/docs/library.md index 4803453..69361ff 100644 --- a/docs/library.md +++ b/docs/library.md @@ -149,7 +149,7 @@ should. ### `query_sync()` ```text -query_sync(crawler: CrawlerInfo | collections.abc.Callable[..., collections.abc.Awaitable[Any]], username: str, **kwargs: Any) -> CrawlerResult +query_sync(crawler: CrawlerInfo | Callable[..., Awaitable[Any]], username: str, **kwargs: Any) -> CrawlerResult Query a crawler synchronously, opening and closing a session for you. @@ -226,7 +226,7 @@ Attributes: Fields name: str meta: CrawlerMeta - query: collections.abc.Callable[..., collections.abc.Awaitable[CrawlerResult]] + query: Callable[..., Awaitable[CrawlerResult]] ``` ### `CrawlerInfo.query_sync()` diff --git a/scripts/generate_library_docs.py b/scripts/generate_library_docs.py index ae1e77d..b3322de 100644 --- a/scripts/generate_library_docs.py +++ b/scripts/generate_library_docs.py @@ -59,7 +59,11 @@ def _doc(obj: Any) -> str: def _readable(annotation: str) -> str: """Drop the module prefixes that only add noise.""" - return annotation.replace("typing.", "").replace("ojhunt.core.models.", "") + return ( + annotation.replace("typing.", "") + .replace("collections.abc.", "") + .replace("ojhunt.core.models.", "") + ) def _signature(fn: Callable[..., Any]) -> inspect.Signature: From 1b446e5c2af2805f3150b761ba21981c3e32f323 Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:35:55 +1000 Subject: [PATCH 04/10] chore(lint): configure ruff for the 0.16 default rule set MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Records the five places the project declines a rule from ruff 0.16's new 413-rule default, so that nobody has to rediscover the reasoning: - BLE001 is ignored project-wide. The catch-all `except Exception` is the crawler contract — any parse or transport failure becomes a RuntimeError the runner reports per crawler, so one judge's surprise never aborts a run. - `fastapi.File` joins extend-immutable-calls. `File(...)` in a parameter default is how FastAPI declares an upload, not a B008 mistake. - FLY002 is suppressed on the CSP in web/app.py. Its fix collapses the list into one 300-character line and deletes the comments inside it, and ADR 0010 records that block as load-bearing. - UP031 is suppressed on the eolymp GraphQL query. .format() needs every brace in the query body doubled. - DTZ007 is suppressed on the chart date parse in web/pdf.py. Those keys are already local-day strings from compute_day_key(); the axis needs their order, not an instant. ADR 0016 carries the argument for taking the new default set rather than pinning `select = ["E4", "E7", "E9", "F"]`, and for reversing the typing convention it collides with. Findings drop from 44 to 16. --- docs/adr/0016-adopt-ruff-default-rule-set.md | 68 ++++++++++++++++++++ docs/development.md | 1 + pyproject.toml | 10 +++ src/ojhunt/crawlers/eolymp.py | 4 +- src/ojhunt/web/app.py | 3 +- src/ojhunt/web/pdf.py | 4 +- 6 files changed, 87 insertions(+), 3 deletions(-) create mode 100644 docs/adr/0016-adopt-ruff-default-rule-set.md diff --git a/docs/adr/0016-adopt-ruff-default-rule-set.md b/docs/adr/0016-adopt-ruff-default-rule-set.md new file mode 100644 index 0000000..afc6dd0 --- /dev/null +++ b/docs/adr/0016-adopt-ruff-default-rule-set.md @@ -0,0 +1,68 @@ +# ADR 0016 — Adopt Ruff's Default Rule Set, Which Reverses the Typing Convention + +**Status:** Accepted + +## Context + +`pyproject.toml` never pinned `[tool.ruff.lint] select`, so the project always linted with +whatever ruff shipped as its default. That default was stable for years at 59 rules +(`E4`, `E7`, `E9`, `F`). Ruff 0.16.0 expanded it to 413 rules. A dependabot bump from +0.15.22 to 0.16.0 therefore turned 0 findings into 534 across 98 files, and CI failed. + +Two of the newly enabled rule groups collide with the codebase on purpose rather than by +accident: + +- `UP006`, `UP007`, `UP035` and `UP045` — 341 of the 534 findings — demand PEP 585 and PEP 604 + syntax (`dict[str, X]`, `X | None`). `docs/dev/python.md` said the opposite: "Use `Dict`, + `List`, `Union` from the `typing` module." Both rules cannot hold. +- `BLE001` flags 23 `except Exception` blocks. Every crawler ends in one by design. + +## Options Considered + +### Option A: Pin the old default set, `select = ["E4", "E7", "E9", "F"]` + +**Rejected because:** it freezes the linter at the 2023 default forever and hides real defects +this bump surfaced. Four of them were genuine: two functions timed their own work by +subtracting `datetime.now()` readings, which an NTP step corrupts, and two rendered timestamps +named no timezone. A rule set chosen to produce zero findings cannot find those. + +### Option B: Adopt the new default set, but keep the typing convention + +**Rejected because:** it needs `ignore = ["UP006", "UP007", "UP035", "UP045"]` — a permanent +exemption whose only argument is that the code already looks that way. The +`format-lint-python.sh` hook runs `ruff check --fix` on every edited file, so without the +exemption the convention is unenforceable anyway, and with it the codebase keeps a style that +Python has deprecated since 3.9. + +### Option C: Adopt the new default set and modernise the typing (chosen) + +The autofix does the mechanical work. `requires-python` is already `>=3.12`, so no annotation +in this repo needs the `typing` spelling for compatibility. + +## Decision + +**Option C.** The default rule set stands as ruff ships it. Three narrow exceptions are +recorded in config, and three at the site: + +| Exception | Where | Why | +|-----------|-------|-----| +| `ignore = ["BLE001"]` | `pyproject.toml` | The catch-all is the crawler contract: any parse or transport failure becomes a `RuntimeError` the runner reports per crawler, so one judge's surprise never aborts a run. | +| `extend-immutable-calls = ["fastapi.File"]` | `pyproject.toml` | `File(...)` in a parameter default is how FastAPI declares an upload. | +| `# noqa: FLY002` | `web/app.py` | The fix collapses the CSP into one 300-character line and deletes the comments inside it. That block is load-bearing — see [ADR 0010](0010-relaxed-csp-for-inline-alpine.md). | +| `# noqa: UP031` | `crawlers/eolymp.py` | `.format()` needs every brace in the GraphQL body doubled. | +| `# noqa: DTZ007` | `web/pdf.py` | The chart axis parses day keys that are already local-day strings. It needs their order, not an instant. | + +`docs/dev/python.md` now states the PEP 585/604 rule, and the `query()` templates in +`docs/dev/crawlers.md` were updated so a new crawler does not reintroduce the old style. + +## Consequences + +- A future ruff release that expands the defaults again will surface findings the same way. + That is accepted: CI runs `./doit.sh lint`, so the bump fails loudly on the dependabot PR + rather than landing silently. +- `Dict`, `List`, `Optional` and `Union` no longer appear in annotations. A patch that adds one + back gets rewritten by the `format-lint-python.sh` hook on the next edit. +- The 413-rule set covers `SIM`, `C4`, `B`, `DTZ`, `RUF`, `PL` and more, so new code meets + checks that were never applied to the code already in the tree. +- `./doit.sh lint` runs `ruff check` only. Ruff 0.16 also formats Python blocks inside + Markdown, and that drift stays invisible to CI until `ruff format --check` joins the task. diff --git a/docs/development.md b/docs/development.md index c4580e3..7010118 100644 --- a/docs/development.md +++ b/docs/development.md @@ -62,3 +62,4 @@ Significant architectural decisions and their rationale are recorded in [`docs/a - [ADR 0013](./adr/0013-lazy-crawler-registry.md) — the registry is the module attribute `ojhunt.crawlers.crawlers`, discovered on first access; the `TYPE_CHECKING` declaration is load-bearing for ruff F822 - [ADR 0014](./adr/0014-generated-crawler-help.md) — crawler `help()` text is generated from `__crawler_meta__` and attached to `CrawlerInfo.__doc__`; crawler module docstrings stay reserved for the license header - [ADR 0015](./adr/0015-submissions-floor-is-solved.md) — every crawler reports at least its `solved` count as `submissions`, so the figure is a lower bound; the rule lives in the crawler files, not in `CrawlerResult` +- [ADR 0016](./adr/0016-adopt-ruff-default-rule-set.md) — ruff's default rule set stands unpinned, which reversed the typing convention to PEP 585/604; the five exceptions (`BLE001`, `fastapi.File`, and three `noqa`s) are listed there diff --git a/pyproject.toml b/pyproject.toml index d29169c..b081e54 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -45,6 +45,16 @@ packages = ["src/ojhunt"] [tool.ruff] exclude = ["archived_crawlers"] +[tool.ruff.lint] +# BLE001: every crawler ends in `except Exception` by contract — any parse or +# transport failure becomes a RuntimeError that the runner reports per crawler, +# so one crawler's surprise never aborts the run. See ADR 0016. +ignore = ["BLE001"] + +[tool.ruff.lint.flake8-bugbear] +# B008: `File(...)` in a parameter default is how FastAPI declares an upload. +extend-immutable-calls = ["fastapi.File"] + [tool.pytest.ini_options] python_files = ["*_test.py", "test_*.py"] testpaths = ["tests", "README.md", "docs/"] diff --git a/src/ojhunt/crawlers/eolymp.py b/src/ojhunt/crawlers/eolymp.py index bbaee9c..9fd46d3 100644 --- a/src/ojhunt/crawlers/eolymp.py +++ b/src/ojhunt/crawlers/eolymp.py @@ -63,6 +63,8 @@ async def query( username = username.strip() + # %-formatting stays: .format() and f-strings would need every brace in the + # GraphQL body below doubled, which hides the query's real shape. query_str = """ { members(first: 1, search: "%s") { @@ -76,7 +78,7 @@ async def query( } } } - """ % username.replace('"', '\\"') + """ % username.replace('"', '\\"') # noqa: UP031 try: async with session.post( diff --git a/src/ojhunt/web/app.py b/src/ojhunt/web/app.py index cae0197..b202404 100644 --- a/src/ojhunt/web/app.py +++ b/src/ojhunt/web/app.py @@ -50,7 +50,8 @@ async def dispatch(self, request: Request, call_next) -> Response: # Content-Security-Policy. Relaxed: 'unsafe-inline'/'unsafe-eval' are required because # index.html uses inline Alpine.js expressions and the standard Alpine build evaluates them # via Function(). Google Fonts is the only third-party origin; everything else is same-origin. -_CSP = "; ".join( +# Keep the list form: it is what carries the per-directive comments below. +_CSP = "; ".join( # noqa: FLY002 [ "default-src 'self'", "script-src 'self' 'unsafe-inline' 'unsafe-eval'", diff --git a/src/ojhunt/web/pdf.py b/src/ojhunt/web/pdf.py index d21b8b2..f60fc8a 100644 --- a/src/ojhunt/web/pdf.py +++ b/src/ojhunt/web/pdf.py @@ -223,7 +223,9 @@ def _section(pdf: _Report, title: str) -> None: def _render_chart_png(history: list[HistoryEntry]) -> bytes: """Render a solved-over-time line chart for all history entries, return PNG bytes.""" - dates = [datetime.strptime(e.key, "%Y-%m-%d") for e in history] + # Naive on purpose: the keys are already local-day strings from compute_day_key(), + # and the axis needs their order, not an instant. + dates = [datetime.strptime(e.key, "%Y-%m-%d") for e in history] # noqa: DTZ007 solved = [e.totalSolved for e in history] fig, ax = plt.subplots(figsize=(7.09, 1.8)) # ~180 mm wide at 96 dpi From ab53dae0b977c905cb11666c458f2d8a94659481 Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:37:04 +1000 Subject: [PATCH 05/10] chore(docs): format python blocks in markdown with ruff 0.16 Ruff 0.16 formats Python blocks inside Markdown, which `ruff format .` now reaches. Four files drifted. The result is worth keeping rather than excluding: these blocks are templates that get copied into real .py files, so matching the formatter is what a reader wants from them. One block was edited by hand first. The standard-test-case list in the crawlers skill puts one test per line, and the formatter would have exploded the first line across three because its trailing comment pushed it past 88 characters. Dropping the redundant "raises" from two comments keeps the shape and the exact match string. --- .claude/skills/ojhunt-crawlers/SKILL.md | 4 ++-- docs/dev/crawlers.md | 20 +++++++++++++++----- docs/dev/e2e.md | 4 +++- docs/dev/testing.md | 3 ++- 4 files changed, 22 insertions(+), 9 deletions(-) diff --git a/.claude/skills/ojhunt-crawlers/SKILL.md b/.claude/skills/ojhunt-crawlers/SKILL.md index 4b108ad..3b54ed0 100644 --- a/.claude/skills/ojhunt-crawlers/SKILL.md +++ b/.claude/skills/ojhunt-crawlers/SKILL.md @@ -191,8 +191,8 @@ Site accessible? Every crawler test file must have all three (see the test template in `docs/dev/crawlers.md`): ```python notest -async def test_user_not_exist(session): ... # raises ValueError "The user does not exist" -async def test_username_with_space(session): ... # raises ValueError +async def test_user_not_exist(session): ... # ValueError "The user does not exist" +async def test_username_with_space(session): ... # ValueError async def test_valid_user(session): ... # asserts solved/submissions/solved_list ``` diff --git a/docs/dev/crawlers.md b/docs/dev/crawlers.md index fdece0b..bcafeea 100644 --- a/docs/dev/crawlers.md +++ b/docs/dev/crawlers.md @@ -43,9 +43,9 @@ All crawlers return: ```python notest { - "solved": int, # Number of accepted problems - "submissions": int, # Total submissions, never below "solved" - "solved_list": list|None # Problem IDs (None if unavailable) + "solved": int, # Number of accepted problems + "submissions": int, # Total submissions, never below "solved" + "solved_list": list | None, # Problem IDs (None if unavailable) } ``` @@ -82,7 +82,10 @@ __crawler_meta__ = { "test_username": "known_active_user", } -async def query(session: aiohttp.ClientSession, username: str, password: Optional[str] = None) -> Dict[str, Union[int, List[str], None]]: + +async def query( + session: aiohttp.ClientSession, username: str, password: Optional[str] = None +) -> Dict[str, Union[int, List[str], None]]: """Query OJ Name for user statistics. Args: @@ -135,7 +138,10 @@ __crawler_meta__ = { "test_username": "known_active_user", } -async def query(session: aiohttp.ClientSession, username: str, password: Optional[str] = None) -> Dict[str, Union[int, List[str], None]]: + +async def query( + session: aiohttp.ClientSession, username: str, password: Optional[str] = None +) -> Dict[str, Union[int, List[str], None]]: """Query Your OJ for user statistics. Args: @@ -183,6 +189,7 @@ __crawler_meta__ = { "test_username": "known_active_user", } + async def query( session: aiohttp.ClientSession, username: str, @@ -229,16 +236,19 @@ from ojhunt.crawlers.example import query, __crawler_meta__ TEST_USERNAME = __crawler_meta__["test_username"] NOT_EXIST_USERNAME = "fmv84zcq3hwu_notexist" + @pytest.mark.asyncio async def test_user_not_exist(session): with pytest.raises(ValueError, match="The user does not exist"): await query(session, NOT_EXIST_USERNAME) + @pytest.mark.asyncio async def test_username_with_space(session): with pytest.raises(ValueError): await query(session, " ") + @pytest.mark.asyncio async def test_valid_user(session): result = await query(session, TEST_USERNAME) diff --git a/docs/dev/e2e.md b/docs/dev/e2e.md index 6897f74..9d977dc 100644 --- a/docs/dev/e2e.md +++ b/docs/dev/e2e.md @@ -36,7 +36,9 @@ See [`docs/dev/testing.md`](testing.md) for shared pytest fixture and assertion covered by `test_query.py`. The success response shape is: ```python notest { - "crawler": "", "username": "", "error": False, + "crawler": "", + "username": "", + "error": False, "data": {"solved": 100, "submissions": 200, "solvedList": ["1A"], "duration": 0.1}, "message": None, } diff --git a/docs/dev/testing.md b/docs/dev/testing.md index 5cee223..4d7d1b6 100644 --- a/docs/dev/testing.md +++ b/docs/dev/testing.md @@ -40,10 +40,11 @@ New page routes must have a corresponding unit test. Use `TestClient` with monke ```python notest from starlette.testclient import TestClient + client = TestClient(app, follow_redirects=False) # File upload syntax: -files={"field": ("name.pdf", bytes_content, "application/pdf")} +files = {"field": ("name.pdf", bytes_content, "application/pdf")} ``` ## Markdown doc tests From 1984ab3e76988c2a345020bb3104e980e38c959b Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:38:50 +1000 Subject: [PATCH 06/10] fix: measure query durations with a monotonic clock run_crawler() and _async_main() timed their own work by subtracting two datetime.now() readings. Wall-clock time is the wrong instrument for that: it can be stepped by NTP or a timezone change mid-query, which yields a duration that is too large, too small, or negative. time.monotonic() cannot run backwards. DTZ005 pointed at these four calls for the narrower reason that they carry no timezone. Adding one would have silenced the rule while leaving the real defect, so the clock changes instead. The reported figure keeps its meaning and its format. Verified with a crawler that sleeps 0.35s: run_crawler reports 0.352. --- src/ojhunt/__main__.py | 7 +++---- src/ojhunt/core/runner.py | 6 +++--- 2 files changed, 6 insertions(+), 7 deletions(-) diff --git a/src/ojhunt/__main__.py b/src/ojhunt/__main__.py index c2c8478..5645a75 100755 --- a/src/ojhunt/__main__.py +++ b/src/ojhunt/__main__.py @@ -8,7 +8,7 @@ import asyncio import inspect import sys -from datetime import datetime +import time import aiohttp @@ -180,12 +180,11 @@ async def _async_main() -> int: if not validate_credentials(queries, crawler_registry, crawler_logins): return 1 - start_time = datetime.now() + start_time = time.monotonic() results = await run_queries( queries, crawler_registry, crawler_logins, args.no_progress ) - end_time = datetime.now() - total_duration = (end_time - start_time).total_seconds() + total_duration = time.monotonic() - start_time return print_report( results, args.show_problems, total_duration, json_output=args.json diff --git a/src/ojhunt/core/runner.py b/src/ojhunt/core/runner.py index 1aabb18..5bab4fa 100644 --- a/src/ojhunt/core/runner.py +++ b/src/ojhunt/core/runner.py @@ -2,7 +2,7 @@ Crawler execution logic. """ -from datetime import datetime +import time from typing import Any import aiohttp @@ -25,10 +25,10 @@ async def run_crawler( Returns: QueryResult with success status and data or error """ - start_time = datetime.now() + start_time = time.monotonic() try: result = await crawler.query(client, username, **kwargs) - duration = (datetime.now() - start_time).total_seconds() + duration = time.monotonic() - start_time return QueryResult( crawler=crawler, username=username, From 3c6258bb588320ba6ea9df04c2dd6407d024358b Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:42:21 +1000 Subject: [PATCH 07/10] fix(web): label rendered timestamps with their timezone MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both timestamps the app renders were naive, so they showed whatever zone the server happened to run in and said nothing about which one that was. A reader in Berlin saw a PDF stamped two hours behind their own clock with no way to tell why. The PDF footer now uses the reader's zone. PdfSnapshot.timezone already carries it, and the day keys in the chart already respect it via compute_day_key() — the footer was the one part of the report that ignored it. The ZoneInfo lookup and its UTC fallback move into _resolve_tz() so both callers share them. /about has no reader zone available, because it is rendered server-side with no JS, so its build time is explicit UTC and labelled as such. Verified by rendering: Europe/Berlin gives "2026-08-11 17:39 CEST", Asia/Shanghai "23:39 CST", an unknown zone falls back to "15:39 UTC", and /about with BUILD_TIME set gives "2025-08-11 08:13:20 UTC". --- src/ojhunt/web/pages.py | 8 ++++++-- src/ojhunt/web/pdf.py | 19 +++++++++++++------ 2 files changed, 19 insertions(+), 8 deletions(-) diff --git a/src/ojhunt/web/pages.py b/src/ojhunt/web/pages.py index 1b0ce8a..5ce9dde 100644 --- a/src/ojhunt/web/pages.py +++ b/src/ojhunt/web/pages.py @@ -5,7 +5,7 @@ import os import random import secrets -from datetime import datetime +from datetime import UTC, datetime from pathlib import Path from fastapi import APIRouter, File, Form, Request, UploadFile @@ -104,7 +104,11 @@ async def about() -> str: if BUILD_TIME: try: ts = int(BUILD_TIME) - build_time_str = datetime.fromtimestamp(ts).strftime("%Y-%m-%d %H:%M:%S") + # UTC, and said so: this page is rendered server-side, so the reader's + # zone is not available here. + build_time_str = datetime.fromtimestamp(ts, UTC).strftime( + "%Y-%m-%d %H:%M:%S UTC" + ) except ValueError: build_time_str = BUILD_TIME return template.render( diff --git a/src/ojhunt/web/pdf.py b/src/ojhunt/web/pdf.py index f60fc8a..117c459 100644 --- a/src/ojhunt/web/pdf.py +++ b/src/ojhunt/web/pdf.py @@ -112,17 +112,21 @@ class PdfEmbeddedData(BaseModel): history: list[HistoryEntry] +def _resolve_tz(iana_timezone: str) -> ZoneInfo: + """Return the named zone, falling back to UTC when the host lacks tzdata.""" + try: + return ZoneInfo(iana_timezone) + except ZoneInfoNotFoundError: + return ZoneInfo("UTC") + + def compute_day_key(iana_timezone: str) -> str: """Return the current YYYY-MM-DD day key using the user's local timezone. The day resets at 4am (not midnight) to avoid splitting a late-night session across two calendar days. """ - try: - tz = ZoneInfo(iana_timezone) - except ZoneInfoNotFoundError: - tz = ZoneInfo("UTC") - now = datetime.now(tz) + now = datetime.now(_resolve_tz(iana_timezone)) if now.hour < 4: now -= timedelta(days=1) return now.strftime("%Y-%m-%d") @@ -271,7 +275,10 @@ def generate_pdf( pdf.set_font(_FONT, "", 9) pdf.set_text_color(100, 100, 100) - generated_at = datetime.now().strftime("%Y-%m-%d %H:%M") + # The reader's zone, not the server's: the day keys in the chart below already + # use it, so the footer would otherwise disagree with the rest of the report. + generated_tz = _resolve_tz(snapshot.timezone) + generated_at = datetime.now(generated_tz).strftime("%Y-%m-%d %H:%M %Z") pdf.cell( 0, 5, From 613eec22c8606dd9a5279ef03eb46a897593ef91 Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:46:02 +1000 Subject: [PATCH 08/10] refactor: clear the remaining ruff 0.16 findings The last ten findings, none of which ruff can fix safely: - crawlers/__init__.py imported List and Union that nothing uses. The autofix rewrote the string annotation on query_sync() but left the import, because it cannot prove a string annotation is the only reference (F401, UP035). - close_http_client() declared `global _CLIENT` but only reads it, so the statement did nothing (PLW0602). init_http_client() keeps its own, which does assign. - Three tests asserted by evaluating a bare attribute inside pytest.raises. The access is the assertion, so binding the result to `_` keeps the intent and satisfies B018 without a suppression comment. - Three temp PDFs used NamedTemporaryFile(delete=False), which SIM115 reads as a leaked handle. _write_temp_pdf() already existed for exactly this, so it now uses mkstemp with a context manager and the other two call sites route through it instead of repeating the pattern. ./doit.sh lint passes. --- src/ojhunt/crawlers/__init__.py | 2 +- src/ojhunt/web/http_client.py | 1 - tests/core/models_test.py | 4 ++-- tests/crawlers/registry_test.py | 2 +- tests/e2e/test_pdf_workflow.py | 30 +++++++++++++++--------------- 5 files changed, 19 insertions(+), 20 deletions(-) diff --git a/src/ojhunt/crawlers/__init__.py b/src/ojhunt/crawlers/__init__.py index 1ba381c..da085c4 100644 --- a/src/ojhunt/crawlers/__init__.py +++ b/src/ojhunt/crawlers/__init__.py @@ -117,7 +117,7 @@ async def main(): from collections.abc import Awaitable, Callable from functools import cache from pathlib import Path -from typing import TYPE_CHECKING, Any, List, Union +from typing import TYPE_CHECKING, Any from ojhunt.core.models import ( CrawlerInfo, diff --git a/src/ojhunt/web/http_client.py b/src/ojhunt/web/http_client.py index 79d102d..aee312f 100644 --- a/src/ojhunt/web/http_client.py +++ b/src/ojhunt/web/http_client.py @@ -22,7 +22,6 @@ async def get_http_client() -> aiohttp.ClientSession: async def close_http_client() -> None: - global _CLIENT await _CLIENT.close() diff --git a/tests/core/models_test.py b/tests/core/models_test.py index 83c30f6..3341cdd 100644 --- a/tests/core/models_test.py +++ b/tests/core/models_test.py @@ -70,7 +70,7 @@ def test_unknown_attribute_raises_attribute_error(): registry = _registry("aizu") with pytest.raises(AttributeError, match="no crawler named 'nope'"): - registry.nope + _ = registry.nope def test_a_near_miss_carries_the_data_python_suggests_from(): @@ -78,7 +78,7 @@ def test_a_near_miss_carries_the_data_python_suggests_from(): registry = _registry("codeforces") with pytest.raises(AttributeError) as excinfo: - registry.codefroces + _ = registry.codefroces assert excinfo.value.name == "codefroces" assert "codeforces" in dir(excinfo.value.obj) diff --git a/tests/crawlers/registry_test.py b/tests/crawlers/registry_test.py index 465e4bc..3c6a822 100644 --- a/tests/crawlers/registry_test.py +++ b/tests/crawlers/registry_test.py @@ -49,7 +49,7 @@ def test_repeated_access_returns_the_same_registry(): def test_unknown_module_attribute_raises_attribute_error(): with pytest.raises(AttributeError, match="has no attribute 'nope'"): - ojhunt.crawlers.nope + _ = ojhunt.crawlers.nope def test_dir_advertises_crawlers(): diff --git a/tests/e2e/test_pdf_workflow.py b/tests/e2e/test_pdf_workflow.py index de45b40..a229e49 100644 --- a/tests/e2e/test_pdf_workflow.py +++ b/tests/e2e/test_pdf_workflow.py @@ -66,11 +66,9 @@ def historical_pdf_path() -> str: ) pdf_bytes = generate_pdf(settings, history, snapshot) - tmp = tempfile.NamedTemporaryFile(suffix=".pdf", delete=False, dir=_TMPDIR) - tmp.write(pdf_bytes) - tmp.close() - yield tmp.name - os.unlink(tmp.name) + path = _write_temp_pdf(pdf_bytes) + yield path + os.unlink(path) # --------------------------------------------------------------------------- @@ -100,11 +98,15 @@ def _make_ojhunt_pdf(context: BrowserContext, username: str = "tourist") -> byte def _write_temp_pdf(pdf_bytes: bytes) -> str: - """Write PDF bytes to a temp file and return the path.""" - tmp = tempfile.NamedTemporaryFile(suffix=".pdf", delete=False, dir=_TMPDIR) - tmp.write(pdf_bytes) - tmp.close() - return tmp.name + """Write bytes to a uniquely named temp file and return the path. + + The caller unlinks it. Playwright needs a real path on disk, so the file + outlives this function and cannot use NamedTemporaryFile's cleanup. + """ + fd, path = tempfile.mkstemp(suffix=".pdf", dir=_TMPDIR) + with os.fdopen(fd, "wb") as handle: + handle.write(pdf_bytes) + return path def _drag_drop_pdf(page: Page, pdf_bytes: bytes, filename: str = "report.pdf") -> None: @@ -200,21 +202,19 @@ def test_upload_pdf_shows_info_when_queries_exist(page: Page, context: BrowserCo @pytest.mark.playwright def test_upload_invalid_file_shows_error(page: Page): """Uploading a non-OJHunt file shows an alert.""" - tmp = tempfile.NamedTemporaryFile(suffix=".pdf", delete=False, dir=_TMPDIR) - tmp.write(b"not a real pdf at all") - tmp.close() + pdf_path = _write_temp_pdf(b"not a real pdf at all") try: page.goto(BASE_URL) _clear_storage(page) dialog_messages = [] page.on("dialog", lambda d: (dialog_messages.append(d.message), d.accept())) - page.set_input_files("input[type='file']", tmp.name) + page.set_input_files("input[type='file']", pdf_path) page.wait_for_timeout(3000) assert any("PDF" in m or "pdf" in m for m in dialog_messages) finally: - os.unlink(tmp.name) + os.unlink(pdf_path) @pytest.mark.playwright From c272d57915e63ae5a6584ea2d17328f437b24779 Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:47:33 +1000 Subject: [PATCH 09/10] docs: adopt the PEP 585/604 typing convention MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit docs/dev/python.md told contributors the opposite of what the code now does: "Use Dict, List, Union from the typing module". Ruff 0.16's defaults rewrite exactly that, and the format-lint hook applies the rewrite on every edit, so the old rule was both wrong and unenforceable. Two sections change: - Typing states the built-in generics and | unions, names the four rules that enforce it, and points at ADR 0016 for why it reversed. - Import order was wrong independently of the bump. It read "standard library → third-party → typing", but typing and collections.abc *are* standard library and I001 sorts them into that first group. The three query() templates in docs/dev/crawlers.md get the same treatment. They are copied verbatim into new crawler files, so a stale template would have reintroduced the old style on the next crawler. Also records the eolymp GraphQL follow-up in docs/BACKLOG.md: passing the username as a query variable would drop both the manual quote escaping and the # noqa: UP031, but it changes the request payload and needs its own network verification. --- docs/BACKLOG.md | 10 ++++++++++ docs/dev/crawlers.md | 18 ++++++++---------- docs/dev/python.md | 17 ++++++++++++++--- 3 files changed, 32 insertions(+), 13 deletions(-) diff --git a/docs/BACKLOG.md b/docs/BACKLOG.md index 96e169d..1f0b3e7 100644 --- a/docs/BACKLOG.md +++ b/docs/BACKLOG.md @@ -53,3 +53,13 @@ texts still need editing by hand. Option: put `explanation` and `credential_args` next to `LoginType.label` and generate the paragraphs from there. + +## Eolymp interpolates the username into its GraphQL query text + +`src/ojhunt/crawlers/eolymp.py` builds its query with `%`-formatting and escapes the username by +hand (`username.replace('"', '\\"')`). GraphQL variables are the right mechanism: pass the query +with a `$search` parameter and send the value in the request's `variables` object. That removes +the manual escaping and the `# noqa: UP031`, because no brace has to survive a format call. + +Left alone because it changes the request payload, so it needs its own network verification +against api.eolymp.com rather than a drive-by in a lint sweep. diff --git a/docs/dev/crawlers.md b/docs/dev/crawlers.md index bcafeea..07b5980 100644 --- a/docs/dev/crawlers.md +++ b/docs/dev/crawlers.md @@ -73,7 +73,6 @@ without it. The module docstring is not available for this — it holds the lice # (copy the full header from an existing crawler) import aiohttp -from typing import Dict, List, Optional, Union __crawler_meta__ = { "title": "OJ Name", @@ -84,8 +83,8 @@ __crawler_meta__ = { async def query( - session: aiohttp.ClientSession, username: str, password: Optional[str] = None -) -> Dict[str, Union[int, List[str], None]]: + session: aiohttp.ClientSession, username: str, password: str | None = None +) -> dict[str, int | list[str] | None]: """Query OJ Name for user statistics. Args: @@ -129,7 +128,6 @@ async def query( import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Optional, Union __crawler_meta__ = { "title": "Your OJ", @@ -140,8 +138,8 @@ __crawler_meta__ = { async def query( - session: aiohttp.ClientSession, username: str, password: Optional[str] = None -) -> Dict[str, Union[int, List[str], None]]: + session: aiohttp.ClientSession, username: str, password: str | None = None +) -> dict[str, int | list[str] | None]: """Query Your OJ for user statistics. Args: @@ -193,10 +191,10 @@ __crawler_meta__ = { async def query( session: aiohttp.ClientSession, username: str, - password: Optional[str] = None, - login_user: Optional[str] = None, - login_password: Optional[str] = None, -) -> Dict[str, Union[int, List[str], None]]: + password: str | None = None, + login_user: str | None = None, + login_password: str | None = None, +) -> dict[str, int | list[str] | None]: """Query Your OJ for user statistics. Your OJ hides profiles from guests, so a login is always required. Any account diff --git a/docs/dev/python.md b/docs/dev/python.md index 766fe74..9489cc1 100644 --- a/docs/dev/python.md +++ b/docs/dev/python.md @@ -2,15 +2,21 @@ ## Import order -Standard library → third-party → typing: +Standard library → third-party → first-party, with a blank line between the groups. +`typing` and `collections.abc` are standard library, so they belong in the first group: ```python notest import re +from collections.abc import Callable + import aiohttp from selectolax.lexbor import LexborHTMLParser -from typing import Dict, List, Union + +from ojhunt.core.models import CrawlerResult ``` +Ruff enforces this order (`I001`) and fixes it, so you do not have to sort by hand. + ## Naming - Files: `snake_case.py` (crawlers), `*_test.py` (tests) @@ -19,7 +25,12 @@ from typing import Dict, List, Union ## Typing -Use `Dict`, `List`, `Union` from the `typing` module — not the `dict[str, ...]` syntax. +Use built-in generics and the `|` union syntax: `dict[str, int]`, `list[str]`, `str | None`. +Do not use `Dict`, `List`, `Optional` or `Union` from `typing` — ruff rewrites them +(`UP006`, `UP007`, `UP035`, `UP045`), and the `format-lint-python.sh` hook applies that +rewrite on every edit. Take `Callable` and `Awaitable` from `collections.abc`, not `typing`. + +See [ADR 0016](../adr/0016-adopt-ruff-default-rule-set.md) for why this reversed. ## Prefer asserts and names over comments From 66f1321ff195032ca7b61bbf8ddccec00198c5c7 Mon Sep 17 00:00:00 2001 From: Shumin Liu Date: Wed, 12 Aug 2026 01:48:40 +1000 Subject: [PATCH 10/10] build: gate the formatter in ./doit.sh lint The lint task ran `ruff check` only, so `ruff format` drift never failed CI. Ruff 0.16 made that gap visible: it formats python blocks inside markdown, and four documentation files had drifted without any check noticing. Both passes now run unconditionally rather than short-circuiting on the first failure, so one `.doit/lint.log` lists every problem instead of hiding the formatter behind the linter. Verified by injecting a badly formatted line into a markdown code block: the task exits 1 and the log shows the diff. --- docs/adr/0016-adopt-ruff-default-rule-set.md | 5 +++-- doit.sh | 13 +++++++++++-- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/docs/adr/0016-adopt-ruff-default-rule-set.md b/docs/adr/0016-adopt-ruff-default-rule-set.md index afc6dd0..acdef0c 100644 --- a/docs/adr/0016-adopt-ruff-default-rule-set.md +++ b/docs/adr/0016-adopt-ruff-default-rule-set.md @@ -64,5 +64,6 @@ recorded in config, and three at the site: back gets rewritten by the `format-lint-python.sh` hook on the next edit. - The 413-rule set covers `SIM`, `C4`, `B`, `DTZ`, `RUF`, `PL` and more, so new code meets checks that were never applied to the code already in the tree. -- `./doit.sh lint` runs `ruff check` only. Ruff 0.16 also formats Python blocks inside - Markdown, and that drift stays invisible to CI until `ruff format --check` joins the task. +- `./doit.sh lint` runs `ruff format --check` beside `ruff check`, because ruff 0.16 also + formats Python blocks inside Markdown and that drift was invisible to a check-only gate. + Both passes always run, so one report lists every problem. diff --git a/doit.sh b/doit.sh index f097275..93b6590 100755 --- a/doit.sh +++ b/doit.sh @@ -205,7 +205,16 @@ status() { logs() { tail -F "$SERVER_LOG"; } lint() { - run_logged lint uv run ruff check . + run_logged lint _ruff_check_and_format +} + +# Both ruff passes always run, so one report lists every problem. Ruff 0.16 also +# formats python blocks inside markdown, which `ruff check` alone does not see. +_ruff_check_and_format() { + local status=0 + uv run ruff check . || status=1 + uv run ruff format --check . || status=1 + return "$status" } gen-docs() { @@ -300,7 +309,7 @@ Server: reap kill orphaned servers whose git worktree has been removed Tests: - lint run ruff linter + lint run ruff linter and formatter check test-unit run unit tests (no network, no playwright) [pytest-args...] test-e2e run e2e tests excluding visual (starts server if needed) [pytest-args...] test-visual run visual regression tests (starts server if needed) [pytest-args...]