From 18470412dfddbc8cbf4e622c913dddcffa232ee8 Mon Sep 17 00:00:00 2001 From: evanlowe <62918515+evanlowe@users.noreply.github.com> Date: Fri, 21 Aug 2026 01:13:57 +0800 Subject: [PATCH 1/6] feat(studio): add scheduled runtime tasks --- frontend/README.md | 24 + frontend/server/cronjobs/__init__.py | 59 ++ frontend/server/cronjobs/repository.py | 643 ++++++++++++++++++ frontend/server/cronjobs/routes.py | 232 +++++++ frontend/server/cronjobs/schemas.py | 250 +++++++ frontend/server/cronjobs/service.py | 340 +++++++++ frontend/service/studio_scheduler/__init__.py | 37 + frontend/service/studio_scheduler/app.py | 141 ++++ frontend/service/studio_scheduler/deploy.py | 322 +++++++++ .../service/studio_scheduler/diagnostics.py | 63 ++ .../service/studio_scheduler/dispatcher.py | 366 ++++++++++ .../service/studio_scheduler/entrypoint.py | 40 ++ frontend/service/studio_scheduler/executor.py | 42 ++ frontend/service/studio_scheduler/http_app.py | 45 ++ .../service/studio_scheduler/local_runner.py | 70 ++ .../studio_scheduler/memory_repository.py | 157 +++++ frontend/service/studio_scheduler/models.py | 485 +++++++++++++ frontend/service/studio_scheduler/ports.py | 97 +++ .../service/studio_scheduler/publisher.py | 95 +++ .../studio_scheduler/runtime_provider.py | 474 +++++++++++++ frontend/service/studio_scheduler/schedule.py | 133 ++++ .../studio_scheduler/tos_repository.py | 457 +++++++++++++ frontend/src/App.tsx | 92 ++- frontend/src/adk/client.ts | 183 +++++ frontend/src/cronjobs/CronJobs.css | 510 ++++++++++++++ frontend/src/cronjobs/CronJobs.tsx | 641 +++++++++++++++++ frontend/src/cronjobs/icons.tsx | 76 +++ frontend/src/cronjobs/model.ts | 59 ++ frontend/src/styles.css | 12 + frontend/src/ui/Sidebar.css | 8 + frontend/src/ui/Sidebar.tsx | 35 + frontend/src/ui/StudioConfirmDialog.tsx | 3 + frontend/tests/cronjobs.test.mjs | 129 ++++ frontend/tests/documentTitle.test.mjs | 1 + pyproject.toml | 8 +- tests/cli/test_studio_deploy_permissions.py | 11 +- tests/cli/test_studio_deploy_target.py | 5 + tests/cli/test_studio_self_update.py | 33 + tests/cli/test_studio_update.py | 67 ++ .../frontend/server/cronjobs/test_cronjobs.py | 521 ++++++++++++++ .../server/test_studio_update_resources.py | 3 +- .../service/studio_scheduler/test_deploy.py | 162 +++++ .../studio_scheduler/test_diagnostics.py | 27 + .../studio_scheduler/test_dispatcher.py | 362 ++++++++++ .../studio_scheduler/test_end_to_end.py | 163 +++++ .../studio_scheduler/test_local_entrypoint.py | 94 +++ .../studio_scheduler/test_local_runner.py | 49 ++ .../studio_scheduler/test_runtime_provider.py | 260 +++++++ .../service/studio_scheduler/test_schedule.py | 88 +++ .../studio_scheduler/test_tos_repository.py | 210 ++++++ veadk/cli/cli_frontend.py | 159 +++++ veadk/cli/studio_deploy_permissions.py | 25 + veadk/cli/studio_package.py | 11 + veadk/cli/studio_self_update.py | 19 + .../{index-sh8QRNzm.js => index-71FSqnQe.js} | 576 ++++++++-------- ...ga.js => MarkdownPromptEditor-BOGec8fb.js} | 2 +- .../{arc-B7D7fKkF.js => arc-Btz4jl6k.js} | 2 +- ...hannel-Cip7x_1m.js => channel-BajLYa9k.js} | 2 +- ...{linear-BmzcZuEF.js => linear-fpsK4bUg.js} | 2 +- veadk/webui/assets/styles/index-DezGLL37.css | 10 + veadk/webui/assets/styles/index-s77xcnYE.css | 10 - ...j5.js => abnfDiagram-N423BO3Z-BsWqgl3v.js} | 2 +- ... architectureDiagram-T3A2C74G-2M-OpiGy.js} | 2 +- ...-.js => blockDiagram-VBNYF7ZC-BrlTwsjf.js} | 2 +- ...rM0V.js => c4Diagram-5PPSVZJV-CiWX4QWA.js} | 2 +- ...8JeziMcZ.js => chunk-2GRJ4B5K-8rhZ018R.js} | 2 +- ...DyH4ylUc.js => chunk-2Q5K7J3B-CIi5pcr7.js} | 2 +- ...B8NYulM6.js => chunk-5RXB4S5H-Cksc9APc.js} | 2 +- ...DW5_javj.js => chunk-5VM5RSS4-DQSNWk-X.js} | 2 +- ...LmTMnw7Y.js => chunk-6Q2QTUOP-M05PcnGQ.js} | 2 +- ...CcKNV6UO.js => chunk-GF5L2VYU-D7DKTkLz.js} | 2 +- ...DKTpV34k.js => chunk-JWPE2WC7-DgyAVOEw.js} | 2 +- ...BJQBtkor.js => chunk-KBJHAD2P-BmOHOQh8.js} | 2 +- ...8wtkB9Bx.js => chunk-RYQCIY6F-CS40XceV.js} | 2 +- ...DqdKXcqv.js => chunk-XXDRQBXY-BUOYlznW.js} | 2 +- .../mermaid/classDiagram-JCYQIIEL-DN0k_PIr.js | 1 - .../mermaid/classDiagram-JCYQIIEL-DxvT2-3O.js | 1 + .../classDiagram-v2-OCEON4UE-DN0k_PIr.js | 1 - .../classDiagram-v2-OCEON4UE-DxvT2-3O.js | 1 + ...a.js => cose-bilkent-JH36ORCC-DkToF0RP.js} | 2 +- ...RcDGzI.js => cynefin-OW5HDTMX-DTsroo5L.js} | 2 +- ...js => cynefinDiagram-MW4NZA55-CGVg7KFL.js} | 2 +- ...jpR8BtJi.js => dagre-VZM6K2ZE-kyA3X9-_.js} | 2 +- ...W8pRlJ.js => diagram-7IWD3JNH-CivQIo97.js} | 2 +- ...-3nBVO.js => diagram-B4RE2ZJO-KP4U0Ep_.js} | 2 +- ...fT50JQ.js => diagram-LBJQPF4R-w8db3UDU.js} | 2 +- ...fRizrn.js => diagram-Q27KOJAE-BK5d491H.js} | 2 +- ...qkXTUU.js => diagram-UB23O5K3-DYfnKSTv.js} | 2 +- ...YD.js => ebnfDiagram-BXEA7PRR-BCUCRYN8.js} | 2 +- ...NU_j.js => erDiagram-JOGREHBK-CkeP9E68.js} | 2 +- ...XI.js => flowDiagram-UKHOOZJN-BurPe9an.js} | 2 +- ...4.js => ganttDiagram-PKOTCBZU-DmOFUlnV.js} | 2 +- ...s => gitGraphDiagram-DS77QQ5N-CWTvXYAU.js} | 2 +- ...PD.js => infoDiagram-6WML65LV-B187BzXO.js} | 2 +- ...s => ishikawaDiagram-WSZJBQD7-B3jQY7e1.js} | 2 +- ...js => journeyDiagram-NVQOT4AX-BIXSNdRR.js} | 2 +- ...=> kanban-definition-27J2QSJJ-BuRrlwZS.js} | 2 +- ...e-DAX2QfGO.js => mermaid.core-CtFGSx1J.js} | 8 +- ...> mindmap-definition-FAOFIHXS-6Ds6_4vd.js} | 2 +- ...Pgl.js => pegDiagram-VL7TDLO6-C32_1iR_.js} | 2 +- ...4cR.js => pieDiagram-7S7Q4E2Y-DFwiwRSS.js} | 2 +- ...s => quadrantDiagram-CIZ2JOQS-DK1qG3le.js} | 2 +- ...s => railroadDiagram-AXF67PYL-PRvupeUB.js} | 2 +- ...> requirementDiagram-LRYGKXZP-D_24PVr7.js} | 2 +- ....js => sankeyDiagram-W5VNT64P-C8Qhvokx.js} | 2 +- ...s => sequenceDiagram-SI44F4Z6-CXNouiEe.js} | 2 +- ...f3.js => sizeCapture-X5ZJPWSS-CMYsneXF.js} | 2 +- ...W.js => stateDiagram-OKZ733FA-DbzKUOce.js} | 2 +- .../stateDiagram-v2-UEYNNEHI-C6zkrfcd.js | 1 - .../stateDiagram-v2-UEYNNEHI-DJDgQJaK.js | 1 + ...YeZk.js => swimlanes-SLNWSIFB-DwNtO8DW.js} | 4 +- .../swimlanesDiagram-ULZ7WXOC-BI7vzbWq.js | 8 + .../swimlanesDiagram-ULZ7WXOC-CPpKyqwL.js | 8 - ... timeline-definition-Z64GVDOM-BMRcJkhU.js} | 2 +- ...lL.js => vennDiagram-T6HMQDX7-DGim9IIK.js} | 2 +- ...js => wardleyDiagram-T6FBY63Y-CcLuxGLj.js} | 2 +- ...js => xychartDiagram-ELKLHX3M-CZs-kR6j.js} | 2 +- veadk/webui/index.html | 4 +- 118 files changed, 8936 insertions(+), 396 deletions(-) create mode 100644 frontend/server/cronjobs/__init__.py create mode 100644 frontend/server/cronjobs/repository.py create mode 100644 frontend/server/cronjobs/routes.py create mode 100644 frontend/server/cronjobs/schemas.py create mode 100644 frontend/server/cronjobs/service.py create mode 100644 frontend/service/studio_scheduler/__init__.py create mode 100644 frontend/service/studio_scheduler/app.py create mode 100644 frontend/service/studio_scheduler/deploy.py create mode 100644 frontend/service/studio_scheduler/diagnostics.py create mode 100644 frontend/service/studio_scheduler/dispatcher.py create mode 100644 frontend/service/studio_scheduler/entrypoint.py create mode 100644 frontend/service/studio_scheduler/executor.py create mode 100644 frontend/service/studio_scheduler/http_app.py create mode 100644 frontend/service/studio_scheduler/local_runner.py create mode 100644 frontend/service/studio_scheduler/memory_repository.py create mode 100644 frontend/service/studio_scheduler/models.py create mode 100644 frontend/service/studio_scheduler/ports.py create mode 100644 frontend/service/studio_scheduler/publisher.py create mode 100644 frontend/service/studio_scheduler/runtime_provider.py create mode 100644 frontend/service/studio_scheduler/schedule.py create mode 100644 frontend/service/studio_scheduler/tos_repository.py create mode 100644 frontend/src/cronjobs/CronJobs.css create mode 100644 frontend/src/cronjobs/CronJobs.tsx create mode 100644 frontend/src/cronjobs/icons.tsx create mode 100644 frontend/src/cronjobs/model.ts create mode 100644 frontend/src/ui/Sidebar.css create mode 100644 frontend/tests/cronjobs.test.mjs create mode 100644 tests/frontend/server/cronjobs/test_cronjobs.py create mode 100644 tests/frontend/service/studio_scheduler/test_deploy.py create mode 100644 tests/frontend/service/studio_scheduler/test_diagnostics.py create mode 100644 tests/frontend/service/studio_scheduler/test_dispatcher.py create mode 100644 tests/frontend/service/studio_scheduler/test_end_to_end.py create mode 100644 tests/frontend/service/studio_scheduler/test_local_entrypoint.py create mode 100644 tests/frontend/service/studio_scheduler/test_local_runner.py create mode 100644 tests/frontend/service/studio_scheduler/test_runtime_provider.py create mode 100644 tests/frontend/service/studio_scheduler/test_schedule.py create mode 100644 tests/frontend/service/studio_scheduler/test_tos_repository.py rename veadk/webui/assets/app/{index-sh8QRNzm.js => index-71FSqnQe.js} (53%) rename veadk/webui/assets/chunks/{MarkdownPromptEditor-9wo3f6ga.js => MarkdownPromptEditor-BOGec8fb.js} (99%) rename veadk/webui/assets/chunks/{arc-B7D7fKkF.js => arc-Btz4jl6k.js} (98%) rename veadk/webui/assets/chunks/{channel-Cip7x_1m.js => channel-BajLYa9k.js} (53%) rename veadk/webui/assets/chunks/{linear-BmzcZuEF.js => linear-fpsK4bUg.js} (98%) create mode 100644 veadk/webui/assets/styles/index-DezGLL37.css delete mode 100644 veadk/webui/assets/styles/index-s77xcnYE.css rename veadk/webui/assets/visualizations/mermaid/{abnfDiagram-N423BO3Z-Cud6OJj5.js => abnfDiagram-N423BO3Z-BsWqgl3v.js} (85%) rename veadk/webui/assets/visualizations/mermaid/{architectureDiagram-T3A2C74G-fRh8vhJG.js => architectureDiagram-T3A2C74G-2M-OpiGy.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{blockDiagram-VBNYF7ZC-gakGyyJ-.js => blockDiagram-VBNYF7ZC-BrlTwsjf.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{c4Diagram-5PPSVZJV-BouUrM0V.js => c4Diagram-5PPSVZJV-CiWX4QWA.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{chunk-2GRJ4B5K-8JeziMcZ.js => chunk-2GRJ4B5K-8rhZ018R.js} (93%) rename veadk/webui/assets/visualizations/mermaid/{chunk-2Q5K7J3B-DyH4ylUc.js => chunk-2Q5K7J3B-CIi5pcr7.js} (66%) rename veadk/webui/assets/visualizations/mermaid/{chunk-5RXB4S5H-B8NYulM6.js => chunk-5RXB4S5H-Cksc9APc.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{chunk-5VM5RSS4-DW5_javj.js => chunk-5VM5RSS4-DQSNWk-X.js} (83%) rename veadk/webui/assets/visualizations/mermaid/{chunk-6Q2QTUOP-LmTMnw7Y.js => chunk-6Q2QTUOP-M05PcnGQ.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{chunk-GF5L2VYU-CcKNV6UO.js => chunk-GF5L2VYU-D7DKTkLz.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{chunk-JWPE2WC7-DKTpV34k.js => chunk-JWPE2WC7-DgyAVOEw.js} (78%) rename veadk/webui/assets/visualizations/mermaid/{chunk-KBJHAD2P-BJQBtkor.js => chunk-KBJHAD2P-BmOHOQh8.js} (87%) rename veadk/webui/assets/visualizations/mermaid/{chunk-RYQCIY6F-8wtkB9Bx.js => chunk-RYQCIY6F-CS40XceV.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{chunk-XXDRQBXY-DqdKXcqv.js => chunk-XXDRQBXY-BUOYlznW.js} (52%) delete mode 100644 veadk/webui/assets/visualizations/mermaid/classDiagram-JCYQIIEL-DN0k_PIr.js create mode 100644 veadk/webui/assets/visualizations/mermaid/classDiagram-JCYQIIEL-DxvT2-3O.js delete mode 100644 veadk/webui/assets/visualizations/mermaid/classDiagram-v2-OCEON4UE-DN0k_PIr.js create mode 100644 veadk/webui/assets/visualizations/mermaid/classDiagram-v2-OCEON4UE-DxvT2-3O.js rename veadk/webui/assets/visualizations/mermaid/{cose-bilkent-JH36ORCC-CPiC1B6a.js => cose-bilkent-JH36ORCC-DkToF0RP.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{cynefin-OW5HDTMX-CVRcDGzI.js => cynefin-OW5HDTMX-DTsroo5L.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{cynefinDiagram-MW4NZA55-oYGksW2C.js => cynefinDiagram-MW4NZA55-CGVg7KFL.js} (98%) rename veadk/webui/assets/visualizations/mermaid/{dagre-VZM6K2ZE-jpR8BtJi.js => dagre-VZM6K2ZE-kyA3X9-_.js} (97%) rename veadk/webui/assets/visualizations/mermaid/{diagram-7IWD3JNH-8KW8pRlJ.js => diagram-7IWD3JNH-CivQIo97.js} (96%) rename veadk/webui/assets/visualizations/mermaid/{diagram-B4RE2ZJO-Zn-3nBVO.js => diagram-B4RE2ZJO-KP4U0Ep_.js} (97%) rename veadk/webui/assets/visualizations/mermaid/{diagram-LBJQPF4R-sifT50JQ.js => diagram-LBJQPF4R-w8db3UDU.js} (93%) rename veadk/webui/assets/visualizations/mermaid/{diagram-Q27KOJAE-D9fRizrn.js => diagram-Q27KOJAE-BK5d491H.js} (98%) rename veadk/webui/assets/visualizations/mermaid/{diagram-UB23O5K3-CgqkXTUU.js => diagram-UB23O5K3-DYfnKSTv.js} (95%) rename veadk/webui/assets/visualizations/mermaid/{ebnfDiagram-BXEA7PRR-_hmiJ3YD.js => ebnfDiagram-BXEA7PRR-BCUCRYN8.js} (87%) rename veadk/webui/assets/visualizations/mermaid/{erDiagram-JOGREHBK-Dj7zNU_j.js => erDiagram-JOGREHBK-CkeP9E68.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{flowDiagram-UKHOOZJN-2fX5nZXI.js => flowDiagram-UKHOOZJN-BurPe9an.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{ganttDiagram-PKOTCBZU-X7kwbkc4.js => ganttDiagram-PKOTCBZU-DmOFUlnV.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{gitGraphDiagram-DS77QQ5N-RzHaRRjn.js => gitGraphDiagram-DS77QQ5N-CWTvXYAU.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{infoDiagram-6WML65LV-NV32ThPD.js => infoDiagram-6WML65LV-B187BzXO.js} (67%) rename veadk/webui/assets/visualizations/mermaid/{ishikawaDiagram-WSZJBQD7-DcfYA8UA.js => ishikawaDiagram-WSZJBQD7-B3jQY7e1.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{journeyDiagram-NVQOT4AX-DpRdtcYa.js => journeyDiagram-NVQOT4AX-BIXSNdRR.js} (98%) rename veadk/webui/assets/visualizations/mermaid/{kanban-definition-27J2QSJJ-T-j7ARpv.js => kanban-definition-27J2QSJJ-BuRrlwZS.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{mermaid.core-DAX2QfGO.js => mermaid.core-CtFGSx1J.js} (98%) rename veadk/webui/assets/visualizations/mermaid/{mindmap-definition-FAOFIHXS-dSs8oJom.js => mindmap-definition-FAOFIHXS-6Ds6_4vd.js} (98%) rename veadk/webui/assets/visualizations/mermaid/{pegDiagram-VL7TDLO6-Db2QQPgl.js => pegDiagram-VL7TDLO6-C32_1iR_.js} (86%) rename veadk/webui/assets/visualizations/mermaid/{pieDiagram-7S7Q4E2Y-b-0AZ4cR.js => pieDiagram-7S7Q4E2Y-DFwiwRSS.js} (95%) rename veadk/webui/assets/visualizations/mermaid/{quadrantDiagram-CIZ2JOQS-DnyqEKV8.js => quadrantDiagram-CIZ2JOQS-DK1qG3le.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{railroadDiagram-AXF67PYL-UgdXcxSH.js => railroadDiagram-AXF67PYL-PRvupeUB.js} (84%) rename veadk/webui/assets/visualizations/mermaid/{requirementDiagram-LRYGKXZP-CVhjvLne.js => requirementDiagram-LRYGKXZP-D_24PVr7.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{sankeyDiagram-W5VNT64P-gmgsn_hR.js => sankeyDiagram-W5VNT64P-C8Qhvokx.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{sequenceDiagram-SI44F4Z6-DvRxUXnN.js => sequenceDiagram-SI44F4Z6-CXNouiEe.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{sizeCapture-X5ZJPWSS-dTWDZWf3.js => sizeCapture-X5ZJPWSS-CMYsneXF.js} (87%) rename veadk/webui/assets/visualizations/mermaid/{stateDiagram-OKZ733FA-B-xg4vQW.js => stateDiagram-OKZ733FA-DbzKUOce.js} (96%) delete mode 100644 veadk/webui/assets/visualizations/mermaid/stateDiagram-v2-UEYNNEHI-C6zkrfcd.js create mode 100644 veadk/webui/assets/visualizations/mermaid/stateDiagram-v2-UEYNNEHI-DJDgQJaK.js rename veadk/webui/assets/visualizations/mermaid/{swimlanes-SLNWSIFB-DG4EYeZk.js => swimlanes-SLNWSIFB-DwNtO8DW.js} (99%) create mode 100644 veadk/webui/assets/visualizations/mermaid/swimlanesDiagram-ULZ7WXOC-BI7vzbWq.js delete mode 100644 veadk/webui/assets/visualizations/mermaid/swimlanesDiagram-ULZ7WXOC-CPpKyqwL.js rename veadk/webui/assets/visualizations/mermaid/{timeline-definition-Z64GVDOM-DwO_fygu.js => timeline-definition-Z64GVDOM-BMRcJkhU.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{vennDiagram-T6HMQDX7-Dsch8ylL.js => vennDiagram-T6HMQDX7-DGim9IIK.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{wardleyDiagram-T6FBY63Y-D4DBl3bi.js => wardleyDiagram-T6FBY63Y-CcLuxGLj.js} (99%) rename veadk/webui/assets/visualizations/mermaid/{xychartDiagram-ELKLHX3M-D97HCDoV.js => xychartDiagram-ELKLHX3M-CZs-kR6j.js} (99%) diff --git a/frontend/README.md b/frontend/README.md index b17c6901b..0eea5a372 100644 --- a/frontend/README.md +++ b/frontend/README.md @@ -612,6 +612,30 @@ veadk studio deploy \ --vefaas-app-name ``` +## Scheduled tasks + +The `定时任务` workspace runs a fixed text prompt on a selected deployed Runtime +Agent. Each occurrence creates an independent Agent session. Schedules support +one-time, daily, weekly, and five-field Cron expressions with an IANA timezone. + +Definitions, locks, execution history, and results live in the private Studio +TOS bucket under `veadk-studio/v1/users/{user_id}/cronjobs/{job_id}`. A derived +minute index lives under `veadk-studio/v1/scheduler/cronjobs/due/{yyyyMMddHHmm}` +so the scheduler never scans user namespaces. + +`veadk studio deploy` also creates or updates a separate stateless VeFaaS +scheduler function and a one-minute cloud timer. Duplicate timer deliveries are +deduplicated with immutable run IDs and TOS conditional writes; an ETag lock +prevents concurrent executions of the same task across Studio replicas. The +scheduler uses the function IAM role to read the Runtime's current endpoint and +version. It does not store user tokens or AK/SK credentials. + +Manual runs are persisted with a `queued` status and placed in the next minute's +due bucket, so they normally start within 60 seconds. This avoids losing a run +when the current minute has already been scanned. When Studio is started with +`veadk studio --vite`, the BFF also starts a local minute scheduler; no separate +local scheduler process is required. + ## Agent usage statistics The `用量统计` tab on a deployed Agent records one invocation after a Studio diff --git a/frontend/server/cronjobs/__init__.py b/frontend/server/cronjobs/__init__.py new file mode 100644 index 000000000..d960447b7 --- /dev/null +++ b/frontend/server/cronjobs/__init__.py @@ -0,0 +1,59 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Durable Studio cronjob management domain.""" + +from .repository import ( + CronjobConflict, + CronjobNotFound, + Stored, + TosCronjobRepository, +) +from .routes import mount_routes +from .schemas import ( + CreateCronjobRequest, + Cronjob, + CronjobIdentity, + CronjobLock, + CronjobRun, + UpdateCronjobRequest, +) +from .service import ( + CronjobAccessDenied, + CronjobAccessPolicy, + CronjobDuePublisher, + CronjobRunQueueUnavailable, + CronjobService, + OwnerOnlyAccessPolicy, +) + +__all__ = [ + "CreateCronjobRequest", + "Cronjob", + "CronjobAccessDenied", + "CronjobAccessPolicy", + "CronjobConflict", + "CronjobDuePublisher", + "CronjobIdentity", + "CronjobLock", + "CronjobNotFound", + "CronjobRun", + "CronjobRunQueueUnavailable", + "CronjobService", + "OwnerOnlyAccessPolicy", + "Stored", + "TosCronjobRepository", + "UpdateCronjobRequest", + "mount_routes", +] diff --git a/frontend/server/cronjobs/repository.py b/frontend/server/cronjobs/repository.py new file mode 100644 index 000000000..6fcbd92dc --- /dev/null +++ b/frontend/server/cronjobs/repository.py @@ -0,0 +1,643 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""TOS persistence for user-owned cronjobs, runs, and execution locks.""" + +from __future__ import annotations + +import asyncio +import json +from collections.abc import Callable +from dataclasses import dataclass +from datetime import datetime, timezone +from typing import Any, Generic, TypeVar, cast +from urllib.parse import quote +from zoneinfo import ZoneInfo + +from pydantic import BaseModel + +from frontend.server.storage import STUDIO_STORAGE_ROOT_PREFIX + +from .schemas import Cronjob, CronjobLock, CronjobRun + +_MAX_OBJECT_BYTES = 2 * 1024 * 1024 +_Value = TypeVar("_Value", bound=BaseModel) + + +class CronjobNotFound(LookupError): + """The requested owner-scoped object does not exist.""" + + +class CronjobConflict(RuntimeError): + """A conditional write failed or an active execution already exists.""" + + +@dataclass(frozen=True) +class Stored(Generic[_Value]): + value: _Value + etag: str + + +def _status_code(error: Exception) -> int | None: + value = getattr(error, "status_code", None) + try: + return int(value) if value is not None else None + except (TypeError, ValueError): + return None + + +class TosCronjobRepository: + """Keep TOS keys and CAS semantics outside business and HTTP layers.""" + + def __init__( + self, + *, + bucket: str, + client_factory: Callable[[], Any], + root_prefix: str = STUDIO_STORAGE_ROOT_PREFIX, + ) -> None: + if not bucket.strip(): + raise ValueError("TOS cronjob storage requires a bucket.") + self._bucket = bucket + self._client_factory = client_factory + self._root_prefix = root_prefix.strip("/") + + async def create_job(self, job: Cronjob) -> Stored[Cronjob]: + return await asyncio.to_thread(self._create_job, job) + + async def get_job(self, owner_id: str, job_id: str) -> Stored[Cronjob]: + return await asyncio.to_thread(self._get_job, owner_id, job_id) + + async def list_jobs(self, owner_id: str) -> list[Cronjob]: + return await asyncio.to_thread(self._list_jobs, owner_id) + + async def update_job(self, job: Cronjob, etag: str) -> Stored[Cronjob]: + return await asyncio.to_thread(self._update_job, job, etag) + + async def delete_job(self, owner_id: str, job_id: str) -> None: + await asyncio.to_thread(self._delete_job, owner_id, job_id) + + async def create_run(self, run: CronjobRun) -> Stored[CronjobRun]: + return await asyncio.to_thread(self._create_run, run) + + async def get_run( + self, owner_id: str, job_id: str, run_id: str + ) -> Stored[CronjobRun]: + return await asyncio.to_thread(self._get_run, owner_id, job_id, run_id) + + async def list_runs(self, owner_id: str, job_id: str) -> list[CronjobRun]: + return await asyncio.to_thread(self._list_runs, owner_id, job_id) + + async def update_run(self, run: CronjobRun, etag: str) -> Stored[CronjobRun]: + return await asyncio.to_thread(self._update_run, run, etag) + + async def request_cancel( + self, owner_id: str, job_id: str, run_id: str, requested_at: datetime + ) -> CronjobRun: + return await asyncio.to_thread( + self._request_cancel, owner_id, job_id, run_id, requested_at + ) + + async def get_lock(self, owner_id: str, job_id: str) -> Stored[CronjobLock] | None: + return await asyncio.to_thread(self._get_lock, owner_id, job_id) + + async def acquire_lock( + self, + owner_id: str, + job_id: str, + run_id: str, + now: datetime, + expires_at: datetime, + ) -> Stored[CronjobLock]: + return await asyncio.to_thread( + self._acquire_lock, owner_id, job_id, run_id, now, expires_at + ) + + async def release_lock( + self, owner_id: str, job_id: str, run_id: str, now: datetime + ) -> None: + await asyncio.to_thread(self._release_lock, owner_id, job_id, run_id, now) + + def scheduler_job_payload(self, job: Cronjob) -> dict[str, Any]: + """Return the exact durable job contract consumed by the dispatcher.""" + return self._job_payload(job) + + def _create_job(self, job: Cronjob) -> Stored[Cronjob]: + key = self._job_key(job.owner_id, job.id) + return self._create(self._client_factory(), key, job) + + def _get_job(self, owner_id: str, job_id: str) -> Stored[Cronjob]: + return self._read( + self._client_factory(), self._job_key(owner_id, job_id), Cronjob + ) + + def _list_jobs(self, owner_id: str) -> list[Cronjob]: + client = self._client_factory() + keys = self._list_keys(client, f"{self._owner_prefix(owner_id)}/") + jobs = [ + self._read(client, key, Cronjob).value + for key in keys + if key.endswith("/job.json") + ] + return sorted(jobs, key=lambda item: (item.created_at, item.id), reverse=True) + + def _update_job(self, job: Cronjob, etag: str) -> Stored[Cronjob]: + return self._replace( + self._client_factory(), self._job_key(job.owner_id, job.id), job, etag + ) + + def _delete_job(self, owner_id: str, job_id: str) -> None: + client = self._client_factory() + prefix = f"{self._job_prefix(owner_id, job_id)}/" + keys = self._list_keys(client, prefix) + if self._job_key(owner_id, job_id) not in keys: + raise CronjobNotFound("定时任务不存在或已被删除。") + for key in keys: + client.delete_object(bucket=self._bucket, key=key) + + def _create_run(self, run: CronjobRun) -> Stored[CronjobRun]: + return self._create( + self._client_factory(), + self._run_key(run.owner_id, run.job_id, run.id), + run, + ) + + def _get_run(self, owner_id: str, job_id: str, run_id: str) -> Stored[CronjobRun]: + return self._read( + self._client_factory(), + self._run_key(owner_id, job_id, run_id), + CronjobRun, + ) + + def _list_runs(self, owner_id: str, job_id: str) -> list[CronjobRun]: + client = self._client_factory() + prefix = f"{self._job_prefix(owner_id, job_id)}/runs/" + runs = [ + self._read(client, key, CronjobRun).value + for key in self._list_keys(client, prefix) + if key.endswith(".json") + ] + return sorted( + runs, + key=lambda item: (item.created_at or item.scheduled_at, item.id), + reverse=True, + ) + + def _update_run(self, run: CronjobRun, etag: str) -> Stored[CronjobRun]: + return self._replace( + self._client_factory(), + self._run_key(run.owner_id, run.job_id, run.id), + run, + etag, + ) + + def _request_cancel( + self, owner_id: str, job_id: str, run_id: str, requested_at: datetime + ) -> CronjobRun: + client = self._client_factory() + key = self._run_key(owner_id, job_id, run_id) + payload, etag = self._read_payload(client, key) + state = str(payload.get("state") or payload.get("status") or "") + if state in {"succeeded", "success", "failed", "cancelled", "skipped"}: + raise CronjobConflict("A completed cronjob run cannot be cancelled.") + if not payload.get("cancelRequested", False): + payload["cancelRequested"] = True + payload["updatedAt"] = self._iso(requested_at) + self._replace_payload(client, key, payload, etag) + return self._run_from_payload(payload) + + def _get_lock(self, owner_id: str, job_id: str) -> Stored[CronjobLock] | None: + try: + client = self._client_factory() + payload, etag = self._read_payload(client, self._lock_key(owner_id, job_id)) + acquired_at = payload.get("acquiredAt") + lock = CronjobLock.model_validate( + { + "jobId": job_id, + "ownerId": owner_id, + "runId": payload["runId"], + "state": payload["state"], + "acquiredAt": acquired_at, + "updatedAt": payload.get("releasedAt") or acquired_at, + "expiresAt": payload["expiresAt"], + } + ) + return Stored(value=lock, etag=etag) + except CronjobNotFound: + return None + + def _acquire_lock( + self, + owner_id: str, + job_id: str, + run_id: str, + now: datetime, + expires_at: datetime, + ) -> Stored[CronjobLock]: + client = self._client_factory() + key = self._lock_key(owner_id, job_id) + lock = CronjobLock( + jobId=job_id, + ownerId=owner_id, + runId=run_id, + state="held", + acquiredAt=now, + updatedAt=now, + expiresAt=expires_at, + ) + try: + return self._create(client, key, lock) + except CronjobConflict: + current = self._read(client, key, CronjobLock) + if current.value.active_at(now): + raise CronjobConflict("This cronjob already has an active run.") + return self._replace(client, key, lock, current.etag) + + def _release_lock( + self, owner_id: str, job_id: str, run_id: str, now: datetime + ) -> None: + current = self._get_lock(owner_id, job_id) + if current is None or current.value.state == "released": + return + if current.value.run_id != run_id: + raise CronjobConflict("Cronjob lock belongs to another run.") + released = current.value.model_copy( + update={"state": "released", "updated_at": now, "expires_at": now} + ) + self._replace( + self._client_factory(), + self._lock_key(owner_id, job_id), + released, + current.etag, + ) + + def _create(self, client: Any, key: str, value: _Value) -> Stored[_Value]: + content = self._encode(value) + try: + output = client.put_object( + bucket=self._bucket, + key=key, + content=content, + content_length=len(content), + content_type="application/json", + forbid_overwrite=True, + ) + except Exception as error: + if _status_code(error) in {409, 412}: + raise CronjobConflict("Cronjob object already exists.") from error + raise + return Stored(value=value, etag=self._write_etag(output, client, key)) + + def _replace( + self, client: Any, key: str, value: _Value, etag: str + ) -> Stored[_Value]: + content = self._encode(value) + try: + output = client.put_object( + bucket=self._bucket, + key=key, + content=content, + content_length=len(content), + content_type="application/json", + if_match=etag, + ) + except Exception as error: + if _status_code(error) in {409, 412}: + raise CronjobConflict( + "Cronjob changed concurrently; reload and try again." + ) from error + if _status_code(error) == 404: + raise CronjobNotFound("定时任务不存在或已被删除。") from error + raise + return Stored(value=value, etag=self._write_etag(output, client, key)) + + def _read(self, client: Any, key: str, model: type[_Value]) -> Stored[_Value]: + payload, etag = self._read_payload(client, key) + if model is Cronjob: + value = self._job_from_payload(payload) + elif model is CronjobRun: + value = self._run_from_payload(payload) + else: + value = model.model_validate(payload) + return cast(Stored[_Value], Stored(value=value, etag=etag)) + + def _read_payload(self, client: Any, key: str) -> tuple[dict[str, Any], str]: + try: + response = client.get_object(bucket=self._bucket, key=key) + except Exception as error: + if _status_code(error) == 404: + raise CronjobNotFound("定时任务不存在或已被删除。") from error + raise + content = response.read(_MAX_OBJECT_BYTES + 1) + if not isinstance(content, bytes) or len(content) > _MAX_OBJECT_BYTES: + raise ValueError("Cronjob object is invalid or too large.") + etag = self._etag(response) + if not etag: + raise RuntimeError("TOS cronjob object did not include an ETag.") + payload = json.loads(content) + if not isinstance(payload, dict): + raise TypeError("Cronjob object must be a JSON object.") + return payload, etag + + def _list_keys(self, client: Any, prefix: str) -> list[str]: + continuation_token = "" + keys: list[str] = [] + while True: + output = client.list_objects_type2( + bucket=self._bucket, + prefix=prefix, + continuation_token=continuation_token, + max_keys=1000, + ) + keys.extend( + str(item.key) for item in (getattr(output, "contents", None) or []) + ) + if not getattr(output, "is_truncated", False): + return keys + continuation_token = str( + getattr(output, "next_continuation_token", "") or "" + ) + if not continuation_token: + raise RuntimeError( + "TOS truncated a cronjob listing without a continuation token." + ) + + def _encode(self, value: BaseModel) -> bytes: + if isinstance(value, Cronjob): + payload = self._job_payload(value) + elif isinstance(value, CronjobRun): + payload = self._run_payload(value) + else: + payload = value.model_dump(mode="json", by_alias=True) + content = json.dumps( + payload, ensure_ascii=False, sort_keys=True, separators=(",", ":") + ).encode("utf-8") + if len(content) > _MAX_OBJECT_BYTES: + raise ValueError("Cronjob object is too large.") + return content + + def _replace_payload( + self, client: Any, key: str, payload: dict[str, Any], etag: str + ) -> None: + content = json.dumps( + payload, ensure_ascii=False, sort_keys=True, separators=(",", ":") + ).encode("utf-8") + try: + client.put_object( + bucket=self._bucket, + key=key, + content=content, + content_length=len(content), + content_type="application/json", + if_match=etag, + ) + except Exception as error: + if _status_code(error) in {409, 412}: + raise CronjobConflict( + "Cronjob changed concurrently; reload and try again." + ) from error + raise + + @classmethod + def _job_payload(cls, job: Cronjob) -> dict[str, Any]: + provider = "volcengine" if job.region.startswith("cn-") else "byteplus" + return { + "userId": job.owner_id, + "ownerId": job.owner_id, + "jobId": job.id, + "name": job.name, + "revision": job.revision, + "enabled": job.enabled, + "prompt": job.prompt, + "runtime": { + "provider": provider, + "runtimeId": job.runtime_id, + "runtimeName": job.runtime_name, + "agentName": job.agent_name, + "region": job.region, + "projectName": "default", + }, + "schedule": cls._scheduler_schedule(job.schedule), + "createdAt": cls._iso(job.created_at), + "updatedAt": cls._iso(job.updated_at), + "nextRunAt": cls._iso(job.next_run_at), + "maxRuntimeSeconds": 1800, + } + + @classmethod + def _job_from_payload(cls, payload: dict[str, Any]) -> Cronjob: + if "runtime" not in payload: + return Cronjob.model_validate(payload) + runtime = payload["runtime"] + schedule = payload["schedule"] + if not isinstance(runtime, dict) or not isinstance(schedule, dict): + raise TypeError("Cronjob runtime and schedule must be objects.") + created_at = payload.get("createdAt") or payload.get("updatedAt") + updated_at = payload.get("updatedAt") or created_at + return Cronjob.model_validate( + { + "jobId": payload["jobId"], + "ownerId": payload.get("userId") or payload["ownerId"], + "name": payload.get("name") or runtime.get("runtimeName") or "定时任务", + "runtimeId": runtime["runtimeId"], + "runtimeName": runtime.get("runtimeName") or runtime["runtimeId"], + "agentName": runtime["agentName"], + "region": runtime["region"], + "prompt": payload["prompt"], + "schedule": cls._api_schedule(schedule), + "enabled": payload["enabled"], + "revision": payload["revision"], + "createdAt": created_at, + "updatedAt": updated_at, + "nextRunAt": payload.get("nextRunAt"), + } + ) + + @classmethod + def _scheduler_schedule(cls, schedule: Any) -> dict[str, Any]: + if schedule.type == "once": + run_at = datetime.fromisoformat(schedule.once_at.replace("Z", "+00:00")) + if run_at.tzinfo is None: + run_at = run_at.replace(tzinfo=ZoneInfo(schedule.timezone)) + return { + "kind": "once", + "timezone": schedule.timezone, + "runAt": cls._iso(run_at), + } + if schedule.type in {"daily", "weekly"}: + hour, minute = (int(item) for item in schedule.time.split(":", 1)) + payload: dict[str, Any] = { + "kind": schedule.type, + "timezone": schedule.timezone, + "hour": hour, + "minute": minute, + } + if schedule.type == "weekly": + payload["weekdays"] = [(schedule.weekday + 6) % 7] + return payload + return { + "kind": "cron", + "timezone": schedule.timezone, + "cron": schedule.cron, + } + + @staticmethod + def _api_schedule(schedule: dict[str, Any]) -> dict[str, Any]: + kind = schedule.get("kind") or schedule.get("type") + timezone_name = str(schedule["timezone"]) + if kind == "once": + raw = schedule.get("runAt") or schedule.get("onceAt") + run_at = datetime.fromisoformat(str(raw).replace("Z", "+00:00")) + if run_at.tzinfo is not None: + run_at = run_at.astimezone(ZoneInfo(timezone_name)) + return { + "type": "once", + "timezone": timezone_name, + "onceAt": run_at.strftime("%Y-%m-%dT%H:%M"), + } + if kind == "daily": + return { + "type": "daily", + "timezone": timezone_name, + "time": f"{int(schedule['hour']):02d}:{int(schedule['minute']):02d}", + } + if kind == "weekly": + weekdays = schedule.get("weekdays") or [0] + return { + "type": "weekly", + "timezone": timezone_name, + "time": f"{int(schedule['hour']):02d}:{int(schedule['minute']):02d}", + "weekday": (int(weekdays[0]) + 1) % 7, + } + return { + "type": "cron", + "timezone": timezone_name, + "cron": schedule.get("cron") or schedule.get("expression"), + } + + @classmethod + def _run_payload(cls, run: CronjobRun) -> dict[str, Any]: + state = {"pending": "preparing", "success": "succeeded"}.get( + run.status, run.status + ) + updated_at = run.finished_at or run.started_at or run.created_at + return { + "userId": run.owner_id, + "jobId": run.job_id, + "runId": run.id, + "revision": run.revision, + "scheduledAt": cls._iso(run.scheduled_at), + "sessionId": run.session_id, + "state": state, + "createdAt": cls._iso(run.created_at or run.scheduled_at), + "updatedAt": cls._iso(updated_at or run.scheduled_at), + "attempt": 0, + "cancelRequested": run.cancellation_requested_at is not None, + "acknowledged": run.status in {"running", "success"}, + "runtimeVersion": run.runtime_version, + "output": run.output, + "error": run.error, + "completedAt": cls._iso(run.finished_at), + } + + @classmethod + def _run_from_payload(cls, payload: dict[str, Any]) -> CronjobRun: + if "state" not in payload: + return CronjobRun.model_validate(payload) + state = str(payload["state"]) + status = { + "preparing": "pending", + "retrying": "retrying", + "succeeded": "success", + }.get(state, state) + updated_at = payload.get("updatedAt") + return CronjobRun.model_validate( + { + "runId": payload["runId"], + "jobId": payload["jobId"], + "ownerId": payload["userId"], + "sessionId": payload["sessionId"], + "status": status, + "scheduledAt": payload["scheduledAt"], + "createdAt": payload.get("createdAt"), + "startedAt": (updated_at if state in {"running", "retrying"} else None), + "finishedAt": payload.get("completedAt"), + "cancellationRequestedAt": ( + updated_at if payload.get("cancelRequested", False) else None + ), + "runtimeVersion": payload.get("runtimeVersion") or "", + "output": payload.get("output") or "", + "error": payload.get("error") or "", + "attempt": payload.get("attempt") or 0, + "revision": payload.get("revision") or 1, + } + ) + + @staticmethod + def _iso(value: datetime | None) -> str | None: + if value is None: + return None + if value.tzinfo is None: + raise ValueError("Cronjob timestamps must include a timezone.") + return value.astimezone(timezone.utc).isoformat().replace("+00:00", "Z") + + def _write_etag(self, output: Any, client: Any, key: str) -> str: + etag = self._etag(output) + if etag: + return etag + response = client.get_object(bucket=self._bucket, key=key) + etag = self._etag(response) + if not etag: + raise RuntimeError("TOS cronjob write did not include an ETag.") + return etag + + @staticmethod + def _etag(value: Any) -> str: + direct = getattr(value, "etag", None) + if direct: + return str(direct) + return str(getattr(getattr(value, "meta", None), "etag", "") or "") + + def _job_key(self, owner_id: str, job_id: str) -> str: + return f"{self._job_prefix(owner_id, job_id)}/job.json" + + def _run_key(self, owner_id: str, job_id: str, run_id: str) -> str: + return f"{self._job_prefix(owner_id, job_id)}/runs/{self._part(run_id)}.json" + + def _lock_key(self, owner_id: str, job_id: str) -> str: + return f"{self._job_prefix(owner_id, job_id)}/lock.json" + + def _job_prefix(self, owner_id: str, job_id: str) -> str: + return f"{self._owner_prefix(owner_id)}/{self._part(job_id)}" + + def _owner_prefix(self, owner_id: str) -> str: + return ( + f"{self._root_prefix}/users/{self._part(owner_id)}/cronjobs" + if self._root_prefix + else f"users/{self._part(owner_id)}/cronjobs" + ) + + @staticmethod + def _part(value: str) -> str: + if not value: + raise ValueError("Cronjob object key part is required.") + return quote(value, safe="") + + +__all__ = [ + "CronjobConflict", + "CronjobNotFound", + "Stored", + "TosCronjobRepository", +] diff --git a/frontend/server/cronjobs/routes.py b/frontend/server/cronjobs/routes.py new file mode 100644 index 000000000..032ab2150 --- /dev/null +++ b/frontend/server/cronjobs/routes.py @@ -0,0 +1,232 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Thin FastAPI management routes for Studio cronjobs.""" + +from __future__ import annotations + +from collections.abc import Callable +from typing import Any + +from fastapi import HTTPException, Query, Request, Response, status + +from .repository import CronjobConflict, CronjobNotFound +from .schemas import ( + CreateCronjobRequest, + Cronjob, + CronjobIdentity, + CronjobListResponse, + CronjobRun, + CronjobRunListResponse, + UpdateCronjobRequest, +) +from .service import ( + CronjobAccessDenied, + CronjobRunQueueUnavailable, + CronjobService, +) + +IdentityResolver = Callable[[Request], CronjobIdentity] +RuntimeAuthorizer = Callable[[Request, str, str], Any] + + +def mount_routes( + app: Any, + service: CronjobService, + identity_resolver: IdentityResolver, + authorize_runtime: RuntimeAuthorizer | None = None, +) -> None: + @app.get("/web/cronjobs", response_model=CronjobListResponse) + async def list_cronjobs( + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> CronjobListResponse: + return await _invoke( + lambda: service.list( + identity_resolver(request), + owner_id=owner_id, + ), + list_response=True, + ) + + @app.post( + "/web/cronjobs", + response_model=Cronjob, + status_code=status.HTTP_201_CREATED, + ) + async def create_cronjob( + body: CreateCronjobRequest, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> Cronjob: + if authorize_runtime is not None: + authorize_runtime(request, body.runtime_id, body.region) + return await _invoke( + lambda: service.create(identity_resolver(request), body, owner_id=owner_id) + ) + + @app.get("/web/cronjobs/{job_id}", response_model=Cronjob) + async def get_cronjob( + job_id: str, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> Cronjob: + return await _invoke( + lambda: service.get(identity_resolver(request), job_id, owner_id=owner_id) + ) + + @app.patch("/web/cronjobs/{job_id}", response_model=Cronjob) + @app.post("/web/cronjobs/{job_id}/update", response_model=Cronjob) + async def update_cronjob( + job_id: str, + body: UpdateCronjobRequest, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> Cronjob: + identity = identity_resolver(request) + if authorize_runtime is not None and ( + body.runtime_id is not None or body.region is not None + ): + current = await _invoke( + lambda: service.get(identity, job_id, owner_id=owner_id) + ) + authorize_runtime( + request, + body.runtime_id or current.runtime_id, + body.region or current.region, + ) + return await _invoke( + lambda: service.update(identity, job_id, body, owner_id=owner_id) + ) + + @app.delete( + "/web/cronjobs/{job_id}", + status_code=status.HTTP_204_NO_CONTENT, + response_class=Response, + ) + async def delete_cronjob( + job_id: str, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> Response: + await _invoke( + lambda: service.delete( + identity_resolver(request), job_id, owner_id=owner_id + ) + ) + return Response(status_code=status.HTTP_204_NO_CONTENT) + + @app.post("/web/cronjobs/{job_id}/enable", response_model=Cronjob) + async def enable_cronjob( + job_id: str, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> Cronjob: + return await _invoke( + lambda: service.enable( + identity_resolver(request), job_id, owner_id=owner_id + ) + ) + + @app.post("/web/cronjobs/{job_id}/disable", response_model=Cronjob) + async def disable_cronjob( + job_id: str, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> Cronjob: + return await _invoke( + lambda: service.disable( + identity_resolver(request), job_id, owner_id=owner_id + ) + ) + + @app.post( + "/web/cronjobs/{job_id}/run", + response_model=CronjobRun, + status_code=status.HTTP_202_ACCEPTED, + ) + async def run_cronjob( + job_id: str, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> CronjobRun: + return await _invoke( + lambda: service.request_run( + identity_resolver(request), job_id, owner_id=owner_id + ) + ) + + @app.get( + "/web/cronjobs/{job_id}/runs", + response_model=CronjobRunListResponse, + ) + async def list_cronjob_runs( + job_id: str, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> CronjobRunListResponse: + return await _invoke( + lambda: service.list_runs( + identity_resolver(request), job_id, owner_id=owner_id + ), + list_response=True, + ) + + @app.post( + "/web/cronjobs/{job_id}/runs/{run_id}/cancel", + response_model=CronjobRun, + ) + async def cancel_cronjob_run( + job_id: str, + run_id: str, + request: Request, + owner_id: str | None = Query(default=None, alias="ownerId", max_length=1024), + ) -> CronjobRun: + return await _invoke( + lambda: service.cancel( + identity_resolver(request), + job_id, + run_id, + owner_id=owner_id, + ) + ) + + +async def _invoke(call: Callable[[], Any], *, list_response: bool = False) -> Any: + try: + result = await call() + except Exception as error: + _raise_api_error(error) + raise + if list_response: + return {"items": result} + return result + + +def _raise_api_error(error: Exception) -> None: + if isinstance(error, CronjobAccessDenied): + raise HTTPException(status_code=403, detail=str(error)) from error + if isinstance(error, CronjobNotFound): + raise HTTPException(status_code=404, detail=str(error)) from error + if isinstance(error, CronjobConflict): + raise HTTPException(status_code=409, detail=str(error)) from error + if isinstance(error, CronjobRunQueueUnavailable): + raise HTTPException(status_code=503, detail=str(error)) from error + raise HTTPException( + status_code=502, + detail="定时任务服务暂时不可用,请稍后重试。", + ) from error + + +__all__ = ["IdentityResolver", "RuntimeAuthorizer", "mount_routes"] diff --git a/frontend/server/cronjobs/schemas.py b/frontend/server/cronjobs/schemas.py new file mode 100644 index 000000000..9e29d4063 --- /dev/null +++ b/frontend/server/cronjobs/schemas.py @@ -0,0 +1,250 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Validated API and persistence contracts for Studio cronjobs.""" + +from __future__ import annotations + +from datetime import datetime +from typing import Annotated, Literal +from zoneinfo import ZoneInfo, ZoneInfoNotFoundError + +from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator + + +class _Model(BaseModel): + model_config = ConfigDict(extra="forbid", populate_by_name=True) + + +class _ZonedSchedule(_Model): + timezone: str = Field(min_length=1, max_length=128) + + @field_validator("timezone") + @classmethod + def validate_timezone(cls, value: str) -> str: + try: + ZoneInfo(value) + except ZoneInfoNotFoundError as error: + raise ValueError( + "schedule timezone must be a valid IANA timezone" + ) from error + return value + + +class OnceSchedule(_ZonedSchedule): + type: Literal["once"] = "once" + once_at: str = Field(alias="onceAt", min_length=1, max_length=64) + + @field_validator("once_at") + @classmethod + def validate_once_at(cls, value: str) -> str: + try: + datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError as error: + raise ValueError("onceAt must be an ISO-8601 datetime") from error + return value + + +class DailySchedule(_ZonedSchedule): + type: Literal["daily"] = "daily" + time: str = Field(pattern=r"^(?:[01]\d|2[0-3]):[0-5]\d$") + + +class WeeklySchedule(_ZonedSchedule): + type: Literal["weekly"] = "weekly" + time: str = Field(pattern=r"^(?:[01]\d|2[0-3]):[0-5]\d$") + weekday: int = Field(ge=0, le=6) + + +class CronSchedule(_ZonedSchedule): + type: Literal["cron"] = "cron" + cron: str = Field(min_length=1, max_length=256) + + @field_validator("cron") + @classmethod + def validate_cron(cls, value: str) -> str: + normalized = " ".join(value.split()) + if len(normalized.split(" ")) != 5: + raise ValueError("cron expression must contain exactly five fields") + return normalized + + +Schedule = Annotated[ + OnceSchedule | DailySchedule | WeeklySchedule | CronSchedule, + Field(discriminator="type"), +] + + +class CreateCronjobRequest(_Model): + name: str = Field(min_length=1, max_length=256) + runtime_id: str = Field(alias="runtimeId", min_length=1, max_length=512) + runtime_name: str = Field(alias="runtimeName", min_length=1, max_length=512) + agent_name: str = Field(alias="agentName", min_length=1, max_length=512) + region: str = Field(min_length=1, max_length=128) + prompt: str = Field(min_length=1, max_length=32_768) + schedule: Schedule + enabled: bool = True + + @field_validator( + "name", "runtime_id", "runtime_name", "agent_name", "region", "prompt" + ) + @classmethod + def strip_required_text(cls, value: str) -> str: + normalized = value.strip() + if not normalized: + raise ValueError("value must not be blank") + return normalized + + +class UpdateCronjobRequest(_Model): + name: str | None = Field(default=None, min_length=1, max_length=256) + runtime_id: str | None = Field( + default=None, alias="runtimeId", min_length=1, max_length=512 + ) + runtime_name: str | None = Field( + default=None, alias="runtimeName", min_length=1, max_length=512 + ) + agent_name: str | None = Field( + default=None, alias="agentName", min_length=1, max_length=512 + ) + region: str | None = Field(default=None, min_length=1, max_length=128) + prompt: str | None = Field(default=None, min_length=1, max_length=32_768) + schedule: Schedule | None = None + enabled: bool | None = None + + @field_validator( + "name", "runtime_id", "runtime_name", "agent_name", "region", "prompt" + ) + @classmethod + def strip_optional_text(cls, value: str | None) -> str | None: + if value is None: + return None + normalized = value.strip() + if not normalized: + raise ValueError("value must not be blank") + return normalized + + @model_validator(mode="after") + def require_change(self) -> UpdateCronjobRequest: + if not self.model_fields_set: + raise ValueError("At least one cronjob field must be updated.") + return self + + +class Cronjob(_Model): + job_id: str = Field(alias="jobId", min_length=1, max_length=128) + owner_id: str = Field(alias="ownerId", min_length=1, max_length=1024) + name: str = Field(min_length=1, max_length=256) + runtime_id: str = Field(alias="runtimeId", min_length=1, max_length=512) + runtime_name: str = Field(alias="runtimeName", min_length=1, max_length=512) + agent_name: str = Field(alias="agentName", min_length=1, max_length=512) + region: str = Field(min_length=1, max_length=128) + prompt: str = Field(min_length=1, max_length=32_768) + schedule: Schedule + enabled: bool = True + revision: int = Field(ge=1) + created_at: datetime = Field(alias="createdAt") + updated_at: datetime = Field(alias="updatedAt") + next_run_at: datetime | None = Field(default=None, alias="nextRunAt") + latest_run: CronjobRun | None = Field(default=None, alias="latestRun") + + @property + def id(self) -> str: + return self.job_id + + +RunStatus = Literal[ + "queued", + "pending", + "running", + "retrying", + "success", + "failed", + "cancelled", + "skipped", +] +RunTrigger = Literal["scheduled", "manual"] + + +class CronjobRun(_Model): + run_id: str = Field(alias="runId", min_length=1, max_length=128) + job_id: str = Field(alias="jobId", min_length=1, max_length=128) + owner_id: str = Field(alias="ownerId", min_length=1, max_length=1024) + session_id: str = Field(alias="sessionId", min_length=1, max_length=512) + status: RunStatus + scheduled_at: datetime = Field(alias="scheduledAt") + created_at: datetime | None = Field(default=None, alias="createdAt") + started_at: datetime | None = Field(default=None, alias="startedAt") + finished_at: datetime | None = Field(default=None, alias="finishedAt") + cancellation_requested_at: datetime | None = Field( + default=None, alias="cancellationRequestedAt" + ) + runtime_version: str = Field(default="", alias="runtimeVersion", max_length=512) + output: str = Field(default="", max_length=1_000_000) + error: str = Field(default="", max_length=32_768) + attempt: int = Field(default=0, ge=0) + revision: int = Field(default=1, ge=1, exclude=True) + + @property + def terminal(self) -> bool: + return self.status in {"success", "failed", "cancelled", "skipped"} + + @property + def id(self) -> str: + return self.run_id + + +class CronjobLock(_Model): + job_id: str = Field(alias="jobId", min_length=1, max_length=128) + owner_id: str = Field(alias="ownerId", min_length=1, max_length=1024) + run_id: str = Field(alias="runId", min_length=1, max_length=128) + state: Literal["held", "released"] + acquired_at: datetime = Field(alias="acquiredAt") + updated_at: datetime = Field(alias="updatedAt") + expires_at: datetime = Field(alias="expiresAt") + + def active_at(self, now: datetime) -> bool: + return self.state == "held" and self.expires_at > now + + +class CronjobIdentity(_Model): + owner_id: str = Field(alias="ownerId", min_length=1, max_length=1024) + is_admin: bool = Field(default=False, alias="isAdmin", exclude=True) + + +class CronjobListResponse(_Model): + items: list[Cronjob] + + +class CronjobRunListResponse(_Model): + items: list[CronjobRun] + + +__all__ = [ + "CreateCronjobRequest", + "CronSchedule", + "Cronjob", + "CronjobIdentity", + "CronjobListResponse", + "CronjobLock", + "CronjobRun", + "CronjobRunListResponse", + "DailySchedule", + "OnceSchedule", + "RunStatus", + "RunTrigger", + "Schedule", + "UpdateCronjobRequest", + "WeeklySchedule", +] diff --git a/frontend/server/cronjobs/service.py b/frontend/server/cronjobs/service.py new file mode 100644 index 000000000..d0bb85c7f --- /dev/null +++ b/frontend/server/cronjobs/service.py @@ -0,0 +1,340 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Ownership, lifecycle, and manual-run orchestration for cronjobs.""" + +from __future__ import annotations + +import asyncio +from collections.abc import Callable +from datetime import datetime, timedelta, timezone +from typing import Protocol +from uuid import uuid4 + +from frontend.service.studio_scheduler.models import CronJob as SchedulerCronJob +from frontend.service.studio_scheduler.schedule import next_scheduled_time + +from .repository import CronjobConflict, TosCronjobRepository +from .schemas import ( + CreateCronjobRequest, + Cronjob, + CronjobIdentity, + CronjobRun, + UpdateCronjobRequest, +) + + +class CronjobAccessDenied(PermissionError): + """The caller may not access the requested owner's cronjobs.""" + + +class CronjobRunQueueUnavailable(RuntimeError): + """Manual runs cannot be published to the durable due queue.""" + + +class CronjobAccessPolicy(Protocol): + def resolve_owner( + self, + identity: CronjobIdentity, + requested_owner_id: str | None, + ) -> str: ... + + +class CronjobDuePublisher(Protocol): + async def publish_next( + self, job: SchedulerCronJob, *, after: datetime + ) -> object | None: ... + + async def publish_run_now( + self, + *, + user_id: str, + job_id: str, + revision: int, + scheduled_at: datetime, + ) -> str: ... + + +class OwnerOnlyAccessPolicy: + """Default policy: callers can only operate on their own namespace.""" + + def resolve_owner( + self, + identity: CronjobIdentity, + requested_owner_id: str | None, + ) -> str: + if requested_owner_id and requested_owner_id != identity.owner_id: + raise CronjobAccessDenied("无权访问其他用户的定时任务。") + return identity.owner_id + + +class CronjobService: + def __init__( + self, + repository: TosCronjobRepository, + *, + access_policy: CronjobAccessPolicy | None = None, + due_publisher: CronjobDuePublisher | None = None, + clock: Callable[[], datetime] | None = None, + id_factory: Callable[[], str] | None = None, + ) -> None: + self.repository = repository + self._access_policy = access_policy or OwnerOnlyAccessPolicy() + self._due_publisher = due_publisher + self._clock = clock or (lambda: datetime.now(timezone.utc)) + self._id_factory = id_factory or (lambda: str(uuid4())) + + async def create( + self, + identity: CronjobIdentity, + request: CreateCronjobRequest, + *, + owner_id: str | None = None, + ) -> Cronjob: + owner = self._owner(identity, owner_id) + now = self._clock() + job = Cronjob( + jobId=self._id_factory(), + ownerId=owner, + **request.model_dump(by_alias=True), + revision=1, + createdAt=now, + updatedAt=now, + ) + if job.enabled: + job = job.model_copy(update={"next_run_at": self._next_time(job, now)}) + created = (await self.repository.create_job(job)).value + if created.enabled: + await self._publish_next(created, after=now) + return created + + async def list( + self, identity: CronjobIdentity, *, owner_id: str | None = None + ) -> list[Cronjob]: + owner = self._owner(identity, owner_id) + jobs = await self.repository.list_jobs(owner) + return list( + await asyncio.gather(*(self._with_latest(owner, job) for job in jobs)) + ) + + async def get( + self, + identity: CronjobIdentity, + job_id: str, + *, + owner_id: str | None = None, + ) -> Cronjob: + owner = self._owner(identity, owner_id) + job = (await self.repository.get_job(owner, job_id)).value + return await self._with_latest(owner, job) + + async def update( + self, + identity: CronjobIdentity, + job_id: str, + patch: UpdateCronjobRequest, + *, + owner_id: str | None = None, + ) -> Cronjob: + owner = self._owner(identity, owner_id) + current = await self.repository.get_job(owner, job_id) + # ``model_copy(update=...)`` intentionally skips validation. Keep nested + # Pydantic values such as ``Schedule`` as models instead of flattening + # them to dictionaries with ``model_dump``. + changes = { + field: getattr(patch, field) + for field in patch.model_fields_set + if getattr(patch, field) is not None + } + updated = current.value.model_copy( + update={ + **changes, + "revision": current.value.revision + 1, + "updated_at": self._clock(), + } + ) + updated = updated.model_copy( + update={ + "next_run_at": ( + self._next_time(updated, updated.updated_at) + if updated.enabled + else None + ) + } + ) + saved = (await self.repository.update_job(updated, current.etag)).value + if saved.enabled: + await self._publish_next(saved, after=saved.updated_at) + return saved + + async def enable( + self, + identity: CronjobIdentity, + job_id: str, + *, + owner_id: str | None = None, + ) -> Cronjob: + return await self._set_enabled(identity, job_id, True, owner_id) + + async def disable( + self, + identity: CronjobIdentity, + job_id: str, + *, + owner_id: str | None = None, + ) -> Cronjob: + return await self._set_enabled(identity, job_id, False, owner_id) + + async def request_run( + self, + identity: CronjobIdentity, + job_id: str, + *, + owner_id: str | None = None, + ) -> CronjobRun: + if self._due_publisher is None: + raise CronjobRunQueueUnavailable("定时任务执行队列尚未配置,请联系管理员。") + owner = self._owner(identity, owner_id) + job = (await self.repository.get_job(owner, job_id)).value + if not job.enabled: + raise CronjobConflict("Enable this cronjob before running it.") + now = self._clock().astimezone(timezone.utc).replace(second=0, microsecond=0) + scheduled_at = now + timedelta(minutes=1) + try: + run_id = await self._due_publisher.publish_run_now( + user_id=owner, + job_id=job.id, + revision=job.revision, + scheduled_at=now, + ) + except Exception as error: + raise CronjobRunQueueUnavailable( + "无法提交定时任务执行请求,请稍后重试。" + ) from error + queued = CronjobRun( + runId=run_id, + jobId=job.id, + ownerId=owner, + sessionId=run_id, + status="queued", + scheduledAt=scheduled_at, + createdAt=now, + ) + try: + return (await self.repository.create_run(queued)).value + except CronjobConflict: + return (await self.repository.get_run(owner, job.id, run_id)).value + + async def list_runs( + self, + identity: CronjobIdentity, + job_id: str, + *, + owner_id: str | None = None, + ) -> list[CronjobRun]: + owner = self._owner(identity, owner_id) + await self.repository.get_job(owner, job_id) + return await self.repository.list_runs(owner, job_id) + + async def cancel( + self, + identity: CronjobIdentity, + job_id: str, + run_id: str, + *, + owner_id: str | None = None, + ) -> CronjobRun: + owner = self._owner(identity, owner_id) + await self.repository.get_job(owner, job_id) + return await self.repository.request_cancel( + owner, job_id, run_id, self._clock() + ) + + async def delete( + self, + identity: CronjobIdentity, + job_id: str, + *, + owner_id: str | None = None, + ) -> None: + owner = self._owner(identity, owner_id) + await self.repository.get_job(owner, job_id) + lock = await self.repository.get_lock(owner, job_id) + if lock is not None and lock.value.active_at(self._clock()): + raise CronjobConflict("Stop the active run before deleting this cronjob.") + await self.repository.delete_job(owner, job_id) + + async def _set_enabled( + self, + identity: CronjobIdentity, + job_id: str, + enabled: bool, + owner_id: str | None, + ) -> Cronjob: + owner = self._owner(identity, owner_id) + current = await self.repository.get_job(owner, job_id) + if current.value.enabled is enabled: + return current.value + updated = current.value.model_copy( + update={ + "enabled": enabled, + "revision": current.value.revision + 1, + "updated_at": self._clock(), + "next_run_at": None, + } + ) + if enabled: + updated = updated.model_copy( + update={"next_run_at": self._next_time(updated, updated.updated_at)} + ) + saved = (await self.repository.update_job(updated, current.etag)).value + if enabled: + await self._publish_next(saved, after=saved.updated_at) + return saved + + async def _publish_next(self, job: Cronjob, *, after: datetime) -> None: + if self._due_publisher is None: + raise CronjobRunQueueUnavailable("定时任务执行队列尚未配置,请联系管理员。") + scheduler_job = SchedulerCronJob.from_dict( + self.repository.scheduler_job_payload(job) + ) + try: + await self._due_publisher.publish_next(scheduler_job, after=after) + except Exception as error: + raise CronjobRunQueueUnavailable( + "无法提交定时任务计划,请稍后重试。" + ) from error + + def _next_time(self, job: Cronjob, after: datetime) -> datetime | None: + scheduler_job = SchedulerCronJob.from_dict( + self.repository.scheduler_job_payload(job) + ) + return next_scheduled_time(scheduler_job.schedule, after) + + async def _with_latest(self, owner_id: str, job: Cronjob) -> Cronjob: + runs = await self.repository.list_runs(owner_id, job.id) + return job.model_copy(update={"latest_run": runs[0] if runs else None}) + + def _owner(self, identity: CronjobIdentity, requested_owner_id: str | None) -> str: + return self._access_policy.resolve_owner(identity, requested_owner_id) + + +__all__ = [ + "CronjobAccessDenied", + "CronjobAccessPolicy", + "CronjobDuePublisher", + "CronjobRunQueueUnavailable", + "CronjobService", + "OwnerOnlyAccessPolicy", +] diff --git a/frontend/service/studio_scheduler/__init__.py b/frontend/service/studio_scheduler/__init__.py new file mode 100644 index 000000000..eeb26637a --- /dev/null +++ b/frontend/service/studio_scheduler/__init__.py @@ -0,0 +1,37 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Stateless, TOS-backed scheduler for Studio cron jobs.""" + +from .dispatcher import Dispatcher +from .executor import ProviderRuntimeExecutor +from .local_runner import run_local_scheduler +from .memory_repository import InMemorySchedulerRepository +from .publisher import DuePublisher +from .runtime_provider import ( + AgentKitRuntimeConnectionResolver, + AgentKitRuntimeProvider, +) +from .tos_repository import TosSchedulerRepository + +__all__ = [ + "AgentKitRuntimeConnectionResolver", + "AgentKitRuntimeProvider", + "Dispatcher", + "DuePublisher", + "InMemorySchedulerRepository", + "ProviderRuntimeExecutor", + "TosSchedulerRepository", + "run_local_scheduler", +] diff --git a/frontend/service/studio_scheduler/app.py b/frontend/service/studio_scheduler/app.py new file mode 100644 index 000000000..5e03f00a4 --- /dev/null +++ b/frontend/service/studio_scheduler/app.py @@ -0,0 +1,141 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Production dependency assembly for the minute-triggered VeFaaS function.""" + +from __future__ import annotations + +import os +from collections.abc import Callable, Mapping +from dataclasses import dataclass +from typing import Any, cast + +from .dispatcher import Dispatcher +from .entrypoint import make_handler +from .executor import ProviderRuntimeExecutor +from .models import ProviderName +from .runtime_provider import ( + AgentKitRuntimeConnectionResolver, + AgentKitRuntimeProvider, + RuntimeConnectionResolver, + resolve_service_credentials, +) +from .tos_repository import TosSchedulerRepository + + +@dataclass(frozen=True) +class SchedulerSettings: + provider: ProviderName + bucket: str + storage_region: str + storage_endpoint: str + replica_id: str + pre_ack_attempts: int = 2 + + def __post_init__(self) -> None: + if self.provider not in {"volcengine", "byteplus"}: + raise ValueError(f"Unsupported scheduler provider: {self.provider}") + if not all( + value.strip() + for value in ( + self.bucket, + self.storage_region, + self.storage_endpoint, + self.replica_id, + ) + ): + raise ValueError("Scheduler storage and replica settings are required") + if self.pre_ack_attempts < 1: + raise ValueError("Scheduler retry attempts must be positive") + + @classmethod + def from_env(cls, source: Mapping[str, str] | None = None) -> SchedulerSettings: + environment = source if source is not None else os.environ + provider_value = str( + environment.get("AGENTKIT_CLOUD_PROVIDER") + or environment.get("CLOUD_PROVIDER") + or "volcengine" + ).strip() + if provider_value not in {"volcengine", "byteplus"}: + raise ValueError(f"Unsupported scheduler provider: {provider_value}") + provider = cast(ProviderName, provider_value) + region = str(environment.get("VEADK_STUDIO_TOS_REGION") or "").strip() + endpoint = str(environment.get("VEADK_STUDIO_TOS_ENDPOINT") or "").strip() + if region and not endpoint: + domain = "bytepluses.com" if provider == "byteplus" else "volces.com" + endpoint = f"tos-{region}.{domain}" + return cls( + provider=provider, + bucket=str(environment.get("VEADK_STUDIO_TOS_BUCKET") or "").strip(), + storage_region=region, + storage_endpoint=endpoint, + replica_id=str( + environment.get("VEFAAS_INSTANCE_NAME") + or environment.get("HOSTNAME") + or "studio-scheduler" + ).strip(), + pre_ack_attempts=int( + str(environment.get("STUDIO_CRONJOB_PRE_ACK_ATTEMPTS") or "2") + ), + ) + + +def create_dispatcher( + settings: SchedulerSettings | None = None, + *, + tos_client_factory: Callable[[], Any] | None = None, + runtime_resolver: RuntimeConnectionResolver | None = None, +) -> Dispatcher: + """Construct all cloud dependencies while preserving injectable test seams.""" + resolved = settings or SchedulerSettings.from_env() + client_factory = tos_client_factory or _tos_client_factory(resolved) + repository = TosSchedulerRepository( + bucket=resolved.bucket, + client_factory=client_factory, + provider=resolved.provider, + ) + resolver = runtime_resolver or AgentKitRuntimeConnectionResolver() + executor = ProviderRuntimeExecutor( + [AgentKitRuntimeProvider(resolved.provider, resolver)] + ) + return Dispatcher( + repository, + executor, + replica_id=resolved.replica_id, + pre_ack_attempts=resolved.pre_ack_attempts, + ) + + +def _tos_client_factory(settings: SchedulerSettings) -> Callable[[], Any]: + def create() -> Any: + import tos + + credentials = resolve_service_credentials(settings.provider) + return tos.TosClientV2( + ak=credentials.access_key, + sk=credentials.secret_key, + security_token=credentials.session_token or None, + endpoint=settings.storage_endpoint, + region=settings.storage_region, + ) + + return create + + +def handler(event: Any, context: Any) -> dict[str, int]: + """VeFaaS entry point; cloud trigger invokes this function once per minute.""" + return make_handler(create_dispatcher())(event, context) + + +__all__ = ["SchedulerSettings", "create_dispatcher", "handler"] diff --git a/frontend/service/studio_scheduler/deploy.py b/frontend/service/studio_scheduler/deploy.py new file mode 100644 index 000000000..fbbe7bc63 --- /dev/null +++ b/frontend/service/studio_scheduler/deploy.py @@ -0,0 +1,322 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Deploy the stateless cronjob scheduler beside Studio.""" + +from __future__ import annotations + +import json +import os +import shutil +import tempfile +import time +import urllib.request +from pathlib import Path +from typing import Any + +from .diagnostics import sanitize_diagnostic + +_TIMER_NAME = "veadk-studio-cronjobs-minute" +_MINUTE_CRONTAB = "* * * * *" + + +def scheduler_function_name(studio_application_name: str) -> str: + """Return a deterministic VeFaaS-safe name kept below the service limit.""" + normalized = studio_application_name.strip().replace("_", "-") + suffix = "-cronjobs" + return f"{normalized[: 64 - len(suffix)].rstrip('-')}{suffix}" + + +def deploy_scheduler( + service: Any, + *, + studio_application_name: str, + package_root: Path, + role_trn: str, + environment: dict[str, str], +) -> tuple[str, str]: + """Create/update the scheduler Function and its idempotent minute trigger.""" + function_name = scheduler_function_name(studio_application_name) + with tempfile.TemporaryDirectory(prefix="studio_cronjob_scheduler_") as tmp: + deployment_root = Path(tmp) + _stage_package(package_root, deployment_root) + function_id = _find_function_id(service, function_name) + if function_id: + service._replace_application_code_bundle( + function_id=function_id, + path=str(deployment_root), + environment_overrides=environment, + ) + else: + function_id = _create_function( + service, + function_name=function_name, + deployment_root=deployment_root, + role_trn=role_trn, + environment=environment, + ) + _install_dependencies(service, function_id) + _release_function(service, function_id) + timer_id = _ensure_minute_timer(service, function_id) + return function_id, timer_id + + +def deploy_scheduler_for_studio_update( + service: Any, + *, + studio_function_id: str, + package_root: Path, + provider: str, + project: str, + environment_overrides: dict[str, str], +) -> tuple[str, str, str]: + """Update the scheduler from the same bundle used by Studio self-update.""" + from volcenginesdkvefaas import GetFunctionRequest + + current_function = service.client.get_function( + GetFunctionRequest(id=studio_function_id) + ) + current_environment = { + str(item.key): str(item.value) + for item in (getattr(current_function, "envs", None) or []) + if getattr(item, "key", None) + } + merged_environment = {**current_environment, **environment_overrides} + storage_bucket = merged_environment.get("VEADK_STUDIO_TOS_BUCKET", "").strip() + storage_region = merged_environment.get("VEADK_STUDIO_TOS_REGION", "").strip() + if not storage_bucket or not storage_region: + raise ValueError("Studio self-update could not resolve scheduler TOS storage") + role_trn = str(getattr(current_function, "role", "") or "").strip() + if not role_trn: + raise ValueError( + "Studio Function does not expose an IAM role for the scheduler" + ) + scheduler_base = ( + merged_environment.get("VEADK_STUDIO_CRONJOB_SCHEDULER_BASE", "").strip() + or str(getattr(current_function, "name", "") or "").strip() + ) + if not scheduler_base: + raise ValueError("Studio Function name is unavailable for scheduler deployment") + function_id, timer_id = deploy_scheduler( + service, + studio_application_name=scheduler_base, + package_root=package_root, + role_trn=role_trn, + environment={ + "CLOUD_PROVIDER": provider, + "AGENTKIT_CLOUD_PROVIDER": provider, + "VEADK_STUDIO_TOS_BUCKET": storage_bucket, + "VEADK_STUDIO_TOS_REGION": storage_region, + "VEADK_STUDIO_TOS_ENDPOINT": merged_environment.get( + "VEADK_STUDIO_TOS_ENDPOINT", "" + ), + "VEADK_STUDIO_PROJECT": project, + }, + ) + return function_id, timer_id, scheduler_base + + +def _stage_package(package_root: Path, destination: Path) -> None: + requirements = package_root / "requirements.txt" + if not requirements.is_file(): + raise ValueError("Studio scheduler package is missing requirements.txt") + shutil.copy2(requirements, destination / requirements.name) + for wheel in package_root.glob("*.whl"): + shutil.copy2(wheel, destination / wheel.name) + run_script = destination / "run.sh" + run_script.write_text( + "#!/bin/bash\n" + "set -e\n" + 'ROOT_DIR="$(cd "$(dirname "$0")" && pwd)"\n' + 'cd "$ROOT_DIR"\n' + 'if [ -d "output" ]; then cd ./output/; fi\n' + "export PYTHONPATH=$PYTHONPATH:./site-packages\n" + "exec python3 -m uvicorn " + "frontend.service.studio_scheduler.http_app:app " + '--host 0.0.0.0 --port "${_FAAS_RUNTIME_PORT:-8000}"\n', + encoding="utf-8", + ) + run_script.chmod(0o755) + + +def _create_function( + service: Any, + *, + function_name: str, + deployment_root: Path, + role_trn: str, + environment: dict[str, str], +) -> str: + import veadk.config + + original_environment = dict(veadk.config.veadk_environments) + original_role = os.environ.get("IAM_ROLE") + try: + veadk.config.veadk_environments.clear() + veadk.config.veadk_environments.update(environment) + os.environ["IAM_ROLE"] = role_trn + _, function_id = service._create_function( + function_name, + str(deployment_root), + ) + return str(function_id) + finally: + veadk.config.veadk_environments.clear() + veadk.config.veadk_environments.update(original_environment) + if original_role is None: + os.environ.pop("IAM_ROLE", None) + else: + os.environ["IAM_ROLE"] = original_role + + +def _find_function_id(service: Any, function_name: str) -> str: + from volcenginesdkvefaas import ListFunctionsRequest + + page_number = 1 + functions: list[Any] = [] + while True: + response = service.client.list_functions( + ListFunctionsRequest(page_number=page_number, page_size=100) + ) + functions.extend(list(getattr(response, "items", []) or [])) + total = int(getattr(response, "total", 0) or 0) + if page_number * 100 >= total: + break + page_number += 1 + matches = [ + item for item in functions if str(getattr(item, "name", "")) == function_name + ] + if len(matches) > 1: + raise RuntimeError(f"Multiple VeFaaS functions are named {function_name}") + return str(getattr(matches[0], "id", "") or "") if matches else "" + + +def _release_function(service: Any, function_id: str) -> None: + from volcenginesdkvefaas import GetReleaseStatusRequest, ReleaseRequest + + service.client.release(ReleaseRequest(function_id=function_id, revision_number=0)) + for _ in range(120): + response = service.client.get_release_status( + GetReleaseStatusRequest(function_id=function_id) + ) + state = str(getattr(response, "status", "") or "").lower() + if "succ" in state or state == "done": + return + if "fail" in state or "error" in state: + detail = sanitize_diagnostic( + " ".join( + str(value or "").strip() + for value in ( + getattr(response, "error_code", ""), + getattr(response, "status_message", ""), + ) + if value + ), + limit=2_000, + ) + suffix = f". {detail}" if detail else "" + raise RuntimeError(f"Scheduler function release failed: {state}{suffix}") + time.sleep(5) + raise RuntimeError("Scheduler function release did not finish in 10 minutes") + + +def _install_dependencies(service: Any, function_id: str) -> None: + from volcenginesdkvefaas import ( + CreateDependencyInstallTaskRequest, + GetDependencyInstallTaskLogDownloadURIRequest, + GetDependencyInstallTaskStatusRequest, + ) + + service.client.create_dependency_install_task( + CreateDependencyInstallTaskRequest(function_id=function_id) + ) + for _ in range(120): + response = service.client.get_dependency_install_task_status( + GetDependencyInstallTaskStatusRequest(function_id=function_id) + ) + state = str(getattr(response, "status", "") or "").lower() + if "succ" in state or state == "done": + return + if "fail" in state or "error" in state: + detail = "" + try: + log_response = ( + service.client.get_dependency_install_task_log_download_uri( + GetDependencyInstallTaskLogDownloadURIRequest( + function_id=function_id + ) + ) + ) + download_url = str( + getattr(log_response, "download_url", "") or "" + ).strip() + if download_url: + with urllib.request.urlopen(download_url, timeout=30) as log_stream: + detail = sanitize_diagnostic( + log_stream.read().decode("utf-8", "replace"), + limit=2_000, + ) + except Exception: # noqa: BLE001 - diagnostics must not mask failure + detail = "" + suffix = f". {detail}" if detail else "" + raise RuntimeError( + f"Scheduler dependency installation failed: {state}{suffix}" + ) + time.sleep(5) + raise RuntimeError("Scheduler dependency installation did not finish in 10 minutes") + + +def _ensure_minute_timer(service: Any, function_id: str) -> str: + from volcenginesdkvefaas import ( + CreateTimerRequest, + ListTriggersRequest, + UpdateTimerRequest, + ) + + response = service.client.list_triggers( + ListTriggersRequest(function_id=function_id) + ) + triggers = list(getattr(response, "items", []) or []) + matches = [ + item + for item in triggers + if str(getattr(item, "name", "")) == _TIMER_NAME + and str(getattr(item, "type", "")).lower() == "timer" + ] + if len(matches) > 1: + raise RuntimeError(f"Multiple VeFaaS timers are named {_TIMER_NAME}") + common = { + "function_id": function_id, + "crontab": _MINUTE_CRONTAB, + "description": "Wake the VeADK Studio cronjob dispatcher every minute", + "enable_concurrency": True, + "enabled": True, + "payload": json.dumps({"source": "veadk-studio-cronjobs"}), + "retries": 0, + } + if matches: + timer_id = str(getattr(matches[0], "id", "") or "") + service.client.update_timer(UpdateTimerRequest(id=timer_id, **common)) + return timer_id + created = service.client.create_timer( + CreateTimerRequest(name=_TIMER_NAME, **common) + ) + return str(getattr(created, "id", "") or "") + + +__all__ = [ + "deploy_scheduler", + "deploy_scheduler_for_studio_update", + "scheduler_function_name", +] diff --git a/frontend/service/studio_scheduler/diagnostics.py b/frontend/service/studio_scheduler/diagnostics.py new file mode 100644 index 000000000..fe3010f52 --- /dev/null +++ b/frontend/service/studio_scheduler/diagnostics.py @@ -0,0 +1,63 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Bounded, redacted diagnostics safe for TOS persistence and the Studio UI.""" + +from __future__ import annotations + +import re +from collections.abc import Iterable + +_MAX_DIAGNOSTIC_CHARS = 4_000 +_SECRET_NAMES = ( + r"authorization|api[_-]?key|access[_-]?key|secret(?:[_-]?key)?|" + r"session[_-]?token|token|cookie" +) +_QUOTED_SECRET_ASSIGNMENT = re.compile( + rf"(?i)([\"'](?:{_SECRET_NAMES})[\"']\s*:\s*)([\"'])(.*?)(\2)" +) +_SECRET_ASSIGNMENT = re.compile(rf"(?i)\b({_SECRET_NAMES})\b(\s*[:=]\s*)([^\s,;]+)") +_BEARER_TOKEN = re.compile(r"(?i)\bBearer\s+[A-Za-z0-9._~+/=-]+") +_JWT_TOKEN = re.compile( + r"\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b" +) + + +def sanitize_diagnostic( + value: object, + *, + secrets: Iterable[str] = (), + limit: int = _MAX_DIAGNOSTIC_CHARS, +) -> str: + """Return one readable diagnostic without credentials or unbounded payloads.""" + text = str(value or "").replace("\x00", "").strip() + for secret in secrets: + if secret: + text = text.replace(str(secret), "[REDACTED]") + text = _BEARER_TOKEN.sub("Bearer [REDACTED]", text) + text = _JWT_TOKEN.sub("[REDACTED]", text) + text = _QUOTED_SECRET_ASSIGNMENT.sub( + lambda match: (f"{match.group(1)}{match.group(2)}[REDACTED]{match.group(4)}"), + text, + ) + text = _SECRET_ASSIGNMENT.sub( + lambda match: f"{match.group(1)}{match.group(2)}[REDACTED]", + text, + ) + if len(text) > limit: + text = f"{text[:limit].rstrip()}\n[truncated]" + return text + + +__all__ = ["sanitize_diagnostic"] diff --git a/frontend/service/studio_scheduler/dispatcher.py b/frontend/service/studio_scheduler/dispatcher.py new file mode 100644 index 000000000..380016803 --- /dev/null +++ b/frontend/service/studio_scheduler/dispatcher.py @@ -0,0 +1,366 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""One-shot dispatcher invoked once per minute by the cloud trigger.""" + +from __future__ import annotations + +import asyncio +from dataclasses import replace +from datetime import datetime, timedelta, timezone +from typing import Literal + +from .diagnostics import sanitize_diagnostic +from .models import ( + CronJob, + DispatchSummary, + DuePointer, + ExecutionRequest, + RuntimeInvocationError, + ScheduledRun, + deterministic_run_id, +) +from .ports import CancellationControl, RuntimeExecutor, SchedulerRepository +from .schedule import next_scheduled_time + +_Outcome = Literal["started", "stale", "skipped", "failed"] + + +class _RunControl(CancellationControl): + def __init__( + self, + repository: SchedulerRepository, + *, + user_id: str, + job_id: str, + run_id: str, + ) -> None: + self._repository = repository + self._user_id = user_id + self._job_id = job_id + self._run_id = run_id + + async def is_cancel_requested(self) -> bool: + run = await self._repository.get_run( + user_id=self._user_id, + job_id=self._job_id, + run_id=self._run_id, + ) + return bool(run and run.cancel_requested) + + +class Dispatcher: + """Dispatch a single UTC minute without retaining in-memory schedule state.""" + + def __init__( + self, + repository: SchedulerRepository, + executor: RuntimeExecutor, + *, + replica_id: str, + pre_ack_attempts: int = 2, + ) -> None: + if not replica_id.strip(): + raise ValueError("replica_id must not be empty") + if pre_ack_attempts < 1: + raise ValueError("pre_ack_attempts must be positive") + self._repository = repository + self._executor = executor + self._replica_id = replica_id + self._pre_ack_attempts = pre_ack_attempts + + async def dispatch_minute(self, now: datetime) -> DispatchSummary: + """List and process only the due bucket for ``now``; missed buckets stay missed.""" + if now.tzinfo is None or now.utcoffset() is None: + raise ValueError("dispatch time must include a timezone") + minute = now.astimezone(timezone.utc).replace(second=0, microsecond=0) + pointers = await self._repository.list_due(minute) + outcomes = await asyncio.gather( + *(self._dispatch(pointer, minute) for pointer in pointers) + ) + return DispatchSummary( + scanned=len(pointers), + started=outcomes.count("started"), + stale=outcomes.count("stale"), + skipped=outcomes.count("skipped"), + failed=outcomes.count("failed"), + ) + + async def _dispatch(self, pointer: DuePointer, now: datetime) -> _Outcome: + if pointer.scheduled_at != now: + return "stale" + job = await self._repository.get_job(pointer.user_id, pointer.job_id) + if not self._is_current(pointer, job): + return "stale" + assert job is not None + + run_id = deterministic_run_id(pointer) + lock = await self._repository.acquire_lock( + user_id=job.user_id, + job_id=job.job_id, + run_id=run_id, + replica_id=self._replica_id, + now=now, + expires_at=now + timedelta(seconds=job.max_runtime_seconds), + ) + if not lock.acquired: + if lock.active_run_id != run_id: + await self._record_skipped(pointer, now, run_id) + await self._write_next_due(job, pointer) + return "skipped" + + if lock.abandoned_run_id: + await self._mark_abandoned(job, lock.abandoned_run_id, now) + + run = ScheduledRun( + user_id=job.user_id, + job_id=job.job_id, + run_id=run_id, + revision=job.revision, + scheduled_at=pointer.scheduled_at, + session_id=run_id, + state="preparing", + created_at=now, + updated_at=now, + ) + try: + if not await self._repository.create_run(run): + existing = await self._repository.get_run( + user_id=run.user_id, + job_id=run.job_id, + run_id=run.run_id, + ) + if existing is None or existing.state != "queued": + return "skipped" + run = await self._repository.update_run( + replace( + existing, + state="preparing", + updated_at=now, + error="", + ) + ) + await self._write_next_due(job, pointer) + return await self._execute(job, run, now) + # This is the durable failure boundary: unknown adapter/storage failures are + # persisted and never retried as an already-acknowledged Runtime call. + except Exception as error: # noqa: BLE001 + existing = await self._repository.get_run( + user_id=run.user_id, + job_id=run.job_id, + run_id=run.run_id, + ) + if existing is not None and existing.state not in { + "succeeded", + "failed", + "cancelled", + "skipped", + }: + await self._repository.update_run( + replace( + existing, + state="failed", + error=str(error), + updated_at=now, + completed_at=now, + ) + ) + return "failed" + finally: + await self._repository.release_lock( + user_id=job.user_id, + job_id=job.job_id, + run_id=run_id, + released_at=now, + ) + + async def _execute( + self, job: CronJob, run: ScheduledRun, now: datetime + ) -> _Outcome: + control = _RunControl( + self._repository, + user_id=run.user_id, + job_id=run.job_id, + run_id=run.run_id, + ) + request = ExecutionRequest( + run_id=run.run_id, + session_id=run.session_id, + user_id=run.user_id, + job_id=run.job_id, + prompt=job.prompt, + runtime=job.runtime, + timeout_seconds=job.max_runtime_seconds, + ) + current = run + for attempt in range(1, self._pre_ack_attempts + 1): + if await control.is_cancel_requested(): + await self._finish(current, now, state="cancelled") + return "started" + current = await self._repository.update_run( + replace( + current, + state="running" if attempt == 1 else "retrying", + attempt=attempt, + updated_at=now, + error="", + ) + ) + try: + result = await self._executor.execute(request, control) + except RuntimeInvocationError as error: + current = replace( + current, + acknowledged=error.acknowledged, + error=str(error), + ) + can_retry = ( + not error.acknowledged + and error.retryable + and attempt < self._pre_ack_attempts + ) + if can_retry: + current = await self._repository.update_run( + replace(current, state="retrying", updated_at=now) + ) + continue + if await control.is_cancel_requested(): + await self._finish(current, now, state="cancelled") + else: + await self._finish(current, now, state="failed", error=str(error)) + return "started" + # Providers may surface non-domain SDK exceptions. They are terminal; + # only an explicit RuntimeInvocationError can opt into pre-ack retry. + except Exception as error: # noqa: BLE001 + detail = sanitize_diagnostic(error) + await self._finish( + current, + now, + state="failed", + error=( + "Runtime execution failed before a structured result was returned." + + (f" Detail: {detail}." if detail else "") + + " Check the scheduler and Runtime logs, then retry this run." + ), + ) + return "started" + + if await control.is_cancel_requested(): + await self._finish(current, now, state="cancelled") + else: + await self._finish( + current, + now, + state="succeeded", + acknowledged=True, + output=result.output, + runtime_version=result.runtime_version, + session_id=result.session_id or current.session_id, + ) + return "started" + return "failed" + + async def _finish( + self, + run: ScheduledRun, + now: datetime, + *, + state: Literal["succeeded", "failed", "cancelled"], + acknowledged: bool | None = None, + output: str = "", + runtime_version: str = "", + session_id: str = "", + error: str = "", + ) -> ScheduledRun: + return await self._repository.update_run( + replace( + run, + state=state, + acknowledged=( + run.acknowledged if acknowledged is None else acknowledged + ), + output=output, + runtime_version=runtime_version, + session_id=session_id or run.session_id, + error=error, + updated_at=now, + completed_at=now, + ) + ) + + async def _write_next_due(self, job: CronJob, pointer: DuePointer) -> None: + next_time = next_scheduled_time(job.schedule, pointer.scheduled_at) + if next_time is None: + return + await self._repository.put_due( + DuePointer( + user_id=job.user_id, + job_id=job.job_id, + revision=job.revision, + scheduled_at=next_time, + ) + ) + + async def _record_skipped( + self, pointer: DuePointer, now: datetime, run_id: str + ) -> None: + run = ScheduledRun( + user_id=pointer.user_id, + job_id=pointer.job_id, + run_id=run_id, + revision=pointer.revision, + scheduled_at=pointer.scheduled_at, + session_id=run_id, + state="skipped", + created_at=now, + updated_at=now, + error="Previous execution is still running", + completed_at=now, + ) + await self._repository.create_run(run) + + async def _mark_abandoned( + self, job: CronJob, abandoned_run_id: str, now: datetime + ) -> None: + abandoned = await self._repository.get_run( + user_id=job.user_id, + job_id=job.job_id, + run_id=abandoned_run_id, + ) + if abandoned is None or abandoned.state in { + "succeeded", + "failed", + "cancelled", + "skipped", + }: + return + await self._repository.update_run( + replace( + abandoned, + state="failed", + error="Scheduler lease expired before completion", + updated_at=now, + completed_at=now, + ) + ) + + @staticmethod + def _is_current(pointer: DuePointer, job: CronJob | None) -> bool: + return bool( + job + and job.enabled + and job.user_id == pointer.user_id + and job.job_id == pointer.job_id + and job.revision == pointer.revision + ) diff --git a/frontend/service/studio_scheduler/entrypoint.py b/frontend/service/studio_scheduler/entrypoint.py new file mode 100644 index 000000000..bbb570af8 --- /dev/null +++ b/frontend/service/studio_scheduler/entrypoint.py @@ -0,0 +1,40 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Dependency-injected entry point for a once-per-minute VeFaaS trigger.""" + +from __future__ import annotations + +import asyncio +from collections.abc import Callable +from datetime import datetime, timezone +from typing import Any + +from .dispatcher import Dispatcher + + +def make_handler(dispatcher: Dispatcher) -> Callable[[Any, Any], dict[str, int]]: + """Build a synchronous cloud handler without retaining scheduling state.""" + + def handler(_event: Any, _context: Any) -> dict[str, int]: + summary = asyncio.run(dispatcher.dispatch_minute(datetime.now(timezone.utc))) + return { + "scanned": summary.scanned, + "started": summary.started, + "stale": summary.stale, + "skipped": summary.skipped, + "failed": summary.failed, + } + + return handler diff --git a/frontend/service/studio_scheduler/executor.py b/frontend/service/studio_scheduler/executor.py new file mode 100644 index 000000000..49a8d15aa --- /dev/null +++ b/frontend/service/studio_scheduler/executor.py @@ -0,0 +1,42 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Thin provider selection boundary shared by Volcengine and BytePlus.""" + +from __future__ import annotations + +from collections.abc import Iterable + +from .models import ExecutionRequest, ExecutionResult, ProviderName +from .ports import CancellationControl, RuntimeProvider + + +class ProviderRuntimeExecutor: + """Delegate to a provider adapter that owns service-identity resolution.""" + + def __init__(self, providers: Iterable[RuntimeProvider]) -> None: + self._providers: dict[ProviderName, RuntimeProvider] = { + provider.provider: provider for provider in providers + } + + async def execute( + self, request: ExecutionRequest, control: CancellationControl + ) -> ExecutionResult: + try: + provider = self._providers[request.runtime.provider] + except KeyError as error: + raise ValueError( + f"Runtime provider is not configured: {request.runtime.provider}" + ) from error + return await provider.execute(request, control) diff --git a/frontend/service/studio_scheduler/http_app.py b/frontend/service/studio_scheduler/http_app.py new file mode 100644 index 000000000..077535b1b --- /dev/null +++ b/frontend/service/studio_scheduler/http_app.py @@ -0,0 +1,45 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""HTTP boundary used by the native VeFaaS minute-timer function.""" + +from __future__ import annotations + +from datetime import datetime, timezone + +from fastapi import FastAPI + +from .app import create_dispatcher + +app = FastAPI(title="VeADK Studio cronjob scheduler", docs_url=None, redoc_url=None) + + +@app.get("/healthz") +async def healthz() -> dict[str, bool]: + return {"ok": True} + + +@app.post("/") +async def dispatch_current_minute() -> dict[str, int]: + summary = await create_dispatcher().dispatch_minute(datetime.now(timezone.utc)) + return { + "scanned": summary.scanned, + "started": summary.started, + "stale": summary.stale, + "skipped": summary.skipped, + "failed": summary.failed, + } + + +__all__ = ["app", "dispatch_current_minute", "healthz"] diff --git a/frontend/service/studio_scheduler/local_runner.py b/frontend/service/studio_scheduler/local_runner.py new file mode 100644 index 000000000..2040d1c14 --- /dev/null +++ b/frontend/service/studio_scheduler/local_runner.py @@ -0,0 +1,70 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Minute-aligned scheduler loop used only by local Studio development.""" + +from __future__ import annotations + +import asyncio +import logging +from collections.abc import Awaitable, Callable +from datetime import datetime, timedelta, timezone +from typing import Protocol + +from .models import DispatchSummary + +logger = logging.getLogger(__name__) + +Clock = Callable[[], datetime] +Sleeper = Callable[[float], Awaitable[None]] + + +class MinuteDispatcher(Protocol): + async def dispatch_minute(self, now: datetime) -> DispatchSummary: ... + + +async def run_local_scheduler( + dispatcher: MinuteDispatcher, + *, + clock: Clock | None = None, + sleep: Sleeper = asyncio.sleep, +) -> None: + """Dispatch each observed UTC minute once until the task is cancelled.""" + get_now = clock or (lambda: datetime.now(timezone.utc)) + last_dispatched: datetime | None = None + while True: + now = get_now() + if now.tzinfo is None or now.utcoffset() is None: + raise ValueError("local scheduler clock must include a timezone") + minute = now.astimezone(timezone.utc).replace(second=0, microsecond=0) + if minute != last_dispatched: + last_dispatched = minute + try: + summary = await dispatcher.dispatch_minute(minute) + logger.debug( + "Local Studio scheduler scanned=%s started=%s failed=%s", + summary.scanned, + summary.started, + summary.failed, + ) + except asyncio.CancelledError: + raise + except Exception: + logger.exception("Local Studio scheduler failed for %s", minute) + next_minute = minute + timedelta(minutes=1) + delay = max(0.01, (next_minute - get_now()).total_seconds()) + await sleep(delay) + + +__all__ = ["run_local_scheduler"] diff --git a/frontend/service/studio_scheduler/memory_repository.py b/frontend/service/studio_scheduler/memory_repository.py new file mode 100644 index 000000000..bbffa09ce --- /dev/null +++ b/frontend/service/studio_scheduler/memory_repository.py @@ -0,0 +1,157 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""In-memory repository for local invocation and deterministic service tests.""" + +from __future__ import annotations + +import asyncio +from dataclasses import replace +from datetime import datetime + +from .models import CronJob, DuePointer, JobLock, LockAttempt, ScheduledRun + + +class InMemorySchedulerRepository: + """Process-local implementation of the same repository contract as TOS.""" + + def __init__(self) -> None: + self.jobs: dict[tuple[str, str], CronJob] = {} + self.due: dict[tuple[str, str, datetime], DuePointer] = {} + self.runs: dict[tuple[str, str, str], ScheduledRun] = {} + self.locks: dict[tuple[str, str], JobLock] = {} + self._mutex = asyncio.Lock() + + async def put_job(self, job: CronJob) -> None: + async with self._mutex: + self.jobs[(job.user_id, job.job_id)] = job + + async def list_due(self, minute: datetime) -> list[DuePointer]: + async with self._mutex: + return sorted( + ( + pointer + for pointer in self.due.values() + if pointer.scheduled_at == minute + ), + key=lambda pointer: (pointer.user_id, pointer.job_id), + ) + + async def get_job(self, user_id: str, job_id: str) -> CronJob | None: + async with self._mutex: + return self.jobs.get((user_id, job_id)) + + async def put_due(self, pointer: DuePointer) -> bool: + key = (pointer.user_id, pointer.job_id, pointer.scheduled_at) + async with self._mutex: + existing = self.due.get(key) + if existing is not None and existing != pointer: + raise ValueError("Due pointer already exists with different data") + self.due[key] = pointer + return existing is None + + async def acquire_lock( + self, + *, + user_id: str, + job_id: str, + run_id: str, + replica_id: str, + now: datetime, + expires_at: datetime, + ) -> LockAttempt: + key = (user_id, job_id) + requested = JobLock( + run_id=run_id, + replica_id=replica_id, + state="held", + acquired_at=now, + expires_at=expires_at, + ) + async with self._mutex: + existing = self.locks.get(key) + if existing and existing.state == "held" and existing.expires_at > now: + return LockAttempt(acquired=False, active_run_id=existing.run_id) + self.locks[key] = requested + abandoned = ( + existing.run_id + if existing is not None and existing.state == "held" + else "" + ) + return LockAttempt(acquired=True, abandoned_run_id=abandoned) + + async def release_lock( + self, + *, + user_id: str, + job_id: str, + run_id: str, + released_at: datetime, + ) -> None: + key = (user_id, job_id) + async with self._mutex: + existing = self.locks.get(key) + if existing and existing.run_id == run_id and existing.state == "held": + self.locks[key] = replace( + existing, + state="released", + released_at=released_at, + ) + + async def create_run(self, run: ScheduledRun) -> bool: + key = (run.user_id, run.job_id, run.run_id) + async with self._mutex: + existing = self.runs.get(key) + if existing is not None: + return False + self.runs[key] = run + return True + + async def update_run(self, run: ScheduledRun) -> ScheduledRun: + key = (run.user_id, run.job_id, run.run_id) + async with self._mutex: + existing = self.runs[key] + merged = replace( + run, + cancel_requested=run.cancel_requested or existing.cancel_requested, + ) + self.runs[key] = merged + return merged + + async def get_run( + self, *, user_id: str, job_id: str, run_id: str + ) -> ScheduledRun | None: + async with self._mutex: + return self.runs.get((user_id, job_id, run_id)) + + async def request_cancel( + self, + *, + user_id: str, + job_id: str, + run_id: str, + requested_at: datetime, + ) -> ScheduledRun | None: + key = (user_id, job_id, run_id) + async with self._mutex: + existing = self.runs.get(key) + if existing is None: + return None + updated = replace( + existing, + cancel_requested=True, + updated_at=requested_at, + ) + self.runs[key] = updated + return updated diff --git a/frontend/service/studio_scheduler/models.py b/frontend/service/studio_scheduler/models.py new file mode 100644 index 000000000..995be0e5b --- /dev/null +++ b/frontend/service/studio_scheduler/models.py @@ -0,0 +1,485 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Small validated domain contracts shared by scheduler adapters.""" + +from __future__ import annotations + +import hashlib +from dataclasses import dataclass +from datetime import datetime, timezone +from typing import Literal, cast +from zoneinfo import ZoneInfo, ZoneInfoNotFoundError + +ProviderName = Literal["volcengine", "byteplus"] +ScheduleKind = Literal["once", "daily", "weekly", "cron"] +RunState = Literal[ + "queued", + "preparing", + "running", + "retrying", + "succeeded", + "failed", + "cancelled", + "skipped", +] + + +def _aware_utc(value: datetime, name: str) -> datetime: + if value.tzinfo is None or value.utcoffset() is None: + raise ValueError(f"{name} must include a timezone") + return value.astimezone(timezone.utc) + + +def _required(data: dict[str, object], camel: str, snake: str | None = None) -> object: + if camel in data: + return data[camel] + if snake and snake in data: + return data[snake] + raise ValueError(f"Missing required field: {camel}") + + +def _datetime(value: object, name: str) -> datetime: + if isinstance(value, datetime): + return _aware_utc(value, name) + if not isinstance(value, str): + raise TypeError(f"{name} must be an ISO-8601 datetime") + try: + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError as error: + raise ValueError(f"{name} must be an ISO-8601 datetime") from error + return _aware_utc(parsed, name) + + +def _iso(value: datetime | None) -> str | None: + if value is None: + return None + return value.astimezone(timezone.utc).isoformat().replace("+00:00", "Z") + + +def _int(value: object, name: str) -> int: + if isinstance(value, bool) or not isinstance(value, (int, str)): + raise TypeError(f"{name} must be an integer") + try: + return int(value) + except ValueError as error: + raise ValueError(f"{name} must be an integer") from error + + +@dataclass(frozen=True) +class Schedule: + """One user-visible schedule, evaluated in its declared IANA timezone.""" + + kind: ScheduleKind + timezone: str + run_at: datetime | None = None + hour: int | None = None + minute: int | None = None + weekdays: tuple[int, ...] = () + cron: str = "" + + def __post_init__(self) -> None: + try: + ZoneInfo(self.timezone) + except ZoneInfoNotFoundError as error: + raise ValueError(f"Unknown schedule timezone: {self.timezone}") from error + if self.kind == "once": + if self.run_at is None: + raise ValueError("once schedule requires runAt") + object.__setattr__(self, "run_at", _aware_utc(self.run_at, "runAt")) + elif self.kind in {"daily", "weekly"}: + if self.hour is None or not 0 <= self.hour <= 23: + raise ValueError(f"{self.kind} schedule requires hour from 0 to 23") + if self.minute is None or not 0 <= self.minute <= 59: + raise ValueError(f"{self.kind} schedule requires minute from 0 to 59") + if self.kind == "weekly" and ( + not self.weekdays or any(day not in range(7) for day in self.weekdays) + ): + raise ValueError("weekly schedule requires weekdays from 0 to 6") + object.__setattr__(self, "weekdays", tuple(sorted(set(self.weekdays)))) + elif self.kind == "cron" and len(self.cron.split()) != 5: + raise ValueError("cron schedule requires a five-field expression") + + @classmethod + def from_dict(cls, data: dict[str, object]) -> Schedule: + kind = str(data.get("kind", data.get("type", ""))) + if kind not in {"once", "daily", "weekly", "cron"}: + raise ValueError(f"Unsupported schedule kind: {kind}") + run_at_value = data.get("runAt", data.get("run_at")) + weekdays_value = data.get("weekdays") or () + if not isinstance(weekdays_value, (list, tuple)): + raise TypeError("weekdays must be an array") + return cls( + kind=cast(ScheduleKind, kind), + timezone=str(_required(data, "timezone")), + run_at=_datetime(run_at_value, "runAt") if run_at_value else None, + hour=_int(data["hour"], "hour") if data.get("hour") is not None else None, + minute=( + _int(data["minute"], "minute") + if data.get("minute") is not None + else None + ), + weekdays=tuple(_int(day, "weekday") for day in weekdays_value), + cron=str(data.get("cron", data.get("expression", "")) or ""), + ) + + def to_dict(self) -> dict[str, object]: + return { + "kind": self.kind, + "timezone": self.timezone, + "runAt": _iso(self.run_at), + "hour": self.hour, + "minute": self.minute, + "weekdays": list(self.weekdays), + "cron": self.cron, + } + + +@dataclass(frozen=True) +class RuntimeTarget: + """Non-secret Runtime coordinates stored with a job.""" + + provider: ProviderName + runtime_id: str + agent_name: str + region: str + project_name: str = "default" + + def __post_init__(self) -> None: + if self.provider not in {"volcengine", "byteplus"}: + raise ValueError(f"Unsupported Runtime provider: {self.provider}") + if not all( + value.strip() + for value in ( + self.runtime_id, + self.agent_name, + self.region, + self.project_name, + ) + ): + raise ValueError("Runtime target fields must not be empty") + + @classmethod + def from_dict(cls, data: dict[str, object]) -> RuntimeTarget: + provider = str(_required(data, "provider")) + return cls( + provider=cast(ProviderName, provider), + runtime_id=str(_required(data, "runtimeId", "runtime_id")), + agent_name=str(_required(data, "agentName", "agent_name")), + region=str(_required(data, "region")), + project_name=str( + data.get("projectName", data.get("project_name", "default")) + ), + ) + + def to_dict(self) -> dict[str, object]: + return { + "provider": self.provider, + "runtimeId": self.runtime_id, + "agentName": self.agent_name, + "region": self.region, + "projectName": self.project_name, + } + + +@dataclass(frozen=True) +class CronJob: + """Durable job definition loaded from a user's TOS namespace.""" + + user_id: str + job_id: str + revision: int + enabled: bool + prompt: str + runtime: RuntimeTarget + schedule: Schedule + max_runtime_seconds: int = 3600 + + def __post_init__(self) -> None: + if not self.user_id or not self.job_id or not self.prompt.strip(): + raise ValueError("Cron job identity and prompt must not be empty") + if self.revision < 1: + raise ValueError("Cron job revision must be positive") + if not 60 <= self.max_runtime_seconds <= 86400: + raise ValueError("maxRuntimeSeconds must be between 60 and 86400") + + @classmethod + def from_dict( + cls, + data: dict[str, object], + *, + default_provider: ProviderName | None = None, + ) -> CronJob: + runtime = data.get("runtime") + schedule = _required(data, "schedule") + if not isinstance(schedule, dict): + raise TypeError("schedule must be an object") + if runtime is None: + provider = data.get("provider") or default_provider + if provider is None: + raise ValueError("Cron job is missing its Runtime provider") + runtime = { + "provider": provider, + "runtimeId": _required(data, "runtimeId", "runtime_id"), + "agentName": _required(data, "agentName", "agent_name"), + "region": _required(data, "region"), + "projectName": data.get("projectName", "default"), + } + if not isinstance(runtime, dict): + raise TypeError("runtime must be an object") + return cls( + user_id=str( + data.get("userId", data.get("user_id", data.get("ownerId", ""))) + ), + job_id=str(_required(data, "jobId", "job_id")), + revision=_int(_required(data, "revision"), "revision"), + enabled=bool(_required(data, "enabled")), + prompt=str(_required(data, "prompt")), + runtime=RuntimeTarget.from_dict(runtime), + schedule=Schedule.from_dict(schedule), + max_runtime_seconds=_int( + data.get("maxRuntimeSeconds", data.get("max_runtime_seconds", 3600)), + "maxRuntimeSeconds", + ), + ) + + def to_dict(self) -> dict[str, object]: + return { + "userId": self.user_id, + "jobId": self.job_id, + "revision": self.revision, + "enabled": self.enabled, + "prompt": self.prompt, + "runtime": self.runtime.to_dict(), + "schedule": self.schedule.to_dict(), + "maxRuntimeSeconds": self.max_runtime_seconds, + } + + +@dataclass(frozen=True) +class DuePointer: + """Minimal global pointer used to find jobs due in one minute.""" + + user_id: str + job_id: str + revision: int + scheduled_at: datetime + + def __post_init__(self) -> None: + if not self.user_id or not self.job_id or self.revision < 1: + raise ValueError("Due pointer identity and revision are invalid") + object.__setattr__( + self, + "scheduled_at", + _aware_utc(self.scheduled_at, "scheduledAt").replace( + second=0, microsecond=0 + ), + ) + + @classmethod + def from_dict(cls, data: dict[str, object]) -> DuePointer: + return cls( + user_id=str(_required(data, "userId", "user_id")), + job_id=str(_required(data, "jobId", "job_id")), + revision=_int(_required(data, "revision"), "revision"), + scheduled_at=_datetime( + _required(data, "scheduledAt", "scheduled_at"), "scheduledAt" + ), + ) + + def to_dict(self) -> dict[str, object]: + return { + "userId": self.user_id, + "jobId": self.job_id, + "revision": self.revision, + "scheduledAt": _iso(self.scheduled_at), + } + + +def deterministic_run_id(pointer: DuePointer) -> str: + """Return the same opaque run id for every delivery of one due pointer.""" + value = "\0".join( + ( + pointer.user_id, + pointer.job_id, + _iso(pointer.scheduled_at) or "", + ) + ) + return hashlib.sha256(value.encode("utf-8")).hexdigest() + + +@dataclass(frozen=True) +class ScheduledRun: + """Complete mutable-by-CAS execution state stored in TOS.""" + + user_id: str + job_id: str + run_id: str + revision: int + scheduled_at: datetime + session_id: str + state: RunState + created_at: datetime + updated_at: datetime + attempt: int = 0 + cancel_requested: bool = False + acknowledged: bool = False + runtime_version: str = "" + output: str = "" + error: str = "" + completed_at: datetime | None = None + + @classmethod + def from_dict(cls, data: dict[str, object]) -> ScheduledRun: + state = str(_required(data, "state")) + valid_states = { + "queued", + "preparing", + "running", + "retrying", + "succeeded", + "failed", + "cancelled", + "skipped", + } + if state not in valid_states: + raise ValueError(f"Unsupported run state: {state}") + completed = data.get("completedAt", data.get("completed_at")) + return cls( + user_id=str(_required(data, "userId", "user_id")), + job_id=str(_required(data, "jobId", "job_id")), + run_id=str(_required(data, "runId", "run_id")), + revision=_int(_required(data, "revision"), "revision"), + scheduled_at=_datetime( + _required(data, "scheduledAt", "scheduled_at"), "scheduledAt" + ), + session_id=str(_required(data, "sessionId", "session_id")), + state=cast(RunState, state), + created_at=_datetime( + _required(data, "createdAt", "created_at"), "createdAt" + ), + updated_at=_datetime( + _required(data, "updatedAt", "updated_at"), "updatedAt" + ), + attempt=_int(data.get("attempt") or 0, "attempt"), + cancel_requested=bool( + data.get("cancelRequested", data.get("cancel_requested", False)) + ), + acknowledged=bool(data.get("acknowledged", False)), + runtime_version=str( + data.get("runtimeVersion", data.get("runtime_version", "")) or "" + ), + output=str(data.get("output") or ""), + error=str(data.get("error") or ""), + completed_at=_datetime(completed, "completedAt") if completed else None, + ) + + def to_dict(self) -> dict[str, object]: + return { + "userId": self.user_id, + "jobId": self.job_id, + "runId": self.run_id, + "revision": self.revision, + "scheduledAt": _iso(self.scheduled_at), + "sessionId": self.session_id, + "state": self.state, + "createdAt": _iso(self.created_at), + "updatedAt": _iso(self.updated_at), + "attempt": self.attempt, + "cancelRequested": self.cancel_requested, + "acknowledged": self.acknowledged, + "runtimeVersion": self.runtime_version, + "output": self.output, + "error": self.error, + "completedAt": _iso(self.completed_at), + } + + +@dataclass(frozen=True) +class JobLock: + run_id: str + replica_id: str + state: Literal["held", "released"] + acquired_at: datetime + expires_at: datetime + released_at: datetime | None = None + + @classmethod + def from_dict(cls, data: dict[str, object]) -> JobLock: + state = str(_required(data, "state")) + if state not in {"held", "released"}: + raise ValueError(f"Unsupported lock state: {state}") + released = data.get("releasedAt") + return cls( + run_id=str(_required(data, "runId")), + replica_id=str(_required(data, "replicaId")), + state=cast(Literal["held", "released"], state), + acquired_at=_datetime(_required(data, "acquiredAt"), "acquiredAt"), + expires_at=_datetime(_required(data, "expiresAt"), "expiresAt"), + released_at=_datetime(released, "releasedAt") if released else None, + ) + + def to_dict(self) -> dict[str, object]: + return { + "runId": self.run_id, + "replicaId": self.replica_id, + "state": self.state, + "acquiredAt": _iso(self.acquired_at), + "expiresAt": _iso(self.expires_at), + "releasedAt": _iso(self.released_at), + } + + +@dataclass(frozen=True) +class LockAttempt: + acquired: bool + active_run_id: str = "" + abandoned_run_id: str = "" + + +@dataclass(frozen=True) +class ExecutionRequest: + run_id: str + session_id: str + user_id: str + job_id: str + prompt: str + runtime: RuntimeTarget + timeout_seconds: int + service_identity: bool = True + + +@dataclass(frozen=True) +class ExecutionResult: + output: str + runtime_version: str = "" + session_id: str = "" + + +class RuntimeInvocationError(RuntimeError): + """Runtime failure annotated with the only safe automatic-retry boundary.""" + + def __init__(self, message: str, *, acknowledged: bool, retryable: bool) -> None: + super().__init__(message) + self.acknowledged = acknowledged + self.retryable = retryable + + +@dataclass(frozen=True) +class DispatchSummary: + scanned: int = 0 + started: int = 0 + stale: int = 0 + skipped: int = 0 + failed: int = 0 diff --git a/frontend/service/studio_scheduler/ports.py b/frontend/service/studio_scheduler/ports.py new file mode 100644 index 000000000..49ced79c6 --- /dev/null +++ b/frontend/service/studio_scheduler/ports.py @@ -0,0 +1,97 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Provider-neutral dependency contracts for the scheduler core.""" + +from __future__ import annotations + +from datetime import datetime +from typing import Protocol + +from .models import ( + CronJob, + DuePointer, + ExecutionRequest, + ExecutionResult, + LockAttempt, + ProviderName, + ScheduledRun, +) + + +class CancellationControl(Protocol): + async def is_cancel_requested(self) -> bool: + """Return the durable cancellation flag for the current run.""" + ... + + +class RuntimeProvider(Protocol): + """Cloud adapter that invokes Runtime using its own service identity.""" + + provider: ProviderName + + async def execute( + self, request: ExecutionRequest, control: CancellationControl + ) -> ExecutionResult: ... + + +class RuntimeExecutor(Protocol): + async def execute( + self, request: ExecutionRequest, control: CancellationControl + ) -> ExecutionResult: ... + + +class SchedulerRepository(Protocol): + async def list_due(self, minute: datetime) -> list[DuePointer]: ... + + async def get_job(self, user_id: str, job_id: str) -> CronJob | None: ... + + async def put_due(self, pointer: DuePointer) -> bool: ... + + async def acquire_lock( + self, + *, + user_id: str, + job_id: str, + run_id: str, + replica_id: str, + now: datetime, + expires_at: datetime, + ) -> LockAttempt: ... + + async def release_lock( + self, + *, + user_id: str, + job_id: str, + run_id: str, + released_at: datetime, + ) -> None: ... + + async def create_run(self, run: ScheduledRun) -> bool: ... + + async def update_run(self, run: ScheduledRun) -> ScheduledRun: ... + + async def get_run( + self, *, user_id: str, job_id: str, run_id: str + ) -> ScheduledRun | None: ... + + async def request_cancel( + self, + *, + user_id: str, + job_id: str, + run_id: str, + requested_at: datetime, + ) -> ScheduledRun | None: ... diff --git a/frontend/service/studio_scheduler/publisher.py b/frontend/service/studio_scheduler/publisher.py new file mode 100644 index 000000000..60d7ca9de --- /dev/null +++ b/frontend/service/studio_scheduler/publisher.py @@ -0,0 +1,95 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Single publishing API used by create/edit and run-now backend operations.""" + +from __future__ import annotations + +from datetime import datetime, timedelta, timezone + +from .models import CronJob, DuePointer, deterministic_run_id +from .ports import SchedulerRepository +from .schedule import next_scheduled_time + + +class DuePublisher: + """Publish immutable minute pointers without duplicating key logic in the BFF.""" + + def __init__(self, repository: SchedulerRepository) -> None: + self._repository = repository + + async def publish(self, job: CronJob, scheduled_at: datetime) -> DuePointer: + return await self.publish_due( + user_id=job.user_id, + job_id=job.job_id, + revision=job.revision, + scheduled_at=scheduled_at, + ) + + async def publish_due( + self, + *, + user_id: str, + job_id: str, + revision: int, + scheduled_at: datetime, + ) -> DuePointer: + """Publish the four-field protocol consumed by the minute dispatcher.""" + pointer = DuePointer( + user_id=user_id, + job_id=job_id, + revision=revision, + scheduled_at=scheduled_at, + ) + await self._repository.put_due(pointer) + return pointer + + async def publish_next(self, job: CronJob, *, after: datetime) -> DuePointer | None: + scheduled_at = next_scheduled_time(job.schedule, after) + if scheduled_at is None: + return None + return await self.publish(job, scheduled_at) + + async def publish_run_now( + self, + job: CronJob | None = None, + *, + user_id: str = "", + job_id: str = "", + revision: int = 0, + scheduled_at: datetime | None = None, + now: datetime | None = None, + ) -> str: + """Publish a next-minute pointer and return its deterministic run id. + + The separate scheduler scans each minute exactly once. Queueing manual work + for the following minute avoids racing a scan that already happened. + """ + requested_at = scheduled_at or now or datetime.now(timezone.utc) + if requested_at.tzinfo is None or requested_at.utcoffset() is None: + raise ValueError("run-now time must include a timezone") + current = requested_at.astimezone(timezone.utc).replace( + second=0, + microsecond=0, + ) + timedelta(minutes=1) + if job is not None: + pointer = await self.publish(job, current) + else: + pointer = await self.publish_due( + user_id=user_id, + job_id=job_id, + revision=revision, + scheduled_at=current, + ) + return deterministic_run_id(pointer) diff --git a/frontend/service/studio_scheduler/runtime_provider.py b/frontend/service/studio_scheduler/runtime_provider.py new file mode 100644 index 000000000..4b0312108 --- /dev/null +++ b/frontend/service/studio_scheduler/runtime_provider.py @@ -0,0 +1,474 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""AgentKit control-plane resolution and cancellable Runtime SSE invocation.""" + +from __future__ import annotations + +import asyncio +import json +import os +from collections.abc import Callable +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Protocol +from urllib.parse import quote + +import httpx + +from .diagnostics import sanitize_diagnostic +from .models import ( + ExecutionRequest, + ExecutionResult, + ProviderName, + RuntimeInvocationError, + RuntimeTarget, +) +from .ports import CancellationControl + +_IAM_CREDENTIAL_PATH = Path("/var/run/secrets/iam/credential") + + +@dataclass(frozen=True) +class ServiceCredentials: + access_key: str + secret_key: str + session_token: str = "" + + +@dataclass(frozen=True) +class RuntimeConnection: + endpoint: str + api_key: str + runtime_version: str + + +class RuntimeConnectionResolver(Protocol): + async def resolve(self, target: RuntimeTarget) -> RuntimeConnection: ... + + +def resolve_service_credentials(provider: ProviderName) -> ServiceCredentials: + """Resolve provider-specific service credentials, falling back to VeFaaS IAM.""" + if provider == "byteplus": + access_key = os.getenv("BYTEPLUS_ACCESS_KEY", "").strip() + secret_key = os.getenv("BYTEPLUS_SECRET_KEY", "").strip() + session_token = os.getenv("BYTEPLUS_SESSION_TOKEN", "").strip() + else: + access_key = os.getenv("VOLCENGINE_ACCESS_KEY", "").strip() + secret_key = os.getenv("VOLCENGINE_SECRET_KEY", "").strip() + session_token = os.getenv("VOLCENGINE_SESSION_TOKEN", "").strip() + if bool(access_key) != bool(secret_key): + raise ValueError( + f"{provider} access key and secret key must be configured together" + ) + if access_key: + return ServiceCredentials(access_key, secret_key, session_token) + if not _IAM_CREDENTIAL_PATH.is_file(): + raise FileNotFoundError("VeFaaS service identity is unavailable") + payload = json.loads(_IAM_CREDENTIAL_PATH.read_text(encoding="utf-8")) + access_key = payload.get("access_key_id") or payload.get("AccessKeyId") + secret_key = payload.get("secret_access_key") or payload.get("SecretAccessKey") + session_token = payload.get("session_token") or payload.get("SessionToken") or "" + if not access_key or not secret_key: + raise ValueError("VeFaaS service identity credential is incomplete") + return ServiceCredentials( + access_key=str(access_key), + secret_key=str(secret_key), + session_token=str(session_token), + ) + + +class AgentKitRuntimeConnectionResolver: + """Read a Runtime endpoint, key, and current version with service identity.""" + + def __init__( + self, + *, + credentials_resolver: Callable[[ProviderName], ServiceCredentials] + | None = None, + runtime_loader: Callable[[RuntimeTarget, ServiceCredentials], Any] + | None = None, + ) -> None: + self._credentials_resolver = credentials_resolver or resolve_service_credentials + self._runtime_loader = runtime_loader or _load_runtime + + async def resolve(self, target: RuntimeTarget) -> RuntimeConnection: + credentials: ServiceCredentials | None = None + try: + credentials = self._credentials_resolver(target.provider) + runtime = await asyncio.to_thread( + self._runtime_loader, + target, + credentials, + ) + return _connection_from_runtime(runtime) + except RuntimeInvocationError: + raise + except Exception as error: + detail = sanitize_diagnostic( + error, + secrets=( + credentials.access_key if credentials else "", + credentials.secret_key if credentials else "", + credentials.session_token if credentials else "", + ), + ) + raise RuntimeInvocationError( + "Unable to resolve Runtime service connection" + + (f". Detail: {detail}" if detail else "") + + ". Check the Runtime ID, region, and scheduler service identity.", + acknowledged=False, + retryable=False, + ) from error + + +def _load_runtime(target: RuntimeTarget, credentials: ServiceCredentials) -> Any: + from agentkit.sdk.runtime import types + from agentkit.sdk.runtime.client import AgentkitRuntimeClient + + client = AgentkitRuntimeClient( + access_key=credentials.access_key, + secret_key=credentials.secret_key, + session_token=credentials.session_token, + region=target.region, + ) + return client.get_runtime(types.GetRuntimeRequest(RuntimeId=target.runtime_id)) + + +def _connection_from_runtime(runtime: Any) -> RuntimeConnection: + endpoint = "" + fallback = "" + for network in getattr(runtime, "network_configurations", None) or (): + candidate = str(getattr(network, "endpoint", "") or "").rstrip("/") + if not candidate: + continue + fallback = fallback or candidate + if str(getattr(network, "network_type", "") or "") == "public": + endpoint = candidate + break + endpoint = endpoint or fallback or str(getattr(runtime, "endpoint", "") or "") + authorizer = getattr(runtime, "authorizer_configuration", None) + key_auth = getattr(authorizer, "key_auth", None) if authorizer else None + api_key = str(getattr(key_auth, "api_key", "") or "") + if not endpoint: + raise RuntimeInvocationError( + "Runtime has no reachable endpoint", + acknowledged=False, + retryable=False, + ) + if not api_key: + raise RuntimeInvocationError( + "Runtime does not expose service key authentication", + acknowledged=False, + retryable=False, + ) + version = str(getattr(runtime, "current_version_number", "") or "") + return RuntimeConnection(endpoint.rstrip("/"), api_key, version) + + +class AgentKitRuntimeProvider: + """Invoke one Runtime through its ADK HTTP API with cancellation polling.""" + + def __init__( + self, + provider: ProviderName, + resolver: RuntimeConnectionResolver, + *, + transport: httpx.AsyncBaseTransport | None = None, + cancel_poll_seconds: float = 1.0, + ) -> None: + if cancel_poll_seconds <= 0: + raise ValueError("cancel_poll_seconds must be positive") + self.provider: ProviderName = provider + self._resolver = resolver + self._transport = transport + self._cancel_poll_seconds = cancel_poll_seconds + + async def execute( + self, + request: ExecutionRequest, + control: CancellationControl, + ) -> ExecutionResult: + connection = await self._resolver.resolve(request.runtime) + headers = { + "Accept": "application/json", + "Authorization": f"Bearer {connection.api_key}", + } + timeout = httpx.Timeout(float(request.timeout_seconds), connect=10.0) + try: + async with httpx.AsyncClient( + timeout=timeout, + transport=self._transport, + ) as client: + session_id = await self._create_session( + client, + connection, + request, + headers, + ) + output = await self._run_sse( + client, + connection, + request, + session_id, + headers, + control, + ) + except RuntimeInvocationError: + raise + except (httpx.ConnectError, httpx.ConnectTimeout) as error: + raise RuntimeInvocationError( + "Runtime connection failed before acknowledgement", + acknowledged=False, + retryable=True, + ) from error + except httpx.HTTPError as error: + raise RuntimeInvocationError( + "Runtime stream failed after request delivery", + acknowledged=True, + retryable=False, + ) from error + return ExecutionResult( + output=output, + runtime_version=connection.runtime_version, + session_id=session_id, + ) + + async def _create_session( + self, + client: httpx.AsyncClient, + connection: RuntimeConnection, + request: ExecutionRequest, + headers: dict[str, str], + ) -> str: + app = quote(request.runtime.agent_name, safe="") + user = quote(request.user_id, safe="") + response = await client.post( + f"{connection.endpoint}/apps/{app}/users/{user}/sessions", + headers={**headers, "Content-Type": "application/json"}, + json={"id": request.session_id}, + ) + _raise_for_runtime_status( + response, + phase="create session", + acknowledged=False, + secrets=(connection.api_key,), + ) + try: + payload = response.json() + except ValueError as error: + raise RuntimeInvocationError( + "Runtime returned an invalid session response", + acknowledged=False, + retryable=False, + ) from error + if not isinstance(payload, dict) or not str(payload.get("id") or ""): + raise RuntimeInvocationError( + "Runtime returned an invalid session identity", + acknowledged=False, + retryable=False, + ) + return str(payload["id"]) + + async def _run_sse( + self, + client: httpx.AsyncClient, + connection: RuntimeConnection, + request: ExecutionRequest, + session_id: str, + headers: dict[str, str], + control: CancellationControl, + ) -> str: + payload = { + "app_name": request.runtime.agent_name, + "user_id": request.user_id, + "session_id": session_id, + "new_message": { + "role": "user", + "parts": [{"text": request.prompt}], + }, + "streaming": True, + } + async with client.stream( + "POST", + f"{connection.endpoint}/run_sse", + headers={**headers, "Content-Type": "application/json"}, + json=payload, + ) as response: + if response.status_code >= 400: + await response.aread() + _raise_for_runtime_status( + response, + phase="run", + acknowledged=False, + secrets=(connection.api_key,), + ) + return await self._read_sse( + response, + control, + secrets=(connection.api_key,), + ) + + async def _read_sse( + self, + response: httpx.Response, + control: CancellationControl, + *, + secrets: tuple[str, ...] = (), + ) -> str: + lines = response.aiter_lines() + pending: asyncio.Future[str] | None = None + final_text = "" + try: + while True: + if await control.is_cancel_requested(): + await response.aclose() + raise RuntimeInvocationError( + "Runtime execution was cancelled", + acknowledged=True, + retryable=False, + ) + pending = pending or asyncio.ensure_future(anext(lines)) + done, _ = await asyncio.wait( + {pending}, + timeout=self._cancel_poll_seconds, + ) + if not done: + continue + try: + line = pending.result() + except StopAsyncIteration: + break + finally: + pending = None + event = _parse_sse_line(line) + if event is None: + continue + error = _event_error(event) + if error: + raise RuntimeInvocationError( + "Runtime stream reported an error. Detail: " + f"{sanitize_diagnostic(error, secrets=secrets)}. " + "Check the Runtime logs, Agent configuration, and model access.", + acknowledged=True, + retryable=False, + ) + text = _event_text(event) + if text and not bool(event.get("partial")): + final_text = text + finally: + if pending is not None and not pending.done(): + pending.cancel() + return final_text + + +def _raise_for_runtime_status( + response: httpx.Response, + *, + phase: str, + acknowledged: bool, + secrets: tuple[str, ...] = (), +) -> None: + if response.status_code < 400: + return + status_code = response.status_code + retryable = status_code in {502, 503, 504} + content_type = response.headers.get("content-type", "").split(";", 1)[0].strip() + detail = _runtime_response_detail(response, secrets=secrets) + if retryable: + recovery = ( + "The scheduler will retry automatically because the Runtime has not " + "acknowledged the request." + ) + elif status_code in {401, 403}: + recovery = "Check the Runtime service key and scheduler service identity." + elif status_code == 404: + recovery = "Check the Runtime Agent name, endpoint, and deployed version." + else: + recovery = "Check the Runtime logs, Agent configuration, and model access, then retry this run." + parts = [f"Runtime {phase} returned HTTP {status_code}."] + if detail: + parts.append(f"Detail: {detail}.") + if content_type and content_type != "application/json": + parts.append(f"Content-Type: {content_type}.") + parts.append(recovery) + raise RuntimeInvocationError( + " ".join(parts), + acknowledged=acknowledged, + retryable=retryable, + ) + + +def _runtime_response_detail( + response: httpx.Response, + *, + secrets: tuple[str, ...], +) -> str: + content_type = response.headers.get("content-type", "").lower() + detail: object = "" + if "json" in content_type: + try: + payload = response.json() + except ValueError: + payload = None + if isinstance(payload, dict): + detail = ( + payload.get("detail") + or payload.get("error") + or payload.get("message") + or payload + ) + elif payload is not None: + detail = payload + if not detail: + detail = response.text + if isinstance(detail, (dict, list)): + detail = json.dumps(detail, ensure_ascii=False, separators=(",", ":")) + return sanitize_diagnostic(detail, secrets=secrets) + + +def _parse_sse_line(line: str) -> dict[str, Any] | None: + if not line.startswith("data:"): + return None + data = line[5:].strip() + if not data or data == "[DONE]": + return None + try: + value = json.loads(data) + except json.JSONDecodeError: + return None + return value if isinstance(value, dict) else None + + +def _event_error(event: dict[str, Any]) -> str: + return str( + event.get("error") + or event.get("errorMessage") + or event.get("error_message") + or "" + ) + + +def _event_text(event: dict[str, Any]) -> str: + content = event.get("content") + if not isinstance(content, dict): + return "" + parts = content.get("parts") + if not isinstance(parts, list): + return "" + return "".join( + str(part.get("text") or "") + for part in parts + if isinstance(part, dict) and part.get("text") is not None + ) diff --git a/frontend/service/studio_scheduler/schedule.py b/frontend/service/studio_scheduler/schedule.py new file mode 100644 index 000000000..31ac35f6d --- /dev/null +++ b/frontend/service/studio_scheduler/schedule.py @@ -0,0 +1,133 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Pure next-occurrence calculation without a resident scheduler process.""" + +from __future__ import annotations + +from collections.abc import Callable +from datetime import datetime, timedelta, timezone +from zoneinfo import ZoneInfo + +from .models import Schedule + +_SEARCH_LIMIT_MINUTES = 366 * 24 * 60 * 5 + + +def next_scheduled_time(schedule: Schedule, after: datetime) -> datetime | None: + """Return the first scheduled UTC minute strictly after ``after``.""" + if after.tzinfo is None or after.utcoffset() is None: + raise ValueError("after must include a timezone") + after = after.astimezone(timezone.utc).replace(second=0, microsecond=0) + if schedule.kind == "once": + return schedule.run_at if schedule.run_at and schedule.run_at > after else None + + zone = ZoneInfo(schedule.timezone) + candidate = after + timedelta(minutes=1) + if schedule.kind == "daily": + return _search(candidate, zone, schedule, lambda local: True) + if schedule.kind == "weekly": + return _search( + candidate, + zone, + schedule, + lambda local: local.weekday() in schedule.weekdays, + ) + + minute, hour, month_day, month, week_day = _parse_cron(schedule.cron) + + def cron_matches(local: datetime) -> bool: + cron_weekday = (local.weekday() + 1) % 7 + day_matches = local.day in month_day + weekday_matches = cron_weekday in week_day + if len(month_day) != 31 and len(week_day) != 7: + calendar_day_matches = day_matches or weekday_matches + else: + calendar_day_matches = day_matches and weekday_matches + return ( + local.minute in minute + and local.hour in hour + and local.month in month + and calendar_day_matches + ) + + for _ in range(_SEARCH_LIMIT_MINUTES): + if cron_matches(candidate.astimezone(zone)): + return candidate + candidate += timedelta(minutes=1) + raise ValueError("cron expression has no occurrence within five years") + + +def _search( + candidate: datetime, + zone: ZoneInfo, + schedule: Schedule, + day_matches: Callable[[datetime], bool], +) -> datetime: + for _ in range(366 * 24 * 60 * 2): + local = candidate.astimezone(zone) + if ( + local.hour == schedule.hour + and local.minute == schedule.minute + and day_matches(local) + ): + return candidate + candidate += timedelta(minutes=1) + raise ValueError("schedule has no occurrence within two years") + + +def _parse_cron( + expression: str, +) -> tuple[set[int], set[int], set[int], set[int], set[int]]: + fields = expression.split() + if len(fields) != 5: + raise ValueError("cron expression must have five fields") + minute = _parse_field(fields[0], 0, 59, "minute") + hour = _parse_field(fields[1], 0, 23, "hour") + month_day = _parse_field(fields[2], 1, 31, "day of month") + month = _parse_field(fields[3], 1, 12, "month") + week_day = _parse_field(fields[4], 0, 7, "day of week") + if 7 in week_day: + week_day.remove(7) + week_day.add(0) + return minute, hour, month_day, month, week_day + + +def _parse_field(field: str, minimum: int, maximum: int, name: str) -> set[int]: + values: set[int] = set() + for part in field.split(","): + base, separator, step_text = part.partition("/") + try: + step = int(step_text) if separator else 1 + except ValueError as error: + raise ValueError(f"cron {name} step is invalid") from error + if step < 1: + raise ValueError(f"cron {name} step must be positive") + if base == "*": + start, end = minimum, maximum + elif "-" in base: + start_text, end_text = base.split("-", 1) + try: + start, end = int(start_text), int(end_text) + except ValueError as error: + raise ValueError(f"cron {name} range is invalid") from error + else: + try: + start = end = int(base) + except ValueError as error: + raise ValueError(f"cron {name} value is invalid") from error + if start < minimum or end > maximum or start > end: + raise ValueError(f"cron {name} value is outside {minimum}-{maximum}") + values.update(range(start, end + 1, step)) + return values diff --git a/frontend/service/studio_scheduler/tos_repository.py b/frontend/service/studio_scheduler/tos_repository.py new file mode 100644 index 000000000..27af2e9f0 --- /dev/null +++ b/frontend/service/studio_scheduler/tos_repository.py @@ -0,0 +1,457 @@ +# Copyright (c) 2025 Beijing Volcano Engine Technology Co., Ltd. and/or its affiliates. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""TOS repository using atomic create and ETag compare-and-set operations.""" + +from __future__ import annotations + +import asyncio +import hashlib +import json +from collections.abc import Callable +from dataclasses import replace +from datetime import datetime, timezone +from typing import Any +from urllib.parse import quote + +from frontend.server.storage import STUDIO_STORAGE_ROOT_PREFIX + +from .models import ( + CronJob, + DuePointer, + JobLock, + LockAttempt, + ProviderName, + ScheduledRun, +) + +_MAX_JSON_BYTES = 1024 * 1024 +_CAS_ATTEMPTS = 5 + + +class SchedulerStorageConflict(RuntimeError): + """An immutable scheduler object already exists with different data.""" + + +class TosSchedulerRepository: + """Persist jobs, due pointers, locks, and run state independently of instances.""" + + def __init__( + self, + *, + bucket: str, + client_factory: Callable[[], Any], + provider: ProviderName = "volcengine", + ) -> None: + if not bucket.strip(): + raise ValueError("TOS scheduler storage requires a bucket") + self._bucket = bucket + self._client_factory = client_factory + self._provider: ProviderName = provider + + async def list_due(self, minute: datetime) -> list[DuePointer]: + return await asyncio.to_thread(self._list_due, minute) + + def _list_due(self, minute: datetime) -> list[DuePointer]: + client = self._client_factory() + prefix = self._due_prefix(minute) + return [ + DuePointer.from_dict(self._get_json(client, key)[0]) + for key in self._list_keys(client, prefix) + ] + + async def get_job(self, user_id: str, job_id: str) -> CronJob | None: + return await asyncio.to_thread(self._get_job, user_id, job_id) + + def _get_job(self, user_id: str, job_id: str) -> CronJob | None: + client = self._client_factory() + try: + payload, _ = self._get_json(client, self._job_key(user_id, job_id)) + except Exception as error: + if _status_code(error) == 404: + return None + raise + return CronJob.from_dict(payload, default_provider=self._provider) + + async def put_due(self, pointer: DuePointer) -> bool: + return await asyncio.to_thread(self._put_due, pointer) + + def _put_due(self, pointer: DuePointer) -> bool: + client = self._client_factory() + key = self._due_key(pointer) + content = _json_bytes(pointer.to_dict()) + try: + self._put(client, key, content, forbid_overwrite=True) + return True + except Exception as error: + if _status_code(error) not in {409, 412}: + raise + for _ in range(_CAS_ATTEMPTS): + existing_payload, etag = self._get_json(client, key) + existing = DuePointer.from_dict(existing_payload) + if existing == pointer: + return False + same_occurrence = ( + existing.user_id == pointer.user_id + and existing.job_id == pointer.job_id + and existing.scheduled_at == pointer.scheduled_at + ) + if not same_occurrence: + raise SchedulerStorageConflict( + f"Due pointer already exists with different identity: {key}" + ) + if existing.revision > pointer.revision: + return False + try: + self._put(client, key, content, if_match=etag) + return True + except Exception as error: + if _status_code(error) not in {409, 412}: + raise + raise SchedulerStorageConflict( + f"Unable to update due pointer after CAS retries: {key}" + ) + + async def acquire_lock( + self, + *, + user_id: str, + job_id: str, + run_id: str, + replica_id: str, + now: datetime, + expires_at: datetime, + ) -> LockAttempt: + return await asyncio.to_thread( + self._acquire_lock, + user_id, + job_id, + run_id, + replica_id, + now, + expires_at, + ) + + def _acquire_lock( + self, + user_id: str, + job_id: str, + run_id: str, + replica_id: str, + now: datetime, + expires_at: datetime, + ) -> LockAttempt: + client = self._client_factory() + key = self._lock_key(user_id, job_id) + requested = JobLock( + run_id=run_id, + replica_id=replica_id, + state="held", + acquired_at=now, + expires_at=expires_at, + ) + content = _json_bytes(requested.to_dict()) + try: + self._put(client, key, content, forbid_overwrite=True) + return LockAttempt(acquired=True) + except Exception as error: + if _status_code(error) not in {409, 412}: + raise + + for _ in range(_CAS_ATTEMPTS): + payload, etag = self._get_json(client, key) + existing = JobLock.from_dict(payload) + if existing.state == "held" and existing.expires_at > now: + return LockAttempt( + acquired=False, + active_run_id=existing.run_id, + ) + try: + self._put(client, key, content, if_match=etag) + abandoned = existing.run_id if existing.state == "held" else "" + return LockAttempt(acquired=True, abandoned_run_id=abandoned) + except Exception as error: + if _status_code(error) not in {409, 412}: + raise + raise SchedulerStorageConflict( + "Unable to acquire cron job lock after CAS retries" + ) + + async def release_lock( + self, + *, + user_id: str, + job_id: str, + run_id: str, + released_at: datetime, + ) -> None: + await asyncio.to_thread( + self._release_lock, user_id, job_id, run_id, released_at + ) + + def _release_lock( + self, user_id: str, job_id: str, run_id: str, released_at: datetime + ) -> None: + client = self._client_factory() + key = self._lock_key(user_id, job_id) + for _ in range(_CAS_ATTEMPTS): + try: + payload, etag = self._get_json(client, key) + except Exception as error: + if _status_code(error) == 404: + return + raise + existing = JobLock.from_dict(payload) + if existing.run_id != run_id or existing.state != "held": + return + released = replace( + existing, + state="released", + released_at=released_at, + ) + try: + self._put( + client, + key, + _json_bytes(released.to_dict()), + if_match=etag, + ) + return + except Exception as error: + if _status_code(error) not in {409, 412}: + raise + raise SchedulerStorageConflict( + "Unable to release cron job lock after CAS retries" + ) + + async def create_run(self, run: ScheduledRun) -> bool: + return await asyncio.to_thread(self._create_run, run) + + def _create_run(self, run: ScheduledRun) -> bool: + client = self._client_factory() + key = self._run_key(run.user_id, run.job_id, run.run_id) + content = _json_bytes(run.to_dict()) + try: + self._put(client, key, content, forbid_overwrite=True) + return True + except Exception as error: + if _status_code(error) not in {409, 412}: + raise + existing, _ = self._get_bytes(client, key) + existing_run = ScheduledRun.from_dict(json.loads(existing)) + if ( + existing_run.user_id, + existing_run.job_id, + existing_run.run_id, + existing_run.scheduled_at, + ) != (run.user_id, run.job_id, run.run_id, run.scheduled_at): + raise SchedulerStorageConflict( + f"Run id already exists with different identity: {run.run_id}" + ) from error + return False + + async def get_run( + self, *, user_id: str, job_id: str, run_id: str + ) -> ScheduledRun | None: + return await asyncio.to_thread(self._get_run, user_id, job_id, run_id) + + def _get_run(self, user_id: str, job_id: str, run_id: str) -> ScheduledRun | None: + client = self._client_factory() + try: + payload, _ = self._get_json(client, self._run_key(user_id, job_id, run_id)) + except Exception as error: + if _status_code(error) == 404: + return None + raise + return ScheduledRun.from_dict(payload) + + async def update_run(self, run: ScheduledRun) -> ScheduledRun: + return await asyncio.to_thread(self._update_run, run) + + def _update_run(self, run: ScheduledRun) -> ScheduledRun: + client = self._client_factory() + key = self._run_key(run.user_id, run.job_id, run.run_id) + for _ in range(_CAS_ATTEMPTS): + payload, etag = self._get_json(client, key) + existing = ScheduledRun.from_dict(payload) + merged = replace( + run, + cancel_requested=run.cancel_requested or existing.cancel_requested, + ) + try: + self._put(client, key, _json_bytes(merged.to_dict()), if_match=etag) + return merged + except Exception as error: + if _status_code(error) not in {409, 412}: + raise + raise SchedulerStorageConflict("Unable to update cron run after CAS retries") + + async def request_cancel( + self, + *, + user_id: str, + job_id: str, + run_id: str, + requested_at: datetime, + ) -> ScheduledRun | None: + return await asyncio.to_thread( + self._request_cancel, + user_id, + job_id, + run_id, + requested_at, + ) + + def _request_cancel( + self, + user_id: str, + job_id: str, + run_id: str, + requested_at: datetime, + ) -> ScheduledRun | None: + client = self._client_factory() + key = self._run_key(user_id, job_id, run_id) + for _ in range(_CAS_ATTEMPTS): + try: + payload, etag = self._get_json(client, key) + except Exception as error: + if _status_code(error) == 404: + return None + raise + existing = ScheduledRun.from_dict(payload) + if existing.cancel_requested: + return existing + updated = replace( + existing, + cancel_requested=True, + updated_at=requested_at, + ) + try: + self._put(client, key, _json_bytes(updated.to_dict()), if_match=etag) + return updated + except Exception as error: + if _status_code(error) not in {409, 412}: + raise + raise SchedulerStorageConflict("Unable to cancel cron run after CAS retries") + + def _list_keys(self, client: Any, prefix: str) -> list[str]: + continuation_token = "" + keys: list[str] = [] + while True: + output = client.list_objects_type2( + bucket=self._bucket, + prefix=prefix, + continuation_token=continuation_token, + max_keys=1000, + ) + keys.extend( + str(item.key) + for item in (getattr(output, "contents", None) or []) + if str(getattr(item, "key", "")).endswith(".json") + ) + if not getattr(output, "is_truncated", False): + return keys + continuation_token = str( + getattr(output, "next_continuation_token", "") or "" + ) + if not continuation_token: + raise RuntimeError("TOS returned a truncated listing without a token") + + def _get_json(self, client: Any, key: str) -> tuple[dict[str, object], str]: + content, etag = self._get_bytes(client, key) + payload = json.loads(content) + if not isinstance(payload, dict): + raise TypeError(f"Scheduler object must be a JSON object: {key}") + return payload, etag + + def _get_bytes(self, client: Any, key: str) -> tuple[bytes, str]: + response = client.get_object(bucket=self._bucket, key=key) + content = response.read() if hasattr(response, "read") else b"".join(response) + if len(content) > _MAX_JSON_BYTES: + raise ValueError(f"Scheduler object is too large: {key}") + etag = str(getattr(response, "etag", "") or "").strip('"') + if not etag: + raise ValueError(f"Scheduler object is missing an ETag: {key}") + return content, etag + + def _put( + self, + client: Any, + key: str, + content: bytes, + *, + forbid_overwrite: bool | None = None, + if_match: str | None = None, + ) -> None: + client.put_object( + bucket=self._bucket, + key=key, + content=content, + content_type="application/json", + forbid_overwrite=forbid_overwrite, + if_match=if_match, + ) + + @staticmethod + def _user_job_prefix(user_id: str, job_id: str) -> str: + return ( + f"{STUDIO_STORAGE_ROOT_PREFIX}/users/{quote(user_id, safe='')}/" + f"cronjobs/{quote(job_id, safe='')}" + ) + + @classmethod + def _job_key(cls, user_id: str, job_id: str) -> str: + return f"{cls._user_job_prefix(user_id, job_id)}/job.json" + + @classmethod + def _lock_key(cls, user_id: str, job_id: str) -> str: + return f"{cls._user_job_prefix(user_id, job_id)}/lock.json" + + @classmethod + def _run_key(cls, user_id: str, job_id: str, run_id: str) -> str: + return ( + f"{cls._user_job_prefix(user_id, job_id)}/runs/" + f"{quote(run_id, safe='')}.json" + ) + + @staticmethod + def _due_prefix(minute: datetime) -> str: + # Due buckets are always named with their UTC minute, independent of host TZ. + stamp = minute.astimezone(timezone.utc) + return ( + f"{STUDIO_STORAGE_ROOT_PREFIX}/scheduler/cronjobs/due/{stamp:%Y%m%d%H%M}/" + ) + + @classmethod + def _due_key(cls, pointer: DuePointer) -> str: + identity = ( + f"{pointer.user_id}\0{pointer.job_id}\0{pointer.scheduled_at.isoformat()}" + ) + name = hashlib.sha256(identity.encode("utf-8")).hexdigest() + return f"{cls._due_prefix(pointer.scheduled_at)}{name}.json" + + +def _json_bytes(payload: dict[str, object]) -> bytes: + return ( + json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")) + + "\n" + ).encode("utf-8") + + +def _status_code(error: Exception) -> int | None: + value = getattr(error, "status_code", None) + try: + return int(value) if value is not None else None + except (TypeError, ValueError): + return None diff --git a/frontend/src/App.tsx b/frontend/src/App.tsx index 910fec59f..007b7ef66 100644 --- a/frontend/src/App.tsx +++ b/frontend/src/App.tsx @@ -102,6 +102,7 @@ import { type MyAgentCardData, } from "./ui/MyAgents"; import { Applications, type ApplicationId } from "./ui/Applications"; +import { CronJobs } from "./cronjobs/CronJobs"; import { getAutomation } from "./automations/registry"; import { SystemInfo } from "./ui/SystemInfo"; import { GitHubIntegration } from "./ui/GitHubIntegration"; @@ -358,6 +359,7 @@ type StudioPageId = | "library" | "search" | "applications" + | "cronjobs" | "agents" | "agent-detail" | "sandbox-agent-detail" @@ -1991,6 +1993,7 @@ export default function App() { }, []); const [applicationsView, setApplicationsView] = useState<"catalog" | ApplicationId | null>(null); + const [cronJobsView, setCronJobsView] = useState(false); // A search result may belong to a different agent; remember it so the // agent-switch effect opens it instead of resetting to a fresh chat. const pendingOpenRef = useRef<{ app: string; sid: string } | null>(null); @@ -2989,6 +2992,8 @@ export default function App() { documentTitleTarget = { kind: "page", title: "问题反馈" }; } else if (systemInfo) { documentTitleTarget = { kind: "page", title: "系统信息" }; + } else if (cronJobsView) { + documentTitleTarget = { kind: "page", title: "定时任务" }; } else if (applicationsView) { documentTitleTarget = { kind: "page", @@ -3637,6 +3642,7 @@ export default function App() { setMyAgents(false); setPageStack([]); setApplicationsView(null); + setCronJobsView(false); setSandboxAgentDetailTarget(null); setSandboxAgentWorkspace(null); } @@ -4559,6 +4565,7 @@ export default function App() { setMyAgents(false); setPageStack([]); setApplicationsView(null); + setCronJobsView(false); startNewChat(); } @@ -5752,6 +5759,7 @@ export default function App() { setMyAgents(true); setPageStack([]); setApplicationsView(null); + setCronJobsView(false); setError(""); }; @@ -5771,10 +5779,32 @@ export default function App() { setSandboxAgentWorkspace(null); setMyAgents(false); setPageStack([]); + setCronJobsView(false); setApplicationsView("catalog"); setError(""); }; + const openCronJobsPage = () => { + setPlatformFeedbackOrigin(null); + if (sandboxSession) exitSandboxSession(); + viewSidRef.current = ""; + setSessionId(""); + setCreateView(null); + setSkillCenter(false); + setAddAgent(false); + setAddMenu(false); + setSearchView(false); + setManageAgents(false); + setAgentDetailTarget(null); + setSandboxAgentDetailTarget(null); + setSandboxAgentWorkspace(null); + setMyAgents(false); + setPageStack([]); + setApplicationsView(null); + setCronJobsView(true); + setError(""); + }; + const talkToWorkspaceAgent = async (agent: AgentEntry) => { setFeedbackCaseReturnAgentId(""); setFeedbackTargetEventId(""); @@ -5832,37 +5862,41 @@ export default function App() { ? "feedback" : skillCenter ? "library" - : applicationsView - ? "applications" - : searchView - ? "search" - : sandboxAgentWorkspace - ? "sandbox-agent-workspace" - : myAgents || manageAgents - ? "agents" - : sandboxSession - ? "sandbox" - : sessionId - ? "conversation" - : createView || addAgent || addMenu - ? "create" - : "new-chat"; + : cronJobsView + ? "cronjobs" + : applicationsView + ? "applications" + : searchView + ? "search" + : sandboxAgentWorkspace + ? "sandbox-agent-workspace" + : myAgents || manageAgents + ? "agents" + : sandboxSession + ? "sandbox" + : sessionId + ? "conversation" + : createView || addAgent || addMenu + ? "create" + : "new-chat"; const sidebarActivePage: SidebarPage = systemInfo ? null : platformFeedbackOrigin !== null ? "feedback" : skillCenter - ? "library" - : applicationsView - ? "applications" - : searchView - ? "search" - : myAgents || manageAgents || sandboxAgentDetailTarget || sandboxAgentWorkspace - ? "agents" - : sessionId || sandboxSession || createView || skillCenter || addAgent || addMenu - ? null - : "new-chat"; + ? "library" + : cronJobsView + ? "cronjobs" + : applicationsView + ? "applications" + : searchView + ? "search" + : myAgents || manageAgents || sandboxAgentDetailTarget || sandboxAgentWorkspace + ? "agents" + : sessionId || sandboxSession || createView || skillCenter || addAgent || addMenu + ? null + : "new-chat"; return (
@@ -5922,6 +5956,7 @@ export default function App() { setMyAgents(false); setPageStack([]); setApplicationsView(null); + setCronJobsView(false); setSearchView(true); setError(""); })} @@ -5944,6 +5979,7 @@ export default function App() { setMyAgents(false); setPageStack([]); setApplicationsView(null); + setCronJobsView(false); setCreateView(null); setImportedDraft(null); setNewRuntimeRegion(defaultCloudRegion(cloudProvider)); @@ -5963,6 +5999,7 @@ export default function App() { setMyAgents(false); setPageStack([]); setApplicationsView(null); + setCronJobsView(false); setSkillCenterLaunch(null); setLibraryTab("skills"); setLibraryPageTitle("技能库"); @@ -5986,6 +6023,7 @@ export default function App() { setMyAgents(false); setPageStack([]); setApplicationsView(null); + setCronJobsView(false); setSessionId(""); setAddMenu(false); setAddAgent(true); @@ -5993,6 +6031,7 @@ export default function App() { })} onMyAgents={() => requestIntelligentNavigation(openMyAgentsPage)} onApplications={() => requestIntelligentNavigation(openApplicationsPage)} + onCronJobs={() => requestIntelligentNavigation(openCronJobsPage)} onSystemInfo={() => requestIntelligentNavigation(() => { pushStudioPage({ page: "system-info", @@ -6027,6 +6066,7 @@ export default function App() { setMyAgents(false); setPageStack([]); setApplicationsView(null); + setCronJobsView(false); setError(""); pickSession(id); })} @@ -6359,6 +6399,8 @@ export default function App() { initialModule={issueFeedbackModuleForPage(platformFeedbackOrigin)} onSubmit={submitPlatformIssueFeedback} /> + ) : cronJobsView ? ( + ) : applicationsView === "coding-agents" ? ( setApplicationsView("catalog")} diff --git a/frontend/src/adk/client.ts b/frontend/src/adk/client.ts index 25aad9e19..14049cd40 100644 --- a/frontend/src/adk/client.ts +++ b/frontend/src/adk/client.ts @@ -3158,6 +3158,189 @@ export interface CloudRuntime { canDelete: boolean; } +export type CronJobScheduleType = "once" | "daily" | "weekly" | "cron"; + +export interface CronJobSchedule { + type: CronJobScheduleType; + timezone: string; + /** ISO local date-time for a one-time schedule. */ + onceAt?: string; + /** HH:mm for daily and weekly schedules. */ + time?: string; + /** 0 (Sunday) through 6 (Saturday), for weekly schedules. */ + weekday?: number; + /** Five-field cron expression for custom schedules. */ + cron?: string; +} + +export type CronJobRunStatus = + | "queued" + | "pending" + | "running" + | "retrying" + | "success" + | "failed" + | "cancelled" + | "skipped"; + +export interface CronJobRun { + runId: string; + jobId: string; + status: CronJobRunStatus; + scheduledAt: string; + startedAt?: string; + finishedAt?: string; + cancellationRequestedAt?: string; + sessionId?: string; + runtimeVersion?: string; + output?: string; + error?: string; + attempt?: number; +} + +export interface CronJob { + jobId: string; + name: string; + runtimeId: string; + runtimeName: string; + agentName: string; + region: string; + prompt: string; + schedule: CronJobSchedule; + enabled: boolean; + nextRunAt?: string; + createdAt: string; + updatedAt: string; + latestRun?: CronJobRun; +} + +export interface CronJobInput { + name: string; + runtimeId: string; + runtimeName: string; + agentName: string; + region: string; + prompt: string; + schedule: CronJobSchedule; + enabled: boolean; +} + +export interface CronJobListResponse { + items: CronJob[]; +} + +export interface CronJobRunListResponse { + items: CronJobRun[]; +} + +function cronJobPath(jobId = ""): string { + return `/web/cronjobs${jobId ? `/${encodeURIComponent(jobId)}` : ""}`; +} + +export async function listCronJobs(signal?: AbortSignal): Promise { + const response = await apiFetch(cronJobPath(), { signal }); + if (!response.ok) { + throw new Error(await httpErrorMessage(response, "加载定时任务失败")); + } + const data = (await response.json()) as CronJobListResponse | CronJob[]; + return Array.isArray(data) ? data : data.items ?? []; +} + +export async function getCronJob( + jobId: string, + signal?: AbortSignal, +): Promise { + const response = await apiFetch(cronJobPath(jobId), { signal }); + if (!response.ok) { + throw new Error(await httpErrorMessage(response, "加载定时任务详情失败")); + } + return (await response.json()) as CronJob; +} + +export async function createCronJob(input: CronJobInput): Promise { + const response = await apiFetch(cronJobPath(), { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(input), + }); + if (!response.ok) { + throw new Error(await httpErrorMessage(response, "创建定时任务失败")); + } + return (await response.json()) as CronJob; +} + +export async function updateCronJob( + jobId: string, + input: CronJobInput, +): Promise { + const response = await apiFetch(`${cronJobPath(jobId)}/update`, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(input), + }); + if (!response.ok) { + throw new Error(await httpErrorMessage(response, "更新定时任务失败")); + } + return (await response.json()) as CronJob; +} + +export async function setCronJobEnabled( + jobId: string, + enabled: boolean, +): Promise { + const action = enabled ? "enable" : "disable"; + const response = await apiFetch(`${cronJobPath(jobId)}/${action}`, { + method: "POST", + }); + if (!response.ok) { + throw new Error( + await httpErrorMessage(response, enabled ? "启用定时任务失败" : "暂停定时任务失败"), + ); + } + return (await response.json()) as CronJob; +} + +export async function runCronJobNow(jobId: string): Promise { + const response = await apiFetch(`${cronJobPath(jobId)}/run`, { method: "POST" }); + if (!response.ok) { + throw new Error(await httpErrorMessage(response, "立即执行定时任务失败")); + } + return (await response.json()) as CronJobRun; +} + +export async function listCronJobRuns( + jobId: string, + signal?: AbortSignal, +): Promise { + const response = await apiFetch(`${cronJobPath(jobId)}/runs`, { signal }); + if (!response.ok) { + throw new Error(await httpErrorMessage(response, "加载执行历史失败")); + } + const data = (await response.json()) as CronJobRunListResponse | CronJobRun[]; + return Array.isArray(data) ? data : data.items ?? []; +} + +export async function cancelCronJobRun( + jobId: string, + runId: string, +): Promise { + const response = await apiFetch( + `${cronJobPath(jobId)}/runs/${encodeURIComponent(runId)}/cancel`, + { method: "POST" }, + ); + if (!response.ok) { + throw new Error(await httpErrorMessage(response, "终止执行失败")); + } + return (await response.json()) as CronJobRun; +} + +export async function deleteCronJob(jobId: string): Promise { + const response = await apiFetch(cronJobPath(jobId), { method: "DELETE" }); + if (!response.ok) { + throw new Error(await httpErrorMessage(response, "删除定时任务失败")); + } +} + /** One page of cloud runtimes plus the token to fetch the next page. */ export interface RuntimePage { runtimes: CloudRuntime[]; diff --git a/frontend/src/cronjobs/CronJobs.css b/frontend/src/cronjobs/CronJobs.css new file mode 100644 index 000000000..8a98e291a --- /dev/null +++ b/frontend/src/cronjobs/CronJobs.css @@ -0,0 +1,510 @@ +.cronjobs-page { + position: relative; + flex: 1; + min-width: 0; + min-height: 0; + display: flex; + flex-direction: column; + overflow: hidden; + padding: 32px 32px 0; + background: hsl(var(--background)); + color: hsl(var(--foreground)); +} + +.cronjobs-page-head, +.cronjobs-detail-head { + flex: 0 0 auto; + display: flex; + align-items: flex-start; + justify-content: space-between; + gap: 24px; +} + +.cronjobs-page-head h1, +.cronjobs-detail-title h1 { + margin: 0; + font-size: 21px; + font-weight: 650; + line-height: 1.25; + letter-spacing: -0.02em; +} + +.cronjobs-page-head p, +.cronjobs-detail-title p { + margin: 6px 0 0; + color: hsl(var(--muted-foreground)); + font-size: 13px; + line-height: 1.5; +} + +.cronjobs-toolbar { + flex: 0 0 auto; + min-height: 36px; + display: flex; + align-items: center; + justify-content: flex-end; + margin-top: 24px; +} + +.cronjobs-button { + min-height: 36px; + display: inline-flex; + align-items: center; + justify-content: center; + gap: 7px; + box-sizing: border-box; + padding: 0 13px; + border: 1px solid hsl(var(--border)); + border-radius: 7px; + background: hsl(var(--panel)); + color: hsl(var(--foreground)); + cursor: pointer; + font: inherit; + font-size: 12.5px; + font-weight: 550; + white-space: nowrap; + transition: background-color 160ms ease, border-color 160ms ease, color 160ms ease; +} + +.cronjobs-button svg { width: 16px; height: 16px; flex: 0 0 16px; } +.cronjobs-button:hover:not(:disabled) { background: hsl(var(--muted)); border-color: hsl(var(--foreground) / 0.17); } +.cronjobs-button.is-primary { border-color: hsl(var(--primary)); background: hsl(var(--primary)); color: hsl(var(--primary-foreground)); } +.cronjobs-button.is-primary:hover:not(:disabled) { border-color: hsl(var(--primary) / 0.88); background: hsl(var(--primary) / 0.88); } +.cronjobs-button.is-danger-quiet { color: hsl(var(--destructive)); } +.cronjobs-button.is-danger-quiet:hover:not(:disabled) { background: hsl(var(--destructive) / 0.08); border-color: hsl(var(--destructive) / 0.24); } +.cronjobs-button:focus-visible, +.cronjobs-icon-button:focus-visible, +.cronjobs-row-actions button:focus-visible, +.cronjobs-name-button:focus-visible, +.cronjobs-run-cancel:focus-visible { outline: 2px solid hsl(var(--ring) / 0.35); outline-offset: 2px; } +.cronjobs-button:disabled, +.cronjobs-icon-button:disabled, +.cronjobs-row-actions button:disabled { cursor: not-allowed; opacity: 0.46; } + +.cronjobs-content { + flex: 1; + min-height: 0; + display: flex; + overflow: hidden; + margin-top: 12px; + padding-bottom: 32px; +} + +.cronjobs-table-wrap { + width: 100%; + min-width: 0; + overflow: auto; + border: 1px solid hsl(var(--border)); + border-radius: 12px; + background: hsl(var(--panel)); + scrollbar-gutter: stable; +} + +.cronjobs-table { + width: 100%; + min-width: 940px; + border-collapse: collapse; + table-layout: fixed; + font-size: 12.5px; +} + +.cronjobs-table th { + position: sticky; + z-index: 1; + top: 0; + height: 42px; + box-sizing: border-box; + padding: 0 14px; + border-bottom: 1px solid hsl(var(--border)); + background: hsl(var(--muted) / 0.56); + color: hsl(var(--muted-foreground)); + font-size: 11.5px; + font-weight: 550; + text-align: left; +} + +.cronjobs-table th:nth-child(1) { width: 19%; } +.cronjobs-table th:nth-child(2) { width: 17%; } +.cronjobs-table th:nth-child(3) { width: 22%; } +.cronjobs-table th:nth-child(4) { width: 10%; } +.cronjobs-table th:nth-child(5) { width: 12%; } +.cronjobs-table th:nth-child(6) { width: 11%; } +.cronjobs-table th:nth-child(7) { width: 112px; } +.cronjobs-table td { + height: 52px; + box-sizing: border-box; + padding: 0 14px; + border-bottom: 1px solid hsl(var(--border)); + color: hsl(var(--muted-foreground)); + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} +.cronjobs-table tr:last-child td { border-bottom: 0; } +.cronjobs-table tbody tr { transition: background-color 140ms ease; } +.cronjobs-table tbody tr:hover { background: hsl(var(--muted) / 0.36); } + +.cronjobs-name-button { + max-width: 100%; + padding: 0; + border: 0; + background: transparent; + color: hsl(var(--foreground)); + cursor: pointer; + font: inherit; + font-weight: 600; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} +.cronjobs-name-button:hover { text-decoration: underline; text-underline-offset: 3px; } +.cronjobs-agent { min-width: 0; display: inline-flex; align-items: center; gap: 7px; max-width: 100%; color: hsl(var(--foreground)); overflow: hidden; text-overflow: ellipsis; } +.cronjobs-agent svg { width: 16px; height: 16px; flex: 0 0 16px; color: hsl(var(--muted-foreground)); } +.cronjobs-enabled, +.cronjobs-disabled { display: inline-flex; align-items: center; min-height: 21px; padding: 0 7px; border-radius: 999px; font-size: 10.5px; font-weight: 600; } +.cronjobs-enabled { background: hsl(145 55% 42% / 0.12); color: hsl(145 58% 28%); } +.cronjobs-disabled { background: hsl(var(--muted)); color: hsl(var(--muted-foreground)); } + +.cronjobs-status { display: inline-flex; align-items: center; gap: 6px; color: hsl(var(--muted-foreground)); font-size: 11.5px; font-weight: 550; } +.cronjobs-status > span { width: 7px; height: 7px; flex: 0 0 7px; border-radius: 50%; background: currentColor; } +.cronjobs-status.is-queued, +.cronjobs-status.is-pending, +.cronjobs-status.is-running, +.cronjobs-status.is-retrying { color: hsl(213 72% 43%); } +.cronjobs-status.is-success { color: hsl(145 58% 31%); } +.cronjobs-status.is-failed { color: hsl(var(--destructive)); } +.cronjobs-status.is-cancelled, +.cronjobs-status.is-skipped, +.cronjobs-status.is-idle { color: hsl(var(--muted-foreground)); } + +.cronjobs-row-actions { display: flex; align-items: center; justify-content: flex-end; gap: 4px; } +.cronjobs-row-actions button, +.cronjobs-icon-button { + width: 30px; + height: 30px; + flex: 0 0 30px; + display: inline-grid; + place-items: center; + padding: 0; + border: 1px solid transparent; + border-radius: 6px; + background: transparent; + color: hsl(var(--muted-foreground)); + cursor: pointer; + transition: background-color 140ms ease, color 140ms ease, border-color 140ms ease; +} +.cronjobs-row-actions button:hover:not(:disabled), +.cronjobs-icon-button:hover:not(:disabled) { background: hsl(var(--muted)); color: hsl(var(--foreground)); } +.cronjobs-row-actions svg, +.cronjobs-icon-button svg { width: 16px; height: 16px; } +.cronjobs-delete-action > span { display: none; } +.cronjobs-icon-button.is-danger:hover:not(:disabled) { background: hsl(var(--destructive) / 0.08); color: hsl(var(--destructive)); } + +.cronjobs-loading, +.cronjobs-state { + width: 100%; + min-height: 300px; + display: grid; + place-items: center; + align-content: center; + color: hsl(var(--muted-foreground)); + text-align: center; +} +.cronjobs-loading { gap: 12px; } +.cronjobs-loading > div { width: min(100%, 760px); height: 48px; border: 1px solid hsl(var(--border)); border-radius: 8px; background: hsl(var(--panel)); } +.cronjobs-state > svg { width: 34px; height: 34px; margin-bottom: 12px; color: hsl(var(--muted-foreground)); } +.cronjobs-state h2 { margin: 0; color: hsl(var(--foreground)); font-size: 15px; font-weight: 600; } +.cronjobs-state p { max-width: 520px; margin: 7px 0 16px; font-size: 12.5px; line-height: 1.55; white-space: pre-wrap; } +.cronjobs-state > span { margin-top: 12px; font-size: 11.5px; } +.cronjobs-state.is-error > svg { color: hsl(var(--destructive)); } + +.cronjobs-banner, +.cronjobs-notice { + flex: 0 0 auto; + margin-top: 16px; + padding: 9px 12px; + border: 1px solid hsl(var(--border)); + border-radius: 8px; + background: hsl(var(--panel)); + color: hsl(var(--foreground)); + font-size: 12px; + line-height: 1.5; + white-space: pre-wrap; +} +.cronjobs-notice { position: absolute; z-index: 50; right: 32px; bottom: 24px; max-width: min(420px, calc(100% - 64px)); box-shadow: 0 10px 30px hsl(var(--foreground) / 0.12); } + +.cronjobs-drawer-backdrop { + position: fixed; + z-index: 110; + inset: 0; + display: flex; + justify-content: flex-end; + background: hsl(var(--foreground) / 0.34); + animation: cronjobs-backdrop-in 140ms ease-out; +} +.cronjobs-drawer { + width: min(520px, 100vw); + height: 100%; + min-height: 0; + display: flex; + flex-direction: column; + overflow: hidden; + background: hsl(var(--panel)); + box-shadow: -14px 0 36px hsl(var(--foreground) / 0.12); + animation: cronjobs-drawer-in 180ms cubic-bezier(0.22, 1, 0.36, 1); +} +.cronjobs-drawer-head { + flex: 0 0 auto; + display: flex; + align-items: flex-start; + justify-content: space-between; + gap: 16px; + padding: 22px 24px 18px; + border-bottom: 1px solid hsl(var(--border)); +} +.cronjobs-drawer-head h2 { margin: 0; font-size: 17px; font-weight: 650; line-height: 1.3; } +.cronjobs-drawer-head p { margin: 6px 0 0; color: hsl(var(--muted-foreground)); font-size: 12.5px; line-height: 1.5; } +.cronjobs-form { flex: 1; min-height: 0; display: flex; flex-direction: column; } +.cronjobs-form-scroll { flex: 1; min-height: 0; overflow-y: auto; padding: 22px 24px 32px; scrollbar-gutter: stable; } +.cronjobs-field { min-width: 0; display: flex; flex-direction: column; gap: 7px; margin-bottom: 18px; } +.cronjobs-field > span, +.cronjobs-fieldset > legend { color: hsl(var(--foreground)); font-size: 12.5px; font-weight: 600; line-height: 1.45; } +.cronjobs-field input, +.cronjobs-field select, +.cronjobs-field textarea { + width: 100%; + min-width: 0; + box-sizing: border-box; + border: 1px solid hsl(var(--border)); + border-radius: 7px; + outline: 0; + background: hsl(var(--background)); + color: hsl(var(--foreground)); + font: inherit; + font-size: 12.5px; + transition: border-color 150ms ease, box-shadow 150ms ease; +} +.cronjobs-field input, +.cronjobs-field select { height: 36px; padding: 0 10px; } +.cronjobs-field textarea { min-height: 116px; resize: vertical; padding: 10px 11px; line-height: 1.55; } +.cronjobs-field input:focus, +.cronjobs-field select:focus, +.cronjobs-field textarea:focus { border-color: hsl(var(--ring) / 0.64); box-shadow: 0 0 0 2px hsl(var(--ring) / 0.11); } +.cronjobs-field input:disabled, +.cronjobs-field select:disabled { cursor: not-allowed; opacity: 0.55; } +.cronjobs-field small { color: hsl(var(--muted-foreground)); font-size: 11px; line-height: 1.45; } +.cronjobs-field textarea + small { text-align: right; } +.cronjobs-fieldset { min-width: 0; margin: 0 0 18px; padding: 0; border: 0; } +.cronjobs-fieldset > legend { margin-bottom: 9px; } +.cronjobs-schedule-types { display: grid; grid-template-columns: repeat(4, 1fr); gap: 4px; margin-bottom: 16px; padding: 4px; border: 1px solid hsl(var(--border)); border-radius: 10px; background: hsl(var(--secondary)); } +.cronjobs-schedule-types label { min-width: 0; cursor: pointer; } +.cronjobs-schedule-types input { position: absolute; opacity: 0; pointer-events: none; } +.cronjobs-schedule-types span { min-height: 34px; display: grid; place-items: center; border: 1px solid transparent; border-radius: 7px; color: hsl(var(--muted-foreground)); font-size: 12px; font-weight: 550; transition: background-color 160ms ease, border-color 160ms ease, color 160ms ease; } +.cronjobs-schedule-types label:hover span { color: hsl(var(--foreground)); background: hsl(var(--panel) / 0.5); } +.cronjobs-schedule-types input:checked + span { border-color: hsl(var(--border)); background: hsl(var(--panel)); color: hsl(var(--foreground)); } +.cronjobs-schedule-types input:focus-visible + span { outline: 2px solid hsl(var(--ring) / 0.3); outline-offset: 2px; } +.cronjobs-fieldset .cronjobs-field:last-child { margin-bottom: 0; } +.cronjobs-inline-fields { display: grid; grid-template-columns: 1fr 1fr; gap: 12px; } +.cronjobs-switch-row { display: flex; align-items: center; justify-content: space-between; gap: 16px; padding: 13px 14px; border: 1px solid hsl(var(--border)); border-radius: 8px; cursor: pointer; } +.cronjobs-switch-row > span { display: flex; flex-direction: column; gap: 3px; } +.cronjobs-switch-row strong { font-size: 12.5px; font-weight: 600; } +.cronjobs-switch-row small { color: hsl(var(--muted-foreground)); font-size: 11px; line-height: 1.4; } +.cronjobs-switch-row input { + position: relative; + width: 34px; + height: 18px; + flex: 0 0 34px; + margin: 0; + border: 1px solid hsl(var(--border)); + border-radius: 999px; + background: hsl(var(--muted)); + appearance: none; + cursor: pointer; + transition: border-color 150ms ease, background-color 150ms ease; +} +.cronjobs-switch-row input::after { + content: ""; + position: absolute; + top: 2px; + left: 2px; + width: 12px; + height: 12px; + border-radius: 50%; + background: hsl(var(--panel)); + box-shadow: 0 1px 2px hsl(var(--foreground) / 0.2); + transition: transform 150ms ease; +} +.cronjobs-switch-row input:checked { + border-color: hsl(var(--primary)); + background: hsl(var(--primary)); +} +.cronjobs-switch-row input:checked::after { transform: translateX(16px); } +.cronjobs-switch-row input:focus-visible { outline: 2px solid hsl(var(--ring) / 0.35); outline-offset: 2px; } +.cronjobs-inline-error { margin-top: 14px; padding: 9px 11px; border-radius: 7px; background: hsl(var(--destructive) / 0.08); color: hsl(var(--destructive)); font-size: 12px; line-height: 1.5; white-space: pre-wrap; } +.cronjobs-drawer-actions { flex: 0 0 auto; display: flex; justify-content: flex-end; gap: 8px; padding: 14px 24px; border-top: 1px solid hsl(var(--border)); background: hsl(var(--panel)); } + +.cronjobs-detail { flex: 1; min-height: 0; display: flex; flex-direction: column; } +.cronjobs-detail-title { min-width: 0; display: flex; align-items: flex-start; gap: 10px; } +.cronjobs-detail-title > div { min-width: 0; } +.cronjobs-detail-title h1, +.cronjobs-detail-title p { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.cronjobs-detail-actions { display: flex; align-items: center; justify-content: flex-end; gap: 8px; } +.cronjobs-detail-scroll { flex: 1; min-height: 0; overflow-y: auto; margin-top: 24px; padding-bottom: 40px; scrollbar-gutter: stable; } +.cronjobs-summary-grid { display: grid; grid-template-columns: minmax(0, 1fr) minmax(320px, 1fr); align-items: stretch; gap: 12px; } +.cronjobs-summary-grid > dl, +.cronjobs-prompt { margin: 0; padding: 18px 20px; border: 1px solid hsl(var(--border)); border-radius: 12px; background: hsl(var(--panel)); } +.cronjobs-summary-grid > dl { display: grid; grid-template-columns: 1fr 1fr; gap: 18px 24px; } +.cronjobs-summary-grid dl div { min-width: 0; } +.cronjobs-summary-grid dt, +.cronjobs-prompt > span, +.cronjobs-run-output > span, +.cronjobs-run-meta > span { color: hsl(var(--muted-foreground)); font-size: 11px; line-height: 1.4; } +.cronjobs-summary-grid dd { margin: 5px 0 0; color: hsl(var(--foreground)); font-size: 12.5px; font-weight: 550; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.cronjobs-prompt p { display: -webkit-box; margin: 7px 0 0; overflow: hidden; color: hsl(var(--foreground)); font-size: 12.5px; line-height: 1.58; white-space: pre-wrap; -webkit-box-orient: vertical; -webkit-line-clamp: 5; } +.cronjobs-history { margin-top: 24px; } +.cronjobs-history > header { display: flex; align-items: flex-start; justify-content: space-between; gap: 16px; margin-bottom: 10px; } +.cronjobs-history h2 { margin: 0; font-size: 15px; font-weight: 620; line-height: 1.4; } +.cronjobs-history header p { margin: 4px 0 0; color: hsl(var(--muted-foreground)); font-size: 11.5px; line-height: 1.5; } +.cronjobs-runs { display: flex; flex-direction: column; border: 1px solid hsl(var(--border)); border-radius: 12px; background: hsl(var(--panel)); overflow: hidden; } +.cronjobs-run { position: relative; display: grid; grid-template-columns: minmax(230px, 0.8fr) minmax(180px, 0.6fr) minmax(260px, 1.4fr); align-items: start; gap: 20px; min-width: 0; padding: 16px 18px; border-bottom: 1px solid hsl(var(--border)); } +.cronjobs-run:last-child { border-bottom: 0; } +.cronjobs-run-main { min-width: 0; display: flex; align-items: flex-start; gap: 14px; } +.cronjobs-run-main > div { min-width: 0; display: flex; flex-direction: column; gap: 4px; } +.cronjobs-run-main strong { font-size: 12px; font-weight: 600; } +.cronjobs-run-main span { color: hsl(var(--muted-foreground)); font-size: 11px; line-height: 1.45; } +.cronjobs-run-meta, +.cronjobs-run-output { min-width: 0; display: flex; flex-direction: column; gap: 5px; } +.cronjobs-run-meta strong { font-size: 11px; font-weight: 500; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.cronjobs-run-output p { display: -webkit-box; margin: 0; overflow: hidden; color: hsl(var(--foreground)); font-size: 11.5px; line-height: 1.5; white-space: pre-wrap; -webkit-box-orient: vertical; -webkit-line-clamp: 3; } +.cronjobs-run-output.is-error p { color: hsl(var(--destructive)); } +.cronjobs-run-error-detail { color: hsl(var(--destructive)); } +.cronjobs-run-error-detail.is-expanded .deploy-error-message-text { + display: block; + max-height: 280px; + overflow: auto; + -webkit-line-clamp: unset; +} +.cronjobs-run-error-detail .deploy-error-message-actions { margin-top: 8px; } +.cronjobs-run-error-detail .deploy-error-message-actions button:focus-visible { outline: 2px solid hsl(var(--ring) / 0.35); outline-offset: 2px; } +.cronjobs-run-cancel { position: absolute; top: 12px; right: 14px; min-height: 28px; padding: 0 9px; border: 1px solid hsl(var(--destructive) / 0.24); border-radius: 6px; background: transparent; color: hsl(var(--destructive)); cursor: pointer; font: inherit; font-size: 11.5px; } +.cronjobs-run-cancel:hover:not(:disabled) { background: hsl(var(--destructive) / 0.08); } +.cronjobs-history-state { min-height: 190px; display: grid; place-items: center; align-content: center; padding: 20px; border: 1px solid hsl(var(--border)); border-radius: 12px; background: hsl(var(--panel)); color: hsl(var(--muted-foreground)); text-align: center; } +.cronjobs-history-state > svg { width: 28px; height: 28px; margin-bottom: 10px; } +.cronjobs-history-state p { margin: 0; color: hsl(var(--foreground)); font-size: 13px; font-weight: 550; } +.cronjobs-history-state span { margin-top: 5px; font-size: 11.5px; } +.cronjobs-history-state.is-error p { max-width: 600px; color: hsl(var(--destructive)); font-weight: 400; line-height: 1.55; white-space: pre-wrap; } +.cronjobs-history-state button { margin-top: 12px; border: 0; background: transparent; color: hsl(var(--primary)); cursor: pointer; font: inherit; font-size: 12px; } + +@keyframes cronjobs-backdrop-in { from { opacity: 0; } } +@keyframes cronjobs-drawer-in { from { transform: translateX(24px); opacity: 0.88; } } + +@media (max-width: 980px) { + .cronjobs-page { padding: 24px 20px 0; } + .cronjobs-summary-grid { grid-template-columns: 1fr; } + .cronjobs-run { grid-template-columns: minmax(220px, 0.8fr) minmax(260px, 1.2fr); } + .cronjobs-run-meta { display: none; } + .cronjobs-notice { right: 20px; max-width: calc(100% - 40px); } +} + +@media (max-width: 900px) { + .cronjobs-page-head, + .cronjobs-detail-head { align-items: stretch; flex-direction: column; gap: 16px; } + .cronjobs-toolbar { min-height: 44px; margin-top: 16px; } + .cronjobs-toolbar .cronjobs-button { min-height: 44px; } + .cronjobs-content { margin-top: 12px; padding-bottom: 20px; } + .cronjobs-table-wrap { overflow-y: auto; border: 0; border-radius: 0; background: transparent; } + .cronjobs-table { min-width: 0; display: block; font-size: 13px; } + .cronjobs-table thead { display: none; } + .cronjobs-table tbody { display: flex; flex-direction: column; gap: 12px; } + .cronjobs-table tr { + display: grid; + grid-template-columns: minmax(0, 1fr) auto; + gap: 12px 16px; + padding: 14px; + border: 1px solid hsl(var(--border)); + border-radius: 12px; + background: hsl(var(--panel)); + } + .cronjobs-table td { + height: auto; + min-width: 0; + display: flex; + flex-direction: column; + align-items: flex-start; + justify-content: center; + padding: 0; + border: 0; + overflow: visible; + white-space: normal; + } + .cronjobs-table tr:last-child td { border: 0; } + .cronjobs-table tbody tr:hover { background: hsl(var(--panel)); } + .cronjobs-table td::before { + content: attr(data-label); + margin-bottom: 4px; + color: hsl(var(--muted-foreground)); + font-size: 11px; + font-weight: 500; + line-height: 1.35; + } + .cronjobs-table td:first-child { grid-column: 1; grid-row: 1; } + .cronjobs-table td:first-child::before, + .cronjobs-table .cronjobs-actions-cell::before { content: none; } + .cronjobs-table td:nth-child(2) { grid-column: 1 / -1; grid-row: 2; } + .cronjobs-table td:nth-child(3) { grid-column: 1 / -1; grid-row: 3; } + .cronjobs-table td:nth-child(4) { grid-column: 1; grid-row: 4; } + .cronjobs-table td:nth-child(5) { grid-column: 1 / -1; grid-row: 5; } + .cronjobs-table td:nth-child(6) { grid-column: 2; grid-row: 4; } + .cronjobs-table .cronjobs-actions-cell { grid-column: 2; grid-row: 1; } + .cronjobs-name-button { + display: -webkit-box; + overflow: hidden; + font-size: 14px; + line-height: 1.45; + text-align: left; + text-overflow: clip; + white-space: normal; + word-break: break-word; + -webkit-box-orient: vertical; + -webkit-line-clamp: 2; + } + .cronjobs-agent { color: hsl(var(--foreground)); white-space: nowrap; } + .cronjobs-row-actions { gap: 8px; } + .cronjobs-row-actions button { width: 44px; height: 44px; flex-basis: 44px; } + .cronjobs-detail-actions { justify-content: flex-start; flex-wrap: wrap; } + .cronjobs-detail-actions .cronjobs-button { min-height: 44px; flex: 1 1 auto; } + .cronjobs-summary-grid > dl { grid-template-columns: 1fr; gap: 14px; } + .cronjobs-run { grid-template-columns: 1fr; gap: 12px; } + .cronjobs-run-meta { display: flex; } + .cronjobs-schedule-types { grid-template-columns: 1fr 1fr; } + .cronjobs-schedule-types span { min-height: 44px; } + .cronjobs-drawer { height: 100dvh; } + .cronjobs-drawer .cronjobs-icon-button { width: 44px; height: 44px; flex-basis: 44px; } + .cronjobs-drawer-actions { padding-bottom: max(12px, env(safe-area-inset-bottom)); } + .cronjobs-drawer-actions .cronjobs-button { min-height: 44px; } +} + +@media (max-width: 520px) { + .cronjobs-page { padding: 20px 16px 0; } + .cronjobs-drawer-head { padding: 18px 16px 14px; } + .cronjobs-form-scroll { padding: 18px 16px 24px; } + .cronjobs-drawer-actions { padding: 12px 16px; } + .cronjobs-inline-fields { grid-template-columns: 1fr; gap: 0; } + .cronjobs-detail-actions { display: grid; grid-template-columns: 1fr 1fr; } + .cronjobs-detail-actions .cronjobs-icon-button { width: 100%; height: 44px; } + .cronjobs-detail-actions .cronjobs-delete-action { display: inline-flex; gap: 7px; font-size: 12.5px; font-weight: 550; } + .cronjobs-detail-actions .cronjobs-delete-action > span { display: inline; } +} + +@media (prefers-reduced-motion: reduce) { + .cronjobs-drawer-backdrop, + .cronjobs-drawer { animation: none; } + .cronjobs-button, + .cronjobs-icon-button, + .cronjobs-row-actions button, + .cronjobs-table tbody tr, + .cronjobs-field input, + .cronjobs-field select, + .cronjobs-field textarea, + .cronjobs-schedule-types span { transition: none; } +} diff --git a/frontend/src/cronjobs/CronJobs.tsx b/frontend/src/cronjobs/CronJobs.tsx new file mode 100644 index 000000000..a91939c9c --- /dev/null +++ b/frontend/src/cronjobs/CronJobs.tsx @@ -0,0 +1,641 @@ +import { + useCallback, + useEffect, + useMemo, + useRef, + useState, + type FormEvent, +} from "react"; +import { + cancelCronJobRun, + createCronJob, + deleteCronJob, + fetchRemoteApps, + getRuntimes, + listCronJobRuns, + listCronJobs, + runCronJobNow, + setCronJobEnabled, + updateCronJob, + type CloudRuntime, + type CronJob, + type CronJobInput, + type CronJobRun, + type CronJobScheduleType, +} from "../adk/client"; +import type { CloudProvider } from "../adk/cloudProvider"; +import { formatCloudRegion } from "../adk/cloudProvider"; +import { StudioConfirmDialog } from "../ui/StudioConfirmDialog"; +import { DeploymentErrorMessage } from "../ui/DeploymentErrorMessage"; +import { TextShimmer } from "../ui/text-shimmer/TextShimmer"; +import { + CronBackIcon, + CronClockIcon, + CronCloseIcon, + CronDeleteIcon, + CronEditIcon, + CronPauseIcon, + CronPlusIcon, + CronRefreshIcon, + CronRunIcon, +} from "./icons"; +import { + CRONJOB_STATUS_LABELS, + WEEKDAY_LABELS, + cronJobIsRunning, + describeCronJobSchedule, + formatCronJobDate, + formatCronJobDuration, +} from "./model"; +import "./CronJobs.css"; + +interface CronJobsProps { + cloudProvider: CloudProvider; +} + +interface CronJobDraft { + name: string; + runtimeId: string; + prompt: string; + scheduleType: CronJobScheduleType; + onceAt: string; + time: string; + weekday: number; + cron: string; + timezone: string; + enabled: boolean; +} + +type ConfirmTarget = + | { kind: "delete"; job: CronJob } + | { kind: "cancel"; job: CronJob; run: CronJobRun }; + +const FALLBACK_TIMEZONE = "Asia/Shanghai"; +const CRONJOB_ACTIVE_REFRESH_MS = 3_000; +const TIMEZONE_OPTIONS = [ + "Asia/Shanghai", + "Asia/Singapore", + "Asia/Tokyo", + "Europe/London", + "America/Los_Angeles", + "America/New_York", + "UTC", +]; + +function browserTimezone(): string { + try { + return Intl.DateTimeFormat().resolvedOptions().timeZone || FALLBACK_TIMEZONE; + } catch { + return FALLBACK_TIMEZONE; + } +} + +function emptyDraft(): CronJobDraft { + const timezone = browserTimezone(); + const tomorrow = new Date(Date.now() + 24 * 60 * 60 * 1000); + tomorrow.setSeconds(0, 0); + const localTomorrow = new Date( + tomorrow.getTime() - tomorrow.getTimezoneOffset() * 60_000, + ); + return { + name: "", + runtimeId: "", + prompt: "", + scheduleType: "daily", + onceAt: localTomorrow.toISOString().slice(0, 16), + time: "09:00", + weekday: 1, + cron: "0 9 * * *", + timezone, + enabled: true, + }; +} + +function draftFromJob(job: CronJob): CronJobDraft { + return { + name: job.name, + runtimeId: job.runtimeId, + prompt: job.prompt, + scheduleType: job.schedule.type, + onceAt: job.schedule.onceAt ?? "", + time: job.schedule.time ?? "09:00", + weekday: job.schedule.weekday ?? 1, + cron: job.schedule.cron ?? "0 9 * * *", + timezone: job.schedule.timezone || FALLBACK_TIMEZONE, + enabled: job.enabled, + }; +} + +function StatusBadge({ run }: { run?: CronJobRun }) { + if (!run) return 尚未执行; + return ( + + + ); +} + +function Drawer({ + job, + runtimes, + cloudProvider, + busy, + onClose, + onSubmit, +}: { + job: CronJob | null; + runtimes: CloudRuntime[]; + cloudProvider: CloudProvider; + busy: boolean; + onClose: () => void; + onSubmit: (input: CronJobInput) => Promise; +}) { + const [draft, setDraft] = useState(() => job ? draftFromJob(job) : emptyDraft()); + const [error, setError] = useState(""); + const [resolvingRuntime, setResolvingRuntime] = useState(false); + const drawerRef = useRef(null); + const nameRef = useRef(null); + const errorRef = useRef(null); + const isBusy = busy || resolvingRuntime; + const busyRef = useRef(isBusy); + const onCloseRef = useRef(onClose); + const zones = useMemo(() => Array.from(new Set([draft.timezone, ...TIMEZONE_OPTIONS])), [draft.timezone]); + + useEffect(() => { + busyRef.current = isBusy; + onCloseRef.current = onClose; + }, [isBusy, onClose]); + + useEffect(() => { + const previousOverflow = document.body.style.overflow; + const previousFocus = document.activeElement instanceof HTMLElement ? document.activeElement : null; + document.body.style.overflow = "hidden"; + nameRef.current?.focus(); + const handleKeyDown = (event: KeyboardEvent) => { + if (event.key === "Escape" && !busyRef.current) { + onCloseRef.current(); + return; + } + if (event.key !== "Tab") return; + const focusable = Array.from( + drawerRef.current?.querySelectorAll( + 'button:not([disabled]), input:not([disabled]), select:not([disabled]), textarea:not([disabled]), [tabindex]:not([tabindex="-1"])', + ) ?? [], + ).filter((element) => !element.hidden && element.getClientRects().length > 0); + if (focusable.length === 0) { + event.preventDefault(); + return; + } + const first = focusable[0]; + const last = focusable[focusable.length - 1]; + const active = document.activeElement; + if (event.shiftKey && (active === first || !drawerRef.current?.contains(active))) { + event.preventDefault(); + last.focus(); + } else if (!event.shiftKey && active === last) { + event.preventDefault(); + first.focus(); + } + }; + window.addEventListener("keydown", handleKeyDown); + return () => { + document.body.style.overflow = previousOverflow; + window.removeEventListener("keydown", handleKeyDown); + if (previousFocus?.isConnected) previousFocus.focus(); + }; + }, []); + + const submit = async (event: FormEvent) => { + event.preventDefault(); + const name = draft.name.trim(); + const prompt = draft.prompt.trim(); + const runtime = runtimes.find((item) => item.runtimeId === draft.runtimeId); + if (!name) return setError("请输入任务名称。"); + if (!runtime) return setError("请选择可用的 Runtime Agent。"); + if (!prompt) return setError("请输入每次执行时发送给 Agent 的文本。"); + if (draft.scheduleType === "once" && !draft.onceAt) return setError("请选择执行时间。"); + if ((draft.scheduleType === "daily" || draft.scheduleType === "weekly") && !draft.time) { + return setError("请选择执行时间。"); + } + const cronFields = draft.cron.trim().split(/\s+/); + if (draft.scheduleType === "cron" && cronFields.length !== 5) { + return setError("Cron 表达式需要包含 5 个字段,例如 0 9 * * *。"); + } + setError(""); + setResolvingRuntime(true); + try { + let agentName = job?.runtimeId === runtime.runtimeId + ? job.agentName.trim() + : ""; + if (!agentName) { + const [runtimeApp] = await fetchRemoteApps("", "", { + runtimeId: runtime.runtimeId, + region: runtime.region, + }); + agentName = runtimeApp?.trim() ?? ""; + } + if (!agentName) { + throw new Error("Runtime Agent 未返回可调用的 appName,请确认 Runtime 已就绪且版本兼容。"); + } + await onSubmit({ + name, + runtimeId: runtime.runtimeId, + runtimeName: runtime.name, + agentName, + region: runtime.region, + prompt, + enabled: draft.enabled, + schedule: { + type: draft.scheduleType, + timezone: draft.timezone, + ...(draft.scheduleType === "once" ? { onceAt: draft.onceAt } : {}), + ...(draft.scheduleType === "daily" ? { time: draft.time } : {}), + ...(draft.scheduleType === "weekly" ? { time: draft.time, weekday: draft.weekday } : {}), + ...(draft.scheduleType === "cron" ? { cron: draft.cron.trim() } : {}), + }, + }); + } catch (cause) { + setError(cause instanceof Error ? cause.message : String(cause)); + window.requestAnimationFrame(() => errorRef.current?.focus()); + } finally { + setResolvingRuntime(false); + } + }; + + return ( +
{ + if (event.target === event.currentTarget && !isBusy) onClose(); + }}> +