From 3358dcbcfe0e318481e0f2e4417208fd8d288b61 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 11 Nov 2025 19:36:21 +0100 Subject: [PATCH 01/38] Added PythonTask output to message column of logs in maestro.db --- examples/1_Old_examples/terraform/maestro.db | Bin 4096 -> 0 bytes maestro.db | Bin 69632 -> 0 bytes .../server/internals/status_manager.py | 26 +++++++++++- src/maestro/server/tasks/python_task.py | 38 +++++++++++++----- tests/maestro.db | Bin 32768 -> 0 bytes 5 files changed, 54 insertions(+), 10 deletions(-) delete mode 100644 examples/1_Old_examples/terraform/maestro.db delete mode 100644 maestro.db delete mode 100644 tests/maestro.db diff --git a/examples/1_Old_examples/terraform/maestro.db b/examples/1_Old_examples/terraform/maestro.db deleted file mode 100644 index 2f7292d0901b002071ec05a1b6f78ffec474647f..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 4096 zcmWFz^vNtqRY=P(%1ta$FlG>7U}9o$P*7lCU|@t|AVoG{WY8(M$dW59lNefqnO`j} zF1aSTlGgH1XeV}EBSC^9eMpPG6i5*uK+!f24cg=-=tIz=MUen~NLv(5Q23>2QzT8B zy!6}|k~=fx$|TLL9Q)GV6?g8;z2|=C+;h*l_sqH1e)CeLWwTDbv8A_IF>)vpjYdAf zvXMySoA7fOe!8Coe2I1cfPbUD@B4lIW~6xQ&20XEB57tPl0T9EyX;@*-p+hI{mabV zbSnLAxJW+~0g3=cfFeK-CbatoN#fd-wR~jwbDuXS49yq-I2ZJ?K}Hd>dlZH6BW;E*Rqb1L)z7FcyV=xl*zEe+XI$AO!xWge zwcXQwZ#S6q;K zy=(4$ivBVm92U^|ssg^b3YtPn|bDph-1Z*BTz%7n2T8sqmCqOiQA7Z;`ybC1mK*G)od z8Ij-9FwLgOR2zq%>HJU0*L?n0`TzF5l0WH(B0v$K2v7tl0u%v?07ZZzKoOt_Py{Ff z6oFr71aeuRK13EDZj=q(Fd%FrcW8P(+T7F|+vVr%?WSGJ9+_T>wi-Ll7P+grZP!dYb9j0ox~Xq( z@0M?F)*E_FpP8Lrj&A8UDyCj7H}oyNw)31`OFuZxMO&MdMzggG)zxZHMU$DE7NfRF zLL@3?`F5pRt>|0LdM!nWd48wTxJmBas=%e>(P^}It5Ma>o5oI~mLMcKjf!2fs=MW8 zZ5Kj$;x3t;raDB$NO!9{X2mLRLfFp>(}WzOT4^?&ohk&AZPuWTskv#)IC>32TL?WS zL>b*~RBmjx$WlN)eJYawYQC6zJNIht;q0GfFK2#|c_}kH^CvT(OuwCe4T|Z9B0v$K z2v7tl0u%v?07c+~g1~334RNL>BTjNG%;^r+I90<)7i32?Ow(s} z_hQxsUceke951j)SrOC~tQ94V*HrbDGiNsb^K)h{Ra$ggP&D2WkxlAVHBM4>g;Pad z<#@}6sttiRu@p4qLyhp9pt6XE7FLj46tTw3==NuS_p#(B^lYj$?>2%|!^Wnga<(BG zoP@B!X`;$=mV|lB(lo`96u)b5k0wx27Bm6NUK3U(Ra$bJunh;xq9AhuHUv%*Y?(7; z4Rf}v%8FsyIu@Ofv$20GyjtY3CLqmg#h6Kz7Ti`O-Vkh6(mBUMA}1M|z!{cD#5lv3CED#4M4tEBv6)nfbK7x5tcZ#!b2?$aqK2^`beLl% zyP}LtRnj!W;1o->K?F>j0|uEK(kw|+c&zJ^9&`iuYlL4B`Jy5t84E9b=c&@dCvGHC zCFFJoS(0L5L*i_kNPwaf36KR&v<+R*MNP7`z3%iJ%l#V?a1jXzDcEbsigyLtK?nrO zwmAW*Dkn*{#A${ta*8U+vZAYoVn`uZb^lh7R#as{Rkcnlv2H7>s3Gj=I)J1DW`@iI zp9C15x}n&*qrv}DR|q=J?>$-pkyfyZd9Ri8)1bzVgVK&$ShKL8U`{p!P-CiMa0apn zXo#1v4otGK&z**Mu!uwN)nZR|1=f%)Rg*KuqSs5MjoX^>mLe^Y`| zT$o^&KpYzA_-^-!QH}Y@h*Rz*$|C;^92W7IO{iFy` z1SkR&0g3=cfFeK9 zpNt$UEks9N&Y4+>m6oGptkSc!{{v*^M+8Y_v>G1pO2E4zdf5ugZ+jlj#ZOyqRw*zBsf4a~rQlxHFm}pd0G_Jcv8}xnPCYP+ z9y#u=7_WJ|VhFZ}u#80ktEwPK?Zg9woY~ncfuQlGurLc?fVpiY>r| z=Q{%92ML8iI&z4KER{~qJ}WMT_U(o-_cm=15Cd}igrRz(Td=}Dr9=9tMx6yJ^=jpo z-A@&-gb|D7naD!v#O#{56xin*#*#E3p3&Sa>tT|vdfOeWSI(~XNf0lF!HMPGH5tY` zka2E*GJ5jEW|_!xX<_!JxYV~JIgF7ELkprhJ9)jr^gLbJwyTvIBuaKQ5D_1z1*z2< zcB`?=LgulG86uHvc7_?2$S@JEbY%7^5Tj7M=rFc8M1qFtI;S@`S#wjb)Ov$HEIB}k z)JMn+6A?WPJYc(X80&tpA{wNLdjm^dk@^ssP7^|0RXipkL#pI^QjuYLDml0(JB+I` z7>Qw3GTGflqNIUB){q1n!=MQhsh&z6ot$x1vX2mmxlj}p=7y6=CW1Vb+~ZDf8+jOa zY@&*pUlCx^2tsuv(NoN!{pVpkzDX!%eg)E7dQ|mrOT-CPu43+v^Ui+tFz)1-su>6Y z)f|K;5t{%Zqbg?~A$y2S_cUL4S9}}kG1`{qNr(L)=dU8Mt* z2rzpGc@v8gN?g@9E+r%CK0<~ZB7@}ri{^htZ2809KVzHq^g)UMMSvne5ugZA1SkR& z0g3=cfFeK{MhMI)Y~N+G2CPR(fDzAcDqqG;kcy{AO6Tnd=4IV*T_eF^z_H# z#}>OZj~=8^I3J&f$IbdqtyR9IS9feUt;(N%XqZTkQ)}@BsN{xSvm0eAkuI|UztR+ zj8)Nx9*ZC0N>SKU>+Vi=vsMigg)p}kKMhZN(oJF_n&Z@R^nCmj)Bg0n;{Rijha#DHYA*II_|p5l zITuSVefi^UMGU=ka<^dVH_8>Ou)-EXMTN6$!F>s;+^THZa6@gUT7_$7qh2dF;awJM zr|P>*-e7COU3Uwi*Ie9dq_opaxEQJ=G&mJ_4Q?CWt|6o>Z`QZ$WlO(hb7sR@zP4L4 zmp^sw@w#O%Z+Xd;m+d?H)^^oyE(_($_U&@-PigPbEj2rkjvG&sk={`Kmxf?+1jo*S}Y`9w(46O zwT9i?skYe1*o{IrF*T;@&4FqcNhB+nMkpoaWl&a#i1jLLPeA1E`y2-j9#54?c!l%li?Zb3@?s(o5kH z2$FY8uT}KrPIB|H#$PC{Hc2k~`e&z;r{B8KUKD*Vq4~9v@3K1tNP_-RHB#VZw=lzl zg%chXV)?dSX_Y}ul-=~=2dRe6vd!J*Ms0xBd6a0Km5mw;AGoJz->J0DAW$lv*4Zdr z+cC*o`E0$$!Yt5a;h@}8V`&{o4Y2oSVi@UQ;r&9o5M=n+YG;Tv@&`N82Fh075aX3M zZcZh+FWzmR62tFE`o~$gZ0y0d_ZoPZ|Gtt5GhZkhkUYQaCeNSP5Zl^t*gk5#J}~!O`v} zNq^%`G>N~&wCBY^uMzu4{Gjr&N4@JQg!$`ToG^zG8S53ZOk_}ROlY+G*m})(1ke!d zgx+;t{SNoOd-?wK-}BuHqe*wyz3JKoPkY&V$dD&FYvAyoajgP4>M4wmmquvP3Sda_ z=rL_HfP8WiJRfBUNqHAW^?*+G4Shj&{r{lu-mXb^^el^=(5`)>F{9`^? zGWKvMY3;%mf_rVk=95x(&oYDsW3s$ksn-tBg7lx7ID~-!Jq@G?48svDyl=P^T(*p6 zH+2om&bs+?&3bL4ModarGrzl0@DeL-K!>lRv#fBI702o?1Sz&;zlg^w;Hp8crzA9;xEGJ0FU zy;QMb?(Y>$SORd_6~|uvlb_X}xW@s%2g~q*ed}X3(un^bkIY9h52XGumYrhZYyao9 zJK5y?>se3?{xraTaqwU7lLe{_^c(JW`>&ZY&{hz7?YT@x8H6%8`?((~>JtE}3_zz9 z-rLFmjzU-7JTjA9`0^tV@)O8%?PqnMU|2?XqfUCdJ{->zviHKp6W04Q+=qtykf1Ia z?$hi{Ordb7_fEJEXeIIWYC5_67O_YBvfcYNQ(xiO+9??95zy3bOh@h5o23SbO6p(+U0Q2KRPPhwzWKaM{>{%@=MG8fm_g2@kWrn21RXvECXO z4gH~^Ki$}G-%8MXF7yX>Yp_K8|Ea|FNcsZ&q92L?MSvne5ugaXdj#It%qLHO@izEv zzT@}&Hl*+J*yf`bE*i2Ny^HrY8e#2!w3oyN#M8Zdv(aZC{69zh98~|`Co!Z422VG_ zckEr{cNIfAiEeqjNQmPF7Rf6D@fqJza-Eo;;3<>-(H<-rd%)9}pO-+f1;aH$>MvM$ zZ#6jbyRw6$2?l9G8uQaV{WZL!9SA46fBXr`$!H46_p%R-`3b-A?VHu#%P~K&zl(|g ze>k}t$-j{MuiSIlA7`&;ew1lu=4ZY(b2R;xbcXpYCY^dU`OD;P;%AAi_`k)=v44nt zdirmsSEv3w`mHG(U5@-|BEYcb%0u%v?07ZZz@PRu)kxF}Zw|f_ zflO7>G{fK&OSL&kFm2Az1d~IWC20zebzRcI27_!iE!*bm96fE@9KuQEn7S#4shB>+9(a;Nu4F&o-2uZqN+nj(@m6Ie} z;xxk+IYpIZS+ZnlIinml@Did&ydSd3J_5M!u6M%SpPmtwtiLKrqdS zwL!j_8iefkQOzqw0R}oRy_{f9s;1kT zC0OQu3f9e;P*jl&DmvJFr$iJrjR(aRXHI(rFrI>SQ(#P-5CsW+((DnU2#P9;?O2RC z<a7688WXSDJ0Jx8M25}L21XQnPb@Fhd1L5N+zKxj-|+s$!QXgK*^aRr*^ZSf#XiSicnhL${XN#bK`op|P*fPZ&_vm0ci?Ihnh^i>6D)h+*kf_37 zQ|0z_l$rDBNVyvp2SSCS2pGr+3~di6ppJ+5{{)Ua7=L6s_oZAi`?<{jW}ct<>5P;9 zQTi(L_sr$g_fwaWf1A9Jzmxc@__yPaBrbj+Y%|(PiU37`B5-d6Ui=7i{Ioltb>|V^ zhPEMN3|2MbOihM)%z-IPcd*8(8b-PxJECElp?S>PQB%AfwW6R3Fq?XcJ5-_}SwSzH zX6C%sz1d@uZ6ih2w2qA)Y#V9dL8$p7kq3g|h^u!i4amj8WK!bqV1maO&}kCj&n0L|ey2O$3P?00WfBAd>3c4>NP~p2On0 zu>O|`RGT6$TfDN$u1zY9}S;L$ytFmI4wvI(-e_%Nyf%;mC;=ay zjD_P}3VNRFgegEski*U*c~7IL0z*~QPCUdcc^ITZe(PijQFsue@JfP9h=O>H*V?IL z%)+v#paX7e-NY6oez1oM>5~$8Fg_j=CRLNcclhinUz}v(9NI;2*RZ#@@>*ye_c;bYp;1p{iKhTYM1~JtH zO=`zx2YAI3Cj*3nR|n#uNP?!~cI-jsnCNK?&nxaN{OzU7(!s5kY+mCGNdm25k_pUI zbg#ZVuqDcKp=9)%S@`*>Vv{eBauT MXSpCO%8EAifA7{;4*&oF diff --git a/src/maestro/server/internals/status_manager.py b/src/maestro/server/internals/status_manager.py index c7ff9c7..0c03038 100644 --- a/src/maestro/server/internals/status_manager.py +++ b/src/maestro/server/internals/status_manager.py @@ -50,12 +50,18 @@ class StatusManager: + # 🆕 Variabile di classe per conservare l'istanza corrente + _instance = None + def __init__(self, db_path: str): self.db_path = db_path self.engine = self._get_engine() self.Session = self._get_session_factory() self._initialize_tables_once() + # 🆕 Memorizza l'istanza globale + StatusManager._instance = self + def _get_engine(self): return create_engine(f'sqlite:///{self.db_path}') @@ -306,6 +312,16 @@ def log_message(self, dag_id: str, execution_id: str, task_id: str, level: str, log = LogORM(dag_id=dag_id, execution_id=execution_id, task_id=task_id, level=level, message=message, thread_id=str(threading.current_thread().ident)) session.add(log) + # 🆕 Metodo helper per aggiungere log in modo più comodo + def add_log(self, dag_id: str, execution_id: str, task_id: str, message: str, level: str = "INFO"): + self.log_message( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id, + level=level, + message=message + ) + def get_execution_logs(self, dag_id: str, execution_id: str = None, limit: int = 100) -> List[Dict[str, Any]]: with self.Session() as session: query = session.query(LogORM).filter_by(dag_id=dag_id) @@ -386,4 +402,12 @@ def delete_dag(self, dag_id: str) -> int: if dag: session.delete(dag) return 1 - return 0 \ No newline at end of file + return 0 + + + # 🆕 Metodo per recuperare l'istanza globale + @classmethod + def get_instance(cls): + if cls._instance is None: + raise RuntimeError("StatusManager has not been initialized yet.") + return cls._instance diff --git a/src/maestro/server/tasks/python_task.py b/src/maestro/server/tasks/python_task.py index 2461c9f..8d51d44 100644 --- a/src/maestro/server/tasks/python_task.py +++ b/src/maestro/server/tasks/python_task.py @@ -1,21 +1,41 @@ from typing import Optional from pydantic import Field +import io +import contextlib +import threading + from maestro.server.tasks.base import BaseTask +from maestro.server.internals.status_manager import StatusManager class PythonTask(BaseTask): """ - Executes a Python snippet provided as a string or a file path. + Executes a Python snippet or script, and logs any printed output. """ + code: Optional[str] = Field(default=None, description="Inline Python code to execute") script_path: Optional[str] = Field(default=None, description="Path to a .py file to execute") def execute_local(self): - if self.code: - exec(self.code, {}) - elif self.script_path: - with open(self.script_path, "r") as f: - code = f.read() - exec(code, {}) - else: - raise ValueError("PythonTask requires either 'code' or 'script_path'.") + buffer = io.StringIO() + + with contextlib.redirect_stdout(buffer), contextlib.redirect_stderr(buffer): + if self.code: + exec(self.code, {}) + elif self.script_path: + with open(self.script_path, "r") as f: + code = f.read() + exec(code, {}) + else: + raise ValueError("PythonTask requires either 'code' or 'script_path'.") + + output = buffer.getvalue().strip() + if output: + sm = StatusManager.get_instance() + sm.add_log( + dag_id=getattr(self, "dag_id", None), + execution_id=getattr(self, "execution_id", None), + task_id=self.task_id, + message=output, + level="INFO" + ) diff --git a/tests/maestro.db b/tests/maestro.db deleted file mode 100644 index 7c2623f8330bf96e76f61b4cad3031d9e89d42f3..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 32768 zcmeI)%TMD*90zcl03lhBw@AoRdl_@vKheLS#~ymxTmOijEA`l8f8#tcgf2+kqV3m;<6&N&`OFyMN$~DRn{LS2 zzF)6cAuCAtC0UlI^{a-<6>!fy;x$~j}5k6-rSsG>6Bv~m_3sC3Ac|z*RS@yPn zHj1f-+%f5I^c8A@v9YC#Xxp#UsEL;o!oxbZoUZysZnqlndURgD?(jO>*f#Vvy(E%$ zN*i0n(i8Tf{)A1q&J=4je~NXtH7jziJ}&7SYuh5{L?cs{mGo7;q;D_lyP{kmYPgj5 zLPgyr+6j$v>1;MRs6J9<$*ns4B>2Lk`xDbT3jOG3Gjcbz=rDAi8Az&QWAZl-qOtP) zgFx(q7e-3NGzY5L<7);=bg|u{n4dUjEXw1@-0L2w3J(J7fHOn?)DU%rF2#garPdxd zwxX{V%bNyUE|&<+FvVc*8pW-hDF0Oh7aN^Rw%3hAI(zv*pT6hWUL+%`TUD45k0_A$15h1M|YE>RZQgLy+l+< zHIJ(8@_=f7`dOm~v=k)HBhuXJfymgn74_>P+VgEID&1Up5OD@ImTAqk8}2mf_N-G% zo1x<2X-#xIvr~7gp~#y#6R{M?Q*;BMtK0dg{Ojv>a{zAE`)8ILngO@{sw1?tu5Nxp zNf&xqxT=?h|Du=q%*}ckALGJJ#`x?kKa-!!%`D{d8p~@-+Tv0{OViaYD}PAz!Uh2d zKmY;|fB*y_009U<00Izzz?&d2nwU(rR-Kv)Xwm2Y%FhzLut5L<5P$##AOHafKmY;| zfB*y_aHj;;<)QNAJJh6sMQ2VcEG{nO=4T3p+-!c9=d8V1J2!9f1#4k`PqSz8v0MJ; zzZpofpZZg3eE$DZQeNJv4x%v-fB*y_009U<00Izz00bZa0SH_nu+U$gOx_^!Pyo{$)S0OhNE Date: Tue, 11 Nov 2025 20:18:05 +0100 Subject: [PATCH 02/38] PythonTask message added to logs table --- src/maestro/server/internals/orchestrator.py | 4 ++++ src/maestro/server/tasks/base.py | 8 +++++++- src/maestro/server/tasks/python_task.py | 15 +++++++++++---- 3 files changed, 22 insertions(+), 5 deletions(-) diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index 2a74d53..bfc7179 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -453,6 +453,10 @@ def _execute_task_async( if db_handler: db_handler.set_context(dag_id, execution_id, task_id) + # 🆕 Passa il contesto alla task (serve per i log) + task.dag_id = dag_id + task.execution_id = execution_id + # Execute the task self.logger.info(f"Executing task: {task_id}") executor_instance = self.executor_factory.get_executor(task.executor) diff --git a/src/maestro/server/tasks/base.py b/src/maestro/server/tasks/base.py index 314ec58..24f0fb7 100644 --- a/src/maestro/server/tasks/base.py +++ b/src/maestro/server/tasks/base.py @@ -1,6 +1,12 @@ +from typing import Optional from maestro.shared.task import Task class BaseTask(Task): """Base class for a task, inheriting from the core Task which is a Pydantic model.""" + + # 🆕 aggiungiamo questi due campi + dag_id: Optional[str] = None + execution_id: Optional[str] = None + def execute_local(self): - raise NotImplementedError("Subclasses must implement this method.") \ No newline at end of file + raise NotImplementedError("Subclasses must implement this method.") diff --git a/src/maestro/server/tasks/python_task.py b/src/maestro/server/tasks/python_task.py index 8d51d44..b4ee339 100644 --- a/src/maestro/server/tasks/python_task.py +++ b/src/maestro/server/tasks/python_task.py @@ -10,7 +10,9 @@ class PythonTask(BaseTask): """ - Executes a Python snippet or script, and logs any printed output. + Executes inline Python code or a .py script. + Captures stdout/stderr (print statements) and writes them into logs table + with proper dag_id, execution_id, and formatted message. """ code: Optional[str] = Field(default=None, description="Inline Python code to execute") @@ -30,12 +32,17 @@ def execute_local(self): raise ValueError("PythonTask requires either 'code' or 'script_path'.") output = buffer.getvalue().strip() + if output: sm = StatusManager.get_instance() + + # 🆕 aggiungiamo prefisso e usiamo i campi reali + message = f"[{self.__class__.__name__}] {output}" + sm.add_log( - dag_id=getattr(self, "dag_id", None), - execution_id=getattr(self, "execution_id", None), + dag_id=self.dag_id, + execution_id=self.execution_id, task_id=self.task_id, - message=output, + message=message, level="INFO" ) From 2a6a83c58290ba4c26000a694df7106ed5088265 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 11 Nov 2025 22:29:05 +0100 Subject: [PATCH 03/38] Added BashTask messages to logs table --- src/maestro/server/internals/orchestrator.py | 74 +++++++++++++------- src/maestro/server/tasks/bash_task.py | 24 +++++-- 2 files changed, 67 insertions(+), 31 deletions(-) diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index bfc7179..044d366 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -18,6 +18,7 @@ from apscheduler.schedulers.background import BackgroundScheduler + class DatabaseLogHandler(logging.Handler): """Custom logging handler that writes to StatusManager database.""" @@ -54,51 +55,69 @@ def emit(self, record): class Orchestrator: - def __init__(self, log_level: str = "INFO", status_manager: Optional[StatusManager] = None, db_path: Optional[str] = "maestro.db"): + + def __init__( + self, + log_level: str = "INFO", + status_manager: Optional[StatusManager] = None, + db_path: Optional[str] = "maestro.db" + ): self.task_registry = TaskRegistry() self.dag_loader = DAGLoader(self.task_registry) self.console = get_console() + if status_manager: self.status_manager = status_manager elif db_path: self.status_manager = StatusManager(db_path) else: self.status_manager = StatusManager() + self.executor_factory = ExecutorFactory() + + # ⬇️ qui configuriamo i logger (vedi sotto) self._setup_logging(log_level) self.executor = ThreadPoolExecutor(max_workers=10) self._execution_stop_events: Dict[str, threading.Event] = {} self.task_executor = ThreadPoolExecutor(max_workers=10) - def _setup_logging(self, log_level: str): - """Setup logging with Rich handler.""" - logging.basicConfig( - level=getattr(logging, log_level.upper()), - format="%(name)s - %(message)s", - handlers=[RichHandler(console=self.console, rich_tracebacks=True)] - ) - # Clear existing handlers from the root logger to prevent duplicate logs - for handler in logging.root.handlers[:]: - logging.root.removeHandler(handler) - - # Add RichHandler for console output - rich_handler = RichHandler(console=self.console, rich_tracebacks=True) - rich_handler.setLevel(getattr(logging, log_level.upper())) - logging.root.addHandler(rich_handler) + """Setup global logging with RichHandler (console) and DatabaseLogHandler (persistent).""" - # Add DatabaseLogHandler for persistence - self.db_handler = DatabaseLogHandler(self.status_manager) - self.db_handler.setLevel(logging.DEBUG) # Capture all levels for database - logging.root.addHandler(self.db_handler) + # Pulisci logging root + logging.shutdown() + for h in logging.root.handlers[:]: + logging.root.removeHandler(h) - # Set the root logger level - logging.root.setLevel(getattr(logging, log_level.upper())) + level = getattr(logging, log_level.upper(), logging.INFO) + logging.root.setLevel(level) - self.logger = logging.getLogger(__name__) + # Console + rich_handler = RichHandler(console=self.console, rich_tracebacks=True) + rich_handler.setLevel(level) - + # DB + self.db_handler = DatabaseLogHandler(self.status_manager) + self.db_handler.setLevel(logging.DEBUG) + + root_logger = logging.getLogger() + root_logger.addHandler(rich_handler) + root_logger.addHandler(self.db_handler) + logging.captureWarnings(True) + root_logger.propagate = True + + # Logger dell’orchestratore + self.logger = logging.getLogger("maestro.orchestrator") + self.logger.setLevel(level) + self.logger.propagate = True + + # 🔧 AGGANCIA il DB handler anche a tutti i logger di maestro già creati + for name in list(logging.root.manager.loggerDict.keys()): + if name.startswith("maestro."): + lg = logging.getLogger(name) + lg.addHandler(self.db_handler) + lg.propagate = True def register_task_type(self, name: str, task_class: Type[BaseTask]): """Register a custom task type.""" @@ -254,23 +273,26 @@ def run_dag( # Set up database logging if execution_id is provided db_handler = None + if execution_id: db_handler = DatabaseLogHandler(self.status_manager) db_handler.set_context(dag_id, execution_id) - # Add the database handler to all task-related loggers task_loggers = [ 'maestro.server.tasks.terraform_task', 'maestro.server.tasks.extended_terraform_task', 'maestro.server.tasks.print_task', + 'maestro.server.tasks.python_task', + 'maestro.server.tasks.bash_task', # ✅ eccolo 'maestro.core.executors.ssh', 'maestro.core.executors.docker', - 'maestro.core.executors.local' + 'maestro.core.executors.local', ] for logger_name in task_loggers: logger = logging.getLogger(logger_name) logger.addHandler(db_handler) + logger.propagate = True try: # Use concurrent execution diff --git a/src/maestro/server/tasks/bash_task.py b/src/maestro/server/tasks/bash_task.py index 475c626..8aec28c 100644 --- a/src/maestro/server/tasks/bash_task.py +++ b/src/maestro/server/tasks/bash_task.py @@ -1,15 +1,16 @@ -# maestro/src/server/task/bash_task.py - import subprocess import logging from maestro.server.tasks.base import BaseTask logger = logging.getLogger(__name__) +logger.propagate = True class BashTask(BaseTask): """ A task that executes a Bash command locally. + Captures stdout and stderr and sends them to the logging system. """ + command: str def execute_local(self): @@ -22,14 +23,27 @@ def execute_local(self): capture_output=True, text=True ) + + # --- LOG STDOUT LINE BY LINE --- + if result.stdout: + for line in result.stdout.strip().splitlines(): + logger.info(f"[BashTask][stdout] {line}") + + # --- LOG STDERR LINE BY LINE --- + if result.stderr: + for line in result.stderr.strip().splitlines(): + logger.error(f"[BashTask][stderr] {line}") + + # --- Set task status --- if result.returncode == 0: - logger.info(f"[BashTask] Output:\n{result.stdout.strip()}") + logger.info("[BashTask] Command completed successfully.") self.status = "completed" else: - logger.error(f"[BashTask] Error:\n{result.stderr.strip()}") + logger.error(f"[BashTask] Command failed with return code {result.returncode}.") self.status = "failed" + except Exception as e: - logger.exception(f"[BashTask] Exception: {e}") + logger.exception(f"[BashTask] Exception while executing command: {e}") self.status = "failed" def to_dict(self): From 2ac3899ff834f3714e2fdcde82280b506e698a69 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Wed, 12 Nov 2025 19:40:58 +0100 Subject: [PATCH 04/38] Fields in executions table completed --- maestro.db | Bin 0 -> 45056 bytes src/maestro/server/internals/status_manager.py | 7 +++++++ 2 files changed, 7 insertions(+) create mode 100644 maestro.db diff --git a/maestro.db b/maestro.db new file mode 100644 index 0000000000000000000000000000000000000000..1470304c5c98feb3bd7c8b9fcca62031c14d7f45 GIT binary patch literal 45056 zcmeI*-%r~{00(fJ03m5<`cM>g)r84QiAqSE-(jlOlp2OcLzf?2tG2RCe1JD6&TPY= ztlMC%v^`AwFSh?++Wvz{le&Ln?|Yf{v^zU7NeZR3piaxziWB?J_T7E%47sy0m5-J+ zn+R)$*-&gD#a-lhp1UCk9LHUv^Eo7H=x?!rn*=s$!o-IGFeZ+6Td zj>Y)>kgcrhFQnMNd!U_zYVqc>bjZPj0ZkR|7R&P=6wA|ML5vEem8wvxEicaq!9Y*a z!ou=OvC4vJ!DY)<=-1Jk%PTi~_PA4Cx?L>a7e18k3)7k!4MyJ&jD{;Qo^u4P&vg1N zR#rA`!&!gPQ|vE6aR19ue>fcGpFMC&Y%A8MG;#Xo?B_=-YZWI;D(t=c$O7l{M?PobP z(Z6lafx7qbBHaCg(Wrkkd_8?d7P?kEPicq?0Y8cUM${SeCL*gsODUX}jz+rIpucdG#jTk`{`!<*G1W zD^s+p%qp`|E#AJv(mu4}`fBGy^$?8izvJ_Vr>6LCzH*Sf(pd*f!#!XJ;jh=evf z=;?HPlRY6$YIrofICWHGJCKmYgXNLEIPWyO{h}kv+B}}>Sf z3TugLnoe4Zz0uvK>hxUhJI7z%i2B=0FnV=pH2mJgQN?6gl|}aMqrYay*st1gI`ij# z;^+$#1Rwwb2tWV=5P$##An@N1`1UUEpZwwDy=ZFH&{f&gRI(dk{RmBs%n1=MFft=V zoG}Dv+=xasH+7wc)lE7I(HUYo2zGkcbamI=$ysXme(7>MknY~;vB<2;!eZhZMuWsv zWsAslQ;k=4n)UeI%DkbHctfEbVbh3{ZQ2{wi4{-ECGxrKt%~xhVrg|**P28zXLpo_ z&T_4i7NLV-&ALV`n!(@sB)*o|}$g5r?_Kp7)_6n z^@c6hhI$m+p>GMXQ-^24*wyA}Qxm+Ays1}cbx)q0?s=9>Br`EljEPA>%+I9?bE$NY z{(R#@e{uAM2?7v+00bZa0SG_<0uX=z1Rwx`|Bt|V-(bKmY;| zfB*y_009U<00Izzz^MxC`T4Qh`pIynH&E)dwLgKKPLL!~b<-}xGOl6aaL^_wxr}O47vVP-h z_xk^HF7*6V3y6b300Izz00bZa0SG_<0uX=z1R!u`0^)FOGH{|11$O=4cVV9kJqvO4 z0}})w009U<00Izz00bZa0SG|gj0C>$1-QxD<%weIns-5Z!`pxG?p_5?6mr=@CY?<04Gpp6uS|S+!sOe}3~RqfY9GWKuRb0MTyaYiXpbtKUU{;GY&vr&WiF9V=aTIHe_yD=g({)% zL%+RoQE&hVKmY;|fB*y_009U<00Izzz@P{eM(DW6>x_x`=0@l^0RK8;07JPEIw0cy z`_K~&98RH7F+pAOHafKmY;|fB*y_009U Date: Wed, 12 Nov 2025 21:49:46 +0100 Subject: [PATCH 05/38] Logs in client implemented successfully --- maestro.db | Bin 45056 -> 0 bytes src/maestro/client/cli.py | 40 ++++++++++++++++++----- src/maestro/server/api/v1/routes/logs.py | 19 +++++++---- 3 files changed, 44 insertions(+), 15 deletions(-) delete mode 100644 maestro.db diff --git a/maestro.db b/maestro.db deleted file mode 100644 index 1470304c5c98feb3bd7c8b9fcca62031c14d7f45..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 45056 zcmeI*-%r~{00(fJ03m5<`cM>g)r84QiAqSE-(jlOlp2OcLzf?2tG2RCe1JD6&TPY= ztlMC%v^`AwFSh?++Wvz{le&Ln?|Yf{v^zU7NeZR3piaxziWB?J_T7E%47sy0m5-J+ zn+R)$*-&gD#a-lhp1UCk9LHUv^Eo7H=x?!rn*=s$!o-IGFeZ+6Td zj>Y)>kgcrhFQnMNd!U_zYVqc>bjZPj0ZkR|7R&P=6wA|ML5vEem8wvxEicaq!9Y*a z!ou=OvC4vJ!DY)<=-1Jk%PTi~_PA4Cx?L>a7e18k3)7k!4MyJ&jD{;Qo^u4P&vg1N zR#rA`!&!gPQ|vE6aR19ue>fcGpFMC&Y%A8MG;#Xo?B_=-YZWI;D(t=c$O7l{M?PobP z(Z6lafx7qbBHaCg(Wrkkd_8?d7P?kEPicq?0Y8cUM${SeCL*gsODUX}jz+rIpucdG#jTk`{`!<*G1W zD^s+p%qp`|E#AJv(mu4}`fBGy^$?8izvJ_Vr>6LCzH*Sf(pd*f!#!XJ;jh=evf z=;?HPlRY6$YIrofICWHGJCKmYgXNLEIPWyO{h}kv+B}}>Sf z3TugLnoe4Zz0uvK>hxUhJI7z%i2B=0FnV=pH2mJgQN?6gl|}aMqrYay*st1gI`ij# z;^+$#1Rwwb2tWV=5P$##An@N1`1UUEpZwwDy=ZFH&{f&gRI(dk{RmBs%n1=MFft=V zoG}Dv+=xasH+7wc)lE7I(HUYo2zGkcbamI=$ysXme(7>MknY~;vB<2;!eZhZMuWsv zWsAslQ;k=4n)UeI%DkbHctfEbVbh3{ZQ2{wi4{-ECGxrKt%~xhVrg|**P28zXLpo_ z&T_4i7NLV-&ALV`n!(@sB)*o|}$g5r?_Kp7)_6n z^@c6hhI$m+p>GMXQ-^24*wyA}Qxm+Ays1}cbx)q0?s=9>Br`EljEPA>%+I9?bE$NY z{(R#@e{uAM2?7v+00bZa0SG_<0uX=z1Rwx`|Bt|V-(bKmY;| zfB*y_009U<00Izzz^MxC`T4Qh`pIynH&E)dwLgKKPLL!~b<-}xGOl6aaL^_wxr}O47vVP-h z_xk^HF7*6V3y6b300Izz00bZa0SG_<0uX=z1R!u`0^)FOGH{|11$O=4cVV9kJqvO4 z0}})w009U<00Izz00bZa0SG|gj0C>$1-QxD<%weIns-5Z!`pxG?p_5?6mr=@CY?<04Gpp6uS|S+!sOe}3~RqfY9GWKuRb0MTyaYiXpbtKUU{;GY&vr&WiF9V=aTIHe_yD=g({)% zL%+RoQE&hVKmY;|fB*y_009U<00Izzz@P{eM(DW6>x_x`=0@l^0RK8;07JPEIw0cy z`_K~&98RH7F+pAOHafKmY;|fB*y_009U Date: Sat, 15 Nov 2025 17:35:04 +0100 Subject: [PATCH 06/38] Added a new set of brand new examples --- ...ic_linear.yaml => 1.1.1.Basic_linear.yaml} | 0 examples/2_New_examples/1.2.1.Long_waits.yaml | 47 +++++ .../1.2.2.Parallel_then_join.yaml | 47 +++++ .../2_New_examples/1.2.3.Bash_python_mix.yaml | 37 ++++ .../1.3.1.Branching_waits_and_joins.yaml | 56 ++++++ .../1.3.2.Cascade_retry_delays.yaml | 43 +++++ .../1.3.3.Parallel_heavy_bash_python.yaml | 37 ++++ ..._bash_chain.yaml => 2.1.1.Bash_chain.yaml} | 0 .../2.2.1.bash_chain_plus_scan.yaml | 33 ++++ .../2.2.2.bash_chain_branch.yaml | 37 ++++ .../2.2.3.bash_chain_with_error_handling.yaml | 35 ++++ .../2.3.1.bash_chain_heavy_diamond.yaml | 76 ++++++++ .../2.3.2.bash_chain_error_storm.yaml | 48 +++++ .../2.3.3.bash_chain_long_pipeline.yaml | 73 +++++++ ...d_retry.yaml => 3.1.1.Wait_and_retry.yaml} | 0 .../3.2.1.wait_and_retry_with_logging.yaml | 46 +++++ .../3.2.2.wait_and_retry_bash_python_mix.yaml | 37 ++++ .../3.2.3.wait_and_retry_branching.yaml | 50 +++++ .../3.3.1.wait_and_retry_heavy_parallel.yaml | 62 ++++++ ...3.3.2.wait_and_retry_cascade_failures.yaml | 54 ++++++ .../3.3.3.wait_and_retry_spiderweb.yaml | 104 ++++++++++ ....yaml => 4.1.1.Conditional_branching.yaml} | 0 .../4.2.1.conditional_branching_extended.yaml | 65 +++++++ ...4.2.2.conditional_branching_with_bash.yaml | 72 +++++++ ...conditional_branching_parallel_python.yaml | 79 ++++++++ ...3.1.conditional_branching_triple_tree.yaml | 112 +++++++++++ .../4.3.2.conditional_branching_matrix.yaml | 136 +++++++++++++ ....3.3.conditional_branching_hypergraph.yaml | 157 +++++++++++++++ ...artbeat.yaml => 5.1.1.Cron_heartbeat.yaml} | 0 .../5.2.1.cron_heartbeat_extended.yaml | 28 +++ .../5.2.2.cron_heartbeat_with_status.yaml | 31 +++ .../5.2.3.cron_heartbeat_light_checks.yaml | 43 +++++ ...5.3.1.cron_heartbeat_full_healthcheck.yaml | 58 ++++++ .../5.3.2.cron_heartbeat_with_warnings.yaml | 51 +++++ .../5.3.3.cron_heartbeat_deep_pipeline.yaml | 71 +++++++ ...ing.yaml => 6.1.1.Scheduled_greeting.yaml} | 0 .../6.2.1.scheduled_greeting_extended.yaml | 59 ++++++ ...6.2.2.scheduled_greeting_language_mix.yaml | 83 ++++++++ .../6.2.3.scheduled_greeting_with_checks.yaml | 60 ++++++ ...1.scheduled_greeting_full_healthcheck.yaml | 125 ++++++++++++ ....3.2.scheduled_greeting_with_warnings.yaml | 107 +++++++++++ .../6.3.3.scheduled_greeting_deep_tree.yaml | 118 ++++++++++++ ...cution.yaml => 7.1.1.Mixed_execution.yaml} | 0 .../7.2.1.mixed_execution_extended.yaml | 83 ++++++++ .../7.2.2.mixed_execution_branching.yaml | 84 ++++++++ .../7.2.3.mixed_execution_random_data.yaml | 86 +++++++++ .../7.3.1.mixed_execution_data_pipeline.yaml | 141 ++++++++++++++ .../7.3.2.mixed_execution_transform_tree.yaml | 139 ++++++++++++++ .../7.3.3.mixed_execution_hyperpipeline.yaml | 180 ++++++++++++++++++ examples/2_New_examples/DAG explanation.ods | Bin 13328 -> 0 bytes 50 files changed, 3090 insertions(+) rename examples/2_New_examples/{1_basic_linear.yaml => 1.1.1.Basic_linear.yaml} (100%) create mode 100644 examples/2_New_examples/1.2.1.Long_waits.yaml create mode 100644 examples/2_New_examples/1.2.2.Parallel_then_join.yaml create mode 100644 examples/2_New_examples/1.2.3.Bash_python_mix.yaml create mode 100644 examples/2_New_examples/1.3.1.Branching_waits_and_joins.yaml create mode 100644 examples/2_New_examples/1.3.2.Cascade_retry_delays.yaml create mode 100644 examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml rename examples/2_New_examples/{2_bash_chain.yaml => 2.1.1.Bash_chain.yaml} (100%) create mode 100644 examples/2_New_examples/2.2.1.bash_chain_plus_scan.yaml create mode 100644 examples/2_New_examples/2.2.2.bash_chain_branch.yaml create mode 100644 examples/2_New_examples/2.2.3.bash_chain_with_error_handling.yaml create mode 100644 examples/2_New_examples/2.3.1.bash_chain_heavy_diamond.yaml create mode 100644 examples/2_New_examples/2.3.2.bash_chain_error_storm.yaml create mode 100644 examples/2_New_examples/2.3.3.bash_chain_long_pipeline.yaml rename examples/2_New_examples/{3_wait_and_retry.yaml => 3.1.1.Wait_and_retry.yaml} (100%) create mode 100644 examples/2_New_examples/3.2.1.wait_and_retry_with_logging.yaml create mode 100644 examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml create mode 100644 examples/2_New_examples/3.2.3.wait_and_retry_branching.yaml create mode 100644 examples/2_New_examples/3.3.1.wait_and_retry_heavy_parallel.yaml create mode 100644 examples/2_New_examples/3.3.2.wait_and_retry_cascade_failures.yaml create mode 100644 examples/2_New_examples/3.3.3.wait_and_retry_spiderweb.yaml rename examples/2_New_examples/{4_conditional_branching.yaml => 4.1.1.Conditional_branching.yaml} (100%) create mode 100644 examples/2_New_examples/4.2.1.conditional_branching_extended.yaml create mode 100644 examples/2_New_examples/4.2.2.conditional_branching_with_bash.yaml create mode 100644 examples/2_New_examples/4.2.3.conditional_branching_parallel_python.yaml create mode 100644 examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml create mode 100644 examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml create mode 100644 examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml rename examples/2_New_examples/{5_cron_heartbeat.yaml => 5.1.1.Cron_heartbeat.yaml} (100%) create mode 100644 examples/2_New_examples/5.2.1.cron_heartbeat_extended.yaml create mode 100644 examples/2_New_examples/5.2.2.cron_heartbeat_with_status.yaml create mode 100644 examples/2_New_examples/5.2.3.cron_heartbeat_light_checks.yaml create mode 100644 examples/2_New_examples/5.3.1.cron_heartbeat_full_healthcheck.yaml create mode 100644 examples/2_New_examples/5.3.2.cron_heartbeat_with_warnings.yaml create mode 100644 examples/2_New_examples/5.3.3.cron_heartbeat_deep_pipeline.yaml rename examples/2_New_examples/{6_scheduled_greeting.yaml => 6.1.1.Scheduled_greeting.yaml} (100%) create mode 100644 examples/2_New_examples/6.2.1.scheduled_greeting_extended.yaml create mode 100644 examples/2_New_examples/6.2.2.scheduled_greeting_language_mix.yaml create mode 100644 examples/2_New_examples/6.2.3.scheduled_greeting_with_checks.yaml create mode 100644 examples/2_New_examples/6.3.1.scheduled_greeting_full_healthcheck.yaml create mode 100644 examples/2_New_examples/6.3.2.scheduled_greeting_with_warnings.yaml create mode 100644 examples/2_New_examples/6.3.3.scheduled_greeting_deep_tree.yaml rename examples/2_New_examples/{7_mixed_execution.yaml => 7.1.1.Mixed_execution.yaml} (100%) create mode 100644 examples/2_New_examples/7.2.1.mixed_execution_extended.yaml create mode 100644 examples/2_New_examples/7.2.2.mixed_execution_branching.yaml create mode 100644 examples/2_New_examples/7.2.3.mixed_execution_random_data.yaml create mode 100644 examples/2_New_examples/7.3.1.mixed_execution_data_pipeline.yaml create mode 100644 examples/2_New_examples/7.3.2.mixed_execution_transform_tree.yaml create mode 100644 examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml delete mode 100644 examples/2_New_examples/DAG explanation.ods diff --git a/examples/2_New_examples/1_basic_linear.yaml b/examples/2_New_examples/1.1.1.Basic_linear.yaml similarity index 100% rename from examples/2_New_examples/1_basic_linear.yaml rename to examples/2_New_examples/1.1.1.Basic_linear.yaml diff --git a/examples/2_New_examples/1.2.1.Long_waits.yaml b/examples/2_New_examples/1.2.1.Long_waits.yaml new file mode 100644 index 0000000..b52c355 --- /dev/null +++ b/examples/2_New_examples/1.2.1.Long_waits.yaml @@ -0,0 +1,47 @@ +# ------------------------------------------------------------------------------------- +# Intermediate Long Waits +# ------------------------------------------------------------------------------------- +# This DAG implements a linear pipeline composed of three consecutive steps, each characterized by an artificial wait phase (a 20-second sleep). The goal is to simulate a sequence of slow processing, allowing testing of the orchestrator's "attached" mode for a total duration of approximately 60 seconds. The structure is simple, but useful for verifying the behavior of streaming logs in a linear flow. +# ------------------------------------------------------------------------------------- + +dag: + name: "intermediate_long_waits" + tasks: + - task_id: "start" + type: "PrintTask" + params: + message: "Starting long wait DAG" + dependencies: [] + + - task_id: "step1" + type: "PythonTask" + params: + code: | + import time + print("Step 1 running...") + time.sleep(20) + dependencies: ["start"] + + - task_id: "step2" + type: "PythonTask" + params: + code: | + import time + print("Step 2 running...") + time.sleep(20) + dependencies: ["step1"] + + - task_id: "step3" + type: "PythonTask" + params: + code: | + import time + print("Step 3 running...") + time.sleep(20) + dependencies: ["step2"] + + - task_id: "finish" + type: "PrintTask" + params: + message: "All steps completed!" + dependencies: ["step3"] diff --git a/examples/2_New_examples/1.2.2.Parallel_then_join.yaml b/examples/2_New_examples/1.2.2.Parallel_then_join.yaml new file mode 100644 index 0000000..cafd4b1 --- /dev/null +++ b/examples/2_New_examples/1.2.2.Parallel_then_join.yaml @@ -0,0 +1,47 @@ +# ------------------------------------------------------------------------------------- +# Parallel then Join +# ------------------------------------------------------------------------------------- +# This DAG involves three parallel tasks of different durations, all dependent on a single entry point. Once completed, they converge into a single final merge task. The maximum duration of the slowest branch (approximately 40 seconds) defines the total execution time. It is particularly suitable for testing streaming visualization of parallel tasks and synchronization during the join phase. +# ------------------------------------------------------------------------------------- + +dag: + name: "intermediate_parallel_then_join" + tasks: + - task_id: "start" + type: "PrintTask" + params: + message: "Starting parallel DAG" + dependencies: [] + + - task_id: "worker_a" + type: "PythonTask" + params: + code: | + import time + print("Worker A...") + time.sleep(30) + dependencies: ["start"] + + - task_id: "worker_b" + type: "PythonTask" + params: + code: | + import time + print("Worker B...") + time.sleep(40) + dependencies: ["start"] + + - task_id: "worker_c" + type: "PythonTask" + params: + code: | + import time + print("Worker C...") + time.sleep(30) + dependencies: ["start"] + + - task_id: "merge" + type: "PrintTask" + params: + message: "Workers completed!" + dependencies: ["worker_a", "worker_b", "worker_c"] diff --git a/examples/2_New_examples/1.2.3.Bash_python_mix.yaml b/examples/2_New_examples/1.2.3.Bash_python_mix.yaml new file mode 100644 index 0000000..2c6a863 --- /dev/null +++ b/examples/2_New_examples/1.2.3.Bash_python_mix.yaml @@ -0,0 +1,37 @@ +# ------------------------------------------------------------------------------------- +# Bash/Python Mix +# ------------------------------------------------------------------------------------- +# A mixed pipeline that alternates Bash and Python tasks, both with significant delays (25-second sleep times each). It is used to test the coexistence and behavior of logs from different executors, as well as the management of sequential wait times. Total duration is approximately 50–60 seconds. +# ------------------------------------------------------------------------------------- + +dag: + name: "intermediate_bash_python_mix" + tasks: + - task_id: "start" + type: "PrintTask" + params: + message: "Starting mixed DAG" + dependencies: [] + + - task_id: "bash_wait" + type: "BashTask" + params: + script: | + echo "Bash waiting 25s..." + sleep 25 + dependencies: ["start"] + + - task_id: "python_wait" + type: "PythonTask" + params: + code: | + import time + print("Python waiting 25s...") + time.sleep(25) + dependencies: ["bash_wait"] + + - task_id: "finish" + type: "PrintTask" + params: + message: "Finished mixed DAG!" + dependencies: ["python_wait"] diff --git a/examples/2_New_examples/1.3.1.Branching_waits_and_joins.yaml b/examples/2_New_examples/1.3.1.Branching_waits_and_joins.yaml new file mode 100644 index 0000000..049e9a9 --- /dev/null +++ b/examples/2_New_examples/1.3.1.Branching_waits_and_joins.yaml @@ -0,0 +1,56 @@ +# ------------------------------------------------------------------------------------- +# Branching, Heavy Waits, and Multi-Join +# ------------------------------------------------------------------------------------- +# This complex DAG combines multiple branching, slow processing, and a final join phase that depends on distinct processing paths. Each branch contains slow subtasks (20–30 seconds) that are combined into a complex pipeline. It is designed to test system scalability, the management of richer graphs, and the correct synchronization of dependencies in the log stream. +# ------------------------------------------------------------------------------------- + + +dag: + name: "difficult_branching_waits_and_joins" + tasks: + - task_id: "start" + type: "PrintTask" + params: + message: "Complex DAG start" + dependencies: [] + + - task_id: "A1" + type: "PythonTask" + params: + code: | + import time; print("A1..."); time.sleep(20) + dependencies: ["start"] + + - task_id: "A2" + type: "PythonTask" + params: + code: | + import time; print("A2..."); time.sleep(25) + dependencies: ["start"] + + - task_id: "B1" + type: "PythonTask" + params: + code: | + import time; print("B1..."); time.sleep(30) + dependencies: ["A1"] + + - task_id: "B2" + type: "PythonTask" + params: + code: | + import time; print("B2..."); time.sleep(30) + dependencies: ["A2"] + + - task_id: "combine" + type: "PrintTask" + params: + message: "Combination done" + dependencies: ["B1", "B2"] + + - task_id: "final" + type: "PythonTask" + params: + code: | + import time; print("Finalizing..."); time.sleep(15) + dependencies: ["combine"] diff --git a/examples/2_New_examples/1.3.2.Cascade_retry_delays.yaml b/examples/2_New_examples/1.3.2.Cascade_retry_delays.yaml new file mode 100644 index 0000000..17b70df --- /dev/null +++ b/examples/2_New_examples/1.3.2.Cascade_retry_delays.yaml @@ -0,0 +1,43 @@ +# ------------------------------------------------------------------------------------- +# Cascade with Retries and Delay Logic +# ------------------------------------------------------------------------------------- +# This DAG tests the system by integrating unstable tasks, automatic retries with configured delays, and a slow pipeline that depends on the success of the critical task. It includes simulated failures (with a 50% probability) to stress the orchestrator's retry logic. The total duration varies based on the retries, ranging from 70 to 100+ seconds. Ideal for validating the robustness of StatusManager, TaskStatus, and logging under error conditions. +# ------------------------------------------------------------------------------------- + +dag: + name: "difficult_cascade_retry_delays" + tasks: + - task_id: "start" + type: "PrintTask" + params: + message: "Start cascade" + dependencies: [] + + - task_id: "unstable_step" + type: "PythonTask" + params: + code: | + import time, random + print("Unstable step (may fail)...") + time.sleep(15) + if random.random() < 0.5: + raise Exception("Simulated failure!") + print("Unstable succeeded") + retries: 3 + retry_delay: 10 + dependencies: ["start"] + + - task_id: "slow_processing" + type: "PythonTask" + params: + code: | + import time + print("Slow processing...") + time.sleep(30) + dependencies: ["unstable_step"] + + - task_id: "finish" + type: "PrintTask" + params: + message: "Cascade completed" + dependencies: ["slow_processing"] diff --git a/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml b/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml new file mode 100644 index 0000000..8241843 --- /dev/null +++ b/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml @@ -0,0 +1,37 @@ +# ------------------------------------------------------------------------------------- +# Heavy Parallel Bash/Python +# ------------------------------------------------------------------------------------- +# A DAG designed to generate a heavy, concurrent load on two separate executors: Bash and Python. Both main tasks start in parallel, each with a significant wait time (45–50 seconds). The final join allows you to check the logger's behavior, the smoothness of the streaming, and the correct handling of very slow parallel tasks. +# ------------------------------------------------------------------------------------- + +dag: + name: "difficult_parallel_heavy_bash_python" + tasks: + - task_id: "start" + type: "PrintTask" + params: + message: "Begin heavy parallel sequence" + dependencies: [] + + - task_id: "bash_long" + type: "BashTask" + params: + script: | + echo "Bash running long task..." + sleep 45 + dependencies: ["start"] + + - task_id: "python_long" + type: "PythonTask" + params: + code: | + import time + print("Python long task...") + time.sleep(50) + dependencies: ["start"] + + - task_id: "final_merge" + type: "PrintTask" + params: + message: "All heavy tasks completed" + dependencies: ["bash_long", "python_long"] diff --git a/examples/2_New_examples/2_bash_chain.yaml b/examples/2_New_examples/2.1.1.Bash_chain.yaml similarity index 100% rename from examples/2_New_examples/2_bash_chain.yaml rename to examples/2_New_examples/2.1.1.Bash_chain.yaml diff --git a/examples/2_New_examples/2.2.1.bash_chain_plus_scan.yaml b/examples/2_New_examples/2.2.1.bash_chain_plus_scan.yaml new file mode 100644 index 0000000..805fdfe --- /dev/null +++ b/examples/2_New_examples/2.2.1.bash_chain_plus_scan.yaml @@ -0,0 +1,33 @@ + +abstract: | + This DAG extends the original bash_chain by adding a disk usage check and a Python + file count operation. It tests sequential task execution and mixed Bash/Python usage. + +dag: + name: "bash_chain_plus_scan" + tasks: + - task_id: "list_files" + type: "BashTask" + params: + command: "ls -l" + dependencies: [] # First task: lists files + + - task_id: "disk_usage" + type: "BashTask" + params: + command: "du -sh ." + dependencies: ["list_files"] # Depends on list_files + + - task_id: "python_count_files" + type: "PythonTask" + params: + script: | + import os + print("FILE_COUNT:", len(os.listdir('.'))) + dependencies: ["disk_usage"] # Python post-processing + + - task_id: "end" + type: "PrintTask" + params: + message: "Scan complete, file count printed." + dependencies: ["python_count_files"] diff --git a/examples/2_New_examples/2.2.2.bash_chain_branch.yaml b/examples/2_New_examples/2.2.2.bash_chain_branch.yaml new file mode 100644 index 0000000..32a1c0b --- /dev/null +++ b/examples/2_New_examples/2.2.2.bash_chain_branch.yaml @@ -0,0 +1,37 @@ + +abstract: | + This DAG introduces branching: one branch lists files, another lists processes. + Each branch performs its own post-processing before merging into a final print task. + +dag: + name: "bash_chain_branch" + tasks: + - task_id: "list_files" + type: "BashTask" + params: + command: "ls -l" + dependencies: [] # Root branch A + + - task_id: "list_processes" + type: "BashTask" + params: + command: "ps aux | head" + dependencies: [] # Root branch B + + - task_id: "count_file_lines" + type: "BashTask" + params: + command: "ls -1 | wc -l" + dependencies: ["list_files"] + + - task_id: "count_processes" + type: "BashTask" + params: + command: "ps aux | wc -l" + dependencies: ["list_processes"] + + - task_id: "final_message" + type: "PrintTask" + params: + message: "Files and process info collected!" + dependencies: ["count_file_lines", "count_processes"] diff --git a/examples/2_New_examples/2.2.3.bash_chain_with_error_handling.yaml b/examples/2_New_examples/2.2.3.bash_chain_with_error_handling.yaml new file mode 100644 index 0000000..228ffd4 --- /dev/null +++ b/examples/2_New_examples/2.2.3.bash_chain_with_error_handling.yaml @@ -0,0 +1,35 @@ + +abstract: | + This DAG tests error handling with fail_fast disabled. A failing BashTask is followed + by Python and Print tasks that continue running regardless of the failure. + +dag: + name: "bash_chain_with_error_handling" + config: + fail_fast: false + + tasks: + - task_id: "list_files" + type: "BashTask" + params: + command: "ls -l" + dependencies: [] + + - task_id: "intentional_error" + type: "BashTask" + params: + command: "non_existing_command_xyz" # Guaranteed to fail + dependencies: ["list_files"] + + - task_id: "python_info" + type: "PythonTask" + params: + script: | + print("This runs even after a failure.") + dependencies: ["intentional_error"] + + - task_id: "end" + type: "PrintTask" + params: + message: "Execution completed with error handling." + dependencies: ["python_info"] diff --git a/examples/2_New_examples/2.3.1.bash_chain_heavy_diamond.yaml b/examples/2_New_examples/2.3.1.bash_chain_heavy_diamond.yaml new file mode 100644 index 0000000..a4dc303 --- /dev/null +++ b/examples/2_New_examples/2.3.1.bash_chain_heavy_diamond.yaml @@ -0,0 +1,76 @@ + +abstract: | + A heavy fan‑out/fan‑in DAG structured like a diamond. Four independent branches + perform system checks before merging into a final summary task. + +dag: + name: "bash_chain_heavy_diamond" + tasks: + - task_id: "start_scan" + type: "PrintTask" + params: + message: "Starting heavy system scan..." + dependencies: [] + + # Branch 1 + - task_id: "files_list" + type: "BashTask" + params: + command: "ls -l" + dependencies: ["start_scan"] + + - task_id: "files_count" + type: "BashTask" + params: + command: "ls -1 | wc -l" + dependencies: ["files_list"] + + # Branch 2 + - task_id: "proc_list" + type: "BashTask" + params: + command: "ps aux" + dependencies: ["start_scan"] + + - task_id: "proc_count" + type: "BashTask" + params: + command: "ps aux | wc -l" + dependencies: ["proc_list"] + + # Branch 3 + - task_id: "disk_usage" + type: "BashTask" + params: + command: "du -sh ." + dependencies: ["start_scan"] + + - task_id: "disk_details" + type: "BashTask" + params: + command: "df -h" + dependencies: ["disk_usage"] + + # Branch 4 + - task_id: "simulated_ping" + type: "BashTask" + params: + command: "echo 'Simulating ping'; sleep 1" + dependencies: ["start_scan"] + + - task_id: "parse_ping" + type: "PythonTask" + params: + script: | + print("Ping simulation parsed.") + dependencies: ["simulated_ping"] + + - task_id: "final_merge" + type: "PrintTask" + params: + message: "All branches completed!" + dependencies: + - "files_count" + - "proc_count" + - "disk_details" + - "parse_ping" diff --git a/examples/2_New_examples/2.3.2.bash_chain_error_storm.yaml b/examples/2_New_examples/2.3.2.bash_chain_error_storm.yaml new file mode 100644 index 0000000..b222619 --- /dev/null +++ b/examples/2_New_examples/2.3.2.bash_chain_error_storm.yaml @@ -0,0 +1,48 @@ + +abstract: | + This DAG intentionally triggers multiple failures in parallel branches while + fail_fast=false ensures the run continues. Useful for stress‑testing logging + and resilience mechanisms. + +dag: + name: "bash_chain_error_storm" + config: + fail_fast: false + + tasks: + - task_id: "start" + type: "PrintTask" + params: + message: "Starting error storm..." + dependencies: [] + + - task_id: "valid_ls" + type: "BashTask" + params: + command: "ls -l" + dependencies: ["start"] + + - task_id: "invalid_cmd_1" + type: "BashTask" + params: + command: "wrongcmd123" + dependencies: ["start"] + + - task_id: "invalid_cmd_2" + type: "BashTask" + params: + command: "another_bad_command" + dependencies: ["valid_ls"] + + - task_id: "python_recover" + type: "PythonTask" + params: + script: | + print("Recovery executed despite failures.") + dependencies: ["invalid_cmd_1", "invalid_cmd_2"] + + - task_id: "end" + type: "PrintTask" + params: + message: "Storm complete!" + dependencies: ["python_recover"] diff --git a/examples/2_New_examples/2.3.3.bash_chain_long_pipeline.yaml b/examples/2_New_examples/2.3.3.bash_chain_long_pipeline.yaml new file mode 100644 index 0000000..3090fdc --- /dev/null +++ b/examples/2_New_examples/2.3.3.bash_chain_long_pipeline.yaml @@ -0,0 +1,73 @@ + +abstract: | + A long sequential pipeline of 12 tasks mixing Print, Bash, and Python. This + stresses long‑chain execution, logging, and scheduling. + +dag: + name: "bash_chain_long_pipeline" + tasks: + - task_id: "t1" + type: "PrintTask" + params: { message: "Start long pipeline" } + dependencies: [] + + - task_id: "t2" + type: "BashTask" + params: { command: "ls -l" } + dependencies: ["t1"] + + - task_id: "t3" + type: "PythonTask" + params: + script: | + print("Step 3 complete") + dependencies: ["t2"] + + - task_id: "t4" + type: "BashTask" + params: { command: "sleep 1 && echo 'Pause done'" } + dependencies: ["t3"] + + - task_id: "t5" + type: "PrintTask" + params: { message: "Halfway there..." } + dependencies: ["t4"] + + - task_id: "t6" + type: "BashTask" + params: { command: "echo 'Running step 6'" } + dependencies: ["t5"] + + - task_id: "t7" + type: "PythonTask" + params: + script: | + print("Step 7: Python OK") + dependencies: ["t6"] + + - task_id: "t8" + type: "BashTask" + params: { command: "echo 'Step 8'" } + dependencies: ["t7"] + + - task_id: "t9" + type: "BashTask" + params: { command: "sleep 1" } + dependencies: ["t8"] + + - task_id: "t10" + type: "PythonTask" + params: + script: | + print("Almost done...") + dependencies: ["t9"] + + - task_id: "t11" + type: "PrintTask" + params: { message: "Final task running..." } + dependencies: ["t10"] + + - task_id: "t12" + type: "PrintTask" + params: { message: "Pipeline complete!" } + dependencies: ["t11"] diff --git a/examples/2_New_examples/3_wait_and_retry.yaml b/examples/2_New_examples/3.1.1.Wait_and_retry.yaml similarity index 100% rename from examples/2_New_examples/3_wait_and_retry.yaml rename to examples/2_New_examples/3.1.1.Wait_and_retry.yaml diff --git a/examples/2_New_examples/3.2.1.wait_and_retry_with_logging.yaml b/examples/2_New_examples/3.2.1.wait_and_retry_with_logging.yaml new file mode 100644 index 0000000..e2cc5cb --- /dev/null +++ b/examples/2_New_examples/3.2.1.wait_and_retry_with_logging.yaml @@ -0,0 +1,46 @@ +abstract: | + This DAG extends the basic retry logic example by adding initialization, multiple unstable Python tasks with different retry behaviors, and a final confirmation step. It is designed to test sequential retry execution, delayed retries, and detailed logging across multiple PythonTask failures. It helps validate orchestrator stability when handling repeated failures and timed retry cycles. + +dag: + name: "wait_and_retry_with_logging" + tasks: + + # Messaggio iniziale + - task_id: "start" + type: "PrintTask" + params: + message: "Starting retry pipeline..." + dependencies: [] + + # Primo task instabile (fail random 50%) + - task_id: "unstable_task_1" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < 0.5: + sys.exit(1) + print("Task 1 succeeded!") + retries: 3 + retry_delay: 2 + dependencies: ["start"] + + # Secondo task instabile (fail random 30%) + - task_id: "unstable_task_2" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < 0.3: + sys.exit(1) + print("Task 2 succeeded!") + retries: 2 + retry_delay: 1 + dependencies: ["unstable_task_1"] + + # Task finale + - task_id: "final_message" + type: "PrintTask" + params: + message: "All retry tasks completed!" + dependencies: ["unstable_task_2"] diff --git a/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml b/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml new file mode 100644 index 0000000..dc6d3fe --- /dev/null +++ b/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml @@ -0,0 +1,37 @@ +abstract: | + This DAG mixes BashTask and PythonTask retry logic in a linear pipeline. The first step is an unstable Bash command that may fail randomly and retry several times; the second step is an unstable PythonTask with its own independent retry policy. This structure tests mixed executor retry handling, error propagation, and consistent log reporting during cross-executor failure scenarios. + +dag: + name: "wait_and_retry_bash_python_mix" + tasks: + + - task_id: "unstable_bash" + type: "BashTask" + params: + command: | + if [ $((RANDOM % 3)) -eq 0 ]; then + echo "Failing Bash task"; exit 1; + else + echo "Bash succeeded!"; + fi + retries: 4 + retry_delay: 1 + dependencies: [] + + - task_id: "unstable_python" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < 0.4: + sys.exit("Python failed") + print("Python succeeded") + retries: 2 + retry_delay: 2 + dependencies: ["unstable_bash"] + + - task_id: "done" + type: "PrintTask" + params: + message: "Mixed retry pipeline finished." + dependencies: ["unstable_python"] diff --git a/examples/2_New_examples/3.2.3.wait_and_retry_branching.yaml b/examples/2_New_examples/3.2.3.wait_and_retry_branching.yaml new file mode 100644 index 0000000..921416d --- /dev/null +++ b/examples/2_New_examples/3.2.3.wait_and_retry_branching.yaml @@ -0,0 +1,50 @@ +abstract: | + This DAG introduces branching into retry logic. A single unstable root PythonTask fans out into two parallel branches—one Bash, one Python—each with retry settings of their own. Both branches must succeed before reaching the final task. This configuration tests concurrency, retry independence across branches, and orchestrator correctness in parallel retry-based workflows. + +dag: + name: "wait_and_retry_branching" + tasks: + + # Primo task instabile (radice) + - task_id: "unstable_root" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < 0.5: + sys.exit(1) + print("Root succeeded!") + retries: 3 + retry_delay: 2 + dependencies: [] + + # Ramo 1 + - task_id: "branch_a" + type: "BashTask" + params: + command: | + if [ $((RANDOM % 2)) -eq 0 ]; then exit 1; fi + echo "Branch A OK" + retries: 2 + retry_delay: 1 + dependencies: ["unstable_root"] + + # Ramo 2 + - task_id: "branch_b" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < 0.3: + sys.exit(1) + print("Branch B OK") + retries: 3 + retry_delay: 1 + dependencies: ["unstable_root"] + + # Fusione + - task_id: "done" + type: "PrintTask" + params: + message: "Branching retry pipeline complete." + dependencies: ["branch_a", "branch_b"] diff --git a/examples/2_New_examples/3.3.1.wait_and_retry_heavy_parallel.yaml b/examples/2_New_examples/3.3.1.wait_and_retry_heavy_parallel.yaml new file mode 100644 index 0000000..8344729 --- /dev/null +++ b/examples/2_New_examples/3.3.1.wait_and_retry_heavy_parallel.yaml @@ -0,0 +1,62 @@ +abstract: | + This DAG is a heavy parallel retry stress test. After an initial PrintTask, four unstable tasks (a mix of Bash and Python) run in parallel, each with its own retry count and delay. This structure pushes the orchestrator’s scheduler to handle simultaneous failures, retries, and logging from multiple executors at once. It validates robustness under parallel retry load. + +dag: + name: "wait_and_retry_heavy_parallel" + tasks: + + - task_id: "start" + type: "PrintTask" + params: + message: "Starting heavy parallel retry pipeline..." + dependencies: [] + + # Quattro task instabili in parallelo + - task_id: "unstable_1" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .5: sys.exit(1) + print("Unstable 1 OK") + retries: 4 + retry_delay: 1 + dependencies: ["start"] + + - task_id: "unstable_2" + type: "BashTask" + params: + command: | + if [ $((RANDOM % 2)) -eq 0 ]; then exit 1; fi + echo "Unstable 2 OK" + retries: 3 + retry_delay: 2 + dependencies: ["start"] + + - task_id: "unstable_3" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .7: sys.exit(1) + print("Unstable 3 OK") + retries: 5 + retry_delay: 1 + dependencies: ["start"] + + - task_id: "unstable_4" + type: "BashTask" + params: + command: | + if [ $((RANDOM % 3)) -eq 0 ]; then exit 1; fi + echo "Unstable 4 OK" + retries: 2 + retry_delay: 3 + dependencies: ["start"] + + # Merging finale + - task_id: "end" + type: "PrintTask" + params: + message: "All unstable tasks have been processed." + dependencies: ["unstable_1", "unstable_2", "unstable_3", "unstable_4"] diff --git a/examples/2_New_examples/3.3.2.wait_and_retry_cascade_failures.yaml b/examples/2_New_examples/3.3.2.wait_and_retry_cascade_failures.yaml new file mode 100644 index 0000000..73ae5a2 --- /dev/null +++ b/examples/2_New_examples/3.3.2.wait_and_retry_cascade_failures.yaml @@ -0,0 +1,54 @@ +abstract: | + This DAG simulates a failure-prone cascading pipeline. Each task in the chain may fail randomly and must retry multiple times, causing accumulated delays and chained retry behavior. This test is ideal for verifying long-chain retry stability, timing correctness, and orchestrator behavior when multiple dependent tasks experience intermittent failures. + +dag: + name: "wait_and_retry_cascade_failures" + tasks: + + - task_id: "t1" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .4: sys.exit(1) + print("t1 OK") + retries: 2 + retry_delay: 1 + dependencies: [] + + - task_id: "t2" + type: "BashTask" + params: + command: | + if [ $((RANDOM % 4)) -eq 0 ]; then exit 1; fi + echo "t2 OK" + retries: 3 + retry_delay: 2 + dependencies: ["t1"] + + - task_id: "t3" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .5: sys.exit(1) + print("t3 OK") + retries: 2 + retry_delay: 1 + dependencies: ["t2"] + + - task_id: "t4" + type: "BashTask" + params: + command: | + if [ $((RANDOM % 3)) -eq 0 ]; then exit 1; fi + echo "t4 OK" + retries: 4 + retry_delay: 1 + dependencies: ["t3"] + + - task_id: "done" + type: "PrintTask" + params: + message: "Cascade pipeline completed." + dependencies: ["t4"] diff --git a/examples/2_New_examples/3.3.3.wait_and_retry_spiderweb.yaml b/examples/2_New_examples/3.3.3.wait_and_retry_spiderweb.yaml new file mode 100644 index 0000000..3c03785 --- /dev/null +++ b/examples/2_New_examples/3.3.3.wait_and_retry_spiderweb.yaml @@ -0,0 +1,104 @@ +abstract: | + This DAG models a retry-heavy “spiderweb” structure: a root unstable PythonTask branches into three two-level sub-branches. After the branches complete, the DAG merges them into a final unstable PythonTask with its own retry policy. The multi-branch, multi-tier layout stresses retry propagation, concurrency handling, and orchestrator performance under complex retry graphs. + +dag: + name: "wait_and_retry_spiderweb" + tasks: + + # ROOT + - task_id: "root" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .5: sys.exit(1) + print("Root OK") + retries: 3 + retry_delay: 2 + dependencies: [] + + # BRANCH A + - task_id: "a1" + type: "BashTask" + params: + command: "if [ $((RANDOM % 2)) -eq 0 ]; then exit 1; fi; echo A1" + retries: 2 + retry_delay: 1 + dependencies: ["root"] + + - task_id: "a2" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .4: sys.exit(1) + print("A2 OK") + retries: 3 + retry_delay: 1 + dependencies: ["a1"] + + # BRANCH B + - task_id: "b1" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .6: sys.exit(1) + print("B1 OK") + retries: 4 + retry_delay: 1 + dependencies: ["root"] + + - task_id: "b2" + type: "BashTask" + params: + command: "if [ $((RANDOM % 3)) -eq 0 ]; then exit 1; fi; echo B2" + retries: 2 + retry_delay: 2 + dependencies: ["b1"] + + # BRANCH C + - task_id: "c1" + type: "BashTask" + params: + command: "if [ $((RANDOM % 2)) -eq 0 ]; then exit 1; fi; echo C1" + retries: 3 + retry_delay: 1 + dependencies: ["root"] + + - task_id: "c2" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .5: sys.exit(1) + print("C2 OK") + retries: 3 + retry_delay: 1 + dependencies: ["c1"] + + # MERGE + - task_id: "merge" + type: "PrintTask" + params: + message: "Branches merged successfully" + dependencies: ["a2", "b2", "c2"] + + # FINAL UNSTABLE TASK + - task_id: "final_unstable" + type: "PythonTask" + params: + script: | + import random, sys + if random.random() < .4: sys.exit(1) + print("Final task OK") + retries: 3 + retry_delay: 2 + dependencies: ["merge"] + + # END + - task_id: "done" + type: "PrintTask" + params: + message: "Spiderweb retry pipeline fully completed." + dependencies: ["final_unstable"] diff --git a/examples/2_New_examples/4_conditional_branching.yaml b/examples/2_New_examples/4.1.1.Conditional_branching.yaml similarity index 100% rename from examples/2_New_examples/4_conditional_branching.yaml rename to examples/2_New_examples/4.1.1.Conditional_branching.yaml diff --git a/examples/2_New_examples/4.2.1.conditional_branching_extended.yaml b/examples/2_New_examples/4.2.1.conditional_branching_extended.yaml new file mode 100644 index 0000000..99921c7 --- /dev/null +++ b/examples/2_New_examples/4.2.1.conditional_branching_extended.yaml @@ -0,0 +1,65 @@ +abstract: | + This DAG extends the original conditional branching example by introducing a + preliminary initialization step, a Python-based condition evaluator, and two + parallel branches that each perform a different kind of system check. Both + branches then converge into a final PrintTask summarizing the pipeline. + + Compared to the original version, this one: + - Adds a starting PrintTask for clarity. + - Splits the execution into two meaningful branches (file listing and process listing). + - Maintains the concept of conditional branching, although both branches run regardless + of the evaluated condition, making it ideal for testing parallel execution behavior. + +dag: + name: "conditional_branching_extended" + + tasks: + # --------------------------------------------------------- + # 1) Initialization step: ensures the DAG begins cleanly. + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting extended conditional branching DAG..." + dependencies: [] + + # --------------------------------------------------------- + # 2) A PythonTask evaluating a random condition. + # Both branches in this example will run regardless, + # but this helps test logic, logging, and Python output. + # --------------------------------------------------------- + - task_id: "evaluate_condition" + type: "PythonTask" + params: + script: | + import random + condition = "A" if random.random() > 0.5 else "B" + print(f"Condition evaluated: {condition}") + dependencies: ["init"] + + # --------------------------------------------------------- + # 3) Branch A: performs a file-system related check. + # --------------------------------------------------------- + - task_id: "branch_a_action" + type: "BashTask" + params: + command: "echo 'Branch A: Listing files'; ls -1 | head -n 5" + dependencies: ["evaluate_condition"] + + # --------------------------------------------------------- + # 4) Branch B: prints some process info. + # --------------------------------------------------------- + - task_id: "branch_b_action" + type: "BashTask" + params: + command: "echo 'Branch B: Listing processes'; ps aux | head -n 5" + dependencies: ["evaluate_condition"] + + # --------------------------------------------------------- + # 5) Final merge: executed only when both branches finish. + # --------------------------------------------------------- + - task_id: "final_message" + type: "PrintTask" + params: + message: "Extended conditional branching DAG finished!" + dependencies: ["branch_a_action", "branch_b_action"] diff --git a/examples/2_New_examples/4.2.2.conditional_branching_with_bash.yaml b/examples/2_New_examples/4.2.2.conditional_branching_with_bash.yaml new file mode 100644 index 0000000..19b3fb5 --- /dev/null +++ b/examples/2_New_examples/4.2.2.conditional_branching_with_bash.yaml @@ -0,0 +1,72 @@ +abstract: | + This DAG expands the original conditional branching concept by incorporating + a Bash-based condition evaluation. Instead of using Python to decide the + branching factor, a BashTask randomly selects a branch and prints the outcome. + + The pipeline includes: + - A PrintTask initializer + - A BashTask that computes a pseudo-random branch selection + - Two parallel branches: one focused on disk usage, the other on directory structure + - A final merge task that acknowledges completion of both branches + + Even though the BashTask “selects” a branch, both branches run in parallel. + This design stresses the orchestrator’s scheduler and ensures the system + handles branching regardless of executor type. + +dag: + name: "conditional_branching_with_bash" + + tasks: + + # --------------------------------------------------------- + # 1) Initialization: simple starting message + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting conditional branching (Bash-based)..." + dependencies: [] + + # --------------------------------------------------------- + # 2) Condition evaluation done inside Bash: + # RANDOM % 2 → 0 or 1 determines “branch A” or “branch B” + # Here, both branches will still run independently, but it + # tests Bash logging, randomness, and execution flow. + # --------------------------------------------------------- + - task_id: "bash_condition" + type: "BashTask" + params: + command: | + if [ $((RANDOM % 2)) -eq 0 ]; then + echo "Condition result: A"; + else + echo "Condition result: B"; + fi + dependencies: ["init"] + + # --------------------------------------------------------- + # 3) Branch A: performs disk-related operations + # --------------------------------------------------------- + - task_id: "branch_a_disk_usage" + type: "BashTask" + params: + command: "echo 'Branch A: Disk usage'; df -h | head -n 5" + dependencies: ["bash_condition"] + + # --------------------------------------------------------- + # 4) Branch B: explores directory structure + # --------------------------------------------------------- + - task_id: "branch_b_list_dirs" + type: "BashTask" + params: + command: "echo 'Branch B: Directory structure'; ls -d */ 2>/dev/null || echo 'No directories'" + dependencies: ["bash_condition"] + + # --------------------------------------------------------- + # 5) Merge: both branches must complete before continuing + # --------------------------------------------------------- + - task_id: "final_message" + type: "PrintTask" + params: + message: "Conditional branching (Bash-based) completed!" + dependencies: ["branch_a_disk_usage", "branch_b_list_dirs"] diff --git a/examples/2_New_examples/4.2.3.conditional_branching_parallel_python.yaml b/examples/2_New_examples/4.2.3.conditional_branching_parallel_python.yaml new file mode 100644 index 0000000..b7563a1 --- /dev/null +++ b/examples/2_New_examples/4.2.3.conditional_branching_parallel_python.yaml @@ -0,0 +1,79 @@ +abstract: | + This DAG extends the conditional branching concept by introducing a Python-driven + evaluation stage followed by *two parallel Python processing branches*. This + pushes the orchestrator’s parallel execution and logging logic without adding + excessive structural complexity. + + The pipeline performs: + - Initialization via PrintTask + - A PythonTask that evaluates a random condition (A or B) + - Two parallel Python tasks: + * branch_a_process — performs numeric computations + * branch_b_process — performs string transformations + - A final PrintTask executed only after both branches complete + + Unlike the original conditional branching DAG, here the branches perform + meaningful Python processing, making it a more realistic mid-level test + of Python executor behavior, concurrency, and DAG synchronization. + +dag: + name: "conditional_branching_parallel_python" + + tasks: + + # --------------------------------------------------------- + # 1) Initialization — just a clean start. + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting conditional branching (parallel Python)..." + dependencies: [] + + # --------------------------------------------------------- + # 2) Python-based condition evaluation. + # The condition is not used to activate/deactivate branches, + # but it is logged to simulate real decision-making behavior. + # --------------------------------------------------------- + - task_id: "evaluate_condition" + type: "PythonTask" + params: + script: | + import random + condition = "A" if random.random() > 0.5 else "B" + print(f"Condition evaluated: {condition}") + dependencies: ["init"] + + # --------------------------------------------------------- + # 3) Branch A — numeric computation. + # --------------------------------------------------------- + - task_id: "branch_a_process" + type: "PythonTask" + params: + script: | + import math + nums = [1, 2, 3, 4, 5] + squares = [n*n for n in nums] + print("Branch A squares:", squares) + dependencies: ["evaluate_condition"] + + # --------------------------------------------------------- + # 4) Branch B — simple text processing. + # --------------------------------------------------------- + - task_id: "branch_b_process" + type: "PythonTask" + params: + script: | + words = ["alpha", "beta", "gamma"] + upper = [w.upper() for w in words] + print("Branch B uppercase:", upper) + dependencies: ["evaluate_condition"] + + # --------------------------------------------------------- + # 5) Final merge. + # --------------------------------------------------------- + - task_id: "final_message" + type: "PrintTask" + params: + message: "Parallel Python branching completed successfully!" + dependencies: ["branch_a_process", "branch_b_process"] diff --git a/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml b/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml new file mode 100644 index 0000000..364165a --- /dev/null +++ b/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml @@ -0,0 +1,112 @@ +abstract: | + This difficult-level DAG expands the original conditional branching structure + into a *triple-branch execution tree*. After an initialization and a Python-driven + condition evaluation stage, the DAG fans out into three independent branches, + each with two processing layers mixing BashTask and PythonTask. + + The result is a large fan-out → fan-in structure intended to stress concurrency, + scheduling, logging, and orchestrator dependency resolution. All three branches + must complete before the pipeline proceeds to the final PrintTask. + +dag: + name: "conditional_branching_triple_tree" + + tasks: + + # --------------------------------------------------------- + # 1) Initialization message. + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting triple-branch conditional pipeline..." + dependencies: [] + + # --------------------------------------------------------- + # 2) Python-based condition evaluation — mostly informational. + # --------------------------------------------------------- + - task_id: "evaluate_condition" + type: "PythonTask" + params: + script: | + import random + condition = random.choice(["A", "B", "C"]) + print(f"Condition evaluated: {condition}") + dependencies: ["init"] + + # --------------------------------------------------------- + # BRANCH 1 — FILE SYSTEM PROCESSING + # --------------------------------------------------------- + + - task_id: "branch1_step1" + type: "BashTask" + params: + command: "echo 'Branch 1: Listing files'; ls -1 | head -n 10" + dependencies: ["evaluate_condition"] + + - task_id: "branch1_step2" + type: "PythonTask" + params: + script: | + import os + print('Branch 1 file count:', len(os.listdir('.'))) + dependencies: ["branch1_step1"] + + # --------------------------------------------------------- + # BRANCH 2 — PROCESS INSPECTION + # --------------------------------------------------------- + + - task_id: "branch2_step1" + type: "BashTask" + params: + command: "echo 'Branch 2: Checking processes'; ps aux | head -n 5" + dependencies: ["evaluate_condition"] + + - task_id: "branch2_step2" + type: "PythonTask" + params: + script: | + print('Branch 2: Process summary collected.') + dependencies: ["branch2_step1"] + + # --------------------------------------------------------- + # BRANCH 3 — TEXT PROCESSING + # --------------------------------------------------------- + + - task_id: "branch3_step1" + type: "PythonTask" + params: + script: | + words = ['lorem', 'ipsum', 'dolor', 'sit', 'amet'] + print('Branch 3 words:', words) + dependencies: ["evaluate_condition"] + + - task_id: "branch3_step2" + type: "PythonTask" + params: + script: | + words = ['lorem', 'ipsum', 'dolor', 'sit', 'amet'] + uppercase = [w.upper() for w in words] + print('Branch 3 uppercase:', uppercase) + dependencies: ["branch3_step1"] + + # --------------------------------------------------------- + # MERGE: require ALL three branches to finish + # --------------------------------------------------------- + - task_id: "merge_results" + type: "PrintTask" + params: + message: "All three branches completed. Consolidating results..." + dependencies: + - "branch1_step2" + - "branch2_step2" + - "branch3_step2" + + # --------------------------------------------------------- + # FINAL TASK + # --------------------------------------------------------- + - task_id: "final" + type: "PrintTask" + params: + message: "Triple-branch conditional pipeline completed successfully!" + dependencies: ["merge_results"] diff --git a/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml b/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml new file mode 100644 index 0000000..99793fe --- /dev/null +++ b/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml @@ -0,0 +1,136 @@ +abstract: | + This difficult-level DAG models a "matrix" of conditional-style branching. + From a single Python-based root evaluation, the DAG fans out into four + independent branches. Each branch contains a 3-step mini-pipeline mixing + PythonTask, BashTask, and PrintTask. + + All four branches run in parallel after the root task and must complete + before the final PrintTask is executed. This structure is ideal for: + - Stress-testing fan-out and fan-in synchronization + - Exercising both Bash and Python executors under concurrent load + - Verifying that logging and task ordering remain understandable in + complex, parallel DAGs. + +dag: + name: "conditional_branching_matrix" + + tasks: + + # --------------------------------------------------------- + # 1) ROOT: condition/value evaluation (informational only). + # --------------------------------------------------------- + - task_id: "root" + type: "PythonTask" + params: + script: | + import random + value = random.random() + print(f"Root evaluated value: {value}") + dependencies: [] + + # --------------------------------------------------------- + # BRANCH 1 (b1_1 → b1_2 → b1_3) + # Python → Bash → Print + # --------------------------------------------------------- + - task_id: "b1_1" + type: "PythonTask" + params: + script: | + nums = [1, 2, 3] + print("Branch 1 - step 1, nums:", nums) + dependencies: ["root"] + + - task_id: "b1_2" + type: "BashTask" + params: + command: "echo 'Branch 1 - step 2, simple Bash step'" + dependencies: ["b1_1"] + + - task_id: "b1_3" + type: "PrintTask" + params: + message: "Branch 1 - step 3 completed." + dependencies: ["b1_2"] + + # --------------------------------------------------------- + # BRANCH 2 (b2_1 → b2_2 → b2_3) + # Bash → Python → Print + # --------------------------------------------------------- + - task_id: "b2_1" + type: "BashTask" + params: + command: "echo 'Branch 2 - step 1, listing current dir'; ls -1 | head -n 5" + dependencies: ["root"] + + - task_id: "b2_2" + type: "PythonTask" + params: + script: | + print("Branch 2 - step 2, post-processing Bash output logically.") + dependencies: ["b2_1"] + + - task_id: "b2_3" + type: "PrintTask" + params: + message: "Branch 2 - step 3 completed." + dependencies: ["b2_2"] + + # --------------------------------------------------------- + # BRANCH 3 (b3_1 → b3_2 → b3_3) + # Python → Bash → Print + # --------------------------------------------------------- + - task_id: "b3_1" + type: "PythonTask" + params: + script: | + words = ["alpha", "beta", "gamma"] + print("Branch 3 - step 1, words:", words) + dependencies: ["root"] + + - task_id: "b3_2" + type: "BashTask" + params: + command: "echo 'Branch 3 - step 2, simple echo from Bash'" + dependencies: ["b3_1"] + + - task_id: "b3_3" + type: "PrintTask" + params: + message: "Branch 3 - step 3 completed." + dependencies: ["b3_2"] + + # --------------------------------------------------------- + # BRANCH 4 (b4_1 → b4_2 → b4_3) + # Bash → Python → Print + # --------------------------------------------------------- + - task_id: "b4_1" + type: "BashTask" + params: + command: "echo 'Branch 4 - step 1, checking date'; date" + dependencies: ["root"] + + - task_id: "b4_2" + type: "PythonTask" + params: + script: | + print("Branch 4 - step 2, Python confirming time was printed above.") + dependencies: ["b4_1"] + + - task_id: "b4_3" + type: "PrintTask" + params: + message: "Branch 4 - step 3 completed." + dependencies: ["b4_2"] + + # --------------------------------------------------------- + # FINAL MERGE: waits for all branch end-nodes. + # --------------------------------------------------------- + - task_id: "final" + type: "PrintTask" + params: + message: "Matrix DAG completed: all four branches are done." + dependencies: + - "b1_3" + - "b2_3" + - "b3_3" + - "b4_3" diff --git a/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml b/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml new file mode 100644 index 0000000..9b56fb0 --- /dev/null +++ b/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml @@ -0,0 +1,157 @@ +abstract: | + This DAG represents a high-complexity “hypergraph” branching structure. + It expands far beyond traditional fan-out/fan-in patterns by: + + - Starting from a Python-based condition evaluation + - Creating three primary branches + - Allowing the branches to interconnect at the second tier + - Generating a tertiary fan-out from the partial merge + - Finally collapsing everything into a unified final task + + This structure is ideal for stress-testing: + - Multi-level synchronization + - Complex dependency graphs + - Mixed Bash/Python execution patterns + - Large-scale parallelism and merge logic + + All tasks run regardless of the evaluated condition (the condition is logged only). + This makes the DAG’s behavior deterministic in execution but non-deterministic in logging. + +dag: + name: "conditional_branching_hypergraph" + + tasks: + + # --------------------------------------------------------- + # 1) ROOT INITIALIZATION + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting hypergraph conditional pipeline..." + dependencies: [] + + # --------------------------------------------------------- + # 2) CONDITION EVALUATION + # --------------------------------------------------------- + - task_id: "evaluate_condition" + type: "PythonTask" + params: + script: | + import random + cond = random.choice(["X", "Y", "Z"]) + print(f"Condition evaluated: {cond}") + dependencies: ["init"] + + # ========================================================= + # PRIMARY FAN-OUT (3 BRANCHES) + # ========================================================= + + # -------- BRANCH A -------- + - task_id: "A1" + type: "PythonTask" + params: + script: | + print("Branch A - Step 1") + dependencies: ["evaluate_condition"] + + - task_id: "A2" + type: "BashTask" + params: + command: "echo 'Branch A - Step 2'" + dependencies: ["A1"] + + # -------- BRANCH B -------- + - task_id: "B1" + type: "BashTask" + params: + command: "echo 'Branch B - Step 1'" + dependencies: ["evaluate_condition"] + + - task_id: "B2" + type: "PythonTask" + params: + script: | + print("Branch B - Step 2") + dependencies: ["B1"] + + # -------- BRANCH C -------- + - task_id: "C1" + type: "PythonTask" + params: + script: | + print("Branch C - Step 1") + dependencies: ["evaluate_condition"] + + - task_id: "C2" + type: "BashTask" + params: + command: "echo 'Branch C - Step 2'" + dependencies: ["C1"] + + # ========================================================= + # SECONDARY INTERCONNECTED LAYER + # ========================================================= + # A2 joins with B1 and C2 (cross connections) + + - task_id: "merge_AB" + type: "PrintTask" + params: + message: "Merging output from A2 and B2" + dependencies: ["A2", "B2"] + + - task_id: "merge_BC" + type: "PrintTask" + params: + message: "Merging output from B2 and C2" + dependencies: ["B2", "C2"] + + - task_id: "merge_CA" + type: "PrintTask" + params: + message: "Merging output from C2 and A2" + dependencies: ["C2", "A2"] + + # ========================================================= + # TERTIARY FAN-OUT FROM PARTIAL MERGES + # ========================================================= + + - task_id: "T1" + type: "PythonTask" + params: + script: | + print("T1: Post-merge computation A+B") + dependencies: ["merge_AB"] + + - task_id: "T2" + type: "PythonTask" + params: + script: | + print("T2: Post-merge computation B+C") + dependencies: ["merge_BC"] + + - task_id: "T3" + type: "PythonTask" + params: + script: | + print("T3: Post-merge computation C+A") + dependencies: ["merge_CA"] + + # ========================================================= + # FINAL MERGE (ALL TERTIARY TASKS) + # ========================================================= + + - task_id: "final_merge" + type: "PrintTask" + params: + message: "All tertiary computations completed. Final consolidation..." + dependencies: ["T1", "T2", "T3"] + + # --------------------------------------------------------- + # 6) FINAL OUTPUT TASK + # --------------------------------------------------------- + - task_id: "final" + type: "PrintTask" + params: + message: "Hypergraph branching pipeline completed!" + dependencies: ["final_merge"] diff --git a/examples/2_New_examples/5_cron_heartbeat.yaml b/examples/2_New_examples/5.1.1.Cron_heartbeat.yaml similarity index 100% rename from examples/2_New_examples/5_cron_heartbeat.yaml rename to examples/2_New_examples/5.1.1.Cron_heartbeat.yaml diff --git a/examples/2_New_examples/5.2.1.cron_heartbeat_extended.yaml b/examples/2_New_examples/5.2.1.cron_heartbeat_extended.yaml new file mode 100644 index 0000000..841cfee --- /dev/null +++ b/examples/2_New_examples/5.2.1.cron_heartbeat_extended.yaml @@ -0,0 +1,28 @@ +abstract: | + This DAG extends the basic cron heartbeat by adding additional informational tasks executed at every scheduled interval. The pipeline logs a heartbeat message, performs lightweight environment checks, and prints a summary. It is designed to test recurring execution, repeated logging consistency, and stable behavior in cron-driven workflows. + +dag: + name: "cron_heartbeat_extended" + cron_schedule: "*/10 * * * *" # Ogni 10 minuti + + tasks: + # 1) Messaggio iniziale, così vedi che il job è partito + - task_id: "heartbeat_start" + type: "PrintTask" + params: + message: "Heartbeat: Maestro cron job triggered." + dependencies: [] + + # 2) Piccolo check di sistema: stampa data/ora e directory corrente + - task_id: "bash_info" + type: "BashTask" + params: + command: "echo 'Time:' $(date) && echo 'PWD:' $(pwd)" + dependencies: ["heartbeat_start"] + + # 3) Messaggio finale di conferma + - task_id: "heartbeat_end" + type: "PrintTask" + params: + message: "Heartbeat extended: basic info logged." + dependencies: ["bash_info"] diff --git a/examples/2_New_examples/5.2.2.cron_heartbeat_with_status.yaml b/examples/2_New_examples/5.2.2.cron_heartbeat_with_status.yaml new file mode 100644 index 0000000..f3ad214 --- /dev/null +++ b/examples/2_New_examples/5.2.2.cron_heartbeat_with_status.yaml @@ -0,0 +1,31 @@ +abstract: | + This DAG enhances the heartbeat concept by logging system status at every cron tick. It performs quick Bash-based system snapshots (disk, memory, processes) alongside the heartbeat message. This setup tests recurring system diagnostics, multi-step scheduled pipelines, and orchestrator stability under repeated periodic commands. + +dag: + name: "cron_heartbeat_with_status" + cron_schedule: "*/10 * * * *" # Ogni 10 minuti + + tasks: + # 1) Valuta uno stato fittizio del sistema + - task_id: "compute_status" + type: "PythonTask" + params: + script: | + import random + status = random.choice(["OK", "DEGRADED", "ERROR"]) + print(f"Computed heartbeat status: {status}") + dependencies: [] + + # 2) Stampa un messaggio leggibile a console + - task_id: "print_status" + type: "PrintTask" + params: + message: "Heartbeat status computed and printed above." + dependencies: ["compute_status"] + + # 3) Scrive un log minimale su file (in /tmp per restare innocui) + - task_id: "log_to_file" + type: "BashTask" + params: + command: "echo \"$(date) - Heartbeat executed\" >> /tmp/maestro_heartbeat.log" + dependencies: ["print_status"] diff --git a/examples/2_New_examples/5.2.3.cron_heartbeat_light_checks.yaml b/examples/2_New_examples/5.2.3.cron_heartbeat_light_checks.yaml new file mode 100644 index 0000000..a1c83b4 --- /dev/null +++ b/examples/2_New_examples/5.2.3.cron_heartbeat_light_checks.yaml @@ -0,0 +1,43 @@ +abstract: | + This DAG performs a lightweight system check on each cron heartbeat. It mixes PrintTask, BashTask, and PythonTask to gather minimal environment data without heavy overhead. It is designed to verify low-impact recurring tasks and ensure the orchestrator remains responsive and stable during frequent scheduled triggers. + +dag: + name: "cron_heartbeat_light_checks" + cron_schedule: "*/10 * * * *" # Ogni 10 minuti + + tasks: + # ROOT: lancia il giro di check + - task_id: "heartbeat_root" + type: "PrintTask" + params: + message: "Heartbeat: starting light system checks..." + dependencies: [] + + # Check pseudo-CPU (simulato) + - task_id: "check_cpu" + type: "PythonTask" + params: + script: | + print("CPU check: (simulated) OK") + dependencies: ["heartbeat_root"] + + # Check disco (solo output informativo) + - task_id: "check_disk" + type: "BashTask" + params: + command: "df -h | head -n 5" + dependencies: ["heartbeat_root"] + + # Check processi (solo prime righe) + - task_id: "check_processes" + type: "BashTask" + params: + command: "ps aux | head -n 5" + dependencies: ["heartbeat_root"] + + # Merge: eseguito solo quando tutti i check sono completati + - task_id: "heartbeat_summary" + type: "PrintTask" + params: + message: "Heartbeat: all light checks completed." + dependencies: ["check_cpu", "check_disk", "check_processes"] diff --git a/examples/2_New_examples/5.3.1.cron_heartbeat_full_healthcheck.yaml b/examples/2_New_examples/5.3.1.cron_heartbeat_full_healthcheck.yaml new file mode 100644 index 0000000..31f0edf --- /dev/null +++ b/examples/2_New_examples/5.3.1.cron_heartbeat_full_healthcheck.yaml @@ -0,0 +1,58 @@ +abstract: | + This DAG represents a full periodic healthcheck pipeline executed at every cron interval. It fans out into multiple parallel branches performing disk, process, and Python runtime diagnostics, then merges them into a summary task. Designed for stress testing recurring parallel workloads, deep logging, and high-frequency health monitoring. + +dag: + name: "cron_heartbeat_full_healthcheck" + cron_schedule: "*/10 * * * *" # Ogni 10 minuti + + tasks: + # ROOT + - task_id: "heartbeat_start" + type: "PrintTask" + params: + message: "Heartbeat: starting full healthcheck..." + dependencies: [] + + # BRANCH CPU + - task_id: "check_cpu" + type: "PythonTask" + params: + script: | + print("CPU usage check (simulated).") + dependencies: ["heartbeat_start"] + + # BRANCH MEMORIA + - task_id: "check_memory" + type: "BashTask" + params: + command: "free -h || echo 'free command not available'" + dependencies: ["heartbeat_start"] + + # BRANCH DISCO + - task_id: "check_disk" + type: "BashTask" + params: + command: "df -h | head -n 10" + dependencies: ["heartbeat_start"] + + # BRANCH PROCESSI + - task_id: "check_processes" + type: "BashTask" + params: + command: "ps aux | head -n 10" + dependencies: ["heartbeat_start"] + + # AGGREGAZIONE LOGICA (qui potresti, in futuro, interpretare output reali) + - task_id: "aggregate_results" + type: "PythonTask" + params: + script: | + print("Aggregating healthcheck results (simulation). Overall: OK") + dependencies: ["check_cpu", "check_memory", "check_disk", "check_processes"] + + # MESSAGGIO FINALE + - task_id: "heartbeat_done" + type: "PrintTask" + params: + message: "Heartbeat full healthcheck completed." + dependencies: ["aggregate_results"] diff --git a/examples/2_New_examples/5.3.2.cron_heartbeat_with_warnings.yaml b/examples/2_New_examples/5.3.2.cron_heartbeat_with_warnings.yaml new file mode 100644 index 0000000..f16d83a --- /dev/null +++ b/examples/2_New_examples/5.3.2.cron_heartbeat_with_warnings.yaml @@ -0,0 +1,51 @@ +abstract: | + This DAG extends the heartbeat model with a warning-generation system. Each branch performs a diagnostic check and emits a simulated warning without failing the pipeline. It tests the orchestrator’s ability to handle scheduled tasks that produce structured warnings, maintain uptime, and avoid fail-fast conditions. + +dag: + name: "cron_heartbeat_with_warnings" + cron_schedule: "*/10 * * * *" # Ogni 10 minuti + + tasks: + # ROOT: decide un livello di "criticità" fittizio + - task_id: "evaluate_status" + type: "PythonTask" + params: + script: | + import random + status = random.choice(["OK", "WARNING"]) + print(f"Evaluated heartbeat status: {status}") + dependencies: [] + + # RAMO OK: controlli leggeri + - task_id: "ok_branch_info" + type: "BashTask" + params: + command: "echo 'OK branch: lightweight checks...'" + dependencies: ["evaluate_status"] + + - task_id: "ok_branch_print" + type: "PrintTask" + params: + message: "OK branch completed." + dependencies: ["ok_branch_info"] + + # RAMO WARNING: logga qualcosa in più + - task_id: "warning_branch_detail" + type: "PythonTask" + params: + script: | + print("WARNING branch: capturing extra diagnostic info (simulated).") + dependencies: ["evaluate_status"] + + - task_id: "warning_branch_log" + type: "BashTask" + params: + command: "echo \"$(date) - WARNING: simulated condition\" >> /tmp/maestro_heartbeat_warn.log" + dependencies: ["warning_branch_detail"] + + # MERGE: sempre eseguito dopo entrambi i rami + - task_id: "heartbeat_summary" + type: "PrintTask" + params: + message: "Heartbeat with warnings: all branches executed." + dependencies: ["ok_branch_print", "warning_branch_log"] diff --git a/examples/2_New_examples/5.3.3.cron_heartbeat_deep_pipeline.yaml b/examples/2_New_examples/5.3.3.cron_heartbeat_deep_pipeline.yaml new file mode 100644 index 0000000..d96922c --- /dev/null +++ b/examples/2_New_examples/5.3.3.cron_heartbeat_deep_pipeline.yaml @@ -0,0 +1,71 @@ +abstract: | + This DAG provides a deep, multi-step cron-based pipeline where each heartbeat triggers a long sequence of Bash and Python transformations. It tests orchestrator stability under long-running scheduled flows, accumulated sequential latency, and executor transitions across many steps. + +dag: + name: "cron_heartbeat_deep_pipeline" + cron_schedule: "*/10 * * * *" # Ogni 10 minuti + + tasks: + # ROOT + - task_id: "heartbeat_root" + type: "PrintTask" + params: + message: "Heartbeat: deep pipeline start." + dependencies: [] + + # ===== BRANCH SYSTEM ===== + - task_id: "sys_info" + type: "BashTask" + params: + command: "uname -a || echo 'uname not available'" + dependencies: ["heartbeat_root"] + + - task_id: "sys_python_analysis" + type: "PythonTask" + params: + script: | + print("System analysis (simulated).") + dependencies: ["sys_info"] + + # ===== BRANCH APPLICATION (simulato) ===== + - task_id: "app_status" + type: "PythonTask" + params: + script: | + print("App status: simulated check OK.") + dependencies: ["heartbeat_root"] + + - task_id: "app_logs_tail" + type: "BashTask" + params: + command: "echo 'Simulated application log tail...'" + dependencies: ["app_status"] + + - task_id: "app_python_summary" + type: "PythonTask" + params: + script: | + print("Application summary (simulated).") + dependencies: ["app_logs_tail"] + + # MERGE: quando entrambe le pipeline (system + app) sono complete + - task_id: "merge_health" + type: "PythonTask" + params: + script: | + print("Merging system and application health (simulated). Overall OK.") + dependencies: ["sys_python_analysis", "app_python_summary"] + + # LOG SU FILE + - task_id: "write_heartbeat_log" + type: "BashTask" + params: + command: "echo \"$(date) - Deep heartbeat completed\" >> /tmp/maestro_deep_heartbeat.log" + dependencies: ["merge_health"] + + # FINALE + - task_id: "heartbeat_done" + type: "PrintTask" + params: + message: "Deep heartbeat pipeline completed." + dependencies: ["write_heartbeat_log"] diff --git a/examples/2_New_examples/6_scheduled_greeting.yaml b/examples/2_New_examples/6.1.1.Scheduled_greeting.yaml similarity index 100% rename from examples/2_New_examples/6_scheduled_greeting.yaml rename to examples/2_New_examples/6.1.1.Scheduled_greeting.yaml diff --git a/examples/2_New_examples/6.2.1.scheduled_greeting_extended.yaml b/examples/2_New_examples/6.2.1.scheduled_greeting_extended.yaml new file mode 100644 index 0000000..e71a546 --- /dev/null +++ b/examples/2_New_examples/6.2.1.scheduled_greeting_extended.yaml @@ -0,0 +1,59 @@ +abstract: | + This DAG extends the basic scheduled_greeting example by adding multiple + steps that validate and log system information immediately after the DAG + triggers at its scheduled start_time. + + The pipeline: + - Begins with a PrintTask announcing execution + - Logs the system time using a BashTask + - Runs a PythonTask that performs a small diagnostic check + - Ends with a closing PrintTask + + This design tests: + - Scheduled one-shot execution + - Multi-step sequential flows after a scheduled trigger + - Mixed Print/Bash/Python execution following start_time activation + +dag: + name: "scheduled_greeting_extended" + start_time: "2025-10-23T09:00:00" + + tasks: + + # --------------------------------------------------------- + # 1) Initialization message triggered at scheduled time + # --------------------------------------------------------- + - task_id: "start_message" + type: "PrintTask" + params: + message: "Scheduled greeting pipeline started!" + dependencies: [] + + # --------------------------------------------------------- + # 2) Simple Bash command to show the actual system date/time + # --------------------------------------------------------- + - task_id: "show_time" + type: "BashTask" + params: + command: "echo 'Current system time: ' $(date)" + dependencies: ["start_message"] + + # --------------------------------------------------------- + # 3) Python diagnostic check + # --------------------------------------------------------- + - task_id: "python_check" + type: "PythonTask" + params: + script: | + import platform + print('Python check executed. Python version:', platform.python_version()) + dependencies: ["show_time"] + + # --------------------------------------------------------- + # 4) Final closing message + # --------------------------------------------------------- + - task_id: "finish" + type: "PrintTask" + params: + message: "Extended scheduled greeting pipeline completed." + dependencies: ["python_check"] diff --git a/examples/2_New_examples/6.2.2.scheduled_greeting_language_mix.yaml b/examples/2_New_examples/6.2.2.scheduled_greeting_language_mix.yaml new file mode 100644 index 0000000..31a72d3 --- /dev/null +++ b/examples/2_New_examples/6.2.2.scheduled_greeting_language_mix.yaml @@ -0,0 +1,83 @@ +abstract: | + This DAG expands the basic scheduled greeting example by introducing + a multilingual greeting flow executed immediately after the scheduled + start_time is reached. + + The pipeline performs: + - A starting PrintTask announcing the scheduled execution + - A PythonTask that selects a random language code ("EN", "IT", "ES") + - Two child PrintTasks, one for each greeting style + (in this version both are executed, regardless of the condition) + - A final PrintTask that confirms the end of the multilingual flow + + This DAG tests: + - Scheduled DAG triggering + - Parallel fan-out from PythonTask + - Execution of multiple PrintTasks representing different output behaviors + - Concurrency and synchronization before the final merge + +dag: + name: "scheduled_greeting_language_mix" + start_time: "2025-10-23T09:00:00" + + tasks: + + # --------------------------------------------------------- + # 1) Initial scheduled activation message + # --------------------------------------------------------- + - task_id: "start" + type: "PrintTask" + params: + message: "Scheduled multilingual greeting starting..." + dependencies: [] + + # --------------------------------------------------------- + # 2) Python selection of a pseudo-random language + # --------------------------------------------------------- + - task_id: "choose_language" + type: "PythonTask" + params: + script: | + import random + lang = random.choice(["EN", "IT", "ES"]) + print(f"Selected language: {lang}") + dependencies: ["start"] + + # --------------------------------------------------------- + # 3) Greeting branch EN + # --------------------------------------------------------- + - task_id: "greet_en" + type: "PrintTask" + params: + message: "Hello! Good morning!" + dependencies: ["choose_language"] + + # --------------------------------------------------------- + # 3b) Greeting branch IT + # --------------------------------------------------------- + - task_id: "greet_it" + type: "PrintTask" + params: + message: "Ciao! Buongiorno!" + dependencies: ["choose_language"] + + # --------------------------------------------------------- + # 3c) Greeting branch ES + # --------------------------------------------------------- + - task_id: "greet_es" + type: "PrintTask" + params: + message: "¡Hola! ¡Buenos días!" + dependencies: ["choose_language"] + + # --------------------------------------------------------- + # 4) Final merge ensuring all greetings completed + # --------------------------------------------------------- + - task_id: "finish" + type: "PrintTask" + params: + message: "Multilingual greeting pipeline completed!" + dependencies: + - "greet_en" + - "greet_it" + - "greet_es" diff --git a/examples/2_New_examples/6.2.3.scheduled_greeting_with_checks.yaml b/examples/2_New_examples/6.2.3.scheduled_greeting_with_checks.yaml new file mode 100644 index 0000000..54ce8a3 --- /dev/null +++ b/examples/2_New_examples/6.2.3.scheduled_greeting_with_checks.yaml @@ -0,0 +1,60 @@ +abstract: | + This DAG enhances the scheduled greeting concept by adding a small diagnostic + pipeline that runs immediately after the scheduled start_time. It mixes Python, + Bash, and Print tasks to validate directory state, create small data, and + log useful information before producing a final scheduled greeting message. + + The flow: + - Scheduled PrintTask announces execution + - BashTask checks the number of files in the working directory + - PythonTask performs a trivial mathematical check + - PrintTask emits a localized greeting + + This DAG tests: + - Sequential validation flow (Bash → Python → Print) + - Proper execution after scheduled activation + - Mixing of executors in a simple but meaningful chain + +dag: + name: "scheduled_greeting_with_checks" + start_time: "2025-10-23T09:00:00" + + tasks: + + # --------------------------------------------------------- + # 1) Start message triggered at schedule time + # --------------------------------------------------------- + - task_id: "start_message" + type: "PrintTask" + params: + message: "Scheduled run initiated: performing system checks..." + dependencies: [] + + # --------------------------------------------------------- + # 2) Bash task: count files in current directory + # --------------------------------------------------------- + - task_id: "count_files" + type: "BashTask" + params: + command: "echo 'Files in working dir:'; ls -1 | wc -l" + dependencies: ["start_message"] + + # --------------------------------------------------------- + # 3) Python diagnostic. + # --------------------------------------------------------- + - task_id: "python_diag" + type: "PythonTask" + params: + script: | + import math + print("Python diagnostic: sqrt(144) is", math.sqrt(144)) + dependencies: ["count_files"] + + # --------------------------------------------------------- + # 4) Final greeting output + # --------------------------------------------------------- + - task_id: "greet" + type: "PrintTask" + params: + message: "System checks passed. Scheduled greeting: Good morning!" + dependencies: ["python_diag"] diff --git a/examples/2_New_examples/6.3.1.scheduled_greeting_full_healthcheck.yaml b/examples/2_New_examples/6.3.1.scheduled_greeting_full_healthcheck.yaml new file mode 100644 index 0000000..5faf9d6 --- /dev/null +++ b/examples/2_New_examples/6.3.1.scheduled_greeting_full_healthcheck.yaml @@ -0,0 +1,125 @@ +abstract: | + This DAG represents a high-complexity scheduled pipeline that performs a + multi-branch system healthcheck immediately after being triggered at the + specified start_time. + + It performs: + - A scheduled initialization message + - A three-branch parallel healthcheck: + * Disk usage check (Bash) + * Process list inspection (Bash) + * Python runtime diagnostic (Python) + - A second tier of validation tasks combining results + - A final merge followed by an end-of-run greeting message + + This DAG stresses: + - Multi-branch parallelism after scheduled activation + - Mixed Python/Bash execution patterns + - Multi-level fan-out/fan-in consolidation + - Readability of logs under intense task parallelism + +dag: + name: "scheduled_greeting_full_healthcheck" + start_time: "2025-10-23T09:00:00" + + tasks: + + # --------------------------------------------------------- + # 1) Scheduled activation + # --------------------------------------------------------- + - task_id: "start" + type: "PrintTask" + params: + message: "Scheduled healthcheck triggered. Beginning diagnostics..." + dependencies: [] + + # ========================================================= + # PRIMARY FAN-OUT — three healthcheck branches + # ========================================================= + + # -------- BRANCH A: DISK HEALTH -------- + - task_id: "disk_usage" + type: "BashTask" + params: + command: "echo 'Disk usage:'; df -h | head -n 5" + dependencies: ["start"] + + - task_id: "disk_inodes" + type: "BashTask" + params: + command: "echo 'Inode usage:'; df -i | head -n 5" + dependencies: ["disk_usage"] + + # -------- BRANCH B: PROCESS HEALTH -------- + - task_id: "proc_list" + type: "BashTask" + params: + command: "echo 'Process snapshot:'; ps aux | head -n 5" + dependencies: ["start"] + + - task_id: "proc_count" + type: "BashTask" + params: + command: "echo 'Process count:'; ps aux | wc -l" + dependencies: ["proc_list"] + + # -------- BRANCH C: PYTHON RUNTIME HEALTH -------- + - task_id: "python_version" + type: "PythonTask" + params: + script: | + import platform + print('Python version:', platform.python_version()) + dependencies: ["start"] + + - task_id: "python_math_check" + type: "PythonTask" + params: + script: | + import math + print('Sanity check: sqrt(256) =', math.sqrt(256)) + dependencies: ["python_version"] + + # ========================================================= + # SECONDARY FAN-IN (partial merges) + # ========================================================= + + - task_id: "merge_disk" + type: "PrintTask" + params: + message: "Disk health checks complete." + dependencies: ["disk_inodes"] + + - task_id: "merge_proc" + type: "PrintTask" + params: + message: "Process health checks complete." + dependencies: ["proc_count"] + + - task_id: "merge_python" + type: "PrintTask" + params: + message: "Python runtime checks complete." + dependencies: ["python_math_check"] + + # ========================================================= + # FINAL MERGE — all health lines must be green + # ========================================================= + + - task_id: "final_merge" + type: "PrintTask" + params: + message: "All system healthchecks completed successfully. Consolidating..." + dependencies: + - "merge_disk" + - "merge_proc" + - "merge_python" + + # --------------------------------------------------------- + # FINAL GREETING + # --------------------------------------------------------- + - task_id: "greet" + type: "PrintTask" + params: + message: "Good morning! System is healthy. Scheduled healthcheck complete." + dependencies: ["final_merge"] diff --git a/examples/2_New_examples/6.3.2.scheduled_greeting_with_warnings.yaml b/examples/2_New_examples/6.3.2.scheduled_greeting_with_warnings.yaml new file mode 100644 index 0000000..4d80cac --- /dev/null +++ b/examples/2_New_examples/6.3.2.scheduled_greeting_with_warnings.yaml @@ -0,0 +1,107 @@ +abstract: | + This DAG is a difficult-level extension of the scheduled greeting concept. + It introduces a warning-style execution flow where some tasks can emit + notices about potential system issues without failing the pipeline. + + After being triggered at the scheduled start_time, the DAG: + - Runs standard initialization + - Executes three diagnostic branches: + * Disk capacity early-warning (Bash) + * Memory usage early-warning (Bash) + * Python environment sanity-check (Python) + - Each branch includes a task that might produce a “warning” message + (but never fails the DAG) + - Merges the three warning lines into a final scheduled greeting + + This structure tests: + - Complex scheduled pipelines + - Multibranch warning logic + - Sequential and parallel execution mixing Bash and Python + - Graceful, non-blocking error/warning reporting + +dag: + name: "scheduled_greeting_with_warnings" + start_time: "2025-10-23T09:00:00" + + tasks: + + # --------------------------------------------------------- + # 1) Scheduled start + # --------------------------------------------------------- + - task_id: "start" + type: "PrintTask" + params: + message: "Scheduled diagnostics starting (warning mode)..." + dependencies: [] + + # ========================================================= + # BRANCH A — DISK EARLY-WARNING + # ========================================================= + - task_id: "disk_free" + type: "BashTask" + params: + command: "echo 'Checking disk free space:'; df -h | head -n 5" + dependencies: ["start"] + + - task_id: "disk_warning" + type: "PythonTask" + params: + script: | + # This is not a real fail: it's a warning generator + print('WARNING: Disk free space could be evaluated as low (simulated).') + dependencies: ["disk_free"] + + # ========================================================= + # BRANCH B — MEMORY EARLY-WARNING + # ========================================================= + - task_id: "mem_usage" + type: "BashTask" + params: + command: "echo 'Checking memory usage:'; free -h || echo 'free command not available'" + dependencies: ["start"] + + - task_id: "mem_warning" + type: "PythonTask" + params: + script: | + print('WARNING: Memory usage seems elevated (simulated).') + dependencies: ["mem_usage"] + + # ========================================================= + # BRANCH C — PYTHON ENVIRONMENT SANITY-CHECK + # ========================================================= + - task_id: "py_info" + type: "PythonTask" + params: + script: | + import platform + print('Python interpreter:', platform.python_version()) + dependencies: ["start"] + + - task_id: "py_warning" + type: "PythonTask" + params: + script: | + print('WARNING: Python environment missing optional modules (simulated).') + dependencies: ["py_info"] + + # ========================================================= + # FINAL MERGE — gather all warnings + # ========================================================= + - task_id: "final_merge" + type: "PrintTask" + params: + message: "All warnings collected. Finalizing scheduled greeting..." + dependencies: + - "disk_warning" + - "mem_warning" + - "py_warning" + + # --------------------------------------------------------- + # FINAL GREETING + # --------------------------------------------------------- + - task_id: "greet" + type: "PrintTask" + params: + message: "Good morning! Scheduled diagnostics completed with warnings." + dependencies: ["final_merge"] diff --git a/examples/2_New_examples/6.3.3.scheduled_greeting_deep_tree.yaml b/examples/2_New_examples/6.3.3.scheduled_greeting_deep_tree.yaml new file mode 100644 index 0000000..a5ef4f1 --- /dev/null +++ b/examples/2_New_examples/6.3.3.scheduled_greeting_deep_tree.yaml @@ -0,0 +1,118 @@ +abstract: | + This DAG represents the most complex variant of the scheduled greeting series. + Instead of a wide multi-branch structure, this version stresses the orchestrator + through a *deep linear pipeline* of many sequential tasks that mix Bash, Python, + and Print operations. + + After starting at the specified start_time, the DAG performs: + - Several low-level checks (environment, directory, process snapshot) + - A multi-step data-processing simulation in Python + - Several Bash tasks that transform intermediate results + - A final merge into a greeting message + + This sequential depth tests: + - Long-chain dependency resolution + - Logging consistency over many tasks + - Accumulated scheduling latency + - Cross-executor transitions (Print → Bash → Python → Bash → Python → ...) + + It is recommended for validating stability during long-running DAGs. + +dag: + name: "scheduled_greeting_deep_pipeline" + start_time: "2025-10-23T09:00:00" + + tasks: + + # --------------------------------------------------------- + # 1) START + # --------------------------------------------------------- + - task_id: "t1_start" + type: "PrintTask" + params: + message: "Deep pipeline scheduled run started..." + dependencies: [] + + # --------------------------------------------------------- + # 2) ENVIRONMENT CHECKS + # --------------------------------------------------------- + - task_id: "t2_show_date" + type: "BashTask" + params: + command: "echo 'Current date:'; date" + dependencies: ["t1_start"] + + - task_id: "t3_list_dir" + type: "BashTask" + params: + command: "echo 'Listing working directory:'; ls -1" + dependencies: ["t2_show_date"] + + - task_id: "t4_process_snapshot" + type: "BashTask" + params: + command: "echo 'Process snapshot:'; ps aux | head -n 5" + dependencies: ["t3_list_dir"] + + # --------------------------------------------------------- + # 3) PYTHON DATA PROCESSING SIMULATION + # --------------------------------------------------------- + - task_id: "t5_generate_numbers" + type: "PythonTask" + params: + script: | + nums = list(range(1, 11)) + print("Generated numbers:", nums) + dependencies: ["t4_process_snapshot"] + + - task_id: "t6_square_numbers" + type: "PythonTask" + params: + script: | + nums = list(range(1, 11)) + squares = [n*n for n in nums] + print("Square numbers:", squares) + dependencies: ["t5_generate_numbers"] + + - task_id: "t7_sum_squares" + type: "PythonTask" + params: + script: | + nums = list(range(1, 11)) + squares = [n*n for n in nums] + print("Sum of squares:", sum(squares)) + dependencies: ["t6_square_numbers"] + + # --------------------------------------------------------- + # 4) BASH DATA TRANSFORMATION SIMULATION + # --------------------------------------------------------- + - task_id: "t8_fake_export" + type: "BashTask" + params: + command: "echo 'Exporting data... (simulated)'" + dependencies: ["t7_sum_squares"] + + - task_id: "t9_fake_compress" + type: "BashTask" + params: + command: "echo 'Compressing data... (simulated)'" + dependencies: ["t8_fake_export"] + + # --------------------------------------------------------- + # 5) PYTHON FINAL CHECK + # --------------------------------------------------------- + - task_id: "t10_python_summary" + type: "PythonTask" + params: + script: | + print("Final Python summary OK.") + dependencies: ["t9_fake_compress"] + + # --------------------------------------------------------- + # 6) FINAL GREETING + # --------------------------------------------------------- + - task_id: "t11_finish" + type: "PrintTask" + params: + message: "Deep scheduled pipeline completed. Good morning!" + dependencies: ["t10_python_summary"] diff --git a/examples/2_New_examples/7_mixed_execution.yaml b/examples/2_New_examples/7.1.1.Mixed_execution.yaml similarity index 100% rename from examples/2_New_examples/7_mixed_execution.yaml rename to examples/2_New_examples/7.1.1.Mixed_execution.yaml diff --git a/examples/2_New_examples/7.2.1.mixed_execution_extended.yaml b/examples/2_New_examples/7.2.1.mixed_execution_extended.yaml new file mode 100644 index 0000000..1a787c2 --- /dev/null +++ b/examples/2_New_examples/7.2.1.mixed_execution_extended.yaml @@ -0,0 +1,83 @@ +abstract: | + This DAG extends the original mixed_execution example by introducing + a multi-step data-processing workflow that mixes PrintTask, PythonTask, + and BashTask execution in a longer and more realistic sequence. + + The pipeline performs: + - Initialization/logging + - Python generation of data.json + - Bash extraction and transformation of data + - Python validation of manipulated results + - Final PrintTask summarizing the process + + This DAG is ideal for testing: + - Sequential mixed-executor behavior + - File creation + reading across Python/Bash boundaries + - Logging clarity during multi-step data transformations + +dag: + name: "mixed_execution_extended" + + tasks: + + # --------------------------------------------------------- + # 1) Initialization + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting extended mixed execution pipeline..." + dependencies: [] + + # --------------------------------------------------------- + # 2) Generate a JSON file using Python + # --------------------------------------------------------- + - task_id: "generate_data" + type: "PythonTask" + params: + script: | + import json + data = {"values": [1, 2, 3, 4, 5, 6]} + with open("data.json", "w") as f: + json.dump(data, f) + print("Generated data.json with 6 values.") + dependencies: ["init"] + + # --------------------------------------------------------- + # 3) Extract values using a Bash pipeline + # --------------------------------------------------------- + - task_id: "extract_numbers" + type: "BashTask" + params: + command: "cat data.json | grep -o '[0-9]' > extracted.txt" + dependencies: ["generate_data"] + + # --------------------------------------------------------- + # 4) Count extracted values (Bash) + # --------------------------------------------------------- + - task_id: "count_numbers" + type: "BashTask" + params: + command: "wc -l extracted.txt" + dependencies: ["extract_numbers"] + + # --------------------------------------------------------- + # 5) Validate inside Python + # --------------------------------------------------------- + - task_id: "python_validation" + type: "PythonTask" + params: + script: | + with open("extracted.txt") as f: + lines = f.readlines() + print("Python validation: extracted", len(lines), "numbers.") + dependencies: ["count_numbers"] + + # --------------------------------------------------------- + # 6) Final message + # --------------------------------------------------------- + - task_id: "finish" + type: "PrintTask" + params: + message: "Extended mixed execution pipeline completed successfully!" + dependencies: ["python_validation"] diff --git a/examples/2_New_examples/7.2.2.mixed_execution_branching.yaml b/examples/2_New_examples/7.2.2.mixed_execution_branching.yaml new file mode 100644 index 0000000..451141e --- /dev/null +++ b/examples/2_New_examples/7.2.2.mixed_execution_branching.yaml @@ -0,0 +1,84 @@ +abstract: | + This DAG expands the mixed_execution pattern by introducing a branching + data-processing pipeline. After generating a JSON file, the DAG splits + into two parallel branches: + + - Branch A (Bash): extracts digits from the file and counts them + - Branch B (Python): loads the JSON and computes a numeric statistic + + Both branches then merge into a final PrintTask. + + This structure tests: + - Mixed Python/Bash execution across parallel branches + - Synchronization via a multi-branch fan-in + - Correct handling of file I/O across executors + - Readability of logs in concurrent flows + +dag: + name: "mixed_execution_branching" + + tasks: + + # --------------------------------------------------------- + # 1) START MESSAGE + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting mixed execution branching pipeline..." + dependencies: [] + + # --------------------------------------------------------- + # 2) GENERATE JSON DATA USING PYTHON + # --------------------------------------------------------- + - task_id: "generate_data" + type: "PythonTask" + params: + script: | + import json + data = {"values": [3, 5, 8, 13, 21]} + with open("numbers.json", "w") as f: + json.dump(data, f) + print("Generated numbers.json with Fibonacci-like values.") + dependencies: ["init"] + + # ========================================================= + # BRANCH A — BASH PROCESSING + # ========================================================= + + - task_id: "bash_extract_digits" + type: "BashTask" + params: + command: "grep -o '[0-9]' numbers.json > digits.txt" + dependencies: ["generate_data"] + + - task_id: "bash_count_digits" + type: "BashTask" + params: + command: "wc -l digits.txt" + dependencies: ["bash_extract_digits"] + + # ========================================================= + # BRANCH B — PYTHON PROCESSING + # ========================================================= + + - task_id: "python_compute_sum" + type: "PythonTask" + params: + script: | + import json + with open("numbers.json") as f: + data = json.load(f) + print("Sum of values:", sum(data["values"])) + dependencies: ["generate_data"] + + # ========================================================= + # FINAL MERGE + # ========================================================= + - task_id: "finish" + type: "PrintTask" + params: + message: "Branching mixed execution pipeline completed!" + dependencies: + - "bash_count_digits" + - "python_compute_sum" diff --git a/examples/2_New_examples/7.2.3.mixed_execution_random_data.yaml b/examples/2_New_examples/7.2.3.mixed_execution_random_data.yaml new file mode 100644 index 0000000..c63d585 --- /dev/null +++ b/examples/2_New_examples/7.2.3.mixed_execution_random_data.yaml @@ -0,0 +1,86 @@ +abstract: | + This DAG extends the mixed execution concept by introducing randomness, + multi-step processing, and mixed Python/Bash transformations. + + The pipeline: + - Generates a random dataset via Python and stores it in random.json + - Uses Bash to extract digits and store them in extracted.txt + - Uses Python to compute statistics (min, max, sum) + - Performs a secondary Bash transformation on the extracted digits + - Merges all tasks into a final PrintTask + + This DAG is ideal for testing: + - Non-deterministic workflow behavior + - Python/Bash interoperability via generated files + - Multi-step mixed executor pipelines + +dag: + name: "mixed_execution_random_data" + + tasks: + + # --------------------------------------------------------- + # 1) Initialization + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting mixed execution random-data pipeline..." + dependencies: [] + + # --------------------------------------------------------- + # 2) Python generates a random dataset + # --------------------------------------------------------- + - task_id: "generate_random_data" + type: "PythonTask" + params: + script: | + import json, random + values = [random.randint(1, 50) for _ in range(10)] + with open("random.json", "w") as f: + json.dump({"values": values}, f) + print("Generated random.json with:", values) + dependencies: ["init"] + + # --------------------------------------------------------- + # 3) Bash extracts digits from JSON (flat extraction) + # --------------------------------------------------------- + - task_id: "bash_extract" + type: "BashTask" + params: + command: "grep -o '[0-9]' random.json > extracted.txt" + dependencies: ["generate_random_data"] + + # --------------------------------------------------------- + # 4) Python computes summary statistics + # --------------------------------------------------------- + - task_id: "python_stats" + type: "PythonTask" + params: + script: | + import json + with open("random.json") as f: + data = json.load(f) + vals = data["values"] + print("Statistics: min =", min(vals), + "max =", max(vals), + "sum =", sum(vals)) + dependencies: ["bash_extract"] + + # --------------------------------------------------------- + # 5) Bash further processes extracted digits + # --------------------------------------------------------- + - task_id: "bash_count_digits" + type: "BashTask" + params: + command: "wc -l extracted.txt" + dependencies: ["python_stats"] + + # --------------------------------------------------------- + # 6) Final summary + # --------------------------------------------------------- + - task_id: "finish" + type: "PrintTask" + params: + message: "Random-data mixed execution pipeline completed!" + dependencies: ["bash_count_digits"] diff --git a/examples/2_New_examples/7.3.1.mixed_execution_data_pipeline.yaml b/examples/2_New_examples/7.3.1.mixed_execution_data_pipeline.yaml new file mode 100644 index 0000000..90ab788 --- /dev/null +++ b/examples/2_New_examples/7.3.1.mixed_execution_data_pipeline.yaml @@ -0,0 +1,141 @@ +abstract: | + This DAG represents a difficult-level mixed execution workflow designed to + emulate a realistic multi-stage data pipeline. + + It performs: + - Initial logging + - Python-based dataset generation (JSON) + - Bash-based filtering, extraction, and transformation + - Python statistical computations on cleaned data + - A two-branch secondary processing stage (parallel) + - A final merge followed by a PrintTask summary + + This DAG stresses: + - Multi-step Python/Bash handoff via intermediate files + - Parallelization of heavy-processing branches + - Long dependency chains + - Strong orchestrator scheduling and logging capabilities + +dag: + name: "mixed_execution_data_pipeline" + + tasks: + + # --------------------------------------------------------- + # 1) Initialization + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting complex mixed execution data pipeline..." + dependencies: [] + + # --------------------------------------------------------- + # 2) Python generates a medium-sized dataset + # --------------------------------------------------------- + - task_id: "generate_dataset" + type: "PythonTask" + params: + script: | + import json, random + values = [random.randint(10, 999) for _ in range(50)] + with open("dataset.json", "w") as f: + json.dump({"values": values}, f) + print("Generated dataset.json with 50 integers.") + dependencies: ["init"] + + # --------------------------------------------------------- + # 3) Bash filters digits from dataset.json + # (raw extraction) + # --------------------------------------------------------- + - task_id: "extract_digits" + type: "BashTask" + params: + command: "grep -o '[0-9]' dataset.json > digits_raw.txt" + dependencies: ["generate_dataset"] + + # --------------------------------------------------------- + # 4) Bash aggregates digits into a cleaner file + # --------------------------------------------------------- + - task_id: "group_digits" + type: "BashTask" + params: + command: "tr -d '\\n' < digits_raw.txt > digits_grouped.txt" + dependencies: ["extract_digits"] + + # --------------------------------------------------------- + # 5) Python interprets digits as numbers & computes basic stats + # --------------------------------------------------------- + - task_id: "compute_stats" + type: "PythonTask" + params: + script: | + with open("digits_grouped.txt") as f: + content = f.read().strip() + if not content: + print("No digits found.") + else: + nums = list(map(int, content)) + print("Digit count:", len(nums)) + print("Digit sum:", sum(nums)) + print("Highest digit:", max(nums)) + print("Lowest digit:", min(nums)) + dependencies: ["group_digits"] + + # ========================================================= + # SECONDARY FAN-OUT (two parallel branches) + # ========================================================= + + # -------- BRANCH A: BASH TRANSFORMATION -------- + - task_id: "branch_a_reverse" + type: "BashTask" + params: + command: "rev digits_grouped.txt > reversed.txt" + dependencies: ["compute_stats"] + + - task_id: "branch_a_count" + type: "BashTask" + params: + command: "wc -m reversed.txt" + dependencies: ["branch_a_reverse"] + + # -------- BRANCH B: PYTHON TRANSFORMATION -------- + - task_id: "branch_b_split" + type: "PythonTask" + params: + script: | + with open("digits_grouped.txt") as f: + data = f.read().strip() + chunks = [data[i:i+5] for i in range(0, len(data), 5)] + print("Chunked data:", chunks) + with open("chunks.txt", "w") as out: + for c in chunks: + out.write(c + "\\n") + dependencies: ["compute_stats"] + + - task_id: "branch_b_linecount" + type: "BashTask" + params: + command: "wc -l chunks.txt" + dependencies: ["branch_b_split"] + + # ========================================================= + # FINAL MERGE + # ========================================================= + + - task_id: "final_merge" + type: "PrintTask" + params: + message: "All branches completed. Consolidating results..." + dependencies: + - "branch_a_count" + - "branch_b_linecount" + + # --------------------------------------------------------- + # FINAL SUMMARY + # --------------------------------------------------------- + - task_id: "finish" + type: "PrintTask" + params: + message: "Complex mixed execution data pipeline successfully completed!" + dependencies: ["final_merge"] diff --git a/examples/2_New_examples/7.3.2.mixed_execution_transform_tree.yaml b/examples/2_New_examples/7.3.2.mixed_execution_transform_tree.yaml new file mode 100644 index 0000000..c74c103 --- /dev/null +++ b/examples/2_New_examples/7.3.2.mixed_execution_transform_tree.yaml @@ -0,0 +1,139 @@ +abstract: | + This DAG represents a difficult-level mixed execution workflow arranged + as a *transformation tree*. After an initial dataset is generated, the + pipeline fans out into three separate transformation branches. Each branch + performs a unique sequence of Python and Bash tasks, including numeric + transformations, string processing, and data restructuring. + + After completing their internal steps, the branches converge into a + second-tier merge node, which triggers a final summary PrintTask. + + This DAG is designed to stress: + - Mixed Python/Bash fan-out patterns + - Tree-like dependency structures + - Multi-file intermediate data handling + - Parallelism in complex data transformation pipelines + +dag: + name: "mixed_execution_transform_tree" + + tasks: + + # --------------------------------------------------------- + # 1) INITIALIZATION + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting transform-tree mixed execution pipeline..." + dependencies: [] + + # --------------------------------------------------------- + # 2) PYTHON: GENERATE DATASET + # --------------------------------------------------------- + - task_id: "generate_dataset" + type: "PythonTask" + params: + script: | + import json, random + data = {"values": [random.randint(1, 100) for _ in range(20)]} + with open("transform_data.json", "w") as f: + json.dump(data, f) + print("Generated transform_data.json with:", data["values"]) + dependencies: ["init"] + + # ========================================================= + # PRIMARY FAN-OUT — 3 TRANSFORMATION BRANCHES + # ========================================================= + + # ---------------------- BRANCH A ------------------------- + # Processes values through scaling & Bash formatting + - task_id: "A1_scale_values" + type: "PythonTask" + params: + script: | + import json + with open("transform_data.json") as f: + data = json.load(f) + scaled = [v * 2 for v in data["values"]] + with open("scaled.txt", "w") as out: + out.write("\\n".join(map(str, scaled))) + print("Branch A scaled values:", scaled) + dependencies: ["generate_dataset"] + + - task_id: "A2_show_head" + type: "BashTask" + params: + command: "echo 'Branch A head:'; head -n 5 scaled.txt" + dependencies: ["A1_scale_values"] + + + # ---------------------- BRANCH B ------------------------- + # Converts numbers into labeled strings, then filters with Bash + - task_id: "B1_label_values" + type: "PythonTask" + params: + script: | + import json + with open("transform_data.json") as f: + data = json.load(f) + with open("labeled.txt", "w") as out: + for v in data["values"]: + out.write(f\"Value_{v}\\n\") + print("Branch B labeled values written to labeled.txt") + dependencies: ["generate_dataset"] + + - task_id: "B2_filter_labels" + type: "BashTask" + params: + command: "grep 'Value_[5-9][0-9]' labeled.txt > filtered_labels.txt || true" + dependencies: ["B1_label_values"] + + + # ---------------------- BRANCH C ------------------------- + # Computes statistics, then restructures data + - task_id: "C1_compute_stats" + type: "PythonTask" + params: + script: | + import json + with open("transform_data.json") as f: + data = json.load(f) + vals = data["values"] + stats = { + "min": min(vals), + "max": max(vals), + "avg": sum(vals)/len(vals) + } + with open("stats.json", "w") as f: + json.dump(stats, f) + print("Branch C stats:", stats) + dependencies: ["generate_dataset"] + + - task_id: "C2_format_stats" + type: "BashTask" + params: + command: "cat stats.json | tr -d '{}\"' | tr ',' '\\n' > stats_formatted.txt" + dependencies: ["C1_compute_stats"] + + # ========================================================= + # SECONDARY FAN-IN — combine outputs of A, B, C + # ========================================================= + + - task_id: "combine_results" + type: "PrintTask" + params: + message: "All transformation branches completed. Consolidating results..." + dependencies: + - "A2_show_head" + - "B2_filter_labels" + - "C2_format_stats" + + # --------------------------------------------------------- + # FINAL SUMMARY + # --------------------------------------------------------- + - task_id: "finish" + type: "PrintTask" + params: + message: "Transform-tree mixed execution pipeline completed successfully!" + dependencies: ["combine_results"] diff --git a/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml b/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml new file mode 100644 index 0000000..e21b7c5 --- /dev/null +++ b/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml @@ -0,0 +1,180 @@ +abstract: | + This DAG represents a “hyperpipeline” workflow — a difficult-level DAG with + multiple levels of branching, cross-branch transformations, wide fan-out and + multi-step merging. It is designed to push the orchestrator to its limits in + terms of: + + - Concurrency + - File I/O across many transformations + - Mixed Python/Bash workloads + - Deep and wide dependency structures + - Multi-level fan-in / fan-out synchronization + + The pipeline: + - Generates a dataset + - Fans out into four primary branches (A–D), each performing two transformations + - Produces intermediate files consumed by a second-level processing wave + - Recombines all results in multiple merge stages + - Ends with a final summary message + + This DAG is suitable for large-scale stress testing. + +dag: + name: "mixed_execution_hyperpipeline" + + tasks: + + # --------------------------------------------------------- + # 1) INITIALIZATION + # --------------------------------------------------------- + - task_id: "init" + type: "PrintTask" + params: + message: "Starting the hyperpipeline..." + dependencies: [] + + # --------------------------------------------------------- + # 2) DATA GENERATION (Python) + # --------------------------------------------------------- + - task_id: "generate_data" + type: "PythonTask" + params: + script: | + import json, random + data = {"values": [random.randint(1, 500) for _ in range(30)]} + with open("hyper_data.json", "w") as f: + json.dump(data, f) + print("Generated hyper_data.json with 30 random values.") + dependencies: ["init"] + + # ========================================================= + # PRIMARY FAN-OUT — 4 BRANCHES A, B, C, D + # Each branch performs two transformation steps. + # ========================================================= + + # ------------------------ BRANCH A ------------------------ + - task_id: "A1_extract_digits" + type: "BashTask" + params: + command: "grep -o '[0-9]' hyper_data.json > A_digits.txt" + dependencies: ["generate_data"] + + - task_id: "A2_count_digits" + type: "BashTask" + params: + command: "wc -l A_digits.txt" + dependencies: ["A1_extract_digits"] + + # ------------------------ BRANCH B ------------------------ + - task_id: "B1_load_and_double" + type: "PythonTask" + params: + script: | + import json + with open("hyper_data.json") as f: + data = json.load(f) + doubled = [v*2 for v in data["values"]] + with open("B_doubled.txt", "w") as out: + out.write("\\n".join(map(str, doubled))) + print("Branch B doubled data.") + dependencies: ["generate_data"] + + - task_id: "B2_show_head" + type: "BashTask" + params: + command: "head -n 5 B_doubled.txt" + dependencies: ["B1_load_and_double"] + + # ------------------------ BRANCH C ------------------------ + - task_id: "C1_compute_stats" + type: "PythonTask" + params: + script: | + import json + with open("hyper_data.json") as f: + data = json.load(f) + vals = data["values"] + stats = { + "avg": sum(vals)/len(vals), + "min": min(vals), + "max": max(vals) + } + with open("C_stats.json", "w") as out: + json.dump(stats, out) + print("Branch C computed stats.") + dependencies: ["generate_data"] + + - task_id: "C2_format_stats" + type: "BashTask" + params: + command: "cat C_stats.json | tr -d '{}\"' | tr ',' '\\n' > C_stats_formatted.txt" + dependencies: ["C1_compute_stats"] + + # ------------------------ BRANCH D ------------------------ + - task_id: "D1_extract_even" + type: "PythonTask" + params: + script: | + import json + with open("hyper_data.json") as f: + data = json.load(f) + evens = [v for v in data["values"] if v % 2 == 0] + with open("D_evens.txt", "w") as out: + out.write("\\n".join(map(str, evens))) + print("Branch D extracted even numbers.") + dependencies: ["generate_data"] + + - task_id: "D2_count_even" + type: "BashTask" + params: + command: "wc -l D_evens.txt" + dependencies: ["D1_extract_even"] + + # ========================================================= + # SECONDARY FAN-IN — merge results from A, B, C, D + # ========================================================= + - task_id: "merge_primary" + type: "PrintTask" + params: + message: "Primary transformations complete. Launching second-stage processing..." + dependencies: + - "A2_count_digits" + - "B2_show_head" + - "C2_format_stats" + - "D2_count_even" + + # ========================================================= + # SECONDARY FAN-OUT — two parallel heavy processing steps + # ========================================================= + - task_id: "S1_consolidate_text" + type: "BashTask" + params: + command: "cat A_digits.txt B_doubled.txt D_evens.txt 2>/dev/null | wc -l" + dependencies: ["merge_primary"] + + - task_id: "S2_summary_python" + type: "PythonTask" + params: + script: | + print('Second-stage summary: Python confirming prior merges completed.') + dependencies: ["merge_primary"] + + # ========================================================= + # FINAL MERGE + # ========================================================= + - task_id: "final_merge" + type: "PrintTask" + params: + message: "Second-stage processing complete. Finalizing hyperpipeline..." + dependencies: + - "S1_consolidate_text" + - "S2_summary_python" + + # --------------------------------------------------------- + # FINAL SUMMARY + # --------------------------------------------------------- + - task_id: "finish" + type: "PrintTask" + params: + message: "Hyperpipeline completed successfully!" + dependencies: ["final_merge"] diff --git a/examples/2_New_examples/DAG explanation.ods b/examples/2_New_examples/DAG explanation.ods deleted file mode 100644 index 58a16a946d2cc34d23cbd61048de869b9cf6a3bb..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 13328 zcmbVz1yo$wvNjsroj?fg65J&;9^73UhXxuClHd^BHMj?N3GNUaf)m``{U`Umd6~H{ zGwZ&;dY#p~d!71roxQ88tLl75K?WKI8v+6z0-{%1Q`_H~Cxi(C0^;ZK{1t?ag$>Zr z%?@a2XJ>6;Z0KkKvSoI*HDLl7f-S&IAUmL~3CP&V259TZ1h#Vk8k&I3fj~!vzhFMY z{ErYlk3>PXrWR&S4*x;}vohN`fXp0#V6c&)1M}Z&;{QhTd0wY~)*}3kmL15>$?m7> zKg{+!J+PypqtpL~m5v~g^*>ul@*68HYz)nSVCJ_Ljy8sN;D0dncMD@`4Kj2D{?l^) z?gD=rYG-H*wEl!#>PNv;PZZh{+JOS9{ykM^n6|aBgoGq8$(+QQy|!p$-%_* zTg*3GUlw$~BYuD2@Qg4sEUW{~K!as^8%KZ!Ec0nH=2A;|jGAIj(Qf+WTB$hOa(H#h zfgw@fr^kmdEU91T6&UX;uA~X+;qA|7h5_W{dj*yE@bPG zsw^7Ud|Ko!%&0^( ztX29NSo=Eaqi`6)y8%B(U!q1inQpfDGew^E%!AclgRc|5S^Nga^x-2jOF_ta8+t#! zmLKZ)E_=k8yB`|6J56+-ER)9?q_Q5$?>6zdJ#9TQjXV*omQ_0gtc8cO&2A&Ns!vH& zCCU_J;NYvT%*SCNAs~FAAt3(C|9`i9h|kLhc674_f|*=xtT(i+Y!~`49@CZ4jZG?$ z{8o)AwArk}qC9~pGs7L1WPN2~XtV3=?<#{$Bg9WfQYqm27#(6#DW>FGcqV7Y2ZF8l zlXX>fr}V!z_D`STak*{k7^`7u=`i}Rd9VB+%X&jQUff-F|MDvcjx8IV(hE4{D=c>&#qMMh8FBDds(+>JEl<87|-OV9gc+ zWVgMz=B|H>_hF$~XWD2(vTf0jY{AR1Wsg5?B|tlZzRA6Sk1CA6a`>^O0aVB+Dk!ojbl!rWId_kNXqZcImfd|eL;5P(`rFJVJP-&FaiQ(jya) zK{y@y*`TB}G>jm5I*%b7t_E6528|B0E9AZG@k($>EQ}e|#QyjT9Od23I2lCvms+JS z$7P`%x?ISMThH~N6(6@QiW?QGJZwGSm()H>gvM5ZRqX3@WX>=OpwPH=JQ(TfR+tD% zVK8+#+ue9mSf#6a>c-xU;JuvS!9vt!6TxscCIU`vHb`6X?H8D3o>-m;4D`iE`9pII zm9JAY*(fEi`fTRaLqxnSQioGgWekr#hSzZk!;g~MkS+EW%K-HI^BgHCq9mT+&{4wE#U$FHZGCHUY!Kx#9rDA&9OW`KqQV?MR0zCgnIiT;f1(uB9uiQ+E|^p)*Y$9G+2S$AZ7u zwV4cj9O-HaB^`8k4;JU0t(-CZ2TX+`a%(z4vExjEeAQrpT{OOOcSxm!8C%%`9>($@ ze(dGjIgF207s#~bO+fti3RD(SjP&y29c!g74R%-i^0*jyzI8}$5lw%_N2-XTxqQ*$ zg^7l!?**@*P5pJsS&b=V zUtYLV2z8F6r9;=$#UzW;3CUY>Dl3UR`5I}5t`~AK-2^!BQ*F*DRD#qNYFmP#GAfnS zAcEYJqcqGc!|`2l@?y*uNj}1E`90mGxB5SV$oitLC0N>!#|S=!wRq#sRfERJ8HN$4 zcy@Uwrk9KOuQC0-3TVkD7Gp)A_F1G{767_|1>Z)sUVF%3fBLS(WRg+Ngg(s?P!|iX z*gH~YO+C)E&#Sh2o82NZPlXfJJ5mungJTib_2aC1`fk|<&G`f?xqA8wkNmtM;ZDWZ z*v#)eBQul>nHyGC49kmGdo}ypgbcmcbaPGNjp93VIeM>mt|SkIb5HPysU(Cpav~+} zvDb~FG4(#878)4H@kvyhDI{dgld6iK->;_N=u9AIDK`_w?C`U++qC2+Rpc$EA%n6- zZ641eyfr6yV`lor9BhaNsYn9V(X?%%vxgYjW{legI>oROEG$VlME#{}nD>TfA*#aW zb-3}2RYhw-s_$aD0!cE98!TP${hrv)aa?)mZ04DbvYZ>f)6B1xwcgNeEbGl?hyjxp zYu!sxcM;^5>|I8=i)F3Ry?P|ia}ZW;=KZt^utYkWyllq9qgJ8W)wxI)su#zorVn4P zG%Pt3y&)ytH2X2(ew~pb0HeUK)im`^HcKZzFPZkGBS*0z{^xnM$*5UIPYio}(dpOAM^wmcB6?$zbO>y`nyNQw7iL3L`u` zZ#J@JE$FD2xklr`fBc$a{-RQzR$Vb?VCtcJU?P9_{7sd?nt)lD>=bmnLD8jWd%4yk zO2E;muY?c84?)r(GB9cyT>EmQN54lI3xA0JV;?>cE&Xb|?>#hWln`4uSVGPu-8thH zOtB;=)Z)d0xO@~o)ita9Ce`(P@-sK3iszHa+Y6Wsak+THf(g|4bO;q=v%9_zT5?b| z9_ATFH(*wUGqopN^@*~pSHOd)$wP~+S9RAH5(hkGj}+o-aFxcOs~p#>X6Gx!x$B<7 zmYp3z!H2PG;#-a?avq@OSGn)2;+rli9&C;t)aPlrWj$-f-Fpx{tX$~6O`}gV$jV$0 z*Qi?>%P$lro>UhnvX+B;hhdK|Z5Q|zJGO$t)mpX-L+>u$CieJJ!!$?$BaHD+3#PBRQtTu?3S#gy*8 zNOzt`w_@R9yRS0CyPbxa?8T2z6niabhYF7~bZbgS!R%v@GB1y3MU&;AygT2=qP;s-nT>q4akq?o0*B;G^^|N<) z!+y!5SSvz=zd%GGb%3%xoBE8GI&~2w=&a@zkDA}AvSUA(R0U1>8dl*^iOr5_7zgC24qN|brL|Q{XpM(sGSfj}8bo_|Te42hF z#6|sW?W{rma~m)49Cs`<1E!iqX=ei0bMdP@*LH>HMBZ%A zz0EQ3_!JGiPw~eq(5>U6v`tCW)7bWSJ#G*XkD-fq%pxbM?Xqo4Sz zs7KL%7*jV^WMuUcSVx6i4Wb$>96^z8giVZ=($-kl;SJV z-wMW7lUMR?0$o2?!yC)sc_R&@=m;5Zbs`VbM*$XH0$(W^zBuTt$pMwV%!4cI;{c3OH$b7qvIY|8*7blKN}1Yq zy5Kc%NVSL-IG{7IGbvD1x_=h!j0T|G0Si2FE1o~BMi18A9CLM8x^;8`rd4~@Ur!L6 zdG1DM_AaoKRaEQX%3R}#9BCkjoj1pwy7rZ47~!u41!X!0GLYI&TWL(b+O3FZ3{9(p zqK(P4DMl5~{&wBv%$wF?BWN}!XvI6O;}<^yVI6BA)oW-m>L$51a#LP4`$b@HhqNg_ zKfgBKwCHon9wjYWP5jmz1%)Mxq0783-$YnYm+T4cA{hlKo}n?9<%SVRYO~V9ELVGf z5UT+Q`+}1iVU6_)uD2?WoK_b;Z`I({k=wVXtLErWc`-ex$#RzQODHU$eYz5zQ?^%J z`jjO)`^5QUiN`$?y_g_|=Ssucs%3~f&JS;Lv1FGCbx~&)^ttCle}7fK89%1#4vZ*w zEk*)=y}*vVO-XDSzOarzSNi(9^(yuk)R6BP+oRpya;uUN?b>!m`#TV=&Z=QAhL<>= z^1NV0+#?W`Vw;K-fI5iAN5Ty;_mu8nuyM$UGsUkWGgLG%AC5N6bQ#m}11hj3VBNj& zyPf|Qh_iQEU|~+Ps5IV~N)?7RY@U#Zdb#;650#i<^JcF~A9}r_fJ&vQwo*7RN7)(+LR z`x(SeZc1bA$YMX`EN!Ngh*m_{FO8r0J7g|5_E|z#S$7P-^eDh!AZhA44SE^!iA~hR zjg2)O$M0;K5_KnW0vBp}ScQy`DJ0s+g|do@BA^D&0gH_eAVr=z?tlH+NRA_7W zgT_oIGoKsG2dpPR(c6JGjfr`rT5VQC7|+I1*f`pK*l)zBiAJR;3=KaDv*(?3%B_0~ z{XSASG1^d5BRqWdn*uI+lv7IUk_rdokBoQSYwciN*Tit08z!ViE4+ zniFL2tThTjHMSu^lqE8J8cey2%iP09XkU;-Dxy&ZU>rVBt)gz7)(!qJ*}wzFHh=Me z=m5nhtGd;n-n$%Uu_gs*tz+}jIbz7VLWWt$m>47(`5aGTQLX(P(6i56HUcSwsy!K6L zB;82C#BotXyv>j->`-V2t)1zP0}7`?&){A*U!_U_JJroOBH5H}|A#6^Onj2`IwM?= zV)Q2yS&4|R&NO~E+omE8bE*# z0!|UNADP1p71{g7ky*i2z*k|K+f(OkE5#ijdT3?DW<4jUv?!^kEH+Ckf3QTXjh`6h zW~9c`;(sSF-2D3PQlgwLt@KA*BGZxG2qRYFEIOw6hu_Rhhb_j7Q z?U{V50`|M{>rs>RP>o8-ZlfnT3`H9xxEKhk3Dw9vHsY=@ib6}m*(()v?hVmjEwHpK z?Aw7GDmolD!K-pt>+sr`RsZa{F2N2xne+}AgO zov{q`{G>cE?~#DaIcj}vEzuNsM@#oX6d0#NdYq}Vd*nwbgQ|p0bh;nS*H_UY9{3-_ zAhKxPcoVa!ecgoQA&^hdQxT#a;gFoiDa_@0las$^)oSAN$#%{ZI);<5G7MsVP&r`8 zPPr?(b38gvZ#E#wsO%z&;;~o8nU?QmcV*4vlXUWn%CI~V%M_vMJE9HN9rxnxSj#7? z1{XX-xh&-pXu^U&CtO$|vk-cVM+d^hPA{ba!u4RQ*u)E*f z0cYj)&}#hc+XC%LVcq<|-80o?_H81+6{ab;xV2lhQ=U&Bp!#AfRjjR}DA_134--mJ zSi;UG%SP@EayVM!n~wxUm#zpmJde;%AJxfjpk}ubCs;Q43J#oOuZ=Mc;Pt~zRbJ2t zWs`OAtU*D-y+$2fv0GXSnv4JiZqG?zb6(j6;6QCBS_MhI035kQ z!a#`}>wj%{Jbqjxb&x6Y`hd_wP$+-39hL-z9*|5J+wM&YPAnbd_i6Mlk-Fe`>Arzu z*49O=D9_Fz%6CU{%ju-b=!>7q2Tg}AzlNARP((l!PBz+u1*M8E84m(Wb9(@=Mw6m@Voe$swD$(&BAh73Z_EFX$_~g~542dgjm8KgPz=uSibC!wW;zkOh|Hk|MvLaOkFk z>!i8WVwL@2081H??kg}Edz3kB!Lvb~iY}Q=LU5+gRXma{f+9;4~WZ_?iIoxLy(J70FwurBOFEcW)5WV2i|eArh!c8Qid`2IpWo-B0Eo zbhiEtd}FQ2lw_~`C53O@Punn+xA0R3Kgn~FeItuhy-0w*RVQyt72j=mL2-1Nou3Y_ z_QvXtE?(VKF~+(wSICYNKx-J2Jk_FYeZ9yK$BtnBBU|U8*zm(e;BAI_<U3S$0s5p z^6uR`Wo2bOJv{>h14~OwFc|FN;o(|cC&Vhk} ziHV7Yg@v`XwY|N)v$M0u$H(V+o}QlM*EUa|M+kP3nqUYBB&?qYBt&xR^CciRO94bw zT;~oJgWa=vF<&T<274}ZNnJdGcqq#!1`j@0)APTDEm%eQ>YerYSC?uJq9! zbx4RIJ7Z27t%&tXmANIMK8&^s(rU#REb9pk&P<~9no=9%O&X8gSI63@?loF3!BA{# zR}St0C6jOkht|Ef;eFN2k@hAj@zN^D^wG){c-KL?aU1`A`_{D9AD_Fe9%L~WR%f(K zxWvwNJnc%R<)R&IyEfbRAc*>u8Wj5DxUWOeZW3l?p8ILGQO-{(iE-3$%+A}1SdrYB z4vH&?29vZwvTGHdxjkhnNb&?=SipWoNrxJH1`y#ry{`VD`LO>TBWm2lcMcZefM1q` zdWR=mggcgJ=lw9WXs#U|rPdOA>E=};(k@S^Oa+?J4~|-+vtZNu7aB{A3WU9BP3b_s z-~udq#;iP=9Qt5xM|VACrlH{na0#>vIqO@>ih@gF6FyMG*C3iuEGkyP5cfB3AEXnF zheRA>vnB1ZiPLBGs4_LwYfU)i0lW(lQ1%349p0|dJtOdK3?ucFo-ql(ye{N z-jSkcV1e~Ai@D~$vy%93+dwSp@WNK2?CiZUct3$TIO{HCi$r?Sy`J&X*|EV>$ntKb zEryzQnRKM(5+Nvt^y7AGdIvjO(Brg)HGi`=eW}KFZ#5OZ-|+I+n<#g_-tVUX0K+kLz>;W^3Pz z#iz_CjC?llKU2)2m_&x;K!2gCgEFAX_U-X1Iq*5-7E8!m^OB zwtjqDPetwBZCCHnO~V_Ue8%#cc9ZGcK>}%5iQmGafV=dQGZ&f5=x4?1JNl&}A0v5o z?b|sMc6*@3a#j%LXH&}J1lDZQY7>_b!Z~V0+;F-VNgmiyhkbTXROz1!6vz}V z4YRah$Yyy$CPe^g<13P1C%d__1QdC}M6crqWlPw>UliNC?}Rz{_3^r}NBG~-3AVZW z13Jez{_~ytwaZ6xu}N2;101A>L{2zynB9kzW}{4!i5vw<_R0Uy4Ss=@2~YA@!Pm9C z8jGd8zyf$;=66PjKIjUPSURsw6@DkD+mcd^z*6lNgcVvCoPdp3Qu2`2;~XnGN66Z( zU)xpd@h2^c=}=_JpLGsh{*>bo%UFFnmqM|Xqs5Ba*h0PALm6w2d6vi#AtRGVG``RR zKh~MF87$3LoHo}VOY5{Fs7W1N6CUn5?n?{aCBq9-!M!W^615i1ccfLz-Zec$df+f4 z?Y+0<@$K~c>*QEBAIa`9dkGCb|H(i>ckZ!JZrq&Sia3B`to^OiTE-R?U&lGcDOayv z=h|Tp)6BhR7{NYHklM|$fE_4!+MnG#o^032$b+#GPO)sS(IQxDECx$end`l(mwYS{ zAeY(G(v3JvR=|sVGVZ6Ri65f!df^aw1*R2d@YTAR9Xo$q$hrwfOd29cRNC2L-tfid6-+s7!gTM%xoFiV6P zp)dyizkKrdL5LYp?X&I8PVVi+8MbI%qL)~pBy!)?*}5w{0Q&T*&RcHsRW=ev&|S>V z?Xj!T#2{uE!2X6?Q8j7`B#$|8>n-QVFAx@T8Tc`MhrsViJ_~0A<3>Wg;azb+k+6R$ zt>ZR{FQ=Fp<@kA60g_X_8_6SqrbVB*^jcdoVDAwaG^8MLhXuGBKty#_hms3*a*cdzmEGQn z|G-7*xC*MgSk!WG8Qow_rd;l25SjEa+`g7YD|r^ha0g!$QZ`Nh2uq6V<*{m3(88%} z7u(n&{jkT7^M0!utHcRUx|g)4bG{J$L}SA@65d-gwh5z!l22y(`cxrHd)r#+-CRlA z9CORmm5g%6_IYUkcNh81XRt=?q$huEe9~I7QK`uk38Y8wdJG?vqrL~tS zpVL`|iX8?i_d*2wol2~V!w4hGdBf{jRDM`7{?3-jIKIM#l8;pmnHH_zE-L`y^VC6z z)ZycMP@VO(K;JUj_|7(JLaZCW9-_4Sa86qVN=-L`n1tsrUvhR;Uf>oPaMol_VV$d8 zj{Loqo~S|IvEHXww0jAdQ@kTSCSKfQlTlP^$HT?)s97fLK8JvBQrKt5x~YI07eEQa zFs#DtsK6MvF3-HCY+Q|wlv|c(=wZ!vXYdgJNFxUtma2}nIaPO4c69>=k6Z{NcW{Vr z!7=lj2Va`c@nBWCWDp?LLNJ7Gn5NWR0{eikw~!;nY-%np){+j*EkFF8yS?e#oq z!f(3HT1UfuU{;^|w%8O((#lk4EPl9#x7^(-XzG!cb?(Jb&2;9x3{$0{7jksYc2w0g zatY#v9NQSZrm-I%>#dcSF=@(UJ0~Ccv%^o*k$t+<46sJ}xV5Y|PWB_e=gIXhIbxKAzc%;K=h0}_~YHUcA z1aNO()R=ztQW<<~`k^DO%rOS(T|rbx@EK<3+EXPr>Vp=@KOllx$5_HGYSs*@S0uyZQ3+cLyO zLT46;BVFe!EKG%;Ql7So^`#e)x;&Rq?P2&+q8-p0xQ{B5Fd#K(eU@h9Ce1d~!B}^h zg~i(l*j@ioEkA!wJ(TM}^c8r;kccm2buZ38m%9>hOe~435=26T+P8*yDPb(6g{;d< zB|P+Kz`yY0vtcwYFeh3NHS`hkNowhC<1{4O5fyHBBPc;IEF!Bwo)OdU&5-BOK`6x{ zC1Hr%MI3IQc)LVv#}gcI-3yhZ*QV%>hwuH07YpT-$~2j^iUaY>dA%N~#_@@pEA30U zwYI!cmN0ZmQW`YasT_Moiicn`>BK#$_L;=K@q*=R50Cp5>{=(p0O2A}XauVF69qGrx& zh%Bj}VyxSHqvuW?L3S@VNth2)G>ZjO>vbgL+ULrm5kh}b8kN#hsC>P$?TUmIq~cVY zl||S635g!_cw1wEIj||_(cX6Yd_K?VsUcPX{RJ4Cd(ti$$ zdbf3uzKVc&zDeA&FUk;HaamV=%ZsdP_JOwHQ6{v>?|mD?q?cmXrfbe=w7C(a;B=Nu zV&G<;Oq!F^^H-JUe_7zi^VDu@bqe$`il5r7Uo< z?dIf*GKZwEU==MX(f5Xg+@KOy6~7*MQ2$g9v{h^38+p2V_|?j7G` z8mZ-}a0d*9X;Fgj66z78`z~uMztfnq79DHr$8`(9tFDC!`(!Om(sQPZoA*-!)1ScX zH-Sxky(&lI*OXEk-P#!~uOHD%WDSPu+T4;&eec7)>F%0|jD^V})s_4T7*9@7=26`k z{iqS;!kK8O@s&9ZZn9G<5$?-oWbOOOL-XvpH~q9|UZUK4Uczj~yCf-aI&OWpEotrT z_U-1wN-23i|6NKGnN`pgMnnFn;6mYYhKk;?X(zw=oAKTTiulwi`o-o5=ysx*J}vwlZN#d(g7vSKPCj8byq%>P5|eZFm65+h>U z`5Im9ipRgZmNiEe4<{r>Ecf$k0f-tnID{84qOl7tGa*>&syP}1;U8BXMo21V(^5;q z(PK_#%GFy;aSZnOd_6l`uJkY9=55qp=8u|7Qsj%N9vU{rqYPr?s`pa_4~RrhN|QkF zEt|qgo`PD5;!3B6+W49>N&+Xk!oH>o&zbFG#~pq`PlW=o5z*3roz8>hArFKb_)VUnN(vUJfI7MPtde&OMZ(MprFeOfsQ5f z#lw9vZ^S+)pX(&UKQ<}>#+9=<4J$ra{yo3*E#%7hg%Tjaf`V6D(sz~*QH6>I*O)lc zowphw)ybazZIPt%P6TL;$M)q!xJ{`Ug?^)Z0|Li!_c z{F&qaGp+h1^3O>B6Fr_E3;4r{|Iqw1H|b}8{Fj73H^+Z5;{RRm&y1I!%%xv~j`}>~ zfAg6BUF*-S{GV)?Uvi4}7jDcSihtMX|D`DZ^6&fpf1vzvv44)de_bX>+P~(a|E~Gx j= Date: Sat, 15 Nov 2025 18:20:52 +0100 Subject: [PATCH 07/38] Fixed 2 broken examples --- examples/2_New_examples/1.2.3.Bash_python_mix.yaml | 7 +++++-- .../2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml | 2 +- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/examples/2_New_examples/1.2.3.Bash_python_mix.yaml b/examples/2_New_examples/1.2.3.Bash_python_mix.yaml index 2c6a863..84666d8 100644 --- a/examples/2_New_examples/1.2.3.Bash_python_mix.yaml +++ b/examples/2_New_examples/1.2.3.Bash_python_mix.yaml @@ -1,7 +1,10 @@ # ------------------------------------------------------------------------------------- # Bash/Python Mix # ------------------------------------------------------------------------------------- -# A mixed pipeline that alternates Bash and Python tasks, both with significant delays (25-second sleep times each). It is used to test the coexistence and behavior of logs from different executors, as well as the management of sequential wait times. Total duration is approximately 50–60 seconds. +# Pipeline mista che alterna task Bash e Python, entrambe con ritardi significativi +# (sleep di 25 secondi ciascuna). Serve a testare la coesistenza e il comportamento +# dei log provenienti da esecutori differenti, oltre alla gestione di tempi di attesa +# in sequenza. Durata complessiva di circa 50 secondi. # ------------------------------------------------------------------------------------- dag: @@ -16,7 +19,7 @@ dag: - task_id: "bash_wait" type: "BashTask" params: - script: | + command: | echo "Bash waiting 25s..." sleep 25 dependencies: ["start"] diff --git a/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml b/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml index 8241843..e992ade 100644 --- a/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml +++ b/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml @@ -16,7 +16,7 @@ dag: - task_id: "bash_long" type: "BashTask" params: - script: | + command: | echo "Bash running long task..." sleep 45 dependencies: ["start"] From ceac3cbcb1b7caa3bde5c45f402bec979955eaba Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 15 Nov 2025 19:15:54 +0100 Subject: [PATCH 08/38] Fixed endpoints in api_client.py --- folder_structure.txt | 180 ++--------------------------- maestro.db | Bin 0 -> 45056 bytes src/maestro/Orchestrator_structure | 53 +++++++++ src/maestro/client/api_client.py | 52 ++++++--- 4 files changed, 96 insertions(+), 189 deletions(-) create mode 100644 maestro.db create mode 100644 src/maestro/Orchestrator_structure diff --git a/folder_structure.txt b/folder_structure.txt index bae86bf..f28cebc 100644 --- a/folder_structure.txt +++ b/folder_structure.txt @@ -1,56 +1,13 @@ . ├── CONCURRENT_EXECUTION.md +├── data.json ├── debug_task_execution.py -├── docs -│   ├── DAG_START_TIME.md -│   ├── maestro-icon.png -│   ├── maestro_icon.svg -│   └── open_issues.md -├── examples -│   ├── 1_Old_examples -│   │   ├── cron_examples_dag.yaml -│   │   ├── cron_scheduled_dag.yaml -│   │   ├── dag_id_generation_example.md -│   │   ├── long_sample_dag.yaml -│   │   ├── playbooks -│   │   │   ├── add_nat_rules.yml -│   │   │   ├── configure-vms.yml -│   │   │   ├── docker_install.yml -│   │   │   ├── install_openfaas.yml -│   │   │   ├── k3s_install.yml -│   │   │   └── remove_nat_rules.yml -│   │   ├── sample_dag2.yaml -│   │   ├── sample_dag.yaml -│   │   ├── scheduled_dag.yaml -│   │   ├── simple_test_dag.yaml -│   │   ├── templates -│   │   │   ├── inventory-nat-rules.ini.tpl -│   │   │   └── ssh_inventory.ini.j2 -│   │   ├── terraform -│   │   │   ├── maestro.db -│   │   │   ├── main.tf -│   │   │   ├── terraform.tfstate -│   │   │   ├── terraform.tfstate.backup -│   │   │   ├── terraform.tfvars.example -│   │   │   ├── tfplan -│   │   │   ├── variables.tf -│   │   │   ├── vm_creation_flow.md -│   │   │   └── wait_for_vm.sh -│   │   └── test_dag.yaml -│   └── 2_New_examples -│   ├── 1_basic_linear.yaml -│   ├── 2_bash_chain.yaml -│   ├── 3_wait_and_retry.yaml -│   ├── 4_conditional_branching.yaml -│   ├── 5_cron_heartbeat.yaml -│   ├── 6_scheduled_greeting.yaml -│   ├── 7_mixed_execution.yaml -│   └── DAG explanation.ods +├── docs (contiene immagini e contenuti descrittivi) +├── examples (contiene gli esempi di DAG per provare l'orchestratore) +├── folder_structure.txt ├── Initial setting.txt ├── maestro.db -├── pictures -│   ├── Screenshot 2025-07-14 at 15.54.28.png -│   └── Screenshot 2025-07-18 at 19.15.10.png +├── pictures (contiene solo immagini di prova) ├── pyproject.toml ├── pytest.ini ├── README.md @@ -61,35 +18,19 @@ │   │   ├── client │   │   │   ├── api_client.py │   │   │   ├── cli.py -│   │   │   ├── __init__.py -│   │   │   └── __pycache__ -│   │   │   ├── api_client.cpython-312.pyc -│   │   │   ├── cli.cpython-312.pyc -│   │   │   └── __init__.cpython-312.pyc +│   │   │   └── __init__.py │   │   ├── __init__.py -│   │   ├── __pycache__ -│   │   │   └── __init__.cpython-312.pyc │   │   ├── server │   │   │   ├── api │   │   │   │   ├── __init__.py -│   │   │   │   ├── __pycache__ -│   │   │   │   │   └── __init__.cpython-312.pyc │   │   │   │   └── v1 │   │   │   │   ├── __init__.py -│   │   │   │   ├── __pycache__ -│   │   │   │   │   ├── __init__.cpython-312.pyc -│   │   │   │   │   └── router.cpython-312.pyc │   │   │   │   ├── router.py │   │   │   │   └── routes │   │   │   │   ├── dags.py │   │   │   │   ├── health.py │   │   │   │   ├── __init__.py -│   │   │   │   ├── logs.py -│   │   │   │   └── __pycache__ -│   │   │   │   ├── dags.cpython-312.pyc -│   │   │   │   ├── health.cpython-312.pyc -│   │   │   │   ├── __init__.cpython-312.pyc -│   │   │   │   └── logs.cpython-312.pyc +│   │   │   │   └── logs.py │   │   │   ├── app.py │   │   │   ├── __init__.py │   │   │   ├── internals @@ -101,61 +42,29 @@ │   │   │   │   │   ├── __init__.py │   │   │   │   │   ├── kubernetes.py │   │   │   │   │   ├── local.py -│   │   │   │   │   ├── __pycache__ -│   │   │   │   │   │   ├── base.cpython-312.pyc -│   │   │   │   │   │   ├── docker.cpython-312.pyc -│   │   │   │   │   │   ├── factory.cpython-312.pyc -│   │   │   │   │   │   ├── __init__.cpython-312.pyc -│   │   │   │   │   │   ├── kubernetes.cpython-312.pyc -│   │   │   │   │   │   ├── local.cpython-312.pyc -│   │   │   │   │   │   └── ssh.cpython-312.pyc │   │   │   │   │   └── ssh.py │   │   │   │   ├── __init__.py │   │   │   │   ├── models.py │   │   │   │   ├── orchestrator.py -│   │   │   │   ├── __pycache__ -│   │   │   │   │   ├── dag_loader.cpython-312.pyc -│   │   │   │   │   ├── __init__.cpython-312.pyc -│   │   │   │   │   ├── models.cpython-312.pyc -│   │   │   │   │   ├── orchestrator.cpython-312.pyc -│   │   │   │   │   ├── status_manager.cpython-312.pyc -│   │   │   │   │   └── task_registry.cpython-312.pyc │   │   │   │   ├── status_manager.py │   │   │   │   └── task_registry.py -│   │   │   ├── __pycache__ -│   │   │   │   ├── app.cpython-312.pyc -│   │   │   │   └── __init__.cpython-312.pyc │   │   │   ├── services │   │   │   │   ├── __init__.py -│   │   │   │   ├── __pycache__ -│   │   │   │   │   ├── __init__.cpython-312.pyc -│   │   │   │   │   └── scheduler_service.cpython-312.pyc │   │   │   │   └── scheduler_service.py │   │   │   └── tasks │   │   │   ├── ansible_task.py │   │   │   ├── base.py +│   │   │   ├── bash_task.py │   │   │   ├── extended_terraform_task.py │   │   │   ├── file_writer_task.py │   │   │   ├── __init__.py │   │   │   ├── print_task.py -│   │   │   ├── __pycache__ -│   │   │   │   ├── ansible_task.cpython-312.pyc -│   │   │   │   ├── base.cpython-312.pyc -│   │   │   │   ├── extended_terraform_task.cpython-312.pyc -│   │   │   │   ├── file_writer_task.cpython-312.pyc -│   │   │   │   ├── __init__.cpython-312.pyc -│   │   │   │   ├── print_task.cpython-312.pyc -│   │   │   │   ├── terraform_task.cpython-312.pyc -│   │   │   │   └── wait_task.cpython-312.pyc +│   │   │   ├── python_task.py │   │   │   ├── terraform_task.py │   │   │   └── wait_task.py │   │   └── shared │   │   ├── dag.py │   │   ├── __init__.py -│   │   ├── __pycache__ -│   │   │   ├── dag.cpython-312.pyc -│   │   │   ├── __init__.cpython-312.pyc -│   │   │   └── task.cpython-312.pyc │   │   └── task.py │   └── maestro.egg-info │   ├── dependency_links.txt @@ -164,77 +73,8 @@ │   ├── requires.txt │   ├── SOURCES.txt │   └── top_level.txt -├── struttura.txt ├── test_dag_status.py ├── test_run_dag.py -├── tests -│   ├── conftest.py -│   ├── integration -│   │   ├── conftest.py -│   │   ├── __pycache__ -│   │   │   ├── conftest.cpython-312-pytest-7.4.4.pyc -│   │   │   ├── conftest.cpython-312-pytest-8.4.1.pyc -│   │   │   ├── test_cli_rest_integration.cpython-312-pytest-7.4.4.pyc -│   │   │   ├── test_cli_rest_integration.cpython-312-pytest-8.4.1.pyc -│   │   │   ├── test_debug_server.cpython-312-pytest-7.4.4.pyc -│   │   │   ├── test_debug_server.cpython-312-pytest-8.4.1.pyc -│   │   │   ├── test_server_api_integration.cpython-312-pytest-7.4.4.pyc -│   │   │   └── test_server_api_integration.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_cli_rest_integration.py -│   │   ├── test_debug_server.py -│   │   └── test_server_api_integration.py -│   ├── maestro.db -│   ├── __pycache__ -│   │   ├── conftest.cpython-312-pytest-7.4.4.pyc -│   │   ├── conftest.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_ansible_task.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_ansible_task.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_ansible_task.cpython-312-pytest-8.4.1.pyc.8528 -│   │   ├── test_api_client.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_api_client.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_cli_attach.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_cli_attach.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_cli_attach.cpython-312-pytest-8.4.1.pyc.8528 -│   │   ├── test_cli_client.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_cli_client.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_cli_integration.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_cli_integration.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_cron_feature.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_cron_feature.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_dag.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_dag.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_dag_id_generation.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_dag_id_generation.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_db_feature.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_db_feature.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_enhanced_cli.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_enhanced_cli.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_extended_terraform_task.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_extended_terraform_task.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_multi_executor.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_multi_executor.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_orchestrator_dagloader.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_orchestrator_dagloader.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_server.cpython-312-pytest-7.4.4.pyc -│   │   ├── test_server.cpython-312-pytest-8.4.1.pyc -│   │   ├── test_status_manager.cpython-312-pytest-7.4.4.pyc -│   │   └── test_status_manager.cpython-312-pytest-8.4.1.pyc -│   ├── test_ansible_task.py -│   ├── test_api_client.py -│   ├── test_cli_attach.py -│   ├── test_cli_client.py -│   ├── test_cli_integration.py -│   ├── test_cron_feature.py -│   ├── test_dag_id_generation.py -│   ├── test_dag.py -│   ├── test_db_feature.py -│   ├── test_enhanced_cli.py -│   ├── test_extended_terraform_task.py -│   ├── test_multi_executor.py -│   ├── test_orchestrator_dagloader.py -│   ├── test_server.py -│   └── test_status_manager.py +├── tests (contiene i test automatici che occorrono a testare l'orchestratore) ├── TEST_SUITE.md └── uv.lock - -37 directories, 201 files diff --git a/maestro.db b/maestro.db new file mode 100644 index 0000000000000000000000000000000000000000..e76510686f7ff8614f312f3432112f558f37d293 GIT binary patch literal 45056 zcmeI5&2JmW6~LF2D2kLs8Z89EHG-_A0uf=8+}RH+s_-h_xa5CEb0pC8? z`EjQSE_wgcspvB<96IEpuSNed^{2?cCcg^*ck*F)Jp4Nd!~+Q+0VIF~kN^@u0^bG# z^$X!(Y$f5VlR`OfJ}5lUjjEkHF!P5+)6zez?Am(Sd{EP^!?Ip0luU2z#Lczr!bX-~ zTz)(IKL6b0{K_)#y#+tlkwd(m91q443E$Ib&Dbq=<>*ZwYDloLaAPUk*)&69Xb5Ptf2EEtQ$e81RYC9WCv zN02#u1Ut&cg7`UifjT718~b{@tQ}G212cD6D^$wIT|f(aXg4`g1Q24?%)AbL@wY+4 z8;f_dG&ol&Rf}eih}wZ=8hO|K>D7g@ZCWhfij_AlFq4h!?d%$j@^HM$JK|sE-Ll5% zmA6*bvWvHuDe*byqBy^ny_H?dF5k?q({wh|g!kHx8@W@oBAj&Lcsw{3yB<4EgN(aQ z*w^^enLsc$J?;DX8%$Tl%Dzqi#yT`bLmG-|l(<71or{p<;Y)GD3bJ{#=-)(#)czF&~}R-rmQ-(>rQBaM#5ZeKpm*xDp?(moH= zQ$A*PPZt5olg`!zCY6otc%0XTB)bW#z$ZId* z_#6JQ*tO~7ib+M)3H=;_W_Fx5wNvot_pfuIuiyg@Bye^DpRM|X@h`8}6VkoIg7wjn zUOKe$mE6I@8EW{2{LDN*(-|{!m7ifPAM5QveEG0ggs_|iZD+G%)DC8C0{ndFBw}X` zpAF}YKuF%KXF9EDG$wUWDVeFfao^N)Rz9_URL-T|UB6k$o2ilk?%S%Q%m?7Zi>93t z^=0#8z4enM#Ux2CRm%JN$3~%MCy$I$k;jD-e7j7qN!mrztj>u7{H9p!oK>jSnkLjxNQ>`L?VwWb04(+b zES@J|s^9YgEWQZ9QZK;L`2rTt39vM5z_hPoQ(xKPBO0OS=p@Xl^g=&edS2C-GoW5@qD+Sx# zp1OjW)1s}!%iNesNh!H%SVpmE7WD(uxPPSY8uo!+We)Yc8rK`U+7UyqMgzipp>1!eWRzJ?&&(ZE`0be+pxvgka{4Z`55gMW z?J{d}d0BB<+-BoSleC&v$1wC>WZhBU3brToYgSqP@c{1Z*(ihbY@h`&ih9Yk_GzEW z+c>nMUK;!zwaslHFQOylX+#|VLCxFMl$^qQoUtW2g%|L&EaJQ*hv9k8AN>ax{SSQL zfdr5M5n8q)bLl$T>+#7^<312(l_@yL(1j*b`h&>&MALXY!M( zMzpl*kwuypl_U{GMIb$8apwO&=c0d(e*P+}7-dHSNB{{S0VIF~kN^@u0!RP}AOR%s zsu55}H{-tk2P*v1=;myw&k+SW|3BgXF&DiT`F-U3Q=bLC41`CYj655;=l=$R@jwD+ zB2W)bgt++T#p%1!70 z3POD}OeKj;FAgFJT^TAlg0$9r@9=CHxqA`r31VavS=9u&J{mfs7-DZR+>5Qd#30=? z#?L4Q=`DtPQN2eDXrzvXHG<(wTjDh-UE;fs zOIt#z8S9)&a*5uPUP=StgOs-DS#!ZA4cK8(S!V|<*r@_WuIY-?Hb)VXFioW#j0!D%C+|OfE}^!n^uag%1g1;f%G(4J-W0F zr7LxW&=NzI*PN(%DojHJNlOdruqBu!s*@9Zx3fM$B@Xr`9%u0^FsBNXl}>cci~9Sf zL_8_zZBH$dC+)yLpCiW>At|DyL|7-!=3|J)tZq0%(c2&dOIOlyhl$(wlX>ZZM+lRRv z?8UYRI|dykPqws%CS*a9l%aFzOBHlRP!Wc&3az0Et%oYSwM7(8QCl{^N%G7&vi*iy z^N$_aPo~Z{uq#ATGvsV#hRu6S3s$1rsfq>rWx|e4My+DCytp^8ZI9HGxE;AlWHEh4 z1sp5Ptfa~fMZig0i0&p|ZZ`SNKI(Tzoj!68%mGss7sV1mlnEvz*Xf2{$*e2z!L4^YUJplRgSF zHm&z9GutA~+rd0>&ywehGc{S@Lg9aNksnR{E&9{Qvx#q}wkAHD{5u5Ufdr5M5?)+p^NcNcI3{WteAZ@7>sn5fxEO zk3J5DVhsW4WP&)n{OP!QQc=>XqIKm@ASRQRG}-?s(9D0FJxB~Qe|X;+9_|%22qG=ID==en`;87Rq&KwnkQOP14GPY ze&P$ooctYo8)n`aLDWRilQ)wpL(&4#z|dw+HnzHNG&JjoNBu!_2g4x=$=5r>8IlZ+l87LvUatf)$Vp8G`2-k1;keM$*Ie|M(ZJN#(Oc0! zME)50G;$|GUktzl2_OL^fCP{L5+W@>mJwEtn2z*_+r3OK_sz^HSB!Jpwr-VDH4jc?F#8ow5Rp_ax! z4{iU0nA5zv${zDCyrkxgVj%$}fCP{L5a06zb}teQc&kN^@u0!RP}AOR$R1dsp{Kmter J2@H?G{{RZ0;hz8i literal 0 HcmV?d00001 diff --git a/src/maestro/Orchestrator_structure b/src/maestro/Orchestrator_structure new file mode 100644 index 0000000..0c8ac3d --- /dev/null +++ b/src/maestro/Orchestrator_structure @@ -0,0 +1,53 @@ + +. +├── client +│   ├── api_client.py +│   ├── cli.py +│   └── __init__.py +├── __init__.py +├── server +│   ├── api +│   │   ├── __init__.py +│   │   └── v1 +│   │   ├── __init__.py +│   │   ├── router.py +│   │   └── routes +│   │   ├── dags.py +│   │   ├── health.py +│   │   ├── __init__.py +│   │   └── logs.py +│   ├── app.py +│   ├── __init__.py +│   ├── internals +│   │   ├── dag_loader.py +│   │   ├── executors +│   │   │   ├── base.py +│   │   │   ├── docker.py +│   │   │   ├── factory.py +│   │   │   ├── __init__.py +│   │   │   ├── kubernetes.py +│   │   │   ├── local.py +│   │   │   └── ssh.py +│   │   ├── __init__.py +│   │   ├── models.py +│   │   ├── orchestrator.py +│   │   ├── status_manager.py +│   │   └── task_registry.py +│   ├── services +│   │   ├── __init__.py +│   │   └── scheduler_service.py +│   └── tasks +│   ├── ansible_task.py +│   ├── base.py +│   ├── bash_task.py +│   ├── extended_terraform_task.py +│   ├── file_writer_task.py +│   ├── __init__.py +│   ├── print_task.py +│   ├── python_task.py +│   ├── terraform_task.py +│   └── wait_task.py +└── shared + ├── dag.py + ├── __init__.py + └── task.py diff --git a/src/maestro/client/api_client.py b/src/maestro/client/api_client.py index 63884c9..e6b3b95 100644 --- a/src/maestro/client/api_client.py +++ b/src/maestro/client/api_client.py @@ -94,42 +94,56 @@ def get_dag_logs_v1(self, response = self._make_request("GET", endpoint, params=params) return response.json() - # TODO: refactor, this is used only for testing - def stream_dag_logs_v1(self, dag_id: str, execution_id: Optional[str] = None, - task_filter: Optional[str] = None, - level_filter: Optional[str] = None) -> Iterator[Dict[str, Any]]: - """Stream logs for a specific DAG execution in real-time using the v1 API""" - endpoint = f"/v1/dags/{dag_id}/attach" + + def stream_dag_logs_v1( + self, + dag_id: str, + execution_id: Optional[str] = None, + task_filter: Optional[str] = None, + level_filter: Optional[str] = None + ) -> Iterator[Dict[str, Any]]: + """ + Stream logs for a specific DAG execution in real-time using the v1 API. + FIXED VERSION — points to the correct server endpoint. + """ + + # Correct endpoint for streaming (SSE) + endpoint = f"/api/v1/logs/{dag_id}/attach" + params = {} - if execution_id: params["execution_id"] = execution_id if task_filter: params["task_filter"] = task_filter if level_filter: params["level_filter"] = level_filter - + + # Build full URL url = f"{self.base_url}{endpoint}" if params: url += "?" + urllib.parse.urlencode(params) - + try: with self.session.get(url, stream=True, timeout=None) as response: response.raise_for_status() - + for line in response.iter_lines(): - if line: - line = line.decode('utf-8') - if line.startswith('data: '): - data = line[6:] # Remove 'data: ' prefix - try: - yield json.loads(data) - except json.JSONDecodeError: - continue + if not line: + continue + + line = line.decode("utf-8").strip() + if line.startswith("data: "): + payload = line[len("data: "):] + try: + yield json.loads(payload) + except json.JSONDecodeError: + continue + except requests.exceptions.ConnectionError: raise ConnectionError(f"Could not connect to Maestro server at {self.base_url}") - except requests.exceptions.HTTPError as e: + except requests.exceptions.HTTPError: raise RuntimeError(f"HTTP {response.status_code}: {response.text}") + def get_running_dags(self) -> Dict[str, Any]: """Get all currently running DAGs""" From 08a3150a7fa92c3725a20e7adf538684b45ddfa7 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 15 Nov 2025 19:30:08 +0100 Subject: [PATCH 09/38] Fixed again api_client.py and now 'maestro attach...' works... but it also works very bad. Need to improve it. --- maestro.db | Bin 45056 -> 0 bytes src/maestro/client/api_client.py | 14 +++++++------- 2 files changed, 7 insertions(+), 7 deletions(-) delete mode 100644 maestro.db diff --git a/maestro.db b/maestro.db deleted file mode 100644 index e76510686f7ff8614f312f3432112f558f37d293..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 45056 zcmeI5&2JmW6~LF2D2kLs8Z89EHG-_A0uf=8+}RH+s_-h_xa5CEb0pC8? z`EjQSE_wgcspvB<96IEpuSNed^{2?cCcg^*ck*F)Jp4Nd!~+Q+0VIF~kN^@u0^bG# z^$X!(Y$f5VlR`OfJ}5lUjjEkHF!P5+)6zez?Am(Sd{EP^!?Ip0luU2z#Lczr!bX-~ zTz)(IKL6b0{K_)#y#+tlkwd(m91q443E$Ib&Dbq=<>*ZwYDloLaAPUk*)&69Xb5Ptf2EEtQ$e81RYC9WCv zN02#u1Ut&cg7`UifjT718~b{@tQ}G212cD6D^$wIT|f(aXg4`g1Q24?%)AbL@wY+4 z8;f_dG&ol&Rf}eih}wZ=8hO|K>D7g@ZCWhfij_AlFq4h!?d%$j@^HM$JK|sE-Ll5% zmA6*bvWvHuDe*byqBy^ny_H?dF5k?q({wh|g!kHx8@W@oBAj&Lcsw{3yB<4EgN(aQ z*w^^enLsc$J?;DX8%$Tl%Dzqi#yT`bLmG-|l(<71or{p<;Y)GD3bJ{#=-)(#)czF&~}R-rmQ-(>rQBaM#5ZeKpm*xDp?(moH= zQ$A*PPZt5olg`!zCY6otc%0XTB)bW#z$ZId* z_#6JQ*tO~7ib+M)3H=;_W_Fx5wNvot_pfuIuiyg@Bye^DpRM|X@h`8}6VkoIg7wjn zUOKe$mE6I@8EW{2{LDN*(-|{!m7ifPAM5QveEG0ggs_|iZD+G%)DC8C0{ndFBw}X` zpAF}YKuF%KXF9EDG$wUWDVeFfao^N)Rz9_URL-T|UB6k$o2ilk?%S%Q%m?7Zi>93t z^=0#8z4enM#Ux2CRm%JN$3~%MCy$I$k;jD-e7j7qN!mrztj>u7{H9p!oK>jSnkLjxNQ>`L?VwWb04(+b zES@J|s^9YgEWQZ9QZK;L`2rTt39vM5z_hPoQ(xKPBO0OS=p@Xl^g=&edS2C-GoW5@qD+Sx# zp1OjW)1s}!%iNesNh!H%SVpmE7WD(uxPPSY8uo!+We)Yc8rK`U+7UyqMgzipp>1!eWRzJ?&&(ZE`0be+pxvgka{4Z`55gMW z?J{d}d0BB<+-BoSleC&v$1wC>WZhBU3brToYgSqP@c{1Z*(ihbY@h`&ih9Yk_GzEW z+c>nMUK;!zwaslHFQOylX+#|VLCxFMl$^qQoUtW2g%|L&EaJQ*hv9k8AN>ax{SSQL zfdr5M5n8q)bLl$T>+#7^<312(l_@yL(1j*b`h&>&MALXY!M( zMzpl*kwuypl_U{GMIb$8apwO&=c0d(e*P+}7-dHSNB{{S0VIF~kN^@u0!RP}AOR%s zsu55}H{-tk2P*v1=;myw&k+SW|3BgXF&DiT`F-U3Q=bLC41`CYj655;=l=$R@jwD+ zB2W)bgt++T#p%1!70 z3POD}OeKj;FAgFJT^TAlg0$9r@9=CHxqA`r31VavS=9u&J{mfs7-DZR+>5Qd#30=? z#?L4Q=`DtPQN2eDXrzvXHG<(wTjDh-UE;fs zOIt#z8S9)&a*5uPUP=StgOs-DS#!ZA4cK8(S!V|<*r@_WuIY-?Hb)VXFioW#j0!D%C+|OfE}^!n^uag%1g1;f%G(4J-W0F zr7LxW&=NzI*PN(%DojHJNlOdruqBu!s*@9Zx3fM$B@Xr`9%u0^FsBNXl}>cci~9Sf zL_8_zZBH$dC+)yLpCiW>At|DyL|7-!=3|J)tZq0%(c2&dOIOlyhl$(wlX>ZZM+lRRv z?8UYRI|dykPqws%CS*a9l%aFzOBHlRP!Wc&3az0Et%oYSwM7(8QCl{^N%G7&vi*iy z^N$_aPo~Z{uq#ATGvsV#hRu6S3s$1rsfq>rWx|e4My+DCytp^8ZI9HGxE;AlWHEh4 z1sp5Ptfa~fMZig0i0&p|ZZ`SNKI(Tzoj!68%mGss7sV1mlnEvz*Xf2{$*e2z!L4^YUJplRgSF zHm&z9GutA~+rd0>&ywehGc{S@Lg9aNksnR{E&9{Qvx#q}wkAHD{5u5Ufdr5M5?)+p^NcNcI3{WteAZ@7>sn5fxEO zk3J5DVhsW4WP&)n{OP!QQc=>XqIKm@ASRQRG}-?s(9D0FJxB~Qe|X;+9_|%22qG=ID==en`;87Rq&KwnkQOP14GPY ze&P$ooctYo8)n`aLDWRilQ)wpL(&4#z|dw+HnzHNG&JjoNBu!_2g4x=$=5r>8IlZ+l87LvUatf)$Vp8G`2-k1;keM$*Ie|M(ZJN#(Oc0! zME)50G;$|GUktzl2_OL^fCP{L5+W@>mJwEtn2z*_+r3OK_sz^HSB!Jpwr-VDH4jc?F#8ow5Rp_ax! z4{iU0nA5zv${zDCyrkxgVj%$}fCP{L5a06zb}teQc&kN^@u0!RP}AOR$R1dsp{Kmter J2@H?G{{RZ0;hz8i diff --git a/src/maestro/client/api_client.py b/src/maestro/client/api_client.py index e6b3b95..81a2b31 100644 --- a/src/maestro/client/api_client.py +++ b/src/maestro/client/api_client.py @@ -95,6 +95,7 @@ def get_dag_logs_v1(self, return response.json() + # TODO: refactor, this is used only for testing def stream_dag_logs_v1( self, dag_id: str, @@ -104,11 +105,11 @@ def stream_dag_logs_v1( ) -> Iterator[Dict[str, Any]]: """ Stream logs for a specific DAG execution in real-time using the v1 API. - FIXED VERSION — points to the correct server endpoint. + FIX: punta all'endpoint /v1/logs/{dag_id}/attach esposto dal server. """ - # Correct endpoint for streaming (SSE) - endpoint = f"/api/v1/logs/{dag_id}/attach" + # 🚩 QUI il vero fix: usare "logs" e non "dags", e mantenere il prefisso /v1 + endpoint = f"/v1/logs/{dag_id}/attach" params = {} if execution_id: @@ -118,7 +119,6 @@ def stream_dag_logs_v1( if level_filter: params["level_filter"] = level_filter - # Build full URL url = f"{self.base_url}{endpoint}" if params: url += "?" + urllib.parse.urlencode(params) @@ -133,15 +133,15 @@ def stream_dag_logs_v1( line = line.decode("utf-8").strip() if line.startswith("data: "): - payload = line[len("data: "):] + data = line[len("data: "):] try: - yield json.loads(payload) + yield json.loads(data) except json.JSONDecodeError: continue except requests.exceptions.ConnectionError: raise ConnectionError(f"Could not connect to Maestro server at {self.base_url}") - except requests.exceptions.HTTPError: + except requests.exceptions.HTTPError as e: raise RuntimeError(f"HTTP {response.status_code}: {response.text}") From 8c18dc12795951db43514dbc2aef7860f653c94f Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sun, 16 Nov 2025 15:22:49 +0100 Subject: [PATCH 10/38] Patches on server components --- examples/2_New_examples/1.2.1.Long_waits.yaml | 3 ++ src/maestro/server/api/v1/routes/logs.py | 52 +++++++++---------- src/maestro/server/internals/models.py | 3 +- src/maestro/server/internals/orchestrator.py | 16 +++--- .../server/internals/status_manager.py | 24 +++++++-- 5 files changed, 60 insertions(+), 38 deletions(-) diff --git a/examples/2_New_examples/1.2.1.Long_waits.yaml b/examples/2_New_examples/1.2.1.Long_waits.yaml index b52c355..6bf03a5 100644 --- a/examples/2_New_examples/1.2.1.Long_waits.yaml +++ b/examples/2_New_examples/1.2.1.Long_waits.yaml @@ -20,6 +20,7 @@ dag: import time print("Step 1 running...") time.sleep(20) + print("Step 1 - Complete successfully!") dependencies: ["start"] - task_id: "step2" @@ -29,6 +30,7 @@ dag: import time print("Step 2 running...") time.sleep(20) + print("Step 2 - Complete successfully!") dependencies: ["step1"] - task_id: "step3" @@ -38,6 +40,7 @@ dag: import time print("Step 3 running...") time.sleep(20) + print("Step 3 - Complete successfully!") dependencies: ["step2"] - task_id: "finish" diff --git a/src/maestro/server/api/v1/routes/logs.py b/src/maestro/server/api/v1/routes/logs.py index afca974..67827d9 100644 --- a/src/maestro/server/api/v1/routes/logs.py +++ b/src/maestro/server/api/v1/routes/logs.py @@ -161,57 +161,57 @@ async def log_streamer(): @router.get("/{dag_id}/attach") async def attach_dag_logs( - dag_id: str, - execution_id: Optional[str] = None, + dag_id: str, + execution_id: Optional[str] = None, task_filter: Optional[str] = Query(None, description="Filter logs by task ID"), level_filter: Optional[str] = Query(None, description="Filter logs by level (INFO, WARNING, ERROR)"), orchestrator: Orchestrator = Depends(get_orchestrator) ): """ - Attaches to the live log stream of a DAG execution (alias for stream). - Compatible with Docker-style attach API. + Minimal patch: stabilizes streaming by fixing ordering & cursor management. + No DB changes required. """ async def log_streamer(): last_timestamp = None displayed_logs = set() - + while True: try: with orchestrator.status_manager as sm: - logs = sm.get_execution_logs(dag_id, execution_id, limit=100) - + logs = sm.get_execution_logs(dag_id, execution_id, limit=200) + + # 👉 SORT logs OLDEST → NEWEST (important!) + logs = sorted(logs, key=lambda x: x["timestamp"]) + # Apply filters if task_filter: logs = [log for log in logs if log["task_id"] == task_filter] if level_filter: logs = [log for log in logs if log["level"].upper() == level_filter.upper()] - - # Filter for new logs only + new_logs = [] + for log in logs: - log_id = f"{log['timestamp']}_{log['task_id']}_{log['level']}_{log['message'][:50]}" - - if log_id not in displayed_logs: - if last_timestamp is None or log["timestamp"] > last_timestamp: + # Build unique log key + log_key = f"{log['timestamp']}_{log['task_id']}_{log['level']}_{log['message'][:50]}" + + # 👉 Ensure ordering works: if timestamp is new or unseen, stream it + if log_key not in displayed_logs: + if last_timestamp is None or log["timestamp"] >= last_timestamp: new_logs.append(log) - displayed_logs.add(log_id) + displayed_logs.add(log_key) last_timestamp = log["timestamp"] - - # Send new logs in simple JSON format - for log in reversed(new_logs): + + # 👉 Stream logs in natural order (no reversed!) + for log in new_logs: yield f"data: {json.dumps(log)}\n\n" - - # Clean up displayed_logs set to prevent memory issues - if len(displayed_logs) > 1000: - displayed_logs.clear() - last_timestamp = None - - await asyncio.sleep(1) - + + await asyncio.sleep(0.5) + except Exception as e: yield f"data: {json.dumps({'error': str(e)})}\n\n" break - + return StreamingResponse( log_streamer(), media_type="text/event-stream", diff --git a/src/maestro/server/internals/models.py b/src/maestro/server/internals/models.py index f75e361..c7f2f9c 100644 --- a/src/maestro/server/internals/models.py +++ b/src/maestro/server/internals/models.py @@ -3,6 +3,7 @@ from sqlalchemy.orm import sessionmaker, relationship, declarative_base from sqlalchemy.sql import func import threading +from datetime import datetime Base = declarative_base() @@ -50,7 +51,7 @@ class LogORM(Base): task_id = Column(String) level = Column(String) message = Column(Text) - timestamp = Column(DateTime, server_default=func.now()) + timestamp = Column(DateTime, default=datetime.utcnow) thread_id = Column(String) def create_db_engine(db_path='maestro.db'): diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index 044d366..b5b4557 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -36,22 +36,26 @@ def set_context(self, dag_id: str, execution_id: str, task_id: str = None): self.current_task_id = task_id def emit(self, record): - """Emit a log record to the database.""" + """Emit a log record to the database, preserving the real timestamp.""" if self.current_dag_id and self.current_execution_id: try: - log_entry = self.format(record) - # Use the provided StatusManager instance + # Timestamp reale del LogRecord + log_timestamp = datetime.fromtimestamp(record.created) + + # Messaggio grezzo (senza metadata RichHandler) + log_message = record.getMessage() + with self.status_manager as sm: sm.log_message( dag_id=self.current_dag_id, execution_id=self.current_execution_id, task_id=self.current_task_id or "system", level=record.levelname, - message=log_entry + message=log_message, + timestamp=log_timestamp ) except Exception: - # Don't let logging errors break the application - pass + pass # Non deve mai rompere l'esecuzione class Orchestrator: diff --git a/src/maestro/server/internals/status_manager.py b/src/maestro/server/internals/status_manager.py index 2dd4723..2c016d7 100644 --- a/src/maestro/server/internals/status_manager.py +++ b/src/maestro/server/internals/status_manager.py @@ -314,19 +314,33 @@ def get_dag_execution_details(self, dag_id: str, execution_id: str = None) -> Di ] } - def log_message(self, dag_id: str, execution_id: str, task_id: str, level: str, message: str): + def log_message(self, dag_id: str, execution_id: str, task_id: str, + level: str, message: str, + timestamp: Optional[datetime] = None): + """Store a log message with an optional externally provided timestamp.""" with self.Session.begin() as session: - log = LogORM(dag_id=dag_id, execution_id=execution_id, task_id=task_id, level=level, message=message, thread_id=str(threading.current_thread().ident)) + log = LogORM( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id, + level=level, + message=message, + timestamp=timestamp or datetime.now(), + thread_id=str(threading.current_thread().ident) + ) session.add(log) - # 🆕 Metodo helper per aggiungere log in modo più comodo - def add_log(self, dag_id: str, execution_id: str, task_id: str, message: str, level: str = "INFO"): + def add_log(self, dag_id: str, execution_id: str, task_id: str, + message: str, level: str = "INFO", + timestamp: Optional[datetime] = None): + """Helper wrapper for log_message() with timestamp support.""" self.log_message( dag_id=dag_id, execution_id=execution_id, task_id=task_id, level=level, - message=message + message=message, + timestamp=timestamp ) def get_execution_logs(self, dag_id: str, execution_id: str = None, limit: int = 100) -> List[Dict[str, Any]]: From 961439f4506427ae52dedb9cd79ab1f540e368d2 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sun, 16 Nov 2025 16:13:16 +0100 Subject: [PATCH 11/38] Fixed Python messages timestamp on live log streaming --- src/maestro/server/internals/models.py | 8 +- src/maestro/server/internals/orchestrator.py | 16 +++- src/maestro/server/tasks/python_task.py | 98 ++++++++++++++------ 3 files changed, 91 insertions(+), 31 deletions(-) diff --git a/src/maestro/server/internals/models.py b/src/maestro/server/internals/models.py index c7f2f9c..a3588ce 100644 --- a/src/maestro/server/internals/models.py +++ b/src/maestro/server/internals/models.py @@ -3,7 +3,7 @@ from sqlalchemy.orm import sessionmaker, relationship, declarative_base from sqlalchemy.sql import func import threading -from datetime import datetime +import datetime Base = declarative_base() @@ -51,7 +51,11 @@ class LogORM(Base): task_id = Column(String) level = Column(String) message = Column(Text) - timestamp = Column(DateTime, default=datetime.utcnow) + timestamp = Column( + DateTime, + default=datetime.datetime.utcnow, # ✔ corretto + nullable=False + ) thread_id = Column(String) def create_db_engine(db_path='maestro.db'): diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index b5b4557..8041437 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -287,7 +287,7 @@ def run_dag( 'maestro.server.tasks.extended_terraform_task', 'maestro.server.tasks.print_task', 'maestro.server.tasks.python_task', - 'maestro.server.tasks.bash_task', # ✅ eccolo + 'maestro.server.tasks.bash_task', 'maestro.core.executors.ssh', 'maestro.core.executors.docker', 'maestro.core.executors.local', @@ -295,8 +295,20 @@ def run_dag( for logger_name in task_loggers: logger = logging.getLogger(logger_name) + + # Pulisce eventuali handler duplicati + logger.handlers.clear() + + # Aggancia DB handler logger.addHandler(db_handler) - logger.propagate = True + + # Aggancia anche RichHandler (console) che sta sul root + for h in logging.getLogger().handlers: + if isinstance(h, RichHandler): + logger.addHandler(h) + + logger.setLevel(logging.INFO) + logger.propagate = False try: # Use concurrent execution diff --git a/src/maestro/server/tasks/python_task.py b/src/maestro/server/tasks/python_task.py index b4ee339..3ecea62 100644 --- a/src/maestro/server/tasks/python_task.py +++ b/src/maestro/server/tasks/python_task.py @@ -3,6 +3,7 @@ import io import contextlib import threading +import sys from maestro.server.tasks.base import BaseTask from maestro.server.internals.status_manager import StatusManager @@ -19,30 +20,73 @@ class PythonTask(BaseTask): script_path: Optional[str] = Field(default=None, description="Path to a .py file to execute") def execute_local(self): - buffer = io.StringIO() - - with contextlib.redirect_stdout(buffer), contextlib.redirect_stderr(buffer): - if self.code: - exec(self.code, {}) - elif self.script_path: - with open(self.script_path, "r") as f: - code = f.read() - exec(code, {}) - else: - raise ValueError("PythonTask requires either 'code' or 'script_path'.") - - output = buffer.getvalue().strip() - - if output: - sm = StatusManager.get_instance() - - # 🆕 aggiungiamo prefisso e usiamo i campi reali - message = f"[{self.__class__.__name__}] {output}" - - sm.add_log( - dag_id=self.dag_id, - execution_id=self.execution_id, - task_id=self.task_id, - message=message, - level="INFO" - ) + """ + Esegue il codice Python catturando stdout/stderr riga per riga + e scrivendolo DIRETTAMENTE nella tabella logs tramite StatusManager, + oltre che sul terminale del server. + """ + + sm = StatusManager.get_instance() + + dag_id = getattr(self, "dag_id", None) + execution_id = getattr(self, "execution_id", None) + task_id = self.task_id + + # Se per qualche motivo non abbiamo contesto, evitiamo di rompere tutto + if not dag_id or not execution_id: + dag_id = dag_id or "unknown_dag" + execution_id = execution_id or "unknown_execution" + + original_stdout = sys.stdout + original_stderr = sys.stderr + + class _LogStream(io.TextIOBase): + def __init__(self, orig_stream, level: str): + self._orig = orig_stream + self._buffer = "" + self._level = level + self._lock = threading.Lock() + + def write(self, s: str) -> int: + # Scrivi comunque sul terminale del server + self._orig.write(s) + self._orig.flush() + + # Accumula nel buffer e spezza per newline + with self._lock: + self._buffer += s + while "\n" in self._buffer: + line, self._buffer = self._buffer.split("\n", 1) + clean = line.rstrip("\r") + if clean.strip(): + # Scrivi nel DB, una riga = un log + sm.add_log( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id, + message=f"[PythonTask] {clean}", + level=self._level + ) + return len(s) + + def flush(self) -> None: + self._orig.flush() + + stdout_stream = _LogStream(original_stdout, level="INFO") + stderr_stream = _LogStream(original_stderr, level="ERROR") + + try: + with contextlib.redirect_stdout(stdout_stream), contextlib.redirect_stderr(stderr_stream): + if self.code: + exec(self.code, {}) + elif self.script_path: + with open(self.script_path, "r") as f: + code = f.read() + exec(code, {}) + else: + raise ValueError("PythonTask requires either 'code' or 'script_path'.") + finally: + # Ripristina gli stream originali + sys.stdout = original_stdout + sys.stderr = original_stderr + From 0f4931bcb8960788742ec139d6c3d396ac35243b Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sun, 16 Nov 2025 18:34:27 +0100 Subject: [PATCH 12/38] Fixed also Bash and Print messages on live log streaming --- examples/2_New_examples/2.1.1.Bash_chain.yaml | 2 +- src/maestro/server/tasks/bash_task.py | 90 +++++++++++++------ src/maestro/server/tasks/print_task.py | 25 +++++- 3 files changed, 83 insertions(+), 34 deletions(-) diff --git a/examples/2_New_examples/2.1.1.Bash_chain.yaml b/examples/2_New_examples/2.1.1.Bash_chain.yaml index f8337a7..9c86877 100644 --- a/examples/2_New_examples/2.1.1.Bash_chain.yaml +++ b/examples/2_New_examples/2.1.1.Bash_chain.yaml @@ -9,7 +9,7 @@ dag: - task_id: "count_lines" type: "BashTask" params: - command: "wc -l <<< 'Counting lines from previous output'" + command: "echo 'Counting lines from previous output' | wc -l" dependencies: ["list_files"] - task_id: "end" type: "PrintTask" diff --git a/src/maestro/server/tasks/bash_task.py b/src/maestro/server/tasks/bash_task.py index 8aec28c..f11fa1a 100644 --- a/src/maestro/server/tasks/bash_task.py +++ b/src/maestro/server/tasks/bash_task.py @@ -1,6 +1,8 @@ import subprocess import logging from maestro.server.tasks.base import BaseTask +from maestro.server.internals.status_manager import StatusManager + logger = logging.getLogger(__name__) logger.propagate = True @@ -13,38 +15,68 @@ class BashTask(BaseTask): command: str + def execute_local(self): - logger.info(f"[BashTask] Executing command: {self.command}") - try: - result = subprocess.run( - self.command, - shell=True, - executable="/bin/bash", - capture_output=True, - text=True + """ + BashTask: esegue il comando e registra stdout/stderr nel DB riga per riga, + con timestamp corretto come PythonTask. + """ + sm = StatusManager.get_instance() + dag_id = self.dag_id + execution_id = self.execution_id + task_id = self.task_id + + process = subprocess.Popen( + self.command, + shell=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + bufsize=1 + ) + + # Leggi STDOUT riga per riga + for line in process.stdout: + clean = line.rstrip("\n") + msg = f"[BashTask][stdout] {clean}" + + print(msg, flush=True) + + sm.add_log( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id, + message=msg, + level="INFO" + ) + + # Leggi STDERR riga per riga + for line in process.stderr: + clean = line.rstrip("\n") + msg = f"[BashTask][stderr] {clean}" + + print(msg, flush=True) + + sm.add_log( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id, + message=msg, + level="ERROR" + ) + + # Ritorno exit code + process.wait() + + if process.returncode != 0: + sm.add_log( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id, + message=f"[BashTask] Failed with exit code {process.returncode}", + level="ERROR" ) - # --- LOG STDOUT LINE BY LINE --- - if result.stdout: - for line in result.stdout.strip().splitlines(): - logger.info(f"[BashTask][stdout] {line}") - - # --- LOG STDERR LINE BY LINE --- - if result.stderr: - for line in result.stderr.strip().splitlines(): - logger.error(f"[BashTask][stderr] {line}") - - # --- Set task status --- - if result.returncode == 0: - logger.info("[BashTask] Command completed successfully.") - self.status = "completed" - else: - logger.error(f"[BashTask] Command failed with return code {result.returncode}.") - self.status = "failed" - - except Exception as e: - logger.exception(f"[BashTask] Exception while executing command: {e}") - self.status = "failed" def to_dict(self): """Include the command field in serialized form.""" diff --git a/src/maestro/server/tasks/print_task.py b/src/maestro/server/tasks/print_task.py index 930c7fa..5adf7c0 100644 --- a/src/maestro/server/tasks/print_task.py +++ b/src/maestro/server/tasks/print_task.py @@ -4,6 +4,8 @@ from rich import get_console from maestro.server.tasks.base import BaseTask +from maestro.server.internals.status_manager import StatusManager + class PrintTask(BaseTask): """A task that prints a message to the console.""" @@ -11,7 +13,22 @@ class PrintTask(BaseTask): delay: Optional[int] = 0 def execute_local(self): - logger = logging.getLogger(__name__) - logger.info(f"[PrintTask] {self.message}") - if self.delay: - time.sleep(self.delay) + """ + PrintTask: registra nel DB lo stesso messaggio che stampa su stdout, + con timestamp corretto e senza dipendere dal logger di orchestrator. + """ + sm = StatusManager.get_instance() + + message = f"[PrintTask] {self.message}" + + # stampa su console server (come ora) + print(message, flush=True) + + # aggiungi al DB + sm.add_log( + dag_id=self.dag_id, + execution_id=self.execution_id, + task_id=self.task_id, + message=message, + level="INFO" + ) From 27a48013c4f52094fe6f9211223202e1f3d977ba Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sun, 16 Nov 2025 19:04:38 +0100 Subject: [PATCH 13/38] 'maestro attach...' command finally works --- src/maestro/client/cli.py | 5 ++++ src/maestro/server/api/v1/routes/logs.py | 38 +++++++++++------------- 2 files changed, 22 insertions(+), 21 deletions(-) diff --git a/src/maestro/client/cli.py b/src/maestro/client/cli.py index b1b4b70..aa3b097 100644 --- a/src/maestro/client/cli.py +++ b/src/maestro/client/cli.py @@ -288,6 +288,11 @@ def handle_exit(signum, frame): console.print(f"[red]Stream error: {log_entry['error']}[/red]") break + # 👉 Evento interno del server: DAG completata → detach automatico + if "event" in log_entry and log_entry["event"] == "DAG_COMPLETED": + console.print("\n[green][DAG COMPLETED] Detaching...[/green]\n") + return + level_style = { "ERROR": "red", "WARNING": "yellow", diff --git a/src/maestro/server/api/v1/routes/logs.py b/src/maestro/server/api/v1/routes/logs.py index 67827d9..c332577 100644 --- a/src/maestro/server/api/v1/routes/logs.py +++ b/src/maestro/server/api/v1/routes/logs.py @@ -171,46 +171,42 @@ async def attach_dag_logs( Minimal patch: stabilizes streaming by fixing ordering & cursor management. No DB changes required. """ + async def log_streamer(): + last_seen = set() last_timestamp = None - displayed_logs = set() while True: try: with orchestrator.status_manager as sm: logs = sm.get_execution_logs(dag_id, execution_id, limit=200) - # 👉 SORT logs OLDEST → NEWEST (important!) + # Ordina dal più vecchio al più recente logs = sorted(logs, key=lambda x: x["timestamp"]) - # Apply filters - if task_filter: - logs = [log for log in logs if log["task_id"] == task_filter] - if level_filter: - logs = [log for log in logs if log["level"].upper() == level_filter.upper()] - new_logs = [] - for log in logs: - # Build unique log key - log_key = f"{log['timestamp']}_{log['task_id']}_{log['level']}_{log['message'][:50]}" + key = f"{log['timestamp']}-{log['task_id']}-{log['message'][:50]}" + if key not in last_seen: + new_logs.append(log) + last_seen.add(key) + last_timestamp = log["timestamp"] - # 👉 Ensure ordering works: if timestamp is new or unseen, stream it - if log_key not in displayed_logs: - if last_timestamp is None or log["timestamp"] >= last_timestamp: - new_logs.append(log) - displayed_logs.add(log_key) - last_timestamp = log["timestamp"] - - # 👉 Stream logs in natural order (no reversed!) + # STREAM DEI LOG for log in new_logs: yield f"data: {json.dumps(log)}\n\n" - await asyncio.sleep(0.5) + # 🔥 TERMINAZIONE AUTOMATICA STREAM + exec_info = sm.get_latest_execution(dag_id) + if exec_info and exec_info["status"] in ("completed", "failed", "cancelled"): + yield 'data: {"event": "DAG_COMPLETED"}\n\n' + return + + await asyncio.sleep(0.3) except Exception as e: yield f"data: {json.dumps({'error': str(e)})}\n\n" - break + return return StreamingResponse( log_streamer(), From 022f3d4773948b425a3dc2fc763fcd7b27f639c9 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sun, 16 Nov 2025 21:49:26 +0100 Subject: [PATCH 14/38] Attachment inverted successfully --- src/maestro/client/cli.py | 101 +++++++++++++++++++++++++++----------- 1 file changed, 71 insertions(+), 30 deletions(-) diff --git a/src/maestro/client/cli.py b/src/maestro/client/cli.py index aa3b097..868d52e 100644 --- a/src/maestro/client/cli.py +++ b/src/maestro/client/cli.py @@ -58,47 +58,88 @@ def create( console.print(f"[red]Error: {e}[/red]") raise typer.Exit(1) + @app.command() def run( - dag_input: str = typer.Argument(..., help="DAG ID or path to DAG YAML file"), - resume: bool = typer.Option(False, "--resume", help="Resume from last checkpoint"), - fail_fast: bool = typer.Option(True, "--fail-fast/--no-fail-fast", help="Stop on first failure"), + dag_file: str = typer.Argument(..., help="Path to DAG YAML file"), + detach: bool = typer.Option(False, "--detach", "-d", help="Run in detached mode"), server_url: str = typer.Option("http://localhost:8000", "--server", help="Maestro server URL") ): - """Run a DAG - either by ID or by creating from a YAML file""" + """ + Run a DAG: default = attached mode (stream logs until completion). + Use --detach to run without attaching to logs. + """ + api_client.base_url = server_url check_server_connection() + console.print(f"[cyan]Creating DAG from file: {dag_file}[/cyan]") + + # 1. Create DAG try: - # Check if the input is a file path or a DAG ID - if os.path.exists(dag_input) and dag_input.endswith(('.yaml', '.yml')): - # It's a file path - create the DAG first - dag_file_path = os.path.abspath(dag_input) - console.print(f"[blue]Creating DAG from file: {dag_file_path}[/blue]") - create_response = api_client.create_dag(dag_file_path) - dag_id = create_response['dag_id'] - console.print(f"[bold green]✓ DAG created successfully with ID: {dag_id}[/bold green]") - else: - # It's a DAG ID - dag_id = dag_input - - # Now run the DAG - response = api_client.run_dag(dag_id, resume, fail_fast) - console.print(f"[bold green]✓ DAG started successfully![/bold green]") - console.print(f"[cyan]DAG ID:[/cyan] {response['dag_id']}") - console.print(f"[cyan]Execution ID:[/cyan] {response['execution_id']}") - console.print(f"[cyan]Status:[/cyan] [yellow]{response['status']}[/yellow]") - console.print() - console.print("[bold blue]Available commands:[/bold blue]") - console.print(f" • [bold]maestro status {response['dag_id']}[/bold] - Check DAG status") - console.print(f" • [bold]maestro log {response['dag_id']}[/bold] - View logs") - console.print(f" • [bold]maestro attach {response['dag_id']}[/bold] - Attach to live log stream") - console.print(f" • [bold]maestro stop {response['dag_id']}[/bold] - Stop execution") - console.print() + result = api_client.create_dag(dag_file) except Exception as e: - console.print(f"[red]Error: {e}[/red]") + console.print(f"[red]Error creating DAG: {e}[/red]") + raise typer.Exit(1) + + dag_id = result.get("dag_id") + if not dag_id: + console.print("[red]Server did not return dag_id[/red]") + raise typer.Exit(1) + + console.print(f"[green]DAG created: {dag_id}[/green]") + + # 2. Run DAG + try: + run_info = api_client.run_dag(dag_id) + except Exception as e: + console.print(f"[red]Error running DAG: {e}[/red]") + raise typer.Exit(1) + + execution_id = run_info.get("execution_id") + if not execution_id: + console.print("[red]Server did not return execution_id[/red]") raise typer.Exit(1) + console.print(f"[green]Execution started: {execution_id}[/green]") + + # 3A. DETACHED MODE → exit immediately + if detach: + console.print("[yellow]Running in detached mode.[/yellow]") + console.print("Use 'maestro attach {dag_id}' to follow logs.") + return + + # 3B. ATTACHED MODE → stream logs live + console.print(f"[bold cyan]Attaching to live logs...[/bold cyan]") + console.print("[dim]Press Ctrl+C to detach[/dim]\n") + + try: + for log_entry in api_client.stream_dag_logs_v1(dag_id, execution_id): + if "event" in log_entry and log_entry["event"] == "DAG_COMPLETED": + console.print("\n[green]DAG completed — detaching.[/green]") + return + + if "level" not in log_entry: + continue # skip malformed entries + + level = log_entry["level"] + ts = log_entry["timestamp"].split("T")[1].split(".")[0] + task = log_entry["task_id"] + msg = log_entry["message"] + + style = { + "ERROR": "red", + "WARNING": "yellow", + "INFO": "green", + "DEBUG": "blue", + }.get(level, "white") + + console.print(f"[dim]{ts}[/dim] [{style}]{level}[/]{style}] [magenta]{task}[/]: {msg}") + + except KeyboardInterrupt: + console.print("\n[yellow]Detached from log stream[/yellow]") + + @app.command() def validate( dag_file: str = typer.Argument(..., help="Path to the DAG YAML file"), From a94d43ff920f5c11972bec45427d92c9174e624f Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sun, 16 Nov 2025 22:49:17 +0100 Subject: [PATCH 15/38] Fine tuning - Inversion of attachment of client console - completed --- src/maestro/client/cli.py | 93 +++++++++++++++++++++++++++++++++------ src/maestro/server/app.py | 27 ++++++++++++ 2 files changed, 107 insertions(+), 13 deletions(-) diff --git a/src/maestro/client/cli.py b/src/maestro/client/cli.py index 868d52e..be8f45e 100644 --- a/src/maestro/client/cli.py +++ b/src/maestro/client/cli.py @@ -17,6 +17,29 @@ from .api_client import MaestroAPIClient +def _print_status_table(console, dag_id: str, execution_id: str, status: dict): + """ + Helper per stampare lo stesso output di 'maestro status'. + """ + from rich.table import Table + + table = Table(title=f"DAG Status — {dag_id} ({execution_id})") + + table.add_column("Task ID", style="magenta") + table.add_column("Status", style="cyan") + table.add_column("Started", style="green") + table.add_column("Completed", style="yellow") + + for task in status.get("tasks", []): + table.add_row( + task["task_id"], + task["status"], + task.get("started_at", "") or "", + task.get("completed_at", "") or "" + ) + + console.print(table) + app = typer.Typer(help="Maestro CLI Client - Communicate with Maestro REST API server") console = Console() @@ -66,8 +89,9 @@ def run( server_url: str = typer.Option("http://localhost:8000", "--server", help="Maestro server URL") ): """ - Run a DAG: default = attached mode (stream logs until completion). - Use --detach to run without attaching to logs. + Run a DAG. + - Default = attached mode (stream logs until completion) + - --detach = detached mode with automatic status reporting at the end """ api_client.base_url = server_url @@ -75,7 +99,7 @@ def run( console.print(f"[cyan]Creating DAG from file: {dag_file}[/cyan]") - # 1. Create DAG + # 1. CREATE DAG try: result = api_client.create_dag(dag_file) except Exception as e: @@ -89,7 +113,7 @@ def run( console.print(f"[green]DAG created: {dag_id}[/green]") - # 2. Run DAG + # 2. RUN DAG try: run_info = api_client.run_dag(dag_id) except Exception as e: @@ -103,24 +127,57 @@ def run( console.print(f"[green]Execution started: {execution_id}[/green]") - # 3A. DETACHED MODE → exit immediately + # ----------------------------------------------------------------------- + # 3A — DETACHED MODE + # ----------------------------------------------------------------------- if detach: - console.print("[yellow]Running in detached mode.[/yellow]") - console.print("Use 'maestro attach {dag_id}' to follow logs.") + + console.print("[yellow]Running in detached mode.[/yellow]\n") + + console.print("Available commands:") + console.print(f" • maestro attach {dag_id} - Attach to live log stream") + console.print(f" • maestro stop {dag_id} or Ctrl+C - Stop execution\n") + + console.print("[dim]...running DAG...[/dim]\n") + + # POLLING FINO A COMPLETAMENTO + while True: + try: + status = api_client.get_dag_status(dag_id, execution_id) + if status["status"] in ("completed", "failed", "cancelled"): + break + time.sleep(1) + except KeyboardInterrupt: + console.print("\n[yellow]Detached (polling stopped).[/yellow]") + return + except Exception: + # se il server fosse temporaneamente non raggiungibile + time.sleep(1) + + console.print(f"\n[green]DAG completed[/green]\n") + + console.print("Available commands:") + console.print(f" • maestro status {dag_id} - Check DAG status") + console.print(f" • maestro log {dag_id} - View logs\n") + return - # 3B. ATTACHED MODE → stream logs live + # ----------------------------------------------------------------------- + # 3B — ATTACHED MODE + # ----------------------------------------------------------------------- + console.print(f"[bold cyan]Attaching to live logs...[/bold cyan]") console.print("[dim]Press Ctrl+C to detach[/dim]\n") try: for log_entry in api_client.stream_dag_logs_v1(dag_id, execution_id): - if "event" in log_entry and log_entry["event"] == "DAG_COMPLETED": - console.print("\n[green]DAG completed — detaching.[/green]") - return + # Evento di fine DAG + if log_entry.get("event") == "DAG_COMPLETED": + console.print("\n[green]DAG completed — detaching.[/green]\n") + break if "level" not in log_entry: - continue # skip malformed entries + continue level = log_entry["level"] ts = log_entry["timestamp"].split("T")[1].split(".")[0] @@ -134,10 +191,20 @@ def run( "DEBUG": "blue", }.get(level, "white") - console.print(f"[dim]{ts}[/dim] [{style}]{level}[/]{style}] [magenta]{task}[/]: {msg}") + console.print(f"[dim]{ts}[/dim] [{style}]{level}[/] [magenta]{task}[/]: {msg}") except KeyboardInterrupt: console.print("\n[yellow]Detached from log stream[/yellow]") + return + + # 🔥 DOPO LO STREAMING → MOSTRA STATUS COMPLETO + console.print("[cyan]DAG final status:[/cyan]\n") + + try: + status = api_client.get_dag_status(dag_id, execution_id) + _print_status_table(console, dag_id, execution_id, status) + except Exception as e: + console.print(f"[red]Unable to fetch DAG status: {e}[/red]") @app.command() diff --git a/src/maestro/server/app.py b/src/maestro/server/app.py index b47a347..de18039 100644 --- a/src/maestro/server/app.py +++ b/src/maestro/server/app.py @@ -473,10 +473,37 @@ async def cleanup_old_executions(days: int = 30): logger.error(f"Failed to cleanup executions: {e}") raise HTTPException(status_code=500, detail=str(e)) + def start_server(host: str = "0.0.0.0", port: int = 8000, log_level: str = "info"): """Start the Maestro API server""" + + import logging + + class PollingFilter(logging.Filter): + """ + Nasconde solo i GET /status del polling interno del client. + Tutti gli altri log HTTP restano visibili. + """ + def filter(self, record): + msg = record.getMessage() + + # es: "GET /v1/dags/foo/status?execution_id=XXXX HTTP/1.1" 200 OK + if "GET" in msg and "/v1/dags/" in msg and "/status" in msg: + return False # → non loggare queste richieste + + return True # → logga tutto il resto + + # Applica il filtro SOLO al logger di accesso HTTP uvicorn + logging.getLogger("uvicorn.access").addFilter(PollingFilter()) + + # Avvia il server normalmente uvicorn.run(app, host=host, port=port, log_level=log_level) + +# def start_server(host: str = "0.0.0.0", port: int = 8000, log_level: str = "info"): + """Start the Maestro API server""" + # uvicorn.run(app, host=host, port=port, log_level=log_level) + def main(): import argparse From fe8efc716843dfdd7ff145b5935fe9b5569db828 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 18 Nov 2025 11:46:57 +0100 Subject: [PATCH 16/38] Fixed logs in parallel execution --- examples/2_New_examples/1.2.1.Long_waits.yaml | 2 +- .../1.2.2.Parallel_then_join.yaml | 5 +- .../1.3.1.Branching_waits_and_joins.yaml | 10 +- .../1.3.2.Cascade_retry_delays.yaml | 11 +- .../1.3.3.Parallel_heavy_bash_python.yaml | 6 +- src/maestro/server/internals/orchestrator.py | 200 ++++++++++-------- src/maestro/server/tasks/python_task.py | 116 +++++----- 7 files changed, 189 insertions(+), 161 deletions(-) diff --git a/examples/2_New_examples/1.2.1.Long_waits.yaml b/examples/2_New_examples/1.2.1.Long_waits.yaml index 6bf03a5..652dea1 100644 --- a/examples/2_New_examples/1.2.1.Long_waits.yaml +++ b/examples/2_New_examples/1.2.1.Long_waits.yaml @@ -5,7 +5,7 @@ # ------------------------------------------------------------------------------------- dag: - name: "intermediate_long_waits" + name: "long_waits" tasks: - task_id: "start" type: "PrintTask" diff --git a/examples/2_New_examples/1.2.2.Parallel_then_join.yaml b/examples/2_New_examples/1.2.2.Parallel_then_join.yaml index cafd4b1..af67ec2 100644 --- a/examples/2_New_examples/1.2.2.Parallel_then_join.yaml +++ b/examples/2_New_examples/1.2.2.Parallel_then_join.yaml @@ -5,7 +5,7 @@ # ------------------------------------------------------------------------------------- dag: - name: "intermediate_parallel_then_join" + name: "parallel_then_join" tasks: - task_id: "start" type: "PrintTask" @@ -20,6 +20,7 @@ dag: import time print("Worker A...") time.sleep(30) + print("Worker A - Task complete.") dependencies: ["start"] - task_id: "worker_b" @@ -29,6 +30,7 @@ dag: import time print("Worker B...") time.sleep(40) + print("Worker B - Task complete.") dependencies: ["start"] - task_id: "worker_c" @@ -38,6 +40,7 @@ dag: import time print("Worker C...") time.sleep(30) + print("Worker C - Task complete.") dependencies: ["start"] - task_id: "merge" diff --git a/examples/2_New_examples/1.3.1.Branching_waits_and_joins.yaml b/examples/2_New_examples/1.3.1.Branching_waits_and_joins.yaml index 049e9a9..71e19b5 100644 --- a/examples/2_New_examples/1.3.1.Branching_waits_and_joins.yaml +++ b/examples/2_New_examples/1.3.1.Branching_waits_and_joins.yaml @@ -6,7 +6,7 @@ dag: - name: "difficult_branching_waits_and_joins" + name: "branching_waits_and_joins" tasks: - task_id: "start" type: "PrintTask" @@ -18,28 +18,28 @@ dag: type: "PythonTask" params: code: | - import time; print("A1..."); time.sleep(20) + import time; print("Running A1..."); time.sleep(20); print("Task A1 completed.") dependencies: ["start"] - task_id: "A2" type: "PythonTask" params: code: | - import time; print("A2..."); time.sleep(25) + import time; print("Running A2..."); time.sleep(25); print("Task A2 completed.") dependencies: ["start"] - task_id: "B1" type: "PythonTask" params: code: | - import time; print("B1..."); time.sleep(30) + import time; print("Running B1..."); time.sleep(30); print("Task B1 completed.") dependencies: ["A1"] - task_id: "B2" type: "PythonTask" params: code: | - import time; print("B2..."); time.sleep(30) + import time; print("Running B2..."); time.sleep(30); print("Task B2 completed.") dependencies: ["A2"] - task_id: "combine" diff --git a/examples/2_New_examples/1.3.2.Cascade_retry_delays.yaml b/examples/2_New_examples/1.3.2.Cascade_retry_delays.yaml index 17b70df..7426fa2 100644 --- a/examples/2_New_examples/1.3.2.Cascade_retry_delays.yaml +++ b/examples/2_New_examples/1.3.2.Cascade_retry_delays.yaml @@ -5,7 +5,7 @@ # ------------------------------------------------------------------------------------- dag: - name: "difficult_cascade_retry_delays" + name: "cascade_retry_delays" tasks: - task_id: "start" type: "PrintTask" @@ -19,8 +19,11 @@ dag: code: | import time, random print("Unstable step (may fail)...") - time.sleep(15) - if random.random() < 0.5: + # time.sleep(15) + generate_random = random.random() + print(f"Generated: {round(generate_random, 2)}") + if generate_random < 0.67: + print("Failure occurred") raise Exception("Simulated failure!") print("Unstable succeeded") retries: 3 @@ -33,7 +36,7 @@ dag: code: | import time print("Slow processing...") - time.sleep(30) + # time.sleep(30) dependencies: ["unstable_step"] - task_id: "finish" diff --git a/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml b/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml index e992ade..21958a7 100644 --- a/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml +++ b/examples/2_New_examples/1.3.3.Parallel_heavy_bash_python.yaml @@ -5,7 +5,7 @@ # ------------------------------------------------------------------------------------- dag: - name: "difficult_parallel_heavy_bash_python" + name: "parallel_heavy_bash_python" tasks: - task_id: "start" type: "PrintTask" @@ -19,6 +19,7 @@ dag: command: | echo "Bash running long task..." sleep 45 + echo "Bash task completed." dependencies: ["start"] - task_id: "python_long" @@ -26,8 +27,9 @@ dag: params: code: | import time - print("Python long task...") + print("Python running long task...") time.sleep(50) + print("Python task completed.") dependencies: ["start"] - task_id: "final_merge" diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index 8041437..93001b3 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -1,6 +1,8 @@ import logging import uuid import threading +from datetime import datetime +import re from concurrent.futures import ThreadPoolExecutor, Future, as_completed from typing import Any, Type, Dict, Optional, Set, List @@ -20,42 +22,68 @@ class DatabaseLogHandler(logging.Handler): - """Custom logging handler that writes to StatusManager database.""" + """Thread-safe logging handler that writes task logs to StatusManager.""" + + _ansi_escape = re.compile(r'\x1B\[[0-?]*[ -/]*[@-~]') def __init__(self, status_manager: StatusManager): super().__init__() self.status_manager = status_manager - self.current_dag_id = None - self.current_execution_id = None - self.current_task_id = None - def set_context(self, dag_id: str, execution_id: str, task_id: str = None): - """Set the current execution context for logging.""" - self.current_dag_id = dag_id - self.current_execution_id = execution_id - self.current_task_id = task_id + # Thread-local context instead of global instance attributes + self._ctx = threading.local() + # ------------------------------- + # CONTEXT HANDLING (THREAD-SAFE) + # ------------------------------- + def set_context(self, dag_id: str, execution_id: str, task_id: str = None): + """Assigns context to the current thread only.""" + self._ctx.dag_id = dag_id + self._ctx.execution_id = execution_id + self._ctx.task_id = task_id + + def clear_context(self): + """Clears thread context after task finishes.""" + self._ctx.dag_id = None + self._ctx.execution_id = None + self._ctx.task_id = None + + # ------------------------------- + # LOG EMISSION + # ------------------------------- def emit(self, record): - """Emit a log record to the database, preserving the real timestamp.""" - if self.current_dag_id and self.current_execution_id: - try: - # Timestamp reale del LogRecord - log_timestamp = datetime.fromtimestamp(record.created) + """Store logs in the database, per-thread context.""" + dag_id = getattr(self._ctx, "dag_id", None) + execution_id = getattr(self._ctx, "execution_id", None) + task_id = getattr(self._ctx, "task_id", None) - # Messaggio grezzo (senza metadata RichHandler) - log_message = record.getMessage() + # If we have no per-thread context: ignore this log + if not dag_id or not execution_id: + return - with self.status_manager as sm: - sm.log_message( - dag_id=self.current_dag_id, - execution_id=self.current_execution_id, - task_id=self.current_task_id or "system", - level=record.levelname, - message=log_message, - timestamp=log_timestamp - ) - except Exception: - pass # Non deve mai rompere l'esecuzione + try: + # Extract real timestamp + timestamp = datetime.fromtimestamp(record.created) + + # Raw message + msg = record.getMessage() + + # Remove ANSI escape sequences (Rich formatting) + msg = self._ansi_escape.sub("", msg) + + with self.status_manager as sm: + sm.log_message( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id or "system", + level=record.levelname, + message=msg, + timestamp=timestamp + ) + + except Exception: + # Log errors should NEVER break execution + pass class Orchestrator: @@ -86,10 +114,10 @@ def __init__( self._execution_stop_events: Dict[str, threading.Event] = {} self.task_executor = ThreadPoolExecutor(max_workers=10) + def _setup_logging(self, log_level: str): - """Setup global logging with RichHandler (console) and DatabaseLogHandler (persistent).""" + """Setup global logging with RichHandler (console) and prepare DB handler.""" - # Pulisci logging root logging.shutdown() for h in logging.root.handlers[:]: logging.root.removeHandler(h) @@ -97,31 +125,24 @@ def _setup_logging(self, log_level: str): level = getattr(logging, log_level.upper(), logging.INFO) logging.root.setLevel(level) - # Console + # Console: solo RichHandler sul root rich_handler = RichHandler(console=self.console, rich_tracebacks=True) rich_handler.setLevel(level) - # DB - self.db_handler = DatabaseLogHandler(self.status_manager) - self.db_handler.setLevel(logging.DEBUG) - root_logger = logging.getLogger() root_logger.addHandler(rich_handler) - root_logger.addHandler(self.db_handler) logging.captureWarnings(True) - root_logger.propagate = True + root_logger.propagate = False - # Logger dell’orchestratore + # Logger dell’orchestratore (non propaga al root) self.logger = logging.getLogger("maestro.orchestrator") self.logger.setLevel(level) - self.logger.propagate = True + self.logger.propagate = False + + # DB handler creato ma NON agganciato al root + self.db_handler = DatabaseLogHandler(self.status_manager) + self.db_handler.setLevel(logging.DEBUG) - # 🔧 AGGANCIA il DB handler anche a tutti i logger di maestro già creati - for name in list(logging.root.manager.loggerDict.keys()): - if name.startswith("maestro."): - lg = logging.getLogger(name) - lg.addHandler(self.db_handler) - lg.propagate = True def register_task_type(self, name: str, task_class: Type[BaseTask]): """Register a custom task type.""" @@ -186,13 +207,13 @@ def run_dag_in_thread( # Use provided execution_id or generate a new one if execution_id is None: execution_id = str(uuid.uuid4()) - + dag.execution_id = execution_id - + # Create a stop event for this execution stop_event = threading.Event() self._execution_stop_events[execution_id] = stop_event - + # Create or update execution record in database synchronously with self.status_manager as sm: # Check if execution already exists (e.g., with 'created' status) @@ -277,10 +298,10 @@ def run_dag( # Set up database logging if execution_id is provided db_handler = None + task_loggers = [] if execution_id: - db_handler = DatabaseLogHandler(self.status_manager) - db_handler.set_context(dag_id, execution_id) + db_handler = self.db_handler task_loggers = [ 'maestro.server.tasks.terraform_task', @@ -322,12 +343,15 @@ def run_dag( stop_event=stop_event, db_handler=db_handler ) + finally: # Clean up database handler if db_handler: for logger_name in task_loggers: logger = logging.getLogger(logger_name) - logger.removeHandler(db_handler) + if db_handler in logger.handlers: + logger.removeHandler(db_handler) + def _run_dag_concurrent( self, @@ -342,14 +366,14 @@ def _run_dag_concurrent( ): """Execute DAG tasks concurrently based on dependencies.""" dag_id = dag.dag_id - + # Initialize task tracking sets pending_tasks: Set[str] = set(dag.tasks.keys()) running_tasks: Dict[str, Future] = {} completed_tasks: Set[str] = set() failed_tasks: Set[str] = set() skipped_tasks: Set[str] = set() - + with self.status_manager as sm: # Handle resume - mark already completed tasks if resume: @@ -364,7 +388,7 @@ def _run_dag_concurrent( progress_tracker.increment_completed() if status_callback: status_callback() - + # Main execution loop while pending_tasks or running_tasks: # Check for cancellation @@ -374,14 +398,14 @@ def _run_dag_concurrent( dag, execution_id, pending_tasks, running_tasks, sm, status_callback ) break - + # Check for completed tasks if running_tasks: completed_futures = [] for task_id, future in list(running_tasks.items()): if future.done(): completed_futures.append((task_id, future)) - + for task_id, future in completed_futures: del running_tasks[task_id] try: @@ -395,16 +419,16 @@ def _run_dag_concurrent( # Cancel all running tasks and stop self._cancel_running_tasks(running_tasks) raise Exception(f"Task {task_id} failed: {e}") - + # Find tasks ready to run ready_tasks = self._find_ready_tasks( dag, pending_tasks, completed_tasks, failed_tasks, skipped_tasks ) - + # Submit ready tasks for execution for task_id in ready_tasks: task = dag.tasks[task_id] - + # Check if dependencies failed if self._has_failed_dependencies(task, failed_tasks, skipped_tasks): self.logger.warning(f"Skipping task {task_id} because its dependencies failed") @@ -415,7 +439,7 @@ def _run_dag_concurrent( if status_callback: status_callback() continue - + # Submit task for execution self.logger.info(f"Submitting task {task_id} for execution") future = self.task_executor.submit( @@ -429,7 +453,7 @@ def _run_dag_concurrent( ) running_tasks[task_id] = future pending_tasks.remove(task_id) - + # Brief sleep to prevent busy waiting if not ready_tasks and running_tasks: threading.Event().wait(0.1) @@ -444,14 +468,14 @@ def _find_ready_tasks( ) -> List[str]: """Find tasks that are ready to run (all dependencies completed).""" ready_tasks = [] - + for task_id in pending_tasks: task = dag.tasks[task_id] - + # Check if all dependencies are completed if all(dep in completed_tasks for dep in task.dependencies): ready_tasks.append(task_id) - + return ready_tasks def _has_failed_dependencies( @@ -466,6 +490,7 @@ def _has_failed_dependencies( for dep in task.dependencies ) + def _execute_task_async( self, task, @@ -477,21 +502,21 @@ def _execute_task_async( ): """Execute a single task asynchronously.""" task_id = task.task_id - + try: # Update status to running task.status = TaskStatus.RUNNING with self.status_manager as sm: sm.set_task_status(dag_id, task_id, "running", execution_id) - + if status_callback: status_callback() - - # Update database handler context for this task + + # Context per-THREAD per questo task if db_handler: db_handler.set_context(dag_id, execution_id, task_id) - - # 🆕 Passa il contesto alla task (serve per i log) + + # (opzionale ma comodo) passa info alla Task task.dag_id = dag_id task.execution_id = execution_id @@ -499,28 +524,33 @@ def _execute_task_async( self.logger.info(f"Executing task: {task_id}") executor_instance = self.executor_factory.get_executor(task.executor) task.execute(executor_instance) - + # Update status to completed task.status = TaskStatus.COMPLETED with self.status_manager as sm: sm.set_task_status(dag_id, task_id, "completed", execution_id) - + if progress_tracker: progress_tracker.increment_completed() if status_callback: status_callback() - + except Exception as e: # Update status to failed task.status = TaskStatus.FAILED with self.status_manager as sm: sm.set_task_status(dag_id, task_id, "failed", execution_id) - + if status_callback: status_callback() - + # Re-raise the exception to be caught by the main loop raise + finally: + # IMPORTANTISSIMO: svuota il contesto per il thread + if db_handler: + db_handler.clear_context() + def _handle_cancellation( self, @@ -534,16 +564,16 @@ def _handle_cancellation( """Handle DAG cancellation.""" # Cancel all running tasks self._cancel_running_tasks(running_tasks) - + # Mark DAG as cancelled sm.update_dag_execution_status(dag.dag_id, execution_id, "cancelled") - + # Mark pending tasks as skipped for task_id in pending_tasks: task = dag.tasks[task_id] task.status = TaskStatus.SKIPPED sm.set_task_status(dag.dag_id, task_id, "skipped", execution_id) - + if status_callback: status_callback() @@ -609,7 +639,7 @@ def get_dag_status(self, dag: DAG) -> Dict[str, Any]: "summary": summary, "total_tasks": len(dag.tasks) } - + def cancel_dag_execution(self, dag_id: str, execution_id: str = None) -> bool: """Cancel a running DAG execution by setting its stop event.""" with self.status_manager as sm: @@ -620,31 +650,31 @@ def cancel_dag_execution(self, dag_id: str, execution_id: str = None) -> bool: if exec_info["dag_id"] == dag_id: execution_id = exec_info["execution_id"] break - + if execution_id and execution_id in self._execution_stop_events: # Set the stop event to signal the thread to stop self._execution_stop_events[execution_id].set() # Update the database status sm.cancel_dag_execution(dag_id, execution_id) return True - + return False - + def stop_all_running_dags(self) -> int: """Stop all currently running DAGs. Returns the number of DAGs stopped.""" stopped_count = 0 - + with self.status_manager as sm: running_dags = sm.get_running_dags() - + for dag_info in running_dags: dag_id = dag_info["dag_id"] execution_id = dag_info["execution_id"] - + if self.cancel_dag_execution(dag_id, execution_id): stopped_count += 1 self.logger.info(f"Stopped DAG {dag_id} (execution: {execution_id})") else: self.logger.warning(f"Failed to stop DAG {dag_id} (execution: {execution_id})") - + return stopped_count diff --git a/src/maestro/server/tasks/python_task.py b/src/maestro/server/tasks/python_task.py index 3ecea62..b69470a 100644 --- a/src/maestro/server/tasks/python_task.py +++ b/src/maestro/server/tasks/python_task.py @@ -12,7 +12,7 @@ class PythonTask(BaseTask): """ Executes inline Python code or a .py script. - Captures stdout/stderr (print statements) and writes them into logs table + Captures print() output and writes it into logs table with proper dag_id, execution_id, and formatted message. """ @@ -21,72 +21,62 @@ class PythonTask(BaseTask): def execute_local(self): """ - Esegue il codice Python catturando stdout/stderr riga per riga - e scrivendolo DIRETTAMENTE nella tabella logs tramite StatusManager, - oltre che sul terminale del server. + Esegue il codice Python catturando le chiamate a print() e + scrivendole DIRETTAMENTE nella tabella logs tramite StatusManager, + senza toccare sys.stdout/sys.stderr globali (thread-safe). """ - sm = StatusManager.get_instance() - dag_id = getattr(self, "dag_id", None) - execution_id = getattr(self, "execution_id", None) + dag_id = getattr(self, "dag_id", None) or "unknown_dag" + execution_id = getattr(self, "execution_id", None) or "unknown_execution" task_id = self.task_id - # Se per qualche motivo non abbiamo contesto, evitiamo di rompere tutto - if not dag_id or not execution_id: - dag_id = dag_id or "unknown_dag" - execution_id = execution_id or "unknown_execution" - - original_stdout = sys.stdout - original_stderr = sys.stderr - - class _LogStream(io.TextIOBase): - def __init__(self, orig_stream, level: str): - self._orig = orig_stream - self._buffer = "" - self._level = level - self._lock = threading.Lock() - - def write(self, s: str) -> int: - # Scrivi comunque sul terminale del server - self._orig.write(s) - self._orig.flush() - - # Accumula nel buffer e spezza per newline - with self._lock: - self._buffer += s - while "\n" in self._buffer: - line, self._buffer = self._buffer.split("\n", 1) - clean = line.rstrip("\r") - if clean.strip(): - # Scrivi nel DB, una riga = un log - sm.add_log( - dag_id=dag_id, - execution_id=execution_id, - task_id=task_id, - message=f"[PythonTask] {clean}", - level=self._level - ) - return len(s) - - def flush(self) -> None: - self._orig.flush() - - stdout_stream = _LogStream(original_stdout, level="INFO") - stderr_stream = _LogStream(original_stderr, level="ERROR") + # useremo la print "vera" solo per la console server + real_print = print + + def _log_line(level: str, text: str): + # stampa su console server + real_print(text, flush=True) + # e registra nel DB + sm.add_log( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id, + message=text, + level=level + ) + + def task_print(*args, **kwargs): + sep = kwargs.get("sep", " ") + end = kwargs.get("end", "\n") + s = sep.join(str(a) for a in args) + end + + # spezza su newline per loggare riga per riga + for line in s.splitlines(): + clean = line.rstrip("\r") + if clean.strip(): + _log_line("INFO", f"[PythonTask] {clean}") + + # Ambiente di esecuzione: print viene shadowata + env = { + "__name__": "__main__", + "print": task_print, + } try: - with contextlib.redirect_stdout(stdout_stream), contextlib.redirect_stderr(stderr_stream): - if self.code: - exec(self.code, {}) - elif self.script_path: - with open(self.script_path, "r") as f: - code = f.read() - exec(code, {}) - else: - raise ValueError("PythonTask requires either 'code' or 'script_path'.") - finally: - # Ripristina gli stream originali - sys.stdout = original_stdout - sys.stderr = original_stderr - + if self.code: + exec(self.code, env) + elif self.script_path: + with open(self.script_path, "r") as f: + code = f.read() + exec(code, env) + else: + raise ValueError("PythonTask requires either 'code' or 'script_path'.") + except Exception: + # logga l'eccezione nel DB + tb = traceback.format_exc() + for line in tb.splitlines(): + if line.strip(): + _log_line("ERROR", f"[PythonTask][exception] {line}") + # rialza per far fallire il task + raise From 6f78122a9fed15f8f4e4fdffb21aa7524d4ed30d Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 18 Nov 2025 13:39:44 +0100 Subject: [PATCH 17/38] Patched retries bugs --- src/maestro/server/internals/orchestrator.py | 128 +++++++++++++------ src/maestro/server/tasks/base.py | 7 +- src/maestro/server/tasks/python_task.py | 1 + 3 files changed, 97 insertions(+), 39 deletions(-) diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index 93001b3..c763ec7 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -3,6 +3,7 @@ import threading from datetime import datetime import re +import time from concurrent.futures import ThreadPoolExecutor, Future, as_completed from typing import Any, Type, Dict, Optional, Set, List @@ -500,56 +501,107 @@ def _execute_task_async( progress_tracker, db_handler ): - """Execute a single task asynchronously.""" + + """Execute a single task asynchronously WITH retry support.""" + task_id = task.task_id + print( + f"[DEBUG] task={task_id}, hasattr(retries)={hasattr(task, 'retries')}, " + f"raw_retries={getattr(task, 'retries', None)!r}, dict={task.__dict__}", + flush=True, + ) + + # Leggiamo i parametri di retry dal task, con fallback robusto + raw_retries = getattr(task, "retries", 0) + raw_retry_delay = getattr(task, "retry_delay", 0) + try: - # Update status to running - task.status = TaskStatus.RUNNING - with self.status_manager as sm: - sm.set_task_status(dag_id, task_id, "running", execution_id) + max_retries = int(raw_retries or 0) + except (TypeError, ValueError): + max_retries = 0 - if status_callback: - status_callback() + try: + retry_delay = int(raw_retry_delay or 0) + except (TypeError, ValueError): + retry_delay = 0 - # Context per-THREAD per questo task - if db_handler: - db_handler.set_context(dag_id, execution_id, task_id) + # Log di debug per capire cosa vede l’orchestratore + self.logger.info( + f"Task {task_id}: configured retries={max_retries}, retry_delay={retry_delay}" + ) - # (opzionale ma comodo) passa info alla Task - task.dag_id = dag_id - task.execution_id = execution_id + attempt = 0 - # Execute the task - self.logger.info(f"Executing task: {task_id}") - executor_instance = self.executor_factory.get_executor(task.executor) - task.execute(executor_instance) + while True: + attempt += 1 - # Update status to completed - task.status = TaskStatus.COMPLETED - with self.status_manager as sm: - sm.set_task_status(dag_id, task_id, "completed", execution_id) + try: + # Stato: running + task.status = TaskStatus.RUNNING + with self.status_manager as sm: + sm.set_task_status(dag_id, task_id, "running", execution_id) - if progress_tracker: - progress_tracker.increment_completed() - if status_callback: - status_callback() + if status_callback: + status_callback() - except Exception as e: - # Update status to failed - task.status = TaskStatus.FAILED - with self.status_manager as sm: - sm.set_task_status(dag_id, task_id, "failed", execution_id) + # Context per logging DB per questo thread/task + if db_handler: + db_handler.set_context(dag_id, execution_id, task_id) - if status_callback: - status_callback() + # Passa info alla task (utile per PythonTask / BashTask) + task.dag_id = dag_id + task.execution_id = execution_id - # Re-raise the exception to be caught by the main loop - raise - finally: - # IMPORTANTISSIMO: svuota il contesto per il thread - if db_handler: - db_handler.clear_context() + # Log del tentativo + self.logger.info( + f"Executing task: {task_id} (attempt {attempt}/{max_retries + 1})" + ) + + executor_instance = self.executor_factory.get_executor(task.executor) + task.execute(executor_instance) + + # Se arrivo qui: successo + task.status = TaskStatus.COMPLETED + with self.status_manager as sm: + sm.set_task_status(dag_id, task_id, "completed", execution_id) + + if progress_tracker: + progress_tracker.increment_completed() + if status_callback: + status_callback() + + return # task completata con successo → si esce + + except Exception as e: + # Fallimento del tentativo corrente + self.logger.error(f"Task {task_id} failed on attempt {attempt}: {e}") + + task.status = TaskStatus.FAILED + with self.status_manager as sm: + sm.set_task_status(dag_id, task_id, "failed", execution_id) + + # Verifica se abbiamo ancora tentativi disponibili + if attempt <= max_retries: + # Logga e attendi prima del retry + self.logger.info( + f"Retrying task {task_id} in {retry_delay} seconds " + f"({attempt}/{max_retries})" + ) + if retry_delay > 0: + time.sleep(retry_delay) + # e poi riparte il while True (nuovo tentativo) + continue + else: + # Nessun retry rimasto → fallimento definitivo + if status_callback: + status_callback() + raise Exception(f"Task {task_id} failed: {e}") from e + + finally: + # Pulizia del contesto di logging + if db_handler: + db_handler.clear_context() def _handle_cancellation( diff --git a/src/maestro/server/tasks/base.py b/src/maestro/server/tasks/base.py index 24f0fb7..8abca3e 100644 --- a/src/maestro/server/tasks/base.py +++ b/src/maestro/server/tasks/base.py @@ -1,12 +1,17 @@ from typing import Optional +from pydantic import Field from maestro.shared.task import Task class BaseTask(Task): """Base class for a task, inheriting from the core Task which is a Pydantic model.""" - # 🆕 aggiungiamo questi due campi + # Contesto runtime dag_id: Optional[str] = None execution_id: Optional[str] = None + # 🆕 Campi che devono essere letti dallo YAML + retries: int = Field(default=0, description="Number of retries on failure") + retry_delay: int = Field(default=0, description="Delay in seconds between retry attempts") + def execute_local(self): raise NotImplementedError("Subclasses must implement this method.") diff --git a/src/maestro/server/tasks/python_task.py b/src/maestro/server/tasks/python_task.py index b69470a..04e10c1 100644 --- a/src/maestro/server/tasks/python_task.py +++ b/src/maestro/server/tasks/python_task.py @@ -1,5 +1,6 @@ from typing import Optional from pydantic import Field +import traceback import io import contextlib import threading From 6d7b850993d266a1400d9f147c7038ec286f82e3 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 25 Nov 2025 15:35:35 +0100 Subject: [PATCH 18/38] Fixed retries in Bash tasks --- .vscode/settings.json | 2 +- examples/2_New_examples/2.1.1.Bash_chain.yaml | 37 +- .../2.2.1.bash_chain_plus_scan.yaml | 36 +- .../2.2.2.bash_chain_branch.yaml | 4 +- .../2.2.3.bash_chain_with_error_handling.yaml | 5 +- .../2.3.1.bash_chain_heavy_diamond.yaml | 25 +- .../2.3.2.bash_chain_error_storm.yaml | 7 +- .../2.3.3.bash_chain_long_pipeline.yaml | 20 +- .../2_New_examples/3.1.1.Wait_and_retry.yaml | 12 +- .../3.2.1.wait_and_retry_with_logging.yaml | 36 +- .../3.2.2.wait_and_retry_bash_python_mix.yaml | 31 +- maestro.db | Bin 0 -> 114688 bytes src/maestro/server/internals/orchestrator.py | 13 +- .../server/internals/status_manager.py | 660 ++++++++++++++---- src/maestro/server/tasks/bash_task.py | 24 +- 15 files changed, 705 insertions(+), 207 deletions(-) create mode 100644 maestro.db diff --git a/.vscode/settings.json b/.vscode/settings.json index 9fcd08d..c88816e 100644 --- a/.vscode/settings.json +++ b/.vscode/settings.json @@ -22,7 +22,7 @@ "git.enableSmartCommit": true, "git.confirmSync": false, "git.autofetch": true, - "workbench.colorTheme": "Dracula Theme", + "workbench.colorTheme": "Ubuntu Color VSCode Dark Highlight", "workbench.iconTheme": "material-icon-theme", "errorLens.enabled": true, "breadcrumbs.enabled": true, diff --git a/examples/2_New_examples/2.1.1.Bash_chain.yaml b/examples/2_New_examples/2.1.1.Bash_chain.yaml index 9c86877..875e20c 100644 --- a/examples/2_New_examples/2.1.1.Bash_chain.yaml +++ b/examples/2_New_examples/2.1.1.Bash_chain.yaml @@ -1,18 +1,51 @@ dag: name: "bash_chain" tasks: + + - task_id: "start_list" + type: "PrintTask" + params: + message: "Listing files..." + dependencies: [] + - task_id: "list_files" type: "BashTask" params: command: "ls -l" - dependencies: [] + dependencies: ["start_list"] + + - task_id: "start_counting_elements" + type: "PrintTask" + params: + message: "Counting elements..." + dependencies: ["list_files"] + + - task_id: "counting_elements" + type: "BashTask" + params: + command: "ls -l | wc -l" + dependencies: ["start_counting_elements"] + + - task_id: "start_counting_lines" + type: "PrintTask" + params: + message: "Counting lines..." + dependencies: ["start_counting_elements"] + - task_id: "count_lines" type: "BashTask" params: command: "echo 'Counting lines from previous output' | wc -l" - dependencies: ["list_files"] + dependencies: ["start_counting_lines"] + - task_id: "end" type: "PrintTask" params: message: "Bash chain complete!" dependencies: ["count_lines"] + + + + + + diff --git a/examples/2_New_examples/2.2.1.bash_chain_plus_scan.yaml b/examples/2_New_examples/2.2.1.bash_chain_plus_scan.yaml index 805fdfe..fdf5fa9 100644 --- a/examples/2_New_examples/2.2.1.bash_chain_plus_scan.yaml +++ b/examples/2_New_examples/2.2.1.bash_chain_plus_scan.yaml @@ -1,31 +1,51 @@ abstract: | - This DAG extends the original bash_chain by adding a disk usage check and a Python - file count operation. It tests sequential task execution and mixed Bash/Python usage. + This DAG extends the example 2.1.1 by adding a disk usage check and a Python file count operation. It tests sequential task execution and mixed Bash/Python usage. dag: name: "bash_chain_plus_scan" tasks: + + - task_id: "start_listing_files" + type: "PrintTask" + params: + message: "Listing files, even hidden ones." + dependencies: [] + - task_id: "list_files" type: "BashTask" params: - command: "ls -l" - dependencies: [] # First task: lists files + command: "ls -a" + dependencies: ["start_listing_files"] # First task: lists files + + - task_id: "start_memory_usage" + type: "PrintTask" + params: + message: "Memory usage..." + dependencies: ["list_files"] - task_id: "disk_usage" type: "BashTask" params: command: "du -sh ." - dependencies: ["list_files"] # Depends on list_files + dependencies: ["start_memory_usage"] # Depends on list_files - - task_id: "python_count_files" + - task_id: "python_name_folder" type: "PythonTask" params: - script: | + code: | import os - print("FILE_COUNT:", len(os.listdir('.'))) + print("Folder name:", os.getcwd()) dependencies: ["disk_usage"] # Python post-processing + - task_id: "python_count_files" + type: "PythonTask" + params: + code: | + import os + print("File count:", len(os.listdir('.'))) + dependencies: ["python_name_folder"] # Python post-processing + - task_id: "end" type: "PrintTask" params: diff --git a/examples/2_New_examples/2.2.2.bash_chain_branch.yaml b/examples/2_New_examples/2.2.2.bash_chain_branch.yaml index 32a1c0b..0436c9c 100644 --- a/examples/2_New_examples/2.2.2.bash_chain_branch.yaml +++ b/examples/2_New_examples/2.2.2.bash_chain_branch.yaml @@ -9,7 +9,7 @@ dag: - task_id: "list_files" type: "BashTask" params: - command: "ls -l" + command: "ls -a" dependencies: [] # Root branch A - task_id: "list_processes" @@ -27,7 +27,7 @@ dag: - task_id: "count_processes" type: "BashTask" params: - command: "ps aux | wc -l" + command: "ps aux --no-heading | wc -l" dependencies: ["list_processes"] - task_id: "final_message" diff --git a/examples/2_New_examples/2.2.3.bash_chain_with_error_handling.yaml b/examples/2_New_examples/2.2.3.bash_chain_with_error_handling.yaml index 228ffd4..aaba2ab 100644 --- a/examples/2_New_examples/2.2.3.bash_chain_with_error_handling.yaml +++ b/examples/2_New_examples/2.2.3.bash_chain_with_error_handling.yaml @@ -1,7 +1,6 @@ abstract: | - This DAG tests error handling with fail_fast disabled. A failing BashTask is followed - by Python and Print tasks that continue running regardless of the failure. + This DAG tests error handling with fail_fast disabled. A failing BashTask is followed by Python and Print tasks that continue running regardless of the failure. dag: name: "bash_chain_with_error_handling" @@ -24,7 +23,7 @@ dag: - task_id: "python_info" type: "PythonTask" params: - script: | + code: | print("This runs even after a failure.") dependencies: ["intentional_error"] diff --git a/examples/2_New_examples/2.3.1.bash_chain_heavy_diamond.yaml b/examples/2_New_examples/2.3.1.bash_chain_heavy_diamond.yaml index a4dc303..ce56e66 100644 --- a/examples/2_New_examples/2.3.1.bash_chain_heavy_diamond.yaml +++ b/examples/2_New_examples/2.3.1.bash_chain_heavy_diamond.yaml @@ -6,6 +6,8 @@ abstract: | dag: name: "bash_chain_heavy_diamond" tasks: + + # Introduction - task_id: "start_scan" type: "PrintTask" params: @@ -16,7 +18,7 @@ dag: - task_id: "files_list" type: "BashTask" params: - command: "ls -l" + command: "ls" dependencies: ["start_scan"] - task_id: "files_count" @@ -29,13 +31,13 @@ dag: - task_id: "proc_list" type: "BashTask" params: - command: "ps aux" + command: "ps aux | head" dependencies: ["start_scan"] - task_id: "proc_count" type: "BashTask" params: - command: "ps aux | wc -l" + command: "ps aux --no-heading | wc -l" dependencies: ["proc_list"] # Branch 3 @@ -52,19 +54,26 @@ dag: dependencies: ["disk_usage"] # Branch 4 - - task_id: "simulated_ping" + - task_id: "ready_to_ping" type: "BashTask" params: - command: "echo 'Simulating ping'; sleep 1" + command: "echo 'Trying to ping...'; sleep 1" dependencies: ["start_scan"] + - task_id: "real_ping" + type: "BashTask" + params: + command: "ping -c 4 8.8.8.8" + dependencies: ["ready_to_ping"] + - task_id: "parse_ping" type: "PythonTask" params: - script: | - print("Ping simulation parsed.") - dependencies: ["simulated_ping"] + code: | + print("Ping done - ready for parsing.") + dependencies: ["real_ping"] + # Final merge - task_id: "final_merge" type: "PrintTask" params: diff --git a/examples/2_New_examples/2.3.2.bash_chain_error_storm.yaml b/examples/2_New_examples/2.3.2.bash_chain_error_storm.yaml index b222619..bebec27 100644 --- a/examples/2_New_examples/2.3.2.bash_chain_error_storm.yaml +++ b/examples/2_New_examples/2.3.2.bash_chain_error_storm.yaml @@ -1,8 +1,7 @@ abstract: | - This DAG intentionally triggers multiple failures in parallel branches while - fail_fast=false ensures the run continues. Useful for stress‑testing logging - and resilience mechanisms. + This DAG intentionally triggers multiple failures in parallel branches while fail_fast=false ensures the run continues. + Useful for stress‑testing logging and resilience mechanisms. dag: name: "bash_chain_error_storm" @@ -37,7 +36,7 @@ dag: - task_id: "python_recover" type: "PythonTask" params: - script: | + code: | print("Recovery executed despite failures.") dependencies: ["invalid_cmd_1", "invalid_cmd_2"] diff --git a/examples/2_New_examples/2.3.3.bash_chain_long_pipeline.yaml b/examples/2_New_examples/2.3.3.bash_chain_long_pipeline.yaml index 3090fdc..121011b 100644 --- a/examples/2_New_examples/2.3.3.bash_chain_long_pipeline.yaml +++ b/examples/2_New_examples/2.3.3.bash_chain_long_pipeline.yaml @@ -8,7 +8,7 @@ dag: tasks: - task_id: "t1" type: "PrintTask" - params: { message: "Start long pipeline" } + params: { message: "Start long pipeline..." } dependencies: [] - task_id: "t2" @@ -19,47 +19,47 @@ dag: - task_id: "t3" type: "PythonTask" params: - script: | + code: | print("Step 3 complete") dependencies: ["t2"] - task_id: "t4" type: "BashTask" - params: { command: "sleep 1 && echo 'Pause done'" } + params: { command: "sleep 1 && echo 'Pause done, task 4 completed.'" } dependencies: ["t3"] - task_id: "t5" type: "PrintTask" - params: { message: "Halfway there..." } + params: { message: "Halfway there... Task 5 done." } dependencies: ["t4"] - task_id: "t6" type: "BashTask" - params: { command: "echo 'Running step 6'" } + params: { command: "echo 'Running step 6' && sleep 2 && echo 'Task 6 successfully executed after 2 seconds pause...'" } dependencies: ["t5"] - task_id: "t7" type: "PythonTask" params: - script: | + code: | print("Step 7: Python OK") dependencies: ["t6"] - task_id: "t8" type: "BashTask" - params: { command: "echo 'Step 8'" } + params: { command: "echo 'Step 8'; echo 'Everything works fine...'" } dependencies: ["t7"] - task_id: "t9" type: "BashTask" - params: { command: "sleep 1" } + params: { command: "sleep 1 && echo 'Another pause executed.'" } dependencies: ["t8"] - task_id: "t10" type: "PythonTask" params: - script: | - print("Almost done...") + code: | + print("Almost done... task 10 completed.") dependencies: ["t9"] - task_id: "t11" diff --git a/examples/2_New_examples/3.1.1.Wait_and_retry.yaml b/examples/2_New_examples/3.1.1.Wait_and_retry.yaml index ba90314..faf0d6a 100644 --- a/examples/2_New_examples/3.1.1.Wait_and_retry.yaml +++ b/examples/2_New_examples/3.1.1.Wait_and_retry.yaml @@ -5,12 +5,16 @@ dag: type: "PythonTask" params: code: | - import random, sys - if random.random() < 0.5: + import random, sys, time + generate_random = random.random() + print(f"Generated: {round(generate_random, 2)}") + if generate_random < 0.67: + print("Failure occurred") sys.exit(1) - print("Success on this run!") + else: + print("Success on this run!") retries: 3 - retry_delay: 2 + retry_delay: 1 dependencies: [] - task_id: "final_message" type: "PrintTask" diff --git a/examples/2_New_examples/3.2.1.wait_and_retry_with_logging.yaml b/examples/2_New_examples/3.2.1.wait_and_retry_with_logging.yaml index e2cc5cb..9299824 100644 --- a/examples/2_New_examples/3.2.1.wait_and_retry_with_logging.yaml +++ b/examples/2_New_examples/3.2.1.wait_and_retry_with_logging.yaml @@ -5,40 +5,48 @@ dag: name: "wait_and_retry_with_logging" tasks: - # Messaggio iniziale + # Initial message - task_id: "start" type: "PrintTask" params: message: "Starting retry pipeline..." dependencies: [] - # Primo task instabile (fail random 50%) + # First unstabile task (fail random 40%) - task_id: "unstable_task_1" type: "PythonTask" params: - script: | - import random, sys - if random.random() < 0.5: + code: | + import random, sys, time + generate_random = random.random() + print(f"Task 1 - Generated: {round(generate_random, 2)}") + if generate_random < 0.4: + print("Failure occurred") sys.exit(1) - print("Task 1 succeeded!") + else: + print("Task 1 - Success on this run!") retries: 3 retry_delay: 2 dependencies: ["start"] - # Secondo task instabile (fail random 30%) + # Second unstable task (fail random 70%) - task_id: "unstable_task_2" type: "PythonTask" params: - script: | - import random, sys - if random.random() < 0.3: + code: | + import random, sys, time + generate_random = random.random() + print(f"Task 2 - Generated: {round(generate_random, 2)}") + if generate_random < 0.7: + print("Failure occurred") sys.exit(1) - print("Task 2 succeeded!") - retries: 2 - retry_delay: 1 + else: + print("Task 2 - Success on this run!") + retries: 3 + retry_delay: 2 dependencies: ["unstable_task_1"] - # Task finale + # Final task - task_id: "final_message" type: "PrintTask" params: diff --git a/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml b/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml index dc6d3fe..2c46162 100644 --- a/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml +++ b/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml @@ -9,23 +9,40 @@ dag: type: "BashTask" params: command: | - if [ $((RANDOM % 3)) -eq 0 ]; then - echo "Failing Bash task"; exit 1; + echo "==== Unstable Bash Task - New run ====" + + # Random integer between 0 and 99 + RANDOM_VALUE=$((RANDOM % 100)) + echo "Unstable Bash Task - Generated value: ${RANDOM_VALUE}" + + # 1/3 of probabilities of failure (when RANDOM_VALUE % 3 == 0) + if [ $((RANDOM_VALUE % 3)) -eq 0 ]; then + echo "Unstable Bash Task - FAIL on this run (exit code 1)" + exit 1 else - echo "Bash succeeded!"; + echo "Unstable Bash Task - SUCCESS on this run (exit code 0)" fi retries: 4 - retry_delay: 1 + retry_delay: 2 dependencies: [] - task_id: "unstable_python" type: "PythonTask" params: - script: | - import random, sys + code: | + import random, sys, time + + # Genera un numero casuale tra 0 e 1 + generate_random = random.random() + print(f"Unstable Python Task - Generated: {round(generate_random, 2)}") + + # Usa una soglia di 0.4 per decidere se fallire oppure no if random.random() < 0.4: + # Fallimento esplicito: il processo termina con errore sys.exit("Python failed") - print("Python succeeded") + else: + # Caso di successo + print("Unstable Python Task - Success on this run!") retries: 2 retry_delay: 2 dependencies: ["unstable_bash"] diff --git a/maestro.db b/maestro.db new file mode 100644 index 0000000000000000000000000000000000000000..0035e55ae88f7d709973afa6684e6a9025a1d85f GIT binary patch literal 114688 zcmeHQe{3YzUEj66Uhn$HUVb>vuS@2fUTxPm>-q8Kmy>JHx%iT6FL&wrTw29pnVorS zPqLnIcW3X;N-ycA(2@$;Lr@SPAOZ=5_yZ&)5C{n*R8flfAyon-pimG{1X3wTfK*ZV zzM0+i?#%AR-dp=F$#)a`yz^$>eBS4Mzu))1Z{GJiYahMf_>?Pom2K1K+tg^{4_tK@W)8=3;dgG`@7elUr*$(zB!rupF}qE*+lMm?oTKGXzH62zd1fW z@k7~%vR{RR_(2Ac0b~FfKn9QjWZ*r?K>b)YJ#&dq)2>FN!Q81@Thy+VX{GS8 zSF9FXdfhKnYHq=IwrR9<{Osk`l`E^<`HRo2ew=&T%ehMzxo8#Ke8(LY>iMzs49_Ru z$oXcm9J^!m@=OE4m6fM2tahn9n{jOJV=I@>KDTmtUNpo7?&75@+{N_^7fx~6OvkOc za~CeHTw#Zy;f1Pi!u<5<3zwemxZ;J&=bvA>{0Z(OtDoTJ9eW|W@S)7;%o?9e1PoR` zS%ztC!L0dS@IB6_7;QoJjn_ugGcz;EA9yinv2Rvift$mxbVu8QL)?6OgF2Ae=4N59 ztsO=5I<;!P;ZJ;7G-%b^#{m&p;1fIsg1BJ8XH|J7wCXz~8D+({8Ju zo$a_)S_$s!Rcu-TF}bq(?CND!6yfm{7s5ZqwPRgiXP&!sdG-9W7a8O8;YkbJ<<)bm zmsc;IU0q|>RfC?0p4(AT`xUK%@Ve~6LV9%O^h~cF40tby!9Qbf+>=VrOiw4jKNa*< z+1sqLpV5wELJYcz@!2NYc6k+Vl1j zdO&jIk0@aPEH22Kvxs~+`SGcq5mqF346qo^6yR!293tf~KjNshR=**MTy_(6O z7DV=Y7>3y~Hq@rzC!6~d_UEIy@8td~_wC%j<^DPM_qo5x{blZJVF4?|4>EuZAOpw% zGJp&q1IPd}fD9l5$N(~c4BSNwJeUPAGa2~x0CVu~Ps6YK*rWSGMD90oU(LOlJC&>DF64eT_kFp)$~~F;78K$K89)Y*0b~FfKn9QjWB?gJ z29N<{02z3XF)%fJBKfjcF7FmBui}g!8$Ox5PF=?=7ihWcRI~RDFD6}Dx$4!bg)OrR z`*Zxv{lk3Huhgo(<8H!6t}VCfxnu0avRT{QVh4+`!N++e%}%sDuUv4Q%`JAeS*D{i z!*bHwfvqGCRA3b<9(Daq>ZVu)MYv_PR4W&@Jnt&4j4(JwuWT1K%}ul7xWnwEnp>%r zN@dtUbPYBfRELfY&t;mJYgXB@C!r6=Gk=lDRi^%aYJKv{lj6i%6UWA1&weXggv0nj z29N<{02x3AkO5=>8F*(fP#+r~TR-vm%F1F&Q%a^RTa-5hQ{jniQ(o7p#7iQPiDAn{ zOOf|xwXKcik{~I(DDslRiORA>mL(~#6GPJVc!`o1Rh@{cC<~II3#uR(iYlsC>*kN2 z`E)&-9a}#cKv9H}K_x}!O~J5uqU$F9u*_+muC3w0ObPRh|^7z!wEoPhqhBrhvU zUV-b40t6WXZeep!Xs`G)+)?p=jF@iFl5^ z@JP#w0kW$~Dv99{guEyb5QFY`vcoNvS1oE16PTf5kq3)YvTdio@$`FCEKtiQH^-a zz3~t&ZwQ3MBwAV~dLG0_0N%UW!|~+U`dlbLHq~rJvw#w?%7hec-qa+U7bUBx$wbf% zp=A#PZ@o7fSp89)Y*0b~FfKn9QjWZ)gkfIPB(B6+ZNyK!Y?{bcf>3v5|^ zWPLH&$4Xen7+L3&{VZ9H1-I;DZ7ChzvX4ce(Wge%89)Y*0b~Ff zc=s^y@Nu^P|3vz!#Kd1t{^B^FekxlS`@Qk+X1*}>O8PrFW3n(cH~t&x-^%_|=GloK z&dyBz$JnQGzccy300(@#=bjoCiBNrzdWYqYVRIauqDUJG zC)Sg;=hD6S>HX~5JaIO)2p85|c!3hUTCK3N>u-7P!Rp{vQgiKf6wT_^!D@JXIW>JU zgx?->y`L?lkFBN_;nGM04qD6cv#GiET38bfTF0ZOQ`3uKBO2=_Q*-S#vG(*+&jWki z6Nwbx&j#S>uHDl|MfZ28GA!E2(KYl(#N% z>igL+ox6})gc>3u*Jou@A5P7+SH`5S&#ET(1TPZpzMq}k6I}(bkBY{-2wor6WcN&U zd#t>&kGe3~BR5S8qYI?47^b8{O z_p89)Y*0b~Ffc&9VKp8uo&|4#2k#E%Rh1IPd}fD9l5$N(~c3?Ku@05X6K+?oOO z|8LC?mLdbl05X6KAOpw%GJp&q1IPd}fD9l5?*<0g{{Lk9hZFEWevkoV02x3AkO5@i zPB3tDCz+mm>v%mUcLW2zI%}Jo1;?IU=4LwzW>0amkVBy27kp=%La|#bm*JRI@!Uey z+M;%?+;-SEtFKhytXF3{Vg`dFtgLa1&O)e;-M!?nL_mEEh9In-Kt&-Qo0iR8s7rz)@XF zl$W05e5hc<<+xj+J-2fH0_VAKu2T(ony1$tpR+)2IB|hth3nZ75n5fY(zmN`ZT;-o z)wQ)=RRVOVvgB;IOho+M9i|GitiM=y&E9~6Em03sa6tL1v)weIQ4CkyHM+2eooUbP z;i?8xgl?)~2b4P7J6^@-DuJY*;;OsVQyeo$p#p?5%7I$AwQXAQILoZo%rfQtiV3m< zUBBUOHsrVv*1?@_RF)6F&w~)|R6vF1OB=H-MR*}do6^+1mMTCWR=k>P&qpdf#YqdG z5Kt9U%k`=WkW8-XZI&IAvmH*z6K)4I)TWkWQ&5U31^p{wmFxKs0J`sYgSNYc<+#+o(!#pC%3E1iJk27%7A4)I2bpm||x$Ox)1_NBIeOFp#A& z^#9TSchUd%doLcpGyQ*}h5r8+6X`F$$LC+zd}II_Kn9QjWZ=#*aMK-5pM2{PSOsWL zuiU-_(0&~Ke|-Lr&;Q#WXmV{^3ef*Y|DSD+X>2OM=l}TpA2!P!(&n*urvI*IX(OPPPj zYHGZDmu51F#r3)A*H;!J)0+LfhqD!) z#X_9UUp#l|#b)%@OD|S^+pGC61+jK-AC1@Ea)m`cC`MkFiL6VaEJ%i?>k?5FLnrn7 zAIKz5u1`;!fY>YBg-vtQ1ijkNyEa)2On~x*X4J+@+^xet;;`hrA&7>cM`7{P4A$cG zjf2B#u^1{sh%ktqVmo?X+9NLv^JxZDbqg2jhF2F9O%|hg=kE{kR*wvC+W;WsMY%gv zK~rPY&fdqM&P~5^8&Kl{%ANIC04$5mY(#0nyuG~J`GBGuNKGULF`|5)zBe0ytUo?b zlr1%CMb$A4Yl*;cS4C8Qej_9JIL&8ldWNeG`Tn4FGd{bPZ@p zaMReTVr(!DAQMrE&35=?C>daE4+U|brk3H=jkqK;|Nnu+A0%=gnfk%WznhdMULXI* z@ulq5%&%sK$3B|=xzTTqemM0LBi|S~KI{*DdFX*;5l+7QKJ^FrOo9(b49oM%1=rc! z^5N+KBpM;QC>0IS;4MLxcw*=puWOdX+p<9=kw}uNNVh#`$jh>l*N7kqF^LdQF=)Y{ z(GRr48#e;XszwwE#=6c`kW-5xW*9e)5OYiM*wCRzXTTv`RafFlcj^Oo97@wFu1|!aa-&TW&jC*DzNAv$7bv+^Vt9>8$bq(bg*McqYY}8JZ9GU+BF3Xx&CB&vYI`$G|dVdStt_cfA)M9YSt#n#atIhjewp$YbOApet7 z^&P8F@gNn_CUp(?h_;~cs%R=a0jtcLB~9T4N)%Bwh-FEGu)?B|7l=-j*i4wbkD&{! zaG&T7(-4b#-q02BH=}Z}^h5|74Dr#zZb@&j8WEiU$;%8eEStwColCQK17azzYoeMr zh$ad#ju%ca#DO`!U5*E3kpZi&XhM8O{0x^3uriB0c(8*u%0V+^Ly65V)W;jm0Hb`= z&1gw>(=IE~hXo+rjDGmLbRX9 za08ee3$PR*Gb0k^Q+PC7X=Tr~4vy^bOmgsq(2$9!#UL*}5<&)(d~}doLd(o?OkhK_ z5}3ExsC@F_aPf~l-MS5e2V>a6=dOgns1X%}GY{=84i6USV2xU%38D=1`iRpcJ-D|x ze3VUS$wt$l!PD%vOn-aNyEX7eBA~S&sO>j_u|i0<}iUjm4hq5qHmKl=aZ|D*qp z{y+Nvo$rc8|9>C;zl{0+?$#M0;z9MbtkN!XU|9-)A?Lvk6m0g%T?8E;jB+UP(C2|+1esc29C-W2a z@vn_@+0SOamN`E5iS)0fhej`^em#{OxiI|Z&^LzUQzplp(xC z5lRqLuRv_4U|2lSb)7F3H4|d(?V?zeEuo~Ut)QKFOi3$Lqptuc!h#Zs4yn9iF*`G7 zTLDlmtz7kL)xwrph4gH`Y@4#Ei54%Iy2_Ix75Ji{s=R5bvgXW1f?QYt6p6;8hG<0=LmLTjsh>R4 z3V^yDv~gkWCp%Vv6T-K|c*OUa5A6j&4HjritJYn30ZOq%&k&!lJovb-Fi6(u=F@CT$2Etuasgf^c3fsnie`X+DCcEAqG z_ySH*Rt3nQ6ER$8guTh+V1W+W_6ALm#dyYZX=!gVd6Z3P33Ai)N@~pXGSCvFHHi%v zOZiYf!1Nvx<~}WNfLK=KsIkEP|L={*OV~%pIUjk@9j{ZOT|LFgt|BwDZ`v2(vqyK+k|6kOZ|9{WO z&$9UcFHHW+88|`)>i4c^ z5+}mA(3ib(dADGB6-QG_rYu_!+AgqH$=f#Nb)8DQBodhzwp_Fn`L;ts;r%E|UQi81 zj=kq~@d^VKyh{FHpbjq@6ws2q0ciDj!1!8-7NS9q1Z^vRjs=1W(a81&P*s*=F^k+~ z1~iDR>=Wo z1MC1%-r8WTt)D%+y0+HKW(jed72ub35*uQdRvY+Xh&?j=EsAp|1@+m;6fNx07d%&>iTwReku;%-qTK$5&PoV=dih0(wy1EpK}VG67OM zX6jitYpS5KTp;bacY;KNEYqB<)>Fqbiy?|k3q_cWZ=g67P!{OEZa54=ROPitGSiCz z8A)cJBO68cljN{B4MFZgNmYf|?O6#+=VJ`j#~#ish5({G3>u3HsW$}3j27tNDpG2c~_miuwSEmH2l!yjAMDHJ# z7$P`=L-ou|W-&w%?UaFoku?~M;+aw)Q=Dw5ilo%j4`k*-ESO?RkDA9-4O_ezjPN`SE~cKe_NvT3TID);wNl|f^HbW4U1&m6{bF{v>3QtG4k zbytJ$Y^MR4djWTYrL*m5~P)d zY_lSfB#^M)ogyoFP1cAUCxVp_B1i;O<(^)uGH@^|G+2(>a(NlRz@(U{Vm*DVyQ&Nr ziJVtpG5|dj?G_nI6j@i4`cN)2E$^8yZ^w7G9mqIo6)GNe{Y~mZ4mLxy1%+2dwirhV zv)rV$qv5lU1+)3w(oI&4`HQZQFbwQ^z0^RCj0L=84n zQeHH5=o_)9!jh>7U1gfEUMduATOz&HMu6qJz<|WKU?pr>R0mr&-piH^5R0^|7?2|q zEJ)O>hb2UWPkl7o-Ifg+h%SL8iS|hVL}v-FwO2El>2O9EOvQ>`*)D9Fn`Xsv71g39 zF+oMBSma@~y2!(lt<75{s|cC1Wf(-YUL68=kn_3$O4Wq~InPwc(1z-%v0n4hfdbJC z$iW;n=pjTnZjkzLy8G;WpeVq;Xi8T)G|`Y%^4e&Jdeq!XtyF>qwZb)6R;!wnYD6oR zAf+aO)ZZ`=@!-;FJk=DUlx)M6M74Ks5FJJ}!GMve(|k(HL5V?BFkNG*Ug|MuEMT(4 z*sLeOBFG9LtdEX#SC7F0(LfLdO^JF;00c^80%`s}HJq8A+Z$4vUQ*2}J3u+rY(=v` zE8w9HAw`=vHOb~h$tr5>YD2h1kHD~{f`Y_+18EsbG(k~xn0$>6WfsHRM+cJuBY|9( zByJ@bMle^pQ6EWW=0YHy!^j|Eh~U4%lYr<54}^$$1*WNnFgf+}iIK@)NaQ{<_1M(^ z!XNlS29N<{02x3AkO5=>89)Y*0b~FfxKj-5p2$p}*mIZ-2i}Fg(6ydw*i<$HC)kSaXNPn4vKK|3?$) zA3Y?Fv07vR89)Y*0b~Ff*bf6YUmj1Ncxwi70PIad+A{#`9mV*6jQ_{@|K9s{H)c27 zlC$A5J@I#U=0v2&KT zyluKIe~*{mJJv75e0Hw z2Tsh}&^?XxVWoS8`mZx<&1LsNtb%u?w5l>IEkCOu}bx+_=#) R9R1pn_C*%900FUA{|}Q 0: time.sleep(retry_delay) - # e poi riparte il while True (nuovo tentativo) continue else: - # Nessun retry rimasto → fallimento definitivo if status_callback: status_callback() raise Exception(f"Task {task_id} failed: {e}") from e + finally: # Pulizia del contesto di logging if db_handler: diff --git a/src/maestro/server/internals/status_manager.py b/src/maestro/server/internals/status_manager.py index 2c016d7..c865698 100644 --- a/src/maestro/server/internals/status_manager.py +++ b/src/maestro/server/internals/status_manager.py @@ -1,88 +1,171 @@ -import threading -from typing import Dict, Optional, List, Any -from datetime import datetime, timedelta -import re +import json import random +import re import string -import json +import threading +from datetime import datetime, timedelta +from typing import Any, Dict, List, Optional + +"""Internal status manager used by the server. + +This module provides a simple SQLite-backed StatusManager used to persist +information about DAGs, executions, tasks and logs. The manager is used by +the server to record runtime status and expose queries for monitoring and +control operations. +""" -from sqlalchemy.orm import sessionmaker from sqlalchemy import create_engine, func +from sqlalchemy.orm import sessionmaker -from .models import DagORM, ExecutionORM, TaskORM, LogORM, Base - -# Docker-like name generator - lists of adjectives and nouns -DOCKER_ADJECTIVES = [ - "amazing", "awesome", "blissful", "bold", "brave", "charming", "clever", "cool", "dazzling", "determined", - "eager", "ecstatic", "elegant", "epic", "exciting", "fantastic", "friendly", "gallant", "gentle", "gracious", - "happy", "hardcore", "inspiring", "jolly", "keen", "kind", "laughing", "loving", "lucid", "magical", - "modest", "naughty", "nervous", "nice", "objective", "optimistic", "peaceful", "pedantic", "pensive", "practical", - "quirky", "relaxed", "romantic", "serene", "sharp", "stoic", "sweet", "tender", "thirsty", "trusting", - "unruffled", "upbeat", "vibrant", "vigilant", "wonderful", "xenial", "youthful", "zealous", "zen" -] - -DOCKER_NOUNS = [ - "albattani", "allen", "almeida", "antonelli", "archimedes", "ardinghelli", "aryabhata", "austin", "babbage", "banach", - "banzai", "bardeen", "bartik", "bassi", "beaver", "bell", "benz", "bhabha", "bhaskara", "black", - "blackburn", "blackwell", "bohr", "booth", "borg", "bose", "bouman", "boyd", "brahmagupta", "brattain", - "brown", "buck", "burnell", "cannon", "carson", "cartwright", "cerf", "chandrasekhar", "chaplygin", "chatelet", - "chatterjee", "chebyshev", "cohen", "chaum", "clarke", "colden", "cori", "cray", "curran", "curie", - "darwin", "davinci", "dewdney", "dhawan", "diffie", "dijkstra", "dirac", "driscoll", "dubinsky", "easley", - "edison", "einstein", "elbakyan", "elgamal", "elion", "ellis", "engelbart", "euclid", "euler", "faraday", - "feistel", "fermat", "fermi", "feynman", "franklin", "gagarin", "galileo", "galois", "ganguly", "gates", - "gauss", "germain", "goldberg", "goldstine", "goldwasser", "golick", "goodall", "gould", "greider", "grothendieck", - "haibt", "hamilton", "haslett", "hawking", "heisenberg", "hermann", "herschel", "hertz", "heyrovsky", "hodgkin", - "hofstadter", "hoover", "hopper", "hugle", "hypatia", "ishizaka", "jackson", "jang", "jemison", "jennings", - "jepsen", "johnson", "joliot", "jones", "kalam", "kapitsa", "kare", "keldysh", "keller", "kepler", - "khorana", "kilby", "kirch", "knuth", "kowalevski", "lalande", "lamarr", "lamport", "leakey", "leavitt", - "lederberg", "lehmann", "lewin", "lichterman", "liskov", "lovelace", "lumiere", "mahavira", "margulis", "matsumoto", - "maxwell", "mayer", "mccarthy", "mcclintock", "mclaren", "mclean", "mcnulty", "mendel", "mendeleev", "menshov", - "merkle", "mestorf", "mirzakhani", "moore", "morse", "murdoch", "moser", "napier", "nash", "neumann", - "newton", "nightingale", "nobel", "noether", "northcutt", "noyce", "panini", "pare", "pascal", "pasteur", - "payne", "perlman", "pike", "poincare", "poitras", "proskuriakova", "ptolemy", "raman", "ramanujan", "ride", - "montalcini", "ritchie", "robinson", "roentgen", "rosalind", "rubin", "saha", "sammet", "sanderson", "shannon", - "shaw", "shirley", "shockley", "shtern", "sinoussi", "snyder", "solomon", "spence", "stallman", "stonebraker", - "sutherland", "swanson", "swartz", "swirles", "taussig", "tereshkova", "tesla", "tharp", "thompson", "torvalds", - "tu", "turing", "varahamihira", "vaughan", "visvesvaraya", "volhard", "wescoff", "wilbur", "wiles", "williams", - "williamson", "wilson", "wing", "wozniak", "wright", "wu", "yalow", "yonath", "zhukovsky" -] +from .models import Base, DagORM, ExecutionORM, LogORM, TaskORM + +# Generatore di nomi in stile Docker - liste compatte di aggettivi e sostantivi +DOCKER_ADJECTIVES = """ +amazing awesome blissful bold brave charming clever cool dazzling determined eager +ecstatic elegant epic exciting fantastic friendly gallant gentle gracious happy +hardcore inspiring jolly keen kind laughing loving lucid magical modest naughty +nervous nice objective optimistic peaceful pedantic pensive practical quirky +relaxed romantic serene sharp stoic sweet tender thirsty trusting unruffled +upbeat vibrant vigilant wonderful xenial youthful zealous zen +""".split() + +DOCKER_NOUNS = """ +albattani allen almeida antonelli archimedes ardinghelli aryabhata austin babbage +banach banzai bardeen bartik bassi beaver bell benz bhabha bhaskara black +blackburn blackwell bohr booth borg bose bouman boyd brahmagupta brattain brown +buck burnell cannon carson cartwright cerf chandrasekhar chaplygin chatelet +chatterjee chebyshev cohen chaum clarke colden cori cray curran curie darwin +davinci dewdney dhawan diffie dijkstra dirac driscoll dubinsky easley edison +einstein elbakyan elgamal elion ellis engelbart euclid euler faraday feistel +fermat fermi feynman franklin gagarin galileo galois ganguly gates gauss germain +goldberg goldstine goldwasser golick goodall gould greider grothendieck haibt +hamilton haslett hawking heisenberg hermann herschel hertz heyrovsky hodgkin +hofstadter hoover hopper hugle hypatia ishizaka jackson jang jemison jennings +jepsen johnson joliot jones kalam kapitsa kare keldysh keller kepler khorana +kilby kirch knuth kowalevski lalande lamarr lamport leakey leavitt lederberg +lehmann lewin lichterman liskov lovelace lumiere mahavira margulis matsumoto +maxwell mayer mccarthy mcclintock mclaren mclean mcnulty mendel mendeleev menshov +merkle mestorf mirzakhani moore morse murdoch moser napier nash neumann newton +nightingale nobel noether northcutt noyce panini pare pascal pasteur payne +perlman pike poincare poitras proskuriakova ptolemy raman ramanujan ride montalcini +ritchie robinson roentgen rosalind rubin saha sammet sanderson shannon shaw +shirley shockley shtern sinoussi snyder solomon spence stallman stonebraker +sutherland swanson swartz swirles taussig tereshkova tesla tharp thompson +torvalds tu turing varahamihira vaughan visvesvaraya volhard wescoff wilbur +wiles williams williamson wilson wing wozniak wright wu yalow yonath zhukovsky +""".split() class StatusManager: + """Manage persistent status for DAGs, executions, tasks and logs. - # 🆕 Variabile di classe per conservare l'istanza corrente + The class provides methods to create and update execution records, + set and query task statuses, record logs, and perform housekeeping. + A single global instance is stored on the class and can be retrieved + with `get_instance()` once initialized. + """ + + # 🆕 Class variable to hold the current singleton instance _instance = None + # ---------------------------------------------------------------------- + + # Construction and initialization + def __init__(self, db_path: str): self.db_path = db_path self.engine = self._get_engine() self.Session = self._get_session_factory() self._initialize_tables_once() - # 🆕 Memorizza l'istanza globale + # 🆕 Store the global singleton instance StatusManager._instance = self + # --------------------------- Critical notes --------------------------- + # + # - Sessions: this class uses SQLAlchemy `Session` and `Session.begin()` + # context managers to ensure transactional boundaries. Callers that + # need long-running or multi-step transactions should obtain their + # own session via `StatusManager.Session()`. + # + # - Concurrency: methods record `thread_id` and `pid` for executions + # and logs, but there is no explicit cross-process locking. If you + # run multiple processes against the same SQLite file consider using + # a server-backed DB (Postgres) to avoid SQLite locking/contention. + # + # - Task fallback: `get_task_status` intentionally falls back to the + # non-execution-scoped task record (execution_id=None) when a + # per-execution record is missing. This supports resume flows where + # tasks may have been persisted before the execution record existed. + # + # - Bulk updates: methods such as `mark_incomplete_tasks_as_failed` + # use `synchronize_session=False` for performance; this avoids + # keeping SQLAlchemy session state in sync and is acceptable here + # because sessions are short-lived and immediately committed. + # + # - ID generation: `generate_unique_dag_id` retries a fixed number of + # times then appends a random suffix. It is designed for convenience + # and not for cryptographic uniqueness. + # + # --------------------------------------------------------------------- + + # ---------------------------------------------------------------------- + # Private method: _get_engine def _get_engine(self): - return create_engine(f'sqlite:///{self.db_path}') + return create_engine(f"sqlite:///{self.db_path}") + # ---------------------------------------------------------------------- + # Private method: _get_session_factory def _get_session_factory(self): return sessionmaker(bind=self.engine) + # ---------------------------------------------------------------------- + # Private method: _initialize_tables_once def _initialize_tables_once(self): with self.engine.connect() as connection: Base.metadata.create_all(self.engine) + # ---------------------------------------------------------------------- + # Metodi del context manager + + # ---------------------------------------------------------------------- + # Metodo: __enter__ def __enter__(self): + """Enter a context and create a new DB session. + + Returns the manager instance with an open `session` attribute. + """ self.session = self.Session() return self + # ---------------------------------------------------------------------- + # Metodo: __exit__ def __exit__(self, exc_type, exc_val, exc_tb): + """Close the previously opened session when exiting the context.""" if self.session: self.session.close() - def set_task_status(self, dag_id: str, task_id: str, status: str, execution_id: str = None): + # ---------------------------------------------------------------------- + # Gestione stato task + + # ---------------------------------------------------------------------- + # Metodo: set_task_status + def set_task_status( + self, dag_id: str, task_id: str, status: str, execution_id: str = None + ): + """Create or update a task record and set its status. + + If the task does not exist it is created. When a task becomes + `running` the `started_at` timestamp is set; when it reaches a + terminal state (`completed`, `failed`, `cancelled`, `skipped`) + the `completed_at` timestamp is set. + """ with self.Session.begin() as session: - task = session.query(TaskORM).filter_by(dag_id=dag_id, id=task_id, execution_id=execution_id).first() + task = ( + session.query(TaskORM) + .filter_by(dag_id=dag_id, id=task_id, execution_id=execution_id) + .first() + ) if not task: task = TaskORM(dag_id=dag_id, id=task_id, execution_id=execution_id) session.add(task) @@ -92,44 +175,99 @@ def set_task_status(self, dag_id: str, task_id: str, status: str, execution_id: elif status in ["completed", "failed", "cancelled", "skipped"]: task.completed_at = datetime.now() - def get_task_status(self, dag_id: str, task_id: str, execution_id: str = None) -> Optional[str]: + # ---------------------------------------------------------------------- + # Metodo: get_task_status + def get_task_status( + self, dag_id: str, task_id: str, execution_id: str = None + ) -> Optional[str]: + """Return the status for a specific task. + + Looks up the task by `dag_id` and `task_id`. If an `execution_id` + is provided the lookup is scoped to that execution; if the record + is missing for the given execution the method falls back to the + globally-scoped task (execution_id=None) to preserve resume flows. + + Returns the status string, or `None` if the task is not found. + """ with self.Session() as session: - task = session.query(TaskORM).filter_by( - dag_id=dag_id, - id=task_id, - execution_id=execution_id - ).first() + task = ( + session.query(TaskORM) + .filter_by(dag_id=dag_id, id=task_id, execution_id=execution_id) + .first() + ) if not task and execution_id is not None: - # Fall back to the task status without execution scoping when - # a specific execution record is missing. This mirrors the - # behaviour of earlier, file-based implementations and keeps - # resume flows working when tasks were persisted before the - # execution record was created. - task = session.query(TaskORM).filter_by( - dag_id=dag_id, - id=task_id, - execution_id=None - ).first() + # Ripiega sullo stato del task senza scoping per esecuzione + # quando manca un record specifico per l'esecuzione. Questo + # replica il comportamento delle implementazioni precedenti + # basate su file e mantiene funzionanti i flussi di resume + # quando i task sono stati persistiti prima della creazione + # del record di esecuzione. + task = ( + session.query(TaskORM) + .filter_by(dag_id=dag_id, id=task_id, execution_id=None) + .first() + ) return task.status if task else None + # ---------------------------------------------------------------------- + # Method: get_dag_status def get_dag_status(self, dag_id: str, execution_id: str = None) -> Dict[str, str]: + """Return a mapping `{task_id: status}` for all tasks of a DAG. + + If `execution_id` is provided the query is scoped to that + execution; otherwise all tasks for the DAG are returned. + """ with self.Session() as session: - tasks = session.query(TaskORM).filter_by(dag_id=dag_id, execution_id=execution_id).all() + tasks = ( + session.query(TaskORM) + .filter_by(dag_id=dag_id, execution_id=execution_id) + .all() + ) return {task.id: task.status for task in tasks} + # ---------------------------------------------------------------------- + # Method: reset_dag_status + # Delete all task records (and executions when execution_id not provided). def reset_dag_status(self, dag_id: str, execution_id: str = None): + """Reset stored status for a DAG. + + If `execution_id` is provided only tasks for that execution are + removed. Otherwise all tasks and execution records for the DAG + are deleted. + """ with self.Session.begin() as session: if execution_id: - session.query(TaskORM).filter_by(dag_id=dag_id, execution_id=execution_id).delete() + session.query(TaskORM).filter_by( + dag_id=dag_id, execution_id=execution_id + ).delete() else: session.query(TaskORM).filter_by(dag_id=dag_id).delete() session.query(ExecutionORM).filter_by(dag_id=dag_id).delete() - def create_dag_execution(self, dag_id: str, execution_id: str, dag_filepath: Optional[str] = None) -> str: + # ---------------------------------------------------------------------- + # Deliver status information about DAG + + # ---------------------------------------------------------------------- + # Method: create_dag_execution + def create_dag_execution( + self, dag_id: str, execution_id: str, dag_filepath: Optional[str] = None + ) -> str: + """Create a new Execution record for `dag_id` and mark it running. + + The method records start time, thread id and pid. If the DAG record + does not exist it is created. Returns the `execution_id`. + """ with self.Session.begin() as session: - execution = ExecutionORM(id=execution_id, dag_id=dag_id, status="running", started_at=datetime.now(), thread_id=str(threading.current_thread().ident), pid=threading.current_thread().ident) + execution = ExecutionORM( + id=execution_id, + dag_id=dag_id, + status="running", + started_at=datetime.now(), + thread_id=str(threading.current_thread().ident), + pid=threading.current_thread().ident, + ) session.add(execution) dag = session.query(DagORM).filter_by(id=dag_id).first() if not dag: @@ -139,26 +277,70 @@ def create_dag_execution(self, dag_id: str, execution_id: str, dag_filepath: Opt dag.dag_filepath = dag_filepath return execution_id - def create_dag_execution_with_status(self, dag_id: str, execution_id: str, status: str) -> str: + # ---------------------------------------------------------------------- + # Method: create_dag_execution_with_status + def create_dag_execution_with_status( + self, dag_id: str, execution_id: str, status: str + ) -> str: + """Create a new Execution record with an explicit status. + + This variant allows creating executions in non-running states + (for example `created` or `failed`). If `status` is not `created` + the `started_at`/`thread_id` are set as well. + """ with self.Session.begin() as session: - execution = ExecutionORM(id=execution_id, dag_id=dag_id, status=status, pid=threading.current_thread().ident) + execution = ExecutionORM( + id=execution_id, + dag_id=dag_id, + status=status, + pid=threading.current_thread().ident, + ) if status != "created": execution.started_at = datetime.now() execution.thread_id = str(threading.current_thread().ident) session.add(execution) return execution_id - def initialize_tasks_for_execution(self, dag_id: str, execution_id: str, task_ids: List[str]): + # ---------------------------------------------------------------------- + # Method: initialize_tasks_for_execution + def initialize_tasks_for_execution( + self, dag_id: str, execution_id: str, task_ids: List[str] + ): + """Insert initial TaskORM rows for a new execution. + + Each task is created with status `pending` and an `insertion_order` + so callers can rely on a deterministic ordering when needed. + """ with self.Session.begin() as session: for i, task_id in enumerate(task_ids): - task = TaskORM(id=task_id, dag_id=dag_id, execution_id=execution_id, status="pending", insertion_order=i) + task = TaskORM( + id=task_id, + dag_id=dag_id, + execution_id=execution_id, + status="pending", + insertion_order=i, + ) session.add(task) + # ---------------------------------------------------------------------- + # Method: update_dag_execution_status def update_dag_execution_status(self, dag_id: str, execution_id: str, status: str): + """Update the status of an existing execution record. + + If transitioning to `running` and `started_at` is missing, the + method sets `started_at`, `thread_id` and `pid`. When a terminal + status is set (`completed`, `failed`, `cancelled`) the + `completed_at` timestamp is recorded. + """ with self.Session.begin() as session: - execution = session.query(ExecutionORM).filter_by(dag_id=dag_id, id=execution_id).first() + execution = ( + session.query(ExecutionORM) + .filter_by(dag_id=dag_id, id=execution_id) + .first() + ) if execution: - # Se la DAG passa da "created" a "running" e non ha started_at, impostalo ora + # If the DAG transitions from "created" to "running" and + # `started_at` is not set yet, set it now. if status == "running" and execution.started_at is None: execution.started_at = datetime.now() execution.thread_id = str(threading.current_thread().ident) @@ -169,18 +351,33 @@ def update_dag_execution_status(self, dag_id: str, execution_id: str, status: st if status in ["completed", "failed", "cancelled"]: execution.completed_at = datetime.now() + # ---------------------------------------------------------------------- + # Method: mark_incomplete_tasks_as_failed def mark_incomplete_tasks_as_failed(self, dag_id: str, execution_id: str): + """Mark all running/pending tasks for an execution as `failed`. + + Used during teardown or when an execution is aborted to ensure + no tasks are left in non-terminal states. For performance the + update uses `synchronize_session=False`. + """ with self.Session.begin() as session: session.query(TaskORM).filter( - TaskORM.dag_id == dag_id, - TaskORM.execution_id == execution_id, - TaskORM.status.in_(["running", "pending"]) + TaskORM.dag_id == dag_id, + TaskORM.execution_id == execution_id, + TaskORM.status.in_(["running", "pending"]), ).update( {TaskORM.status: "failed", TaskORM.completed_at: datetime.now()}, - synchronize_session=False + synchronize_session=False, ) + # ---------------------------------------------------------------------- + # Method: get_running_dags def get_running_dags(self) -> List[Dict[str, Any]]: + """Return a list of currently running executions with metadata. + + Each item contains `dag_id`, `execution_id`, `started_at`, + `thread_id` and `pid`. + """ with self.Session() as session: executions = session.query(ExecutionORM).filter_by(status="running").all() return [ @@ -189,11 +386,15 @@ def get_running_dags(self) -> List[Dict[str, Any]]: "execution_id": e.id, "started_at": e.started_at.isoformat() if e.started_at else None, "thread_id": e.thread_id, - "pid": e.pid - } for e in executions + "pid": e.pid, + } + for e in executions ] + # ---------------------------------------------------------------------- + # Method: get_dags_by_status def get_dags_by_status(self, status: str) -> List[Dict[str, Any]]: + """Return executions filtered by `status` with relevant metadata.""" with self.Session() as session: executions = session.query(ExecutionORM).filter_by(status=status).all() return [ @@ -201,67 +402,143 @@ def get_dags_by_status(self, status: str) -> List[Dict[str, Any]]: "dag_id": e.dag_id, "execution_id": e.id, "started_at": e.started_at.isoformat() if e.started_at else None, - "completed_at": e.completed_at.isoformat() if e.completed_at else None, + "completed_at": ( + e.completed_at.isoformat() if e.completed_at else None + ), "thread_id": e.thread_id, - "pid": e.pid - } for e in executions + "pid": e.pid, + } + for e in executions ] + # ---------------------------------------------------------------------- + # Method: get_all_dags def get_all_dags(self) -> List[Dict[str, Any]]: + """Return a list of all known DAGs and their latest execution. + + Each entry includes the DAG id and a summary of the most recent + execution (if any). + """ with self.Session() as session: dags = session.query(DagORM).all() result = [] for dag in dags: - latest_execution = session.query(ExecutionORM).filter_by(dag_id=dag.id).order_by(ExecutionORM.started_at.desc()).first() - result.append({ - "dag_id": dag.id, - "execution_id": latest_execution.id if latest_execution else None, - "status": latest_execution.status if latest_execution else "created", - "started_at": latest_execution.started_at.isoformat() if latest_execution and latest_execution.started_at else None, - "completed_at": latest_execution.completed_at.isoformat() if latest_execution and latest_execution.completed_at else None, - "thread_id": latest_execution.thread_id if latest_execution else None, - "pid": latest_execution.pid if latest_execution else None - }) + latest_execution = ( + session.query(ExecutionORM) + .filter_by(dag_id=dag.id) + .order_by(ExecutionORM.started_at.desc()) + .first() + ) + result.append( + { + "dag_id": dag.id, + "execution_id": ( + latest_execution.id if latest_execution else None + ), + "status": ( + latest_execution.status if latest_execution else "created" + ), + "started_at": ( + latest_execution.started_at.isoformat() + if latest_execution and latest_execution.started_at + else None + ), + "completed_at": ( + latest_execution.completed_at.isoformat() + if latest_execution and latest_execution.completed_at + else None + ), + "thread_id": ( + latest_execution.thread_id if latest_execution else None + ), + "pid": latest_execution.pid if latest_execution else None, + } + ) return result + # ---------------------------------------------------------------------- + # Method: get_dag_summary def get_dag_summary(self) -> Dict[str, Any]: + """Return aggregated counts for executions and DAGs. + + The result contains `total_executions`, `unique_dags` and a + mapping `status_counts` showing how many executions are in each + status. + """ with self.Session() as session: - status_counts = session.query(ExecutionORM.status, func.count(ExecutionORM.status)).group_by(ExecutionORM.status).all() + status_counts = ( + session.query(ExecutionORM.status, func.count(ExecutionORM.status)) + .group_by(ExecutionORM.status) + .all() + ) total_count = session.query(ExecutionORM).count() unique_dags = session.query(DagORM).count() return { "total_executions": total_count, "unique_dags": unique_dags, - "status_counts": dict(status_counts) + "status_counts": dict(status_counts), } + # ---------------------------------------------------------------------- + # Method: get_dag_history def get_dag_history(self, dag_id: str) -> List[Dict[str, Any]]: + """Return the execution history for a DAG ordered by start time. + + Each entry contains `execution_id`, `status`, timestamps and + thread/pid information. + """ with self.Session() as session: - executions = session.query(ExecutionORM).filter_by(dag_id=dag_id).order_by(ExecutionORM.started_at.desc()).all() + executions = ( + session.query(ExecutionORM) + .filter_by(dag_id=dag_id) + .order_by(ExecutionORM.started_at.desc()) + .all() + ) return [ { "execution_id": e.id, "status": e.status, "started_at": e.started_at.isoformat() if e.started_at else None, - "completed_at": e.completed_at.isoformat() if e.completed_at else None, + "completed_at": ( + e.completed_at.isoformat() if e.completed_at else None + ), "thread_id": e.thread_id, - "pid": e.pid - } for e in executions + "pid": e.pid, + } + for e in executions ] + # ---------------------------------------------------------------------- + # Method: cleanup_old_executions def cleanup_old_executions(self, days_to_keep: int = 30): + """Delete executions older than `days_to_keep` days and return count. + + Use with care: this permanently removes execution and related task + records. + """ with self.Session.begin() as session: cutoff_date = datetime.now() - timedelta(days=days_to_keep) - executions_to_delete = session.query(ExecutionORM).filter(ExecutionORM.started_at < cutoff_date).all() + executions_to_delete = ( + session.query(ExecutionORM) + .filter(ExecutionORM.started_at < cutoff_date) + .all() + ) for execution in executions_to_delete: session.delete(execution) return len(executions_to_delete) - def cancel_dag_execution(self, - dag_id: str, - execution_id: str = None) -> bool: + # ---------------------------------------------------------------------- + # Method: cancel_dag_execution + def cancel_dag_execution(self, dag_id: str, execution_id: str = None) -> bool: + """Cancel a running execution and mark its tasks as `cancelled`. + + If `execution_id` is None the first running execution for the DAG + is targeted. + """ with self.Session.begin() as session: - query = session.query(ExecutionORM).filter_by(dag_id=dag_id, status="running") + query = session.query(ExecutionORM).filter_by( + dag_id=dag_id, status="running" + ) if execution_id: query = query.filter_by(id=execution_id) execution = query.first() @@ -269,21 +546,31 @@ def cancel_dag_execution(self, # Mark the execution as cancelled execution.status = "cancelled" execution.completed_at = datetime.now() - + # Mark all incomplete tasks (running, pending) as cancelled session.query(TaskORM).filter( TaskORM.dag_id == dag_id, TaskORM.execution_id == execution.id, - TaskORM.status.in_(["running", "pending"]) + TaskORM.status.in_(["running", "pending"]), ).update( {TaskORM.status: "cancelled", TaskORM.completed_at: datetime.now()}, - synchronize_session=False + synchronize_session=False, ) - + return True return False - def get_dag_execution_details(self, dag_id: str, execution_id: str = None) -> Dict[str, Any]: + # ---------------------------------------------------------------------- + # Method: get_dag_execution_details + def get_dag_execution_details( + self, dag_id: str, execution_id: str = None + ) -> Dict[str, Any]: + """Return detailed information for a specific execution. + + If `execution_id` is omitted the latest execution is returned. The + result includes the list of tasks with their timestamps and + statuses. + """ with self.Session() as session: query = session.query(ExecutionORM).filter_by(dag_id=dag_id) if execution_id: @@ -295,28 +582,52 @@ def get_dag_execution_details(self, dag_id: str, execution_id: str = None) -> Di if not execution: return {} - tasks = session.query(TaskORM).filter_by(dag_id=dag_id, execution_id=execution.id).order_by(TaskORM.insertion_order).all() + tasks = ( + session.query(TaskORM) + .filter_by(dag_id=dag_id, execution_id=execution.id) + .order_by(TaskORM.insertion_order) + .all() + ) return { "execution_id": execution.id, "status": execution.status, - "started_at": execution.started_at.isoformat() if execution.started_at else None, - "completed_at": execution.completed_at.isoformat() if execution.completed_at else None, + "started_at": ( + execution.started_at.isoformat() if execution.started_at else None + ), + "completed_at": ( + execution.completed_at.isoformat() + if execution.completed_at + else None + ), "thread_id": execution.thread_id, "pid": execution.pid, "tasks": [ { "task_id": task.id, "status": task.status, - "started_at": task.started_at.isoformat() if task.started_at else None, - "completed_at": task.completed_at.isoformat() if task.completed_at else None, - "thread_id": task.thread_id - } for task in tasks - ] + "started_at": ( + task.started_at.isoformat() if task.started_at else None + ), + "completed_at": ( + task.completed_at.isoformat() if task.completed_at else None + ), + "thread_id": task.thread_id, + } + for task in tasks + ], } - def log_message(self, dag_id: str, execution_id: str, task_id: str, - level: str, message: str, - timestamp: Optional[datetime] = None): + # ---------------------------------------------------------------------- + # Method: log_message + def log_message( + self, + dag_id: str, + execution_id: str, + task_id: str, + level: str, + message: str, + timestamp: Optional[datetime] = None, + ): """Store a log message with an optional externally provided timestamp.""" with self.Session.begin() as session: log = LogORM( @@ -326,13 +637,21 @@ def log_message(self, dag_id: str, execution_id: str, task_id: str, level=level, message=message, timestamp=timestamp or datetime.now(), - thread_id=str(threading.current_thread().ident) + thread_id=str(threading.current_thread().ident), ) session.add(log) - def add_log(self, dag_id: str, execution_id: str, task_id: str, - message: str, level: str = "INFO", - timestamp: Optional[datetime] = None): + # ---------------------------------------------------------------------- + # Method: add_log + def add_log( + self, + dag_id: str, + execution_id: str, + task_id: str, + message: str, + level: str = "INFO", + timestamp: Optional[datetime] = None, + ): """Helper wrapper for log_message() with timestamp support.""" self.log_message( dag_id=dag_id, @@ -340,10 +659,18 @@ def add_log(self, dag_id: str, execution_id: str, task_id: str, task_id=task_id, level=level, message=message, - timestamp=timestamp + timestamp=timestamp, ) - def get_execution_logs(self, dag_id: str, execution_id: str = None, limit: int = 100) -> List[Dict[str, Any]]: + # ---------------------------------------------------------------------- + # Method: get_execution_logs + def get_execution_logs( + self, dag_id: str, execution_id: str = None, limit: int = 100 + ) -> List[Dict[str, Any]]: + """Return the most recent logs for a DAG/execution up to `limit`. + + Logs are returned ordered by timestamp descending. + """ with self.Session() as session: query = session.query(LogORM).filter_by(dag_id=dag_id) if execution_id: @@ -355,44 +682,88 @@ def get_execution_logs(self, dag_id: str, execution_id: str = None, limit: int = "level": log.level, "message": log.message, "timestamp": log.timestamp.isoformat(), - "thread_id": log.thread_id - } for log in logs + "thread_id": log.thread_id, + } + for log in logs ] + # ---------------------------------------------------------------------- + # Method: validate_dag_id def validate_dag_id(self, dag_id: str) -> bool: + """Return True if `dag_id` is a non-empty alphanumeric identifier. + + Allowed characters are letters, digits, underscores and hyphens. + """ if not dag_id: return False - return bool(re.match(r'^[a-zA-Z0-9_-]+$', dag_id)) + return bool(re.match(r"^[a-zA-Z0-9_-]+$", dag_id)) + # ---------------------------------------------------------------------- + # Method: check_dag_id_uniqueness def check_dag_id_uniqueness(self, dag_id: str) -> bool: + """Return True if no DagORM with `dag_id` exists in the DB.""" with self.Session() as session: return session.query(DagORM).filter_by(id=dag_id).first() is None + # ---------------------------------------------------------------------- + # Method: generate_unique_dag_id def generate_unique_dag_id(self) -> str: + """Generate a human-friendly, likely-unique dag id. + + It tries `max_attempts` times to pick a combination that does not + collide with existing DAG ids. If all attempts fail it appends a + short random suffix. + """ max_attempts = 100 for _ in range(max_attempts): dag_id = f"{random.choice(DOCKER_ADJECTIVES)}_{random.choice(DOCKER_NOUNS)}" if self.check_dag_id_uniqueness(dag_id): return dag_id base_name = f"{random.choice(DOCKER_ADJECTIVES)}_{random.choice(DOCKER_NOUNS)}" - suffix = ''.join(random.choices(string.ascii_lowercase + string.digits, k=6)) + suffix = "".join(random.choices(string.ascii_lowercase + string.digits, k=6)) return f"{base_name}_{suffix}" + # ---------------------------------------------------------------------- + # Method: get_latest_execution def get_latest_execution(self, dag_id: str) -> Optional[Dict[str, Any]]: + """Return a summary dict for the latest execution of `dag_id`. + + Returns `None` if no execution exists. + """ with self.Session() as session: - execution = session.query(ExecutionORM).filter_by(dag_id=dag_id).order_by(ExecutionORM.started_at.desc()).first() + execution = ( + session.query(ExecutionORM) + .filter_by(dag_id=dag_id) + .order_by(ExecutionORM.started_at.desc()) + .first() + ) if execution: return { "execution_id": execution.id, "status": execution.status, - "started_at": execution.started_at.isoformat() if execution.started_at else None, - "completed_at": execution.completed_at.isoformat() if execution.completed_at else None, + "started_at": ( + execution.started_at.isoformat() + if execution.started_at + else None + ), + "completed_at": ( + execution.completed_at.isoformat() + if execution.completed_at + else None + ), "thread_id": execution.thread_id, - "pid": execution.pid + "pid": execution.pid, } return None + # ---------------------------------------------------------------------- + # Method: save_dag_definition def save_dag_definition(self, dag, dag_filepath: Optional[str] = None): + """Persist the DAG definition JSON into the `DagORM.definition` field. + + If the DagORM does not exist a new record is created. `dag` is + expected to expose a `dag_id` attribute and `to_dict()` method. + """ with self.Session.begin() as session: dag_orm = session.query(DagORM).filter_by(id=dag.dag_id).first() if not dag_orm: @@ -402,7 +773,14 @@ def save_dag_definition(self, dag, dag_filepath: Optional[str] = None): if dag_filepath: dag_orm.dag_filepath = dag_filepath + # ---------------------------------------------------------------------- + # Method: get_dag_definition def get_dag_definition(self, dag_id: str) -> Optional[Dict]: + """Return the stored DAG definition as a Python object. + + If the stored value is valid JSON it is decoded; otherwise the raw + stored value is returned (for backward compatibility). + """ with self.Session() as session: dag = session.query(DagORM).filter_by(id=dag_id).first() if not dag: @@ -417,7 +795,13 @@ def get_dag_definition(self, dag_id: str) -> Optional[Dict]: # If the stored definition is not valid JSON, fall back to returning the raw value return dag.definition + # ---------------------------------------------------------------------- + # Method: delete_dag def delete_dag(self, dag_id: str) -> int: + """Delete the DAG record (and by cascade its executions/tasks if set). + + Returns 1 if a record was deleted, 0 otherwise. + """ with self.Session.begin() as session: dag = session.query(DagORM).filter_by(id=dag_id).first() if dag: @@ -425,10 +809,14 @@ def delete_dag(self, dag_id: str) -> int: return 1 return 0 - - # 🆕 Metodo per recuperare l'istanza globale + # ---------------------------------------------------------------------- + # Method: get_instance (class method) @classmethod def get_instance(cls): + """Return the previously initialized global StatusManager. + + Raises a RuntimeError if the manager has not been initialized. + """ if cls._instance is None: raise RuntimeError("StatusManager has not been initialized yet.") return cls._instance diff --git a/src/maestro/server/tasks/bash_task.py b/src/maestro/server/tasks/bash_task.py index f11fa1a..34d22cd 100644 --- a/src/maestro/server/tasks/bash_task.py +++ b/src/maestro/server/tasks/bash_task.py @@ -29,6 +29,7 @@ def execute_local(self): process = subprocess.Popen( self.command, shell=True, + executable="/bin/bash", stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, @@ -66,17 +67,36 @@ def execute_local(self): ) # Ritorno exit code + process.wait() - if process.returncode != 0: + rc = process.returncode + + if rc != 0: + msg = f"[BashTask] Failed with exit code {rc}" + print(msg, flush=True) sm.add_log( dag_id=dag_id, execution_id=execution_id, task_id=task_id, - message=f"[BashTask] Failed with exit code {process.returncode}", + message=msg, level="ERROR" ) + # QUI: solleva un'eccezione → farà scattare il retry + raise Exception(f"BashTask failed with exit code {rc}") + + # Se arrivo qui, exit code 0 -> success + success_msg = "[BashTask] Completed successfully (exit code 0)" + print(success_msg, flush=True) + sm.add_log( + dag_id=dag_id, + execution_id=execution_id, + task_id=task_id, + message=success_msg, + level="INFO" + ) + def to_dict(self): """Include the command field in serialized form.""" From b447460992909c7f7bd422abcfea96f56c435bbc Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Wed, 26 Nov 2025 00:11:41 +0100 Subject: [PATCH 19/38] Updated examples --- data.json | 1 - .../3.2.2.wait_and_retry_bash_python_mix.yaml | 6 +- .../3.2.3.wait_and_retry_branching.yaml | 17 +++--- .../3.3.1.wait_and_retry_heavy_parallel.yaml | 10 ++-- ...3.3.2.wait_and_retry_cascade_failures.yaml | 12 ++-- .../3.3.3.wait_and_retry_spiderweb.yaml | 36 +++++++----- .../4.1.1.Conditional_branching.yaml | 3 + .../4.1.2. WORKING_Conditional_branching.yaml | 52 ++++++++++++++++++ .../5.2.1.cron_heartbeat_extended.yaml | 10 ++-- .../5.2.2.cron_heartbeat_with_status.yaml | 15 +++-- .../5.2.3.cron_heartbeat_light_checks.yaml | 30 +++++++--- ...5.3.1.cron_heartbeat_full_healthcheck.yaml | 27 +++++++-- .../5.3.2.cron_heartbeat_with_warnings.yaml | 12 ++-- .../5.3.3.cron_heartbeat_deep_pipeline.yaml | 16 +++--- .../6.2.1.scheduled_greeting_extended.yaml | 2 +- ...6.2.2.scheduled_greeting_language_mix.yaml | 6 +- .../6.2.3.scheduled_greeting_with_checks.yaml | 7 +-- ...1.scheduled_greeting_full_healthcheck.yaml | 8 +-- ....3.2.scheduled_greeting_with_warnings.yaml | 14 ++--- .../6.3.3.scheduled_greeting_deep_tree.yaml | 12 ++-- .../2_New_examples/7.1.1.Mixed_execution.yaml | 7 ++- .../7.2.1.mixed_execution_extended.yaml | 13 ++--- .../7.2.2.mixed_execution_branching.yaml | 4 +- .../7.2.3.mixed_execution_random_data.yaml | 6 +- .../7.3.1.mixed_execution_data_pipeline.yaml | 6 +- .../7.3.2.mixed_execution_transform_tree.yaml | 16 +++--- .../7.3.3.mixed_execution_hyperpipeline.yaml | 14 ++--- maestro.db | Bin 114688 -> 0 bytes runtime/tmp/maestro_heartbeat.log | 1 + 29 files changed, 230 insertions(+), 133 deletions(-) delete mode 100644 data.json create mode 100644 examples/2_New_examples/4.1.2. WORKING_Conditional_branching.yaml delete mode 100644 maestro.db create mode 100644 runtime/tmp/maestro_heartbeat.log diff --git a/data.json b/data.json deleted file mode 100644 index 4136c6f..0000000 --- a/data.json +++ /dev/null @@ -1 +0,0 @@ -{"values": [1, 2, 3, 4, 5]} \ No newline at end of file diff --git a/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml b/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml index 2c46162..7762559 100644 --- a/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml +++ b/examples/2_New_examples/3.2.2.wait_and_retry_bash_python_mix.yaml @@ -32,13 +32,13 @@ dag: code: | import random, sys, time - # Genera un numero casuale tra 0 e 1 + # generate a random number between 0 and 1 generate_random = random.random() print(f"Unstable Python Task - Generated: {round(generate_random, 2)}") - # Usa una soglia di 0.4 per decidere se fallire oppure no + # Use the threshold of 40% if random.random() < 0.4: - # Fallimento esplicito: il processo termina con errore + # Explicit failyure - exit with error status 1 sys.exit("Python failed") else: # Caso di successo diff --git a/examples/2_New_examples/3.2.3.wait_and_retry_branching.yaml b/examples/2_New_examples/3.2.3.wait_and_retry_branching.yaml index 921416d..f9a762a 100644 --- a/examples/2_New_examples/3.2.3.wait_and_retry_branching.yaml +++ b/examples/2_New_examples/3.2.3.wait_and_retry_branching.yaml @@ -5,35 +5,36 @@ dag: name: "wait_and_retry_branching" tasks: - # Primo task instabile (radice) + # First unstable task (root) - task_id: "unstable_root" type: "PythonTask" params: - script: | + code: | import random, sys if random.random() < 0.5: sys.exit(1) - print("Root succeeded!") + else: + print("Root succeeded!") retries: 3 retry_delay: 2 dependencies: [] - # Ramo 1 + # Branch 1 - task_id: "branch_a" type: "BashTask" params: command: | if [ $((RANDOM % 2)) -eq 0 ]; then exit 1; fi echo "Branch A OK" - retries: 2 + retries: 3 retry_delay: 1 dependencies: ["unstable_root"] - # Ramo 2 + # Branch 2 - task_id: "branch_b" type: "PythonTask" params: - script: | + code: | import random, sys if random.random() < 0.3: sys.exit(1) @@ -42,7 +43,7 @@ dag: retry_delay: 1 dependencies: ["unstable_root"] - # Fusione + # Final merging - task_id: "done" type: "PrintTask" params: diff --git a/examples/2_New_examples/3.3.1.wait_and_retry_heavy_parallel.yaml b/examples/2_New_examples/3.3.1.wait_and_retry_heavy_parallel.yaml index 8344729..ebf565a 100644 --- a/examples/2_New_examples/3.3.1.wait_and_retry_heavy_parallel.yaml +++ b/examples/2_New_examples/3.3.1.wait_and_retry_heavy_parallel.yaml @@ -15,7 +15,7 @@ dag: - task_id: "unstable_1" type: "PythonTask" params: - script: | + code: | import random, sys if random.random() < .5: sys.exit(1) print("Unstable 1 OK") @@ -29,14 +29,14 @@ dag: command: | if [ $((RANDOM % 2)) -eq 0 ]; then exit 1; fi echo "Unstable 2 OK" - retries: 3 + retries: 4 retry_delay: 2 dependencies: ["start"] - task_id: "unstable_3" type: "PythonTask" params: - script: | + code: | import random, sys if random.random() < .7: sys.exit(1) print("Unstable 3 OK") @@ -50,8 +50,8 @@ dag: command: | if [ $((RANDOM % 3)) -eq 0 ]; then exit 1; fi echo "Unstable 4 OK" - retries: 2 - retry_delay: 3 + retries: 6 + retry_delay: 0 dependencies: ["start"] # Merging finale diff --git a/examples/2_New_examples/3.3.2.wait_and_retry_cascade_failures.yaml b/examples/2_New_examples/3.3.2.wait_and_retry_cascade_failures.yaml index 73ae5a2..fce57a5 100644 --- a/examples/2_New_examples/3.3.2.wait_and_retry_cascade_failures.yaml +++ b/examples/2_New_examples/3.3.2.wait_and_retry_cascade_failures.yaml @@ -8,7 +8,7 @@ dag: - task_id: "t1" type: "PythonTask" params: - script: | + code: | import random, sys if random.random() < .4: sys.exit(1) print("t1 OK") @@ -22,19 +22,19 @@ dag: command: | if [ $((RANDOM % 4)) -eq 0 ]; then exit 1; fi echo "t2 OK" - retries: 3 - retry_delay: 2 + retries: 1 + retry_delay: 5 dependencies: ["t1"] - task_id: "t3" type: "PythonTask" params: - script: | + code: | import random, sys if random.random() < .5: sys.exit(1) print("t3 OK") retries: 2 - retry_delay: 1 + retry_delay: 2 dependencies: ["t2"] - task_id: "t4" @@ -43,7 +43,7 @@ dag: command: | if [ $((RANDOM % 3)) -eq 0 ]; then exit 1; fi echo "t4 OK" - retries: 4 + retries: 6 retry_delay: 1 dependencies: ["t3"] diff --git a/examples/2_New_examples/3.3.3.wait_and_retry_spiderweb.yaml b/examples/2_New_examples/3.3.3.wait_and_retry_spiderweb.yaml index 3c03785..9ff480e 100644 --- a/examples/2_New_examples/3.3.3.wait_and_retry_spiderweb.yaml +++ b/examples/2_New_examples/3.3.3.wait_and_retry_spiderweb.yaml @@ -9,9 +9,11 @@ dag: - task_id: "root" type: "PythonTask" params: - script: | + code: | import random, sys - if random.random() < .5: sys.exit(1) + if random.random() < .5: + print("Root failed") + sys.exit(1) print("Root OK") retries: 3 retry_delay: 2 @@ -21,7 +23,7 @@ dag: - task_id: "a1" type: "BashTask" params: - command: "if [ $((RANDOM % 2)) -eq 0 ]; then exit 1; fi; echo A1" + command: "if [ $((RANDOM % 2)) -eq 0 ]; then echo 'Failed A1'; exit 1; fi; echo A1 OK" retries: 2 retry_delay: 1 dependencies: ["root"] @@ -29,9 +31,11 @@ dag: - task_id: "a2" type: "PythonTask" params: - script: | + code: | import random, sys - if random.random() < .4: sys.exit(1) + if random.random() < .4: + print("Failed A2") + sys.exit(1) print("A2 OK") retries: 3 retry_delay: 1 @@ -41,9 +45,11 @@ dag: - task_id: "b1" type: "PythonTask" params: - script: | + code: | import random, sys - if random.random() < .6: sys.exit(1) + if random.random() < .6: + print("Failed B1") + sys.exit(1) print("B1 OK") retries: 4 retry_delay: 1 @@ -52,7 +58,7 @@ dag: - task_id: "b2" type: "BashTask" params: - command: "if [ $((RANDOM % 3)) -eq 0 ]; then exit 1; fi; echo B2" + command: "if [ $((RANDOM % 3)) -eq 0 ]; then echo 'Failed B2'; exit 1; fi; echo B2 OK" retries: 2 retry_delay: 2 dependencies: ["b1"] @@ -61,7 +67,7 @@ dag: - task_id: "c1" type: "BashTask" params: - command: "if [ $((RANDOM % 2)) -eq 0 ]; then exit 1; fi; echo C1" + command: "if [ $((RANDOM % 2)) -eq 0 ]; then echo 'Failed C1'; exit 1; fi; echo C1 OK" retries: 3 retry_delay: 1 dependencies: ["root"] @@ -69,9 +75,11 @@ dag: - task_id: "c2" type: "PythonTask" params: - script: | + code: | import random, sys - if random.random() < .5: sys.exit(1) + if random.random() < .5: + print("Failed C2") + sys.exit(1) print("C2 OK") retries: 3 retry_delay: 1 @@ -88,9 +96,11 @@ dag: - task_id: "final_unstable" type: "PythonTask" params: - script: | + code: | import random, sys - if random.random() < .4: sys.exit(1) + if random.random() < .4: + print("Final task failed") + sys.exit(1) print("Final task OK") retries: 3 retry_delay: 2 diff --git a/examples/2_New_examples/4.1.1.Conditional_branching.yaml b/examples/2_New_examples/4.1.1.Conditional_branching.yaml index 826000b..fe747bb 100644 --- a/examples/2_New_examples/4.1.1.Conditional_branching.yaml +++ b/examples/2_New_examples/4.1.1.Conditional_branching.yaml @@ -1,6 +1,7 @@ dag: name: "conditional_branching" tasks: + - task_id: "check_condition" type: "PythonTask" params: @@ -9,11 +10,13 @@ dag: result = "branch_a" if random.random() > 0.5 else "branch_b" print(f"Next step: {result}") dependencies: [] + - task_id: "branch_a" type: "PrintTask" params: message: "Running branch A" dependencies: ["check_condition"] + - task_id: "branch_b" type: "PrintTask" params: diff --git a/examples/2_New_examples/4.1.2. WORKING_Conditional_branching.yaml b/examples/2_New_examples/4.1.2. WORKING_Conditional_branching.yaml new file mode 100644 index 0000000..a547547 --- /dev/null +++ b/examples/2_New_examples/4.1.2. WORKING_Conditional_branching.yaml @@ -0,0 +1,52 @@ +dag: + name: "conditional_branching" + tasks: + + - task_id: "check_condition" + type: "PythonTask" + params: + code: | + import random, pathlib + + # Decidi quale ramo eseguire + result = "branch_a" if random.random() > 0.5 else "branch_b" + print(f"[check_condition] Next step: {result}") + + # Salva la scelta in un file (es. nella working dir dell'esecuzione) + path = pathlib.Path("branch_choice.txt") + path.write_text(result) + dependencies: [] + + - task_id: "branch_a" + type: "PythonTask" + params: + code: | + import pathlib + + path = pathlib.Path("branch_choice.txt") + if not path.exists(): + print("[branch_a] No branch_choice.txt found, cannot decide → skipping A") + else: + choice = path.read_text().strip() + if choice == "branch_a": + print("[branch_a] I am the chosen branch! Running branch A logic...") + else: + print(f"[branch_a] Branch choice is '{choice}', not 'branch_a' → skipping A") + dependencies: ["check_condition"] + + - task_id: "branch_b" + type: "PythonTask" + params: + code: | + import pathlib + + path = pathlib.Path("branch_choice.txt") + if not path.exists(): + print("[branch_b] No branch_choice.txt found, cannot decide → skipping B") + else: + choice = path.read_text().strip() + if choice == "branch_b": + print("[branch_b] I am the chosen branch! Running branch B logic...") + else: + print(f"[branch_b] Branch choice is '{choice}', not 'branch_b' → skipping B") + dependencies: ["check_condition"] diff --git a/examples/2_New_examples/5.2.1.cron_heartbeat_extended.yaml b/examples/2_New_examples/5.2.1.cron_heartbeat_extended.yaml index 841cfee..9a0b6ba 100644 --- a/examples/2_New_examples/5.2.1.cron_heartbeat_extended.yaml +++ b/examples/2_New_examples/5.2.1.cron_heartbeat_extended.yaml @@ -3,24 +3,24 @@ abstract: | dag: name: "cron_heartbeat_extended" - cron_schedule: "*/10 * * * *" # Ogni 10 minuti + cron_schedule: "*/10 * * * *" # Each 10 minutes tasks: - # 1) Messaggio iniziale, così vedi che il job è partito + # 1) Initial message to check the start - task_id: "heartbeat_start" type: "PrintTask" params: message: "Heartbeat: Maestro cron job triggered." dependencies: [] - # 2) Piccolo check di sistema: stampa data/ora e directory corrente + # 2) System check - task_id: "bash_info" type: "BashTask" params: - command: "echo 'Time:' $(date) && echo 'PWD:' $(pwd)" + command: "echo 'Time:' $(date); echo 'PWD:' $(pwd)" dependencies: ["heartbeat_start"] - # 3) Messaggio finale di conferma + # 3) Final message - task_id: "heartbeat_end" type: "PrintTask" params: diff --git a/examples/2_New_examples/5.2.2.cron_heartbeat_with_status.yaml b/examples/2_New_examples/5.2.2.cron_heartbeat_with_status.yaml index f3ad214..2848392 100644 --- a/examples/2_New_examples/5.2.2.cron_heartbeat_with_status.yaml +++ b/examples/2_New_examples/5.2.2.cron_heartbeat_with_status.yaml @@ -3,29 +3,32 @@ abstract: | dag: name: "cron_heartbeat_with_status" - cron_schedule: "*/10 * * * *" # Ogni 10 minuti + cron_schedule: "*/10 * * * *" # Each 10 minutes tasks: - # 1) Valuta uno stato fittizio del sistema + # 1) Evaluate a fake state of the system - task_id: "compute_status" type: "PythonTask" params: - script: | + code: | import random status = random.choice(["OK", "DEGRADED", "ERROR"]) print(f"Computed heartbeat status: {status}") dependencies: [] - # 2) Stampa un messaggio leggibile a console + # 2) Print a readable message to the console - task_id: "print_status" type: "PrintTask" params: message: "Heartbeat status computed and printed above." dependencies: ["compute_status"] - # 3) Scrive un log minimale su file (in /tmp per restare innocui) + # 3) Write a minimal log on the console (in /tmp) - task_id: "log_to_file" type: "BashTask" params: - command: "echo \"$(date) - Heartbeat executed\" >> /tmp/maestro_heartbeat.log" + command: | + mkdir -p runtime/tmp + echo "$(date) - Heartbeat executed" >> runtime/tmp/maestro_heartbeat.log dependencies: ["print_status"] + diff --git a/examples/2_New_examples/5.2.3.cron_heartbeat_light_checks.yaml b/examples/2_New_examples/5.2.3.cron_heartbeat_light_checks.yaml index a1c83b4..a3cc226 100644 --- a/examples/2_New_examples/5.2.3.cron_heartbeat_light_checks.yaml +++ b/examples/2_New_examples/5.2.3.cron_heartbeat_light_checks.yaml @@ -6,36 +6,52 @@ dag: cron_schedule: "*/10 * * * *" # Ogni 10 minuti tasks: - # ROOT: lancia il giro di check + # ROOT: start check routine - task_id: "heartbeat_root" type: "PrintTask" params: message: "Heartbeat: starting light system checks..." dependencies: [] - # Check pseudo-CPU (simulato) - task_id: "check_cpu" type: "PythonTask" params: - script: | - print("CPU check: (simulated) OK") + code: | + import time + + def read_cpu_times(): + with open("/proc/stat", "r") as f: + fields = f.readline().strip().split()[1:] + return list(map(int, fields)) + + t1 = read_cpu_times() + time.sleep(0.1) + t2 = read_cpu_times() + + idle1, total1 = t1[3], sum(t1) + idle2, total2 = t2[3], sum(t2) + + idle_delta = idle2 - idle1 + total_delta = total2 - total1 + + usage = 100 * (1 - idle_delta / total_delta) + + print(f"CPU Usage (real): {usage:.2f}%") dependencies: ["heartbeat_root"] - # Check disco (solo output informativo) - task_id: "check_disk" type: "BashTask" params: command: "df -h | head -n 5" dependencies: ["heartbeat_root"] - # Check processi (solo prime righe) - task_id: "check_processes" type: "BashTask" params: command: "ps aux | head -n 5" dependencies: ["heartbeat_root"] - # Merge: eseguito solo quando tutti i check sono completati + # Final merge - task_id: "heartbeat_summary" type: "PrintTask" params: diff --git a/examples/2_New_examples/5.3.1.cron_heartbeat_full_healthcheck.yaml b/examples/2_New_examples/5.3.1.cron_heartbeat_full_healthcheck.yaml index 31f0edf..85115ee 100644 --- a/examples/2_New_examples/5.3.1.cron_heartbeat_full_healthcheck.yaml +++ b/examples/2_New_examples/5.3.1.cron_heartbeat_full_healthcheck.yaml @@ -17,8 +17,27 @@ dag: - task_id: "check_cpu" type: "PythonTask" params: - script: | - print("CPU usage check (simulated).") + code: | + import time + + def read_cpu_times(): + with open("/proc/stat", "r") as f: + fields = f.readline().strip().split()[1:] + return list(map(int, fields)) + + t1 = read_cpu_times() + time.sleep(0.1) + t2 = read_cpu_times() + + idle1, total1 = t1[3], sum(t1) + idle2, total2 = t2[3], sum(t2) + + idle_delta = idle2 - idle1 + total_delta = total2 - total1 + + usage = 100 * (1 - idle_delta / total_delta) + + print(f"CPU Usage: {usage:.2f}%") dependencies: ["heartbeat_start"] # BRANCH MEMORIA @@ -42,11 +61,11 @@ dag: command: "ps aux | head -n 10" dependencies: ["heartbeat_start"] - # AGGREGAZIONE LOGICA (qui potresti, in futuro, interpretare output reali) + # AGGREGAZIONE LOGICA - task_id: "aggregate_results" type: "PythonTask" params: - script: | + code: | print("Aggregating healthcheck results (simulation). Overall: OK") dependencies: ["check_cpu", "check_memory", "check_disk", "check_processes"] diff --git a/examples/2_New_examples/5.3.2.cron_heartbeat_with_warnings.yaml b/examples/2_New_examples/5.3.2.cron_heartbeat_with_warnings.yaml index f16d83a..dc57766 100644 --- a/examples/2_New_examples/5.3.2.cron_heartbeat_with_warnings.yaml +++ b/examples/2_New_examples/5.3.2.cron_heartbeat_with_warnings.yaml @@ -6,17 +6,17 @@ dag: cron_schedule: "*/10 * * * *" # Ogni 10 minuti tasks: - # ROOT: decide un livello di "criticità" fittizio + # ROOT: set a critical (fake) threshold - task_id: "evaluate_status" type: "PythonTask" params: - script: | + code: | import random status = random.choice(["OK", "WARNING"]) print(f"Evaluated heartbeat status: {status}") dependencies: [] - # RAMO OK: controlli leggeri + # OK branch - task_id: "ok_branch_info" type: "BashTask" params: @@ -29,11 +29,11 @@ dag: message: "OK branch completed." dependencies: ["ok_branch_info"] - # RAMO WARNING: logga qualcosa in più + # WARNING branch - task_id: "warning_branch_detail" type: "PythonTask" params: - script: | + code: | print("WARNING branch: capturing extra diagnostic info (simulated).") dependencies: ["evaluate_status"] @@ -43,7 +43,7 @@ dag: command: "echo \"$(date) - WARNING: simulated condition\" >> /tmp/maestro_heartbeat_warn.log" dependencies: ["warning_branch_detail"] - # MERGE: sempre eseguito dopo entrambi i rami + # Final marge - task_id: "heartbeat_summary" type: "PrintTask" params: diff --git a/examples/2_New_examples/5.3.3.cron_heartbeat_deep_pipeline.yaml b/examples/2_New_examples/5.3.3.cron_heartbeat_deep_pipeline.yaml index d96922c..208c8bc 100644 --- a/examples/2_New_examples/5.3.3.cron_heartbeat_deep_pipeline.yaml +++ b/examples/2_New_examples/5.3.3.cron_heartbeat_deep_pipeline.yaml @@ -23,15 +23,15 @@ dag: - task_id: "sys_python_analysis" type: "PythonTask" params: - script: | + code: | print("System analysis (simulated).") dependencies: ["sys_info"] - # ===== BRANCH APPLICATION (simulato) ===== + # ===== BRANCH APPLICATION (simulated) ===== - task_id: "app_status" type: "PythonTask" params: - script: | + code: | print("App status: simulated check OK.") dependencies: ["heartbeat_root"] @@ -44,26 +44,26 @@ dag: - task_id: "app_python_summary" type: "PythonTask" params: - script: | + code: | print("Application summary (simulated).") dependencies: ["app_logs_tail"] - # MERGE: quando entrambe le pipeline (system + app) sono complete + # Final merge of all branches - task_id: "merge_health" type: "PythonTask" params: - script: | + code: | print("Merging system and application health (simulated). Overall OK.") dependencies: ["sys_python_analysis", "app_python_summary"] - # LOG SU FILE + # LOG ON FILE - task_id: "write_heartbeat_log" type: "BashTask" params: command: "echo \"$(date) - Deep heartbeat completed\" >> /tmp/maestro_deep_heartbeat.log" dependencies: ["merge_health"] - # FINALE + # FINAL MESSAGE - task_id: "heartbeat_done" type: "PrintTask" params: diff --git a/examples/2_New_examples/6.2.1.scheduled_greeting_extended.yaml b/examples/2_New_examples/6.2.1.scheduled_greeting_extended.yaml index e71a546..fad5377 100644 --- a/examples/2_New_examples/6.2.1.scheduled_greeting_extended.yaml +++ b/examples/2_New_examples/6.2.1.scheduled_greeting_extended.yaml @@ -44,7 +44,7 @@ dag: - task_id: "python_check" type: "PythonTask" params: - script: | + code: | import platform print('Python check executed. Python version:', platform.python_version()) dependencies: ["show_time"] diff --git a/examples/2_New_examples/6.2.2.scheduled_greeting_language_mix.yaml b/examples/2_New_examples/6.2.2.scheduled_greeting_language_mix.yaml index 31a72d3..ce685be 100644 --- a/examples/2_New_examples/6.2.2.scheduled_greeting_language_mix.yaml +++ b/examples/2_New_examples/6.2.2.scheduled_greeting_language_mix.yaml @@ -1,7 +1,5 @@ abstract: | - This DAG expands the basic scheduled greeting example by introducing - a multilingual greeting flow executed immediately after the scheduled - start_time is reached. + This DAG expands the basic scheduled greeting example by introducing a multilingual greeting flow executed immediately after the scheduled start_time is reached. The pipeline performs: - A starting PrintTask announcing the scheduled execution @@ -37,7 +35,7 @@ dag: - task_id: "choose_language" type: "PythonTask" params: - script: | + code: | import random lang = random.choice(["EN", "IT", "ES"]) print(f"Selected language: {lang}") diff --git a/examples/2_New_examples/6.2.3.scheduled_greeting_with_checks.yaml b/examples/2_New_examples/6.2.3.scheduled_greeting_with_checks.yaml index 54ce8a3..ffa81e5 100644 --- a/examples/2_New_examples/6.2.3.scheduled_greeting_with_checks.yaml +++ b/examples/2_New_examples/6.2.3.scheduled_greeting_with_checks.yaml @@ -1,8 +1,5 @@ abstract: | - This DAG enhances the scheduled greeting concept by adding a small diagnostic - pipeline that runs immediately after the scheduled start_time. It mixes Python, - Bash, and Print tasks to validate directory state, create small data, and - log useful information before producing a final scheduled greeting message. + This DAG enhances the scheduled greeting concept by adding a small diagnostic pipeline that runs immediately after the scheduled start_time. It mixes Python, Bash, and Print tasks to validate directory state, create small data, and log useful information before producing a final scheduled greeting message. The flow: - Scheduled PrintTask announces execution @@ -45,7 +42,7 @@ dag: - task_id: "python_diag" type: "PythonTask" params: - script: | + code: | import math print("Python diagnostic: sqrt(144) is", math.sqrt(144)) dependencies: ["count_files"] diff --git a/examples/2_New_examples/6.3.1.scheduled_greeting_full_healthcheck.yaml b/examples/2_New_examples/6.3.1.scheduled_greeting_full_healthcheck.yaml index 5faf9d6..ea74a21 100644 --- a/examples/2_New_examples/6.3.1.scheduled_greeting_full_healthcheck.yaml +++ b/examples/2_New_examples/6.3.1.scheduled_greeting_full_healthcheck.yaml @@ -1,7 +1,5 @@ abstract: | - This DAG represents a high-complexity scheduled pipeline that performs a - multi-branch system healthcheck immediately after being triggered at the - specified start_time. + This DAG represents a high-complexity scheduled pipeline that performs a multi-branch system healthcheck immediately after being triggered at the specified start_time. It performs: - A scheduled initialization message @@ -67,7 +65,7 @@ dag: - task_id: "python_version" type: "PythonTask" params: - script: | + code: | import platform print('Python version:', platform.python_version()) dependencies: ["start"] @@ -75,7 +73,7 @@ dag: - task_id: "python_math_check" type: "PythonTask" params: - script: | + code: | import math print('Sanity check: sqrt(256) =', math.sqrt(256)) dependencies: ["python_version"] diff --git a/examples/2_New_examples/6.3.2.scheduled_greeting_with_warnings.yaml b/examples/2_New_examples/6.3.2.scheduled_greeting_with_warnings.yaml index 4d80cac..7f910fa 100644 --- a/examples/2_New_examples/6.3.2.scheduled_greeting_with_warnings.yaml +++ b/examples/2_New_examples/6.3.2.scheduled_greeting_with_warnings.yaml @@ -1,7 +1,6 @@ abstract: | This DAG is a difficult-level extension of the scheduled greeting concept. - It introduces a warning-style execution flow where some tasks can emit - notices about potential system issues without failing the pipeline. + It introduces a warning-style execution flow where some tasks can emit notices about potential system issues without failing the pipeline. After being triggered at the scheduled start_time, the DAG: - Runs standard initialization @@ -9,8 +8,7 @@ abstract: | * Disk capacity early-warning (Bash) * Memory usage early-warning (Bash) * Python environment sanity-check (Python) - - Each branch includes a task that might produce a “warning” message - (but never fails the DAG) + - Each branch includes a task that might produce a “warning” message (but never fails the DAG) - Merges the three warning lines into a final scheduled greeting This structure tests: @@ -46,7 +44,7 @@ dag: - task_id: "disk_warning" type: "PythonTask" params: - script: | + code: | # This is not a real fail: it's a warning generator print('WARNING: Disk free space could be evaluated as low (simulated).') dependencies: ["disk_free"] @@ -63,7 +61,7 @@ dag: - task_id: "mem_warning" type: "PythonTask" params: - script: | + code: | print('WARNING: Memory usage seems elevated (simulated).') dependencies: ["mem_usage"] @@ -73,7 +71,7 @@ dag: - task_id: "py_info" type: "PythonTask" params: - script: | + code: | import platform print('Python interpreter:', platform.python_version()) dependencies: ["start"] @@ -81,7 +79,7 @@ dag: - task_id: "py_warning" type: "PythonTask" params: - script: | + code: | print('WARNING: Python environment missing optional modules (simulated).') dependencies: ["py_info"] diff --git a/examples/2_New_examples/6.3.3.scheduled_greeting_deep_tree.yaml b/examples/2_New_examples/6.3.3.scheduled_greeting_deep_tree.yaml index a5ef4f1..f6c2f84 100644 --- a/examples/2_New_examples/6.3.3.scheduled_greeting_deep_tree.yaml +++ b/examples/2_New_examples/6.3.3.scheduled_greeting_deep_tree.yaml @@ -1,8 +1,6 @@ abstract: | This DAG represents the most complex variant of the scheduled greeting series. - Instead of a wide multi-branch structure, this version stresses the orchestrator - through a *deep linear pipeline* of many sequential tasks that mix Bash, Python, - and Print operations. + Instead of a wide multi-branch structure, this version stresses the orchestrator through a *deep linear pipeline* of many sequential tasks that mix Bash, Python, and Print operations. After starting at the specified start_time, the DAG performs: - Several low-level checks (environment, directory, process snapshot) @@ -60,7 +58,7 @@ dag: - task_id: "t5_generate_numbers" type: "PythonTask" params: - script: | + code: | nums = list(range(1, 11)) print("Generated numbers:", nums) dependencies: ["t4_process_snapshot"] @@ -68,7 +66,7 @@ dag: - task_id: "t6_square_numbers" type: "PythonTask" params: - script: | + code: | nums = list(range(1, 11)) squares = [n*n for n in nums] print("Square numbers:", squares) @@ -77,7 +75,7 @@ dag: - task_id: "t7_sum_squares" type: "PythonTask" params: - script: | + code: | nums = list(range(1, 11)) squares = [n*n for n in nums] print("Sum of squares:", sum(squares)) @@ -104,7 +102,7 @@ dag: - task_id: "t10_python_summary" type: "PythonTask" params: - script: | + code: | print("Final Python summary OK.") dependencies: ["t9_fake_compress"] diff --git a/examples/2_New_examples/7.1.1.Mixed_execution.yaml b/examples/2_New_examples/7.1.1.Mixed_execution.yaml index 4abc615..1f8f9c5 100644 --- a/examples/2_New_examples/7.1.1.Mixed_execution.yaml +++ b/examples/2_New_examples/7.1.1.Mixed_execution.yaml @@ -1,26 +1,31 @@ dag: name: "mixed_execution" tasks: + - task_id: "init" type: "PrintTask" params: message: "Starting mixed DAG..." dependencies: [] + - task_id: "generate_data" type: "PythonTask" params: code: | - import json + import json, os + print(os.getcwd()) data = {"values": [1, 2, 3, 4, 5]} with open("data.json", "w") as f: json.dump(data, f) print("Generated data.json") dependencies: ["init"] + - task_id: "count_values" type: "BashTask" params: command: "cat data.json | grep -o '[0-9]' | wc -l" dependencies: ["generate_data"] + - task_id: "finish" type: "PrintTask" params: diff --git a/examples/2_New_examples/7.2.1.mixed_execution_extended.yaml b/examples/2_New_examples/7.2.1.mixed_execution_extended.yaml index 1a787c2..68a2ce2 100644 --- a/examples/2_New_examples/7.2.1.mixed_execution_extended.yaml +++ b/examples/2_New_examples/7.2.1.mixed_execution_extended.yaml @@ -1,7 +1,5 @@ abstract: | - This DAG extends the original mixed_execution example by introducing - a multi-step data-processing workflow that mixes PrintTask, PythonTask, - and BashTask execution in a longer and more realistic sequence. + This DAG extends the original mixed_execution example by introducing a multi-step data-processing workflow that mixes PrintTask, PythonTask, and BashTask execution in a longer and more realistic sequence. The pipeline performs: - Initialization/logging @@ -35,8 +33,9 @@ dag: - task_id: "generate_data" type: "PythonTask" params: - script: | - import json + code: | + import json, os + print(os.getcwd()) data = {"values": [1, 2, 3, 4, 5, 6]} with open("data.json", "w") as f: json.dump(data, f) @@ -49,7 +48,7 @@ dag: - task_id: "extract_numbers" type: "BashTask" params: - command: "cat data.json | grep -o '[0-9]' > extracted.txt" + command: "cat data.json | grep -o '[0-9]' > extracted.txt; echo 'Created extracted.txt'" dependencies: ["generate_data"] # --------------------------------------------------------- @@ -67,7 +66,7 @@ dag: - task_id: "python_validation" type: "PythonTask" params: - script: | + code: | with open("extracted.txt") as f: lines = f.readlines() print("Python validation: extracted", len(lines), "numbers.") diff --git a/examples/2_New_examples/7.2.2.mixed_execution_branching.yaml b/examples/2_New_examples/7.2.2.mixed_execution_branching.yaml index 451141e..d163422 100644 --- a/examples/2_New_examples/7.2.2.mixed_execution_branching.yaml +++ b/examples/2_New_examples/7.2.2.mixed_execution_branching.yaml @@ -34,7 +34,7 @@ dag: - task_id: "generate_data" type: "PythonTask" params: - script: | + code: | import json data = {"values": [3, 5, 8, 13, 21]} with open("numbers.json", "w") as f: @@ -65,7 +65,7 @@ dag: - task_id: "python_compute_sum" type: "PythonTask" params: - script: | + code: | import json with open("numbers.json") as f: data = json.load(f) diff --git a/examples/2_New_examples/7.2.3.mixed_execution_random_data.yaml b/examples/2_New_examples/7.2.3.mixed_execution_random_data.yaml index c63d585..e5089f5 100644 --- a/examples/2_New_examples/7.2.3.mixed_execution_random_data.yaml +++ b/examples/2_New_examples/7.2.3.mixed_execution_random_data.yaml @@ -34,7 +34,7 @@ dag: - task_id: "generate_random_data" type: "PythonTask" params: - script: | + code: | import json, random values = [random.randint(1, 50) for _ in range(10)] with open("random.json", "w") as f: @@ -57,7 +57,7 @@ dag: - task_id: "python_stats" type: "PythonTask" params: - script: | + code: | import json with open("random.json") as f: data = json.load(f) @@ -73,7 +73,7 @@ dag: - task_id: "bash_count_digits" type: "BashTask" params: - command: "wc -l extracted.txt" + command: "echo 'Number of digits:'; wc -l extracted.txt" dependencies: ["python_stats"] # --------------------------------------------------------- diff --git a/examples/2_New_examples/7.3.1.mixed_execution_data_pipeline.yaml b/examples/2_New_examples/7.3.1.mixed_execution_data_pipeline.yaml index 90ab788..c794521 100644 --- a/examples/2_New_examples/7.3.1.mixed_execution_data_pipeline.yaml +++ b/examples/2_New_examples/7.3.1.mixed_execution_data_pipeline.yaml @@ -36,7 +36,7 @@ dag: - task_id: "generate_dataset" type: "PythonTask" params: - script: | + code: | import json, random values = [random.randint(10, 999) for _ in range(50)] with open("dataset.json", "w") as f: @@ -69,7 +69,7 @@ dag: - task_id: "compute_stats" type: "PythonTask" params: - script: | + code: | with open("digits_grouped.txt") as f: content = f.read().strip() if not content: @@ -103,7 +103,7 @@ dag: - task_id: "branch_b_split" type: "PythonTask" params: - script: | + code: | with open("digits_grouped.txt") as f: data = f.read().strip() chunks = [data[i:i+5] for i in range(0, len(data), 5)] diff --git a/examples/2_New_examples/7.3.2.mixed_execution_transform_tree.yaml b/examples/2_New_examples/7.3.2.mixed_execution_transform_tree.yaml index c74c103..cbe8de1 100644 --- a/examples/2_New_examples/7.3.2.mixed_execution_transform_tree.yaml +++ b/examples/2_New_examples/7.3.2.mixed_execution_transform_tree.yaml @@ -34,7 +34,7 @@ dag: - task_id: "generate_dataset" type: "PythonTask" params: - script: | + code: | import json, random data = {"values": [random.randint(1, 100) for _ in range(20)]} with open("transform_data.json", "w") as f: @@ -51,13 +51,13 @@ dag: - task_id: "A1_scale_values" type: "PythonTask" params: - script: | + code: | import json with open("transform_data.json") as f: data = json.load(f) scaled = [v * 2 for v in data["values"]] with open("scaled.txt", "w") as out: - out.write("\\n".join(map(str, scaled))) + out.write("\n".join(map(str, scaled))) print("Branch A scaled values:", scaled) dependencies: ["generate_dataset"] @@ -73,13 +73,13 @@ dag: - task_id: "B1_label_values" type: "PythonTask" params: - script: | + code: | import json with open("transform_data.json") as f: - data = json.load(f) + data = json.load(f) with open("labeled.txt", "w") as out: - for v in data["values"]: - out.write(f\"Value_{v}\\n\") + for v in data["values"]: + out.write(f"Value_{v}\n") print("Branch B labeled values written to labeled.txt") dependencies: ["generate_dataset"] @@ -95,7 +95,7 @@ dag: - task_id: "C1_compute_stats" type: "PythonTask" params: - script: | + code: | import json with open("transform_data.json") as f: data = json.load(f) diff --git a/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml b/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml index e21b7c5..a41bf17 100644 --- a/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml +++ b/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml @@ -39,7 +39,7 @@ dag: - task_id: "generate_data" type: "PythonTask" params: - script: | + code: | import json, random data = {"values": [random.randint(1, 500) for _ in range(30)]} with open("hyper_data.json", "w") as f: @@ -69,13 +69,13 @@ dag: - task_id: "B1_load_and_double" type: "PythonTask" params: - script: | + code: | import json with open("hyper_data.json") as f: data = json.load(f) doubled = [v*2 for v in data["values"]] with open("B_doubled.txt", "w") as out: - out.write("\\n".join(map(str, doubled))) + out.write("\n".join(map(str, doubled))) print("Branch B doubled data.") dependencies: ["generate_data"] @@ -89,7 +89,7 @@ dag: - task_id: "C1_compute_stats" type: "PythonTask" params: - script: | + code: | import json with open("hyper_data.json") as f: data = json.load(f) @@ -114,13 +114,13 @@ dag: - task_id: "D1_extract_even" type: "PythonTask" params: - script: | + code: | import json with open("hyper_data.json") as f: data = json.load(f) evens = [v for v in data["values"] if v % 2 == 0] with open("D_evens.txt", "w") as out: - out.write("\\n".join(map(str, evens))) + out.write("\n".join(map(str, evens))) print("Branch D extracted even numbers.") dependencies: ["generate_data"] @@ -155,7 +155,7 @@ dag: - task_id: "S2_summary_python" type: "PythonTask" params: - script: | + code: | print('Second-stage summary: Python confirming prior merges completed.') dependencies: ["merge_primary"] diff --git a/maestro.db b/maestro.db deleted file mode 100644 index 0035e55ae88f7d709973afa6684e6a9025a1d85f..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 114688 zcmeHQe{3YzUEj66Uhn$HUVb>vuS@2fUTxPm>-q8Kmy>JHx%iT6FL&wrTw29pnVorS zPqLnIcW3X;N-ycA(2@$;Lr@SPAOZ=5_yZ&)5C{n*R8flfAyon-pimG{1X3wTfK*ZV zzM0+i?#%AR-dp=F$#)a`yz^$>eBS4Mzu))1Z{GJiYahMf_>?Pom2K1K+tg^{4_tK@W)8=3;dgG`@7elUr*$(zB!rupF}qE*+lMm?oTKGXzH62zd1fW z@k7~%vR{RR_(2Ac0b~FfKn9QjWZ*r?K>b)YJ#&dq)2>FN!Q81@Thy+VX{GS8 zSF9FXdfhKnYHq=IwrR9<{Osk`l`E^<`HRo2ew=&T%ehMzxo8#Ke8(LY>iMzs49_Ru z$oXcm9J^!m@=OE4m6fM2tahn9n{jOJV=I@>KDTmtUNpo7?&75@+{N_^7fx~6OvkOc za~CeHTw#Zy;f1Pi!u<5<3zwemxZ;J&=bvA>{0Z(OtDoTJ9eW|W@S)7;%o?9e1PoR` zS%ztC!L0dS@IB6_7;QoJjn_ugGcz;EA9yinv2Rvift$mxbVu8QL)?6OgF2Ae=4N59 ztsO=5I<;!P;ZJ;7G-%b^#{m&p;1fIsg1BJ8XH|J7wCXz~8D+({8Ju zo$a_)S_$s!Rcu-TF}bq(?CND!6yfm{7s5ZqwPRgiXP&!sdG-9W7a8O8;YkbJ<<)bm zmsc;IU0q|>RfC?0p4(AT`xUK%@Ve~6LV9%O^h~cF40tby!9Qbf+>=VrOiw4jKNa*< z+1sqLpV5wELJYcz@!2NYc6k+Vl1j zdO&jIk0@aPEH22Kvxs~+`SGcq5mqF346qo^6yR!293tf~KjNshR=**MTy_(6O z7DV=Y7>3y~Hq@rzC!6~d_UEIy@8td~_wC%j<^DPM_qo5x{blZJVF4?|4>EuZAOpw% zGJp&q1IPd}fD9l5$N(~c4BSNwJeUPAGa2~x0CVu~Ps6YK*rWSGMD90oU(LOlJC&>DF64eT_kFp)$~~F;78K$K89)Y*0b~FfKn9QjWB?gJ z29N<{02z3XF)%fJBKfjcF7FmBui}g!8$Ox5PF=?=7ihWcRI~RDFD6}Dx$4!bg)OrR z`*Zxv{lk3Huhgo(<8H!6t}VCfxnu0avRT{QVh4+`!N++e%}%sDuUv4Q%`JAeS*D{i z!*bHwfvqGCRA3b<9(Daq>ZVu)MYv_PR4W&@Jnt&4j4(JwuWT1K%}ul7xWnwEnp>%r zN@dtUbPYBfRELfY&t;mJYgXB@C!r6=Gk=lDRi^%aYJKv{lj6i%6UWA1&weXggv0nj z29N<{02x3AkO5=>8F*(fP#+r~TR-vm%F1F&Q%a^RTa-5hQ{jniQ(o7p#7iQPiDAn{ zOOf|xwXKcik{~I(DDslRiORA>mL(~#6GPJVc!`o1Rh@{cC<~II3#uR(iYlsC>*kN2 z`E)&-9a}#cKv9H}K_x}!O~J5uqU$F9u*_+muC3w0ObPRh|^7z!wEoPhqhBrhvU zUV-b40t6WXZeep!Xs`G)+)?p=jF@iFl5^ z@JP#w0kW$~Dv99{guEyb5QFY`vcoNvS1oE16PTf5kq3)YvTdio@$`FCEKtiQH^-a zz3~t&ZwQ3MBwAV~dLG0_0N%UW!|~+U`dlbLHq~rJvw#w?%7hec-qa+U7bUBx$wbf% zp=A#PZ@o7fSp89)Y*0b~FfKn9QjWZ)gkfIPB(B6+ZNyK!Y?{bcf>3v5|^ zWPLH&$4Xen7+L3&{VZ9H1-I;DZ7ChzvX4ce(Wge%89)Y*0b~Ff zc=s^y@Nu^P|3vz!#Kd1t{^B^FekxlS`@Qk+X1*}>O8PrFW3n(cH~t&x-^%_|=GloK z&dyBz$JnQGzccy300(@#=bjoCiBNrzdWYqYVRIauqDUJG zC)Sg;=hD6S>HX~5JaIO)2p85|c!3hUTCK3N>u-7P!Rp{vQgiKf6wT_^!D@JXIW>JU zgx?->y`L?lkFBN_;nGM04qD6cv#GiET38bfTF0ZOQ`3uKBO2=_Q*-S#vG(*+&jWki z6Nwbx&j#S>uHDl|MfZ28GA!E2(KYl(#N% z>igL+ox6})gc>3u*Jou@A5P7+SH`5S&#ET(1TPZpzMq}k6I}(bkBY{-2wor6WcN&U zd#t>&kGe3~BR5S8qYI?47^b8{O z_p89)Y*0b~Ffc&9VKp8uo&|4#2k#E%Rh1IPd}fD9l5$N(~c3?Ku@05X6K+?oOO z|8LC?mLdbl05X6KAOpw%GJp&q1IPd}fD9l5?*<0g{{Lk9hZFEWevkoV02x3AkO5@i zPB3tDCz+mm>v%mUcLW2zI%}Jo1;?IU=4LwzW>0amkVBy27kp=%La|#bm*JRI@!Uey z+M;%?+;-SEtFKhytXF3{Vg`dFtgLa1&O)e;-M!?nL_mEEh9In-Kt&-Qo0iR8s7rz)@XF zl$W05e5hc<<+xj+J-2fH0_VAKu2T(ony1$tpR+)2IB|hth3nZ75n5fY(zmN`ZT;-o z)wQ)=RRVOVvgB;IOho+M9i|GitiM=y&E9~6Em03sa6tL1v)weIQ4CkyHM+2eooUbP z;i?8xgl?)~2b4P7J6^@-DuJY*;;OsVQyeo$p#p?5%7I$AwQXAQILoZo%rfQtiV3m< zUBBUOHsrVv*1?@_RF)6F&w~)|R6vF1OB=H-MR*}do6^+1mMTCWR=k>P&qpdf#YqdG z5Kt9U%k`=WkW8-XZI&IAvmH*z6K)4I)TWkWQ&5U31^p{wmFxKs0J`sYgSNYc<+#+o(!#pC%3E1iJk27%7A4)I2bpm||x$Ox)1_NBIeOFp#A& z^#9TSchUd%doLcpGyQ*}h5r8+6X`F$$LC+zd}II_Kn9QjWZ=#*aMK-5pM2{PSOsWL zuiU-_(0&~Ke|-Lr&;Q#WXmV{^3ef*Y|DSD+X>2OM=l}TpA2!P!(&n*urvI*IX(OPPPj zYHGZDmu51F#r3)A*H;!J)0+LfhqD!) z#X_9UUp#l|#b)%@OD|S^+pGC61+jK-AC1@Ea)m`cC`MkFiL6VaEJ%i?>k?5FLnrn7 zAIKz5u1`;!fY>YBg-vtQ1ijkNyEa)2On~x*X4J+@+^xet;;`hrA&7>cM`7{P4A$cG zjf2B#u^1{sh%ktqVmo?X+9NLv^JxZDbqg2jhF2F9O%|hg=kE{kR*wvC+W;WsMY%gv zK~rPY&fdqM&P~5^8&Kl{%ANIC04$5mY(#0nyuG~J`GBGuNKGULF`|5)zBe0ytUo?b zlr1%CMb$A4Yl*;cS4C8Qej_9JIL&8ldWNeG`Tn4FGd{bPZ@p zaMReTVr(!DAQMrE&35=?C>daE4+U|brk3H=jkqK;|Nnu+A0%=gnfk%WznhdMULXI* z@ulq5%&%sK$3B|=xzTTqemM0LBi|S~KI{*DdFX*;5l+7QKJ^FrOo9(b49oM%1=rc! z^5N+KBpM;QC>0IS;4MLxcw*=puWOdX+p<9=kw}uNNVh#`$jh>l*N7kqF^LdQF=)Y{ z(GRr48#e;XszwwE#=6c`kW-5xW*9e)5OYiM*wCRzXTTv`RafFlcj^Oo97@wFu1|!aa-&TW&jC*DzNAv$7bv+^Vt9>8$bq(bg*McqYY}8JZ9GU+BF3Xx&CB&vYI`$G|dVdStt_cfA)M9YSt#n#atIhjewp$YbOApet7 z^&P8F@gNn_CUp(?h_;~cs%R=a0jtcLB~9T4N)%Bwh-FEGu)?B|7l=-j*i4wbkD&{! zaG&T7(-4b#-q02BH=}Z}^h5|74Dr#zZb@&j8WEiU$;%8eEStwColCQK17azzYoeMr zh$ad#ju%ca#DO`!U5*E3kpZi&XhM8O{0x^3uriB0c(8*u%0V+^Ly65V)W;jm0Hb`= z&1gw>(=IE~hXo+rjDGmLbRX9 za08ee3$PR*Gb0k^Q+PC7X=Tr~4vy^bOmgsq(2$9!#UL*}5<&)(d~}doLd(o?OkhK_ z5}3ExsC@F_aPf~l-MS5e2V>a6=dOgns1X%}GY{=84i6USV2xU%38D=1`iRpcJ-D|x ze3VUS$wt$l!PD%vOn-aNyEX7eBA~S&sO>j_u|i0<}iUjm4hq5qHmKl=aZ|D*qp z{y+Nvo$rc8|9>C;zl{0+?$#M0;z9MbtkN!XU|9-)A?Lvk6m0g%T?8E;jB+UP(C2|+1esc29C-W2a z@vn_@+0SOamN`E5iS)0fhej`^em#{OxiI|Z&^LzUQzplp(xC z5lRqLuRv_4U|2lSb)7F3H4|d(?V?zeEuo~Ut)QKFOi3$Lqptuc!h#Zs4yn9iF*`G7 zTLDlmtz7kL)xwrph4gH`Y@4#Ei54%Iy2_Ix75Ji{s=R5bvgXW1f?QYt6p6;8hG<0=LmLTjsh>R4 z3V^yDv~gkWCp%Vv6T-K|c*OUa5A6j&4HjritJYn30ZOq%&k&!lJovb-Fi6(u=F@CT$2Etuasgf^c3fsnie`X+DCcEAqG z_ySH*Rt3nQ6ER$8guTh+V1W+W_6ALm#dyYZX=!gVd6Z3P33Ai)N@~pXGSCvFHHi%v zOZiYf!1Nvx<~}WNfLK=KsIkEP|L={*OV~%pIUjk@9j{ZOT|LFgt|BwDZ`v2(vqyK+k|6kOZ|9{WO z&$9UcFHHW+88|`)>i4c^ z5+}mA(3ib(dADGB6-QG_rYu_!+AgqH$=f#Nb)8DQBodhzwp_Fn`L;ts;r%E|UQi81 zj=kq~@d^VKyh{FHpbjq@6ws2q0ciDj!1!8-7NS9q1Z^vRjs=1W(a81&P*s*=F^k+~ z1~iDR>=Wo z1MC1%-r8WTt)D%+y0+HKW(jed72ub35*uQdRvY+Xh&?j=EsAp|1@+m;6fNx07d%&>iTwReku;%-qTK$5&PoV=dih0(wy1EpK}VG67OM zX6jitYpS5KTp;bacY;KNEYqB<)>Fqbiy?|k3q_cWZ=g67P!{OEZa54=ROPitGSiCz z8A)cJBO68cljN{B4MFZgNmYf|?O6#+=VJ`j#~#ish5({G3>u3HsW$}3j27tNDpG2c~_miuwSEmH2l!yjAMDHJ# z7$P`=L-ou|W-&w%?UaFoku?~M;+aw)Q=Dw5ilo%j4`k*-ESO?RkDA9-4O_ezjPN`SE~cKe_NvT3TID);wNl|f^HbW4U1&m6{bF{v>3QtG4k zbytJ$Y^MR4djWTYrL*m5~P)d zY_lSfB#^M)ogyoFP1cAUCxVp_B1i;O<(^)uGH@^|G+2(>a(NlRz@(U{Vm*DVyQ&Nr ziJVtpG5|dj?G_nI6j@i4`cN)2E$^8yZ^w7G9mqIo6)GNe{Y~mZ4mLxy1%+2dwirhV zv)rV$qv5lU1+)3w(oI&4`HQZQFbwQ^z0^RCj0L=84n zQeHH5=o_)9!jh>7U1gfEUMduATOz&HMu6qJz<|WKU?pr>R0mr&-piH^5R0^|7?2|q zEJ)O>hb2UWPkl7o-Ifg+h%SL8iS|hVL}v-FwO2El>2O9EOvQ>`*)D9Fn`Xsv71g39 zF+oMBSma@~y2!(lt<75{s|cC1Wf(-YUL68=kn_3$O4Wq~InPwc(1z-%v0n4hfdbJC z$iW;n=pjTnZjkzLy8G;WpeVq;Xi8T)G|`Y%^4e&Jdeq!XtyF>qwZb)6R;!wnYD6oR zAf+aO)ZZ`=@!-;FJk=DUlx)M6M74Ks5FJJ}!GMve(|k(HL5V?BFkNG*Ug|MuEMT(4 z*sLeOBFG9LtdEX#SC7F0(LfLdO^JF;00c^80%`s}HJq8A+Z$4vUQ*2}J3u+rY(=v` zE8w9HAw`=vHOb~h$tr5>YD2h1kHD~{f`Y_+18EsbG(k~xn0$>6WfsHRM+cJuBY|9( zByJ@bMle^pQ6EWW=0YHy!^j|Eh~U4%lYr<54}^$$1*WNnFgf+}iIK@)NaQ{<_1M(^ z!XNlS29N<{02x3AkO5=>89)Y*0b~FfxKj-5p2$p}*mIZ-2i}Fg(6ydw*i<$HC)kSaXNPn4vKK|3?$) zA3Y?Fv07vR89)Y*0b~Ff*bf6YUmj1Ncxwi70PIad+A{#`9mV*6jQ_{@|K9s{H)c27 zlC$A5J@I#U=0v2&KT zyluKIe~*{mJJv75e0Hw z2Tsh}&^?XxVWoS8`mZx<&1LsNtb%u?w5l>IEkCOu}bx+_=#) R9R1pn_C*%900FUA{|}Q Date: Tue, 2 Dec 2025 07:47:19 +0100 Subject: [PATCH 20/38] Updated examples + testing passage of files --- .../7.3.3.mixed_execution_hyperpipeline.yaml | 4 +- .../8.1.1.save_and_reload_basic.yaml | 40 +++++++++ .../8.2.1.generate_filter_summarize.yaml | 58 +++++++++++++ .../8.2.2.json_extract_process.yaml | 50 ++++++++++++ .../8.2.3.parallel_save_and_merge.yaml | 57 +++++++++++++ .../8.3.1.csv_fanout_merge.yaml | 54 +++++++++++++ .../8.3.2.deep_transform_tree.yaml | 52 ++++++++++++ .../8.3.3.massive_fanout_consolidation.yaml | 76 ++++++++++++++++++ maestro.db | Bin 0 -> 45056 bytes runtime/tmp/maestro_heartbeat.log | 1 - 10 files changed, 388 insertions(+), 4 deletions(-) create mode 100644 examples/2_New_examples/8.1.1.save_and_reload_basic.yaml create mode 100644 examples/2_New_examples/8.2.1.generate_filter_summarize.yaml create mode 100644 examples/2_New_examples/8.2.2.json_extract_process.yaml create mode 100644 examples/2_New_examples/8.2.3.parallel_save_and_merge.yaml create mode 100644 examples/2_New_examples/8.3.1.csv_fanout_merge.yaml create mode 100644 examples/2_New_examples/8.3.2.deep_transform_tree.yaml create mode 100644 examples/2_New_examples/8.3.3.massive_fanout_consolidation.yaml create mode 100644 maestro.db delete mode 100644 runtime/tmp/maestro_heartbeat.log diff --git a/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml b/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml index a41bf17..732b617 100644 --- a/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml +++ b/examples/2_New_examples/7.3.3.mixed_execution_hyperpipeline.yaml @@ -1,7 +1,5 @@ abstract: | - This DAG represents a “hyperpipeline” workflow — a difficult-level DAG with - multiple levels of branching, cross-branch transformations, wide fan-out and - multi-step merging. It is designed to push the orchestrator to its limits in + This DAG represents a “hyperpipeline” workflow — a difficult-level DAG with multiple levels of branching, cross-branch transformations, wide fan-out and multi-step merging. It is designed to push the orchestrator to its limits in terms of: - Concurrency diff --git a/examples/2_New_examples/8.1.1.save_and_reload_basic.yaml b/examples/2_New_examples/8.1.1.save_and_reload_basic.yaml new file mode 100644 index 0000000..18d8d9d --- /dev/null +++ b/examples/2_New_examples/8.1.1.save_and_reload_basic.yaml @@ -0,0 +1,40 @@ +abstract: | + This DAG demonstrates the most essential "write → read" workflow. + A PythonTask writes a small text file to /tmp/maestro_data/, a second PythonTask loads the file back into memory, and a PrintTask confirms completion. + It is the minimal example of inter-task data persistence and file-based communication within Maestro. + +dag: + name: "save_and_reload_basic" + + tasks: + + - task_id: "write_file" + type: "PythonTask" + params: + code: | + import os + path = "/tmp/maestro_data" + os.makedirs(path, exist_ok=True) + + filepath = os.path.join(path, "hello.txt") + with open(filepath, "w") as f: + f.write("Hello world from Maestro!") + + print(f"Written file: {filepath}") + dependencies: [] + + - task_id: "read_file" + type: "PythonTask" + params: + code: | + filepath = "/tmp/maestro_data/hello.txt" + with open(filepath) as f: + content = f.read() + print("Read content:", content) + dependencies: ["write_file"] + + - task_id: "finish" + type: "PrintTask" + params: + message: "Basic file save/load completed!" + dependencies: ["read_file"] diff --git a/examples/2_New_examples/8.2.1.generate_filter_summarize.yaml b/examples/2_New_examples/8.2.1.generate_filter_summarize.yaml new file mode 100644 index 0000000..0b03c7a --- /dev/null +++ b/examples/2_New_examples/8.2.1.generate_filter_summarize.yaml @@ -0,0 +1,58 @@ +abstract: | + This intermediate-level DAG builds a simple numeric processing pipeline. + A PythonTask generates a numeric dataset and stores it in /tmp/maestro_data/. + A BashTask filters the values using a command-line expression, creating a refined dataset. + Finally, a PythonTask performs lightweight statistical analysis on the filtered results. + The DAG demonstrates cross-language task cooperation and basic sequential data transformation. + +dag: + name: "generate_filter_summarize" + + tasks: + + # Create dataset + - task_id: "create_dataset" + type: "PythonTask" + params: + code: | + import os, random + + path = "/tmp/maestro_data" + os.makedirs(path, exist_ok=True) + + outfile = f"{path}/numbers.txt" + with open(outfile, "w") as f: + for _ in range(20): + f.write(str(random.randint(1,100)) + "\n") + + print("Generated:", outfile) + dependencies: [] + + # Bash filters numbers ≥ 50 + - task_id: "filter_numbers" + type: "BashTask" + params: + command: "grep -E '^[5-9][0-9]$' /tmp/maestro_data/numbers.txt > /tmp/maestro_data/high_numbers.txt || true" + dependencies: ["create_dataset"] + + # Python computes summary + - task_id: "summarize" + type: "PythonTask" + params: + code: | + import numpy as np + + with open("/tmp/maestro_data/high_numbers.txt") as f: + data = [int(x) for x in f.read().split()] + + if data: + print("High numbers summary: min =", min(data), "max =", max(data), "avg =", sum(data)/len(data)) + else: + print("No high numbers found.") + dependencies: ["filter_numbers"] + + - task_id: "finish" + type: "PrintTask" + params: + message: "Intermediate pipeline completed." + dependencies: ["summarize"] diff --git a/examples/2_New_examples/8.2.2.json_extract_process.yaml b/examples/2_New_examples/8.2.2.json_extract_process.yaml new file mode 100644 index 0000000..862ea3f --- /dev/null +++ b/examples/2_New_examples/8.2.2.json_extract_process.yaml @@ -0,0 +1,50 @@ +abstract: | + This DAG processes structured JSON data across Python and Bash tasks. + It begins by generating a random-value JSON file. + A BashTask uses jq to extract values into a flat, line-based file. + A final PythonTask loads these extracted values and performs counting and parity-based analysis. + The workflow highlights hybrid processing and external tool integration for JSON pipelines + +dag: + name: "json_extract_process" + + tasks: + + - task_id: "make_json" + type: "PythonTask" + params: + code: | + import json, os, random + + path = "/tmp/maestro_data" + os.makedirs(path, exist_ok=True) + + data = {"values": [random.randint(0,10) for _ in range(15)]} + with open(f"{path}/data.json", "w") as f: + json.dump(data, f) + + print("JSON created:", data) + dependencies: [] + + - task_id: "extract_values" + type: "BashTask" + params: + command: "jq '.values[]' /tmp/maestro_data/data.json > /tmp/maestro_data/flat_values.txt" + dependencies: ["make_json"] + + - task_id: "analyze" + type: "PythonTask" + params: + code: | + with open("/tmp/maestro_data/flat_values.txt") as f: + arr = [int(x) for x in f.read().split()] + + print("Count =", len(arr)) + print("Even numbers =", [x for x in arr if x % 2 == 0]) + dependencies: ["extract_values"] + + - task_id: "finish" + type: "PrintTask" + params: + message: "JSON extract & analyze completed!" + dependencies: ["analyze"] diff --git a/examples/2_New_examples/8.2.3.parallel_save_and_merge.yaml b/examples/2_New_examples/8.2.3.parallel_save_and_merge.yaml new file mode 100644 index 0000000..b8d4dc0 --- /dev/null +++ b/examples/2_New_examples/8.2.3.parallel_save_and_merge.yaml @@ -0,0 +1,57 @@ +abstract: | + This DAG illustrates parallel fan-out and fan-in behavior. + Two Python tasks independently generate text files in /tmp/maestro_data/, operating concurrently. + Two Bash tasks then read each file individually. + Finally, a PrintTask acts as a merge point, executing only after all branches complete. + The example emphasizes Maestro’s concurrent execution capabilities and multi-branch synchronization. + +dag: + name: "parallel_save_and_merge" + + tasks: + + - task_id: "init" + type: "PrintTask" + params: + message: "Starting parallel save/load workflow..." + dependencies: [] + + - task_id: "save_a" + type: "PythonTask" + params: + code: | + import os + path="/tmp/maestro_data"; os.makedirs(path, exist_ok=True) + with open(f"{path}/A.txt", "w") as f: + f.write("Alpha content") + print("Saved A.txt") + dependencies: ["init"] + + - task_id: "save_b" + type: "PythonTask" + params: + code: | + import os + path="/tmp/maestro_data"; os.makedirs(path, exist_ok=True) + with open(f"{path}/B.txt", "w") as f: + f.write("Beta content") + print("Saved B.txt") + dependencies: ["init"] + + - task_id: "read_a" + type: "BashTask" + params: + command: "cat /tmp/maestro_data/A.txt" + dependencies: ["save_a"] + + - task_id: "read_b" + type: "BashTask" + params: + command: "cat /tmp/maestro_data/B.txt" + dependencies: ["save_b"] + + - task_id: "merge" + type: "PrintTask" + params: + message: "Both A and B files processed!" + dependencies: ["read_a", "read_b"] diff --git a/examples/2_New_examples/8.3.1.csv_fanout_merge.yaml b/examples/2_New_examples/8.3.1.csv_fanout_merge.yaml new file mode 100644 index 0000000..b8f847e --- /dev/null +++ b/examples/2_New_examples/8.3.1.csv_fanout_merge.yaml @@ -0,0 +1,54 @@ +abstract: | + This advanced DAG processes a single CSV dataset through three independent transformation branches. + Branch A sorts the numbers, Branch B filters high-valued entries using Bash, and Branch C computes statistics. + Once all branches finish, their results converge in a final PrintTask. + This pattern exercises parallelism, cross-branch data divergence, and fan-in consolidation in a real-world mini workflow. + +dag: + name: "csv_fanout_merge" + + tasks: + + - task_id: "create_csv" + type: "PythonTask" + params: + code: | + import os, random + path="/tmp/maestro_data"; os.makedirs(path, exist_ok=True) + with open(f"{path}/data.csv", "w") as f: + for _ in range(30): + f.write(str(random.randint(1,100)) + "\n") + print("CSV created.") + dependencies: [] + + - task_id: "branch_a" + type: "PythonTask" + params: + code: | + vals = [int(x) for x in open("/tmp/maestro_data/data.csv")] + with open("/tmp/maestro_data/sorted.txt","w") as f: + f.write("\n".join(map(str, sorted(vals)))) + print("Branch A sorted numbers written.") + dependencies: ["create_csv"] + + - task_id: "branch_b" + type: "BashTask" + params: + command: "grep '^[5-9][0-9]$' /tmp/maestro_data/data.csv > /tmp/maestro_data/high.txt || true" + dependencies: ["create_csv"] + + - task_id: "branch_c" + type: "PythonTask" + params: + code: | + vals=[int(x) for x in open("/tmp/maestro_data/data.csv")] + with open("/tmp/maestro_data/stats.txt","w") as f: + f.write(f"min={min(vals)} max={max(vals)} avg={round(sum(vals)/len(vals), 2)}") + print("Branch C stats created.") + dependencies: ["create_csv"] + + - task_id: "merge" + type: "PrintTask" + params: + message: "Fanout/merge CSV pipeline completed!" + dependencies: ["branch_a", "branch_b", "branch_c"] diff --git a/examples/2_New_examples/8.3.2.deep_transform_tree.yaml b/examples/2_New_examples/8.3.2.deep_transform_tree.yaml new file mode 100644 index 0000000..9caa011 --- /dev/null +++ b/examples/2_New_examples/8.3.2.deep_transform_tree.yaml @@ -0,0 +1,52 @@ +abstract: | + This DAG models a multi-stage transformation tree with Python and Bash cooperating across nested steps. + Raw JSON is transformed into a flat list, reshaped through sorting and deduplication, and finally aggregated into summary metrics. + The pipeline demonstrates complex sequential transformations where each stage depends on intermediate file products. + It showcases deep, multi-layered processing and reusable artifact generation. + +dag: + name: "deep_transform_tree" + + tasks: + + - task_id: "raw_json" + type: "PythonTask" + params: + code: | + import os, json, random + path="/tmp/maestro_data"; os.makedirs(path, exist_ok=True) + data={"items":[random.randint(1,50) for _ in range(15)]} + json.dump(data, open(f"{path}/raw.json","w")) + print("Raw JSON created.") + dependencies: [] + + - task_id: "extract_list" + type: "PythonTask" + params: + code: | + import json + vals=json.load(open("/tmp/maestro_data/raw.json"))["items"] + open("/tmp/maestro_data/list.txt","w").write("\n".join(map(str,vals))) + print("List extracted.") + dependencies: ["raw_json"] + + - task_id: "bash_transform" + type: "BashTask" + params: + command: "sort -n /tmp/maestro_data/list.txt | uniq > /tmp/maestro_data/unique_sorted.txt" + dependencies: ["extract_list"] + + - task_id: "final_aggregate" + type: "PythonTask" + params: + code: | + vals=[int(x) for x in open("/tmp/maestro_data/unique_sorted.txt")] + print("Final sum=", sum(vals)) + print("Count=", len(vals)) + dependencies: ["bash_transform"] + + - task_id: "finish" + type: "PrintTask" + params: + message: "Deep transform tree completed!" + dependencies: ["final_aggregate"] diff --git a/examples/2_New_examples/8.3.3.massive_fanout_consolidation.yaml b/examples/2_New_examples/8.3.3.massive_fanout_consolidation.yaml new file mode 100644 index 0000000..31605d3 --- /dev/null +++ b/examples/2_New_examples/8.3.3.massive_fanout_consolidation.yaml @@ -0,0 +1,76 @@ +abstract: | + This DAG performs a high-parallelism numeric partitioning workflow. + Starting from a 50-number seed file, five BashTask branches independently filter different numeric ranges in parallel. + A PythonTask then merges all filtered subsets and computes global statistics. + The example stresses Maestro’s ability to handle wide fan-out, large intermediate files, and multi-branch consolidation at scale. + +dag: + name: "massive_fanout_consolidation" + + tasks: + + - task_id: "seed" + type: "PythonTask" + params: + code: | + import os, random + path="/tmp/maestro_data"; os.makedirs(path, exist_ok=True) + nums=[random.randint(1,200) for _ in range(50)] + open(f"{path}/seed.txt","w").write("\n".join(map(str,nums))) + print("Seed dataset created.") + dependencies: [] + + # 5 parallel branches filtering different ranges + - task_id: "range_1_40" + type: "BashTask" + params: + command: "awk '$1>=1 && $1<=40' /tmp/maestro_data/seed.txt > /tmp/maestro_data/r1.txt" + dependencies: ["seed"] + + - task_id: "range_41_80" + type: "BashTask" + params: + command: "awk '$1>=41 && $1<=80' /tmp/maestro_data/seed.txt > /tmp/maestro_data/r2.txt" + dependencies: ["seed"] + + - task_id: "range_81_120" + type: "BashTask" + params: + command: "awk '$1>=81 && $1<=120' /tmp/maestro_data/seed.txt > /tmp/maestro_data/r3.txt" + dependencies: ["seed"] + + - task_id: "range_121_160" + type: "BashTask" + params: + command: "awk '$1>=121 && $1<=160' /tmp/maestro_data/seed.txt > /tmp/maestro_data/r4.txt" + dependencies: ["seed"] + + - task_id: "range_161_200" + type: "BashTask" + params: + command: "awk '$1>=161 && $1<=200' /tmp/maestro_data/seed.txt > /tmp/maestro_data/r5.txt" + dependencies: ["seed"] + + # Merge final stats + - task_id: "merge_ranges" + type: "PythonTask" + params: + code: | + import glob + vals=[] + for file in glob.glob("/tmp/maestro_data/r*.txt"): + vals.extend(int(x) for x in open(file)) + print("Total collected values:", len(vals)) + print("Sum:", sum(vals)) + dependencies: + - "range_1_40" + - "range_41_80" + - "range_81_120" + - "range_121_160" + - "range_161_200" + + - task_id: "finish" + type: "PrintTask" + params: + message: "Massive fan-out consolidation complete!" + dependencies: ["merge_ranges"] diff --git a/maestro.db b/maestro.db new file mode 100644 index 0000000000000000000000000000000000000000..e8dcc8272395834eb43ec64d0ab805cc6e4c1041 GIT binary patch literal 45056 zcmeI&+iv4T7{GCRw@n&%OUp%NfrK=@B9_`k(Ow~hz-`@9ktU1lqFr%ed7Ei#$cfll z^$OH*kdSyE-T^Ln4IY6@X52Q3y^U6j#Etn!vg3>oGvDv~CI>kuUo_)Xiu1v!7p3Bf zv0<2|@u?7oVN}$zs+PNle!X<}LOq*v+lw|U#^aa2tgZiTShb&x_4n5QT>Im#zaIUz zy87r>>mBQN)yM+@1Q0*~0R#~EUkY4rSXH~dV_rAnen(!#S79_v&MswV+?Av7>%og8 z?8~b(9F6;78uw&z@9LiKxq&B+TKnEJ@!HcxyCsU35cOOS&h_J!s=c#gPS(@tMfa5+ zi>KFaM-aHXP4AwITeY|&zI6S)&s@LW_^{!KRyz=_(`Iv9ShZZ&;-J}f1Kq3w50f-f z-%od&?cLlNPyM4~*MBZP_nwP-+;J@DlUl_-*)fewF!{EtzSf4(I2~lKU-fJhBd{jl zRjRgan?HS(MVv;-H>z{Bs^((LnnZm*pxnqh(M33mHMdJ%$+K}95BiG^(7}$AJ10$` zT1KhtgzC%rzRGZLbnNNovq5jzmG|sOFGn)!%w1m}9rqJC%K9CQI&!4SB=8PBU+*e7 zZi}1YZ;Sb{I{L_iw(lJswzPQt<{(G--ht}h54_HxzU+%$h&&T`cqTqJs_+4l1F2WD1f6E_ppC9anTImMg8 zU;K6;&*OfqC*QO*bm5)HT{(=>%lQ&IQ&YDvrNzE?;GQ-Ev3KgLi57(V4d5hjkDuN% zdDWZX`wuGiN1Kbn)I$v$di_v+%&zE<+BLn}&?_DYAb Date: Tue, 2 Dec 2025 09:13:10 +0100 Subject: [PATCH 21/38] Checked task.py and status_manager.py -'skipped' logic already implemented --- maestro.db | Bin 45056 -> 0 bytes 1 file changed, 0 insertions(+), 0 deletions(-) delete mode 100644 maestro.db diff --git a/maestro.db b/maestro.db deleted file mode 100644 index e8dcc8272395834eb43ec64d0ab805cc6e4c1041..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 45056 zcmeI&+iv4T7{GCRw@n&%OUp%NfrK=@B9_`k(Ow~hz-`@9ktU1lqFr%ed7Ei#$cfll z^$OH*kdSyE-T^Ln4IY6@X52Q3y^U6j#Etn!vg3>oGvDv~CI>kuUo_)Xiu1v!7p3Bf zv0<2|@u?7oVN}$zs+PNle!X<}LOq*v+lw|U#^aa2tgZiTShb&x_4n5QT>Im#zaIUz zy87r>>mBQN)yM+@1Q0*~0R#~EUkY4rSXH~dV_rAnen(!#S79_v&MswV+?Av7>%og8 z?8~b(9F6;78uw&z@9LiKxq&B+TKnEJ@!HcxyCsU35cOOS&h_J!s=c#gPS(@tMfa5+ zi>KFaM-aHXP4AwITeY|&zI6S)&s@LW_^{!KRyz=_(`Iv9ShZZ&;-J}f1Kq3w50f-f z-%od&?cLlNPyM4~*MBZP_nwP-+;J@DlUl_-*)fewF!{EtzSf4(I2~lKU-fJhBd{jl zRjRgan?HS(MVv;-H>z{Bs^((LnnZm*pxnqh(M33mHMdJ%$+K}95BiG^(7}$AJ10$` zT1KhtgzC%rzRGZLbnNNovq5jzmG|sOFGn)!%w1m}9rqJC%K9CQI&!4SB=8PBU+*e7 zZi}1YZ;Sb{I{L_iw(lJswzPQt<{(G--ht}h54_HxzU+%$h&&T`cqTqJs_+4l1F2WD1f6E_ppC9anTImMg8 zU;K6;&*OfqC*QO*bm5)HT{(=>%lQ&IQ&YDvrNzE?;GQ-Ev3KgLi57(V4d5hjkDuN% zdDWZX`wuGiN1Kbn)I$v$di_v+%&zE<+BLn}&?_DYAb Date: Tue, 2 Dec 2025 10:19:42 +0100 Subject: [PATCH 22/38] Added output column to TaskORM for storing task results --- src/maestro/server/internals/models.py | 1 + 1 file changed, 1 insertion(+) diff --git a/src/maestro/server/internals/models.py b/src/maestro/server/internals/models.py index a3588ce..97d01bb 100644 --- a/src/maestro/server/internals/models.py +++ b/src/maestro/server/internals/models.py @@ -39,6 +39,7 @@ class TaskORM(Base): started_at = Column(DateTime) completed_at = Column(DateTime) thread_id = Column(String) + output = Column(Text) insertion_order = Column(Integer) execution = relationship("ExecutionORM", back_populates="tasks") From 13e36ce8e7841ced8f3a41ed9756a1ebe1e2f207 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 2 Dec 2025 10:25:15 +0100 Subject: [PATCH 23/38] Added set_task_output and get_task_output methods to status_manager.py --- .../server/internals/status_manager.py | 54 +++++++++++++++++++ 1 file changed, 54 insertions(+) diff --git a/src/maestro/server/internals/status_manager.py b/src/maestro/server/internals/status_manager.py index c865698..663b948 100644 --- a/src/maestro/server/internals/status_manager.py +++ b/src/maestro/server/internals/status_manager.py @@ -175,6 +175,36 @@ def set_task_status( elif status in ["completed", "failed", "cancelled", "skipped"]: task.completed_at = datetime.now() + # ---------------------------------------------------------------------- + # Metodo: set_task_output + def set_task_output( + self, + dag_id: str, + task_id: str, + execution_id: str, + output: Any + ): + """Store JSON-serializable output for a task.""" + with self.Session.begin() as session: + task = ( + session.query(TaskORM) + .filter_by(dag_id=dag_id, id=task_id, execution_id=execution_id) + .first() + ) + + if not task: + # Should never happen if initialization is correct, + # but we allow automatic creation for safety. + task = TaskORM( + dag_id=dag_id, + id=task_id, + execution_id=execution_id, + status="pending", + ) + session.add(task) + + task.output = json.dumps(output) + # ---------------------------------------------------------------------- # Metodo: get_task_status def get_task_status( @@ -211,6 +241,30 @@ def get_task_status( return task.status if task else None + # ---------------------------------------------------------------------- + # Metodo: get_task_output + def get_task_output( + self, + dag_id: str, + task_id: str, + execution_id: str + ) -> Any: + """Return the decoded output of a task, or None if not found.""" + with self.Session() as session: + task = ( + session.query(TaskORM) + .filter_by(dag_id=dag_id, id=task_id, execution_id=execution_id) + .first() + ) + + if not task or task.output is None: + return None + + try: + return json.loads(task.output) + except json.JSONDecodeError: + return task.output # fallback: raw string + # ---------------------------------------------------------------------- # Method: get_dag_status def get_dag_status(self, dag_id: str, execution_id: str = None) -> Dict[str, str]: From 91f8d5b2a441ab054b6538566177276c9ca081e2 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 2 Dec 2025 10:33:19 +0100 Subject: [PATCH 24/38] Added structured output support with __output__, result, and output variables to python_task.py --- src/maestro/server/tasks/python_task.py | 30 +++++++++++++++++++++++-- 1 file changed, 28 insertions(+), 2 deletions(-) diff --git a/src/maestro/server/tasks/python_task.py b/src/maestro/server/tasks/python_task.py index 04e10c1..0faf8b2 100644 --- a/src/maestro/server/tasks/python_task.py +++ b/src/maestro/server/tasks/python_task.py @@ -73,11 +73,37 @@ def task_print(*args, **kwargs): exec(code, env) else: raise ValueError("PythonTask requires either 'code' or 'script_path'.") + + # Output catch + + output = None + + # Priorità 1: __output__ + if "__output__" in env: + output = env["__output__"] + + # Priorità 2: result (pattern: result = main()) + elif "result" in env: + output = env["result"] + + # Priorità 3: output (fallback) + elif "output" in env: + output = env["output"] + + # Solo se c'è un output significativo, salvalo nel DB + if output is not None: + sm.set_task_output( + dag_id=dag_id, + task_id=task_id, + execution_id=execution_id, + output=output + ) + except Exception: - # logga l'eccezione nel DB + # log eccezione... tb = traceback.format_exc() for line in tb.splitlines(): if line.strip(): _log_line("ERROR", f"[PythonTask][exception] {line}") - # rialza per far fallire il task raise + From 41a0851c39b93c60b0569e4cb7ae430f081957b8 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 2 Dec 2025 10:44:12 +0100 Subject: [PATCH 25/38] Added parsing and support for task-level 'condition' in YAML, patching task.py and dag_loader.py --- src/maestro/server/internals/dag_loader.py | 9 ++++++++- src/maestro/shared/task.py | 1 + 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/src/maestro/server/internals/dag_loader.py b/src/maestro/server/internals/dag_loader.py index 3c7a7ae..688d000 100644 --- a/src/maestro/server/internals/dag_loader.py +++ b/src/maestro/server/internals/dag_loader.py @@ -88,10 +88,17 @@ def _create_task_from_config(self, task_config: Dict[str, Any], dag_file_path: s if not task_class: raise ValueError(f"Unknown task type: {task_type_name}") - # Merge params into main config (for backward compatibility) + # Merge params + preserve condition + params = task_config.pop("params", {}) + condition = task_config.get("condition") # keep condition if present + task_config.update(params) + # Reinserisci condition se c'era + if condition is not None: + task_config["condition"] = condition + # Add DAG file path task_config["dag_file_path"] = dag_file_path diff --git a/src/maestro/shared/task.py b/src/maestro/shared/task.py index 883bbab..8e12ba5 100644 --- a/src/maestro/shared/task.py +++ b/src/maestro/shared/task.py @@ -29,6 +29,7 @@ class Task(BaseModel, ABC): executor: str = "local" # New field for executor type on_success: Optional[Callable] = None # Callback for when the task completes successfully on_failure: Optional[Callable] = None # Callback for when the task fails + condition: Optional[str] = None class Config: arbitrary_types_allowed = True From 1cf7240ef76e99a39dc432921d3d93aad2947e0c Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 2 Dec 2025 11:05:35 +0100 Subject: [PATCH 26/38] Patched orchestrator.py --- src/maestro/server/internals/orchestrator.py | 67 ++++++++++++++++++-- 1 file changed, 62 insertions(+), 5 deletions(-) diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index ed4dfd8..463fcbe 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -467,15 +467,44 @@ def _find_ready_tasks( failed_tasks: Set[str], skipped_tasks: Set[str] ) -> List[str]: - """Find tasks that are ready to run (all dependencies completed).""" + """Find tasks that are ready to run (all dependencies completed AND condition == True).""" ready_tasks = [] + dag_id = dag.dag_id + execution_id = dag.execution_id + sm = StatusManager.get_instance() - for task_id in pending_tasks: + # We must skip tasks whose condition is False + to_skip = [] + + for task_id in list(pending_tasks): task = dag.tasks[task_id] - # Check if all dependencies are completed - if all(dep in completed_tasks for dep in task.dependencies): - ready_tasks.append(task_id) + # Check dependencies completion + if not all(dep in completed_tasks for dep in task.dependencies): + continue + + # Check condition (if present) + condition = getattr(task, "condition", None) + if condition: + if not self._evaluate_condition(task, dag_id, execution_id): + # Mark SKIPPED + task.status = TaskStatus.SKIPPED + sm.set_task_status(dag_id, task_id, "skipped", execution_id) + skipped_tasks.add(task_id) + to_skip.append(task_id) + + self.logger.info( + f"Skipping task {task_id} due to condition: {condition}" + ) + continue + + # If we reach here → task is ready + ready_tasks.append(task_id) + + # Remove skipped tasks from pending set + for tid in to_skip: + if tid in pending_tasks: + pending_tasks.remove(tid) return ready_tasks @@ -491,6 +520,34 @@ def _has_failed_dependencies( for dep in task.dependencies ) + def _evaluate_condition(self, task, dag_id: str, execution_id: str) -> bool: + """Evaluate the boolean condition of a task using output_of().""" + condition = getattr(task, "condition", None) + if not condition: + return True # No condition → task is allowed + + sm = StatusManager.get_instance() + + # Helper to retrieve outputs from PythonTask + def output_of(tid: str): + result_json = sm.get_task_output(dag_id, tid, execution_id) + return result_json # Can be dict, list, int, str depending on task return + + try: + # Safe environment + safe_globals = { + "__builtins__": {}, + "output_of": output_of, + } + safe_locals = {} + + return bool(eval(condition, safe_globals, safe_locals)) + + except Exception as e: + self.logger.error( + f"Condition evaluation failed for task {task.task_id}: {e}" + ) + return False def _execute_task_async( self, From f04029b511b8574c88c66235a7eb1bb8613a8740 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 2 Dec 2025 18:01:16 +0100 Subject: [PATCH 27/38] Intermediate commit --- .../4.1.1.Conditional_branching.yaml | 31 +++++++++---------- src/maestro/server/internals/models.py | 2 +- src/maestro/server/internals/orchestrator.py | 7 +++++ 3 files changed, 22 insertions(+), 18 deletions(-) diff --git a/examples/2_New_examples/4.1.1.Conditional_branching.yaml b/examples/2_New_examples/4.1.1.Conditional_branching.yaml index fe747bb..6d53455 100644 --- a/examples/2_New_examples/4.1.1.Conditional_branching.yaml +++ b/examples/2_New_examples/4.1.1.Conditional_branching.yaml @@ -1,24 +1,21 @@ dag: - name: "conditional_branching" tasks: - - - task_id: "check_condition" + - task_id: check_condition type: "PythonTask" - params: - code: | - import random - result = "branch_a" if random.random() > 0.5 else "branch_b" - print(f"Next step: {result}") - dependencies: [] + code: | + import random + result = "branch_a" if random.random() > 0.5 else "branch_b" + print(f"Next step: {result}") + task_output = {"branch": result} - - task_id: "branch_a" + - task_id: branch_a type: "PrintTask" - params: - message: "Running branch A" - dependencies: ["check_condition"] + message: "Eseguo branch A" + dependencies: [check_condition] + condition: "output_of('check_condition')['branch'] == 'branch_a'" - - task_id: "branch_b" + - task_id: branch_b type: "PrintTask" - params: - message: "Running branch B" - dependencies: ["check_condition"] + message: "Eseguo branch B" + dependencies: [check_condition] + condition: "output_of('check_condition')['branch'] == 'branch_b'" diff --git a/src/maestro/server/internals/models.py b/src/maestro/server/internals/models.py index 97d01bb..f2325cd 100644 --- a/src/maestro/server/internals/models.py +++ b/src/maestro/server/internals/models.py @@ -39,7 +39,7 @@ class TaskORM(Base): started_at = Column(DateTime) completed_at = Column(DateTime) thread_id = Column(String) - output = Column(Text) + output = Column(Text, nullable=True) insertion_order = Column(Integer) execution = relationship("ExecutionORM", back_populates="tasks") diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index 463fcbe..3faa807 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -549,6 +549,13 @@ def output_of(tid: str): ) return False + print("[DEBUG] condition eval:", + task.task_id, + expr, + "output_of(check_condition)=", + safe_evaluator_globals["output_of"]("check_condition"), + flush=True) + def _execute_task_async( self, task, From 3265b303736e8bcc481f06dbe24a93e157e4d67e Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 2 Dec 2025 22:09:07 +0100 Subject: [PATCH 28/38] Intermediate passage 2 --- .../4.1.1.Conditional_branching.yaml | 8 ++- .../4.2.1.conditional_branching_extended.yaml | 11 +++- ...4.2.2.conditional_branching_with_bash.yaml | 8 ++- src/maestro/server/internals/orchestrator.py | 56 ++++++++++++------- src/maestro/server/tasks/bash_task.py | 38 ++++++------- src/maestro/server/tasks/python_task.py | 20 ++++--- 6 files changed, 87 insertions(+), 54 deletions(-) diff --git a/examples/2_New_examples/4.1.1.Conditional_branching.yaml b/examples/2_New_examples/4.1.1.Conditional_branching.yaml index 6d53455..48ecbac 100644 --- a/examples/2_New_examples/4.1.1.Conditional_branching.yaml +++ b/examples/2_New_examples/4.1.1.Conditional_branching.yaml @@ -4,9 +4,11 @@ dag: type: "PythonTask" code: | import random - result = "branch_a" if random.random() > 0.5 else "branch_b" - print(f"Next step: {result}") - task_output = {"branch": result} + random_number = random.random() + print("Random number:", round(random_number, 2)) + branch = "branch_a" if random_number > 0.5 else "branch_b" + print(f"Next step: {branch}") + result = {"branch": branch} - task_id: branch_a type: "PrintTask" diff --git a/examples/2_New_examples/4.2.1.conditional_branching_extended.yaml b/examples/2_New_examples/4.2.1.conditional_branching_extended.yaml index 99921c7..adfe1fd 100644 --- a/examples/2_New_examples/4.2.1.conditional_branching_extended.yaml +++ b/examples/2_New_examples/4.2.1.conditional_branching_extended.yaml @@ -31,10 +31,13 @@ dag: - task_id: "evaluate_condition" type: "PythonTask" params: - script: | + code: | import random - condition = "A" if random.random() > 0.5 else "B" + random_number = round(random.random(), 2) + print("Number:", random_number) + condition = "A" if random_number > 0.5 else "B" print(f"Condition evaluated: {condition}") + task_output = {"branch": condition} dependencies: ["init"] # --------------------------------------------------------- @@ -45,6 +48,7 @@ dag: params: command: "echo 'Branch A: Listing files'; ls -1 | head -n 5" dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == 'A'" # --------------------------------------------------------- # 4) Branch B: prints some process info. @@ -52,8 +56,9 @@ dag: - task_id: "branch_b_action" type: "BashTask" params: - command: "echo 'Branch B: Listing processes'; ps aux | head -n 5" + command: "echo 'Branch B: Listing processes'; echo 'Branch B executing!'; sleep 1; echo 'Done'" dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == 'B'" # --------------------------------------------------------- # 5) Final merge: executed only when both branches finish. diff --git a/examples/2_New_examples/4.2.2.conditional_branching_with_bash.yaml b/examples/2_New_examples/4.2.2.conditional_branching_with_bash.yaml index 19b3fb5..ed34837 100644 --- a/examples/2_New_examples/4.2.2.conditional_branching_with_bash.yaml +++ b/examples/2_New_examples/4.2.2.conditional_branching_with_bash.yaml @@ -38,9 +38,11 @@ dag: params: command: | if [ $((RANDOM % 2)) -eq 0 ]; then - echo "Condition result: A"; + echo "Condition result: A" + echo "OUTPUT: A" else - echo "Condition result: B"; + echo "Condition result: B" + echo "OUTPUT: B" fi dependencies: ["init"] @@ -52,6 +54,7 @@ dag: params: command: "echo 'Branch A: Disk usage'; df -h | head -n 5" dependencies: ["bash_condition"] + condition: "output_of('bash_condition') == 'A'" # --------------------------------------------------------- # 4) Branch B: explores directory structure @@ -61,6 +64,7 @@ dag: params: command: "echo 'Branch B: Directory structure'; ls -d */ 2>/dev/null || echo 'No directories'" dependencies: ["bash_condition"] + condition: "output_of('bash_condition') == 'B'" # --------------------------------------------------------- # 5) Merge: both branches must complete before continuing diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index 3faa807..f0d6353 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -459,6 +459,7 @@ def _run_dag_concurrent( if not ready_tasks and running_tasks: threading.Event().wait(0.1) + def _find_ready_tasks( self, dag: DAG, @@ -467,58 +468,75 @@ def _find_ready_tasks( failed_tasks: Set[str], skipped_tasks: Set[str] ) -> List[str]: - """Find tasks that are ready to run (all dependencies completed AND condition == True).""" - ready_tasks = [] + """ + Find tasks that are ready to run. + + Un task è "ready" quando: + - tutte le sue dipendenze sono in stato terminale + (completed, failed o skipped) + - e, se ha una condition, questa viene valutata a True. + + I task la cui condition risulta False vengono marcati come SKIPPED qui. + """ + ready_tasks: List[str] = [] + dag_id = dag.dag_id execution_id = dag.execution_id sm = StatusManager.get_instance() - # We must skip tasks whose condition is False - to_skip = [] + # Stati "terminali": la dipendenza non è più né pending né running + terminal_deps = completed_tasks | failed_tasks | skipped_tasks - for task_id in list(pending_tasks): + # Task da rimuovere da pending perché skippati per condition False + to_skip: List[str] = [] + + for task_id in pending_tasks: task = dag.tasks[task_id] - # Check dependencies completion - if not all(dep in completed_tasks for dep in task.dependencies): + # Dipendenze ancora non "chiuse"? Non è pronto. + if not all(dep in terminal_deps for dep in task.dependencies): continue - # Check condition (if present) + # Se ha una condition, valutiamola. condition = getattr(task, "condition", None) if condition: if not self._evaluate_condition(task, dag_id, execution_id): - # Mark SKIPPED + # Condition False → SKIPPED qui task.status = TaskStatus.SKIPPED - sm.set_task_status(dag_id, task_id, "skipped", execution_id) skipped_tasks.add(task_id) + sm.set_task_status(dag_id, task_id, "skipped", execution_id) to_skip.append(task_id) - self.logger.info( f"Skipping task {task_id} due to condition: {condition}" ) - continue + continue # non è ready - # If we reach here → task is ready + # Se siamo qui: tutte le dipendenze sono terminali + # e la condition (se presente) è True → task pronto ready_tasks.append(task_id) - # Remove skipped tasks from pending set + # Rimuovi dai pending i task skippati per condition False for tid in to_skip: if tid in pending_tasks: pending_tasks.remove(tid) return ready_tasks + def _has_failed_dependencies( self, task, failed_tasks: Set[str], skipped_tasks: Set[str] ) -> bool: - """Check if task has any failed or skipped dependencies.""" - return any( - dep in failed_tasks or dep in skipped_tasks - for dep in task.dependencies - ) + """ + A task is blocked ONLY if one of its dependencies FAILED. + + Skipped dependencies do not block downstream tasks — this enables + branching with a final merge step, where only one branch runs. + """ + return any(dep in failed_tasks for dep in task.dependencies) + def _evaluate_condition(self, task, dag_id: str, execution_id: str) -> bool: """Evaluate the boolean condition of a task using output_of().""" diff --git a/src/maestro/server/tasks/bash_task.py b/src/maestro/server/tasks/bash_task.py index 34d22cd..3988651 100644 --- a/src/maestro/server/tasks/bash_task.py +++ b/src/maestro/server/tasks/bash_task.py @@ -3,24 +3,13 @@ from maestro.server.tasks.base import BaseTask from maestro.server.internals.status_manager import StatusManager - logger = logging.getLogger(__name__) logger.propagate = True class BashTask(BaseTask): - """ - A task that executes a Bash command locally. - Captures stdout and stderr and sends them to the logging system. - """ - command: str - def execute_local(self): - """ - BashTask: esegue il comando e registra stdout/stderr nel DB riga per riga, - con timestamp corretto come PythonTask. - """ sm = StatusManager.get_instance() dag_id = self.dag_id execution_id = self.execution_id @@ -36,7 +25,9 @@ def execute_local(self): bufsize=1 ) - # Leggi STDOUT riga per riga + captured_output = None # <--- nuovo! + + # STDOUT for line in process.stdout: clean = line.rstrip("\n") msg = f"[BashTask][stdout] {clean}" @@ -51,6 +42,16 @@ def execute_local(self): level="INFO" ) + # NEW: detect structured output: OUTPUT: ... + if clean.lstrip().startswith("OUTPUT:"): + value = clean[len("OUTPUT:"):].strip() + captured_output = value + + raw = clean.lstrip() + if raw.startswith("OUTPUT:"): + value = raw.split("OUTPUT:", 1)[1].strip() + captured_output = value + # Leggi STDERR riga per riga for line in process.stderr: clean = line.rstrip("\n") @@ -66,10 +67,8 @@ def execute_local(self): level="ERROR" ) - # Ritorno exit code - + # Attendi la fine del processo process.wait() - rc = process.returncode if rc != 0: @@ -82,11 +81,13 @@ def execute_local(self): message=msg, level="ERROR" ) - - # QUI: solleva un'eccezione → farà scattare il retry raise Exception(f"BashTask failed with exit code {rc}") - # Se arrivo qui, exit code 0 -> success + # NEW — persist captured output (if present) + if captured_output is not None: + sm.set_task_output(dag_id, execution_id, task_id, captured_output) + + # Success log finale success_msg = "[BashTask] Completed successfully (exit code 0)" print(success_msg, flush=True) sm.add_log( @@ -97,7 +98,6 @@ def execute_local(self): level="INFO" ) - def to_dict(self): """Include the command field in serialized form.""" base = super().to_dict() diff --git a/src/maestro/server/tasks/python_task.py b/src/maestro/server/tasks/python_task.py index 0faf8b2..c278228 100644 --- a/src/maestro/server/tasks/python_task.py +++ b/src/maestro/server/tasks/python_task.py @@ -74,19 +74,24 @@ def task_print(*args, **kwargs): else: raise ValueError("PythonTask requires either 'code' or 'script_path'.") - # Output catch - + # ----------------------------- + # 1) Cattura output dal task + # ----------------------------- output = None - # Priorità 1: __output__ - if "__output__" in env: + # Priorità 1: task_output (come stai usando nella DAG) + if "task_output" in env: + output = env["task_output"] + + # Priorità 2: __output__ + elif "__output__" in env: output = env["__output__"] - # Priorità 2: result (pattern: result = main()) + # Priorità 3: result (pattern: result = ...) elif "result" in env: output = env["result"] - # Priorità 3: output (fallback) + # Priorità 4: output (fallback) elif "output" in env: output = env["output"] @@ -96,7 +101,7 @@ def task_print(*args, **kwargs): dag_id=dag_id, task_id=task_id, execution_id=execution_id, - output=output + output=output, ) except Exception: @@ -106,4 +111,3 @@ def task_print(*args, **kwargs): if line.strip(): _log_line("ERROR", f"[PythonTask][exception] {line}") raise - From eb89c9062773057cb2f47c903b846ea650423f3c Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Tue, 2 Dec 2025 22:57:17 +0100 Subject: [PATCH 29/38] Intermediate checkpoint 3 - basic functioning --- ...conditional_branching_parallel_python.yaml | 13 ++++--- ...3.1.conditional_branching_triple_tree.yaml | 18 ++++++---- .../4.3.2.conditional_branching_matrix.yaml | 30 ++++++++-------- ....3.3.conditional_branching_hypergraph.yaml | 21 +++++++----- .../server/internals/status_manager.py | 34 ++++++++++++------- src/maestro/server/tasks/bash_task.py | 15 ++++---- 6 files changed, 76 insertions(+), 55 deletions(-) diff --git a/examples/2_New_examples/4.2.3.conditional_branching_parallel_python.yaml b/examples/2_New_examples/4.2.3.conditional_branching_parallel_python.yaml index b7563a1..90f1890 100644 --- a/examples/2_New_examples/4.2.3.conditional_branching_parallel_python.yaml +++ b/examples/2_New_examples/4.2.3.conditional_branching_parallel_python.yaml @@ -38,10 +38,13 @@ dag: - task_id: "evaluate_condition" type: "PythonTask" params: - script: | + code: | import random - condition = "A" if random.random() > 0.5 else "B" + random_value = round(random.random(), 2) + print("Value:", random_value) + condition = "A" if random_value > 0.5 else "B" print(f"Condition evaluated: {condition}") + task_output = {"branch": condition} dependencies: ["init"] # --------------------------------------------------------- @@ -50,12 +53,13 @@ dag: - task_id: "branch_a_process" type: "PythonTask" params: - script: | + code: | import math nums = [1, 2, 3, 4, 5] squares = [n*n for n in nums] print("Branch A squares:", squares) dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == 'A'" # --------------------------------------------------------- # 4) Branch B — simple text processing. @@ -63,11 +67,12 @@ dag: - task_id: "branch_b_process" type: "PythonTask" params: - script: | + code: | words = ["alpha", "beta", "gamma"] upper = [w.upper() for w in words] print("Branch B uppercase:", upper) dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == 'B'" # --------------------------------------------------------- # 5) Final merge. diff --git a/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml b/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml index 364165a..db32861 100644 --- a/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml +++ b/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml @@ -28,10 +28,11 @@ dag: - task_id: "evaluate_condition" type: "PythonTask" params: - script: | + code: | import random - condition = random.choice(["A", "B", "C"]) - print(f"Condition evaluated: {condition}") + condition = random.choice(["1", "2", "3"]) + print("Condition evaluated:", condition) + task_output = {"branch": condition} dependencies: ["init"] # --------------------------------------------------------- @@ -43,11 +44,12 @@ dag: params: command: "echo 'Branch 1: Listing files'; ls -1 | head -n 10" dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == '1'" - task_id: "branch1_step2" type: "PythonTask" params: - script: | + code: | import os print('Branch 1 file count:', len(os.listdir('.'))) dependencies: ["branch1_step1"] @@ -61,11 +63,12 @@ dag: params: command: "echo 'Branch 2: Checking processes'; ps aux | head -n 5" dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == '2'" - task_id: "branch2_step2" type: "PythonTask" params: - script: | + code: | print('Branch 2: Process summary collected.') dependencies: ["branch2_step1"] @@ -76,15 +79,16 @@ dag: - task_id: "branch3_step1" type: "PythonTask" params: - script: | + code: | words = ['lorem', 'ipsum', 'dolor', 'sit', 'amet'] print('Branch 3 words:', words) dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == '3'" - task_id: "branch3_step2" type: "PythonTask" params: - script: | + code: | words = ['lorem', 'ipsum', 'dolor', 'sit', 'amet'] uppercase = [w.upper() for w in words] print('Branch 3 uppercase:', uppercase) diff --git a/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml b/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml index 99793fe..b0a95a4 100644 --- a/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml +++ b/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml @@ -1,15 +1,11 @@ abstract: | This difficult-level DAG models a "matrix" of conditional-style branching. - From a single Python-based root evaluation, the DAG fans out into four - independent branches. Each branch contains a 3-step mini-pipeline mixing - PythonTask, BashTask, and PrintTask. + From a single Python-based root evaluation, the DAG fans out into four independent branches. Each branch contains a 3-step mini-pipeline mixing PythonTask, BashTask, and PrintTask. - All four branches run in parallel after the root task and must complete - before the final PrintTask is executed. This structure is ideal for: + All four branches run in parallel after the root task and must complete before the final PrintTask is executed. This structure is ideal for: - Stress-testing fan-out and fan-in synchronization - Exercising both Bash and Python executors under concurrent load - - Verifying that logging and task ordering remain understandable in - complex, parallel DAGs. + - Verifying that logging and task ordering remain understandable in complex, parallel DAGs. dag: name: "conditional_branching_matrix" @@ -22,10 +18,12 @@ dag: - task_id: "root" type: "PythonTask" params: - script: | + code: | import random - value = random.random() - print(f"Root evaluated value: {value}") + branches = ['1', '2', '3', '4'] + branch = random.choice(branches) + print(f"Root evaluated value: {branch}") + task_output = {"branch": branch} dependencies: [] # --------------------------------------------------------- @@ -35,10 +33,11 @@ dag: - task_id: "b1_1" type: "PythonTask" params: - script: | + code: | nums = [1, 2, 3] print("Branch 1 - step 1, nums:", nums) dependencies: ["root"] + condition: "output_of('evaluate_condition')['branch'] == '1'" - task_id: "b1_2" type: "BashTask" @@ -61,11 +60,12 @@ dag: params: command: "echo 'Branch 2 - step 1, listing current dir'; ls -1 | head -n 5" dependencies: ["root"] + condition: "output_of('evaluate_condition')['branch'] == '2'" - task_id: "b2_2" type: "PythonTask" params: - script: | + code: | print("Branch 2 - step 2, post-processing Bash output logically.") dependencies: ["b2_1"] @@ -82,10 +82,11 @@ dag: - task_id: "b3_1" type: "PythonTask" params: - script: | + code: | words = ["alpha", "beta", "gamma"] print("Branch 3 - step 1, words:", words) dependencies: ["root"] + condition: "output_of('evaluate_condition')['branch'] == '3'" - task_id: "b3_2" type: "BashTask" @@ -108,11 +109,12 @@ dag: params: command: "echo 'Branch 4 - step 1, checking date'; date" dependencies: ["root"] + condition: "output_of('evaluate_condition')['branch'] == '4'" - task_id: "b4_2" type: "PythonTask" params: - script: | + code: | print("Branch 4 - step 2, Python confirming time was printed above.") dependencies: ["b4_1"] diff --git a/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml b/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml index 9b56fb0..6bef54f 100644 --- a/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml +++ b/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml @@ -37,10 +37,12 @@ dag: - task_id: "evaluate_condition" type: "PythonTask" params: - script: | + code: | import random - cond = random.choice(["X", "Y", "Z"]) + branches = ["A", "B", "C"] + cond = random.choice(branches) print(f"Condition evaluated: {cond}") + task_output = {"branch": cond} dependencies: ["init"] # ========================================================= @@ -51,9 +53,10 @@ dag: - task_id: "A1" type: "PythonTask" params: - script: | + code: | print("Branch A - Step 1") dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == 'A'" - task_id: "A2" type: "BashTask" @@ -67,11 +70,12 @@ dag: params: command: "echo 'Branch B - Step 1'" dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == 'B'" - task_id: "B2" type: "PythonTask" params: - script: | + code: | print("Branch B - Step 2") dependencies: ["B1"] @@ -79,9 +83,10 @@ dag: - task_id: "C1" type: "PythonTask" params: - script: | + code: | print("Branch C - Step 1") dependencies: ["evaluate_condition"] + condition: "output_of('evaluate_condition')['branch'] == 'C'" - task_id: "C2" type: "BashTask" @@ -119,21 +124,21 @@ dag: - task_id: "T1" type: "PythonTask" params: - script: | + code: | print("T1: Post-merge computation A+B") dependencies: ["merge_AB"] - task_id: "T2" type: "PythonTask" params: - script: | + code: | print("T2: Post-merge computation B+C") dependencies: ["merge_BC"] - task_id: "T3" type: "PythonTask" params: - script: | + code: | print("T3: Post-merge computation C+A") dependencies: ["merge_CA"] diff --git a/src/maestro/server/internals/status_manager.py b/src/maestro/server/internals/status_manager.py index 663b948..7976e24 100644 --- a/src/maestro/server/internals/status_manager.py +++ b/src/maestro/server/internals/status_manager.py @@ -182,9 +182,22 @@ def set_task_output( dag_id: str, task_id: str, execution_id: str, - output: Any - ): - """Store JSON-serializable output for a task.""" + output: Any, + ) -> None: + """ + Salva l'output di un task nella colonna `output` della tabella tasks. + + NOTA IMPORTANTE: + - NON crea nuove righe nella tabella tasks. + - Aggiorna SOLO il record esistente identificato da + (dag_id, task_id, execution_id). + """ + # serializza in JSON se è dict/list, altrimenti stringa semplice + if isinstance(output, (dict, list)): + encoded = json.dumps(output) + else: + encoded = str(output) + with self.Session.begin() as session: task = ( session.query(TaskORM) @@ -193,17 +206,12 @@ def set_task_output( ) if not task: - # Should never happen if initialization is correct, - # but we allow automatic creation for safety. - task = TaskORM( - dag_id=dag_id, - id=task_id, - execution_id=execution_id, - status="pending", - ) - session.add(task) + # Non creiamo una nuova riga qui: se non troviamo il task, + # c'è qualcosa di incoerente a monte. + # Volendo, potresti loggare un warning. + return - task.output = json.dumps(output) + task.output = encoded # ---------------------------------------------------------------------- # Metodo: get_task_status diff --git a/src/maestro/server/tasks/bash_task.py b/src/maestro/server/tasks/bash_task.py index 3988651..3a9f693 100644 --- a/src/maestro/server/tasks/bash_task.py +++ b/src/maestro/server/tasks/bash_task.py @@ -25,7 +25,7 @@ def execute_local(self): bufsize=1 ) - captured_output = None # <--- nuovo! + captured_output = None # STDOUT for line in process.stdout: @@ -43,14 +43,11 @@ def execute_local(self): ) # NEW: detect structured output: OUTPUT: ... - if clean.lstrip().startswith("OUTPUT:"): - value = clean[len("OUTPUT:"):].strip() - captured_output = value - raw = clean.lstrip() - if raw.startswith("OUTPUT:"): - value = raw.split("OUTPUT:", 1)[1].strip() - captured_output = value + raw = clean.lstrip() + if raw.startswith("OUTPUT:"): + value = raw.split("OUTPUT:", 1)[1].strip() + captured_output = value # Leggi STDERR riga per riga for line in process.stderr: @@ -85,7 +82,7 @@ def execute_local(self): # NEW — persist captured output (if present) if captured_output is not None: - sm.set_task_output(dag_id, execution_id, task_id, captured_output) + sm.set_task_output(dag_id, task_id, execution_id, captured_output) # Success log finale success_msg = "[BashTask] Completed successfully (exit code 0)" From 1ea4c85c5b4592d00e108f91e50a29decc9e741c Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Fri, 5 Dec 2025 21:51:21 +0100 Subject: [PATCH 30/38] First tests - on main Tasks --- data.json | 1 + maestro.db | Bin 0 -> 45056 bytes src/maestro/server/internals/task_registry.py | 11 +- src/maestro/server/tasks/bash_task.py | 25 +- tests/conftest.py | 599 ------------------ tests/integration/conftest.py | 102 --- .../integration/test_cli_rest_integration.py | 51 -- tests/integration/test_debug_server.py | 101 --- .../test_server_api_integration.py | 310 --------- tests/test_ansible_task.py | 343 ---------- tests/test_api_client.py | 528 --------------- tests/test_base_task.py | 30 + tests/test_bash_task.py | 95 +++ tests/test_cli_attach.py | 399 ------------ tests/test_cli_client.py | 422 ------------ tests/test_cli_integration.py | 520 --------------- tests/test_cron_feature.py | 42 -- tests/test_dag.py | 139 ---- tests/test_dag_id_generation.py | 259 -------- tests/test_db_feature.py | 107 ---- tests/test_enhanced_cli.py | 441 ------------- tests/test_extended_terraform_task.py | 366 ----------- tests/test_multi_executor.py | 286 --------- tests/test_orchestrator_dagloader.py | 266 -------- tests/test_print_task.py | 50 ++ tests/test_python_task.py | 151 +++++ tests/test_server.py | 335 ---------- tests/test_status_manager.py | 350 ---------- 28 files changed, 340 insertions(+), 5989 deletions(-) create mode 100644 data.json create mode 100644 maestro.db delete mode 100644 tests/conftest.py delete mode 100644 tests/integration/conftest.py delete mode 100644 tests/integration/test_cli_rest_integration.py delete mode 100644 tests/integration/test_debug_server.py delete mode 100644 tests/integration/test_server_api_integration.py delete mode 100644 tests/test_ansible_task.py delete mode 100644 tests/test_api_client.py create mode 100644 tests/test_base_task.py create mode 100644 tests/test_bash_task.py delete mode 100644 tests/test_cli_attach.py delete mode 100644 tests/test_cli_client.py delete mode 100644 tests/test_cli_integration.py delete mode 100644 tests/test_cron_feature.py delete mode 100644 tests/test_dag.py delete mode 100644 tests/test_dag_id_generation.py delete mode 100644 tests/test_db_feature.py delete mode 100644 tests/test_enhanced_cli.py delete mode 100644 tests/test_extended_terraform_task.py delete mode 100644 tests/test_multi_executor.py delete mode 100644 tests/test_orchestrator_dagloader.py create mode 100644 tests/test_print_task.py create mode 100644 tests/test_python_task.py delete mode 100644 tests/test_server.py delete mode 100644 tests/test_status_manager.py diff --git a/data.json b/data.json new file mode 100644 index 0000000..4136c6f --- /dev/null +++ b/data.json @@ -0,0 +1 @@ +{"values": [1, 2, 3, 4, 5]} \ No newline at end of file diff --git a/maestro.db b/maestro.db new file mode 100644 index 0000000000000000000000000000000000000000..0deeaf9192592cefd88e5eaea54bd03e545267e9 GIT binary patch literal 45056 zcmeI4UvJyi6~HN5mMt~1QVfES0mE*z1r||B7DZ9AQgoZ*$gVN}NoBcfVHpHNUfWD1 zQX#3>M&@lww}%14zQEr16*gcGeJL;md)R9~LZ9}uhoRU*&m|>UROQBz9VBiKEz0Em zd+zU?OY+=H-R%$7b%&CD(`qRW5xJXOAi({C5RT(+!TuKP{VfCs!Tt&S4-6fjcX*4- zKKWxb_CGEXKH_3Oiv2tKulX-;{4MhT8-I#KBL9Fu+>ihgKmter2_OL^@ckgry&0KJ zY-IvnK{qsdtRKrt+g1;$)@f2peq=thWrH3&vehwUM{m*5*teEz)k?ie);8}~KO!&8 zPPR75Xb~jktAX!k=VlX`OyF6}Q64tGQR8U#a8H7I<=%RAOyQBRu8|KbwdMCKwUkg6 zc(S=wC!0I#>uC}R`>G}@>sysN3xAm%>d%heGYHJ&n+Q;OV)sIO^*Z2tk z^KdA!oe6L*LHltNo?~UD?sH-YAjER0Cd1SCeb8`yZKKM9RkPJ@(lHUv zp+yyKsC<^%>^SX?L+aI!T&%ibQ_B^@v@~jg)zqu2)f!7Vf;CON)}+a>=sZigvQ?|D zt!^>}q`X8tsa031wd&?_b(`h0UB!&1^%XRnV-Vrxi}3txDDht6JcYVykOlYO+_Sf5 zW)tyv;P*4G>YC<(&GwK_aV(^#w_b~V%JJ)si9L{L1ShL=B&bQB(B`n-TGY0c13F;N zdM)U()gGwPSNT}AjVR-#ut)f3Kb)RT+`b+7^tUdPL1XN*qo9w^S@>ku{fdmY4V~%5 z7W$HK#9%0~diy-vIwK(qXU8+$#ei#N&nhlaX5wt0@=^2*|NN^1-PaABb-saVFvHu| zo3yPshr=eMLQfv);pBdGrLwbLC(AoE=xBABT@SYFm5m2uIxk!k7p6jqcjMihgKmter2_OL^aIFda`A#6j|K-kO@jz)d6~mF+imfWm zr%CqkrEAH%B<^c_09)Xdh~js7IiGI#vByG&Q4^FHkA`7h=ugLB39Wpb8JBf(qc^j zuV^)!&3dArrFTx+lqG&(>4sAWw<~!Hr5sR$TJUHrYl@>l^wZ=JYkwr9etpnAy}u#o zRWK@sQg5|Q%OQ_!(?}E3ZWwLGHf7q`1L~+pT8ifz1_MjpBTpO2CrYzJ?FQuB71AW1 zCSsZt(nQ)jZ5T&7I4mX@Q>p=!9VN>OZGa9N$)iS+CyGt>?=}nq8;i+momLyh3LGU( z_MsB5;zn|{r%nw94F-ikJTnGcwe&XYt!{tmS=qoUPMky27?_1>cA#T>0*_jSFYfgk zh2KEm80%=Y6vMp^sxZ_vPz3okIk0G(WK6QKo6D5<79iwEB^mcR(O1Mh#XcOUf%&&K zySn+hWiRygqW4+B8$y2QlIUK{&qmrve~|AYm&Mc5!JXm4H;Q~NFJ**$CMOYL>8_N! zD@hT!%Liir;oyTC5yYv6C#~gfc zLjp(u2_OL^fCP{L5`9Om2T^zqFJSRG}>Sz4iVJ0^0CQe%YJJ&kCYgDCWk+NRqfDibrF{1%Gfb*G@lPg^ zYWVl%u5A4u#_y5k{(_Cf?x-p(4!95NiJXuWEcWTZ54|-?a{M?c#iEca7e=L=3wlyY zS1aYN?P#!8PD-YAzMQK#7 zpH6vlO{B)`IA%ryNB{{S0VIF~kN^@u0!RP}Ac0p!pc|Tw+~IfNb#hq7 zZFn@|e@zsHVj=+~ afCP{L5 0: - break - time.sleep(1) # Wait a bit before retrying - assert len(logs) > 0 - assert any("task1" in log["task_id"] for log in logs) - -""" def test_attach_dag(client: TestClient, sample_dag_file: str): - # Create and run a DAG first - client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_attach"}) - client.post("/v1/dags/test_dag_attach/run", json={"dag_id": "test_dag_attach"}) - - # Stream logs - messages = [] - with client.stream("GET", "/v1/dags/test_dag_attach/attach") as response: - for chunk in response.iter_bytes(): - messages.append(chunk.decode("utf-8")) - if "Task 'task2' completed" in chunk.decode("utf-8"): - break - - assert any("task1" in msg for msg in messages) - assert any("task3" in msg for msg in messages) """ - -def test_rm_dag(client: TestClient, sample_dag_file: str): - # Create a DAG - client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_rm"}) - - # Remove the DAG - response = client.delete("/v1/dags/test_dag_rm") - assert response.status_code == 200 - assert "removed successfully" in response.json()["message"] - - # Verify it's gone - response = client.get("/v1/dags?filter=all") - assert response.status_code == 200 - assert not any(d["dag_id"] == "test_dag_rm" for d in response.json()) - -def test_rm_running_dag_force(client: TestClient, sample_dag_file: str): - # Create and run a DAG - client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_rm_force"}) - client.post("/v1/dags/test_dag_rm_force/run", json={"dag_id": "test_dag_rm_force"}) - time.sleep(1) # Give it a moment to start running - - # Try to remove without force (should fail) - response = client.delete("/v1/dags/test_dag_rm_force") - assert response.status_code == 409 # Conflict - assert "currently running" in response.json()["detail"] - - # Remove with force - response = client.delete("/v1/dags/test_dag_rm_force?force=true") - assert response.status_code == 200 - assert "removed successfully" in response.json()["message"] - - # Verify it's gone - response = client.get("/v1/dags?filter=all") - assert response.status_code == 200 - assert not any(d["dag_id"] == "test_dag_rm_force" for d in response.json()) - -def test_ls_dags(client: TestClient, sample_dag_file: str): - # Create a few DAGs for testing ls - client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "ls_dag_1"}) - client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "ls_dag_2"}) - client.post("/v1/dags/ls_dag_1/run", json={"dag_id": "ls_dag_1"}) - time.sleep(1) # Give it a moment to start running - - # Test ls all - response = client.get("/v1/dags?filter=all") - assert response.status_code == 200 - dags = response.json() - - assert any(d["dag_id"] == "ls_dag_1" for d in dags) - assert any(d["dag_id"] == "ls_dag_2" for d in dags) - - # Test ls active - response = client.get("/v1/dags?filter=active") - assert response.status_code == 200 - dags = response.json() - print(f"\n[DEBUG] Active DAGs response: {dags}") - assert any(d["dag_id"] == "ls_dag_1" for d in dags) - assert not any(d["dag_id"] == "ls_dag_2" for d in dags) # ls_dag_2 is not running - - # Test ls terminated (after ls_dag_1 completes) - time.sleep(5) # Wait for ls_dag_1 to complete - response = client.get("/v1/dags?filter=terminated") - assert response.status_code == 200 - dags = response.json() - print(f"\n[DEBUG] Terminated DAGs response: {dags}") - assert any(d["dag_id"] == "ls_dag_1" and d["completed_at"] is not None for d in dags) - assert not any(d["dag_id"] == "ls_dag_2" for d in dags) # ls_dag_2 was never run - -def test_stop_dag(client: TestClient, sample_dag_file: str): - # Create and run a DAG - client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_stop"}) - run_response = client.post("/v1/dags/test_dag_stop/run", json={"dag_id": "test_dag_stop"}) - execution_id = run_response.json()["execution_id"] - time.sleep(1) # Give it a moment to start running - - # Stop the DAG - response = client.post(f"/v1/dags/test_dag_stop/stop?execution_id={execution_id}") - assert response.status_code == 200 - assert "Successfully stopped" in response.json()["message"] - - # Verify status is cancelled or completed (if it finished before cancellation took effect) - status_response = client.get(f"/dags/test_dag_stop/status?execution_id={execution_id}") - assert status_response.status_code == 200 - # The DAG might complete before cancellation takes effect in test environment - assert status_response.json()["status"] in ["cancelled", "completed"] - -def test_resume_dag(client: TestClient, sample_dag_file: str): - # Create and run a DAG, then stop it - client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_resume"}) - run_response = client.post("/v1/dags/test_dag_resume/run", json={"dag_id": "test_dag_resume"}) - execution_id = run_response.json()["execution_id"] - time.sleep(1) # Give it a moment to start running - client.post(f"/v1/dags/test_dag_resume/stop?execution_id={execution_id}") - time.sleep(1) # Give it a moment to stop - - # Resume the DAG - response = client.post(f"/v1/dags/test_dag_resume/resume?execution_id={execution_id}") - assert response.status_code == 200 - assert "resumed with new execution ID" in response.json()["message"] - - # Verify a new execution is running - time.sleep(5) # Wait for it to complete - status_response = client.get(f"/dags/test_dag_resume/status") - assert status_response.status_code == 200 - assert status_response.json()["status"] == "completed" diff --git a/tests/test_ansible_task.py b/tests/test_ansible_task.py deleted file mode 100644 index 9a88715..0000000 --- a/tests/test_ansible_task.py +++ /dev/null @@ -1,343 +0,0 @@ -import pytest -import os -import tempfile -from pathlib import Path -from unittest.mock import patch, MagicMock, call -from maestro.server.tasks.ansible_task import AnsibleTask - - -class TestAnsibleTask: - """Test suite for AnsibleTask.""" - - def test_ansible_task_creation(self): - """Test basic AnsibleTask creation.""" - task = AnsibleTask( - task_id='test_ansible_task', - playbook='test.yml', - inventory='inventory.ini' - ) - - assert task.task_id == 'test_ansible_task' - assert task.playbook == 'test.yml' - assert task.inventory == 'inventory.ini' - assert task.private_data_dir == './' - assert task.verbosity == 1 - assert task.extra_vars == {} - assert task.become_user is None - - def test_ansible_task_with_all_fields(self): - """Test AnsibleTask creation with all fields.""" - task = AnsibleTask( - task_id='test_ansible_task', - playbook='test.yml', - inventory='inventory.ini', - private_data_dir='/tmp/ansible', - verbosity=2, - extra_vars={'env': 'prod', 'version': '1.0'}, - become_user='root' - ) - - assert task.task_id == 'test_ansible_task' - assert task.playbook == 'test.yml' - assert task.inventory == 'inventory.ini' - assert task.private_data_dir == '/tmp/ansible' - assert task.verbosity == 2 - assert task.extra_vars == {'env': 'prod', 'version': '1.0'} - assert task.become_user == 'root' - - def test_get_absolute_paths_with_dag_file(self): - """Test get_absolute_paths when dag_file_path is provided.""" - with tempfile.TemporaryDirectory() as temp_dir: - dag_file = Path(temp_dir) / 'test.yml' - dag_file.touch() - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini', - private_data_dir='./ansible', - dag_file_path=str(dag_file) - ) - - playbook_path, inventory_path, data_dir = task.get_absolute_paths() - - assert playbook_path == str(Path(temp_dir) / 'playbook.yml') - assert inventory_path == str(Path(temp_dir) / 'inventory.ini') - assert data_dir == str(Path(temp_dir) / 'ansible') - - def test_get_absolute_paths_without_dag_file(self): - """Test get_absolute_paths when dag_file_path is not provided.""" - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini', - private_data_dir='./ansible' - ) - - playbook_path, inventory_path, data_dir = task.get_absolute_paths() - - # Should resolve relative to current directory - assert playbook_path == str(Path('playbook.yml').resolve()) - assert inventory_path == str(Path('inventory.ini').resolve()) - assert data_dir == str(Path('./ansible').resolve()) - - def test_get_absolute_paths_with_absolute_paths(self): - """Test get_absolute_paths when paths are already absolute.""" - with tempfile.TemporaryDirectory() as temp_dir: - playbook_abs = Path(temp_dir) / 'playbook.yml' - inventory_abs = Path(temp_dir) / 'inventory.ini' - data_dir_abs = Path(temp_dir) / 'ansible' - - task = AnsibleTask( - task_id='test_ansible_task', - playbook=str(playbook_abs), - inventory=str(inventory_abs), - private_data_dir=str(data_dir_abs), - dag_file_path='/some/other/path/dag.yml' - ) - - playbook_path, inventory_path, data_dir = task.get_absolute_paths() - - # Should use absolute paths as-is - assert playbook_path == str(playbook_abs) - assert inventory_path == str(inventory_abs) - assert data_dir == str(data_dir_abs) - - @patch('maestro.server.tasks.ansible_task.ansible_runner.run') - @patch('maestro.server.tasks.ansible_task.os.path.exists') - @patch('maestro.server.tasks.ansible_task.get_console') - def test_execute_local_success(self, mock_get_console, mock_exists, mock_ansible_run): - """Test successful execution of ansible task.""" - mock_console = MagicMock() - mock_get_console.return_value = mock_console - mock_exists.return_value = True - - # Mock successful ansible run - mock_result = MagicMock() - mock_result.status = 'successful' - mock_ansible_run.return_value = mock_result - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini' - ) - - task.execute_local() - - # Verify console output - mock_console.print.assert_any_call("[AnsibleTask] Executing 'test_ansible_task'") - mock_console.print.assert_any_call("[AnsibleTask] Task 'test_ansible_task' completed successfully.") - - # Verify ansible_runner was called correctly - mock_ansible_run.assert_called_once() - - @patch('maestro.server.tasks.ansible_task.ansible_runner.run') - @patch('maestro.server.tasks.ansible_task.os.path.exists') - @patch('maestro.server.tasks.ansible_task.get_console') - def test_execute_local_failure(self, mock_get_console, mock_exists, mock_ansible_run): - """Test failed execution of ansible task.""" - mock_console = MagicMock() - mock_get_console.return_value = mock_console - mock_exists.return_value = True - - # Mock failed ansible run - mock_result = MagicMock() - mock_result.status = 'failed' - mock_ansible_run.return_value = mock_result - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini' - ) - - with pytest.raises(Exception) as exc_info: - task.execute_local() - - assert "failed with status: failed" in str(exc_info.value) - - # Verify console output - mock_console.print.assert_any_call("[AnsibleTask] Executing 'test_ansible_task'") - mock_console.print.assert_any_call( - "[AnsibleTask] Task 'test_ansible_task' failed with status: failed", - style="red" - ) - - @patch('maestro.server.tasks.ansible_task.os.path.exists') - @patch('maestro.server.tasks.ansible_task.get_console') - def test_execute_local_playbook_not_found(self, mock_get_console, mock_exists): - """Test execution when playbook file doesn't exist.""" - mock_console = MagicMock() - mock_get_console.return_value = mock_console - - # Mock playbook not existing - def mock_exists_func(path): - return 'playbook.yml' not in path - - mock_exists.side_effect = mock_exists_func - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini' - ) - - with pytest.raises(FileNotFoundError) as exc_info: - task.execute_local() - - assert "Playbook not found" in str(exc_info.value) - - @patch('maestro.server.tasks.ansible_task.os.path.exists') - @patch('maestro.server.tasks.ansible_task.get_console') - def test_execute_local_inventory_not_found(self, mock_get_console, mock_exists): - """Test execution when inventory file doesn't exist.""" - mock_console = MagicMock() - mock_get_console.return_value = mock_console - - # Mock inventory not existing - def mock_exists_func(path): - return 'inventory.ini' not in path - - mock_exists.side_effect = mock_exists_func - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini' - ) - - with pytest.raises(FileNotFoundError) as exc_info: - task.execute_local() - - assert "Inventory not found" in str(exc_info.value) - - @patch('maestro.server.tasks.ansible_task.ansible_runner.run') - @patch('maestro.server.tasks.ansible_task.os.path.exists') - @patch('maestro.server.tasks.ansible_task.get_console') - def test_execute_local_with_extra_vars(self, mock_get_console, mock_exists, mock_ansible_run): - """Test execution with extra variables.""" - mock_console = MagicMock() - mock_get_console.return_value = mock_console - mock_exists.return_value = True - - # Mock successful ansible run - mock_result = MagicMock() - mock_result.status = 'successful' - mock_ansible_run.return_value = mock_result - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini', - extra_vars={'env': 'test', 'debug': True}, - verbosity=2 - ) - - task.execute_local() - - # Verify ansible_runner was called with correct parameters - mock_ansible_run.assert_called_once() - call_args = mock_ansible_run.call_args - assert call_args[1]['verbosity'] == 2 - assert call_args[1]['quiet'] is False - - @patch('maestro.server.tasks.ansible_task.ansible_runner.run') - @patch('maestro.server.tasks.ansible_task.os.path.exists') - @patch('maestro.server.tasks.ansible_task.get_console') - def test_execute_local_with_become_user(self, mock_get_console, mock_exists, mock_ansible_run): - """Test execution with become_user option.""" - mock_console = MagicMock() - mock_get_console.return_value = mock_console - mock_exists.return_value = True - - # Mock successful ansible run - mock_result = MagicMock() - mock_result.status = 'successful' - mock_ansible_run.return_value = mock_result - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini', - become_user='root' - ) - - task.execute_local() - - # Verify ansible_runner was called - mock_ansible_run.assert_called_once() - - @patch('maestro.server.tasks.ansible_task.ansible_runner.run') - @patch('maestro.server.tasks.ansible_task.os.path.exists') - @patch('maestro.server.tasks.ansible_task.get_console') - def test_execute_local_console_output(self, mock_get_console, mock_exists, mock_ansible_run): - """Test that console output shows correct paths.""" - mock_console = MagicMock() - mock_get_console.return_value = mock_console - mock_exists.return_value = True - - # Mock successful ansible run - mock_result = MagicMock() - mock_result.status = 'successful' - mock_ansible_run.return_value = mock_result - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini' - ) - - task.execute_local() - - # Verify console output includes path information - assert mock_console.print.call_count >= 4 - mock_console.print.assert_any_call("[AnsibleTask] Executing 'test_ansible_task'") - - # Check that playbook and inventory paths are printed - calls = mock_console.print.call_args_list - playbook_call = any("[AnsibleTask] Using playbook:" in str(call) for call in calls) - inventory_call = any("[AnsibleTask] Using inventory:" in str(call) for call in calls) - - assert playbook_call, "Playbook path should be printed" - assert inventory_call, "Inventory path should be printed" - - def test_ansible_task_field_defaults(self): - """Test that default field values are set correctly.""" - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbook.yml', - inventory='inventory.ini' - ) - - # Test default values - assert task.private_data_dir == './' - assert task.verbosity == 1 - assert task.extra_vars == {} - assert task.become_user is None - - def test_ansible_task_with_complex_paths(self): - """Test ansible task with complex relative and absolute paths.""" - with tempfile.TemporaryDirectory() as temp_dir: - dag_file = Path(temp_dir) / 'dag.yml' - dag_file.touch() - - # Create subdirectories - playbook_dir = Path(temp_dir) / 'playbooks' - playbook_dir.mkdir() - inventory_dir = Path(temp_dir) / 'inventories' - inventory_dir.mkdir() - - task = AnsibleTask( - task_id='test_ansible_task', - playbook='playbooks/site.yml', - inventory='inventories/prod.ini', - private_data_dir='./data', - dag_file_path=str(dag_file) - ) - - playbook_path, inventory_path, data_dir = task.get_absolute_paths() - - assert playbook_path == str(Path(temp_dir) / 'playbooks' / 'site.yml') - assert inventory_path == str(Path(temp_dir) / 'inventories' / 'prod.ini') - assert data_dir == str(Path(temp_dir) / 'data') diff --git a/tests/test_api_client.py b/tests/test_api_client.py deleted file mode 100644 index 387813f..0000000 --- a/tests/test_api_client.py +++ /dev/null @@ -1,528 +0,0 @@ -#!/usr/bin/env python3 -""" -Comprehensive test suite for the Maestro API client. - -This test suite ensures high coverage of all API client methods and error scenarios, -using the responses library to mock HTTP calls. -""" - -import pytest -import responses -import requests -import json -from unittest.mock import patch, Mock -from maestro.client.api_client import MaestroAPIClient - - -class TestMaestroAPIClient: - """Test suite for MaestroAPIClient.""" - - def setup_method(self): - """Set up test fixtures.""" - self.client = MaestroAPIClient(base_url="http://localhost:8000") - self.mock_dag_response = { - "dag_id": "test-dag", - "execution_id": "exec-123", - "status": "submitted", - "submitted_at": "2025-07-18T18:33:35Z" - } - - @pytest.fixture - def mock_responses(self): - """Mock HTTP responses.""" - with responses.RequestsMock() as rsps: - yield rsps - - # Constructor Tests - @pytest.mark.unit - def test_client_initialization(self): - """Test client initialization with default and custom values.""" - # Default initialization - client = MaestroAPIClient() - assert client.base_url == "http://localhost:8000" - assert client.timeout == 30 - - # Custom initialization - client = MaestroAPIClient(base_url="http://custom:9000", timeout=60) - assert client.base_url == "http://custom:9000" - assert client.timeout == 60 - - @pytest.mark.unit - def test_client_initialization_strips_trailing_slash(self): - """Test that trailing slashes are stripped from base_url.""" - client = MaestroAPIClient(base_url="http://localhost:8000/") - assert client.base_url == "http://localhost:8000" - - # Health Check Tests - @pytest.mark.unit - def test_health_check_success(self, mock_responses): - """Test successful health check.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/", - json={"status": "healthy", "timestamp": "2025-07-18T18:33:35Z"}, - status=200 - ) - - result = self.client.health_check() - - assert result["status"] == "healthy" - assert result["timestamp"] == "2025-07-18T18:33:35Z" - assert len(mock_responses.calls) == 1 - - @pytest.mark.unit - def test_health_check_connection_error(self, mock_responses): - """Test health check with connection error.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/", - body=requests.exceptions.ConnectionError("Connection failed") - ) - - with pytest.raises(ConnectionError) as exc_info: - self.client.health_check() - - assert "Could not connect to Maestro server" in str(exc_info.value) - - # Get DAG Status Tests - @pytest.mark.unit - def test_get_dag_status_success(self, mock_responses): - """Test successful DAG status retrieval.""" - status_response = { - "execution_id": "exec-123", - "status": "running", - "started_at": "2025-07-18T18:33:35Z", - "completed_at": None, - "thread_id": "thread-1", - "tasks": [] - } - - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/dags/test-dag/status", - json=status_response, - status=200 - ) - - result = self.client.get_dag_status("test-dag") - - assert result == status_response - assert len(mock_responses.calls) == 1 - - @pytest.mark.unit - def test_get_dag_status_with_execution_id(self, mock_responses): - """Test DAG status retrieval with execution ID.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/dags/test-dag/status", - json={"execution_id": "exec-specific"}, - status=200 - ) - - result = self.client.get_dag_status("test-dag", execution_id="exec-specific") - - # Check that execution_id parameter was passed - assert mock_responses.calls[0].request.url.endswith("?execution_id=exec-specific") - - @pytest.mark.unit - def test_get_dag_status_not_found(self, mock_responses): - """Test DAG status retrieval for non-existent DAG.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/dags/nonexistent/status", - status=404 - ) - - with pytest.raises(FileNotFoundError) as exc_info: - self.client.get_dag_status("nonexistent") - - assert "Resource not found" in str(exc_info.value) - - # Get DAG Logs Tests - @pytest.mark.unit - def test_get_dag_logs_success(self, mock_responses): - """Test successful DAG logs retrieval.""" - logs_response = { - "logs": [ - { - "timestamp": "2025-07-18T18:33:35Z", - "level": "INFO", - "task_id": "task-1", - "message": "Task started" - } - ], - "total_count": 1 - } - - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/logs/test-dag", - json=logs_response, - status=200 - ) - - result = self.client.get_dag_logs_v1("test-dag") - - assert result == logs_response - assert len(mock_responses.calls) == 1 - - @pytest.mark.unit - def test_get_dag_logs_with_filters(self, mock_responses): - """Test DAG logs retrieval with filters.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/logs/test-dag", - json={"logs": [], "total_count": 0}, - status=200 - ) - - result = self.client.get_dag_logs_v1( - "test-dag", - execution_id="exec-123", - limit=50, - task_filter="specific-task", - level_filter="ERROR" - ) - - # Check URL parameters - request_url = mock_responses.calls[0].request.url - assert "limit=50" in request_url - assert "execution_id=exec-123" in request_url - assert "task_filter=specific-task" in request_url - assert "level_filter=ERROR" in request_url - - # Stream DAG Logs Tests - @pytest.mark.unit - def test_stream_dag_logs_success(self, mock_responses): - """Test successful DAG logs streaming.""" - # Mock SSE stream response - stream_data = [ - "data: {\"timestamp\": \"2025-07-18T18:33:35Z\", \"level\": \"INFO\", \"task_id\": \"task-1\", \"message\": \"Log 1\"}\n", - "data: {\"timestamp\": \"2025-07-18T18:33:36Z\", \"level\": \"ERROR\", \"task_id\": \"task-2\", \"message\": \"Log 2\"}\n" - ] - - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/dags/test-dag/attach", - body="".join(stream_data), - status=200, - stream=True - ) - - logs = list(self.client.stream_dag_logs_v1("test-dag")) - - assert len(logs) == 2 - assert logs[0]["message"] == "Log 1" - assert logs[1]["message"] == "Log 2" - - @pytest.mark.unit - def test_stream_dag_logs_with_filters(self, mock_responses): - """Test DAG logs streaming with filters.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/dags/test-dag/attach", - body="", - status=200, - stream=True - ) - - list(self.client.stream_dag_logs_v1( - "test-dag", - execution_id="exec-123", - task_filter="specific-task", - level_filter="ERROR" - )) - - # Check URL parameters - request_url = mock_responses.calls[0].request.url - assert "execution_id=exec-123" in request_url - assert "task_filter=specific-task" in request_url - assert "level_filter=ERROR" in request_url - - @pytest.mark.unit - def test_stream_dag_logs_invalid_json(self, mock_responses): - """Test DAG logs streaming with invalid JSON.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/dags/test-dag/attach", - body="data: {invalid json}\n", - status=200, - stream=True - ) - - logs = list(self.client.stream_dag_logs_v1("test-dag")) - - # Should skip invalid JSON lines - assert len(logs) == 0 - - # Get Running DAGs Tests - @pytest.mark.unit - def test_get_running_dags_success(self, mock_responses): - """Test successful running DAGs retrieval.""" - running_response = { - "running_dags": [ - { - "dag_id": "dag-1", - "execution_id": "exec-1", - "started_at": "2025-07-18T18:33:35Z", - "thread_id": 123 - } - ], - "count": 1 - } - - mock_responses.add( - responses.GET, - "http://localhost:8000/dags/running", - json=running_response, - status=200 - ) - - result = self.client.get_running_dags() - - assert result == running_response - assert len(mock_responses.calls) == 1 - - # Cancel DAG Tests - @pytest.mark.unit - def test_cancel_dag_success(self, mock_responses): - """Test successful DAG cancellation.""" - cancel_response = { - "success": True, - "message": "DAG cancelled successfully" - } - - mock_responses.add( - responses.POST, - "http://localhost:8000/dags/test-dag/cancel", - json=cancel_response, - status=200 - ) - - result = self.client.cancel_dag("test-dag") - - assert result == cancel_response - assert len(mock_responses.calls) == 1 - - @pytest.mark.unit - def test_cancel_dag_with_execution_id(self, mock_responses): - """Test DAG cancellation with execution ID.""" - mock_responses.add( - responses.POST, - "http://localhost:8000/dags/test-dag/cancel", - json={"success": True}, - status=200 - ) - - result = self.client.cancel_dag("test-dag", execution_id="exec-123") - - # Check that execution_id parameter was passed - assert mock_responses.calls[0].request.url.endswith("?execution_id=exec-123") - - # Validate DAG Tests - @pytest.mark.unit - def test_validate_dag_success(self, mock_responses): - """Test successful DAG validation.""" - validate_response = { - "valid": True, - "dag_id": "test-dag", - "tasks": [ - { - "task_id": "task-1", - "type": "python", - "dependencies": [] - } - ], - "total_tasks": 1 - } - - mock_responses.add( - responses.POST, - "http://localhost:8000/dags/validate", - json=validate_response, - status=200 - ) - - result = self.client.validate_dag("/path/to/dag.yaml") - - assert result == validate_response - assert len(mock_responses.calls) == 1 - - # Verify request payload - request_body = json.loads(mock_responses.calls[0].request.body) - assert request_body["dag_file_path"] == "/path/to/dag.yaml" - - @pytest.mark.unit - def test_validate_dag_invalid(self, mock_responses): - """Test DAG validation failure.""" - validate_response = { - "valid": False, - "error": "Invalid DAG structure" - } - - mock_responses.add( - responses.POST, - "http://localhost:8000/dags/validate", - json=validate_response, - status=200 - ) - - result = self.client.validate_dag("/path/to/invalid.yaml") - - assert result == validate_response - assert not result["valid"] - - # Cleanup Tests - @pytest.mark.unit - def test_cleanup_old_executions_success(self, mock_responses): - """Test successful cleanup of old executions.""" - cleanup_response = { - "message": "Cleaned up 5 old executions" - } - - mock_responses.add( - responses.DELETE, - "http://localhost:8000/dags/cleanup", - json=cleanup_response, - status=200 - ) - - result = self.client.cleanup_old_executions(days=7) - - assert result == cleanup_response - assert len(mock_responses.calls) == 1 - - # Check that days parameter was passed - assert mock_responses.calls[0].request.url.endswith("?days=7") - - @pytest.mark.unit - def test_cleanup_old_executions_default_days(self, mock_responses): - """Test cleanup with default days parameter.""" - mock_responses.add( - responses.DELETE, - "http://localhost:8000/dags/cleanup", - json={"message": "Cleaned up"}, - status=200 - ) - - result = self.client.cleanup_old_executions() - - # Check that default days=30 was used - assert mock_responses.calls[0].request.url.endswith("?days=30") - - - @pytest.mark.unit - def test_list_dags_with_status_filter(self, mock_responses): - """Test DAG listing with status filter.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/v1/dags", - json={"dags": [], "count": 0, "title": "Running DAGs"}, - status=200 - ) - - result = self.client.list_dags(status_filter="running") - - # Check that status parameter was passed - assert mock_responses.calls[0].request.url.endswith("?status=active") - - # Server Status Tests - @pytest.mark.unit - def test_is_server_running_true(self, mock_responses): - """Test server running check returns True.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/", - json={"status": "healthy"}, - status=200 - ) - - result = self.client.is_server_running() - - assert result is True - - @pytest.mark.unit - def test_is_server_running_false(self, mock_responses): - """Test server running check returns False.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/", - body=requests.exceptions.ConnectionError("Connection failed") - ) - - result = self.client.is_server_running() - - assert result is False - - @pytest.mark.unit - def test_wait_for_server_success(self, mock_responses): - """Test successful server wait.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/", - json={"status": "healthy"}, - status=200 - ) - - result = self.client.wait_for_server(max_wait_time=1) - - assert result is True - - @pytest.mark.unit - def test_wait_for_server_timeout(self, mock_responses): - """Test server wait timeout.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/", - body=requests.exceptions.ConnectionError("Connection failed") - ) - - result = self.client.wait_for_server(max_wait_time=1) - - assert result is False - - # Error Handling Tests - @pytest.mark.unit - def test_make_request_timeout_error(self, mock_responses): - """Test timeout error handling.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/", - body=requests.exceptions.Timeout("Request timed out") - ) - - with pytest.raises(TimeoutError) as exc_info: - self.client.health_check() - - assert "Request to http://localhost:8000/ timed out" in str(exc_info.value) - - @pytest.mark.unit - def test_make_request_http_error(self, mock_responses): - """Test HTTP error handling.""" - mock_responses.add( - responses.GET, - "http://localhost:8000/", - json={"error": "Server error"}, - status=500 - ) - - with pytest.raises(RuntimeError) as exc_info: - self.client.health_check() - - assert "HTTP 500" in str(exc_info.value) - - @pytest.mark.unit - def test_make_request_custom_timeout(self): - """Test client with custom timeout.""" - client = MaestroAPIClient(timeout=5) - - with patch.object(client.session, 'request') as mock_request: - mock_request.return_value.json.return_value = {"status": "ok"} - mock_request.return_value.raise_for_status.return_value = None - - client.health_check() - - # Verify that custom timeout was used - mock_request.assert_called_once_with( - "GET", "http://localhost:8000/", timeout=5 - ) diff --git a/tests/test_base_task.py b/tests/test_base_task.py new file mode 100644 index 0000000..d307b1c --- /dev/null +++ b/tests/test_base_task.py @@ -0,0 +1,30 @@ +import pytest + +from maestro.server.tasks.base import BaseTask + + +def test_basetask_initialization_defaults(): + task = BaseTask(task_id="t1") + + assert task.retries == 0 + assert task.retry_delay == 0 + assert task.dag_id is None + assert task.execution_id is None + + +def test_basetask_custom_fields(): + task = BaseTask( + task_id="t1", retries=3, retry_delay=10, dag_id="D1", execution_id="E1" + ) + + assert task.retries == 3 + assert task.retry_delay == 10 + assert task.dag_id == "D1" + assert task.execution_id == "E1" + + +def test_basetask_execute_local_not_implemented(): + task = BaseTask(task_id="t1") + + with pytest.raises(NotImplementedError): + task.execute_local() diff --git a/tests/test_bash_task.py b/tests/test_bash_task.py new file mode 100644 index 0000000..9e31ba3 --- /dev/null +++ b/tests/test_bash_task.py @@ -0,0 +1,95 @@ +from unittest.mock import Mock, patch + +import pytest + +from maestro.server.internals.status_manager import StatusManager +from maestro.server.tasks.bash_task import BashTask + + +@pytest.fixture +def mock_status_manager(): + sm = Mock() + sm.add_log = Mock() + sm.set_task_output = Mock() + # Patch get_instance() to return our mock + with patch.object(StatusManager, "get_instance", return_value=sm): + yield sm + + +@pytest.fixture +def mock_popen(): + """Creates a fake Popen object we can control.""" + mock_process = Mock() + + # Default: no output, success + mock_process.stdout = [] + mock_process.stderr = [] + mock_process.returncode = 0 + + mock_process.wait = Mock() + return mock_process + + +def test_bash_task_logs_stdout(mock_status_manager, mock_popen): + mock_popen.stdout = ["hello\n", "world\n"] + + with patch("subprocess.Popen", return_value=mock_popen): + task = BashTask(task_id="t1", command="echo test") + task.dag_id = "D1" + task.execution_id = "E1" + + task.execute_local() + + # Check logs + assert mock_status_manager.add_log.call_count == 2 + mock_status_manager.add_log.assert_any_call( + dag_id="D1", + execution_id="E1", + task_id="t1", + message="[BashTask][stdout] hello", + level="INFO", + ) + + +def test_bash_task_logs_stderr(mock_status_manager, mock_popen): + mock_popen.stderr = ["error line\n"] + + with patch("subprocess.Popen", return_value=mock_popen): + task = BashTask(task_id="t1", command="bad") + task.dag_id = "D1" + task.execution_id = "E1" + + task.execute_local() + + mock_status_manager.add_log.assert_any_call( + dag_id="D1", + execution_id="E1", + task_id="t1", + message="[BashTask][stderr] error line", + level="ERROR", + ) + + +def test_bash_task_captures_output_line(mock_status_manager, mock_popen): + mock_popen.stdout = ["OUTPUT: 42\n"] + + with patch("subprocess.Popen", return_value=mock_popen): + task = BashTask(task_id="t1", command="echo OUTPUT:") + task.dag_id = "D1" + task.execution_id = "E1" + + task.execute_local() + + mock_status_manager.set_task_output.assert_called_once_with("D1", "t1", "E1", "42") + + +def test_bash_task_fails_on_nonzero_exit_code(mock_status_manager, mock_popen): + mock_popen.returncode = 3 + + with patch("subprocess.Popen", return_value=mock_popen): + task = BashTask(task_id="t1", command="exit 3") + task.dag_id = "D1" + task.execution_id = "E1" + + with pytest.raises(Exception, match="exit code 3"): + task.execute_local() diff --git a/tests/test_cli_attach.py b/tests/test_cli_attach.py deleted file mode 100644 index 5633c79..0000000 --- a/tests/test_cli_attach.py +++ /dev/null @@ -1,399 +0,0 @@ -#!/usr/bin/env python3 -""" -Test suite for the attach command and streaming functionality. - -This test suite focuses on the complex attach command which involves -signal handling, streaming, and user interaction. -""" - -import pytest -import signal -import time -import unittest.mock -from unittest.mock import patch, ANY -from typer.testing import CliRunner -from maestro.client.cli import app - - -class TestAttachCommand: - """Test suite for the attach command.""" - - def setup_method(self): - """Set up test fixtures.""" - self.runner = CliRunner() - - @pytest.fixture - def mock_api_client(self, mocker): - """Mock API client.""" - return mocker.patch('maestro.cli_client.api_client', autospec=True) - - @pytest.fixture - def mock_check_server(self, mocker): - """Mock server connection check.""" - return mocker.patch('maestro.cli_client.check_server_connection') - - @pytest.fixture - def mock_signal(self, mocker): - """Mock signal handling.""" - return mocker.patch('maestro.cli_client.signal') - - @pytest.mark.unit - def test_attach_command_success(self, mock_api_client, mock_check_server, mock_signal): - """Test successful attach command execution.""" - # Mock streaming logs generator - mock_logs = [ - { - "timestamp": "2025-07-18T18:33:35.123456Z", - "level": "INFO", - "task_id": "task-1", - "message": "Starting task" - }, - { - "timestamp": "2025-07-18T18:33:36.123456Z", - "level": "ERROR", - "task_id": "task-2", - "message": "Task failed" - } - ] - - mock_api_client.stream_dag_logs.return_value = iter(mock_logs) - - result = self.runner.invoke(app, ['attach', 'test-dag']) - - assert result.exit_code == 0 - assert "Attaching to live logs for DAG: test-dag" in result.output - assert "Press Ctrl+C to detach" in result.output - assert "Starting task" in result.output - assert "Task failed" in result.output - - # Verify the API was called correctly - mock_api_client.stream_dag_logs.assert_called_once_with('test-dag', None, None, None) - - # Verify server connection was checked - mock_check_server.assert_called_once() - - # Signal handling test (optional - may not be implemented yet) - # This test will pass whether signal handling is implemented or not - if mock_signal.signal.called: - print("Signal handling is implemented") - calls = mock_signal.signal.call_args_list - print(f"Signal calls: {calls}") - # If it's implemented, verify basic functionality - assert len(calls) >= 1, "At least one signal handler should be registered" - else: - print("Signal handling not yet implemented - this is OK for now") - - @pytest.mark.unit - def test_attach_command_with_filters(self, mock_api_client, mock_check_server, mock_signal): - """Test attach command with filters.""" - mock_api_client.stream_dag_logs.return_value = iter([]) - - result = self.runner.invoke(app, [ - 'attach', 'test-dag', - '--execution-id', 'exec-123', - '--task', 'specific-task', - '--level', 'ERROR' - ]) - - assert result.exit_code == 0 - mock_api_client.stream_dag_logs.assert_called_once_with( - 'test-dag', 'exec-123', 'specific-task', 'ERROR' - ) - - @pytest.mark.unit - def test_attach_command_with_execution_id(self, mock_api_client, mock_check_server, mock_signal): - """Test attach command with execution ID.""" - mock_api_client.stream_dag_logs.return_value = iter([]) - - result = self.runner.invoke(app, [ - 'attach', 'test-dag', - '--execution-id', 'exec-specific' - ]) - - assert result.exit_code == 0 - assert "Execution ID: exec-specific" in result.output - - @pytest.mark.unit - def test_attach_command_stream_error(self, mock_api_client, mock_check_server, mock_signal): - """Test attach command with stream error.""" - mock_logs = [ - {"error": "Stream connection lost"} - ] - - mock_api_client.stream_dag_logs.return_value = iter(mock_logs) - - result = self.runner.invoke(app, ['attach', 'test-dag']) - - assert result.exit_code == 0 - assert "Stream error: Stream connection lost" in result.output - - @pytest.mark.unit - def test_attach_command_keyboard_interrupt(self, mock_api_client, mock_check_server, mock_signal): - """Test attach command with keyboard interrupt.""" - - def mock_stream_logs(*args, **kwargs): - """Mock streaming that raises KeyboardInterrupt.""" - yield {"timestamp": "2025-07-18T18:33:35Z", "level": "INFO", "task_id": "task-1", "message": "Starting"} - raise KeyboardInterrupt("User interrupted") - - mock_api_client.stream_dag_logs.side_effect = mock_stream_logs - - result = self.runner.invoke(app, ['attach', 'test-dag']) - - assert result.exit_code == 0 - assert "Detached from log stream" in result.output - - @pytest.mark.unit - def test_attach_command_general_exception(self, mock_api_client, mock_check_server, mock_signal): - """Test attach command with general exception.""" - mock_api_client.stream_dag_logs.side_effect = RuntimeError("Stream error") - - result = self.runner.invoke(app, ['attach', 'test-dag']) - - assert result.exit_code == 1 - assert "Error: Stream error" in result.output - - @pytest.mark.unit - def test_attach_command_custom_server_url(self, mock_api_client, mock_check_server, mock_signal): - """Test attach command with custom server URL.""" - mock_api_client.stream_dag_logs.return_value = iter([]) - - result = self.runner.invoke(app, [ - 'attach', 'test-dag', - '--server', 'http://custom:9000' - ]) - - assert result.exit_code == 0 - # Verify that the API client base_url was updated - assert mock_api_client.base_url == 'http://custom:9000' - - @pytest.mark.unit - def test_attach_command_log_timestamp_parsing(self, mock_api_client, mock_check_server, mock_signal): - """Test attach command log timestamp parsing.""" - mock_logs = [ - { - "timestamp": "2025-07-18T18:33:35.123456Z", - "level": "INFO", - "task_id": "task-1", - "message": "Test with full timestamp" - }, - { - "timestamp": "18:33:35", - "level": "DEBUG", - "task_id": "task-2", - "message": "Test with time only" - } - ] - - mock_api_client.stream_dag_logs.return_value = iter(mock_logs) - - result = self.runner.invoke(app, ['attach', 'test-dag']) - - assert result.exit_code == 0 - assert "18:33:35" in result.output - assert "Test with full timestamp" in result.output - assert "Test with time only" in result.output - - @pytest.mark.unit - def test_attach_command_log_level_styling(self, mock_api_client, mock_check_server, mock_signal): - """Test attach command log level styling.""" - mock_logs = [ - { - "timestamp": "2025-07-18T18:33:35Z", - "level": "ERROR", - "task_id": "task-1", - "message": "Error message" - }, - { - "timestamp": "2025-07-18T18:33:36Z", - "level": "WARNING", - "task_id": "task-2", - "message": "Warning message" - }, - { - "timestamp": "2025-07-18T18:33:37Z", - "level": "INFO", - "task_id": "task-3", - "message": "Info message" - }, - { - "timestamp": "2025-07-18T18:33:38Z", - "level": "DEBUG", - "task_id": "task-4", - "message": "Debug message" - }, - { - "timestamp": "2025-07-18T18:33:39Z", - "level": "UNKNOWN", - "task_id": "task-5", - "message": "Unknown level message" - } - ] - - mock_api_client.stream_dag_logs.return_value = iter(mock_logs) - - result = self.runner.invoke(app, ['attach', 'test-dag']) - - assert result.exit_code == 0 - # All messages should be present - assert "Error message" in result.output - assert "Warning message" in result.output - assert "Info message" in result.output - assert "Debug message" in result.output - - @pytest.mark.unit - def test_attach_command_basic_functionality(self, mock_api_client, mock_check_server, mock_signal): - """Test basic attach command functionality without signal handling.""" - # Mock streaming logs generator - mock_logs = [ - { - "timestamp": "2025-07-18T18:33:35.123456Z", - "level": "INFO", - "task_id": "task-1", - "message": "Basic functionality test" - } - ] - - mock_api_client.stream_dag_logs.return_value = iter(mock_logs) - - result = self.runner.invoke(app, ['attach', 'test-dag']) - - # Test the core functionality - assert result.exit_code == 0 - assert "Attaching to live logs for DAG: test-dag" in result.output - assert "Basic functionality test" in result.output - - # Verify the API was called correctly - mock_api_client.stream_dag_logs.assert_called_once_with('test-dag', None, None, None) - - # Verify server connection was checked - mock_check_server.assert_called_once() - - -class TestAttachCommandSignalHandling: - """Separate test class for signal handling to isolate these tests.""" - - def setup_method(self): - """Set up test fixtures.""" - self.runner = CliRunner() - - @pytest.fixture - def mock_api_client(self, mocker): - """Mock API client.""" - return mocker.patch('maestro.cli_client.api_client', autospec=True) - - @pytest.fixture - def mock_check_server(self, mocker): - """Mock server connection check.""" - return mocker.patch('maestro.cli_client.check_server_connection') - - @pytest.fixture - def mock_signal(self, mocker): - """Mock signal handling.""" - return mocker.patch('maestro.cli_client.signal') - - @pytest.mark.unit - def test_signal_handling_if_implemented(self, mock_api_client, mock_check_server, mock_signal): - """Test signal handling if it's implemented in the attach command.""" - mock_api_client.stream_dag_logs.return_value = iter([]) - - result = self.runner.invoke(app, ['attach', 'test-dag']) - - assert result.exit_code == 0 - - # Check if signal handling is implemented - if mock_signal.signal.called: - # If signal handling is implemented, verify it - calls = mock_signal.signal.call_args_list - - # Check that at least one signal was registered - assert len(calls) >= 1, "At least one signal handler should be registered" - - # Check for SIGINT specifically - sigint_calls = [call for call in calls if call[0][0] == signal.SIGINT] - if sigint_calls: - # Verify the handler is callable - handler = sigint_calls[0][0][1] - assert callable(handler), "SIGINT handler should be callable" - - print(f"Signal handling is implemented. Registered signals: {[call[0][0] for call in calls]}") - else: - print("Signal handling not implemented in attach command") - # This is fine - the test passes either way - - -class TestStreamingIntegration: - """Integration tests for streaming functionality.""" - - @pytest.mark.integration - def test_attach_command_integration(self, mocker): - """Test attach command integration with API client.""" - runner = CliRunner() - - # Mock the API client's stream method to return a controlled stream - mock_api_client = mocker.patch('maestro.cli_client.api_client') - mock_check_server = mocker.patch('maestro.cli_client.check_server_connection') - - # Create a controlled stream that ends after a few messages - def controlled_stream(*args, **kwargs): - messages = [ - { - "timestamp": "2025-07-18T18:33:35Z", - "level": "INFO", - "task_id": "task-1", - "message": "Task started" - }, - { - "timestamp": "2025-07-18T18:33:36Z", - "level": "INFO", - "task_id": "task-1", - "message": "Task running" - }, - { - "timestamp": "2025-07-18T18:33:37Z", - "level": "INFO", - "task_id": "task-1", - "message": "Task completed" - } - ] - - for msg in messages: - yield msg - - mock_api_client.stream_dag_logs.side_effect = controlled_stream - - result = runner.invoke(app, ['attach', 'test-dag-integration']) - - assert result.exit_code == 0 - assert "Attaching to live logs for DAG: test-dag-integration" in result.output - assert "Task started" in result.output - assert "Task running" in result.output - assert "Task completed" in result.output - - @pytest.mark.slow - def test_attach_command_long_running_stream(self, mocker): - """Test attach command with a longer running stream.""" - runner = CliRunner() - - mock_api_client = mocker.patch('maestro.cli_client.api_client') - mock_check_server = mocker.patch('maestro.cli_client.check_server_connection') - - # Simulate a longer running stream - def long_stream(*args, **kwargs): - for i in range(10): - yield { - "timestamp": f"2025-07-18T18:33:{35 + i:02d}Z", - "level": "INFO", - "task_id": f"task-{i + 1}", - "message": f"Processing item {i + 1}" - } - # Small delay to simulate real streaming - time.sleep(0.01) - - mock_api_client.stream_dag_logs.side_effect = long_stream - - result = runner.invoke(app, ['attach', 'test-dag-long']) - - assert result.exit_code == 0 - assert "Processing item 1" in result.output - assert "Processing item 10" in result.output \ No newline at end of file diff --git a/tests/test_cli_client.py b/tests/test_cli_client.py deleted file mode 100644 index 9428d88..0000000 --- a/tests/test_cli_client.py +++ /dev/null @@ -1,422 +0,0 @@ -#!/usr/bin/env python3 -""" -Comprehensive test suite for the Maestro CLI client. - -This test suite ensures high coverage of all CLI commands and edge cases, -using mocks to isolate the client from server dependencies. -""" - -import pytest -from unittest.mock import patch, Mock, MagicMock -from typer.testing import CliRunner -from maestro.client.cli import app, check_server_connection -import tempfile -import os -from io import StringIO -import typer - - -class TestCliClient: - """Test suite for CLI client commands.""" - - def setup_method(self): - """Set up test fixtures.""" - self.runner = CliRunner() - self.mock_response_data = { - 'dag_id': 'test-dag-123', - 'execution_id': 'exec-456', - 'status': 'submitted', - 'submitted_at': '2025-07-18T18:33:35Z' - } - - @pytest.fixture - def mock_api_client(self, mocker): - """Mock API client with all methods.""" - return mocker.patch('maestro.cli_client.api_client', autospec=True) - - @pytest.fixture - def mock_check_server(self, mocker): - """Mock server connection check.""" - return mocker.patch('maestro.cli_client.check_server_connection') - - @pytest.fixture - def temp_dag_file(self): - """Create a temporary DAG file for testing.""" - with tempfile.NamedTemporaryFile(mode='w', suffix='.yaml', delete=False) as f: - f.write("dag_id: test-dag\ntasks: []") - yield f.name - os.unlink(f.name) - - - - # Status Command Tests - @pytest.mark.unit - def test_status_command_success(self, mock_api_client, mock_check_server): - """Test successful status retrieval.""" - mock_api_client.get_dag_status.return_value = { - 'execution_id': 'exec-123', - 'status': 'running', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': None, - 'thread_id': 'thread-1', - 'tasks': [{ - 'task_id': 'task-1', - 'status': 'completed', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': '2025-07-18T18:34:35Z' - }] - } - - result = self.runner.invoke(app, ['status', 'test-dag']) - - assert result.exit_code == 0 - assert "DAG Status: test-dag" in result.output - assert "running" in result.output - assert "task-1" in result.output - - @pytest.mark.unit - def test_status_command_with_execution_id(self, mock_api_client, mock_check_server): - """Test status retrieval with specific execution ID.""" - mock_api_client.get_dag_status.return_value = { - 'execution_id': 'exec-specific', - 'status': 'completed', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': '2025-07-18T18:34:35Z', - 'thread_id': None, - 'tasks': [] - } - - result = self.runner.invoke(app, ['status', 'test-dag', '--execution-id', 'exec-specific']) - - assert result.exit_code == 0 - mock_api_client.get_dag_status.assert_called_once_with('test-dag', 'exec-specific') - - @pytest.mark.unit - def test_status_command_not_found(self, mock_api_client, mock_check_server): - """Test status retrieval for non-existent DAG.""" - mock_api_client.get_dag_status.side_effect = FileNotFoundError("DAG not found") - - result = self.runner.invoke(app, ['status', 'nonexistent-dag']) - - assert result.exit_code == 1 - assert "DAG execution not found" in result.output - - # Logs Command Tests - @pytest.mark.unit - def test_logs_command_success(self, mock_api_client, mock_check_server): - """Test successful logs retrieval.""" - mock_api_client.get_dag_logs_v1.return_value = [ - { - 'timestamp': '2025-07-18T18:33:35.123456Z', - 'level': 'INFO', - 'task_id': 'task-1', - 'message': 'Task started' - }, - { - 'timestamp': '2025-07-18T18:33:36.123456Z', - 'level': 'ERROR', - 'task_id': 'task-2', - 'message': 'Task failed' - } - ] - - result = self.runner.invoke(app, ['log', 'test-dag']) - - assert "Task started" in result.output - assert "Task failed" in result.output - - @pytest.mark.unit - def test_logs_command_with_filters(self, mock_api_client, mock_check_server): - """Test logs retrieval with filters.""" - mock_api_client.get_dag_logs_v1.return_value = [] - - result = self.runner.invoke(app, [ - 'log', 'test-dag', - '--limit', '50', - '--task', 'specific-task', - '--level', 'ERROR' - ]) - - assert result.exit_code == 0 - mock_api_client.get_dag_logs_v1.assert_called_once_with( - 'test-dag', None, 50, 'specific-task', 'ERROR' - ) - - @pytest.mark.unit - def test_logs_command_no_logs(self, mock_api_client, mock_check_server): - """Test logs retrieval when no logs exist.""" - mock_api_client.get_dag_logs_v1.return_value = [] - - result = self.runner.invoke(app, ['log', 'test-dag']) - - assert result.exit_code == 0 - assert "No logs found for DAG: test-dag" in result.output - - # Running Command Tests - @pytest.mark.unit - def test_running_command_success(self, mock_api_client, mock_check_server): - """Test successful running DAGs retrieval.""" - mock_api_client.list_dags_v1.return_value = [ - { - 'dag_id': 'dag-1', - 'execution_id': 'exec-1', - 'status': 'running', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': None, - 'thread_id': 123 - }, - { - 'dag_id': 'dag-2', - 'execution_id': 'exec-2', - 'status': 'running', - 'started_at': '2025-07-18T18:34:35Z', - 'completed_at': None, - 'thread_id': 456 - } - ] - - result = self.runner.invoke(app, ['ls', '--filter', 'active']) - - assert result.exit_code == 0 - assert "Maestro DAGs" in result.output - assert "dag-1" in result.output - assert "dag-2" in result.output - assert "Total DAGs: 2" in result.output - - @pytest.mark.unit - def test_running_command_no_running_dags(self, mock_api_client, mock_check_server): - """Test running DAGs retrieval when none are running.""" - mock_api_client.list_dags_v1.return_value = [] - - result = self.runner.invoke(app, ['ls', '--filter', 'active']) - - assert result.exit_code == 0 - assert "No DAGs found with filter 'active'" in result.output - - # Cancel Command Tests - @pytest.mark.unit - def test_cancel_command_success(self, mock_api_client, mock_check_server): - """Test successful DAG cancellation.""" - mock_api_client.stop_dag.return_value = { - 'message': 'DAG cancelled successfully' - } - - result = self.runner.invoke(app, ['stop', 'test-dag']) - - assert result.exit_code == 0 - assert "DAG cancelled successfully" in result.output - - @pytest.mark.unit - def test_cancel_command_not_running(self, mock_api_client, mock_check_server): - """Test cancellation of non-running DAG.""" - mock_api_client.stop_dag.return_value = { - 'message': 'DAG is not running' - } - - result = self.runner.invoke(app, ['stop', 'test-dag']) - - assert result.exit_code == 0 - assert "DAG is not running" in result.output - - # Validate Command Tests - @pytest.mark.unit - def test_validate_command_success(self, mock_api_client, mock_check_server, temp_dag_file): - """Test successful DAG validation.""" - mock_api_client.validate_dag.return_value = { - 'valid': True, - 'dag_id': 'test-dag', - 'tasks': [ - { - 'task_id': 'task-1', - 'type': 'python', - 'dependencies': [] - }, - { - 'task_id': 'task-2', - 'type': 'shell', - 'dependencies': ['task-1'] - } - ], - 'total_tasks': 2 - } - - result = self.runner.invoke(app, ['validate', temp_dag_file]) - - assert result.exit_code == 0 - assert "✓ DAG is valid" in result.output - assert "test-dag" in result.output - assert "task-1" in result.output - assert "task-2" in result.output - assert "Total tasks: 2" in result.output - - @pytest.mark.unit - def test_validate_command_invalid(self, mock_api_client, mock_check_server, temp_dag_file): - """Test DAG validation failure.""" - mock_api_client.validate_dag.return_value = { - 'valid': False, - 'error': 'Invalid DAG structure' - } - - result = self.runner.invoke(app, ['validate', temp_dag_file]) - - assert result.exit_code == 1 - assert "✗ DAG validation failed" in result.output - assert "Invalid DAG structure" in result.output - - # Cleanup Command Tests - @pytest.mark.unit - def test_cleanup_command_success(self, mock_api_client, mock_check_server): - """Test successful cleanup.""" - mock_api_client.cleanup_old_executions.return_value = { - 'message': 'Cleaned up 5 old executions' - } - - result = self.runner.invoke(app, ['cleanup', '--days', '7']) - - assert result.exit_code == 0 - assert "Cleaned up 5 old executions" in result.output - mock_api_client.cleanup_old_executions.assert_called_once_with(7) - - # List Command Tests - @pytest.mark.unit - def test_list_command_success(self, mock_api_client, mock_check_server): - """Test successful DAG listing.""" - mock_api_client.list_dags_v1.return_value = [ - { - 'dag_id': 'dag-1', - 'execution_id': 'exec-1', - 'status': 'completed', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': '2025-07-18T18:34:35Z', - 'thread_id': 123 - }, - { - 'dag_id': 'dag-2', - 'execution_id': 'exec-2', - 'status': 'running', - 'started_at': '2025-07-18T18:35:35Z', - 'completed_at': None, - 'thread_id': 456 - } - ] - - result = self.runner.invoke(app, ['ls']) - - assert result.exit_code == 0 - assert "Maestro DAGs" in result.output - assert "dag-1" in result.output - assert "dag-2" in result.output - assert "Total DAGs: 2" in result.output - - @pytest.mark.unit - def test_list_command_with_status_filter(self, mock_api_client, mock_check_server): - """Test DAG listing with status filter.""" - mock_api_client.list_dags_v1.return_value = [] - - result = self.runner.invoke(app, ['ls', '--filter', 'running']) - - assert result.exit_code == 0 - mock_api_client.list_dags_v1.assert_called_once_with('running') - - @pytest.mark.unit - def test_list_command_active_flag(self, mock_api_client, mock_check_server): - """Test DAG listing with active flag.""" - mock_api_client.list_dags_v1.return_value = [] - - result = self.runner.invoke(app, ['ls', '--filter', 'active']) - - assert result.exit_code == 0 - mock_api_client.list_dags_v1.assert_called_once_with('active') - - @pytest.mark.unit - def test_list_command_no_dags(self, mock_api_client, mock_check_server): - """Test DAG listing when no DAGs exist.""" - mock_api_client.list_dags_v1.return_value = [] - - result = self.runner.invoke(app, ['ls']) - - assert result.exit_code == 0 - assert "No DAGs found" in result.output - - # Server Commands Tests - @pytest.mark.unit - def test_server_status_command_running(self, mock_api_client, mock_check_server): - """Test server status when running.""" - mock_api_client.health_check.return_value = { - 'status': 'healthy', - 'timestamp': '2025-07-18T18:33:35Z' - } - - result = self.runner.invoke(app, ['server', 'status']) - - assert result.exit_code == 0 - assert "Server is running" in result.output - assert "healthy" in result.output - - @pytest.mark.unit - def test_server_status_command_not_running(self, mock_api_client, mock_check_server): - """Test server status when not running.""" - mock_api_client.health_check.side_effect = ConnectionError("Connection failed") - - result = self.runner.invoke(app, ['server', 'status']) - - assert result.exit_code == 1 - assert "Server is not running" in result.output - - @pytest.mark.unit - def test_server_start_command_daemon(self, mock_api_client, mock_check_server, mocker): - """Test server start in daemon mode.""" - mock_popen = mocker.patch('maestro.cli_client.subprocess.Popen') - mock_api_client.wait_for_server.return_value = True - - result = self.runner.invoke(app, ['server', 'start', '--daemon']) - - assert result.exit_code == 0 - assert "Maestro server started" in result.output - mock_popen.assert_called_once() - - @pytest.mark.unit - def test_server_start_command_daemon_fail(self, mock_api_client, mock_check_server, mocker): - """Test server start daemon failure.""" - mock_popen = mocker.patch('maestro.cli_client.subprocess.Popen') - mock_api_client.wait_for_server.return_value = False - - result = self.runner.invoke(app, ['server', 'start', '--daemon']) - - assert result.exit_code == 1 - assert "Failed to start server" in result.output - - @pytest.mark.unit - def test_server_stop_command(self, mock_api_client, mock_check_server): - """Test server stop command.""" - result = self.runner.invoke(app, ['server', 'stop']) - - assert result.exit_code == 0 - assert "Server stop command not implemented" in result.output - - -class TestServerConnection: - """Test suite for server connection functionality.""" - - @pytest.mark.unit - def test_check_server_connection_success(self, mocker): - """Test successful server connection check.""" - mock_api_client = mocker.patch('maestro.cli_client.api_client') - mock_api_client.is_server_running.return_value = True - - # Should not raise any exception - check_server_connection() - mock_api_client.is_server_running.assert_called_once() - - @pytest.mark.unit - def test_check_server_connection_failure(self, mocker): - """Test server connection failure.""" - mock_api_client = mocker.patch('maestro.cli_client.api_client') - mock_api_client.is_server_running.return_value = False - mock_console = mocker.patch('maestro.cli_client.console') - - with pytest.raises(typer.Exit) as exc_info: - check_server_connection() - - assert exc_info.value.exit_code == 1 - mock_console.print.assert_called() \ No newline at end of file diff --git a/tests/test_cli_integration.py b/tests/test_cli_integration.py deleted file mode 100644 index b1c997f..0000000 --- a/tests/test_cli_integration.py +++ /dev/null @@ -1,520 +0,0 @@ -#!/usr/bin/env python3 -""" -Integration tests for the Maestro CLI client. - -These tests focus on testing the interactions between different components -and edge cases that might occur in real usage scenarios. -""" - -import pytest -import tempfile -import os -from unittest.mock import patch, Mock -from typer.testing import CliRunner -from maestro.client.cli import app - - -class TestCliIntegration: - """Integration tests for CLI client.""" - - def setup_method(self): - """Set up test fixtures.""" - self.runner = CliRunner() - - @pytest.fixture - def mock_api_client(self, mocker): - """Mock API client.""" - return mocker.patch('maestro.cli_client.api_client', autospec=True) - - @pytest.fixture - def mock_check_server(self, mocker): - """Mock server connection check.""" - return mocker.patch('maestro.cli_client.check_server_connection') - - @pytest.fixture - def temp_dag_file(self): - """Create a temporary DAG file.""" - with tempfile.NamedTemporaryFile(mode='w', suffix='.yaml', delete=False) as f: - f.write(""" -dag_id: test-integration-dag -tasks: - - task_id: task1 - type: shell - command: echo "Hello World" - - task_id: task2 - type: shell - command: echo "Task 2" - depends_on: [task1] -""") - yield f.name - os.unlink(f.name) - - @pytest.mark.integration - def test_full_dag_workflow(self, mock_api_client, mock_check_server, temp_dag_file): - """Test complete DAG workflow: submit -> status -> logs -> cancel.""" - # Mock responses for different stages - submit_response = { - 'dag_id': 'test-integration-dag', - 'execution_id': 'exec-integration-123', - 'status': 'submitted', - 'submitted_at': '2025-07-18T18:33:35Z' - } - - status_response = { - 'execution_id': 'exec-integration-123', - 'status': 'running', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': None, - 'thread_id': 'thread-123', - 'tasks': [ - { - 'task_id': 'task1', - 'status': 'completed', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': '2025-07-18T18:33:40Z' - }, - { - 'task_id': 'task2', - 'status': 'running', - 'started_at': '2025-07-18T18:33:40Z', - 'completed_at': None - } - ] - } - - logs_response = { - 'logs': [ - { - 'timestamp': '2025-07-18T18:33:35Z', - 'level': 'INFO', - 'task_id': 'task1', - 'message': 'Starting task1' - }, - { - 'timestamp': '2025-07-18T18:33:37Z', - 'level': 'INFO', - 'task_id': 'task1', - 'message': 'Hello World' - }, - { - 'timestamp': '2025-07-18T18:33:40Z', - 'level': 'INFO', - 'task_id': 'task2', - 'message': 'Starting task2' - } - ], - 'total_count': 3 - } - - cancel_response = { - 'success': True, - 'message': 'DAG cancelled successfully' - } - - # mock_api_client.submit_dag.return_value = submit_response #submit api removed - mock_api_client.get_dag_status.return_value = status_response - mock_api_client.get_dag_logs.return_value = logs_response - mock_api_client.cancel_dag.return_value = cancel_response - - # Step 1: Submit DAG - result = self.runner.invoke(app, ['submit', temp_dag_file]) - assert result.exit_code == 0 - assert '✓ DAG submitted successfully!' in result.output - assert 'test-integration-dag' in result.output - - # Step 2: Check status - result = self.runner.invoke(app, ['status', 'test-integration-dag']) - assert result.exit_code == 0 - assert 'DAG Status: test-integration-dag' in result.output - assert 'running' in result.output - assert 'task1' in result.output - assert 'task2' in result.output - - # Step 3: Get logs - result = self.runner.invoke(app, ['logs', 'test-integration-dag']) - assert result.exit_code == 0 - assert 'Logs: test-integration-dag' in result.output - assert 'Hello World' in result.output - assert 'Starting task2' in result.output - - # Step 4: Cancel DAG - result = self.runner.invoke(app, ['cancel', 'test-integration-dag']) - assert result.exit_code == 0 - assert 'DAG cancelled successfully' in result.output - - # Verify all API calls were made - # mock_api_client.submit_dag.assert_called_once() - mock_api_client.get_dag_status.assert_called_once() - mock_api_client.get_dag_logs.assert_called_once() - mock_api_client.cancel_dag.assert_called_once() - - @pytest.mark.integration - def test_error_handling_chain(self, mock_api_client, mock_check_server): - """Test error handling across multiple commands.""" - # Test server connection error - mock_check_server.side_effect = SystemExit(1) - - result = self.runner.invoke(app, ['status', 'test-dag']) - assert result.exit_code == 1 - - # Reset mock - mock_check_server.side_effect = None - - # Test DAG not found error - mock_api_client.get_dag_status.side_effect = FileNotFoundError("DAG not found") - - result = self.runner.invoke(app, ['status', 'nonexistent-dag']) - assert result.exit_code == 1 - assert 'DAG execution not found' in result.output - - # Test API error - mock_api_client.get_dag_status.side_effect = RuntimeError("API Error") - - result = self.runner.invoke(app, ['status', 'error-dag']) - assert result.exit_code == 1 - assert 'Error: API Error' in result.output - - @pytest.mark.integration - def test_dag_lifecycle_states(self, mock_api_client, mock_check_server, temp_dag_file): - """Test DAG through different lifecycle states.""" - # Test 1: Submitted state - # mock_api_client.submit_dag.return_value = { - # 'dag_id': 'lifecycle-dag', - # 'execution_id': 'exec-lifecycle', - # 'status': 'submitted', - # 'submitted_at': '2025-07-18T18:33:35Z' - # } - - result = self.runner.invoke(app, ['submit', temp_dag_file]) - assert result.exit_code == 0 - assert 'submitted' in result.output - - # Test 2: Running state - mock_api_client.get_dag_status.return_value = { - 'execution_id': 'exec-lifecycle', - 'status': 'running', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': None, - 'thread_id': 'thread-123', - 'tasks': [ - { - 'task_id': 'task1', - 'status': 'running', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': None - } - ] - } - - result = self.runner.invoke(app, ['status', 'lifecycle-dag']) - assert result.exit_code == 0 - assert 'running' in result.output - - # Test 3: Completed state - mock_api_client.get_dag_status.return_value = { - 'execution_id': 'exec-lifecycle', - 'status': 'completed', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': '2025-07-18T18:35:35Z', - 'thread_id': None, - 'tasks': [ - { - 'task_id': 'task1', - 'status': 'completed', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': '2025-07-18T18:35:35Z' - } - ] - } - - result = self.runner.invoke(app, ['status', 'lifecycle-dag']) - assert result.exit_code == 0 - assert 'completed' in result.output - - # Test 4: Failed state - mock_api_client.get_dag_status.return_value = { - 'execution_id': 'exec-lifecycle', - 'status': 'failed', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': '2025-07-18T18:34:35Z', - 'thread_id': None, - 'tasks': [ - { - 'task_id': 'task1', - 'status': 'failed', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': '2025-07-18T18:34:35Z' - } - ] - } - - result = self.runner.invoke(app, ['status', 'lifecycle-dag']) - assert result.exit_code == 0 - assert 'failed' in result.output - - @pytest.mark.integration - def test_multiple_dag_management(self, mock_api_client, mock_check_server): - """Test managing multiple DAGs simultaneously.""" - # Mock multiple running DAGs - mock_api_client.get_running_dags.return_value = { - 'running_dags': [ - { - 'dag_id': 'dag-1', - 'execution_id': 'exec-1', - 'started_at': '2025-07-18T18:33:35Z', - 'thread_id': 123 - }, - { - 'dag_id': 'dag-2', - 'execution_id': 'exec-2', - 'started_at': '2025-07-18T18:34:35Z', - 'thread_id': 456 - }, - { - 'dag_id': 'dag-3', - 'execution_id': 'exec-3', - 'started_at': '2025-07-18T18:35:35Z', - 'thread_id': 789 - } - ], - 'count': 3 - } - - result = self.runner.invoke(app, ['running']) - assert result.exit_code == 0 - assert 'dag-1' in result.output - assert 'dag-2' in result.output - assert 'dag-3' in result.output - assert 'Total running DAGs: 3' in result.output - - # Test list with different filters - mock_api_client.list_dags.return_value = { - 'dags': [ - { - 'dag_id': 'completed-dag', - 'execution_id': 'exec-completed', - 'status': 'completed', - 'started_at': '2025-07-18T18:30:35Z', - 'completed_at': '2025-07-18T18:32:35Z', - 'thread_id': None - } - ], - 'count': 1, - 'title': 'Completed DAGs' - } - - result = self.runner.invoke(app, ['list', '--status', 'completed']) - assert result.exit_code == 0 - assert 'completed-' in result.output # DAG ID is truncated in Rich table - assert 'Completed DAGs' in result.output - - @pytest.mark.integration - def test_dag_validation_workflow(self, mock_api_client, mock_check_server, temp_dag_file): - """Test DAG validation before submission.""" - # Test valid DAG - mock_api_client.validate_dag.return_value = { - 'valid': True, - 'dag_id': 'test-integration-dag', - 'tasks': [ - { - 'task_id': 'task1', - 'type': 'shell', - 'dependencies': [] - }, - { - 'task_id': 'task2', - 'type': 'shell', - 'dependencies': ['task1'] - } - ], - 'total_tasks': 2 - } - - result = self.runner.invoke(app, ['validate', temp_dag_file]) - assert result.exit_code == 0 - assert '✓ DAG is valid' in result.output - assert 'test-integration-dag' in result.output - assert 'task1' in result.output - assert 'task2' in result.output - assert 'Total tasks: 2' in result.output - - # Test invalid DAG - mock_api_client.validate_dag.return_value = { - 'valid': False, - 'error': 'Circular dependency detected between task1 and task2' - } - - result = self.runner.invoke(app, ['validate', temp_dag_file]) - assert result.exit_code == 1 - assert '✗ DAG validation failed' in result.output - assert 'Circular dependency detected' in result.output - - @pytest.mark.integration - def test_server_management_workflow(self, mock_api_client, mock_check_server, mocker): - """Test server management commands.""" - # Test server status when not running - mock_api_client.health_check.side_effect = ConnectionError("Connection failed") - - result = self.runner.invoke(app, ['server', 'status']) - assert result.exit_code == 1 - assert 'Server is not running' in result.output - - # Test server status when running - mock_api_client.health_check.side_effect = None - mock_api_client.health_check.return_value = { - 'status': 'healthy', - 'timestamp': '2025-07-18T18:33:35Z' - } - - result = self.runner.invoke(app, ['server', 'status']) - assert result.exit_code == 0 - assert 'Server is running' in result.output - assert 'healthy' in result.output - - # Test server start daemon mode - mock_popen = mocker.patch('maestro.cli_client.subprocess.Popen') - mock_api_client.wait_for_server.return_value = True - - result = self.runner.invoke(app, ['server', 'start', '--daemon', '--port', '9000']) - assert result.exit_code == 0 - assert 'Maestro server started' in result.output - - # Verify subprocess was called with correct arguments - mock_popen.assert_called_once() - call_args = mock_popen.call_args[0][0] - assert '--port' in call_args - assert '9000' in call_args - - @pytest.mark.integration - def test_log_filtering_and_display(self, mock_api_client, mock_check_server): - """Test log filtering and display functionality.""" - # Test logs with different levels and tasks - mock_api_client.get_dag_logs.return_value = { - 'logs': [ - { - 'timestamp': '2025-07-18T18:33:35Z', - 'level': 'DEBUG', - 'task_id': 'task1', - 'message': 'Debug message from task1' - }, - { - 'timestamp': '2025-07-18T18:33:36Z', - 'level': 'INFO', - 'task_id': 'task1', - 'message': 'Info message from task1' - }, - { - 'timestamp': '2025-07-18T18:33:37Z', - 'level': 'WARNING', - 'task_id': 'task2', - 'message': 'Warning message from task2' - }, - { - 'timestamp': '2025-07-18T18:33:38Z', - 'level': 'ERROR', - 'task_id': 'task2', - 'message': 'Error message from task2' - } - ], - 'total_count': 4 - } - - # Test with different filtering combinations - result = self.runner.invoke(app, [ - 'logs', 'test-dag', - '--limit', '10', - '--task', 'task1', - '--level', 'INFO' - ]) - - assert result.exit_code == 0 - assert 'Logs: test-dag' in result.output - - # Verify API was called with correct parameters - mock_api_client.get_dag_logs.assert_called_with( - 'test-dag', None, 10, 'task1', 'INFO' - ) - - @pytest.mark.integration - def test_cleanup_workflow(self, mock_api_client, mock_check_server): - """Test cleanup workflow with different scenarios.""" - # Test successful cleanup - mock_api_client.cleanup_old_executions.return_value = { - 'message': 'Cleaned up 15 old executions (older than 7 days)' - } - - result = self.runner.invoke(app, ['cleanup', '--days', '7']) - assert result.exit_code == 0 - assert 'Cleaned up 15 old executions' in result.output - - # Test cleanup with no old executions - mock_api_client.cleanup_old_executions.return_value = { - 'message': 'No old executions found to clean up' - } - - result = self.runner.invoke(app, ['cleanup', '--days', '30']) - assert result.exit_code == 0 - assert 'No old executions found' in result.output - - # Test cleanup with default days - result = self.runner.invoke(app, ['cleanup']) - assert result.exit_code == 0 - mock_api_client.cleanup_old_executions.assert_called_with(30) - - @pytest.mark.integration - def test_edge_cases_and_error_recovery(self, mock_api_client, mock_check_server, temp_dag_file): - """Test edge cases and error recovery scenarios.""" - # Test with very long DAG ID - long_dag_id = 'a' * 255 - mock_api_client.get_dag_status.return_value = { - 'execution_id': 'exec-long', - 'status': 'running', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': None, - 'thread_id': 'thread-123', - 'tasks': [] - } - - result = self.runner.invoke(app, ['status', long_dag_id]) - assert result.exit_code == 0 - - # Test with empty logs - mock_api_client.get_dag_logs.return_value = { - 'logs': [], - 'total_count': 0 - } - - result = self.runner.invoke(app, ['logs', 'empty-dag']) - assert result.exit_code == 0 - assert 'No logs found' in result.output - - # Test with malformed timestamp - mock_api_client.get_dag_logs.return_value = { - 'logs': [ - { - 'timestamp': 'invalid-timestamp', - 'level': 'INFO', - 'task_id': 'task1', - 'message': 'Test message' - } - ], - 'total_count': 1 - } - - result = self.runner.invoke(app, ['logs', 'malformed-dag']) - assert result.exit_code == 0 - assert 'Test message' in result.output - - # Test with missing task data - mock_api_client.get_dag_status.return_value = { - 'execution_id': 'exec-missing', - 'status': 'running', - 'started_at': '2025-07-18T18:33:35Z', - 'completed_at': None, - 'thread_id': None, - 'tasks': [] - } - - result = self.runner.invoke(app, ['status', 'missing-tasks-dag']) - assert result.exit_code == 0 - # Should handle missing tasks gracefully diff --git a/tests/test_cron_feature.py b/tests/test_cron_feature.py deleted file mode 100644 index 8a228a9..0000000 --- a/tests/test_cron_feature.py +++ /dev/null @@ -1,42 +0,0 @@ -import pytest -from datetime import datetime, timedelta -from maestro.shared.dag import DAG - -def test_valid_cron_expression(): - cron_schedule = "0 9 * * *" # Every day at 9:00 AM - dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - assert dag.cron_schedule == cron_schedule - -def test_invalid_cron_expression(): - invalid_cron = "invalid cron" - with pytest.raises(ValueError, match="Invalid cron expression"): - DAG(dag_id="test_dag", cron_schedule=invalid_cron) - -def test_dag_ready_to_start(): - cron_schedule = "* * * * *" # Every minute - dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - now = datetime.now() - assert dag.is_ready_to_start(now) - -def test_dag_not_ready_to_start(): - cron_schedule = "0 9 * * *" # Every day at 9:00 AM - dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - now = datetime.now().replace(hour=10) - assert not dag.is_ready_to_start(now) - -def test_dag_next_run_time(): - cron_schedule = "0 9 * * *" # Every day at 9:00 AM - dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - now = datetime(2024, 1, 1, 8, 0, 0) # 8:00 AM - next_run = dag.get_next_run_time(now) - assert next_run.hour == 9 - assert next_run.minute == 0 - -def test_dag_next_run_time_wrap_around(): - cron_schedule = "0 9 * * *" # Every day at 9:00 AM - dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - now = datetime(2024, 1, 1, 10, 0, 0) # After the run time - next_run = dag.get_next_run_time(now) - assert next_run.hour == 9 - assert next_run.day == 2 # Next day - diff --git a/tests/test_dag.py b/tests/test_dag.py deleted file mode 100644 index cf2792b..0000000 --- a/tests/test_dag.py +++ /dev/null @@ -1,139 +0,0 @@ - -import pytest -from datetime import datetime, timedelta - -from maestro.shared.dag import DAG -from maestro.shared.task import Task - -class DummyTask(Task): - def execute_local(self): - pass - -def test_dag_add_task(): - dag = DAG() - task = DummyTask(task_id="test_task") - dag.add_task(task) - assert "test_task" in dag.tasks - -def test_dag_validation(): - dag = DAG() - task1 = DummyTask(task_id="task1") - task2 = DummyTask(task_id="task2", dependencies=["task1"]) - dag.add_task(task1) - dag.add_task(task2) - dag.validate() - -def test_dag_cycle_detection(): - dag = DAG() - task1 = DummyTask(task_id="task1", dependencies=["task3"]) - task2 = DummyTask(task_id="task2", dependencies=["task1"]) - task3 = DummyTask(task_id="task3", dependencies=["task2"]) - dag.add_task(task1) - dag.add_task(task2) - dag.add_task(task3) - with pytest.raises(ValueError, match="DAG has a cycle."): - dag.validate() - -def test_dag_with_start_time(): - """Test DAG creation with start_time parameter.""" - start_time = datetime(2024, 1, 1, 9, 0, 0) - dag = DAG(dag_id="test_dag", start_time=start_time) - - assert dag.dag_id == "test_dag" - assert dag.start_time == start_time - -def test_dag_without_start_time(): - """Test DAG creation without start_time parameter.""" - dag = DAG(dag_id="test_dag") - - assert dag.dag_id == "test_dag" - assert dag.start_time is None - -def test_dag_is_ready_to_start_future(): - """Test DAG readiness check with future start time.""" - future_time = datetime.now() + timedelta(hours=1) - dag = DAG(dag_id="test_dag", start_time=future_time) - - assert not dag.is_ready_to_start() - assert dag.time_until_start() > 0 - -def test_dag_is_ready_to_start_past(): - """Test DAG readiness check with past start time.""" - past_time = datetime.now() - timedelta(hours=1) - dag = DAG(dag_id="test_dag", start_time=past_time) - - assert dag.is_ready_to_start() - assert dag.time_until_start() == 0 - -def test_dag_is_ready_to_start_no_time(): - """Test DAG readiness check without start time.""" - dag = DAG(dag_id="test_dag") - - assert dag.is_ready_to_start() - assert dag.time_until_start() is None - -def test_dag_with_cron_schedule(): - """Test DAG creation with cron schedule.""" - cron_schedule = "0 9 * * *" # Every day at 9:00 AM - dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - - assert dag.dag_id == "test_dag" - assert dag.cron_schedule == cron_schedule - assert dag.start_time is None - -def test_dag_with_invalid_cron_schedule(): - """Test DAG creation with invalid cron schedule.""" - invalid_cron = "invalid cron" - - with pytest.raises(ValueError, match="Invalid cron expression"): - DAG(dag_id="test_dag", cron_schedule=invalid_cron) - -def test_dag_with_both_start_time_and_cron(): - """Test DAG creation with both start_time and cron_schedule (should fail).""" - start_time = datetime(2024, 1, 1, 9, 0, 0) - cron_schedule = "0 9 * * *" - - with pytest.raises(ValueError, match="Cannot specify both start_time and cron_schedule"): - DAG(dag_id="test_dag", start_time=start_time, cron_schedule=cron_schedule) - -def test_dag_cron_schedule_readiness(): - """Test DAG readiness check with cron schedule.""" - # Test with a cron that runs every minute - cron_schedule = "* * * * *" - dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - - # Should be ready most of the time since it runs every minute - current_time = datetime.now() - assert dag.is_ready_to_start(current_time) - - # Test with specific time - test_time = datetime(2024, 1, 1, 9, 0, 30) # 30 seconds after 9:00 AM - assert dag.is_ready_to_start(test_time) - -def test_dag_cron_next_run_time(): - """Test getting next run time for cron scheduled DAG.""" - cron_schedule = "0 9 * * *" # Every day at 9:00 AM - dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - - test_time = datetime(2024, 1, 1, 8, 0, 0) # 8:00 AM - next_run = dag.get_next_run_time(test_time) - - assert next_run is not None - assert next_run.hour == 9 - assert next_run.minute == 0 - -def test_dag_schedule_descriptions(): - """Test schedule description methods.""" - # Test with start_time - start_time = datetime(2024, 1, 1, 9, 0, 0) - dag_start_time = DAG(dag_id="test_dag", start_time=start_time) - assert "One-time execution" in dag_start_time.get_schedule_description() - - # Test with cron schedule - cron_schedule = "0 9 * * *" - dag_cron = DAG(dag_id="test_dag", cron_schedule=cron_schedule) - assert "Cron schedule" in dag_cron.get_schedule_description() - - # Test with no schedule - dag_no_schedule = DAG(dag_id="test_dag") - assert "No schedule" in dag_no_schedule.get_schedule_description() diff --git a/tests/test_dag_id_generation.py b/tests/test_dag_id_generation.py deleted file mode 100644 index 264e450..0000000 --- a/tests/test_dag_id_generation.py +++ /dev/null @@ -1,259 +0,0 @@ -""" -Test cases for DAG ID generation and validation functionality. -""" - -import pytest -import re -from unittest.mock import Mock, patch, MagicMock -from maestro.server.internals.status_manager import StatusManager, DOCKER_ADJECTIVES, DOCKER_NOUNS -import tempfile -import os -import sqlite3 - - -class TestDockerLikeNameGeneration: - """Test Docker-like name generation functionality.""" - - def setup_method(self): - """Set up a temporary database for each test.""" - self.temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") - self.temp_db.close() - self.sm = StatusManager(self.temp_db.name) - - def teardown_method(self): - """Clean up the temporary database.""" - if os.path.exists(self.temp_db.name): - os.unlink(self.temp_db.name) - - def test_generate_docker_like_name_format(self): - """Test that generated names follow the adjective_noun format.""" - with self.sm as sm: - name = sm.generate_unique_dag_id() - assert "_" in name - parts = name.split("_") - # Should have at least adjective and noun (might have suffix if collision) - assert len(parts) >= 2 - adjective = parts[0] - noun = parts[1] - assert adjective in DOCKER_ADJECTIVES - assert noun in DOCKER_NOUNS - - def test_generate_docker_like_name_randomness(self): - """Test that generated names are different (mostly).""" - with self.sm as sm: - names = [sm.generate_unique_dag_id() for _ in range(10)] - # Most names should be different (allowing for some duplicates due to randomness) - unique_names = set(names) - assert len(unique_names) >= 7 # Allow for some duplicates - - def test_generate_docker_like_name_valid_characters(self): - """Test that generated names only contain valid characters.""" - with self.sm as sm: - name = sm.generate_unique_dag_id() - # Should only contain alphanumeric, underscores, and hyphens - assert re.match(r'^[a-zA-Z0-9_-]+$', name) - - -class TestDAGIDValidation: - """Test DAG ID validation functionality.""" - - def setup_method(self): - """Set up a temporary database for each test.""" - self.temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") - self.temp_db.close() - self.sm = StatusManager(self.temp_db.name) - - def teardown_method(self): - """Clean up the temporary database.""" - if os.path.exists(self.temp_db.name): - os.unlink(self.temp_db.name) - - def test_validate_dag_id_valid_cases(self): - """Test validation with valid DAG IDs.""" - valid_ids = [ - "simple_dag", - "dag-with-hyphens", - "dag_with_123_numbers", - "CamelCaseDAG", - "simple", - "a1b2c3", - "test-dag_123" - ] - - with self.sm as sm: - for dag_id in valid_ids: - assert sm.validate_dag_id(dag_id), f"'{dag_id}' should be valid" - - def test_validate_dag_id_invalid_cases(self): - """Test validation with invalid DAG IDs.""" - invalid_ids = [ - "", # Empty string - "dag with spaces", # Contains spaces - "dag@with@symbols", # Contains special characters - "dag.with.dots", # Contains dots - "dag/with/slashes", # Contains slashes - "dag#with#hash", # Contains hash - "dag%with%percent", # Contains percent - None, # None value - ] - - with self.sm as sm: - for dag_id in invalid_ids: - assert not sm.validate_dag_id(dag_id), f"'{dag_id}' should be invalid" - - -class TestDAGIDUniquenessCheck: - """Test DAG ID uniqueness checking functionality.""" - - def setup_method(self): - """Set up a temporary database for each test.""" - self.temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") - self.temp_db.close() - self.sm = StatusManager(self.temp_db.name) - - def teardown_method(self): - """Clean up the temporary database.""" - if os.path.exists(self.temp_db.name): - os.unlink(self.temp_db.name) - - def test_check_dag_id_uniqueness_unique(self): - """Test uniqueness check when DAG ID is unique.""" - with self.sm as sm: - # Add some existing DAGs - sm._create_dag_if_not_exists('existing_dag_1') - sm._create_dag_if_not_exists('existing_dag_2') - - # Test with a new DAG ID - assert sm.check_dag_id_uniqueness('new_dag_id') is True - - def test_check_dag_id_uniqueness_duplicate(self): - """Test uniqueness check when DAG ID already exists.""" - with self.sm as sm: - # Add some existing DAGs - sm._create_dag_if_not_exists('existing_dag_1') - sm._create_dag_if_not_exists('existing_dag_2') - - # Test with an existing DAG ID - assert sm.check_dag_id_uniqueness('existing_dag_1') is False - - def test_check_dag_id_uniqueness_error_handling(self): - """Test uniqueness check error handling.""" - # Create a temporary file for the database that we'll remove - temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") - temp_db.close() - sm = StatusManager(temp_db.name) - - # Remove the database file to simulate an error during operation - os.unlink(temp_db.name) - - # Now trying to connect should fail - with pytest.raises(sqlite3.OperationalError): - with sm as s: - s.check_dag_id_uniqueness('test_dag') - - def test_check_dag_id_uniqueness_empty_database(self): - """Test uniqueness check with empty database.""" - with self.sm as sm: - # Any DAG ID should be unique in empty database - assert sm.check_dag_id_uniqueness('any_dag_id') is True - - -class TestUniqueDAGIDGeneration: - """Test unique DAG ID generation functionality.""" - - def setup_method(self): - """Set up a temporary database for each test.""" - self.temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") - self.temp_db.close() - self.sm = StatusManager(self.temp_db.name) - - def teardown_method(self): - """Clean up the temporary database.""" - if os.path.exists(self.temp_db.name): - os.unlink(self.temp_db.name) - - def test_generate_unique_dag_id_first_attempt(self): - """Test successful generation on first attempt.""" - with self.sm as sm: - result = sm.generate_unique_dag_id() - - # Should be a valid format - assert "_" in result - parts = result.split("_") - assert len(parts) >= 2 - assert parts[0] in DOCKER_ADJECTIVES - assert parts[1] in DOCKER_NOUNS - - def test_generate_unique_dag_id_retry_logic(self): - """Test retry logic when first attempts fail.""" - with self.sm as sm: - # Pre-populate database with many DAGs to increase chance of collision - for adj in DOCKER_ADJECTIVES[:10]: # Use first 10 adjectives - for noun in DOCKER_NOUNS[:10]: # Use first 10 nouns - sm._create_dag_if_not_exists(f"{adj}_{noun}") - - # Generate a new unique ID - result = sm.generate_unique_dag_id() - - # Should still get a unique ID - assert sm.check_dag_id_uniqueness(result) - assert "_" in result - - @patch('maestro.server.internals.status_manager.random.choices') - def test_generate_unique_dag_id_fallback_suffix(self, mock_choices): - """Test fallback to suffix when max attempts reached.""" - mock_choices.return_value = ['a', 'b', 'c', 'd', 'e', 'f'] - - with self.sm as sm: - # Pre-populate database with ALL possible combinations - # This is impractical in reality but simulates the worst case - # Instead, we'll mock the check_dag_id_uniqueness method - original_check = sm.check_dag_id_uniqueness - call_count = 0 - - def mock_check(dag_id): - nonlocal call_count - call_count += 1 - # Return False for first 100 calls, then True - if call_count <= 100: - return False - return original_check(dag_id) - - sm.check_dag_id_uniqueness = mock_check - - result = sm.generate_unique_dag_id() - - # Should have a suffix - assert result.endswith("_abcdef") - assert call_count >= 100 # Should have tried at least 100 times - - def test_generate_unique_dag_id_valid_format(self): - """Test that generated unique DAG ID has valid format.""" - with self.sm as sm: - result = sm.generate_unique_dag_id() - - assert sm.validate_dag_id(result) - - -class TestDockerListContents: - """Test that Docker name lists contain expected content.""" - - def test_docker_adjectives_not_empty(self): - """Test that adjectives list is not empty.""" - assert len(DOCKER_ADJECTIVES) > 0 - assert all(isinstance(adj, str) for adj in DOCKER_ADJECTIVES) - - def test_docker_nouns_not_empty(self): - """Test that nouns list is not empty.""" - assert len(DOCKER_NOUNS) > 0 - assert all(isinstance(noun, str) for noun in DOCKER_NOUNS) - - def test_docker_names_valid_format(self): - """Test that all names in lists are valid for DAG IDs.""" - # Test adjectives - for adj in DOCKER_ADJECTIVES: - assert re.match(r'^[a-zA-Z0-9_-]+$', adj), f"Invalid adjective: {adj}" - - # Test nouns - for noun in DOCKER_NOUNS: - assert re.match(r'^[a-zA-Z0-9_-]+$', noun), f"Invalid noun: {noun}" diff --git a/tests/test_db_feature.py b/tests/test_db_feature.py deleted file mode 100644 index 1cc97bf..0000000 --- a/tests/test_db_feature.py +++ /dev/null @@ -1,107 +0,0 @@ -import pytest -import os -import uuid - -from maestro.server.internals.orchestrator import Orchestrator -from maestro.server.tasks.base import BaseTask -from maestro.shared.task import TaskStatus -from maestro.server.internals.status_manager import StatusManager - -# Define a dummy task for testing -class DummyPrintTask(BaseTask): - message: str - executed: bool = False - - def execute_local(self): - self.executed = True - -@pytest.fixture -def db_path(tmp_path): - return os.path.join(tmp_path, "test_maestro.db") - -@pytest.fixture -def orchestrator(db_path): - orch = Orchestrator(log_level="CRITICAL", db_path=db_path) - orch.register_task_type("print_task", DummyPrintTask) - return orch - -@pytest.fixture -def dag_filepath(tmp_path): - content = """ -dag: - tasks: - - task_id: task1 - type: print_task - message: "Hello from task1" - - task_id: task2 - type: print_task - message: "Hello from task2" - dependencies: [task1] -""" - f = tmp_path / "test_dag.yaml" - f.write_text(content) - return str(f) - -def test_db_creation(orchestrator, db_path): - with orchestrator.status_manager: - pass # Entering the context creates the DB - assert os.path.exists(db_path) - -def test_save_state(orchestrator, dag_filepath, db_path): - dag = orchestrator.load_dag_from_file(dag_filepath) - execution_id = orchestrator.run_dag_in_thread(dag) - - # Wait for execution to complete - import time - time.sleep(2) - - with StatusManager(db_path) as sm: - assert sm.get_task_status(dag.dag_id, "task1", execution_id) == "completed" - assert sm.get_task_status(dag.dag_id, "task2", execution_id) == "completed" - -def test_resume_execution(orchestrator, dag_filepath, db_path): - dag = orchestrator.load_dag_from_file(dag_filepath) - - # First, create an execution and set task1 as completed - execution_id = str(uuid.uuid4()) - with StatusManager(db_path) as sm: - # Create the execution and DAG records - sm.save_dag_definition(dag) - sm.create_dag_execution(dag.dag_id, execution_id) - sm.initialize_tasks_for_execution(dag.dag_id, execution_id, list(dag.tasks.keys())) - # Mark task1 as completed - sm.set_task_status(dag.dag_id, "task1", "completed", execution_id) - - assert not dag.tasks["task1"].executed - assert not dag.tasks["task2"].executed - - # Run the DAG with resume=True using the same execution_id - orchestrator.run_dag(dag, execution_id=execution_id, resume=True) - - assert not dag.tasks["task1"].executed # Should not re-execute - assert dag.tasks["task2"].executed # Should execute - assert dag.tasks["task1"].status == TaskStatus.COMPLETED - assert dag.tasks["task2"].status == TaskStatus.COMPLETED - -def test_reset_execution(orchestrator, dag_filepath, db_path): - dag1 = orchestrator.load_dag_from_file(dag_filepath) - execution_id1 = orchestrator.run_dag_in_thread(dag1) - - # Wait for execution to complete - import time - time.sleep(2) - - assert dag1.tasks["task1"].executed - - dag2 = orchestrator.load_dag_from_file(dag_filepath) - assert not dag2.tasks["task1"].executed # New DAG instance has fresh tasks - - execution_id2 = orchestrator.run_dag_in_thread(dag2, resume=False) - time.sleep(2) - - assert dag2.tasks["task1"].executed - with StatusManager(db_path) as sm: - # Both executions should show completed - assert sm.get_task_status(dag1.dag_id, "task1", execution_id1) == "completed" - assert sm.get_task_status(dag2.dag_id, "task1", execution_id2) == "completed" - diff --git a/tests/test_enhanced_cli.py b/tests/test_enhanced_cli.py deleted file mode 100644 index 5c580d5..0000000 --- a/tests/test_enhanced_cli.py +++ /dev/null @@ -1,441 +0,0 @@ -import pytest -import time -import uuid -from unittest.mock import Mock, patch, MagicMock -from datetime import datetime, timedelta - -from maestro.server.internals.status_manager import StatusManager -from maestro.server.internals.orchestrator import Orchestrator -from maestro.shared.dag import DAG -from maestro.server.tasks.base import BaseTask - - -class SimpleTask(BaseTask): - """Simple task for CLI testing.""" - message: str = "test" - executed: bool = False - - def execute_local(self): - self.executed = True - - -@pytest.fixture -def status_manager(tmp_path): - """Create status manager with test database.""" - db_path = tmp_path / "test_cli.db" - return StatusManager(str(db_path)) - - -@pytest.fixture -def orchestrator_with_data(tmp_path): - """Create orchestrator with test data.""" - db_path = tmp_path / "test_cli_orchestrator.db" - orchestrator = Orchestrator(log_level="CRITICAL", db_path=str(db_path)) - orchestrator.register_task_type("simple_task", SimpleTask) - return orchestrator - - -class TestStatusManagerCLIFeatures: - """Test status manager features used by CLI.""" - - def test_create_dag_execution(self, status_manager): - """Test creating DAG execution records.""" - dag_id = "test_dag" - execution_id = str(uuid.uuid4()) - - with status_manager as sm: - result = sm.create_dag_execution(dag_id, execution_id) - assert result == execution_id - - # Verify execution was created - details = sm.get_dag_execution_details(dag_id, execution_id) - assert details["execution_id"] == execution_id - assert details["status"] == "running" - assert details["started_at"] is not None - - def test_update_dag_execution_status(self, status_manager): - """Test updating DAG execution status.""" - dag_id = "test_dag" - execution_id = str(uuid.uuid4()) - - with status_manager as sm: - sm.create_dag_execution(dag_id, execution_id) - - # Update status to completed - sm.update_dag_execution_status(dag_id, execution_id, "completed") - - # Verify status update - details = sm.get_dag_execution_details(dag_id, execution_id) - assert details["status"] == "completed" - assert details["completed_at"] is not None - - def test_get_running_dags(self, status_manager): - """Test retrieving running DAGs.""" - dag_id1 = "running_dag_1" - dag_id2 = "running_dag_2" - dag_id3 = "completed_dag" - - with status_manager as sm: - # Create running DAGs - exec_id1 = sm.create_dag_execution(dag_id1, str(uuid.uuid4())) - exec_id2 = sm.create_dag_execution(dag_id2, str(uuid.uuid4())) - exec_id3 = sm.create_dag_execution(dag_id3, str(uuid.uuid4())) - - # Complete one DAG - sm.update_dag_execution_status(dag_id3, exec_id3, "completed") - - # Get running DAGs - running_dags = sm.get_running_dags() - - # Should only return running DAGs - assert len(running_dags) == 2 - running_dag_ids = [dag["dag_id"] for dag in running_dags] - assert dag_id1 in running_dag_ids - assert dag_id2 in running_dag_ids - assert dag_id3 not in running_dag_ids - - def test_get_dags_by_status(self, status_manager): - """Test retrieving DAGs by specific status.""" - with status_manager as sm: - # Create DAGs with different statuses - dag_ids = [] - for i, status in enumerate(["running", "completed", "failed", "running"]): - dag_id = f"dag_{i}" - execution_id = str(uuid.uuid4()) - sm.create_dag_execution(dag_id, execution_id) - if status != "running": - sm.update_dag_execution_status(dag_id, execution_id, status) - dag_ids.append(dag_id) - - # Test getting running DAGs - running_dags = sm.get_dags_by_status("running") - assert len(running_dags) == 2 - - # Test getting completed DAGs - completed_dags = sm.get_dags_by_status("completed") - assert len(completed_dags) == 1 - - # Test getting failed DAGs - failed_dags = sm.get_dags_by_status("failed") - assert len(failed_dags) == 1 - - def test_get_all_dags(self, status_manager): - """Test retrieving all DAGs.""" - with status_manager as sm: - # Create DAGs with different statuses - dag_count = 5 - for i in range(dag_count): - dag_id = f"dag_{i}" - execution_id = str(uuid.uuid4()) - sm.create_dag_execution(dag_id, execution_id) - if i % 2 == 0: - sm.update_dag_execution_status(dag_id, execution_id, "completed") - - # Get all DAGs - all_dags = sm.get_all_dags() - assert len(all_dags) == dag_count - - # Verify all have required fields - for dag in all_dags: - assert "dag_id" in dag - assert "execution_id" in dag - assert "status" in dag - assert "started_at" in dag - - def test_get_dag_summary(self, status_manager): - """Test getting DAG summary statistics.""" - with status_manager as sm: - # Create DAGs with different statuses - statuses = ["running", "completed", "failed", "completed", "running"] - for i, status in enumerate(statuses): - dag_id = f"dag_{i}" - execution_id = str(uuid.uuid4()) - sm.create_dag_execution(dag_id, execution_id) - if status != "running": - sm.update_dag_execution_status(dag_id, execution_id, status) - - # Get summary - summary = sm.get_dag_summary() - - # Verify summary structure - assert "total_executions" in summary - assert "unique_dags" in summary - assert "status_counts" in summary - - # Verify counts - assert summary["total_executions"] == 5 - assert summary["unique_dags"] == 5 - assert summary["status_counts"]["running"] == 2 - assert summary["status_counts"]["completed"] == 2 - assert summary["status_counts"]["failed"] == 1 - - def test_get_dag_history(self, status_manager): - """Test getting DAG execution history.""" - dag_id = "test_dag" - - with status_manager as sm: - # Create multiple executions for the same DAG - execution_ids = [] - for i in range(3): - execution_id = str(uuid.uuid4()) - sm.create_dag_execution(dag_id, execution_id) - sm.update_dag_execution_status(dag_id, execution_id, "completed") - execution_ids.append(execution_id) - - # Get history - history = sm.get_dag_history(dag_id) - - # Verify history - assert len(history) == 3 - for execution in history: - assert execution["execution_id"] in execution_ids - assert execution["status"] == "completed" - - def test_cleanup_old_executions(self, status_manager): - """Test cleaning up old execution records.""" - with status_manager as sm: - # Create some executions - dag_id = "test_dag" - execution_ids = [] - for i in range(5): - execution_id = str(uuid.uuid4()) - sm.create_dag_execution(dag_id, execution_id) - sm.update_dag_execution_status(dag_id, execution_id, "completed") - execution_ids.append(execution_id) - - # Clean up all executions (0 days to keep) - deleted_count = sm.cleanup_old_executions(0) - - # Verify cleanup - assert deleted_count == 5 - - # Verify all executions are gone - all_dags = sm.get_all_dags() - assert len(all_dags) == 0 - - def test_cancel_dag_execution(self, status_manager): - """Test cancelling DAG executions.""" - with status_manager as sm: - # Create running executions - dag_id = "test_dag" - execution_id1 = str(uuid.uuid4()) - execution_id2 = str(uuid.uuid4()) - - sm.create_dag_execution(dag_id, execution_id1) - sm.create_dag_execution(dag_id, execution_id2) - - # Cancel specific execution - success = sm.cancel_dag_execution(dag_id, execution_id1) - assert success is True - - # Verify cancellation - details = sm.get_dag_execution_details(dag_id, execution_id1) - assert details["status"] == "cancelled" - assert details["completed_at"] is not None - - # Other execution should still be running - details2 = sm.get_dag_execution_details(dag_id, execution_id2) - assert details2["status"] == "running" - - def test_log_message_and_retrieval(self, status_manager): - """Test logging messages and retrieving them.""" - with status_manager as sm: - dag_id = "test_dag" - execution_id = str(uuid.uuid4()) - task_id = "test_task" - - sm.create_dag_execution(dag_id, execution_id) - - # Log some messages - messages = [ - ("INFO", "Task started"), - ("DEBUG", "Processing data"), - ("INFO", "Task completed"), - ("ERROR", "Something went wrong") - ] - - for level, message in messages: - sm.log_message(dag_id, execution_id, task_id, level, message) - - # Retrieve logs - logs = sm.get_execution_logs(dag_id, execution_id) - - assert len(logs) == len(messages) - - # Verify messages (retrieved logs are in descending order) - retrieved_messages = [log['message'] for log in reversed(logs)] - original_messages = [msg[1] for msg in messages] - assert retrieved_messages == original_messages - - def test_get_dag_execution_details(self, status_manager): - """Test getting detailed DAG execution information.""" - with status_manager as sm: - dag_id = "test_dag" - execution_id = str(uuid.uuid4()) - - # Create execution - sm.create_dag_execution(dag_id, execution_id) - - # Add some task statuses with execution_id - sm.set_task_status(dag_id, "task1", "completed", execution_id) - sm.set_task_status(dag_id, "task2", "running", execution_id) - - # Get details - details = sm.get_dag_execution_details(dag_id, execution_id) - - # Verify details structure - assert details["execution_id"] == execution_id - assert details["status"] == "running" - assert "started_at" in details - assert "tasks" in details - - # Verify task details - task_ids = [task["task_id"] for task in details["tasks"]] - assert "task1" in task_ids - assert "task2" in task_ids - - -class TestCLIIntegration: - """Test CLI integration with orchestrator and status manager.""" - - def test_orchestrator_run_dag_in_thread_cli_integration(self, orchestrator_with_data): - """Test run_dag_in_thread for CLI integration.""" - # Create simple DAG - dag = DAG(dag_id="cli_test_dag") - task = SimpleTask(task_id="simple_task", message="Hello CLI") - dag.add_task(task) - - # Run in thread (simulating CLI async execution) - execution_id = orchestrator_with_data.run_dag_in_thread(dag) - - # Wait for completion - time.sleep(0.2) - - # Verify execution was tracked - with orchestrator_with_data.status_manager as sm: - details = sm.get_dag_execution_details(dag.dag_id, execution_id) - assert details["execution_id"] == execution_id - assert details["status"] == "completed" - - def test_dag_status_monitoring_cli_scenario(self, orchestrator_with_data): - """Test DAG status monitoring scenario for CLI.""" - dag = DAG(dag_id="monitor_dag") - task = SimpleTask(task_id="monitor_task", message="Monitor me") - dag.add_task(task) - - # Start execution - execution_id = orchestrator_with_data.run_dag_in_thread(dag) - - # Wait for completion first - time.sleep(0.3) - - # Monitor execution (simulating CLI monitor command) - use separate context - with orchestrator_with_data.status_manager as sm: - # Check final status - details = sm.get_dag_execution_details(dag.dag_id, execution_id) - assert details["status"] == "completed" - - def test_resume_functionality_cli_scenario(self, orchestrator_with_data): - """Test resume functionality for CLI.""" - dag = DAG(dag_id="resume_dag") - task1 = SimpleTask(task_id="task1", message="First task") - task2 = SimpleTask(task_id="task2", message="Second task", dependencies=["task1"]) - dag.add_task(task1) - dag.add_task(task2) - - # Simulate partial execution - with orchestrator_with_data.status_manager as sm: - sm.set_task_status(dag.dag_id, "task1", "completed") - sm.set_task_status(dag.dag_id, "task2", "pending") - - # Resume execution - execution_id = orchestrator_with_data.run_dag_in_thread(dag, resume=True) - - # Wait for completion - time.sleep(0.2) - - # Verify resume worked - with orchestrator_with_data.status_manager as sm: - details = sm.get_dag_execution_details(dag.dag_id, execution_id) - assert details["status"] == "completed" - - # task1 should not have been executed again - assert not dag.tasks["task1"].executed - # task2 should have been executed - assert dag.tasks["task2"].executed - - -class TestCLIDataStructures: - """Test data structures and formats expected by CLI.""" - - def test_dag_execution_details_format(self, status_manager): - """Test that DAG execution details have expected format for CLI.""" - with status_manager as sm: - dag_id = "format_test_dag" - execution_id = str(uuid.uuid4()) - - # Create execution with tasks - sm.create_dag_execution(dag_id, execution_id) - sm.set_task_status(dag_id, "task1", "completed", execution_id) - sm.set_task_status(dag_id, "task2", "running", execution_id) - - # Get details - details = sm.get_dag_execution_details(dag_id, execution_id) - - # Verify CLI-expected format - required_fields = ["execution_id", "status", "started_at", "completed_at", "thread_id", "pid", "tasks"] - for field in required_fields: - assert field in details - - # Verify task format - assert len(details["tasks"]) == 2 - for task in details["tasks"]: - task_fields = ["task_id", "status", "started_at", "completed_at", "thread_id"] - for field in task_fields: - assert field in task - - def test_logs_format_for_cli(self, status_manager): - """Test that logs have expected format for CLI display.""" - with status_manager as sm: - dag_id = "log_format_dag" - execution_id = str(uuid.uuid4()) - - sm.create_dag_execution(dag_id, execution_id) - - # Log some messages - sm.log_message(dag_id, execution_id, "task1", "INFO", "Test message") - - logs = sm.get_execution_logs(dag_id, execution_id) - assert len(logs) == 1 - log_entry = logs[0] - - expected_keys = ["task_id", "level", "message", "timestamp", "thread_id"] - for key in expected_keys: - assert key in log_entry - - # Verify log levels are valid - assert log_entry["level"] in ["DEBUG", "INFO", "WARNING", "ERROR"] - - def test_summary_format_for_cli(self, status_manager): - """Test that summary has expected format for CLI display.""" - with status_manager as sm: - # Create some test data - for i in range(5): - dag_id = f"summary_dag_{i}" - execution_id = str(uuid.uuid4()) - sm.create_dag_execution(dag_id, execution_id) - if i % 2 == 0: - sm.update_dag_execution_status(dag_id, execution_id, "completed") - - # Get summary - summary = sm.get_dag_summary() - - # Verify CLI-expected format - required_fields = ["total_executions", "unique_dags", "status_counts"] - for field in required_fields: - assert field in summary - - # Verify status_counts is a dictionary - assert isinstance(summary["status_counts"], dict) - assert summary["total_executions"] == 5 - assert summary["unique_dags"] == 5 diff --git a/tests/test_extended_terraform_task.py b/tests/test_extended_terraform_task.py deleted file mode 100644 index 6b69f66..0000000 --- a/tests/test_extended_terraform_task.py +++ /dev/null @@ -1,366 +0,0 @@ -import pytest -import os -import tempfile -from pathlib import Path -from unittest.mock import patch, MagicMock, call -import shutil - -# Import with proper error handling -try: - from maestro.server.tasks.extended_terraform_task import ( - ExtendedTerraformTask, - check_command_exists, - print_status, - print_success, - print_warning, - print_error, - print_header - ) -except ImportError: - pytest.skip("ExtendedTerraformTask not available", allow_module_level=True) - - -class TestExtendedTerraformTask: - - def setup_method(self): - """Setup method to ensure clean state for each test.""" - self.original_cwd = os.getcwd() - - def teardown_method(self): - """Cleanup method to restore original state.""" - os.chdir(self.original_cwd) - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_full_workflow_execution(self, mock_which, mock_subprocess): - """Test executing the full Terraform workflow.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - workflow_mode=True - ) - - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - - result = task.execute_local() - - # Verify the workflow executed successfully - assert result is not None - # Adjusted number based on: init, validate, fmt (check), fmt, plan, show, apply - assert mock_subprocess.call_count == 7 - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_single_command_execution(self, mock_which, mock_subprocess): - """Test executing a single Terraform command.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - command='init' - ) - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - - task.execute_local() - - mock_subprocess.assert_called_once() - - def test_missing_workflow_and_command(self): - """Test handling of missing workflow_mode and command.""" - with patch('maestro.server.tasks.terraform_task.shutil.which', return_value='/usr/bin/terraform'): - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.' - ) - - with pytest.raises(ValueError) as exc_info: - task.execute_local() - assert "either 'workflow_mode' must be true or a 'command' must be provided" in str(exc_info.value) - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.extended_terraform_task.os.path.exists') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_full_workflow_with_existing_terraform_dir(self, mock_which, mock_exists, mock_subprocess): - """Test full workflow when .terraform directory exists.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - workflow_mode=True - ) - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - mock_exists.return_value = True - - with patch('maestro.server.tasks.extended_terraform_task.Path.is_dir', return_value=True): - task.execute_local() - - # Should call init with -upgrade flag - assert mock_subprocess.call_count == 7 - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.extended_terraform_task.os.environ.get') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_full_workflow_with_skip_plan(self, mock_which, mock_environ, mock_subprocess): - """Test full workflow with SKIP_PLAN environment variable.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - workflow_mode=True - ) - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - mock_environ.return_value = "true" - - task.execute_local() - - # Should call fewer commands when SKIP_PLAN is true - assert mock_subprocess.call_count == 4 # init, validate, fmt (check), fmt - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.extended_terraform_task.os.chdir') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_full_workflow_directory_change(self, mock_which, mock_chdir, mock_subprocess): - """Test that the workflow changes directories correctly.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='/tmp/test', - workflow_mode=True - ) - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - - task.execute_local() - - # Should change to the working directory and back - assert mock_chdir.call_count == 2 - - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_full_workflow_exception_handling(self, mock_which, mock_subprocess): - """Test that exceptions in workflow are handled properly.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - workflow_mode=True - ) - mock_subprocess.side_effect = Exception("Test error") - - with pytest.raises(Exception) as exc_info: - task.execute_local() - assert "Test error" in str(exc_info.value) - - def test_workflow_mode_field(self): - """Test that workflow_mode field is properly set.""" - with patch('maestro.server.tasks.terraform_task.shutil.which', return_value='/usr/bin/terraform'): - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - workflow_mode=True - ) - assert task.workflow_mode is True - - task_no_workflow = ExtendedTerraformTask( - task_id='test_task2', - working_dir='.' - ) - assert task_no_workflow.workflow_mode is False - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_single_command_with_workspace(self, mock_which, mock_subprocess): - """Test single command execution with workspace.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - command='plan', - workspace='test-workspace' - ) - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - - task.execute_local() - - # Should call workspace select first, then the command - assert mock_subprocess.call_count == 2 - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_single_command_with_variables(self, mock_which, mock_subprocess): - """Test single command execution with variables.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - command='plan', - vars={'env': 'test', 'region': 'us-east-1'} - ) - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - - task.execute_local() - - mock_subprocess.assert_called_once() - - def test_print_functions(self): - """Test the print utility functions.""" - with patch('maestro.server.tasks.extended_terraform_task.logging.getLogger') as mock_logger: - mock_log = MagicMock() - mock_logger.return_value = mock_log - - print_status("Test status") - print_success("Test success") - print_warning("Test warning") - print_error("Test error") - print_header("Test header") - - # Verify logging calls were made - assert mock_log.info.call_count == 3 # status, success, header - assert mock_log.warning.call_count == 1 # warning - assert mock_log.error.call_count == 1 # error - - def test_check_command_exists(self): - """Test the check_command_exists function.""" - with patch('maestro.server.tasks.extended_terraform_task.shutil.which') as mock_which: - mock_which.return_value = '/usr/bin/terraform' - assert check_command_exists('terraform') is True - - mock_which.return_value = None - assert check_command_exists('nonexistent') is False - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.terraform_task.shutil.which') - def test_workflow_returns_value(self, mock_which, mock_subprocess): - """Test that the workflow returns the expected value.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - workflow_mode=True - ) - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - - # The _run_full_workflow method should return 1 - result = task._run_full_workflow() - assert result == 1 - - @pytest.mark.skipif( - not shutil.which("terraform") and not shutil.which("tofu"), - reason="Neither terraform nor tofu available in PATH" - ) - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - @patch('maestro.server.tasks.terraform_task.shutil.which') - @patch('maestro.server.tasks.extended_terraform_task.os.makedirs') - def test_workflow_with_relative_path(self, mock_makedirs, mock_which, mock_subprocess): - """Test workflow execution with relative working directory.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - - with tempfile.TemporaryDirectory() as temp_dir: - # Create terraform directory for the test - terraform_dir = Path(temp_dir) / 'terraform' - terraform_dir.mkdir() - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='./terraform', - workflow_mode=True, - dag_file_path=str(Path(temp_dir) / 'test.yaml') - ) - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - - task.execute_local() - - assert mock_subprocess.call_count == 7 - - -# Alternative approach: Create a separate test configuration for environments without terraform -class TestExtendedTerraformTaskMocked: - """Tests that run entirely with mocks, regardless of terraform availability.""" - - @patch('maestro.server.tasks.terraform_task.shutil.which') - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - def test_mock_full_workflow_execution(self, mock_subprocess, mock_which): - """Test executing the full Terraform workflow with complete mocking.""" - # Mock terraform being available - mock_which.return_value = '/usr/bin/terraform' - mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - workflow_mode=True - ) - - task.execute_local() - - # Verify the workflow executed successfully - assert mock_subprocess.call_count == 7 - - @patch('maestro.server.tasks.terraform_task.shutil.which') - @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') - def test_mock_terraform_not_available(self, mock_subprocess, mock_which): - """Test behavior when terraform is not available.""" - # Mock terraform NOT being available - mock_which.return_value = None - - task = ExtendedTerraformTask( - task_id='test_task', - working_dir='.', - workflow_mode=True - ) - - with pytest.raises(FileNotFoundError) as exc_info: - task.execute_local() - - assert "Neither 'tofu' nor 'terraform' command found" in str(exc_info.value) \ No newline at end of file diff --git a/tests/test_multi_executor.py b/tests/test_multi_executor.py deleted file mode 100644 index dcd959a..0000000 --- a/tests/test_multi_executor.py +++ /dev/null @@ -1,286 +0,0 @@ -import pytest -import time -from unittest.mock import Mock, patch, MagicMock - -from maestro.server.internals.orchestrator import Orchestrator -from maestro.shared.dag import DAG -from maestro.server.tasks.base import BaseTask -from maestro.shared.task import TaskStatus -from maestro.server.internals.executors.factory import ExecutorFactory -from maestro.server.internals.executors.local import LocalExecutor -from maestro.server.internals.executors.base import BaseExecutor - - -class MultiExecutorTestTask(BaseTask): - """Test task for multi-executor testing.""" - executed: bool = False - executor_used: str = None - - def execute_local(self): - self.executed = True - self.executor_used = "local" - - -class CustomExecutor(BaseExecutor): - """Custom executor for testing.""" - def __init__(self): - self.executed_tasks = [] - - def execute(self, task): - self.executed_tasks.append(task.task_id) - task.execute_local() - - -@pytest.fixture -def orchestrator_with_executors(tmp_path): - """Create orchestrator with multiple executors.""" - db_path = tmp_path / "test_multi_executor.db" - orchestrator = Orchestrator(log_level="CRITICAL", db_path=str(db_path)) - orchestrator.register_task_type("test_task", MultiExecutorTestTask) - return orchestrator - - -@pytest.fixture -def dag_with_different_executors(): - """Create DAG with tasks using different executors.""" - dag = DAG(dag_id="multi_executor_dag") - - # Tasks with different executors - task1 = MultiExecutorTestTask(task_id="local_task", executor="local") - task2 = MultiExecutorTestTask(task_id="ssh_task", executor="ssh") - task3 = MultiExecutorTestTask(task_id="docker_task", executor="docker") - - dag.add_task(task1) - dag.add_task(task2) - dag.add_task(task3) - - return dag - - -class TestMultiExecutorSupport: - """Test multi-executor support functionality.""" - - def test_executor_factory_default_executors(self): - """Test that ExecutorFactory has default executors registered.""" - factory = ExecutorFactory() - - # Test getting default executors - local_executor = factory.get_executor("local") - assert isinstance(local_executor, LocalExecutor) - - # Test that unknown executor raises error - with pytest.raises(ValueError, match="Unknown executor: unknown"): - factory.get_executor("unknown") - - def test_executor_factory_register_custom_executor(self): - """Test registering custom executor.""" - factory = ExecutorFactory() - - # Register custom executor - factory.register_executor("custom", CustomExecutor) - - # Test getting custom executor - custom_executor = factory.get_executor("custom") - assert isinstance(custom_executor, CustomExecutor) - - def test_task_executor_field_default(self): - """Test that task executor field defaults to 'local'.""" - task = MultiExecutorTestTask(task_id="test_task") - assert task.executor == "local" - - def test_task_executor_field_custom(self): - """Test setting custom executor for task.""" - task = MultiExecutorTestTask(task_id="test_task", executor="ssh") - assert task.executor == "ssh" - - def test_dag_execution_with_different_executors(self, orchestrator_with_executors, dag_with_different_executors): - """Test DAG execution with tasks using different executors.""" - # Create custom executors for testing - class TestSshExecutor(BaseExecutor): - def execute(self, task): - task.execute_local() - task.executor_used = "ssh" - - class TestDockerExecutor(BaseExecutor): - def execute(self, task): - task.execute_local() - task.executor_used = "docker" - - # Register custom executors - orchestrator_with_executors.executor_factory.register_executor("ssh", TestSshExecutor) - orchestrator_with_executors.executor_factory.register_executor("docker", TestDockerExecutor) - - # Run the DAG - orchestrator_with_executors.run_dag(dag_with_different_executors) - - # Verify all tasks were executed - assert dag_with_different_executors.tasks["local_task"].executed - assert dag_with_different_executors.tasks["ssh_task"].executed - assert dag_with_different_executors.tasks["docker_task"].executed - - # Verify correct executors were used - assert dag_with_different_executors.tasks["local_task"].executor_used == "local" - assert dag_with_different_executors.tasks["ssh_task"].executor_used == "ssh" - assert dag_with_different_executors.tasks["docker_task"].executor_used == "docker" - - def test_concurrent_execution_with_different_executors(self, orchestrator_with_executors): - """Test concurrent execution with different executors.""" - dag = DAG(dag_id="concurrent_multi_executor_dag") - - # Create tasks with different executors - task1 = MultiExecutorTestTask(task_id="local_task1", executor="local") - task2 = MultiExecutorTestTask(task_id="local_task2", executor="local") - - dag.add_task(task1) - dag.add_task(task2) - - # Run concurrently - execution_id = orchestrator_with_executors.run_dag_in_thread(dag) - - # Wait for completion - time.sleep(0.2) - - # Verify execution completed - with orchestrator_with_executors.status_manager as sm: - details = sm.get_dag_execution_details(dag.dag_id, execution_id) - assert details["status"] == "completed" - - # Verify tasks were executed - assert dag.tasks["local_task1"].executed - assert dag.tasks["local_task2"].executed - - def test_executor_failure_handling(self, orchestrator_with_executors): - """Test handling of executor failures.""" - dag = DAG(dag_id="executor_failure_dag") - - # Create task with non-existent executor - task = MultiExecutorTestTask(task_id="invalid_executor_task", executor="non_existent") - dag.add_task(task) - - # Run should fail with unknown executor - with pytest.raises(Exception, match="Unknown executor: non_existent"): - orchestrator_with_executors.run_dag(dag) - - def test_custom_executor_registration_and_usage(self, orchestrator_with_executors): - """Test registering and using custom executor.""" - # Register custom executor - orchestrator_with_executors.executor_factory.register_executor("custom", CustomExecutor) - - # Create DAG with custom executor task - dag = DAG(dag_id="custom_executor_dag") - task = MultiExecutorTestTask(task_id="custom_task", executor="custom") - dag.add_task(task) - - # Run DAG - orchestrator_with_executors.run_dag(dag) - - # Verify task was executed - assert task.executed - - def test_mixed_executor_dependencies(self, orchestrator_with_executors): - """Test DAG with tasks using different executors and dependencies.""" - dag = DAG(dag_id="mixed_executor_dependencies_dag") - - # Create tasks with dependencies and different executors - task1 = MultiExecutorTestTask(task_id="local_root", executor="local") - task2 = MultiExecutorTestTask(task_id="local_dependent", executor="local", dependencies=["local_root"]) - - dag.add_task(task1) - dag.add_task(task2) - - # Run DAG - orchestrator_with_executors.run_dag(dag) - - # Verify execution order and completion - assert task1.executed - assert task2.executed - assert task1.status == TaskStatus.COMPLETED - assert task2.status == TaskStatus.COMPLETED - - def test_executor_factory_thread_safety(self): - """Test that ExecutorFactory is thread-safe.""" - from concurrent.futures import ThreadPoolExecutor as TPE - - factory = ExecutorFactory() - - def get_executor(name): - return factory.get_executor(name) - - # Test concurrent access to executor factory - with TPE(max_workers=5) as executor: - futures = [] - for _ in range(10): - future = executor.submit(get_executor, "local") - futures.append(future) - - # All should complete without errors - results = [f.result() for f in futures] - - # All should return LocalExecutor instances - for result in results: - assert isinstance(result, LocalExecutor) - - -class TestExecutorIntegration: - """Test executor integration with the orchestrator.""" - - def test_orchestrator_uses_correct_executor(self, orchestrator_with_executors): - """Test that orchestrator uses the correct executor for each task.""" - dag = DAG(dag_id="executor_integration_dag") - - # Create tasks with specific executors - local_task = MultiExecutorTestTask(task_id="local_task", executor="local") - dag.add_task(local_task) - - # Mock the executor factory to track calls - with patch.object(orchestrator_with_executors.executor_factory, 'get_executor', wraps=orchestrator_with_executors.executor_factory.get_executor) as mock_get_executor: - orchestrator_with_executors.run_dag(dag) - - # Verify get_executor was called with correct executor name - mock_get_executor.assert_called_with("local") - - def test_executor_context_isolation(self, orchestrator_with_executors): - """Test that different executors don't interfere with each other.""" - # Register two custom executors - executor1 = CustomExecutor() - executor2 = CustomExecutor() - - orchestrator_with_executors.executor_factory.register_executor("custom1", lambda: executor1) - orchestrator_with_executors.executor_factory.register_executor("custom2", lambda: executor2) - - dag = DAG(dag_id="executor_isolation_dag") - - task1 = MultiExecutorTestTask(task_id="task1", executor="custom1") - task2 = MultiExecutorTestTask(task_id="task2", executor="custom2") - - dag.add_task(task1) - dag.add_task(task2) - - # Run DAG - orchestrator_with_executors.run_dag(dag) - - # Verify each executor only executed its own task - assert "task1" in executor1.executed_tasks - assert "task1" not in executor2.executed_tasks - assert "task2" in executor2.executed_tasks - assert "task2" not in executor1.executed_tasks - - def test_executor_error_propagation(self, orchestrator_with_executors): - """Test that executor errors are properly propagated.""" - # Create a failing executor - class FailingExecutor(BaseExecutor): - def execute(self, task): - raise Exception("Executor failed") - - orchestrator_with_executors.executor_factory.register_executor("failing", FailingExecutor) - - dag = DAG(dag_id="failing_executor_dag") - task = MultiExecutorTestTask(task_id="failing_task", executor="failing") - dag.add_task(task) - - # Run should fail with executor error - with pytest.raises(Exception, match="Task failing_task failed: Executor failed"): - orchestrator_with_executors.run_dag(dag) - - # Verify task status is marked as failed - assert task.status == TaskStatus.FAILED diff --git a/tests/test_orchestrator_dagloader.py b/tests/test_orchestrator_dagloader.py deleted file mode 100644 index 7ea5715..0000000 --- a/tests/test_orchestrator_dagloader.py +++ /dev/null @@ -1,266 +0,0 @@ -import pytest -from datetime import datetime - -from maestro.server.internals.orchestrator import Orchestrator -from maestro.shared.dag import DAG -from maestro.server.tasks.base import BaseTask -from maestro.shared.task import TaskStatus -from maestro.server.internals.executors.local import LocalExecutor - -# Define a dummy task for testing -class DummyPrintTask(BaseTask): - message: str - executed: bool = False - - def execute_local(self): - self.executed = True - print(f"DummyPrintTask {self.task_id}: {self.message}") - - -@pytest.fixture -def orchestrator(): - orch = Orchestrator(log_level="CRITICAL") # Suppress logging during tests - orch.register_task_type("print_task", DummyPrintTask) - return orch - -@pytest.fixture -def dag_filepath(tmp_path): - content = """ -dag: - tasks: - - task_id: task1 - type: print_task - message: "Hello from task1" - - task_id: task2 - type: print_task - message: "Hello from task2" - dependencies: [task1] -""" - f = tmp_path / "test_dag.yaml" - f.write_text(content) - return str(f) - -@pytest.fixture -def invalid_dag_filepath(tmp_path): - content = """ -dag: - tasks: - - task_id: task1 - type: non_existent_task - message: "Hello from task1" -""" - f = tmp_path / "invalid_dag.yaml" - f.write_text(content) - return str(f) - -@pytest.fixture -def dag_with_start_time_filepath(tmp_path): - content = """ -dag: - name: "scheduled_dag" - start_time: "2024-01-01T09:00:00" - tasks: - - task_id: task1 - type: print_task - message: "Hello from scheduled task1" - - task_id: task2 - type: print_task - message: "Hello from scheduled task2" - dependencies: [task1] -""" - f = tmp_path / "scheduled_dag.yaml" - f.write_text(content) - return str(f) - - -@pytest.fixture -def dag_with_cron_schedule_filepath(tmp_path): - content = """ -dag: - name: "cron_scheduled_dag" - cron_schedule: "0 9 * * *" # Every day at 9:00 AM - tasks: - - task_id: task1 - type: print_task - message: "Hello from scheduled task1" - - task_id: task2 - type: print_task - message: "Hello from scheduled task2" - dependencies: [task1] -""" - f = tmp_path / "cron_scheduled_dag.yaml" - f.write_text(content) - return str(f) - - -@pytest.fixture -def dag_with_invalid_start_time_filepath(tmp_path): - content = """ -dag: - name: "invalid_scheduled_dag" - start_time: "invalid-date-format" - tasks: - - task_id: task1 - type: print_task - message: "Hello from task1" -""" - f = tmp_path / "invalid_scheduled_dag.yaml" - f.write_text(content) - return str(f) - -def test_orchestrator_load_dag_from_file(orchestrator, dag_filepath): - dag = orchestrator.load_dag_from_file(dag_filepath) - assert isinstance(dag, DAG) - assert "task1" in dag.tasks - assert "task2" in dag.tasks - assert dag.tasks["task2"].dependencies == ["task1"] - -def test_orchestrator_load_invalid_dag(orchestrator, invalid_dag_filepath): - with pytest.raises(ValueError, match="Unknown task type: non_existent_task"): - orchestrator.load_dag_from_file(invalid_dag_filepath) - -def test_orchestrator_run_dag(orchestrator, dag_filepath): - dag = orchestrator.load_dag_from_file(dag_filepath) - - orchestrator.run_dag(dag) - - assert dag.tasks["task1"].status == TaskStatus.COMPLETED - assert dag.tasks["task2"].status == TaskStatus.COMPLETED - assert dag.tasks["task1"].executed # Check if execute was called - assert dag.tasks["task2"].executed # Check if execute was called - -def test_orchestrator_run_dag_fail_fast(orchestrator, tmp_path): - # Create a DAG with a failing task - content = """ -dag: - tasks: - - task_id: failing_task - type: failing_task_type - - task_id: subsequent_task - type: print_task - message: "This should not run" - dependencies: [failing_task] -""" - f = tmp_path / "failing_dag.yaml" - f.write_text(content) - - class FailingTask(BaseTask): - def execute_local(self): - raise Exception("Simulated task failure") - - orchestrator.register_task_type("failing_task_type", FailingTask) - - dag = orchestrator.load_dag_from_file(str(f)) - - with pytest.raises(Exception, match="Task failing_task failed: Simulated task failure"): - orchestrator.run_dag(dag, fail_fast=True) - - assert dag.tasks["failing_task"].status == TaskStatus.FAILED - assert dag.tasks["subsequent_task"].status == TaskStatus.PENDING # Should not have run - -def test_orchestrator_run_dag_no_fail_fast(orchestrator, tmp_path): - # Create a DAG with a failing task - content = """ -dag: - tasks: - - task_id: failing_task - type: failing_task_type - - task_id: subsequent_task - type: print_task - message: "This should run" - dependencies: [failing_task] -""" - f = tmp_path / "no_fail_fast_dag.yaml" - f.write_text(content) - - class FailingTask(BaseTask): - def execute_local(self): - raise Exception("Simulated task failure") - - orchestrator.register_task_type("failing_task_type", FailingTask) - - dag = orchestrator.load_dag_from_file(str(f)) - - orchestrator.run_dag(dag, fail_fast=False) - - assert dag.tasks["failing_task"].status == TaskStatus.FAILED - assert dag.tasks["subsequent_task"].status == TaskStatus.SKIPPED - - def test_orchestrator_run_dag_with_skipped_task_marks_dag_as_failed(orchestrator, tmp_path): - # Create a DAG with a failing task, which should cause the next to be skipped - content = """ - dag: - tasks: - - task_id: failing_task - type: failing_task_type - - task_id: subsequent_task - type: print_task - message: "This should be skipped" - dependencies: [failing_task] - """ - f = tmp_path / "dag_with_skipped.yaml" - f.write_text(content) - - class FailingTask(BaseTask): - def execute_local(self): - raise Exception("Simulated task failure") - - orchestrator.register_task_type("failing_task_type", FailingTask) - - dag = orchestrator.load_dag_from_file(str(f)) - - # Use run_dag_in_thread to get the final status - execution_id = orchestrator.run_dag_in_thread(dag, fail_fast=False) - - # Wait for the DAG to finish - time.sleep(1) # Adjust as needed - - with orchestrator.status_manager as sm: - dag_execution_details = sm.get_dag_execution_details(dag.dag_id, execution_id) - assert dag_execution_details["status"] == "failed" - -def test_orchestrator_load_dag_with_start_time(orchestrator, dag_with_start_time_filepath): - """Test loading a DAG with start_time parameter from YAML file.""" - dag = orchestrator.load_dag_from_file(dag_with_start_time_filepath) - - assert isinstance(dag, DAG) - assert dag.start_time is not None - assert dag.start_time == datetime(2024, 1, 1, 9, 0, 0) - assert "task1" in dag.tasks - assert "task2" in dag.tasks - assert dag.tasks["task2"].dependencies == ["task1"] - -def test_orchestrator_load_dag_with_invalid_start_time(orchestrator, dag_with_invalid_start_time_filepath): - """Test loading a DAG with invalid start_time format from YAML file.""" - with pytest.raises(ValueError, match="Invalid start_time format"): - orchestrator.load_dag_from_file(dag_with_invalid_start_time_filepath) - -def test_orchestrator_load_dag_without_start_time(orchestrator, dag_filepath): - """Test loading a DAG without start_time parameter from YAML file.""" - dag = orchestrator.load_dag_from_file(dag_filepath) - - assert isinstance(dag, DAG) - assert dag.start_time is None - assert dag.is_ready_to_start() # Should be ready to start immediately - assert dag.time_until_start() is None - -def test_orchestrator_load_dag_with_cron_schedule(orchestrator, dag_with_cron_schedule_filepath): - """Test loading a DAG with cron schedule from YAML file.""" - dag = orchestrator.load_dag_from_file(dag_with_cron_schedule_filepath) - - assert isinstance(dag, DAG) - assert dag.cron_schedule == "0 9 * * *" - assert dag.start_time is None - assert "task1" in dag.tasks - assert "task2" in dag.tasks - assert dag.tasks["task2"].dependencies == ["task1"] - - # Test schedule description - assert "Cron schedule" in dag.get_schedule_description() - - # Test next run time - next_run = dag.get_next_run_time() - assert next_run is not None - assert next_run.hour == 9 - assert next_run.minute == 0 - diff --git a/tests/test_print_task.py b/tests/test_print_task.py new file mode 100644 index 0000000..1cf6a90 --- /dev/null +++ b/tests/test_print_task.py @@ -0,0 +1,50 @@ +from unittest.mock import MagicMock, patch + +import pytest + +from maestro.server.internals.status_manager import StatusManager +from maestro.server.tasks.print_task import PrintTask + + +# --- FIXTURE per StatusManager mockato +@pytest.fixture +def mock_status_manager(): + mock = MagicMock() + with patch.object(StatusManager, "get_instance", return_value=mock): + yield mock + + +# --- TEST 1: La PrintTask stampa correttamente +def test_print_task_prints_message(mock_status_manager): + task = PrintTask( + task_id="T1", dag_id="DAGX", execution_id="EX1", message="Hello World" + ) + + with patch("builtins.print") as mock_print: + task.execute_local() + mock_print.assert_called_with("[PrintTask] Hello World", flush=True) + + +# --- TEST 2: Viene chiamato add_log con i parametri corretti +def test_print_task_adds_log_to_status_manager(mock_status_manager): + task = PrintTask(task_id="T1", dag_id="DAGX", execution_id="EX1", message="ABC") + + with patch("builtins.print"): + task.execute_local() + + mock_status_manager.add_log.assert_called_once_with( + dag_id="DAGX", + execution_id="EX1", + task_id="T1", + message="[PrintTask] ABC", + level="INFO", + ) + + +# --- TEST 3: message vuoto +def test_print_task_empty_message(mock_status_manager): + task = PrintTask(task_id="T1", dag_id="DAGX", execution_id="EX1", message="") + + with patch("builtins.print") as mock_print: + task.execute_local() + mock_print.assert_called_with("[PrintTask] ", flush=True) diff --git a/tests/test_python_task.py b/tests/test_python_task.py new file mode 100644 index 0000000..39725cc --- /dev/null +++ b/tests/test_python_task.py @@ -0,0 +1,151 @@ +# tests/test_python_task.py + +from unittest.mock import MagicMock, patch + +import pytest + +from maestro.server.tasks.python_task import PythonTask + + +# FIXTURE: finto StatusManager +@pytest.fixture +def fake_status_manager(): + """ + Restituisce un finto StatusManager con metodi + add_log() e set_task_output() tracciabili. + """ + fake = MagicMock() + fake.add_log = MagicMock() + fake.set_task_output = MagicMock() + return fake + + +@pytest.fixture +def patched_status_manager(fake_status_manager): + with patch( + "maestro.server.tasks.python_task.StatusManager.get_instance", + return_value=fake_status_manager, + ): + yield fake_status_manager + + +# TEST 1 — Cattura del print (logging INFO) +def test_python_task_captures_print(patched_status_manager): + task = PythonTask( + task_id="t1", dag_id="dagX", execution_id="exec1", code="print('ciao')" + ) + + task.execute_local() + + patched_status_manager.add_log.assert_any_call( + dag_id="dagX", + execution_id="exec1", + task_id="t1", + level="INFO", + message="[PythonTask] ciao", + ) + + +# TEST 2 — Esecuzione inline + cattura output di print() +def test_python_task_captures_task_output(patched_status_manager): + task = PythonTask( + task_id="t1", dag_id="dagX", execution_id="exec1", code="task_output = 123" + ) + + task.execute_local() + + patched_status_manager.set_task_output.assert_called_once_with( + dag_id="dagX", task_id="t1", execution_id="exec1", output=123 + ) + + +# TEST 3 — Esecuzione da file + + +def test_python_task_executes_script_file(patched_status_manager, tmp_path): + script_path = tmp_path / "myscript.py" + script_path.write_text("task_output = 99") + + task = PythonTask( + task_id="t1", dag_id="dagX", execution_id="exec1", script_path=str(script_path) + ) + + task.execute_local() + + patched_status_manager.set_task_output.assert_called_once_with( + dag_id="dagX", task_id="t1", execution_id="exec1", output=99 + ) + + +# TEST 4 — Priorità dell’output + + +# 4.1 Caso task_output +def test_python_task_priority_task_output(patched_status_manager): + task = PythonTask( + task_id="t1", + dag_id="d", + execution_id="e", + code="task_output = 'A'; __output__='B'; result='C'; output='D'", + ) + task.execute_local() + + patched_status_manager.set_task_output.assert_called_once_with( + dag_id="d", task_id="t1", execution_id="e", output="A" + ) + + +# 4.2 Caso __output__ +def test_python_task_priority_dunder_output(patched_status_manager): + task = PythonTask( + task_id="t1", + dag_id="d", + execution_id="e", + code="__output__='B'; result='C'; output='D'", + ) + task.execute_local() + + patched_status_manager.set_task_output.assert_called_once_with( + dag_id="d", task_id="t1", execution_id="e", output="B" + ) + + +# 4.3 Caso result +def test_python_task_priority_result(patched_status_manager): + task = PythonTask( + task_id="t1", dag_id="d", execution_id="e", code="result='C'; output='D'" + ) + task.execute_local() + + patched_status_manager.set_task_output.assert_called_once_with( + dag_id="d", task_id="t1", execution_id="e", output="C" + ) + + +# 4.4 Caso output +def test_python_task_priority_output(patched_status_manager): + task = PythonTask(task_id="t1", dag_id="d", execution_id="e", code="output='D'") + task.execute_local() + + patched_status_manager.set_task_output.assert_called_once_with( + dag_id="d", task_id="t1", execution_id="e", output="D" + ) + + +# TEST 5 — Gestione eccezioni +def test_python_task_logs_and_raises_exceptions(patched_status_manager): + task = PythonTask( + task_id="t1", dag_id="d", execution_id="e", code="raise ValueError('BOOM')" + ) + + with pytest.raises(ValueError): + task.execute_local() + + # Controllo che abbia loggato almeno un errore + error_logs = [ + call + for call in patched_status_manager.add_log.call_args_list + if call.kwargs.get("level") == "ERROR" + ] + + assert len(error_logs) > 0 diff --git a/tests/test_server.py b/tests/test_server.py deleted file mode 100644 index bcd0ab1..0000000 --- a/tests/test_server.py +++ /dev/null @@ -1,335 +0,0 @@ - -import pytest -from fastapi.testclient import TestClient -from unittest.mock import patch, MagicMock -import os -import sys -from datetime import datetime -from contextlib import asynccontextmanager - -# Add the src directory to the Python path -sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), '../src'))) - -from maestro.server.app import app, lifespan -from maestro.shared.dag import DAG -from maestro.shared.task import Task - -# Create a concrete implementation of the abstract Task class for testing purposes -class ConcreteTask(Task): - def execute_local(self, **kwargs): - # This method is abstract in the base class, so we provide a minimal implementation. - print(f"Executing task {self.name} with action: {self.action}") - return f"Output of {self.name}" - -# Sample DAG for testing -@pytest.fixture -def sample_dag(): - task1 = ConcreteTask(task_id="task1", name="Task 1", action="echo 'Task 1'", dependencies=[]) - task2 = ConcreteTask(task_id="task2", name="Task 2", action="echo 'Task 2'", dependencies=["task1"]) - dag = DAG(dag_id="sample_dag") - dag.add_task(task1) - dag.add_task(task2) - return dag - -@pytest.fixture -def client(sample_dag): - # This fixture provides a test client for the FastAPI app. - # It patches the global `orchestrator` object that is used by the API endpoints. - - # 1. Create a mock for the Orchestrator. - mock_orchestrator = MagicMock() - mock_orchestrator.load_dag_from_file.return_value = sample_dag - mock_orchestrator.run_dag_in_thread.return_value = "test_execution_id" - - # 2. The original `lifespan` function creates a real Orchestrator. We need to prevent that. - # We replace the app's lifespan with a dummy one for the tests. - @asynccontextmanager - async def mock_lifespan(app): - # This context manager does nothing, preventing the real orchestrator creation. - yield - - app.router.lifespan_context = mock_lifespan - - # 3. Patch the global `orchestrator` variable in the `app` module with our mock. - # This is the instance that the endpoint functions will actually use. - with patch('maestro.server.app.orchestrator', mock_orchestrator): - with TestClient(app) as test_client: - yield test_client - - -def test_root(client): - response = client.get("/") - assert response.status_code == 200 - json_response = response.json() - assert json_response["message"] == "Maestro API Server" - assert json_response["status"] == "running" - -def test_submit_dag_success(client, sample_dag): - with patch('maestro.server.app.orchestrator.load_dag_from_file', - return_value=sample_dag): - with patch('maestro.server.app.orchestrator.run_dag_in_thread', - return_value="test_execution_id") as mock_run: - response = client.post("/dags/submit", - json={"dag_file_path": "path/to/dag.yaml", - "resume": False, - "fail_fast": True}) - - assert response.status_code == 200 - json_response = response.json() - assert json_response["dag_id"] == "sample_dag" - assert json_response["execution_id"] == "test_execution_id" - assert json_response["status"] == "submitted" - assert "submitted_at" in json_response - mock_run.assert_called_once_with(dag=sample_dag, - resume=False, - fail_fast=True) - -def test_submit_dag_file_not_found(client): - with patch('maestro.server.app.orchestrator.load_dag_from_file', side_effect=FileNotFoundError("DAG file not found")): - response = client.post("/dags/submit", json={"dag_file_path": "non_existent.yaml"}) - assert response.status_code == 400 - assert "DAG file not found" in response.json()["detail"] - -def test_get_dag_status_success(client): - mock_status_manager = MagicMock() - mock_status_manager.get_dag_execution_details.return_value = { - "execution_id": "test_execution_id", - "status": "running", - "started_at": datetime.now().isoformat(), - "completed_at": None, - "thread_id": "thread_123", - "tasks": [] - } - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/sample_dag/status?execution_id=test_execution_id") - assert response.status_code == 200 - json_response = response.json() - assert json_response["dag_id"] == "sample_dag" - assert json_response["execution_id"] == "test_execution_id" - assert json_response["status"] == "running" - -def test_get_dag_status_server_error(client): - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): - response = client.get("/dags/sample_dag/status") - assert response.status_code == 500 - assert "DB error" in response.json()["detail"] - -def test_get_dag_status_not_found(client): - mock_status_manager = MagicMock() - mock_status_manager.get_dag_execution_details.return_value = None - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/unknown_dag/status") - assert response.status_code == 404 - assert "DAG execution not found" in response.json()["detail"] - -def test_get_dag_logs_success(client): - mock_logs = [ - {"task_id": "task1", "level": "INFO", "message": "Task 1 started", "timestamp": datetime.now().isoformat(), "thread_id": "thread_123"} - ] - mock_status_manager = MagicMock() - mock_status_manager.get_execution_logs.return_value = mock_logs - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/sample_dag/logs") - assert response.status_code == 200 - json_response = response.json() - assert json_response["dag_id"] == "sample_dag" - assert len(json_response["logs"]) == 1 - assert json_response["logs"][0]["task_id"] == "task1" - -import asyncio - -def test_get_dag_logs_with_filters(client): - mock_logs = [ - {"task_id": "task1", "level": "INFO", "message": "Task 1 started", "timestamp": datetime.now().isoformat(), "thread_id": "thread_123"}, - {"task_id": "task2", "level": "DEBUG", "message": "Task 2 debug", "timestamp": datetime.now().isoformat(), "thread_id": "thread_123"} - ] - mock_status_manager = MagicMock() - mock_status_manager.get_execution_logs.return_value = mock_logs - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/sample_dag/logs?task_filter=task1&level_filter=INFO") - assert response.status_code == 200 - json_response = response.json() - assert len(json_response["logs"]) == 1 - assert json_response["logs"][0]["task_id"] == "task1" - assert json_response["logs"][0]["level"] == "INFO" - - -@pytest.mark.asyncio -async def test_stream_dag_logs(client): - mock_status_manager = MagicMock() - logs_queue = asyncio.Queue() - - async def mock_get_logs(*args, **kwargs): - try: - log = await asyncio.wait_for(logs_queue.get(), timeout=1.0) - return [log] - except asyncio.TimeoutError: - return [] - - mock_status_manager.get_execution_logs.side_effect = mock_get_logs - - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/sample_dag/logs/stream") - assert response.status_code == 200 - - # Simulate adding a log entry - log_entry = {"task_id": "task1", "level": "INFO", "message": "A new log", "timestamp": datetime.now().isoformat(), "thread_id": "thread_123"} - await logs_queue.put(log_entry) - - # The client-side implementation to read from the stream would be more complex. - # For this test, we are mainly ensuring the endpoint can be hit and returns a streaming response. - # A more thorough test would involve a client that can handle server-sent events. - -def test_get_dag_logs_server_error(client): - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): - response = client.get("/dags/sample_dag/logs") - assert response.status_code == 500 - assert "DB error" in response.json()["detail"] - - -def test_get_running_dags_empty(client): - mock_status_manager = MagicMock() - mock_status_manager.get_running_dags.return_value = [] - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/running") - assert response.status_code == 200 - json_response = response.json() - assert json_response["count"] == 0 - assert json_response["running_dags"] == [] - -def test_get_running_dags(client): - running_dags = [{"dag_id": "dag1", "execution_id": "exec1"}] - mock_status_manager = MagicMock() - mock_status_manager.get_running_dags.return_value = running_dags - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/running") - assert response.status_code == 200 - json_response = response.json() - assert json_response["count"] == 1 - assert json_response["running_dags"][0]["dag_id"] == "dag1" - -def test_get_running_dags_server_error(client): - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): - response = client.get("/dags/running") - assert response.status_code == 500 - assert "DB error" in response.json()["detail"] - -def test_list_dags_all(client): - all_dags = [{"dag_id": "dag1", "status": "completed"}, {"dag_id": "dag2", "status": "failed"}] - mock_status_manager = MagicMock() - mock_status_manager.get_all_dags.return_value = all_dags - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/list") - assert response.status_code == 200 - json_response = response.json() - assert json_response["count"] == 2 - assert json_response["title"] == "All DAGs" - -def test_list_dags_with_status_filter(client): - completed_dags = [{"dag_id": "dag1", "status": "completed"}] - mock_status_manager = MagicMock() - mock_status_manager.get_dags_by_status.return_value = completed_dags - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.get("/dags/list?status=completed") - assert response.status_code == 200 - json_response = response.json() - assert json_response["count"] == 1 - assert json_response["title"] == "DAGs with status: completed" - mock_status_manager.get_dags_by_status.assert_called_once_with("completed") - -def test_list_dags_server_error(client): - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): - response = client.get("/dags/list") - assert response.status_code == 500 - assert "DB error" in response.json()["detail"] - -def test_cancel_dag_success(client): - mock_status_manager = MagicMock() - mock_status_manager.cancel_dag_execution.return_value = True - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.post("/dags/sample_dag/cancel?execution_id=exec1") - assert response.status_code == 200 - assert response.json()["success"] is True - mock_status_manager.cancel_dag_execution.assert_called_once_with("sample_dag", "exec1") - -def test_cancel_dag_not_found(client): - mock_status_manager = MagicMock() - mock_status_manager.cancel_dag_execution.return_value = False - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.post("/dags/sample_dag/cancel") - assert response.status_code == 200 - assert response.json()["success"] is False - -def test_cancel_dag_server_error(client): - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): - response = client.post("/dags/sample_dag/cancel") - assert response.status_code == 500 - assert "DB error" in response.json()["detail"] - -def test_validate_dag_success(client, sample_dag): - with patch('maestro.server.app.orchestrator.load_dag_from_file', return_value=sample_dag): - response = client.post("/dags/validate", json={"dag_file_path": "path/to/dag.yaml"}) - assert response.status_code == 200 - json_response = response.json() - assert json_response["valid"] is True - assert json_response["dag_id"] == "sample_dag" - assert json_response["total_tasks"] == 2 - assert json_response["execution_order"] == ["task1", "task2"] - -def test_validate_dag_invalid(client): - with patch('maestro.server.app.orchestrator.load_dag_from_file', side_effect=Exception("Invalid DAG")): - response = client.post("/dags/validate", json={"dag_file_path": "path/to/invalid_dag.yaml"}) - assert response.status_code == 200 - json_response = response.json() - assert json_response["valid"] is False - assert "Invalid DAG" in json_response["error"] - -def test_validate_dag_missing_path(client): - response = client.post("/dags/validate", json={}) - assert response.status_code == 200 - json_response = response.json() - assert json_response["valid"] is False - assert "dag_file_path is required" in json_response["error"] - -def test_cleanup_old_executions(client): - mock_status_manager = MagicMock() - mock_status_manager.cleanup_old_executions.return_value = 5 - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): - response = client.delete("/dags/cleanup?days=15") - assert response.status_code == 200 - json_response = response.json() - assert json_response["deleted_count"] == 5 - assert "Deleted 5 execution records older than 15 days" in json_response["message"] - mock_status_manager.cleanup_old_executions.assert_called_once_with(15) - -def test_cleanup_old_executions_server_error(client): - with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): - response = client.delete("/dags/cleanup") - assert response.status_code == 500 - assert "DB error" in response.json()["detail"] - -def test_main_start_server(): - with patch('maestro.server.app.uvicorn.run') as mock_run: - with patch('argparse.ArgumentParser') as mock_argparse: - mock_argparse.return_value.parse_args.return_value = MagicMock(host="127.0.0.1", port=8080, log_level="debug") - from maestro.server.app import main - main() - mock_run.assert_called_once() - - -@pytest.mark.asyncio -async def test_lifespan(monkeypatch): - mock_orchestrator = MagicMock() - monkeypatch.setattr("maestro.server.app.Orchestrator", lambda **kwargs: mock_orchestrator) - - # Import the app module to access the orchestrator variable - from maestro.server import app as app_module - - async with lifespan(app): - # Access the orchestrator from the app module, not as a global in test scope - assert app_module.orchestrator is not None - - mock_orchestrator.executor.shutdown.assert_called_once_with(wait=True) - - - diff --git a/tests/test_status_manager.py b/tests/test_status_manager.py deleted file mode 100644 index 1bb29d6..0000000 --- a/tests/test_status_manager.py +++ /dev/null @@ -1,350 +0,0 @@ -import pytest -import threading -import uuid -from maestro.server.internals.status_manager import StatusManager -from maestro.server.internals.models import DagORM -from datetime import datetime, timedelta - -@pytest.fixture -def status_manager(): - """Create a fresh in-memory StatusManager for each test.""" - # Use a unique database path for each test to avoid shared state - import tempfile - import os - temp_file = tempfile.NamedTemporaryFile(delete=False) - temp_file.close() - - sm = StatusManager(temp_file.name) - yield sm - - # Clean up the temporary file - try: - os.unlink(temp_file.name) - except (FileNotFoundError, PermissionError): - pass - - -def test_set_and_get_task_status(status_manager): - dag_id = "dag_123" - task_id = "task_1" - - with status_manager as sm: - # Set status without execution_id (should create default execution) - sm.set_task_status(dag_id, task_id, "completed") - status = sm.get_task_status(dag_id, task_id) - assert status == "completed" - - # Set status with specific execution_id - execution_id = "exec_456" - # Create execution first - sm.create_dag_execution(dag_id, execution_id) - sm.set_task_status(dag_id, task_id, "running", execution_id) - status = sm.get_task_status(dag_id, task_id, execution_id) - assert status == "running" - - -def test_get_dag_status(status_manager): - dag_id = "dag_123" - execution_id = "exec_456" - - with status_manager as sm: - # Create execution before setting task status - sm.create_dag_execution(dag_id, execution_id) - sm.set_task_status(dag_id, "task_1", "completed", execution_id) - sm.set_task_status(dag_id, "task_2", "running", execution_id) - - dag_status = sm.get_dag_status(dag_id, execution_id) - assert dag_status["task_1"] == "completed" - assert dag_status["task_2"] == "running" - - -def test_reset_dag_status(status_manager): - dag_id = "dag_123" - execution_id = "exec_456" - - with status_manager as sm: - # Create execution before setting task status - sm.create_dag_execution(dag_id, execution_id) - sm.set_task_status(dag_id, "task_1", "completed", execution_id) - sm.reset_dag_status(dag_id, execution_id) - - dag_status = sm.get_dag_status(dag_id, execution_id) - assert not dag_status - - -def test_create_dag_execution(status_manager): - dag_id = "dag_123" - execution_id = "exec_456" - - with status_manager as sm: - sm.create_dag_execution(dag_id, execution_id) - execution_details = sm.get_dag_execution_details(dag_id, execution_id) - assert execution_details["execution_id"] == execution_id - - -def test_logging_and_retrieval(status_manager): - dag_id = "dag_123" - execution_id = "exec_456" - task_id = "task_1" - - with status_manager as sm: - # Create execution before logging - sm.create_dag_execution(dag_id, execution_id) - sm.log_message(dag_id, execution_id, task_id, "INFO", "Test log message.") - logs = sm.get_execution_logs(dag_id, execution_id) - - assert len(logs) == 1 - assert logs[0]["message"] == "Test log message." - - -def test_cleanup_old_executions(status_manager): - dag_id = "dag_123" - old_execution_id = "old_exec" - new_execution_id = "new_exec" - old_date = datetime.now().replace(year=2000).isoformat() - - with status_manager as sm: - # Create old execution by direct SQL insertion - sm._conn.execute(""" - INSERT OR IGNORE INTO dags (id) VALUES (?) - """, (dag_id,)) - sm._conn.execute(""" - INSERT INTO executions (id, dag_id, started_at, status, thread_id, pid) - VALUES (?, ?, ?, ?, ?, ?)""", - (old_execution_id, dag_id, old_date, "completed", str(threading.get_ident()), str(threading.get_ident())) - ) - - # Create new execution - sm.create_dag_execution(dag_id, new_execution_id) - - # Clean up executions older than a year - cleaned = sm.cleanup_old_executions(days_to_keep=365) - - assert cleaned == 1 - executions = sm.get_dag_history(dag_id) - assert len(executions) == 1 - assert executions[0]["execution_id"] == new_execution_id - - -def test_get_task_status_fallback(status_manager): - """Test the fallback logic for getting task status.""" - dag_id = "dag_123" - task_id = "task_1" - execution_id = "exec_456" - - with status_manager as sm: - # Set task status without specific execution_id (uses "default") - sm.set_task_status(dag_id, task_id, "completed") - - # Create a new execution but don't set task status for it - sm.create_dag_execution(dag_id, execution_id) - - # Should fallback to default execution when task not found for specific execution - status = sm.get_task_status(dag_id, task_id, execution_id) - assert status == "completed" - - -def test_get_running_dags(status_manager): - """Test getting all running DAGs.""" - dag_id1 = "dag_1" - dag_id2 = "dag_2" - execution_id1 = "exec_1" - execution_id2 = "exec_2" - - with status_manager as sm: - # Create running executions - sm.create_dag_execution(dag_id1, execution_id1) - sm.create_dag_execution(dag_id2, execution_id2) - - # Mark one as completed - sm.update_dag_execution_status(dag_id1, execution_id1, "completed") - - running_dags = sm.get_running_dags() - assert len(running_dags) == 1 - assert running_dags[0]["dag_id"] == dag_id2 - assert running_dags[0]["execution_id"] == execution_id2 - - -def test_get_dags_by_status(status_manager): - """Test getting DAGs by status.""" - dag_id1 = "dag_1" - dag_id2 = "dag_2" - execution_id1 = "exec_1" - execution_id2 = "exec_2" - - with status_manager as sm: - # Create executions - sm.create_dag_execution(dag_id1, execution_id1) - sm.create_dag_execution(dag_id2, execution_id2) - - # Update statuses - sm.update_dag_execution_status(dag_id1, execution_id1, "completed") - sm.update_dag_execution_status(dag_id2, execution_id2, "failed") - - completed_dags = sm.get_dags_by_status("completed") - failed_dags = sm.get_dags_by_status("failed") - - assert len(completed_dags) == 1 - assert completed_dags[0]["dag_id"] == dag_id1 - - assert len(failed_dags) == 1 - assert failed_dags[0]["dag_id"] == dag_id2 - - -def test_get_all_dags(status_manager): - """Test getting all DAGs.""" - dag_id1 = "dag_1" - dag_id2 = "dag_2" - execution_id1 = "exec_1" - execution_id2 = "exec_2" - - with status_manager as sm: - sm.create_dag_execution(dag_id1, execution_id1) - sm.create_dag_execution(dag_id2, execution_id2) - - all_dags = sm.get_all_dags() - assert len(all_dags) == 2 - - dag_ids = [dag["dag_id"] for dag in all_dags] - assert dag_id1 in dag_ids - assert dag_id2 in dag_ids - - -def test_get_dag_summary(status_manager): - """Test getting DAG summary statistics.""" - with status_manager as sm: - # Create multiple executions with different statuses - for i in range(3): - dag_id = f"dag_{i}" - execution_id = f"exec_{i}" - sm.create_dag_execution(dag_id, execution_id) - - if i == 0: - sm.update_dag_execution_status(dag_id, execution_id, "completed") - elif i == 1: - sm.update_dag_execution_status(dag_id, execution_id, "failed") - # i == 2 remains "running" - - summary = sm.get_dag_summary() - - assert summary["total_executions"] == 3 - assert summary["unique_dags"] == 3 - assert summary["status_counts"]["completed"] == 1 - assert summary["status_counts"]["failed"] == 1 - assert summary["status_counts"]["running"] == 1 - - -def test_cancel_dag_execution(status_manager): - """Test cancelling a DAG execution.""" - dag_id = "dag_123" - execution_id = "exec_456" - - with status_manager as sm: - sm.create_dag_execution(dag_id, execution_id) - - # Cancel the execution - cancelled = sm.cancel_dag_execution(dag_id, execution_id) - assert cancelled is True - - # Check the status was updated - details = sm.get_dag_execution_details(dag_id, execution_id) - assert details["status"] == "cancelled" - assert details["completed_at"] is not None - - -def test_mark_incomplete_tasks_as_failed(status_manager): - """Test marking incomplete tasks as failed.""" - dag_id = "dag_123" - execution_id = "exec_456" - - with status_manager as sm: - sm.create_dag_execution(dag_id, execution_id) - - # Set various task statuses - sm.set_task_status(dag_id, "task_1", "running", execution_id) - sm.set_task_status(dag_id, "task_2", "pending", execution_id) - sm.set_task_status(dag_id, "task_3", "completed", execution_id) - - # Mark incomplete tasks as failed - sm.mark_incomplete_tasks_as_failed(dag_id, execution_id) - - # Check results - assert sm.get_task_status(dag_id, "task_1", execution_id) == "failed" - assert sm.get_task_status(dag_id, "task_2", execution_id) == "failed" - assert sm.get_task_status(dag_id, "task_3", execution_id) == "completed" # Should remain unchanged - - -def test_initialize_tasks_for_execution(status_manager): - """Test initializing tasks for an execution.""" - dag_id = "dag_123" - execution_id = "exec_456" - task_ids = ["task_1", "task_2", "task_3"] - - with status_manager as sm: - sm.create_dag_execution(dag_id, execution_id) - sm.initialize_tasks_for_execution(dag_id, execution_id, task_ids) - - # All tasks should be pending - for task_id in task_ids: - status = sm.get_task_status(dag_id, task_id, execution_id) - assert status == "pending" - - -def test_task_status_with_timestamps(status_manager): - """Test that task status updates include proper timestamps.""" - dag_id = "dag_123" - execution_id = "exec_456" - task_id = "task_1" - - with status_manager as sm: - sm.create_dag_execution(dag_id, execution_id) - - # Set to running (should set started_at) - sm.set_task_status(dag_id, task_id, "running", execution_id) - - # Set to completed (should set completed_at) - sm.set_task_status(dag_id, task_id, "completed", execution_id) - - # Get execution details to check timestamps - details = sm.get_dag_execution_details(dag_id, execution_id) - task = next(t for t in details["tasks"] if t["task_id"] == task_id) - - assert task["started_at"] is not None - assert task["completed_at"] is not None - assert task["status"] == "completed" - - -def test_get_dag_definition_returns_saved_definition(status_manager): - class DummyDag: - dag_id = "dummy_dag" - - def to_dict(self): - return {"dag_id": self.dag_id, "tasks": {}} - - with status_manager as sm: - dag = DummyDag() - sm.save_dag_definition(dag) - - stored_definition = sm.get_dag_definition(dag.dag_id) - - assert stored_definition == dag.to_dict() - - -def test_save_dag_definition_persists_filepath(status_manager): - class DummyDag: - dag_id = "dummy_dag" - - def to_dict(self): - return {"dag_id": self.dag_id, "tasks": {}} - - dag_file_path = "/tmp/workflows/sample.yaml" - - with status_manager as sm: - dag = DummyDag() - sm.save_dag_definition(dag, dag_file_path) - - with sm.Session() as session: - stored_dag = session.query(DagORM).filter_by(id=dag.dag_id).first() - assert stored_dag is not None - assert stored_dag.dag_filepath == dag_file_path - From 4595da4c012cf7603cfdc409b50fe1c6fac5e7e1 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 6 Dec 2025 10:38:18 +0100 Subject: [PATCH 31/38] Added test for dag_loader.py and task_registry.py --- src/maestro/server/internals/dag_loader.py | 62 +++-- tests/test_dag_loader.py | 263 +++++++++++++++++++++ tests/test_task_registry.py | 63 +++++ 3 files changed, 363 insertions(+), 25 deletions(-) create mode 100644 tests/test_dag_loader.py create mode 100644 tests/test_task_registry.py diff --git a/src/maestro/server/internals/dag_loader.py b/src/maestro/server/internals/dag_loader.py index 688d000..d1ac898 100644 --- a/src/maestro/server/internals/dag_loader.py +++ b/src/maestro/server/internals/dag_loader.py @@ -1,21 +1,25 @@ -import yaml -from typing import Dict, Any, Type, Optional -from pathlib import Path from datetime import datetime +from pathlib import Path +from typing import Any, Dict, Optional, Type + +import yaml from pydantic import BaseModel, ValidationError +from maestro.server.internals.task_registry import TaskRegistry +from maestro.server.tasks.base import BaseTask from maestro.shared.dag import DAG from maestro.shared.task import Task, TaskStatus -from maestro.server.tasks.base import BaseTask -from maestro.server.internals.task_registry import TaskRegistry + class DAGConfig(BaseModel): """Schema for DAG configuration validation.""" + dag: Dict[str, Any] class Config: extra = "allow" + class DAGLoader: def __init__(self, task_registry: TaskRegistry): self.task_registry = task_registry @@ -40,20 +44,20 @@ def load_dag_from_file(self, filepath: str, dag_id: Optional[str] = None) -> DAG # Extract scheduling configuration start_time = None cron_schedule = None - + if "start_time" in config.dag and "cron_schedule" in config.dag: raise ValueError("Cannot specify both start_time and cron_schedule") - + if "start_time" in config.dag: start_time_str = config.dag["start_time"] try: start_time = self._parse_datetime(start_time_str) except ValueError as e: raise ValueError(f"Invalid start_time format: {e}") - + if "cron_schedule" in config.dag: cron_schedule = config.dag["cron_schedule"] - + dag = DAG(dag_id=dag_id, start_time=start_time, cron_schedule=cron_schedule) dag_config = config.dag @@ -66,7 +70,9 @@ def load_dag_from_file(self, filepath: str, dag_id: Optional[str] = None) -> DAG task = self._create_task_from_config(task_config, filepath) dag.add_task(task) except Exception as e: - raise ValueError(f"Error creating task '{task_config.get('task_id', 'unknown')}': {e}") + raise ValueError( + f"Error creating task '{task_config.get('task_id', 'unknown')}': {e}" + ) try: dag.validate() @@ -75,7 +81,9 @@ def load_dag_from_file(self, filepath: str, dag_id: Optional[str] = None) -> DAG return dag - def _create_task_from_config(self, task_config: Dict[str, Any], dag_file_path: str) -> BaseTask: + def _create_task_from_config( + self, task_config: Dict[str, Any], dag_file_path: str + ) -> BaseTask: """Create a task instance from configuration.""" task_config = task_config.copy() # Don't modify original @@ -110,26 +118,28 @@ def _create_task_from_config(self, task_config: Dict[str, Any], dag_file_path: s return task_class(**task_config) except Exception as e: raise ValueError(f"Error instantiating {task_type_name}: {e}") - + def _parse_datetime(self, datetime_str: str) -> datetime: """Parse datetime string in various formats.""" # Common datetime formats to try formats = [ - "%Y-%m-%d %H:%M:%S", # 2023-12-25 14:30:00 - "%Y-%m-%dT%H:%M:%S", # 2023-12-25T14:30:00 (ISO format) - "%Y-%m-%dT%H:%M:%SZ", # 2023-12-25T14:30:00Z (UTC) - "%Y-%m-%d %H:%M", # 2023-12-25 14:30 - "%Y-%m-%dT%H:%M", # 2023-12-25T14:30 - "%Y-%m-%d", # 2023-12-25 (assumes 00:00:00) + "%Y-%m-%d %H:%M:%S", # 2023-12-25 14:30:00 + "%Y-%m-%dT%H:%M:%S", # 2023-12-25T14:30:00 (ISO format) + "%Y-%m-%dT%H:%M:%SZ", # 2023-12-25T14:30:00Z (UTC) + "%Y-%m-%d %H:%M", # 2023-12-25 14:30 + "%Y-%m-%dT%H:%M", # 2023-12-25T14:30 + "%Y-%m-%d", # 2023-12-25 (assumes 00:00:00) ] - + for fmt in formats: try: return datetime.strptime(datetime_str, fmt) except ValueError: continue - - raise ValueError(f"Unable to parse datetime string '{datetime_str}'. Supported formats: {formats}") + + raise ValueError( + f"Unable to parse datetime string '{datetime_str}'. Supported formats: {formats}" + ) def load_dag_from_dict(self, dag_dict: Dict[str, Any]) -> DAG: """Load a DAG from a dictionary representation.""" @@ -142,21 +152,23 @@ def load_dag_from_dict(self, dag_dict: Dict[str, Any]) -> DAG: # Get tasks from the dictionary tasks = dag_dict.get("tasks", {}) - + # Handle tasks as a dictionary (from to_dict()) for task_id, task_config in tasks.items(): if isinstance(task_config, dict): # Add task_id to the config if not present if "task_id" not in task_config: task_config["task_id"] = task_id - + # Ensure type field exists (from the serialized task model) # The type might be stored as the class name without 'Task' suffix if "type" not in task_config and task_config.get("task_id"): # Try to infer type from other fields or use a default if "playbook" in task_config: task_config["type"] = "AnsibleTask" - elif "working_dir" in task_config and "workflow_mode" in task_config: + elif ( + "working_dir" in task_config and "workflow_mode" in task_config + ): task_config["type"] = "ExtendedTerraformTask" elif "message" in task_config: task_config["type"] = "PrintTask" @@ -169,7 +181,7 @@ def load_dag_from_dict(self, dag_dict: Dict[str, Any]) -> DAG: task_config["type"] = "PrintTask" if "message" not in task_config: task_config["message"] = f"Task {task_id}" - + task = self._create_task_from_config(task_config, None) dag.add_task(task) diff --git a/tests/test_dag_loader.py b/tests/test_dag_loader.py new file mode 100644 index 0000000..5f65767 --- /dev/null +++ b/tests/test_dag_loader.py @@ -0,0 +1,263 @@ +from typing import Any, Optional +from unittest.mock import Mock, patch + +import pytest +from pydantic import Field + +from maestro.server.internals.dag_loader import DAGLoader +from maestro.server.internals.task_registry import TaskRegistry +from maestro.server.tasks.base import BaseTask +from maestro.shared.dag import DAG + + +# ============================== +# DummyTask globale per tutti i test +# ============================== +class DummyTask(BaseTask): + """ + DummyTask per i test: + - eredita correttamente da BaseTask (che è un Pydantic model) + - permette campi extra (foo, message, ecc.) + - dichiara foo e message come opzionali per eliminare warning VSCode + """ + + foo: Optional[Any] = Field(default=None) + message: Optional[str] = Field(default=None) + + class Config: + extra = "allow" # <-- LA PARTE FONDAMENTALE + + def execute_local(self): + pass + + +# ----------------------------- +# Fixtures +# ----------------------------- +@pytest.fixture +def mock_task_registry(): + registry = Mock(spec=TaskRegistry) + # Ora registry.get() restituisce sempre DummyTask + registry.get.return_value = DummyTask + return registry + + +@pytest.fixture +def dag_loader(mock_task_registry): + return DAGLoader(task_registry=mock_task_registry) + + +# ----------------------------- +# Test: load DAG from dictionary +# ----------------------------- +def test_load_dag_from_dict_minimal(dag_loader): + dag_dict = {"dag_id": "dag1", "tasks": {"task1": {"type": "DummyTask"}}} + dag = dag_loader.load_dag_from_dict(dag_dict) + + assert isinstance(dag, DAG) + assert len(dag.tasks) == 1 + + # Accediamo al task corretto dal dict + task = dag.tasks["task1"] # task_id è la chiave del dict + assert task.task_id == "task1" + + +def test_load_dag_from_dict_infers_printtask(dag_loader, mock_task_registry): + # Registriamo DummyTask come PrintTask fittizio + mock_task_registry.register("PrintTask", DummyTask) + dag_loader.task_registry = mock_task_registry + + dag_dict = {"dag_id": "dag2", "tasks": {"t1": {"message": "hello world"}}} + + dag = dag_loader.load_dag_from_dict(dag_dict) + + # Accediamo al task + task = dag.tasks["t1"] + assert task.task_id == "t1" + assert hasattr(task, "message") + assert task.message == "hello world" + + +def test_load_dag_from_dict_invalid_type_raises(dag_loader, mock_task_registry): + dag_dict = {"dag_id": "dag3", "tasks": {"t1": {"type": "UnknownTask"}}} + # Make registry return None + mock_task_registry.get.return_value = None + with pytest.raises(ValueError, match="Unknown task type"): + dag_loader.load_dag_from_dict(dag_dict) + + +# ----------------------------- +# Test: datetime parsing +# ----------------------------- +@pytest.mark.parametrize( + "dt_str,expected", + [ + ("2023-12-25 14:30:00", (2023, 12, 25, 14, 30, 0)), + ("2023-12-25", (2023, 12, 25, 0, 0, 0)), + ("2023-12-25T14:30", (2023, 12, 25, 14, 30, 0)), + ], +) +def test_parse_datetime_formats(dag_loader, dt_str, expected): + dt = dag_loader._parse_datetime(dt_str) + assert (dt.year, dt.month, dt.day, dt.hour, dt.minute, dt.second) == expected + + +def test_parse_datetime_invalid_raises(dag_loader): + with pytest.raises(ValueError, match="Unable to parse datetime string"): + dag_loader._parse_datetime("invalid-date") + + +# ----------------------------- +# Test: create task from config merges params and preserves condition +# ----------------------------- +def test_create_task_from_config_merges_params_and_condition( + dag_loader, mock_task_registry +): + # Registriamo DummyTask nel registry per il test + mock_task_registry.register("DummyTask", DummyTask) + dag_loader.task_registry = mock_task_registry + + task_config = { + "type": "DummyTask", + "task_id": "t1", + "params": {"foo": 42}, + "condition": "some_condition", + } + task = dag_loader._create_task_from_config(task_config, "dummy.yaml") + + # Verifica attributi + assert task.task_id == "t1" + assert hasattr(task, "foo") + assert task.foo == 42 + assert hasattr(task, "condition") + assert task.condition == "some_condition" + + +# ----------------------------- +# Test: load DAG from file raises on missing file +# ----------------------------- +def test_load_dag_from_file_missing_file(dag_loader): + with pytest.raises(FileNotFoundError): + dag_loader.load_dag_from_file("/non/existent/file.yaml") + + +# ----------------------------- +# Test: load DAG from file raises on invalid YAML +# ----------------------------- +def test_load_dag_from_file_invalid_yaml(dag_loader): + # Patch open to return invalid YAML + with patch("builtins.open", mock_open(read_data=":::")): + with pytest.raises(ValueError, match="Invalid YAML configuration"): + dag_loader.load_dag_from_file("dummy.yaml") + + +# Utility for mocking open +from io import StringIO +from unittest.mock import mock_open, patch + + +# ----------------------------- +# Test: load DAG from file with valid YAML +# ----------------------------- +def test_load_dag_from_file_valid_yaml(dag_loader, mock_task_registry): + yaml_content = """ +dag: + dag_id: my_dag + tasks: + - type: DummyTask + task_id: task1 +""" + m = mock_open(read_data=yaml_content) + with patch("builtins.open", m): + dag = dag_loader.load_dag_from_file("dummy.yaml") + + assert isinstance(dag, DAG) + assert len(dag.tasks) == 1 + + # Accediamo correttamente al task + task = list(dag.tasks.values())[0] # oppure dag.tasks["task1"] se chiave nota + assert task.task_id == "task1" + + +# ----------------------------- +# Test: load DAG from file with both start_time and cron_schedule -> ValueError +# ----------------------------- +def test_load_dag_from_file_both_start_and_cron(dag_loader): + yaml_content = """ +dag: + dag_id: dag1 + start_time: "2023-12-25 14:30:00" + cron_schedule: "* * * * *" + tasks: + - type: DummyTask + task_id: t1 +""" + m = mock_open(read_data=yaml_content) + with patch("builtins.open", m): + with pytest.raises( + ValueError, match="Cannot specify both start_time and cron_schedule" + ): + dag_loader.load_dag_from_file("dummy.yaml") + + +# ----------------------------- +# Test: load DAG from file with invalid start_time format -> ValueError +# ----------------------------- +def test_load_dag_from_file_invalid_start_time(dag_loader): + yaml_content = """ +dag: + dag_id: dag1 + start_time: "not-a-date" + tasks: + - type: DummyTask + task_id: t1 +""" + m = mock_open(read_data=yaml_content) + with patch("builtins.open", m): + with pytest.raises(ValueError, match="Invalid start_time format"): + dag_loader.load_dag_from_file("dummy.yaml") + + +# ----------------------------- +# Test: load DAG from file missing 'tasks' field -> ValueError +# ----------------------------- +def test_load_dag_from_file_missing_tasks(dag_loader): + yaml_content = """ +dag: + dag_id: dag1 +""" + m = mock_open(read_data=yaml_content) + with patch("builtins.open", m): + with pytest.raises( + ValueError, match="DAG configuration must contain 'tasks' field" + ): + dag_loader.load_dag_from_file("dummy.yaml") + + +# ----------------------------- +# Test integrato DAGLoader con più task +# ----------------------------- +def test_load_dag_from_dict_multiple_tasks(dag_loader, mock_task_registry): + # Registriamo DummyTask + mock_task_registry.register("DummyTask", DummyTask) + dag_loader.task_registry = mock_task_registry + + dag_dict = { + "dag_id": "integration_dag", + "tasks": { + "task1": {"type": "DummyTask", "task_id": "task1"}, + "task2": {"type": "DummyTask", "task_id": "task2"}, + "task3": {"type": "DummyTask", "task_id": "task3"}, + }, + } + + dag = dag_loader.load_dag_from_dict(dag_dict) + + # Verifica DAG + assert isinstance(dag, DAG) + assert dag.dag_id == "integration_dag" + assert len(dag.tasks) == 3 + + # Ora accediamo ai task correttamente (dag.tasks è dict) + task_ids = {task.task_id for task in dag.tasks.values()} + assert task_ids == {"task1", "task2", "task3"} diff --git a/tests/test_task_registry.py b/tests/test_task_registry.py new file mode 100644 index 0000000..c3ec794 --- /dev/null +++ b/tests/test_task_registry.py @@ -0,0 +1,63 @@ +import pytest + +from maestro.server.internals.task_registry import TaskRegistry +from maestro.server.tasks.base import BaseTask + + +def test_builtin_tasks_are_loaded(): + registry = TaskRegistry() + types = registry.list_types() + + # Controlliamo che i task principali siano presenti + expected = { + "PrintTask", + "FileWriterTask", + "WaitTask", + "TerraformTask", + "AnsibleTask", + "ExtendedTerraformTask", + "PythonTask", + "BashTask", + } + + for name in expected: + assert name in types, f"Expected built-in task '{name}' to be registered" + + +def test_register_valid_task(): + registry = TaskRegistry() + + class DummyTask(BaseTask): + pass + + registry.register("DummyTask", DummyTask) + + assert registry.get("DummyTask") is DummyTask + assert "DummyTask" in registry.list_types() + + +def test_register_invalid_task(): + registry = TaskRegistry() + + class NotATask: + pass + + with pytest.raises(ValueError): + registry.register("BadTask", NotATask) + + +def test_get_unknown_task_returns_none(): + registry = TaskRegistry() + assert registry.get("UNKNOWN") is None + + +def test_list_types_returns_copy_not_reference(): + registry = TaskRegistry() + types1 = registry.list_types() + types2 = registry.list_types() + + # devono essere uguali come contenuto... + assert types1 == types2 + + # ...ma NON lo stesso oggetto + assert types1 is not types2 From 734923d8abee4639e5324b8153711c94fcf3a7b1 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 6 Dec 2025 21:04:12 +0100 Subject: [PATCH 32/38] Test on status_manager-py - work-in-progress --- data.json | 1 - .../8.2.1.generate_filter_summarize.yaml | 2 +- maestro.db | Bin 45056 -> 0 bytes tests/Vecchi_test/conftest.py | 599 ++++++++++++++++++ tests/Vecchi_test/integration/conftest.py | 102 +++ .../integration/test_cli_rest_integration.py | 51 ++ .../integration/test_debug_server.py | 101 +++ .../test_server_api_integration.py | 310 +++++++++ tests/Vecchi_test/test_ansible_task.py | 343 ++++++++++ tests/Vecchi_test/test_api_client.py | 528 +++++++++++++++ tests/Vecchi_test/test_cli_attach.py | 399 ++++++++++++ tests/Vecchi_test/test_cli_client.py | 422 ++++++++++++ tests/Vecchi_test/test_cli_integration.py | 520 +++++++++++++++ tests/Vecchi_test/test_cron_feature.py | 42 ++ tests/Vecchi_test/test_dag.py | 139 ++++ tests/Vecchi_test/test_dag_id_generation.py | 259 ++++++++ tests/Vecchi_test/test_db_feature.py | 107 ++++ tests/Vecchi_test/test_enhanced_cli.py | 441 +++++++++++++ .../test_extended_terraform_task.py | 366 +++++++++++ tests/Vecchi_test/test_multi_executor.py | 286 +++++++++ .../test_orchestrator_dagloader.py | 266 ++++++++ tests/Vecchi_test/test_server.py | 335 ++++++++++ tests/Vecchi_test/test_status_manager.py | 350 ++++++++++ tests/Work_in_progress/test_status_manager.py | 294 +++++++++ 24 files changed, 6261 insertions(+), 2 deletions(-) delete mode 100644 data.json delete mode 100644 maestro.db create mode 100644 tests/Vecchi_test/conftest.py create mode 100644 tests/Vecchi_test/integration/conftest.py create mode 100644 tests/Vecchi_test/integration/test_cli_rest_integration.py create mode 100644 tests/Vecchi_test/integration/test_debug_server.py create mode 100644 tests/Vecchi_test/integration/test_server_api_integration.py create mode 100644 tests/Vecchi_test/test_ansible_task.py create mode 100644 tests/Vecchi_test/test_api_client.py create mode 100644 tests/Vecchi_test/test_cli_attach.py create mode 100644 tests/Vecchi_test/test_cli_client.py create mode 100644 tests/Vecchi_test/test_cli_integration.py create mode 100644 tests/Vecchi_test/test_cron_feature.py create mode 100644 tests/Vecchi_test/test_dag.py create mode 100644 tests/Vecchi_test/test_dag_id_generation.py create mode 100644 tests/Vecchi_test/test_db_feature.py create mode 100644 tests/Vecchi_test/test_enhanced_cli.py create mode 100644 tests/Vecchi_test/test_extended_terraform_task.py create mode 100644 tests/Vecchi_test/test_multi_executor.py create mode 100644 tests/Vecchi_test/test_orchestrator_dagloader.py create mode 100644 tests/Vecchi_test/test_server.py create mode 100644 tests/Vecchi_test/test_status_manager.py create mode 100644 tests/Work_in_progress/test_status_manager.py diff --git a/data.json b/data.json deleted file mode 100644 index 4136c6f..0000000 --- a/data.json +++ /dev/null @@ -1 +0,0 @@ -{"values": [1, 2, 3, 4, 5]} \ No newline at end of file diff --git a/examples/2_New_examples/8.2.1.generate_filter_summarize.yaml b/examples/2_New_examples/8.2.1.generate_filter_summarize.yaml index 0b03c7a..44a489e 100644 --- a/examples/2_New_examples/8.2.1.generate_filter_summarize.yaml +++ b/examples/2_New_examples/8.2.1.generate_filter_summarize.yaml @@ -46,7 +46,7 @@ dag: data = [int(x) for x in f.read().split()] if data: - print("High numbers summary: min =", min(data), "max =", max(data), "avg =", sum(data)/len(data)) + print("High numbers summary: min =", min(data), "max =", max(data), "avg =", round(sum(data)/len(data), 2)) else: print("No high numbers found.") dependencies: ["filter_numbers"] diff --git a/maestro.db b/maestro.db deleted file mode 100644 index 0deeaf9192592cefd88e5eaea54bd03e545267e9..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 45056 zcmeI4UvJyi6~HN5mMt~1QVfES0mE*z1r||B7DZ9AQgoZ*$gVN}NoBcfVHpHNUfWD1 zQX#3>M&@lww}%14zQEr16*gcGeJL;md)R9~LZ9}uhoRU*&m|>UROQBz9VBiKEz0Em zd+zU?OY+=H-R%$7b%&CD(`qRW5xJXOAi({C5RT(+!TuKP{VfCs!Tt&S4-6fjcX*4- zKKWxb_CGEXKH_3Oiv2tKulX-;{4MhT8-I#KBL9Fu+>ihgKmter2_OL^@ckgry&0KJ zY-IvnK{qsdtRKrt+g1;$)@f2peq=thWrH3&vehwUM{m*5*teEz)k?ie);8}~KO!&8 zPPR75Xb~jktAX!k=VlX`OyF6}Q64tGQR8U#a8H7I<=%RAOyQBRu8|KbwdMCKwUkg6 zc(S=wC!0I#>uC}R`>G}@>sysN3xAm%>d%heGYHJ&n+Q;OV)sIO^*Z2tk z^KdA!oe6L*LHltNo?~UD?sH-YAjER0Cd1SCeb8`yZKKM9RkPJ@(lHUv zp+yyKsC<^%>^SX?L+aI!T&%ibQ_B^@v@~jg)zqu2)f!7Vf;CON)}+a>=sZigvQ?|D zt!^>}q`X8tsa031wd&?_b(`h0UB!&1^%XRnV-Vrxi}3txDDht6JcYVykOlYO+_Sf5 zW)tyv;P*4G>YC<(&GwK_aV(^#w_b~V%JJ)si9L{L1ShL=B&bQB(B`n-TGY0c13F;N zdM)U()gGwPSNT}AjVR-#ut)f3Kb)RT+`b+7^tUdPL1XN*qo9w^S@>ku{fdmY4V~%5 z7W$HK#9%0~diy-vIwK(qXU8+$#ei#N&nhlaX5wt0@=^2*|NN^1-PaABb-saVFvHu| zo3yPshr=eMLQfv);pBdGrLwbLC(AoE=xBABT@SYFm5m2uIxk!k7p6jqcjMihgKmter2_OL^aIFda`A#6j|K-kO@jz)d6~mF+imfWm zr%CqkrEAH%B<^c_09)Xdh~js7IiGI#vByG&Q4^FHkA`7h=ugLB39Wpb8JBf(qc^j zuV^)!&3dArrFTx+lqG&(>4sAWw<~!Hr5sR$TJUHrYl@>l^wZ=JYkwr9etpnAy}u#o zRWK@sQg5|Q%OQ_!(?}E3ZWwLGHf7q`1L~+pT8ifz1_MjpBTpO2CrYzJ?FQuB71AW1 zCSsZt(nQ)jZ5T&7I4mX@Q>p=!9VN>OZGa9N$)iS+CyGt>?=}nq8;i+momLyh3LGU( z_MsB5;zn|{r%nw94F-ikJTnGcwe&XYt!{tmS=qoUPMky27?_1>cA#T>0*_jSFYfgk zh2KEm80%=Y6vMp^sxZ_vPz3okIk0G(WK6QKo6D5<79iwEB^mcR(O1Mh#XcOUf%&&K zySn+hWiRygqW4+B8$y2QlIUK{&qmrve~|AYm&Mc5!JXm4H;Q~NFJ**$CMOYL>8_N! zD@hT!%Liir;oyTC5yYv6C#~gfc zLjp(u2_OL^fCP{L5`9Om2T^zqFJSRG}>Sz4iVJ0^0CQe%YJJ&kCYgDCWk+NRqfDibrF{1%Gfb*G@lPg^ zYWVl%u5A4u#_y5k{(_Cf?x-p(4!95NiJXuWEcWTZ54|-?a{M?c#iEca7e=L=3wlyY zS1aYN?P#!8PD-YAzMQK#7 zpH6vlO{B)`IA%ryNB{{S0VIF~kN^@u0!RP}Ac0p!pc|Tw+~IfNb#hq7 zZFn@|e@zsHVj=+~ afCP{L5 0: + break + time.sleep(1) # Wait a bit before retrying + assert len(logs) > 0 + assert any("task1" in log["task_id"] for log in logs) + +""" def test_attach_dag(client: TestClient, sample_dag_file: str): + # Create and run a DAG first + client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_attach"}) + client.post("/v1/dags/test_dag_attach/run", json={"dag_id": "test_dag_attach"}) + + # Stream logs + messages = [] + with client.stream("GET", "/v1/dags/test_dag_attach/attach") as response: + for chunk in response.iter_bytes(): + messages.append(chunk.decode("utf-8")) + if "Task 'task2' completed" in chunk.decode("utf-8"): + break + + assert any("task1" in msg for msg in messages) + assert any("task3" in msg for msg in messages) """ + +def test_rm_dag(client: TestClient, sample_dag_file: str): + # Create a DAG + client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_rm"}) + + # Remove the DAG + response = client.delete("/v1/dags/test_dag_rm") + assert response.status_code == 200 + assert "removed successfully" in response.json()["message"] + + # Verify it's gone + response = client.get("/v1/dags?filter=all") + assert response.status_code == 200 + assert not any(d["dag_id"] == "test_dag_rm" for d in response.json()) + +def test_rm_running_dag_force(client: TestClient, sample_dag_file: str): + # Create and run a DAG + client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_rm_force"}) + client.post("/v1/dags/test_dag_rm_force/run", json={"dag_id": "test_dag_rm_force"}) + time.sleep(1) # Give it a moment to start running + + # Try to remove without force (should fail) + response = client.delete("/v1/dags/test_dag_rm_force") + assert response.status_code == 409 # Conflict + assert "currently running" in response.json()["detail"] + + # Remove with force + response = client.delete("/v1/dags/test_dag_rm_force?force=true") + assert response.status_code == 200 + assert "removed successfully" in response.json()["message"] + + # Verify it's gone + response = client.get("/v1/dags?filter=all") + assert response.status_code == 200 + assert not any(d["dag_id"] == "test_dag_rm_force" for d in response.json()) + +def test_ls_dags(client: TestClient, sample_dag_file: str): + # Create a few DAGs for testing ls + client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "ls_dag_1"}) + client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "ls_dag_2"}) + client.post("/v1/dags/ls_dag_1/run", json={"dag_id": "ls_dag_1"}) + time.sleep(1) # Give it a moment to start running + + # Test ls all + response = client.get("/v1/dags?filter=all") + assert response.status_code == 200 + dags = response.json() + + assert any(d["dag_id"] == "ls_dag_1" for d in dags) + assert any(d["dag_id"] == "ls_dag_2" for d in dags) + + # Test ls active + response = client.get("/v1/dags?filter=active") + assert response.status_code == 200 + dags = response.json() + print(f"\n[DEBUG] Active DAGs response: {dags}") + assert any(d["dag_id"] == "ls_dag_1" for d in dags) + assert not any(d["dag_id"] == "ls_dag_2" for d in dags) # ls_dag_2 is not running + + # Test ls terminated (after ls_dag_1 completes) + time.sleep(5) # Wait for ls_dag_1 to complete + response = client.get("/v1/dags?filter=terminated") + assert response.status_code == 200 + dags = response.json() + print(f"\n[DEBUG] Terminated DAGs response: {dags}") + assert any(d["dag_id"] == "ls_dag_1" and d["completed_at"] is not None for d in dags) + assert not any(d["dag_id"] == "ls_dag_2" for d in dags) # ls_dag_2 was never run + +def test_stop_dag(client: TestClient, sample_dag_file: str): + # Create and run a DAG + client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_stop"}) + run_response = client.post("/v1/dags/test_dag_stop/run", json={"dag_id": "test_dag_stop"}) + execution_id = run_response.json()["execution_id"] + time.sleep(1) # Give it a moment to start running + + # Stop the DAG + response = client.post(f"/v1/dags/test_dag_stop/stop?execution_id={execution_id}") + assert response.status_code == 200 + assert "Successfully stopped" in response.json()["message"] + + # Verify status is cancelled or completed (if it finished before cancellation took effect) + status_response = client.get(f"/dags/test_dag_stop/status?execution_id={execution_id}") + assert status_response.status_code == 200 + # The DAG might complete before cancellation takes effect in test environment + assert status_response.json()["status"] in ["cancelled", "completed"] + +def test_resume_dag(client: TestClient, sample_dag_file: str): + # Create and run a DAG, then stop it + client.post("/v1/dags/create", json={"dag_file_path": sample_dag_file, "dag_id": "test_dag_resume"}) + run_response = client.post("/v1/dags/test_dag_resume/run", json={"dag_id": "test_dag_resume"}) + execution_id = run_response.json()["execution_id"] + time.sleep(1) # Give it a moment to start running + client.post(f"/v1/dags/test_dag_resume/stop?execution_id={execution_id}") + time.sleep(1) # Give it a moment to stop + + # Resume the DAG + response = client.post(f"/v1/dags/test_dag_resume/resume?execution_id={execution_id}") + assert response.status_code == 200 + assert "resumed with new execution ID" in response.json()["message"] + + # Verify a new execution is running + time.sleep(5) # Wait for it to complete + status_response = client.get(f"/dags/test_dag_resume/status") + assert status_response.status_code == 200 + assert status_response.json()["status"] == "completed" diff --git a/tests/Vecchi_test/test_ansible_task.py b/tests/Vecchi_test/test_ansible_task.py new file mode 100644 index 0000000..9a88715 --- /dev/null +++ b/tests/Vecchi_test/test_ansible_task.py @@ -0,0 +1,343 @@ +import pytest +import os +import tempfile +from pathlib import Path +from unittest.mock import patch, MagicMock, call +from maestro.server.tasks.ansible_task import AnsibleTask + + +class TestAnsibleTask: + """Test suite for AnsibleTask.""" + + def test_ansible_task_creation(self): + """Test basic AnsibleTask creation.""" + task = AnsibleTask( + task_id='test_ansible_task', + playbook='test.yml', + inventory='inventory.ini' + ) + + assert task.task_id == 'test_ansible_task' + assert task.playbook == 'test.yml' + assert task.inventory == 'inventory.ini' + assert task.private_data_dir == './' + assert task.verbosity == 1 + assert task.extra_vars == {} + assert task.become_user is None + + def test_ansible_task_with_all_fields(self): + """Test AnsibleTask creation with all fields.""" + task = AnsibleTask( + task_id='test_ansible_task', + playbook='test.yml', + inventory='inventory.ini', + private_data_dir='/tmp/ansible', + verbosity=2, + extra_vars={'env': 'prod', 'version': '1.0'}, + become_user='root' + ) + + assert task.task_id == 'test_ansible_task' + assert task.playbook == 'test.yml' + assert task.inventory == 'inventory.ini' + assert task.private_data_dir == '/tmp/ansible' + assert task.verbosity == 2 + assert task.extra_vars == {'env': 'prod', 'version': '1.0'} + assert task.become_user == 'root' + + def test_get_absolute_paths_with_dag_file(self): + """Test get_absolute_paths when dag_file_path is provided.""" + with tempfile.TemporaryDirectory() as temp_dir: + dag_file = Path(temp_dir) / 'test.yml' + dag_file.touch() + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini', + private_data_dir='./ansible', + dag_file_path=str(dag_file) + ) + + playbook_path, inventory_path, data_dir = task.get_absolute_paths() + + assert playbook_path == str(Path(temp_dir) / 'playbook.yml') + assert inventory_path == str(Path(temp_dir) / 'inventory.ini') + assert data_dir == str(Path(temp_dir) / 'ansible') + + def test_get_absolute_paths_without_dag_file(self): + """Test get_absolute_paths when dag_file_path is not provided.""" + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini', + private_data_dir='./ansible' + ) + + playbook_path, inventory_path, data_dir = task.get_absolute_paths() + + # Should resolve relative to current directory + assert playbook_path == str(Path('playbook.yml').resolve()) + assert inventory_path == str(Path('inventory.ini').resolve()) + assert data_dir == str(Path('./ansible').resolve()) + + def test_get_absolute_paths_with_absolute_paths(self): + """Test get_absolute_paths when paths are already absolute.""" + with tempfile.TemporaryDirectory() as temp_dir: + playbook_abs = Path(temp_dir) / 'playbook.yml' + inventory_abs = Path(temp_dir) / 'inventory.ini' + data_dir_abs = Path(temp_dir) / 'ansible' + + task = AnsibleTask( + task_id='test_ansible_task', + playbook=str(playbook_abs), + inventory=str(inventory_abs), + private_data_dir=str(data_dir_abs), + dag_file_path='/some/other/path/dag.yml' + ) + + playbook_path, inventory_path, data_dir = task.get_absolute_paths() + + # Should use absolute paths as-is + assert playbook_path == str(playbook_abs) + assert inventory_path == str(inventory_abs) + assert data_dir == str(data_dir_abs) + + @patch('maestro.server.tasks.ansible_task.ansible_runner.run') + @patch('maestro.server.tasks.ansible_task.os.path.exists') + @patch('maestro.server.tasks.ansible_task.get_console') + def test_execute_local_success(self, mock_get_console, mock_exists, mock_ansible_run): + """Test successful execution of ansible task.""" + mock_console = MagicMock() + mock_get_console.return_value = mock_console + mock_exists.return_value = True + + # Mock successful ansible run + mock_result = MagicMock() + mock_result.status = 'successful' + mock_ansible_run.return_value = mock_result + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini' + ) + + task.execute_local() + + # Verify console output + mock_console.print.assert_any_call("[AnsibleTask] Executing 'test_ansible_task'") + mock_console.print.assert_any_call("[AnsibleTask] Task 'test_ansible_task' completed successfully.") + + # Verify ansible_runner was called correctly + mock_ansible_run.assert_called_once() + + @patch('maestro.server.tasks.ansible_task.ansible_runner.run') + @patch('maestro.server.tasks.ansible_task.os.path.exists') + @patch('maestro.server.tasks.ansible_task.get_console') + def test_execute_local_failure(self, mock_get_console, mock_exists, mock_ansible_run): + """Test failed execution of ansible task.""" + mock_console = MagicMock() + mock_get_console.return_value = mock_console + mock_exists.return_value = True + + # Mock failed ansible run + mock_result = MagicMock() + mock_result.status = 'failed' + mock_ansible_run.return_value = mock_result + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini' + ) + + with pytest.raises(Exception) as exc_info: + task.execute_local() + + assert "failed with status: failed" in str(exc_info.value) + + # Verify console output + mock_console.print.assert_any_call("[AnsibleTask] Executing 'test_ansible_task'") + mock_console.print.assert_any_call( + "[AnsibleTask] Task 'test_ansible_task' failed with status: failed", + style="red" + ) + + @patch('maestro.server.tasks.ansible_task.os.path.exists') + @patch('maestro.server.tasks.ansible_task.get_console') + def test_execute_local_playbook_not_found(self, mock_get_console, mock_exists): + """Test execution when playbook file doesn't exist.""" + mock_console = MagicMock() + mock_get_console.return_value = mock_console + + # Mock playbook not existing + def mock_exists_func(path): + return 'playbook.yml' not in path + + mock_exists.side_effect = mock_exists_func + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini' + ) + + with pytest.raises(FileNotFoundError) as exc_info: + task.execute_local() + + assert "Playbook not found" in str(exc_info.value) + + @patch('maestro.server.tasks.ansible_task.os.path.exists') + @patch('maestro.server.tasks.ansible_task.get_console') + def test_execute_local_inventory_not_found(self, mock_get_console, mock_exists): + """Test execution when inventory file doesn't exist.""" + mock_console = MagicMock() + mock_get_console.return_value = mock_console + + # Mock inventory not existing + def mock_exists_func(path): + return 'inventory.ini' not in path + + mock_exists.side_effect = mock_exists_func + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini' + ) + + with pytest.raises(FileNotFoundError) as exc_info: + task.execute_local() + + assert "Inventory not found" in str(exc_info.value) + + @patch('maestro.server.tasks.ansible_task.ansible_runner.run') + @patch('maestro.server.tasks.ansible_task.os.path.exists') + @patch('maestro.server.tasks.ansible_task.get_console') + def test_execute_local_with_extra_vars(self, mock_get_console, mock_exists, mock_ansible_run): + """Test execution with extra variables.""" + mock_console = MagicMock() + mock_get_console.return_value = mock_console + mock_exists.return_value = True + + # Mock successful ansible run + mock_result = MagicMock() + mock_result.status = 'successful' + mock_ansible_run.return_value = mock_result + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini', + extra_vars={'env': 'test', 'debug': True}, + verbosity=2 + ) + + task.execute_local() + + # Verify ansible_runner was called with correct parameters + mock_ansible_run.assert_called_once() + call_args = mock_ansible_run.call_args + assert call_args[1]['verbosity'] == 2 + assert call_args[1]['quiet'] is False + + @patch('maestro.server.tasks.ansible_task.ansible_runner.run') + @patch('maestro.server.tasks.ansible_task.os.path.exists') + @patch('maestro.server.tasks.ansible_task.get_console') + def test_execute_local_with_become_user(self, mock_get_console, mock_exists, mock_ansible_run): + """Test execution with become_user option.""" + mock_console = MagicMock() + mock_get_console.return_value = mock_console + mock_exists.return_value = True + + # Mock successful ansible run + mock_result = MagicMock() + mock_result.status = 'successful' + mock_ansible_run.return_value = mock_result + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini', + become_user='root' + ) + + task.execute_local() + + # Verify ansible_runner was called + mock_ansible_run.assert_called_once() + + @patch('maestro.server.tasks.ansible_task.ansible_runner.run') + @patch('maestro.server.tasks.ansible_task.os.path.exists') + @patch('maestro.server.tasks.ansible_task.get_console') + def test_execute_local_console_output(self, mock_get_console, mock_exists, mock_ansible_run): + """Test that console output shows correct paths.""" + mock_console = MagicMock() + mock_get_console.return_value = mock_console + mock_exists.return_value = True + + # Mock successful ansible run + mock_result = MagicMock() + mock_result.status = 'successful' + mock_ansible_run.return_value = mock_result + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini' + ) + + task.execute_local() + + # Verify console output includes path information + assert mock_console.print.call_count >= 4 + mock_console.print.assert_any_call("[AnsibleTask] Executing 'test_ansible_task'") + + # Check that playbook and inventory paths are printed + calls = mock_console.print.call_args_list + playbook_call = any("[AnsibleTask] Using playbook:" in str(call) for call in calls) + inventory_call = any("[AnsibleTask] Using inventory:" in str(call) for call in calls) + + assert playbook_call, "Playbook path should be printed" + assert inventory_call, "Inventory path should be printed" + + def test_ansible_task_field_defaults(self): + """Test that default field values are set correctly.""" + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbook.yml', + inventory='inventory.ini' + ) + + # Test default values + assert task.private_data_dir == './' + assert task.verbosity == 1 + assert task.extra_vars == {} + assert task.become_user is None + + def test_ansible_task_with_complex_paths(self): + """Test ansible task with complex relative and absolute paths.""" + with tempfile.TemporaryDirectory() as temp_dir: + dag_file = Path(temp_dir) / 'dag.yml' + dag_file.touch() + + # Create subdirectories + playbook_dir = Path(temp_dir) / 'playbooks' + playbook_dir.mkdir() + inventory_dir = Path(temp_dir) / 'inventories' + inventory_dir.mkdir() + + task = AnsibleTask( + task_id='test_ansible_task', + playbook='playbooks/site.yml', + inventory='inventories/prod.ini', + private_data_dir='./data', + dag_file_path=str(dag_file) + ) + + playbook_path, inventory_path, data_dir = task.get_absolute_paths() + + assert playbook_path == str(Path(temp_dir) / 'playbooks' / 'site.yml') + assert inventory_path == str(Path(temp_dir) / 'inventories' / 'prod.ini') + assert data_dir == str(Path(temp_dir) / 'data') diff --git a/tests/Vecchi_test/test_api_client.py b/tests/Vecchi_test/test_api_client.py new file mode 100644 index 0000000..387813f --- /dev/null +++ b/tests/Vecchi_test/test_api_client.py @@ -0,0 +1,528 @@ +#!/usr/bin/env python3 +""" +Comprehensive test suite for the Maestro API client. + +This test suite ensures high coverage of all API client methods and error scenarios, +using the responses library to mock HTTP calls. +""" + +import pytest +import responses +import requests +import json +from unittest.mock import patch, Mock +from maestro.client.api_client import MaestroAPIClient + + +class TestMaestroAPIClient: + """Test suite for MaestroAPIClient.""" + + def setup_method(self): + """Set up test fixtures.""" + self.client = MaestroAPIClient(base_url="http://localhost:8000") + self.mock_dag_response = { + "dag_id": "test-dag", + "execution_id": "exec-123", + "status": "submitted", + "submitted_at": "2025-07-18T18:33:35Z" + } + + @pytest.fixture + def mock_responses(self): + """Mock HTTP responses.""" + with responses.RequestsMock() as rsps: + yield rsps + + # Constructor Tests + @pytest.mark.unit + def test_client_initialization(self): + """Test client initialization with default and custom values.""" + # Default initialization + client = MaestroAPIClient() + assert client.base_url == "http://localhost:8000" + assert client.timeout == 30 + + # Custom initialization + client = MaestroAPIClient(base_url="http://custom:9000", timeout=60) + assert client.base_url == "http://custom:9000" + assert client.timeout == 60 + + @pytest.mark.unit + def test_client_initialization_strips_trailing_slash(self): + """Test that trailing slashes are stripped from base_url.""" + client = MaestroAPIClient(base_url="http://localhost:8000/") + assert client.base_url == "http://localhost:8000" + + # Health Check Tests + @pytest.mark.unit + def test_health_check_success(self, mock_responses): + """Test successful health check.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/", + json={"status": "healthy", "timestamp": "2025-07-18T18:33:35Z"}, + status=200 + ) + + result = self.client.health_check() + + assert result["status"] == "healthy" + assert result["timestamp"] == "2025-07-18T18:33:35Z" + assert len(mock_responses.calls) == 1 + + @pytest.mark.unit + def test_health_check_connection_error(self, mock_responses): + """Test health check with connection error.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/", + body=requests.exceptions.ConnectionError("Connection failed") + ) + + with pytest.raises(ConnectionError) as exc_info: + self.client.health_check() + + assert "Could not connect to Maestro server" in str(exc_info.value) + + # Get DAG Status Tests + @pytest.mark.unit + def test_get_dag_status_success(self, mock_responses): + """Test successful DAG status retrieval.""" + status_response = { + "execution_id": "exec-123", + "status": "running", + "started_at": "2025-07-18T18:33:35Z", + "completed_at": None, + "thread_id": "thread-1", + "tasks": [] + } + + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/dags/test-dag/status", + json=status_response, + status=200 + ) + + result = self.client.get_dag_status("test-dag") + + assert result == status_response + assert len(mock_responses.calls) == 1 + + @pytest.mark.unit + def test_get_dag_status_with_execution_id(self, mock_responses): + """Test DAG status retrieval with execution ID.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/dags/test-dag/status", + json={"execution_id": "exec-specific"}, + status=200 + ) + + result = self.client.get_dag_status("test-dag", execution_id="exec-specific") + + # Check that execution_id parameter was passed + assert mock_responses.calls[0].request.url.endswith("?execution_id=exec-specific") + + @pytest.mark.unit + def test_get_dag_status_not_found(self, mock_responses): + """Test DAG status retrieval for non-existent DAG.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/dags/nonexistent/status", + status=404 + ) + + with pytest.raises(FileNotFoundError) as exc_info: + self.client.get_dag_status("nonexistent") + + assert "Resource not found" in str(exc_info.value) + + # Get DAG Logs Tests + @pytest.mark.unit + def test_get_dag_logs_success(self, mock_responses): + """Test successful DAG logs retrieval.""" + logs_response = { + "logs": [ + { + "timestamp": "2025-07-18T18:33:35Z", + "level": "INFO", + "task_id": "task-1", + "message": "Task started" + } + ], + "total_count": 1 + } + + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/logs/test-dag", + json=logs_response, + status=200 + ) + + result = self.client.get_dag_logs_v1("test-dag") + + assert result == logs_response + assert len(mock_responses.calls) == 1 + + @pytest.mark.unit + def test_get_dag_logs_with_filters(self, mock_responses): + """Test DAG logs retrieval with filters.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/logs/test-dag", + json={"logs": [], "total_count": 0}, + status=200 + ) + + result = self.client.get_dag_logs_v1( + "test-dag", + execution_id="exec-123", + limit=50, + task_filter="specific-task", + level_filter="ERROR" + ) + + # Check URL parameters + request_url = mock_responses.calls[0].request.url + assert "limit=50" in request_url + assert "execution_id=exec-123" in request_url + assert "task_filter=specific-task" in request_url + assert "level_filter=ERROR" in request_url + + # Stream DAG Logs Tests + @pytest.mark.unit + def test_stream_dag_logs_success(self, mock_responses): + """Test successful DAG logs streaming.""" + # Mock SSE stream response + stream_data = [ + "data: {\"timestamp\": \"2025-07-18T18:33:35Z\", \"level\": \"INFO\", \"task_id\": \"task-1\", \"message\": \"Log 1\"}\n", + "data: {\"timestamp\": \"2025-07-18T18:33:36Z\", \"level\": \"ERROR\", \"task_id\": \"task-2\", \"message\": \"Log 2\"}\n" + ] + + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/dags/test-dag/attach", + body="".join(stream_data), + status=200, + stream=True + ) + + logs = list(self.client.stream_dag_logs_v1("test-dag")) + + assert len(logs) == 2 + assert logs[0]["message"] == "Log 1" + assert logs[1]["message"] == "Log 2" + + @pytest.mark.unit + def test_stream_dag_logs_with_filters(self, mock_responses): + """Test DAG logs streaming with filters.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/dags/test-dag/attach", + body="", + status=200, + stream=True + ) + + list(self.client.stream_dag_logs_v1( + "test-dag", + execution_id="exec-123", + task_filter="specific-task", + level_filter="ERROR" + )) + + # Check URL parameters + request_url = mock_responses.calls[0].request.url + assert "execution_id=exec-123" in request_url + assert "task_filter=specific-task" in request_url + assert "level_filter=ERROR" in request_url + + @pytest.mark.unit + def test_stream_dag_logs_invalid_json(self, mock_responses): + """Test DAG logs streaming with invalid JSON.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/dags/test-dag/attach", + body="data: {invalid json}\n", + status=200, + stream=True + ) + + logs = list(self.client.stream_dag_logs_v1("test-dag")) + + # Should skip invalid JSON lines + assert len(logs) == 0 + + # Get Running DAGs Tests + @pytest.mark.unit + def test_get_running_dags_success(self, mock_responses): + """Test successful running DAGs retrieval.""" + running_response = { + "running_dags": [ + { + "dag_id": "dag-1", + "execution_id": "exec-1", + "started_at": "2025-07-18T18:33:35Z", + "thread_id": 123 + } + ], + "count": 1 + } + + mock_responses.add( + responses.GET, + "http://localhost:8000/dags/running", + json=running_response, + status=200 + ) + + result = self.client.get_running_dags() + + assert result == running_response + assert len(mock_responses.calls) == 1 + + # Cancel DAG Tests + @pytest.mark.unit + def test_cancel_dag_success(self, mock_responses): + """Test successful DAG cancellation.""" + cancel_response = { + "success": True, + "message": "DAG cancelled successfully" + } + + mock_responses.add( + responses.POST, + "http://localhost:8000/dags/test-dag/cancel", + json=cancel_response, + status=200 + ) + + result = self.client.cancel_dag("test-dag") + + assert result == cancel_response + assert len(mock_responses.calls) == 1 + + @pytest.mark.unit + def test_cancel_dag_with_execution_id(self, mock_responses): + """Test DAG cancellation with execution ID.""" + mock_responses.add( + responses.POST, + "http://localhost:8000/dags/test-dag/cancel", + json={"success": True}, + status=200 + ) + + result = self.client.cancel_dag("test-dag", execution_id="exec-123") + + # Check that execution_id parameter was passed + assert mock_responses.calls[0].request.url.endswith("?execution_id=exec-123") + + # Validate DAG Tests + @pytest.mark.unit + def test_validate_dag_success(self, mock_responses): + """Test successful DAG validation.""" + validate_response = { + "valid": True, + "dag_id": "test-dag", + "tasks": [ + { + "task_id": "task-1", + "type": "python", + "dependencies": [] + } + ], + "total_tasks": 1 + } + + mock_responses.add( + responses.POST, + "http://localhost:8000/dags/validate", + json=validate_response, + status=200 + ) + + result = self.client.validate_dag("/path/to/dag.yaml") + + assert result == validate_response + assert len(mock_responses.calls) == 1 + + # Verify request payload + request_body = json.loads(mock_responses.calls[0].request.body) + assert request_body["dag_file_path"] == "/path/to/dag.yaml" + + @pytest.mark.unit + def test_validate_dag_invalid(self, mock_responses): + """Test DAG validation failure.""" + validate_response = { + "valid": False, + "error": "Invalid DAG structure" + } + + mock_responses.add( + responses.POST, + "http://localhost:8000/dags/validate", + json=validate_response, + status=200 + ) + + result = self.client.validate_dag("/path/to/invalid.yaml") + + assert result == validate_response + assert not result["valid"] + + # Cleanup Tests + @pytest.mark.unit + def test_cleanup_old_executions_success(self, mock_responses): + """Test successful cleanup of old executions.""" + cleanup_response = { + "message": "Cleaned up 5 old executions" + } + + mock_responses.add( + responses.DELETE, + "http://localhost:8000/dags/cleanup", + json=cleanup_response, + status=200 + ) + + result = self.client.cleanup_old_executions(days=7) + + assert result == cleanup_response + assert len(mock_responses.calls) == 1 + + # Check that days parameter was passed + assert mock_responses.calls[0].request.url.endswith("?days=7") + + @pytest.mark.unit + def test_cleanup_old_executions_default_days(self, mock_responses): + """Test cleanup with default days parameter.""" + mock_responses.add( + responses.DELETE, + "http://localhost:8000/dags/cleanup", + json={"message": "Cleaned up"}, + status=200 + ) + + result = self.client.cleanup_old_executions() + + # Check that default days=30 was used + assert mock_responses.calls[0].request.url.endswith("?days=30") + + + @pytest.mark.unit + def test_list_dags_with_status_filter(self, mock_responses): + """Test DAG listing with status filter.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/v1/dags", + json={"dags": [], "count": 0, "title": "Running DAGs"}, + status=200 + ) + + result = self.client.list_dags(status_filter="running") + + # Check that status parameter was passed + assert mock_responses.calls[0].request.url.endswith("?status=active") + + # Server Status Tests + @pytest.mark.unit + def test_is_server_running_true(self, mock_responses): + """Test server running check returns True.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/", + json={"status": "healthy"}, + status=200 + ) + + result = self.client.is_server_running() + + assert result is True + + @pytest.mark.unit + def test_is_server_running_false(self, mock_responses): + """Test server running check returns False.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/", + body=requests.exceptions.ConnectionError("Connection failed") + ) + + result = self.client.is_server_running() + + assert result is False + + @pytest.mark.unit + def test_wait_for_server_success(self, mock_responses): + """Test successful server wait.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/", + json={"status": "healthy"}, + status=200 + ) + + result = self.client.wait_for_server(max_wait_time=1) + + assert result is True + + @pytest.mark.unit + def test_wait_for_server_timeout(self, mock_responses): + """Test server wait timeout.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/", + body=requests.exceptions.ConnectionError("Connection failed") + ) + + result = self.client.wait_for_server(max_wait_time=1) + + assert result is False + + # Error Handling Tests + @pytest.mark.unit + def test_make_request_timeout_error(self, mock_responses): + """Test timeout error handling.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/", + body=requests.exceptions.Timeout("Request timed out") + ) + + with pytest.raises(TimeoutError) as exc_info: + self.client.health_check() + + assert "Request to http://localhost:8000/ timed out" in str(exc_info.value) + + @pytest.mark.unit + def test_make_request_http_error(self, mock_responses): + """Test HTTP error handling.""" + mock_responses.add( + responses.GET, + "http://localhost:8000/", + json={"error": "Server error"}, + status=500 + ) + + with pytest.raises(RuntimeError) as exc_info: + self.client.health_check() + + assert "HTTP 500" in str(exc_info.value) + + @pytest.mark.unit + def test_make_request_custom_timeout(self): + """Test client with custom timeout.""" + client = MaestroAPIClient(timeout=5) + + with patch.object(client.session, 'request') as mock_request: + mock_request.return_value.json.return_value = {"status": "ok"} + mock_request.return_value.raise_for_status.return_value = None + + client.health_check() + + # Verify that custom timeout was used + mock_request.assert_called_once_with( + "GET", "http://localhost:8000/", timeout=5 + ) diff --git a/tests/Vecchi_test/test_cli_attach.py b/tests/Vecchi_test/test_cli_attach.py new file mode 100644 index 0000000..5633c79 --- /dev/null +++ b/tests/Vecchi_test/test_cli_attach.py @@ -0,0 +1,399 @@ +#!/usr/bin/env python3 +""" +Test suite for the attach command and streaming functionality. + +This test suite focuses on the complex attach command which involves +signal handling, streaming, and user interaction. +""" + +import pytest +import signal +import time +import unittest.mock +from unittest.mock import patch, ANY +from typer.testing import CliRunner +from maestro.client.cli import app + + +class TestAttachCommand: + """Test suite for the attach command.""" + + def setup_method(self): + """Set up test fixtures.""" + self.runner = CliRunner() + + @pytest.fixture + def mock_api_client(self, mocker): + """Mock API client.""" + return mocker.patch('maestro.cli_client.api_client', autospec=True) + + @pytest.fixture + def mock_check_server(self, mocker): + """Mock server connection check.""" + return mocker.patch('maestro.cli_client.check_server_connection') + + @pytest.fixture + def mock_signal(self, mocker): + """Mock signal handling.""" + return mocker.patch('maestro.cli_client.signal') + + @pytest.mark.unit + def test_attach_command_success(self, mock_api_client, mock_check_server, mock_signal): + """Test successful attach command execution.""" + # Mock streaming logs generator + mock_logs = [ + { + "timestamp": "2025-07-18T18:33:35.123456Z", + "level": "INFO", + "task_id": "task-1", + "message": "Starting task" + }, + { + "timestamp": "2025-07-18T18:33:36.123456Z", + "level": "ERROR", + "task_id": "task-2", + "message": "Task failed" + } + ] + + mock_api_client.stream_dag_logs.return_value = iter(mock_logs) + + result = self.runner.invoke(app, ['attach', 'test-dag']) + + assert result.exit_code == 0 + assert "Attaching to live logs for DAG: test-dag" in result.output + assert "Press Ctrl+C to detach" in result.output + assert "Starting task" in result.output + assert "Task failed" in result.output + + # Verify the API was called correctly + mock_api_client.stream_dag_logs.assert_called_once_with('test-dag', None, None, None) + + # Verify server connection was checked + mock_check_server.assert_called_once() + + # Signal handling test (optional - may not be implemented yet) + # This test will pass whether signal handling is implemented or not + if mock_signal.signal.called: + print("Signal handling is implemented") + calls = mock_signal.signal.call_args_list + print(f"Signal calls: {calls}") + # If it's implemented, verify basic functionality + assert len(calls) >= 1, "At least one signal handler should be registered" + else: + print("Signal handling not yet implemented - this is OK for now") + + @pytest.mark.unit + def test_attach_command_with_filters(self, mock_api_client, mock_check_server, mock_signal): + """Test attach command with filters.""" + mock_api_client.stream_dag_logs.return_value = iter([]) + + result = self.runner.invoke(app, [ + 'attach', 'test-dag', + '--execution-id', 'exec-123', + '--task', 'specific-task', + '--level', 'ERROR' + ]) + + assert result.exit_code == 0 + mock_api_client.stream_dag_logs.assert_called_once_with( + 'test-dag', 'exec-123', 'specific-task', 'ERROR' + ) + + @pytest.mark.unit + def test_attach_command_with_execution_id(self, mock_api_client, mock_check_server, mock_signal): + """Test attach command with execution ID.""" + mock_api_client.stream_dag_logs.return_value = iter([]) + + result = self.runner.invoke(app, [ + 'attach', 'test-dag', + '--execution-id', 'exec-specific' + ]) + + assert result.exit_code == 0 + assert "Execution ID: exec-specific" in result.output + + @pytest.mark.unit + def test_attach_command_stream_error(self, mock_api_client, mock_check_server, mock_signal): + """Test attach command with stream error.""" + mock_logs = [ + {"error": "Stream connection lost"} + ] + + mock_api_client.stream_dag_logs.return_value = iter(mock_logs) + + result = self.runner.invoke(app, ['attach', 'test-dag']) + + assert result.exit_code == 0 + assert "Stream error: Stream connection lost" in result.output + + @pytest.mark.unit + def test_attach_command_keyboard_interrupt(self, mock_api_client, mock_check_server, mock_signal): + """Test attach command with keyboard interrupt.""" + + def mock_stream_logs(*args, **kwargs): + """Mock streaming that raises KeyboardInterrupt.""" + yield {"timestamp": "2025-07-18T18:33:35Z", "level": "INFO", "task_id": "task-1", "message": "Starting"} + raise KeyboardInterrupt("User interrupted") + + mock_api_client.stream_dag_logs.side_effect = mock_stream_logs + + result = self.runner.invoke(app, ['attach', 'test-dag']) + + assert result.exit_code == 0 + assert "Detached from log stream" in result.output + + @pytest.mark.unit + def test_attach_command_general_exception(self, mock_api_client, mock_check_server, mock_signal): + """Test attach command with general exception.""" + mock_api_client.stream_dag_logs.side_effect = RuntimeError("Stream error") + + result = self.runner.invoke(app, ['attach', 'test-dag']) + + assert result.exit_code == 1 + assert "Error: Stream error" in result.output + + @pytest.mark.unit + def test_attach_command_custom_server_url(self, mock_api_client, mock_check_server, mock_signal): + """Test attach command with custom server URL.""" + mock_api_client.stream_dag_logs.return_value = iter([]) + + result = self.runner.invoke(app, [ + 'attach', 'test-dag', + '--server', 'http://custom:9000' + ]) + + assert result.exit_code == 0 + # Verify that the API client base_url was updated + assert mock_api_client.base_url == 'http://custom:9000' + + @pytest.mark.unit + def test_attach_command_log_timestamp_parsing(self, mock_api_client, mock_check_server, mock_signal): + """Test attach command log timestamp parsing.""" + mock_logs = [ + { + "timestamp": "2025-07-18T18:33:35.123456Z", + "level": "INFO", + "task_id": "task-1", + "message": "Test with full timestamp" + }, + { + "timestamp": "18:33:35", + "level": "DEBUG", + "task_id": "task-2", + "message": "Test with time only" + } + ] + + mock_api_client.stream_dag_logs.return_value = iter(mock_logs) + + result = self.runner.invoke(app, ['attach', 'test-dag']) + + assert result.exit_code == 0 + assert "18:33:35" in result.output + assert "Test with full timestamp" in result.output + assert "Test with time only" in result.output + + @pytest.mark.unit + def test_attach_command_log_level_styling(self, mock_api_client, mock_check_server, mock_signal): + """Test attach command log level styling.""" + mock_logs = [ + { + "timestamp": "2025-07-18T18:33:35Z", + "level": "ERROR", + "task_id": "task-1", + "message": "Error message" + }, + { + "timestamp": "2025-07-18T18:33:36Z", + "level": "WARNING", + "task_id": "task-2", + "message": "Warning message" + }, + { + "timestamp": "2025-07-18T18:33:37Z", + "level": "INFO", + "task_id": "task-3", + "message": "Info message" + }, + { + "timestamp": "2025-07-18T18:33:38Z", + "level": "DEBUG", + "task_id": "task-4", + "message": "Debug message" + }, + { + "timestamp": "2025-07-18T18:33:39Z", + "level": "UNKNOWN", + "task_id": "task-5", + "message": "Unknown level message" + } + ] + + mock_api_client.stream_dag_logs.return_value = iter(mock_logs) + + result = self.runner.invoke(app, ['attach', 'test-dag']) + + assert result.exit_code == 0 + # All messages should be present + assert "Error message" in result.output + assert "Warning message" in result.output + assert "Info message" in result.output + assert "Debug message" in result.output + + @pytest.mark.unit + def test_attach_command_basic_functionality(self, mock_api_client, mock_check_server, mock_signal): + """Test basic attach command functionality without signal handling.""" + # Mock streaming logs generator + mock_logs = [ + { + "timestamp": "2025-07-18T18:33:35.123456Z", + "level": "INFO", + "task_id": "task-1", + "message": "Basic functionality test" + } + ] + + mock_api_client.stream_dag_logs.return_value = iter(mock_logs) + + result = self.runner.invoke(app, ['attach', 'test-dag']) + + # Test the core functionality + assert result.exit_code == 0 + assert "Attaching to live logs for DAG: test-dag" in result.output + assert "Basic functionality test" in result.output + + # Verify the API was called correctly + mock_api_client.stream_dag_logs.assert_called_once_with('test-dag', None, None, None) + + # Verify server connection was checked + mock_check_server.assert_called_once() + + +class TestAttachCommandSignalHandling: + """Separate test class for signal handling to isolate these tests.""" + + def setup_method(self): + """Set up test fixtures.""" + self.runner = CliRunner() + + @pytest.fixture + def mock_api_client(self, mocker): + """Mock API client.""" + return mocker.patch('maestro.cli_client.api_client', autospec=True) + + @pytest.fixture + def mock_check_server(self, mocker): + """Mock server connection check.""" + return mocker.patch('maestro.cli_client.check_server_connection') + + @pytest.fixture + def mock_signal(self, mocker): + """Mock signal handling.""" + return mocker.patch('maestro.cli_client.signal') + + @pytest.mark.unit + def test_signal_handling_if_implemented(self, mock_api_client, mock_check_server, mock_signal): + """Test signal handling if it's implemented in the attach command.""" + mock_api_client.stream_dag_logs.return_value = iter([]) + + result = self.runner.invoke(app, ['attach', 'test-dag']) + + assert result.exit_code == 0 + + # Check if signal handling is implemented + if mock_signal.signal.called: + # If signal handling is implemented, verify it + calls = mock_signal.signal.call_args_list + + # Check that at least one signal was registered + assert len(calls) >= 1, "At least one signal handler should be registered" + + # Check for SIGINT specifically + sigint_calls = [call for call in calls if call[0][0] == signal.SIGINT] + if sigint_calls: + # Verify the handler is callable + handler = sigint_calls[0][0][1] + assert callable(handler), "SIGINT handler should be callable" + + print(f"Signal handling is implemented. Registered signals: {[call[0][0] for call in calls]}") + else: + print("Signal handling not implemented in attach command") + # This is fine - the test passes either way + + +class TestStreamingIntegration: + """Integration tests for streaming functionality.""" + + @pytest.mark.integration + def test_attach_command_integration(self, mocker): + """Test attach command integration with API client.""" + runner = CliRunner() + + # Mock the API client's stream method to return a controlled stream + mock_api_client = mocker.patch('maestro.cli_client.api_client') + mock_check_server = mocker.patch('maestro.cli_client.check_server_connection') + + # Create a controlled stream that ends after a few messages + def controlled_stream(*args, **kwargs): + messages = [ + { + "timestamp": "2025-07-18T18:33:35Z", + "level": "INFO", + "task_id": "task-1", + "message": "Task started" + }, + { + "timestamp": "2025-07-18T18:33:36Z", + "level": "INFO", + "task_id": "task-1", + "message": "Task running" + }, + { + "timestamp": "2025-07-18T18:33:37Z", + "level": "INFO", + "task_id": "task-1", + "message": "Task completed" + } + ] + + for msg in messages: + yield msg + + mock_api_client.stream_dag_logs.side_effect = controlled_stream + + result = runner.invoke(app, ['attach', 'test-dag-integration']) + + assert result.exit_code == 0 + assert "Attaching to live logs for DAG: test-dag-integration" in result.output + assert "Task started" in result.output + assert "Task running" in result.output + assert "Task completed" in result.output + + @pytest.mark.slow + def test_attach_command_long_running_stream(self, mocker): + """Test attach command with a longer running stream.""" + runner = CliRunner() + + mock_api_client = mocker.patch('maestro.cli_client.api_client') + mock_check_server = mocker.patch('maestro.cli_client.check_server_connection') + + # Simulate a longer running stream + def long_stream(*args, **kwargs): + for i in range(10): + yield { + "timestamp": f"2025-07-18T18:33:{35 + i:02d}Z", + "level": "INFO", + "task_id": f"task-{i + 1}", + "message": f"Processing item {i + 1}" + } + # Small delay to simulate real streaming + time.sleep(0.01) + + mock_api_client.stream_dag_logs.side_effect = long_stream + + result = runner.invoke(app, ['attach', 'test-dag-long']) + + assert result.exit_code == 0 + assert "Processing item 1" in result.output + assert "Processing item 10" in result.output \ No newline at end of file diff --git a/tests/Vecchi_test/test_cli_client.py b/tests/Vecchi_test/test_cli_client.py new file mode 100644 index 0000000..9428d88 --- /dev/null +++ b/tests/Vecchi_test/test_cli_client.py @@ -0,0 +1,422 @@ +#!/usr/bin/env python3 +""" +Comprehensive test suite for the Maestro CLI client. + +This test suite ensures high coverage of all CLI commands and edge cases, +using mocks to isolate the client from server dependencies. +""" + +import pytest +from unittest.mock import patch, Mock, MagicMock +from typer.testing import CliRunner +from maestro.client.cli import app, check_server_connection +import tempfile +import os +from io import StringIO +import typer + + +class TestCliClient: + """Test suite for CLI client commands.""" + + def setup_method(self): + """Set up test fixtures.""" + self.runner = CliRunner() + self.mock_response_data = { + 'dag_id': 'test-dag-123', + 'execution_id': 'exec-456', + 'status': 'submitted', + 'submitted_at': '2025-07-18T18:33:35Z' + } + + @pytest.fixture + def mock_api_client(self, mocker): + """Mock API client with all methods.""" + return mocker.patch('maestro.cli_client.api_client', autospec=True) + + @pytest.fixture + def mock_check_server(self, mocker): + """Mock server connection check.""" + return mocker.patch('maestro.cli_client.check_server_connection') + + @pytest.fixture + def temp_dag_file(self): + """Create a temporary DAG file for testing.""" + with tempfile.NamedTemporaryFile(mode='w', suffix='.yaml', delete=False) as f: + f.write("dag_id: test-dag\ntasks: []") + yield f.name + os.unlink(f.name) + + + + # Status Command Tests + @pytest.mark.unit + def test_status_command_success(self, mock_api_client, mock_check_server): + """Test successful status retrieval.""" + mock_api_client.get_dag_status.return_value = { + 'execution_id': 'exec-123', + 'status': 'running', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': None, + 'thread_id': 'thread-1', + 'tasks': [{ + 'task_id': 'task-1', + 'status': 'completed', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': '2025-07-18T18:34:35Z' + }] + } + + result = self.runner.invoke(app, ['status', 'test-dag']) + + assert result.exit_code == 0 + assert "DAG Status: test-dag" in result.output + assert "running" in result.output + assert "task-1" in result.output + + @pytest.mark.unit + def test_status_command_with_execution_id(self, mock_api_client, mock_check_server): + """Test status retrieval with specific execution ID.""" + mock_api_client.get_dag_status.return_value = { + 'execution_id': 'exec-specific', + 'status': 'completed', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': '2025-07-18T18:34:35Z', + 'thread_id': None, + 'tasks': [] + } + + result = self.runner.invoke(app, ['status', 'test-dag', '--execution-id', 'exec-specific']) + + assert result.exit_code == 0 + mock_api_client.get_dag_status.assert_called_once_with('test-dag', 'exec-specific') + + @pytest.mark.unit + def test_status_command_not_found(self, mock_api_client, mock_check_server): + """Test status retrieval for non-existent DAG.""" + mock_api_client.get_dag_status.side_effect = FileNotFoundError("DAG not found") + + result = self.runner.invoke(app, ['status', 'nonexistent-dag']) + + assert result.exit_code == 1 + assert "DAG execution not found" in result.output + + # Logs Command Tests + @pytest.mark.unit + def test_logs_command_success(self, mock_api_client, mock_check_server): + """Test successful logs retrieval.""" + mock_api_client.get_dag_logs_v1.return_value = [ + { + 'timestamp': '2025-07-18T18:33:35.123456Z', + 'level': 'INFO', + 'task_id': 'task-1', + 'message': 'Task started' + }, + { + 'timestamp': '2025-07-18T18:33:36.123456Z', + 'level': 'ERROR', + 'task_id': 'task-2', + 'message': 'Task failed' + } + ] + + result = self.runner.invoke(app, ['log', 'test-dag']) + + assert "Task started" in result.output + assert "Task failed" in result.output + + @pytest.mark.unit + def test_logs_command_with_filters(self, mock_api_client, mock_check_server): + """Test logs retrieval with filters.""" + mock_api_client.get_dag_logs_v1.return_value = [] + + result = self.runner.invoke(app, [ + 'log', 'test-dag', + '--limit', '50', + '--task', 'specific-task', + '--level', 'ERROR' + ]) + + assert result.exit_code == 0 + mock_api_client.get_dag_logs_v1.assert_called_once_with( + 'test-dag', None, 50, 'specific-task', 'ERROR' + ) + + @pytest.mark.unit + def test_logs_command_no_logs(self, mock_api_client, mock_check_server): + """Test logs retrieval when no logs exist.""" + mock_api_client.get_dag_logs_v1.return_value = [] + + result = self.runner.invoke(app, ['log', 'test-dag']) + + assert result.exit_code == 0 + assert "No logs found for DAG: test-dag" in result.output + + # Running Command Tests + @pytest.mark.unit + def test_running_command_success(self, mock_api_client, mock_check_server): + """Test successful running DAGs retrieval.""" + mock_api_client.list_dags_v1.return_value = [ + { + 'dag_id': 'dag-1', + 'execution_id': 'exec-1', + 'status': 'running', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': None, + 'thread_id': 123 + }, + { + 'dag_id': 'dag-2', + 'execution_id': 'exec-2', + 'status': 'running', + 'started_at': '2025-07-18T18:34:35Z', + 'completed_at': None, + 'thread_id': 456 + } + ] + + result = self.runner.invoke(app, ['ls', '--filter', 'active']) + + assert result.exit_code == 0 + assert "Maestro DAGs" in result.output + assert "dag-1" in result.output + assert "dag-2" in result.output + assert "Total DAGs: 2" in result.output + + @pytest.mark.unit + def test_running_command_no_running_dags(self, mock_api_client, mock_check_server): + """Test running DAGs retrieval when none are running.""" + mock_api_client.list_dags_v1.return_value = [] + + result = self.runner.invoke(app, ['ls', '--filter', 'active']) + + assert result.exit_code == 0 + assert "No DAGs found with filter 'active'" in result.output + + # Cancel Command Tests + @pytest.mark.unit + def test_cancel_command_success(self, mock_api_client, mock_check_server): + """Test successful DAG cancellation.""" + mock_api_client.stop_dag.return_value = { + 'message': 'DAG cancelled successfully' + } + + result = self.runner.invoke(app, ['stop', 'test-dag']) + + assert result.exit_code == 0 + assert "DAG cancelled successfully" in result.output + + @pytest.mark.unit + def test_cancel_command_not_running(self, mock_api_client, mock_check_server): + """Test cancellation of non-running DAG.""" + mock_api_client.stop_dag.return_value = { + 'message': 'DAG is not running' + } + + result = self.runner.invoke(app, ['stop', 'test-dag']) + + assert result.exit_code == 0 + assert "DAG is not running" in result.output + + # Validate Command Tests + @pytest.mark.unit + def test_validate_command_success(self, mock_api_client, mock_check_server, temp_dag_file): + """Test successful DAG validation.""" + mock_api_client.validate_dag.return_value = { + 'valid': True, + 'dag_id': 'test-dag', + 'tasks': [ + { + 'task_id': 'task-1', + 'type': 'python', + 'dependencies': [] + }, + { + 'task_id': 'task-2', + 'type': 'shell', + 'dependencies': ['task-1'] + } + ], + 'total_tasks': 2 + } + + result = self.runner.invoke(app, ['validate', temp_dag_file]) + + assert result.exit_code == 0 + assert "✓ DAG is valid" in result.output + assert "test-dag" in result.output + assert "task-1" in result.output + assert "task-2" in result.output + assert "Total tasks: 2" in result.output + + @pytest.mark.unit + def test_validate_command_invalid(self, mock_api_client, mock_check_server, temp_dag_file): + """Test DAG validation failure.""" + mock_api_client.validate_dag.return_value = { + 'valid': False, + 'error': 'Invalid DAG structure' + } + + result = self.runner.invoke(app, ['validate', temp_dag_file]) + + assert result.exit_code == 1 + assert "✗ DAG validation failed" in result.output + assert "Invalid DAG structure" in result.output + + # Cleanup Command Tests + @pytest.mark.unit + def test_cleanup_command_success(self, mock_api_client, mock_check_server): + """Test successful cleanup.""" + mock_api_client.cleanup_old_executions.return_value = { + 'message': 'Cleaned up 5 old executions' + } + + result = self.runner.invoke(app, ['cleanup', '--days', '7']) + + assert result.exit_code == 0 + assert "Cleaned up 5 old executions" in result.output + mock_api_client.cleanup_old_executions.assert_called_once_with(7) + + # List Command Tests + @pytest.mark.unit + def test_list_command_success(self, mock_api_client, mock_check_server): + """Test successful DAG listing.""" + mock_api_client.list_dags_v1.return_value = [ + { + 'dag_id': 'dag-1', + 'execution_id': 'exec-1', + 'status': 'completed', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': '2025-07-18T18:34:35Z', + 'thread_id': 123 + }, + { + 'dag_id': 'dag-2', + 'execution_id': 'exec-2', + 'status': 'running', + 'started_at': '2025-07-18T18:35:35Z', + 'completed_at': None, + 'thread_id': 456 + } + ] + + result = self.runner.invoke(app, ['ls']) + + assert result.exit_code == 0 + assert "Maestro DAGs" in result.output + assert "dag-1" in result.output + assert "dag-2" in result.output + assert "Total DAGs: 2" in result.output + + @pytest.mark.unit + def test_list_command_with_status_filter(self, mock_api_client, mock_check_server): + """Test DAG listing with status filter.""" + mock_api_client.list_dags_v1.return_value = [] + + result = self.runner.invoke(app, ['ls', '--filter', 'running']) + + assert result.exit_code == 0 + mock_api_client.list_dags_v1.assert_called_once_with('running') + + @pytest.mark.unit + def test_list_command_active_flag(self, mock_api_client, mock_check_server): + """Test DAG listing with active flag.""" + mock_api_client.list_dags_v1.return_value = [] + + result = self.runner.invoke(app, ['ls', '--filter', 'active']) + + assert result.exit_code == 0 + mock_api_client.list_dags_v1.assert_called_once_with('active') + + @pytest.mark.unit + def test_list_command_no_dags(self, mock_api_client, mock_check_server): + """Test DAG listing when no DAGs exist.""" + mock_api_client.list_dags_v1.return_value = [] + + result = self.runner.invoke(app, ['ls']) + + assert result.exit_code == 0 + assert "No DAGs found" in result.output + + # Server Commands Tests + @pytest.mark.unit + def test_server_status_command_running(self, mock_api_client, mock_check_server): + """Test server status when running.""" + mock_api_client.health_check.return_value = { + 'status': 'healthy', + 'timestamp': '2025-07-18T18:33:35Z' + } + + result = self.runner.invoke(app, ['server', 'status']) + + assert result.exit_code == 0 + assert "Server is running" in result.output + assert "healthy" in result.output + + @pytest.mark.unit + def test_server_status_command_not_running(self, mock_api_client, mock_check_server): + """Test server status when not running.""" + mock_api_client.health_check.side_effect = ConnectionError("Connection failed") + + result = self.runner.invoke(app, ['server', 'status']) + + assert result.exit_code == 1 + assert "Server is not running" in result.output + + @pytest.mark.unit + def test_server_start_command_daemon(self, mock_api_client, mock_check_server, mocker): + """Test server start in daemon mode.""" + mock_popen = mocker.patch('maestro.cli_client.subprocess.Popen') + mock_api_client.wait_for_server.return_value = True + + result = self.runner.invoke(app, ['server', 'start', '--daemon']) + + assert result.exit_code == 0 + assert "Maestro server started" in result.output + mock_popen.assert_called_once() + + @pytest.mark.unit + def test_server_start_command_daemon_fail(self, mock_api_client, mock_check_server, mocker): + """Test server start daemon failure.""" + mock_popen = mocker.patch('maestro.cli_client.subprocess.Popen') + mock_api_client.wait_for_server.return_value = False + + result = self.runner.invoke(app, ['server', 'start', '--daemon']) + + assert result.exit_code == 1 + assert "Failed to start server" in result.output + + @pytest.mark.unit + def test_server_stop_command(self, mock_api_client, mock_check_server): + """Test server stop command.""" + result = self.runner.invoke(app, ['server', 'stop']) + + assert result.exit_code == 0 + assert "Server stop command not implemented" in result.output + + +class TestServerConnection: + """Test suite for server connection functionality.""" + + @pytest.mark.unit + def test_check_server_connection_success(self, mocker): + """Test successful server connection check.""" + mock_api_client = mocker.patch('maestro.cli_client.api_client') + mock_api_client.is_server_running.return_value = True + + # Should not raise any exception + check_server_connection() + mock_api_client.is_server_running.assert_called_once() + + @pytest.mark.unit + def test_check_server_connection_failure(self, mocker): + """Test server connection failure.""" + mock_api_client = mocker.patch('maestro.cli_client.api_client') + mock_api_client.is_server_running.return_value = False + mock_console = mocker.patch('maestro.cli_client.console') + + with pytest.raises(typer.Exit) as exc_info: + check_server_connection() + + assert exc_info.value.exit_code == 1 + mock_console.print.assert_called() \ No newline at end of file diff --git a/tests/Vecchi_test/test_cli_integration.py b/tests/Vecchi_test/test_cli_integration.py new file mode 100644 index 0000000..b1c997f --- /dev/null +++ b/tests/Vecchi_test/test_cli_integration.py @@ -0,0 +1,520 @@ +#!/usr/bin/env python3 +""" +Integration tests for the Maestro CLI client. + +These tests focus on testing the interactions between different components +and edge cases that might occur in real usage scenarios. +""" + +import pytest +import tempfile +import os +from unittest.mock import patch, Mock +from typer.testing import CliRunner +from maestro.client.cli import app + + +class TestCliIntegration: + """Integration tests for CLI client.""" + + def setup_method(self): + """Set up test fixtures.""" + self.runner = CliRunner() + + @pytest.fixture + def mock_api_client(self, mocker): + """Mock API client.""" + return mocker.patch('maestro.cli_client.api_client', autospec=True) + + @pytest.fixture + def mock_check_server(self, mocker): + """Mock server connection check.""" + return mocker.patch('maestro.cli_client.check_server_connection') + + @pytest.fixture + def temp_dag_file(self): + """Create a temporary DAG file.""" + with tempfile.NamedTemporaryFile(mode='w', suffix='.yaml', delete=False) as f: + f.write(""" +dag_id: test-integration-dag +tasks: + - task_id: task1 + type: shell + command: echo "Hello World" + - task_id: task2 + type: shell + command: echo "Task 2" + depends_on: [task1] +""") + yield f.name + os.unlink(f.name) + + @pytest.mark.integration + def test_full_dag_workflow(self, mock_api_client, mock_check_server, temp_dag_file): + """Test complete DAG workflow: submit -> status -> logs -> cancel.""" + # Mock responses for different stages + submit_response = { + 'dag_id': 'test-integration-dag', + 'execution_id': 'exec-integration-123', + 'status': 'submitted', + 'submitted_at': '2025-07-18T18:33:35Z' + } + + status_response = { + 'execution_id': 'exec-integration-123', + 'status': 'running', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': None, + 'thread_id': 'thread-123', + 'tasks': [ + { + 'task_id': 'task1', + 'status': 'completed', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': '2025-07-18T18:33:40Z' + }, + { + 'task_id': 'task2', + 'status': 'running', + 'started_at': '2025-07-18T18:33:40Z', + 'completed_at': None + } + ] + } + + logs_response = { + 'logs': [ + { + 'timestamp': '2025-07-18T18:33:35Z', + 'level': 'INFO', + 'task_id': 'task1', + 'message': 'Starting task1' + }, + { + 'timestamp': '2025-07-18T18:33:37Z', + 'level': 'INFO', + 'task_id': 'task1', + 'message': 'Hello World' + }, + { + 'timestamp': '2025-07-18T18:33:40Z', + 'level': 'INFO', + 'task_id': 'task2', + 'message': 'Starting task2' + } + ], + 'total_count': 3 + } + + cancel_response = { + 'success': True, + 'message': 'DAG cancelled successfully' + } + + # mock_api_client.submit_dag.return_value = submit_response #submit api removed + mock_api_client.get_dag_status.return_value = status_response + mock_api_client.get_dag_logs.return_value = logs_response + mock_api_client.cancel_dag.return_value = cancel_response + + # Step 1: Submit DAG + result = self.runner.invoke(app, ['submit', temp_dag_file]) + assert result.exit_code == 0 + assert '✓ DAG submitted successfully!' in result.output + assert 'test-integration-dag' in result.output + + # Step 2: Check status + result = self.runner.invoke(app, ['status', 'test-integration-dag']) + assert result.exit_code == 0 + assert 'DAG Status: test-integration-dag' in result.output + assert 'running' in result.output + assert 'task1' in result.output + assert 'task2' in result.output + + # Step 3: Get logs + result = self.runner.invoke(app, ['logs', 'test-integration-dag']) + assert result.exit_code == 0 + assert 'Logs: test-integration-dag' in result.output + assert 'Hello World' in result.output + assert 'Starting task2' in result.output + + # Step 4: Cancel DAG + result = self.runner.invoke(app, ['cancel', 'test-integration-dag']) + assert result.exit_code == 0 + assert 'DAG cancelled successfully' in result.output + + # Verify all API calls were made + # mock_api_client.submit_dag.assert_called_once() + mock_api_client.get_dag_status.assert_called_once() + mock_api_client.get_dag_logs.assert_called_once() + mock_api_client.cancel_dag.assert_called_once() + + @pytest.mark.integration + def test_error_handling_chain(self, mock_api_client, mock_check_server): + """Test error handling across multiple commands.""" + # Test server connection error + mock_check_server.side_effect = SystemExit(1) + + result = self.runner.invoke(app, ['status', 'test-dag']) + assert result.exit_code == 1 + + # Reset mock + mock_check_server.side_effect = None + + # Test DAG not found error + mock_api_client.get_dag_status.side_effect = FileNotFoundError("DAG not found") + + result = self.runner.invoke(app, ['status', 'nonexistent-dag']) + assert result.exit_code == 1 + assert 'DAG execution not found' in result.output + + # Test API error + mock_api_client.get_dag_status.side_effect = RuntimeError("API Error") + + result = self.runner.invoke(app, ['status', 'error-dag']) + assert result.exit_code == 1 + assert 'Error: API Error' in result.output + + @pytest.mark.integration + def test_dag_lifecycle_states(self, mock_api_client, mock_check_server, temp_dag_file): + """Test DAG through different lifecycle states.""" + # Test 1: Submitted state + # mock_api_client.submit_dag.return_value = { + # 'dag_id': 'lifecycle-dag', + # 'execution_id': 'exec-lifecycle', + # 'status': 'submitted', + # 'submitted_at': '2025-07-18T18:33:35Z' + # } + + result = self.runner.invoke(app, ['submit', temp_dag_file]) + assert result.exit_code == 0 + assert 'submitted' in result.output + + # Test 2: Running state + mock_api_client.get_dag_status.return_value = { + 'execution_id': 'exec-lifecycle', + 'status': 'running', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': None, + 'thread_id': 'thread-123', + 'tasks': [ + { + 'task_id': 'task1', + 'status': 'running', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': None + } + ] + } + + result = self.runner.invoke(app, ['status', 'lifecycle-dag']) + assert result.exit_code == 0 + assert 'running' in result.output + + # Test 3: Completed state + mock_api_client.get_dag_status.return_value = { + 'execution_id': 'exec-lifecycle', + 'status': 'completed', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': '2025-07-18T18:35:35Z', + 'thread_id': None, + 'tasks': [ + { + 'task_id': 'task1', + 'status': 'completed', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': '2025-07-18T18:35:35Z' + } + ] + } + + result = self.runner.invoke(app, ['status', 'lifecycle-dag']) + assert result.exit_code == 0 + assert 'completed' in result.output + + # Test 4: Failed state + mock_api_client.get_dag_status.return_value = { + 'execution_id': 'exec-lifecycle', + 'status': 'failed', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': '2025-07-18T18:34:35Z', + 'thread_id': None, + 'tasks': [ + { + 'task_id': 'task1', + 'status': 'failed', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': '2025-07-18T18:34:35Z' + } + ] + } + + result = self.runner.invoke(app, ['status', 'lifecycle-dag']) + assert result.exit_code == 0 + assert 'failed' in result.output + + @pytest.mark.integration + def test_multiple_dag_management(self, mock_api_client, mock_check_server): + """Test managing multiple DAGs simultaneously.""" + # Mock multiple running DAGs + mock_api_client.get_running_dags.return_value = { + 'running_dags': [ + { + 'dag_id': 'dag-1', + 'execution_id': 'exec-1', + 'started_at': '2025-07-18T18:33:35Z', + 'thread_id': 123 + }, + { + 'dag_id': 'dag-2', + 'execution_id': 'exec-2', + 'started_at': '2025-07-18T18:34:35Z', + 'thread_id': 456 + }, + { + 'dag_id': 'dag-3', + 'execution_id': 'exec-3', + 'started_at': '2025-07-18T18:35:35Z', + 'thread_id': 789 + } + ], + 'count': 3 + } + + result = self.runner.invoke(app, ['running']) + assert result.exit_code == 0 + assert 'dag-1' in result.output + assert 'dag-2' in result.output + assert 'dag-3' in result.output + assert 'Total running DAGs: 3' in result.output + + # Test list with different filters + mock_api_client.list_dags.return_value = { + 'dags': [ + { + 'dag_id': 'completed-dag', + 'execution_id': 'exec-completed', + 'status': 'completed', + 'started_at': '2025-07-18T18:30:35Z', + 'completed_at': '2025-07-18T18:32:35Z', + 'thread_id': None + } + ], + 'count': 1, + 'title': 'Completed DAGs' + } + + result = self.runner.invoke(app, ['list', '--status', 'completed']) + assert result.exit_code == 0 + assert 'completed-' in result.output # DAG ID is truncated in Rich table + assert 'Completed DAGs' in result.output + + @pytest.mark.integration + def test_dag_validation_workflow(self, mock_api_client, mock_check_server, temp_dag_file): + """Test DAG validation before submission.""" + # Test valid DAG + mock_api_client.validate_dag.return_value = { + 'valid': True, + 'dag_id': 'test-integration-dag', + 'tasks': [ + { + 'task_id': 'task1', + 'type': 'shell', + 'dependencies': [] + }, + { + 'task_id': 'task2', + 'type': 'shell', + 'dependencies': ['task1'] + } + ], + 'total_tasks': 2 + } + + result = self.runner.invoke(app, ['validate', temp_dag_file]) + assert result.exit_code == 0 + assert '✓ DAG is valid' in result.output + assert 'test-integration-dag' in result.output + assert 'task1' in result.output + assert 'task2' in result.output + assert 'Total tasks: 2' in result.output + + # Test invalid DAG + mock_api_client.validate_dag.return_value = { + 'valid': False, + 'error': 'Circular dependency detected between task1 and task2' + } + + result = self.runner.invoke(app, ['validate', temp_dag_file]) + assert result.exit_code == 1 + assert '✗ DAG validation failed' in result.output + assert 'Circular dependency detected' in result.output + + @pytest.mark.integration + def test_server_management_workflow(self, mock_api_client, mock_check_server, mocker): + """Test server management commands.""" + # Test server status when not running + mock_api_client.health_check.side_effect = ConnectionError("Connection failed") + + result = self.runner.invoke(app, ['server', 'status']) + assert result.exit_code == 1 + assert 'Server is not running' in result.output + + # Test server status when running + mock_api_client.health_check.side_effect = None + mock_api_client.health_check.return_value = { + 'status': 'healthy', + 'timestamp': '2025-07-18T18:33:35Z' + } + + result = self.runner.invoke(app, ['server', 'status']) + assert result.exit_code == 0 + assert 'Server is running' in result.output + assert 'healthy' in result.output + + # Test server start daemon mode + mock_popen = mocker.patch('maestro.cli_client.subprocess.Popen') + mock_api_client.wait_for_server.return_value = True + + result = self.runner.invoke(app, ['server', 'start', '--daemon', '--port', '9000']) + assert result.exit_code == 0 + assert 'Maestro server started' in result.output + + # Verify subprocess was called with correct arguments + mock_popen.assert_called_once() + call_args = mock_popen.call_args[0][0] + assert '--port' in call_args + assert '9000' in call_args + + @pytest.mark.integration + def test_log_filtering_and_display(self, mock_api_client, mock_check_server): + """Test log filtering and display functionality.""" + # Test logs with different levels and tasks + mock_api_client.get_dag_logs.return_value = { + 'logs': [ + { + 'timestamp': '2025-07-18T18:33:35Z', + 'level': 'DEBUG', + 'task_id': 'task1', + 'message': 'Debug message from task1' + }, + { + 'timestamp': '2025-07-18T18:33:36Z', + 'level': 'INFO', + 'task_id': 'task1', + 'message': 'Info message from task1' + }, + { + 'timestamp': '2025-07-18T18:33:37Z', + 'level': 'WARNING', + 'task_id': 'task2', + 'message': 'Warning message from task2' + }, + { + 'timestamp': '2025-07-18T18:33:38Z', + 'level': 'ERROR', + 'task_id': 'task2', + 'message': 'Error message from task2' + } + ], + 'total_count': 4 + } + + # Test with different filtering combinations + result = self.runner.invoke(app, [ + 'logs', 'test-dag', + '--limit', '10', + '--task', 'task1', + '--level', 'INFO' + ]) + + assert result.exit_code == 0 + assert 'Logs: test-dag' in result.output + + # Verify API was called with correct parameters + mock_api_client.get_dag_logs.assert_called_with( + 'test-dag', None, 10, 'task1', 'INFO' + ) + + @pytest.mark.integration + def test_cleanup_workflow(self, mock_api_client, mock_check_server): + """Test cleanup workflow with different scenarios.""" + # Test successful cleanup + mock_api_client.cleanup_old_executions.return_value = { + 'message': 'Cleaned up 15 old executions (older than 7 days)' + } + + result = self.runner.invoke(app, ['cleanup', '--days', '7']) + assert result.exit_code == 0 + assert 'Cleaned up 15 old executions' in result.output + + # Test cleanup with no old executions + mock_api_client.cleanup_old_executions.return_value = { + 'message': 'No old executions found to clean up' + } + + result = self.runner.invoke(app, ['cleanup', '--days', '30']) + assert result.exit_code == 0 + assert 'No old executions found' in result.output + + # Test cleanup with default days + result = self.runner.invoke(app, ['cleanup']) + assert result.exit_code == 0 + mock_api_client.cleanup_old_executions.assert_called_with(30) + + @pytest.mark.integration + def test_edge_cases_and_error_recovery(self, mock_api_client, mock_check_server, temp_dag_file): + """Test edge cases and error recovery scenarios.""" + # Test with very long DAG ID + long_dag_id = 'a' * 255 + mock_api_client.get_dag_status.return_value = { + 'execution_id': 'exec-long', + 'status': 'running', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': None, + 'thread_id': 'thread-123', + 'tasks': [] + } + + result = self.runner.invoke(app, ['status', long_dag_id]) + assert result.exit_code == 0 + + # Test with empty logs + mock_api_client.get_dag_logs.return_value = { + 'logs': [], + 'total_count': 0 + } + + result = self.runner.invoke(app, ['logs', 'empty-dag']) + assert result.exit_code == 0 + assert 'No logs found' in result.output + + # Test with malformed timestamp + mock_api_client.get_dag_logs.return_value = { + 'logs': [ + { + 'timestamp': 'invalid-timestamp', + 'level': 'INFO', + 'task_id': 'task1', + 'message': 'Test message' + } + ], + 'total_count': 1 + } + + result = self.runner.invoke(app, ['logs', 'malformed-dag']) + assert result.exit_code == 0 + assert 'Test message' in result.output + + # Test with missing task data + mock_api_client.get_dag_status.return_value = { + 'execution_id': 'exec-missing', + 'status': 'running', + 'started_at': '2025-07-18T18:33:35Z', + 'completed_at': None, + 'thread_id': None, + 'tasks': [] + } + + result = self.runner.invoke(app, ['status', 'missing-tasks-dag']) + assert result.exit_code == 0 + # Should handle missing tasks gracefully diff --git a/tests/Vecchi_test/test_cron_feature.py b/tests/Vecchi_test/test_cron_feature.py new file mode 100644 index 0000000..8a228a9 --- /dev/null +++ b/tests/Vecchi_test/test_cron_feature.py @@ -0,0 +1,42 @@ +import pytest +from datetime import datetime, timedelta +from maestro.shared.dag import DAG + +def test_valid_cron_expression(): + cron_schedule = "0 9 * * *" # Every day at 9:00 AM + dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + assert dag.cron_schedule == cron_schedule + +def test_invalid_cron_expression(): + invalid_cron = "invalid cron" + with pytest.raises(ValueError, match="Invalid cron expression"): + DAG(dag_id="test_dag", cron_schedule=invalid_cron) + +def test_dag_ready_to_start(): + cron_schedule = "* * * * *" # Every minute + dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + now = datetime.now() + assert dag.is_ready_to_start(now) + +def test_dag_not_ready_to_start(): + cron_schedule = "0 9 * * *" # Every day at 9:00 AM + dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + now = datetime.now().replace(hour=10) + assert not dag.is_ready_to_start(now) + +def test_dag_next_run_time(): + cron_schedule = "0 9 * * *" # Every day at 9:00 AM + dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + now = datetime(2024, 1, 1, 8, 0, 0) # 8:00 AM + next_run = dag.get_next_run_time(now) + assert next_run.hour == 9 + assert next_run.minute == 0 + +def test_dag_next_run_time_wrap_around(): + cron_schedule = "0 9 * * *" # Every day at 9:00 AM + dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + now = datetime(2024, 1, 1, 10, 0, 0) # After the run time + next_run = dag.get_next_run_time(now) + assert next_run.hour == 9 + assert next_run.day == 2 # Next day + diff --git a/tests/Vecchi_test/test_dag.py b/tests/Vecchi_test/test_dag.py new file mode 100644 index 0000000..cf2792b --- /dev/null +++ b/tests/Vecchi_test/test_dag.py @@ -0,0 +1,139 @@ + +import pytest +from datetime import datetime, timedelta + +from maestro.shared.dag import DAG +from maestro.shared.task import Task + +class DummyTask(Task): + def execute_local(self): + pass + +def test_dag_add_task(): + dag = DAG() + task = DummyTask(task_id="test_task") + dag.add_task(task) + assert "test_task" in dag.tasks + +def test_dag_validation(): + dag = DAG() + task1 = DummyTask(task_id="task1") + task2 = DummyTask(task_id="task2", dependencies=["task1"]) + dag.add_task(task1) + dag.add_task(task2) + dag.validate() + +def test_dag_cycle_detection(): + dag = DAG() + task1 = DummyTask(task_id="task1", dependencies=["task3"]) + task2 = DummyTask(task_id="task2", dependencies=["task1"]) + task3 = DummyTask(task_id="task3", dependencies=["task2"]) + dag.add_task(task1) + dag.add_task(task2) + dag.add_task(task3) + with pytest.raises(ValueError, match="DAG has a cycle."): + dag.validate() + +def test_dag_with_start_time(): + """Test DAG creation with start_time parameter.""" + start_time = datetime(2024, 1, 1, 9, 0, 0) + dag = DAG(dag_id="test_dag", start_time=start_time) + + assert dag.dag_id == "test_dag" + assert dag.start_time == start_time + +def test_dag_without_start_time(): + """Test DAG creation without start_time parameter.""" + dag = DAG(dag_id="test_dag") + + assert dag.dag_id == "test_dag" + assert dag.start_time is None + +def test_dag_is_ready_to_start_future(): + """Test DAG readiness check with future start time.""" + future_time = datetime.now() + timedelta(hours=1) + dag = DAG(dag_id="test_dag", start_time=future_time) + + assert not dag.is_ready_to_start() + assert dag.time_until_start() > 0 + +def test_dag_is_ready_to_start_past(): + """Test DAG readiness check with past start time.""" + past_time = datetime.now() - timedelta(hours=1) + dag = DAG(dag_id="test_dag", start_time=past_time) + + assert dag.is_ready_to_start() + assert dag.time_until_start() == 0 + +def test_dag_is_ready_to_start_no_time(): + """Test DAG readiness check without start time.""" + dag = DAG(dag_id="test_dag") + + assert dag.is_ready_to_start() + assert dag.time_until_start() is None + +def test_dag_with_cron_schedule(): + """Test DAG creation with cron schedule.""" + cron_schedule = "0 9 * * *" # Every day at 9:00 AM + dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + + assert dag.dag_id == "test_dag" + assert dag.cron_schedule == cron_schedule + assert dag.start_time is None + +def test_dag_with_invalid_cron_schedule(): + """Test DAG creation with invalid cron schedule.""" + invalid_cron = "invalid cron" + + with pytest.raises(ValueError, match="Invalid cron expression"): + DAG(dag_id="test_dag", cron_schedule=invalid_cron) + +def test_dag_with_both_start_time_and_cron(): + """Test DAG creation with both start_time and cron_schedule (should fail).""" + start_time = datetime(2024, 1, 1, 9, 0, 0) + cron_schedule = "0 9 * * *" + + with pytest.raises(ValueError, match="Cannot specify both start_time and cron_schedule"): + DAG(dag_id="test_dag", start_time=start_time, cron_schedule=cron_schedule) + +def test_dag_cron_schedule_readiness(): + """Test DAG readiness check with cron schedule.""" + # Test with a cron that runs every minute + cron_schedule = "* * * * *" + dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + + # Should be ready most of the time since it runs every minute + current_time = datetime.now() + assert dag.is_ready_to_start(current_time) + + # Test with specific time + test_time = datetime(2024, 1, 1, 9, 0, 30) # 30 seconds after 9:00 AM + assert dag.is_ready_to_start(test_time) + +def test_dag_cron_next_run_time(): + """Test getting next run time for cron scheduled DAG.""" + cron_schedule = "0 9 * * *" # Every day at 9:00 AM + dag = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + + test_time = datetime(2024, 1, 1, 8, 0, 0) # 8:00 AM + next_run = dag.get_next_run_time(test_time) + + assert next_run is not None + assert next_run.hour == 9 + assert next_run.minute == 0 + +def test_dag_schedule_descriptions(): + """Test schedule description methods.""" + # Test with start_time + start_time = datetime(2024, 1, 1, 9, 0, 0) + dag_start_time = DAG(dag_id="test_dag", start_time=start_time) + assert "One-time execution" in dag_start_time.get_schedule_description() + + # Test with cron schedule + cron_schedule = "0 9 * * *" + dag_cron = DAG(dag_id="test_dag", cron_schedule=cron_schedule) + assert "Cron schedule" in dag_cron.get_schedule_description() + + # Test with no schedule + dag_no_schedule = DAG(dag_id="test_dag") + assert "No schedule" in dag_no_schedule.get_schedule_description() diff --git a/tests/Vecchi_test/test_dag_id_generation.py b/tests/Vecchi_test/test_dag_id_generation.py new file mode 100644 index 0000000..264e450 --- /dev/null +++ b/tests/Vecchi_test/test_dag_id_generation.py @@ -0,0 +1,259 @@ +""" +Test cases for DAG ID generation and validation functionality. +""" + +import pytest +import re +from unittest.mock import Mock, patch, MagicMock +from maestro.server.internals.status_manager import StatusManager, DOCKER_ADJECTIVES, DOCKER_NOUNS +import tempfile +import os +import sqlite3 + + +class TestDockerLikeNameGeneration: + """Test Docker-like name generation functionality.""" + + def setup_method(self): + """Set up a temporary database for each test.""" + self.temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") + self.temp_db.close() + self.sm = StatusManager(self.temp_db.name) + + def teardown_method(self): + """Clean up the temporary database.""" + if os.path.exists(self.temp_db.name): + os.unlink(self.temp_db.name) + + def test_generate_docker_like_name_format(self): + """Test that generated names follow the adjective_noun format.""" + with self.sm as sm: + name = sm.generate_unique_dag_id() + assert "_" in name + parts = name.split("_") + # Should have at least adjective and noun (might have suffix if collision) + assert len(parts) >= 2 + adjective = parts[0] + noun = parts[1] + assert adjective in DOCKER_ADJECTIVES + assert noun in DOCKER_NOUNS + + def test_generate_docker_like_name_randomness(self): + """Test that generated names are different (mostly).""" + with self.sm as sm: + names = [sm.generate_unique_dag_id() for _ in range(10)] + # Most names should be different (allowing for some duplicates due to randomness) + unique_names = set(names) + assert len(unique_names) >= 7 # Allow for some duplicates + + def test_generate_docker_like_name_valid_characters(self): + """Test that generated names only contain valid characters.""" + with self.sm as sm: + name = sm.generate_unique_dag_id() + # Should only contain alphanumeric, underscores, and hyphens + assert re.match(r'^[a-zA-Z0-9_-]+$', name) + + +class TestDAGIDValidation: + """Test DAG ID validation functionality.""" + + def setup_method(self): + """Set up a temporary database for each test.""" + self.temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") + self.temp_db.close() + self.sm = StatusManager(self.temp_db.name) + + def teardown_method(self): + """Clean up the temporary database.""" + if os.path.exists(self.temp_db.name): + os.unlink(self.temp_db.name) + + def test_validate_dag_id_valid_cases(self): + """Test validation with valid DAG IDs.""" + valid_ids = [ + "simple_dag", + "dag-with-hyphens", + "dag_with_123_numbers", + "CamelCaseDAG", + "simple", + "a1b2c3", + "test-dag_123" + ] + + with self.sm as sm: + for dag_id in valid_ids: + assert sm.validate_dag_id(dag_id), f"'{dag_id}' should be valid" + + def test_validate_dag_id_invalid_cases(self): + """Test validation with invalid DAG IDs.""" + invalid_ids = [ + "", # Empty string + "dag with spaces", # Contains spaces + "dag@with@symbols", # Contains special characters + "dag.with.dots", # Contains dots + "dag/with/slashes", # Contains slashes + "dag#with#hash", # Contains hash + "dag%with%percent", # Contains percent + None, # None value + ] + + with self.sm as sm: + for dag_id in invalid_ids: + assert not sm.validate_dag_id(dag_id), f"'{dag_id}' should be invalid" + + +class TestDAGIDUniquenessCheck: + """Test DAG ID uniqueness checking functionality.""" + + def setup_method(self): + """Set up a temporary database for each test.""" + self.temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") + self.temp_db.close() + self.sm = StatusManager(self.temp_db.name) + + def teardown_method(self): + """Clean up the temporary database.""" + if os.path.exists(self.temp_db.name): + os.unlink(self.temp_db.name) + + def test_check_dag_id_uniqueness_unique(self): + """Test uniqueness check when DAG ID is unique.""" + with self.sm as sm: + # Add some existing DAGs + sm._create_dag_if_not_exists('existing_dag_1') + sm._create_dag_if_not_exists('existing_dag_2') + + # Test with a new DAG ID + assert sm.check_dag_id_uniqueness('new_dag_id') is True + + def test_check_dag_id_uniqueness_duplicate(self): + """Test uniqueness check when DAG ID already exists.""" + with self.sm as sm: + # Add some existing DAGs + sm._create_dag_if_not_exists('existing_dag_1') + sm._create_dag_if_not_exists('existing_dag_2') + + # Test with an existing DAG ID + assert sm.check_dag_id_uniqueness('existing_dag_1') is False + + def test_check_dag_id_uniqueness_error_handling(self): + """Test uniqueness check error handling.""" + # Create a temporary file for the database that we'll remove + temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") + temp_db.close() + sm = StatusManager(temp_db.name) + + # Remove the database file to simulate an error during operation + os.unlink(temp_db.name) + + # Now trying to connect should fail + with pytest.raises(sqlite3.OperationalError): + with sm as s: + s.check_dag_id_uniqueness('test_dag') + + def test_check_dag_id_uniqueness_empty_database(self): + """Test uniqueness check with empty database.""" + with self.sm as sm: + # Any DAG ID should be unique in empty database + assert sm.check_dag_id_uniqueness('any_dag_id') is True + + +class TestUniqueDAGIDGeneration: + """Test unique DAG ID generation functionality.""" + + def setup_method(self): + """Set up a temporary database for each test.""" + self.temp_db = tempfile.NamedTemporaryFile(delete=False, suffix=".db") + self.temp_db.close() + self.sm = StatusManager(self.temp_db.name) + + def teardown_method(self): + """Clean up the temporary database.""" + if os.path.exists(self.temp_db.name): + os.unlink(self.temp_db.name) + + def test_generate_unique_dag_id_first_attempt(self): + """Test successful generation on first attempt.""" + with self.sm as sm: + result = sm.generate_unique_dag_id() + + # Should be a valid format + assert "_" in result + parts = result.split("_") + assert len(parts) >= 2 + assert parts[0] in DOCKER_ADJECTIVES + assert parts[1] in DOCKER_NOUNS + + def test_generate_unique_dag_id_retry_logic(self): + """Test retry logic when first attempts fail.""" + with self.sm as sm: + # Pre-populate database with many DAGs to increase chance of collision + for adj in DOCKER_ADJECTIVES[:10]: # Use first 10 adjectives + for noun in DOCKER_NOUNS[:10]: # Use first 10 nouns + sm._create_dag_if_not_exists(f"{adj}_{noun}") + + # Generate a new unique ID + result = sm.generate_unique_dag_id() + + # Should still get a unique ID + assert sm.check_dag_id_uniqueness(result) + assert "_" in result + + @patch('maestro.server.internals.status_manager.random.choices') + def test_generate_unique_dag_id_fallback_suffix(self, mock_choices): + """Test fallback to suffix when max attempts reached.""" + mock_choices.return_value = ['a', 'b', 'c', 'd', 'e', 'f'] + + with self.sm as sm: + # Pre-populate database with ALL possible combinations + # This is impractical in reality but simulates the worst case + # Instead, we'll mock the check_dag_id_uniqueness method + original_check = sm.check_dag_id_uniqueness + call_count = 0 + + def mock_check(dag_id): + nonlocal call_count + call_count += 1 + # Return False for first 100 calls, then True + if call_count <= 100: + return False + return original_check(dag_id) + + sm.check_dag_id_uniqueness = mock_check + + result = sm.generate_unique_dag_id() + + # Should have a suffix + assert result.endswith("_abcdef") + assert call_count >= 100 # Should have tried at least 100 times + + def test_generate_unique_dag_id_valid_format(self): + """Test that generated unique DAG ID has valid format.""" + with self.sm as sm: + result = sm.generate_unique_dag_id() + + assert sm.validate_dag_id(result) + + +class TestDockerListContents: + """Test that Docker name lists contain expected content.""" + + def test_docker_adjectives_not_empty(self): + """Test that adjectives list is not empty.""" + assert len(DOCKER_ADJECTIVES) > 0 + assert all(isinstance(adj, str) for adj in DOCKER_ADJECTIVES) + + def test_docker_nouns_not_empty(self): + """Test that nouns list is not empty.""" + assert len(DOCKER_NOUNS) > 0 + assert all(isinstance(noun, str) for noun in DOCKER_NOUNS) + + def test_docker_names_valid_format(self): + """Test that all names in lists are valid for DAG IDs.""" + # Test adjectives + for adj in DOCKER_ADJECTIVES: + assert re.match(r'^[a-zA-Z0-9_-]+$', adj), f"Invalid adjective: {adj}" + + # Test nouns + for noun in DOCKER_NOUNS: + assert re.match(r'^[a-zA-Z0-9_-]+$', noun), f"Invalid noun: {noun}" diff --git a/tests/Vecchi_test/test_db_feature.py b/tests/Vecchi_test/test_db_feature.py new file mode 100644 index 0000000..1cc97bf --- /dev/null +++ b/tests/Vecchi_test/test_db_feature.py @@ -0,0 +1,107 @@ +import pytest +import os +import uuid + +from maestro.server.internals.orchestrator import Orchestrator +from maestro.server.tasks.base import BaseTask +from maestro.shared.task import TaskStatus +from maestro.server.internals.status_manager import StatusManager + +# Define a dummy task for testing +class DummyPrintTask(BaseTask): + message: str + executed: bool = False + + def execute_local(self): + self.executed = True + +@pytest.fixture +def db_path(tmp_path): + return os.path.join(tmp_path, "test_maestro.db") + +@pytest.fixture +def orchestrator(db_path): + orch = Orchestrator(log_level="CRITICAL", db_path=db_path) + orch.register_task_type("print_task", DummyPrintTask) + return orch + +@pytest.fixture +def dag_filepath(tmp_path): + content = """ +dag: + tasks: + - task_id: task1 + type: print_task + message: "Hello from task1" + - task_id: task2 + type: print_task + message: "Hello from task2" + dependencies: [task1] +""" + f = tmp_path / "test_dag.yaml" + f.write_text(content) + return str(f) + +def test_db_creation(orchestrator, db_path): + with orchestrator.status_manager: + pass # Entering the context creates the DB + assert os.path.exists(db_path) + +def test_save_state(orchestrator, dag_filepath, db_path): + dag = orchestrator.load_dag_from_file(dag_filepath) + execution_id = orchestrator.run_dag_in_thread(dag) + + # Wait for execution to complete + import time + time.sleep(2) + + with StatusManager(db_path) as sm: + assert sm.get_task_status(dag.dag_id, "task1", execution_id) == "completed" + assert sm.get_task_status(dag.dag_id, "task2", execution_id) == "completed" + +def test_resume_execution(orchestrator, dag_filepath, db_path): + dag = orchestrator.load_dag_from_file(dag_filepath) + + # First, create an execution and set task1 as completed + execution_id = str(uuid.uuid4()) + with StatusManager(db_path) as sm: + # Create the execution and DAG records + sm.save_dag_definition(dag) + sm.create_dag_execution(dag.dag_id, execution_id) + sm.initialize_tasks_for_execution(dag.dag_id, execution_id, list(dag.tasks.keys())) + # Mark task1 as completed + sm.set_task_status(dag.dag_id, "task1", "completed", execution_id) + + assert not dag.tasks["task1"].executed + assert not dag.tasks["task2"].executed + + # Run the DAG with resume=True using the same execution_id + orchestrator.run_dag(dag, execution_id=execution_id, resume=True) + + assert not dag.tasks["task1"].executed # Should not re-execute + assert dag.tasks["task2"].executed # Should execute + assert dag.tasks["task1"].status == TaskStatus.COMPLETED + assert dag.tasks["task2"].status == TaskStatus.COMPLETED + +def test_reset_execution(orchestrator, dag_filepath, db_path): + dag1 = orchestrator.load_dag_from_file(dag_filepath) + execution_id1 = orchestrator.run_dag_in_thread(dag1) + + # Wait for execution to complete + import time + time.sleep(2) + + assert dag1.tasks["task1"].executed + + dag2 = orchestrator.load_dag_from_file(dag_filepath) + assert not dag2.tasks["task1"].executed # New DAG instance has fresh tasks + + execution_id2 = orchestrator.run_dag_in_thread(dag2, resume=False) + time.sleep(2) + + assert dag2.tasks["task1"].executed + with StatusManager(db_path) as sm: + # Both executions should show completed + assert sm.get_task_status(dag1.dag_id, "task1", execution_id1) == "completed" + assert sm.get_task_status(dag2.dag_id, "task1", execution_id2) == "completed" + diff --git a/tests/Vecchi_test/test_enhanced_cli.py b/tests/Vecchi_test/test_enhanced_cli.py new file mode 100644 index 0000000..5c580d5 --- /dev/null +++ b/tests/Vecchi_test/test_enhanced_cli.py @@ -0,0 +1,441 @@ +import pytest +import time +import uuid +from unittest.mock import Mock, patch, MagicMock +from datetime import datetime, timedelta + +from maestro.server.internals.status_manager import StatusManager +from maestro.server.internals.orchestrator import Orchestrator +from maestro.shared.dag import DAG +from maestro.server.tasks.base import BaseTask + + +class SimpleTask(BaseTask): + """Simple task for CLI testing.""" + message: str = "test" + executed: bool = False + + def execute_local(self): + self.executed = True + + +@pytest.fixture +def status_manager(tmp_path): + """Create status manager with test database.""" + db_path = tmp_path / "test_cli.db" + return StatusManager(str(db_path)) + + +@pytest.fixture +def orchestrator_with_data(tmp_path): + """Create orchestrator with test data.""" + db_path = tmp_path / "test_cli_orchestrator.db" + orchestrator = Orchestrator(log_level="CRITICAL", db_path=str(db_path)) + orchestrator.register_task_type("simple_task", SimpleTask) + return orchestrator + + +class TestStatusManagerCLIFeatures: + """Test status manager features used by CLI.""" + + def test_create_dag_execution(self, status_manager): + """Test creating DAG execution records.""" + dag_id = "test_dag" + execution_id = str(uuid.uuid4()) + + with status_manager as sm: + result = sm.create_dag_execution(dag_id, execution_id) + assert result == execution_id + + # Verify execution was created + details = sm.get_dag_execution_details(dag_id, execution_id) + assert details["execution_id"] == execution_id + assert details["status"] == "running" + assert details["started_at"] is not None + + def test_update_dag_execution_status(self, status_manager): + """Test updating DAG execution status.""" + dag_id = "test_dag" + execution_id = str(uuid.uuid4()) + + with status_manager as sm: + sm.create_dag_execution(dag_id, execution_id) + + # Update status to completed + sm.update_dag_execution_status(dag_id, execution_id, "completed") + + # Verify status update + details = sm.get_dag_execution_details(dag_id, execution_id) + assert details["status"] == "completed" + assert details["completed_at"] is not None + + def test_get_running_dags(self, status_manager): + """Test retrieving running DAGs.""" + dag_id1 = "running_dag_1" + dag_id2 = "running_dag_2" + dag_id3 = "completed_dag" + + with status_manager as sm: + # Create running DAGs + exec_id1 = sm.create_dag_execution(dag_id1, str(uuid.uuid4())) + exec_id2 = sm.create_dag_execution(dag_id2, str(uuid.uuid4())) + exec_id3 = sm.create_dag_execution(dag_id3, str(uuid.uuid4())) + + # Complete one DAG + sm.update_dag_execution_status(dag_id3, exec_id3, "completed") + + # Get running DAGs + running_dags = sm.get_running_dags() + + # Should only return running DAGs + assert len(running_dags) == 2 + running_dag_ids = [dag["dag_id"] for dag in running_dags] + assert dag_id1 in running_dag_ids + assert dag_id2 in running_dag_ids + assert dag_id3 not in running_dag_ids + + def test_get_dags_by_status(self, status_manager): + """Test retrieving DAGs by specific status.""" + with status_manager as sm: + # Create DAGs with different statuses + dag_ids = [] + for i, status in enumerate(["running", "completed", "failed", "running"]): + dag_id = f"dag_{i}" + execution_id = str(uuid.uuid4()) + sm.create_dag_execution(dag_id, execution_id) + if status != "running": + sm.update_dag_execution_status(dag_id, execution_id, status) + dag_ids.append(dag_id) + + # Test getting running DAGs + running_dags = sm.get_dags_by_status("running") + assert len(running_dags) == 2 + + # Test getting completed DAGs + completed_dags = sm.get_dags_by_status("completed") + assert len(completed_dags) == 1 + + # Test getting failed DAGs + failed_dags = sm.get_dags_by_status("failed") + assert len(failed_dags) == 1 + + def test_get_all_dags(self, status_manager): + """Test retrieving all DAGs.""" + with status_manager as sm: + # Create DAGs with different statuses + dag_count = 5 + for i in range(dag_count): + dag_id = f"dag_{i}" + execution_id = str(uuid.uuid4()) + sm.create_dag_execution(dag_id, execution_id) + if i % 2 == 0: + sm.update_dag_execution_status(dag_id, execution_id, "completed") + + # Get all DAGs + all_dags = sm.get_all_dags() + assert len(all_dags) == dag_count + + # Verify all have required fields + for dag in all_dags: + assert "dag_id" in dag + assert "execution_id" in dag + assert "status" in dag + assert "started_at" in dag + + def test_get_dag_summary(self, status_manager): + """Test getting DAG summary statistics.""" + with status_manager as sm: + # Create DAGs with different statuses + statuses = ["running", "completed", "failed", "completed", "running"] + for i, status in enumerate(statuses): + dag_id = f"dag_{i}" + execution_id = str(uuid.uuid4()) + sm.create_dag_execution(dag_id, execution_id) + if status != "running": + sm.update_dag_execution_status(dag_id, execution_id, status) + + # Get summary + summary = sm.get_dag_summary() + + # Verify summary structure + assert "total_executions" in summary + assert "unique_dags" in summary + assert "status_counts" in summary + + # Verify counts + assert summary["total_executions"] == 5 + assert summary["unique_dags"] == 5 + assert summary["status_counts"]["running"] == 2 + assert summary["status_counts"]["completed"] == 2 + assert summary["status_counts"]["failed"] == 1 + + def test_get_dag_history(self, status_manager): + """Test getting DAG execution history.""" + dag_id = "test_dag" + + with status_manager as sm: + # Create multiple executions for the same DAG + execution_ids = [] + for i in range(3): + execution_id = str(uuid.uuid4()) + sm.create_dag_execution(dag_id, execution_id) + sm.update_dag_execution_status(dag_id, execution_id, "completed") + execution_ids.append(execution_id) + + # Get history + history = sm.get_dag_history(dag_id) + + # Verify history + assert len(history) == 3 + for execution in history: + assert execution["execution_id"] in execution_ids + assert execution["status"] == "completed" + + def test_cleanup_old_executions(self, status_manager): + """Test cleaning up old execution records.""" + with status_manager as sm: + # Create some executions + dag_id = "test_dag" + execution_ids = [] + for i in range(5): + execution_id = str(uuid.uuid4()) + sm.create_dag_execution(dag_id, execution_id) + sm.update_dag_execution_status(dag_id, execution_id, "completed") + execution_ids.append(execution_id) + + # Clean up all executions (0 days to keep) + deleted_count = sm.cleanup_old_executions(0) + + # Verify cleanup + assert deleted_count == 5 + + # Verify all executions are gone + all_dags = sm.get_all_dags() + assert len(all_dags) == 0 + + def test_cancel_dag_execution(self, status_manager): + """Test cancelling DAG executions.""" + with status_manager as sm: + # Create running executions + dag_id = "test_dag" + execution_id1 = str(uuid.uuid4()) + execution_id2 = str(uuid.uuid4()) + + sm.create_dag_execution(dag_id, execution_id1) + sm.create_dag_execution(dag_id, execution_id2) + + # Cancel specific execution + success = sm.cancel_dag_execution(dag_id, execution_id1) + assert success is True + + # Verify cancellation + details = sm.get_dag_execution_details(dag_id, execution_id1) + assert details["status"] == "cancelled" + assert details["completed_at"] is not None + + # Other execution should still be running + details2 = sm.get_dag_execution_details(dag_id, execution_id2) + assert details2["status"] == "running" + + def test_log_message_and_retrieval(self, status_manager): + """Test logging messages and retrieving them.""" + with status_manager as sm: + dag_id = "test_dag" + execution_id = str(uuid.uuid4()) + task_id = "test_task" + + sm.create_dag_execution(dag_id, execution_id) + + # Log some messages + messages = [ + ("INFO", "Task started"), + ("DEBUG", "Processing data"), + ("INFO", "Task completed"), + ("ERROR", "Something went wrong") + ] + + for level, message in messages: + sm.log_message(dag_id, execution_id, task_id, level, message) + + # Retrieve logs + logs = sm.get_execution_logs(dag_id, execution_id) + + assert len(logs) == len(messages) + + # Verify messages (retrieved logs are in descending order) + retrieved_messages = [log['message'] for log in reversed(logs)] + original_messages = [msg[1] for msg in messages] + assert retrieved_messages == original_messages + + def test_get_dag_execution_details(self, status_manager): + """Test getting detailed DAG execution information.""" + with status_manager as sm: + dag_id = "test_dag" + execution_id = str(uuid.uuid4()) + + # Create execution + sm.create_dag_execution(dag_id, execution_id) + + # Add some task statuses with execution_id + sm.set_task_status(dag_id, "task1", "completed", execution_id) + sm.set_task_status(dag_id, "task2", "running", execution_id) + + # Get details + details = sm.get_dag_execution_details(dag_id, execution_id) + + # Verify details structure + assert details["execution_id"] == execution_id + assert details["status"] == "running" + assert "started_at" in details + assert "tasks" in details + + # Verify task details + task_ids = [task["task_id"] for task in details["tasks"]] + assert "task1" in task_ids + assert "task2" in task_ids + + +class TestCLIIntegration: + """Test CLI integration with orchestrator and status manager.""" + + def test_orchestrator_run_dag_in_thread_cli_integration(self, orchestrator_with_data): + """Test run_dag_in_thread for CLI integration.""" + # Create simple DAG + dag = DAG(dag_id="cli_test_dag") + task = SimpleTask(task_id="simple_task", message="Hello CLI") + dag.add_task(task) + + # Run in thread (simulating CLI async execution) + execution_id = orchestrator_with_data.run_dag_in_thread(dag) + + # Wait for completion + time.sleep(0.2) + + # Verify execution was tracked + with orchestrator_with_data.status_manager as sm: + details = sm.get_dag_execution_details(dag.dag_id, execution_id) + assert details["execution_id"] == execution_id + assert details["status"] == "completed" + + def test_dag_status_monitoring_cli_scenario(self, orchestrator_with_data): + """Test DAG status monitoring scenario for CLI.""" + dag = DAG(dag_id="monitor_dag") + task = SimpleTask(task_id="monitor_task", message="Monitor me") + dag.add_task(task) + + # Start execution + execution_id = orchestrator_with_data.run_dag_in_thread(dag) + + # Wait for completion first + time.sleep(0.3) + + # Monitor execution (simulating CLI monitor command) - use separate context + with orchestrator_with_data.status_manager as sm: + # Check final status + details = sm.get_dag_execution_details(dag.dag_id, execution_id) + assert details["status"] == "completed" + + def test_resume_functionality_cli_scenario(self, orchestrator_with_data): + """Test resume functionality for CLI.""" + dag = DAG(dag_id="resume_dag") + task1 = SimpleTask(task_id="task1", message="First task") + task2 = SimpleTask(task_id="task2", message="Second task", dependencies=["task1"]) + dag.add_task(task1) + dag.add_task(task2) + + # Simulate partial execution + with orchestrator_with_data.status_manager as sm: + sm.set_task_status(dag.dag_id, "task1", "completed") + sm.set_task_status(dag.dag_id, "task2", "pending") + + # Resume execution + execution_id = orchestrator_with_data.run_dag_in_thread(dag, resume=True) + + # Wait for completion + time.sleep(0.2) + + # Verify resume worked + with orchestrator_with_data.status_manager as sm: + details = sm.get_dag_execution_details(dag.dag_id, execution_id) + assert details["status"] == "completed" + + # task1 should not have been executed again + assert not dag.tasks["task1"].executed + # task2 should have been executed + assert dag.tasks["task2"].executed + + +class TestCLIDataStructures: + """Test data structures and formats expected by CLI.""" + + def test_dag_execution_details_format(self, status_manager): + """Test that DAG execution details have expected format for CLI.""" + with status_manager as sm: + dag_id = "format_test_dag" + execution_id = str(uuid.uuid4()) + + # Create execution with tasks + sm.create_dag_execution(dag_id, execution_id) + sm.set_task_status(dag_id, "task1", "completed", execution_id) + sm.set_task_status(dag_id, "task2", "running", execution_id) + + # Get details + details = sm.get_dag_execution_details(dag_id, execution_id) + + # Verify CLI-expected format + required_fields = ["execution_id", "status", "started_at", "completed_at", "thread_id", "pid", "tasks"] + for field in required_fields: + assert field in details + + # Verify task format + assert len(details["tasks"]) == 2 + for task in details["tasks"]: + task_fields = ["task_id", "status", "started_at", "completed_at", "thread_id"] + for field in task_fields: + assert field in task + + def test_logs_format_for_cli(self, status_manager): + """Test that logs have expected format for CLI display.""" + with status_manager as sm: + dag_id = "log_format_dag" + execution_id = str(uuid.uuid4()) + + sm.create_dag_execution(dag_id, execution_id) + + # Log some messages + sm.log_message(dag_id, execution_id, "task1", "INFO", "Test message") + + logs = sm.get_execution_logs(dag_id, execution_id) + assert len(logs) == 1 + log_entry = logs[0] + + expected_keys = ["task_id", "level", "message", "timestamp", "thread_id"] + for key in expected_keys: + assert key in log_entry + + # Verify log levels are valid + assert log_entry["level"] in ["DEBUG", "INFO", "WARNING", "ERROR"] + + def test_summary_format_for_cli(self, status_manager): + """Test that summary has expected format for CLI display.""" + with status_manager as sm: + # Create some test data + for i in range(5): + dag_id = f"summary_dag_{i}" + execution_id = str(uuid.uuid4()) + sm.create_dag_execution(dag_id, execution_id) + if i % 2 == 0: + sm.update_dag_execution_status(dag_id, execution_id, "completed") + + # Get summary + summary = sm.get_dag_summary() + + # Verify CLI-expected format + required_fields = ["total_executions", "unique_dags", "status_counts"] + for field in required_fields: + assert field in summary + + # Verify status_counts is a dictionary + assert isinstance(summary["status_counts"], dict) + assert summary["total_executions"] == 5 + assert summary["unique_dags"] == 5 diff --git a/tests/Vecchi_test/test_extended_terraform_task.py b/tests/Vecchi_test/test_extended_terraform_task.py new file mode 100644 index 0000000..6b69f66 --- /dev/null +++ b/tests/Vecchi_test/test_extended_terraform_task.py @@ -0,0 +1,366 @@ +import pytest +import os +import tempfile +from pathlib import Path +from unittest.mock import patch, MagicMock, call +import shutil + +# Import with proper error handling +try: + from maestro.server.tasks.extended_terraform_task import ( + ExtendedTerraformTask, + check_command_exists, + print_status, + print_success, + print_warning, + print_error, + print_header + ) +except ImportError: + pytest.skip("ExtendedTerraformTask not available", allow_module_level=True) + + +class TestExtendedTerraformTask: + + def setup_method(self): + """Setup method to ensure clean state for each test.""" + self.original_cwd = os.getcwd() + + def teardown_method(self): + """Cleanup method to restore original state.""" + os.chdir(self.original_cwd) + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_full_workflow_execution(self, mock_which, mock_subprocess): + """Test executing the full Terraform workflow.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + workflow_mode=True + ) + + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + + result = task.execute_local() + + # Verify the workflow executed successfully + assert result is not None + # Adjusted number based on: init, validate, fmt (check), fmt, plan, show, apply + assert mock_subprocess.call_count == 7 + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_single_command_execution(self, mock_which, mock_subprocess): + """Test executing a single Terraform command.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + command='init' + ) + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + + task.execute_local() + + mock_subprocess.assert_called_once() + + def test_missing_workflow_and_command(self): + """Test handling of missing workflow_mode and command.""" + with patch('maestro.server.tasks.terraform_task.shutil.which', return_value='/usr/bin/terraform'): + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.' + ) + + with pytest.raises(ValueError) as exc_info: + task.execute_local() + assert "either 'workflow_mode' must be true or a 'command' must be provided" in str(exc_info.value) + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.extended_terraform_task.os.path.exists') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_full_workflow_with_existing_terraform_dir(self, mock_which, mock_exists, mock_subprocess): + """Test full workflow when .terraform directory exists.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + workflow_mode=True + ) + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + mock_exists.return_value = True + + with patch('maestro.server.tasks.extended_terraform_task.Path.is_dir', return_value=True): + task.execute_local() + + # Should call init with -upgrade flag + assert mock_subprocess.call_count == 7 + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.extended_terraform_task.os.environ.get') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_full_workflow_with_skip_plan(self, mock_which, mock_environ, mock_subprocess): + """Test full workflow with SKIP_PLAN environment variable.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + workflow_mode=True + ) + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + mock_environ.return_value = "true" + + task.execute_local() + + # Should call fewer commands when SKIP_PLAN is true + assert mock_subprocess.call_count == 4 # init, validate, fmt (check), fmt + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.extended_terraform_task.os.chdir') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_full_workflow_directory_change(self, mock_which, mock_chdir, mock_subprocess): + """Test that the workflow changes directories correctly.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='/tmp/test', + workflow_mode=True + ) + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + + task.execute_local() + + # Should change to the working directory and back + assert mock_chdir.call_count == 2 + + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_full_workflow_exception_handling(self, mock_which, mock_subprocess): + """Test that exceptions in workflow are handled properly.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + workflow_mode=True + ) + mock_subprocess.side_effect = Exception("Test error") + + with pytest.raises(Exception) as exc_info: + task.execute_local() + assert "Test error" in str(exc_info.value) + + def test_workflow_mode_field(self): + """Test that workflow_mode field is properly set.""" + with patch('maestro.server.tasks.terraform_task.shutil.which', return_value='/usr/bin/terraform'): + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + workflow_mode=True + ) + assert task.workflow_mode is True + + task_no_workflow = ExtendedTerraformTask( + task_id='test_task2', + working_dir='.' + ) + assert task_no_workflow.workflow_mode is False + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_single_command_with_workspace(self, mock_which, mock_subprocess): + """Test single command execution with workspace.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + command='plan', + workspace='test-workspace' + ) + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + + task.execute_local() + + # Should call workspace select first, then the command + assert mock_subprocess.call_count == 2 + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_single_command_with_variables(self, mock_which, mock_subprocess): + """Test single command execution with variables.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + command='plan', + vars={'env': 'test', 'region': 'us-east-1'} + ) + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + + task.execute_local() + + mock_subprocess.assert_called_once() + + def test_print_functions(self): + """Test the print utility functions.""" + with patch('maestro.server.tasks.extended_terraform_task.logging.getLogger') as mock_logger: + mock_log = MagicMock() + mock_logger.return_value = mock_log + + print_status("Test status") + print_success("Test success") + print_warning("Test warning") + print_error("Test error") + print_header("Test header") + + # Verify logging calls were made + assert mock_log.info.call_count == 3 # status, success, header + assert mock_log.warning.call_count == 1 # warning + assert mock_log.error.call_count == 1 # error + + def test_check_command_exists(self): + """Test the check_command_exists function.""" + with patch('maestro.server.tasks.extended_terraform_task.shutil.which') as mock_which: + mock_which.return_value = '/usr/bin/terraform' + assert check_command_exists('terraform') is True + + mock_which.return_value = None + assert check_command_exists('nonexistent') is False + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.terraform_task.shutil.which') + def test_workflow_returns_value(self, mock_which, mock_subprocess): + """Test that the workflow returns the expected value.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + workflow_mode=True + ) + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + + # The _run_full_workflow method should return 1 + result = task._run_full_workflow() + assert result == 1 + + @pytest.mark.skipif( + not shutil.which("terraform") and not shutil.which("tofu"), + reason="Neither terraform nor tofu available in PATH" + ) + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + @patch('maestro.server.tasks.terraform_task.shutil.which') + @patch('maestro.server.tasks.extended_terraform_task.os.makedirs') + def test_workflow_with_relative_path(self, mock_makedirs, mock_which, mock_subprocess): + """Test workflow execution with relative working directory.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + + with tempfile.TemporaryDirectory() as temp_dir: + # Create terraform directory for the test + terraform_dir = Path(temp_dir) / 'terraform' + terraform_dir.mkdir() + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='./terraform', + workflow_mode=True, + dag_file_path=str(Path(temp_dir) / 'test.yaml') + ) + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + + task.execute_local() + + assert mock_subprocess.call_count == 7 + + +# Alternative approach: Create a separate test configuration for environments without terraform +class TestExtendedTerraformTaskMocked: + """Tests that run entirely with mocks, regardless of terraform availability.""" + + @patch('maestro.server.tasks.terraform_task.shutil.which') + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + def test_mock_full_workflow_execution(self, mock_subprocess, mock_which): + """Test executing the full Terraform workflow with complete mocking.""" + # Mock terraform being available + mock_which.return_value = '/usr/bin/terraform' + mock_subprocess.return_value = MagicMock(returncode=0, stdout='', stderr='') + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + workflow_mode=True + ) + + task.execute_local() + + # Verify the workflow executed successfully + assert mock_subprocess.call_count == 7 + + @patch('maestro.server.tasks.terraform_task.shutil.which') + @patch('maestro.server.tasks.extended_terraform_task.subprocess.run') + def test_mock_terraform_not_available(self, mock_subprocess, mock_which): + """Test behavior when terraform is not available.""" + # Mock terraform NOT being available + mock_which.return_value = None + + task = ExtendedTerraformTask( + task_id='test_task', + working_dir='.', + workflow_mode=True + ) + + with pytest.raises(FileNotFoundError) as exc_info: + task.execute_local() + + assert "Neither 'tofu' nor 'terraform' command found" in str(exc_info.value) \ No newline at end of file diff --git a/tests/Vecchi_test/test_multi_executor.py b/tests/Vecchi_test/test_multi_executor.py new file mode 100644 index 0000000..dcd959a --- /dev/null +++ b/tests/Vecchi_test/test_multi_executor.py @@ -0,0 +1,286 @@ +import pytest +import time +from unittest.mock import Mock, patch, MagicMock + +from maestro.server.internals.orchestrator import Orchestrator +from maestro.shared.dag import DAG +from maestro.server.tasks.base import BaseTask +from maestro.shared.task import TaskStatus +from maestro.server.internals.executors.factory import ExecutorFactory +from maestro.server.internals.executors.local import LocalExecutor +from maestro.server.internals.executors.base import BaseExecutor + + +class MultiExecutorTestTask(BaseTask): + """Test task for multi-executor testing.""" + executed: bool = False + executor_used: str = None + + def execute_local(self): + self.executed = True + self.executor_used = "local" + + +class CustomExecutor(BaseExecutor): + """Custom executor for testing.""" + def __init__(self): + self.executed_tasks = [] + + def execute(self, task): + self.executed_tasks.append(task.task_id) + task.execute_local() + + +@pytest.fixture +def orchestrator_with_executors(tmp_path): + """Create orchestrator with multiple executors.""" + db_path = tmp_path / "test_multi_executor.db" + orchestrator = Orchestrator(log_level="CRITICAL", db_path=str(db_path)) + orchestrator.register_task_type("test_task", MultiExecutorTestTask) + return orchestrator + + +@pytest.fixture +def dag_with_different_executors(): + """Create DAG with tasks using different executors.""" + dag = DAG(dag_id="multi_executor_dag") + + # Tasks with different executors + task1 = MultiExecutorTestTask(task_id="local_task", executor="local") + task2 = MultiExecutorTestTask(task_id="ssh_task", executor="ssh") + task3 = MultiExecutorTestTask(task_id="docker_task", executor="docker") + + dag.add_task(task1) + dag.add_task(task2) + dag.add_task(task3) + + return dag + + +class TestMultiExecutorSupport: + """Test multi-executor support functionality.""" + + def test_executor_factory_default_executors(self): + """Test that ExecutorFactory has default executors registered.""" + factory = ExecutorFactory() + + # Test getting default executors + local_executor = factory.get_executor("local") + assert isinstance(local_executor, LocalExecutor) + + # Test that unknown executor raises error + with pytest.raises(ValueError, match="Unknown executor: unknown"): + factory.get_executor("unknown") + + def test_executor_factory_register_custom_executor(self): + """Test registering custom executor.""" + factory = ExecutorFactory() + + # Register custom executor + factory.register_executor("custom", CustomExecutor) + + # Test getting custom executor + custom_executor = factory.get_executor("custom") + assert isinstance(custom_executor, CustomExecutor) + + def test_task_executor_field_default(self): + """Test that task executor field defaults to 'local'.""" + task = MultiExecutorTestTask(task_id="test_task") + assert task.executor == "local" + + def test_task_executor_field_custom(self): + """Test setting custom executor for task.""" + task = MultiExecutorTestTask(task_id="test_task", executor="ssh") + assert task.executor == "ssh" + + def test_dag_execution_with_different_executors(self, orchestrator_with_executors, dag_with_different_executors): + """Test DAG execution with tasks using different executors.""" + # Create custom executors for testing + class TestSshExecutor(BaseExecutor): + def execute(self, task): + task.execute_local() + task.executor_used = "ssh" + + class TestDockerExecutor(BaseExecutor): + def execute(self, task): + task.execute_local() + task.executor_used = "docker" + + # Register custom executors + orchestrator_with_executors.executor_factory.register_executor("ssh", TestSshExecutor) + orchestrator_with_executors.executor_factory.register_executor("docker", TestDockerExecutor) + + # Run the DAG + orchestrator_with_executors.run_dag(dag_with_different_executors) + + # Verify all tasks were executed + assert dag_with_different_executors.tasks["local_task"].executed + assert dag_with_different_executors.tasks["ssh_task"].executed + assert dag_with_different_executors.tasks["docker_task"].executed + + # Verify correct executors were used + assert dag_with_different_executors.tasks["local_task"].executor_used == "local" + assert dag_with_different_executors.tasks["ssh_task"].executor_used == "ssh" + assert dag_with_different_executors.tasks["docker_task"].executor_used == "docker" + + def test_concurrent_execution_with_different_executors(self, orchestrator_with_executors): + """Test concurrent execution with different executors.""" + dag = DAG(dag_id="concurrent_multi_executor_dag") + + # Create tasks with different executors + task1 = MultiExecutorTestTask(task_id="local_task1", executor="local") + task2 = MultiExecutorTestTask(task_id="local_task2", executor="local") + + dag.add_task(task1) + dag.add_task(task2) + + # Run concurrently + execution_id = orchestrator_with_executors.run_dag_in_thread(dag) + + # Wait for completion + time.sleep(0.2) + + # Verify execution completed + with orchestrator_with_executors.status_manager as sm: + details = sm.get_dag_execution_details(dag.dag_id, execution_id) + assert details["status"] == "completed" + + # Verify tasks were executed + assert dag.tasks["local_task1"].executed + assert dag.tasks["local_task2"].executed + + def test_executor_failure_handling(self, orchestrator_with_executors): + """Test handling of executor failures.""" + dag = DAG(dag_id="executor_failure_dag") + + # Create task with non-existent executor + task = MultiExecutorTestTask(task_id="invalid_executor_task", executor="non_existent") + dag.add_task(task) + + # Run should fail with unknown executor + with pytest.raises(Exception, match="Unknown executor: non_existent"): + orchestrator_with_executors.run_dag(dag) + + def test_custom_executor_registration_and_usage(self, orchestrator_with_executors): + """Test registering and using custom executor.""" + # Register custom executor + orchestrator_with_executors.executor_factory.register_executor("custom", CustomExecutor) + + # Create DAG with custom executor task + dag = DAG(dag_id="custom_executor_dag") + task = MultiExecutorTestTask(task_id="custom_task", executor="custom") + dag.add_task(task) + + # Run DAG + orchestrator_with_executors.run_dag(dag) + + # Verify task was executed + assert task.executed + + def test_mixed_executor_dependencies(self, orchestrator_with_executors): + """Test DAG with tasks using different executors and dependencies.""" + dag = DAG(dag_id="mixed_executor_dependencies_dag") + + # Create tasks with dependencies and different executors + task1 = MultiExecutorTestTask(task_id="local_root", executor="local") + task2 = MultiExecutorTestTask(task_id="local_dependent", executor="local", dependencies=["local_root"]) + + dag.add_task(task1) + dag.add_task(task2) + + # Run DAG + orchestrator_with_executors.run_dag(dag) + + # Verify execution order and completion + assert task1.executed + assert task2.executed + assert task1.status == TaskStatus.COMPLETED + assert task2.status == TaskStatus.COMPLETED + + def test_executor_factory_thread_safety(self): + """Test that ExecutorFactory is thread-safe.""" + from concurrent.futures import ThreadPoolExecutor as TPE + + factory = ExecutorFactory() + + def get_executor(name): + return factory.get_executor(name) + + # Test concurrent access to executor factory + with TPE(max_workers=5) as executor: + futures = [] + for _ in range(10): + future = executor.submit(get_executor, "local") + futures.append(future) + + # All should complete without errors + results = [f.result() for f in futures] + + # All should return LocalExecutor instances + for result in results: + assert isinstance(result, LocalExecutor) + + +class TestExecutorIntegration: + """Test executor integration with the orchestrator.""" + + def test_orchestrator_uses_correct_executor(self, orchestrator_with_executors): + """Test that orchestrator uses the correct executor for each task.""" + dag = DAG(dag_id="executor_integration_dag") + + # Create tasks with specific executors + local_task = MultiExecutorTestTask(task_id="local_task", executor="local") + dag.add_task(local_task) + + # Mock the executor factory to track calls + with patch.object(orchestrator_with_executors.executor_factory, 'get_executor', wraps=orchestrator_with_executors.executor_factory.get_executor) as mock_get_executor: + orchestrator_with_executors.run_dag(dag) + + # Verify get_executor was called with correct executor name + mock_get_executor.assert_called_with("local") + + def test_executor_context_isolation(self, orchestrator_with_executors): + """Test that different executors don't interfere with each other.""" + # Register two custom executors + executor1 = CustomExecutor() + executor2 = CustomExecutor() + + orchestrator_with_executors.executor_factory.register_executor("custom1", lambda: executor1) + orchestrator_with_executors.executor_factory.register_executor("custom2", lambda: executor2) + + dag = DAG(dag_id="executor_isolation_dag") + + task1 = MultiExecutorTestTask(task_id="task1", executor="custom1") + task2 = MultiExecutorTestTask(task_id="task2", executor="custom2") + + dag.add_task(task1) + dag.add_task(task2) + + # Run DAG + orchestrator_with_executors.run_dag(dag) + + # Verify each executor only executed its own task + assert "task1" in executor1.executed_tasks + assert "task1" not in executor2.executed_tasks + assert "task2" in executor2.executed_tasks + assert "task2" not in executor1.executed_tasks + + def test_executor_error_propagation(self, orchestrator_with_executors): + """Test that executor errors are properly propagated.""" + # Create a failing executor + class FailingExecutor(BaseExecutor): + def execute(self, task): + raise Exception("Executor failed") + + orchestrator_with_executors.executor_factory.register_executor("failing", FailingExecutor) + + dag = DAG(dag_id="failing_executor_dag") + task = MultiExecutorTestTask(task_id="failing_task", executor="failing") + dag.add_task(task) + + # Run should fail with executor error + with pytest.raises(Exception, match="Task failing_task failed: Executor failed"): + orchestrator_with_executors.run_dag(dag) + + # Verify task status is marked as failed + assert task.status == TaskStatus.FAILED diff --git a/tests/Vecchi_test/test_orchestrator_dagloader.py b/tests/Vecchi_test/test_orchestrator_dagloader.py new file mode 100644 index 0000000..7ea5715 --- /dev/null +++ b/tests/Vecchi_test/test_orchestrator_dagloader.py @@ -0,0 +1,266 @@ +import pytest +from datetime import datetime + +from maestro.server.internals.orchestrator import Orchestrator +from maestro.shared.dag import DAG +from maestro.server.tasks.base import BaseTask +from maestro.shared.task import TaskStatus +from maestro.server.internals.executors.local import LocalExecutor + +# Define a dummy task for testing +class DummyPrintTask(BaseTask): + message: str + executed: bool = False + + def execute_local(self): + self.executed = True + print(f"DummyPrintTask {self.task_id}: {self.message}") + + +@pytest.fixture +def orchestrator(): + orch = Orchestrator(log_level="CRITICAL") # Suppress logging during tests + orch.register_task_type("print_task", DummyPrintTask) + return orch + +@pytest.fixture +def dag_filepath(tmp_path): + content = """ +dag: + tasks: + - task_id: task1 + type: print_task + message: "Hello from task1" + - task_id: task2 + type: print_task + message: "Hello from task2" + dependencies: [task1] +""" + f = tmp_path / "test_dag.yaml" + f.write_text(content) + return str(f) + +@pytest.fixture +def invalid_dag_filepath(tmp_path): + content = """ +dag: + tasks: + - task_id: task1 + type: non_existent_task + message: "Hello from task1" +""" + f = tmp_path / "invalid_dag.yaml" + f.write_text(content) + return str(f) + +@pytest.fixture +def dag_with_start_time_filepath(tmp_path): + content = """ +dag: + name: "scheduled_dag" + start_time: "2024-01-01T09:00:00" + tasks: + - task_id: task1 + type: print_task + message: "Hello from scheduled task1" + - task_id: task2 + type: print_task + message: "Hello from scheduled task2" + dependencies: [task1] +""" + f = tmp_path / "scheduled_dag.yaml" + f.write_text(content) + return str(f) + + +@pytest.fixture +def dag_with_cron_schedule_filepath(tmp_path): + content = """ +dag: + name: "cron_scheduled_dag" + cron_schedule: "0 9 * * *" # Every day at 9:00 AM + tasks: + - task_id: task1 + type: print_task + message: "Hello from scheduled task1" + - task_id: task2 + type: print_task + message: "Hello from scheduled task2" + dependencies: [task1] +""" + f = tmp_path / "cron_scheduled_dag.yaml" + f.write_text(content) + return str(f) + + +@pytest.fixture +def dag_with_invalid_start_time_filepath(tmp_path): + content = """ +dag: + name: "invalid_scheduled_dag" + start_time: "invalid-date-format" + tasks: + - task_id: task1 + type: print_task + message: "Hello from task1" +""" + f = tmp_path / "invalid_scheduled_dag.yaml" + f.write_text(content) + return str(f) + +def test_orchestrator_load_dag_from_file(orchestrator, dag_filepath): + dag = orchestrator.load_dag_from_file(dag_filepath) + assert isinstance(dag, DAG) + assert "task1" in dag.tasks + assert "task2" in dag.tasks + assert dag.tasks["task2"].dependencies == ["task1"] + +def test_orchestrator_load_invalid_dag(orchestrator, invalid_dag_filepath): + with pytest.raises(ValueError, match="Unknown task type: non_existent_task"): + orchestrator.load_dag_from_file(invalid_dag_filepath) + +def test_orchestrator_run_dag(orchestrator, dag_filepath): + dag = orchestrator.load_dag_from_file(dag_filepath) + + orchestrator.run_dag(dag) + + assert dag.tasks["task1"].status == TaskStatus.COMPLETED + assert dag.tasks["task2"].status == TaskStatus.COMPLETED + assert dag.tasks["task1"].executed # Check if execute was called + assert dag.tasks["task2"].executed # Check if execute was called + +def test_orchestrator_run_dag_fail_fast(orchestrator, tmp_path): + # Create a DAG with a failing task + content = """ +dag: + tasks: + - task_id: failing_task + type: failing_task_type + - task_id: subsequent_task + type: print_task + message: "This should not run" + dependencies: [failing_task] +""" + f = tmp_path / "failing_dag.yaml" + f.write_text(content) + + class FailingTask(BaseTask): + def execute_local(self): + raise Exception("Simulated task failure") + + orchestrator.register_task_type("failing_task_type", FailingTask) + + dag = orchestrator.load_dag_from_file(str(f)) + + with pytest.raises(Exception, match="Task failing_task failed: Simulated task failure"): + orchestrator.run_dag(dag, fail_fast=True) + + assert dag.tasks["failing_task"].status == TaskStatus.FAILED + assert dag.tasks["subsequent_task"].status == TaskStatus.PENDING # Should not have run + +def test_orchestrator_run_dag_no_fail_fast(orchestrator, tmp_path): + # Create a DAG with a failing task + content = """ +dag: + tasks: + - task_id: failing_task + type: failing_task_type + - task_id: subsequent_task + type: print_task + message: "This should run" + dependencies: [failing_task] +""" + f = tmp_path / "no_fail_fast_dag.yaml" + f.write_text(content) + + class FailingTask(BaseTask): + def execute_local(self): + raise Exception("Simulated task failure") + + orchestrator.register_task_type("failing_task_type", FailingTask) + + dag = orchestrator.load_dag_from_file(str(f)) + + orchestrator.run_dag(dag, fail_fast=False) + + assert dag.tasks["failing_task"].status == TaskStatus.FAILED + assert dag.tasks["subsequent_task"].status == TaskStatus.SKIPPED + + def test_orchestrator_run_dag_with_skipped_task_marks_dag_as_failed(orchestrator, tmp_path): + # Create a DAG with a failing task, which should cause the next to be skipped + content = """ + dag: + tasks: + - task_id: failing_task + type: failing_task_type + - task_id: subsequent_task + type: print_task + message: "This should be skipped" + dependencies: [failing_task] + """ + f = tmp_path / "dag_with_skipped.yaml" + f.write_text(content) + + class FailingTask(BaseTask): + def execute_local(self): + raise Exception("Simulated task failure") + + orchestrator.register_task_type("failing_task_type", FailingTask) + + dag = orchestrator.load_dag_from_file(str(f)) + + # Use run_dag_in_thread to get the final status + execution_id = orchestrator.run_dag_in_thread(dag, fail_fast=False) + + # Wait for the DAG to finish + time.sleep(1) # Adjust as needed + + with orchestrator.status_manager as sm: + dag_execution_details = sm.get_dag_execution_details(dag.dag_id, execution_id) + assert dag_execution_details["status"] == "failed" + +def test_orchestrator_load_dag_with_start_time(orchestrator, dag_with_start_time_filepath): + """Test loading a DAG with start_time parameter from YAML file.""" + dag = orchestrator.load_dag_from_file(dag_with_start_time_filepath) + + assert isinstance(dag, DAG) + assert dag.start_time is not None + assert dag.start_time == datetime(2024, 1, 1, 9, 0, 0) + assert "task1" in dag.tasks + assert "task2" in dag.tasks + assert dag.tasks["task2"].dependencies == ["task1"] + +def test_orchestrator_load_dag_with_invalid_start_time(orchestrator, dag_with_invalid_start_time_filepath): + """Test loading a DAG with invalid start_time format from YAML file.""" + with pytest.raises(ValueError, match="Invalid start_time format"): + orchestrator.load_dag_from_file(dag_with_invalid_start_time_filepath) + +def test_orchestrator_load_dag_without_start_time(orchestrator, dag_filepath): + """Test loading a DAG without start_time parameter from YAML file.""" + dag = orchestrator.load_dag_from_file(dag_filepath) + + assert isinstance(dag, DAG) + assert dag.start_time is None + assert dag.is_ready_to_start() # Should be ready to start immediately + assert dag.time_until_start() is None + +def test_orchestrator_load_dag_with_cron_schedule(orchestrator, dag_with_cron_schedule_filepath): + """Test loading a DAG with cron schedule from YAML file.""" + dag = orchestrator.load_dag_from_file(dag_with_cron_schedule_filepath) + + assert isinstance(dag, DAG) + assert dag.cron_schedule == "0 9 * * *" + assert dag.start_time is None + assert "task1" in dag.tasks + assert "task2" in dag.tasks + assert dag.tasks["task2"].dependencies == ["task1"] + + # Test schedule description + assert "Cron schedule" in dag.get_schedule_description() + + # Test next run time + next_run = dag.get_next_run_time() + assert next_run is not None + assert next_run.hour == 9 + assert next_run.minute == 0 + diff --git a/tests/Vecchi_test/test_server.py b/tests/Vecchi_test/test_server.py new file mode 100644 index 0000000..bcd0ab1 --- /dev/null +++ b/tests/Vecchi_test/test_server.py @@ -0,0 +1,335 @@ + +import pytest +from fastapi.testclient import TestClient +from unittest.mock import patch, MagicMock +import os +import sys +from datetime import datetime +from contextlib import asynccontextmanager + +# Add the src directory to the Python path +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), '../src'))) + +from maestro.server.app import app, lifespan +from maestro.shared.dag import DAG +from maestro.shared.task import Task + +# Create a concrete implementation of the abstract Task class for testing purposes +class ConcreteTask(Task): + def execute_local(self, **kwargs): + # This method is abstract in the base class, so we provide a minimal implementation. + print(f"Executing task {self.name} with action: {self.action}") + return f"Output of {self.name}" + +# Sample DAG for testing +@pytest.fixture +def sample_dag(): + task1 = ConcreteTask(task_id="task1", name="Task 1", action="echo 'Task 1'", dependencies=[]) + task2 = ConcreteTask(task_id="task2", name="Task 2", action="echo 'Task 2'", dependencies=["task1"]) + dag = DAG(dag_id="sample_dag") + dag.add_task(task1) + dag.add_task(task2) + return dag + +@pytest.fixture +def client(sample_dag): + # This fixture provides a test client for the FastAPI app. + # It patches the global `orchestrator` object that is used by the API endpoints. + + # 1. Create a mock for the Orchestrator. + mock_orchestrator = MagicMock() + mock_orchestrator.load_dag_from_file.return_value = sample_dag + mock_orchestrator.run_dag_in_thread.return_value = "test_execution_id" + + # 2. The original `lifespan` function creates a real Orchestrator. We need to prevent that. + # We replace the app's lifespan with a dummy one for the tests. + @asynccontextmanager + async def mock_lifespan(app): + # This context manager does nothing, preventing the real orchestrator creation. + yield + + app.router.lifespan_context = mock_lifespan + + # 3. Patch the global `orchestrator` variable in the `app` module with our mock. + # This is the instance that the endpoint functions will actually use. + with patch('maestro.server.app.orchestrator', mock_orchestrator): + with TestClient(app) as test_client: + yield test_client + + +def test_root(client): + response = client.get("/") + assert response.status_code == 200 + json_response = response.json() + assert json_response["message"] == "Maestro API Server" + assert json_response["status"] == "running" + +def test_submit_dag_success(client, sample_dag): + with patch('maestro.server.app.orchestrator.load_dag_from_file', + return_value=sample_dag): + with patch('maestro.server.app.orchestrator.run_dag_in_thread', + return_value="test_execution_id") as mock_run: + response = client.post("/dags/submit", + json={"dag_file_path": "path/to/dag.yaml", + "resume": False, + "fail_fast": True}) + + assert response.status_code == 200 + json_response = response.json() + assert json_response["dag_id"] == "sample_dag" + assert json_response["execution_id"] == "test_execution_id" + assert json_response["status"] == "submitted" + assert "submitted_at" in json_response + mock_run.assert_called_once_with(dag=sample_dag, + resume=False, + fail_fast=True) + +def test_submit_dag_file_not_found(client): + with patch('maestro.server.app.orchestrator.load_dag_from_file', side_effect=FileNotFoundError("DAG file not found")): + response = client.post("/dags/submit", json={"dag_file_path": "non_existent.yaml"}) + assert response.status_code == 400 + assert "DAG file not found" in response.json()["detail"] + +def test_get_dag_status_success(client): + mock_status_manager = MagicMock() + mock_status_manager.get_dag_execution_details.return_value = { + "execution_id": "test_execution_id", + "status": "running", + "started_at": datetime.now().isoformat(), + "completed_at": None, + "thread_id": "thread_123", + "tasks": [] + } + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/sample_dag/status?execution_id=test_execution_id") + assert response.status_code == 200 + json_response = response.json() + assert json_response["dag_id"] == "sample_dag" + assert json_response["execution_id"] == "test_execution_id" + assert json_response["status"] == "running" + +def test_get_dag_status_server_error(client): + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): + response = client.get("/dags/sample_dag/status") + assert response.status_code == 500 + assert "DB error" in response.json()["detail"] + +def test_get_dag_status_not_found(client): + mock_status_manager = MagicMock() + mock_status_manager.get_dag_execution_details.return_value = None + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/unknown_dag/status") + assert response.status_code == 404 + assert "DAG execution not found" in response.json()["detail"] + +def test_get_dag_logs_success(client): + mock_logs = [ + {"task_id": "task1", "level": "INFO", "message": "Task 1 started", "timestamp": datetime.now().isoformat(), "thread_id": "thread_123"} + ] + mock_status_manager = MagicMock() + mock_status_manager.get_execution_logs.return_value = mock_logs + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/sample_dag/logs") + assert response.status_code == 200 + json_response = response.json() + assert json_response["dag_id"] == "sample_dag" + assert len(json_response["logs"]) == 1 + assert json_response["logs"][0]["task_id"] == "task1" + +import asyncio + +def test_get_dag_logs_with_filters(client): + mock_logs = [ + {"task_id": "task1", "level": "INFO", "message": "Task 1 started", "timestamp": datetime.now().isoformat(), "thread_id": "thread_123"}, + {"task_id": "task2", "level": "DEBUG", "message": "Task 2 debug", "timestamp": datetime.now().isoformat(), "thread_id": "thread_123"} + ] + mock_status_manager = MagicMock() + mock_status_manager.get_execution_logs.return_value = mock_logs + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/sample_dag/logs?task_filter=task1&level_filter=INFO") + assert response.status_code == 200 + json_response = response.json() + assert len(json_response["logs"]) == 1 + assert json_response["logs"][0]["task_id"] == "task1" + assert json_response["logs"][0]["level"] == "INFO" + + +@pytest.mark.asyncio +async def test_stream_dag_logs(client): + mock_status_manager = MagicMock() + logs_queue = asyncio.Queue() + + async def mock_get_logs(*args, **kwargs): + try: + log = await asyncio.wait_for(logs_queue.get(), timeout=1.0) + return [log] + except asyncio.TimeoutError: + return [] + + mock_status_manager.get_execution_logs.side_effect = mock_get_logs + + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/sample_dag/logs/stream") + assert response.status_code == 200 + + # Simulate adding a log entry + log_entry = {"task_id": "task1", "level": "INFO", "message": "A new log", "timestamp": datetime.now().isoformat(), "thread_id": "thread_123"} + await logs_queue.put(log_entry) + + # The client-side implementation to read from the stream would be more complex. + # For this test, we are mainly ensuring the endpoint can be hit and returns a streaming response. + # A more thorough test would involve a client that can handle server-sent events. + +def test_get_dag_logs_server_error(client): + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): + response = client.get("/dags/sample_dag/logs") + assert response.status_code == 500 + assert "DB error" in response.json()["detail"] + + +def test_get_running_dags_empty(client): + mock_status_manager = MagicMock() + mock_status_manager.get_running_dags.return_value = [] + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/running") + assert response.status_code == 200 + json_response = response.json() + assert json_response["count"] == 0 + assert json_response["running_dags"] == [] + +def test_get_running_dags(client): + running_dags = [{"dag_id": "dag1", "execution_id": "exec1"}] + mock_status_manager = MagicMock() + mock_status_manager.get_running_dags.return_value = running_dags + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/running") + assert response.status_code == 200 + json_response = response.json() + assert json_response["count"] == 1 + assert json_response["running_dags"][0]["dag_id"] == "dag1" + +def test_get_running_dags_server_error(client): + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): + response = client.get("/dags/running") + assert response.status_code == 500 + assert "DB error" in response.json()["detail"] + +def test_list_dags_all(client): + all_dags = [{"dag_id": "dag1", "status": "completed"}, {"dag_id": "dag2", "status": "failed"}] + mock_status_manager = MagicMock() + mock_status_manager.get_all_dags.return_value = all_dags + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/list") + assert response.status_code == 200 + json_response = response.json() + assert json_response["count"] == 2 + assert json_response["title"] == "All DAGs" + +def test_list_dags_with_status_filter(client): + completed_dags = [{"dag_id": "dag1", "status": "completed"}] + mock_status_manager = MagicMock() + mock_status_manager.get_dags_by_status.return_value = completed_dags + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.get("/dags/list?status=completed") + assert response.status_code == 200 + json_response = response.json() + assert json_response["count"] == 1 + assert json_response["title"] == "DAGs with status: completed" + mock_status_manager.get_dags_by_status.assert_called_once_with("completed") + +def test_list_dags_server_error(client): + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): + response = client.get("/dags/list") + assert response.status_code == 500 + assert "DB error" in response.json()["detail"] + +def test_cancel_dag_success(client): + mock_status_manager = MagicMock() + mock_status_manager.cancel_dag_execution.return_value = True + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.post("/dags/sample_dag/cancel?execution_id=exec1") + assert response.status_code == 200 + assert response.json()["success"] is True + mock_status_manager.cancel_dag_execution.assert_called_once_with("sample_dag", "exec1") + +def test_cancel_dag_not_found(client): + mock_status_manager = MagicMock() + mock_status_manager.cancel_dag_execution.return_value = False + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.post("/dags/sample_dag/cancel") + assert response.status_code == 200 + assert response.json()["success"] is False + +def test_cancel_dag_server_error(client): + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): + response = client.post("/dags/sample_dag/cancel") + assert response.status_code == 500 + assert "DB error" in response.json()["detail"] + +def test_validate_dag_success(client, sample_dag): + with patch('maestro.server.app.orchestrator.load_dag_from_file', return_value=sample_dag): + response = client.post("/dags/validate", json={"dag_file_path": "path/to/dag.yaml"}) + assert response.status_code == 200 + json_response = response.json() + assert json_response["valid"] is True + assert json_response["dag_id"] == "sample_dag" + assert json_response["total_tasks"] == 2 + assert json_response["execution_order"] == ["task1", "task2"] + +def test_validate_dag_invalid(client): + with patch('maestro.server.app.orchestrator.load_dag_from_file', side_effect=Exception("Invalid DAG")): + response = client.post("/dags/validate", json={"dag_file_path": "path/to/invalid_dag.yaml"}) + assert response.status_code == 200 + json_response = response.json() + assert json_response["valid"] is False + assert "Invalid DAG" in json_response["error"] + +def test_validate_dag_missing_path(client): + response = client.post("/dags/validate", json={}) + assert response.status_code == 200 + json_response = response.json() + assert json_response["valid"] is False + assert "dag_file_path is required" in json_response["error"] + +def test_cleanup_old_executions(client): + mock_status_manager = MagicMock() + mock_status_manager.cleanup_old_executions.return_value = 5 + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(return_value=mock_status_manager)): + response = client.delete("/dags/cleanup?days=15") + assert response.status_code == 200 + json_response = response.json() + assert json_response["deleted_count"] == 5 + assert "Deleted 5 execution records older than 15 days" in json_response["message"] + mock_status_manager.cleanup_old_executions.assert_called_once_with(15) + +def test_cleanup_old_executions_server_error(client): + with patch('maestro.server.app.orchestrator.status_manager', __enter__=MagicMock(side_effect=Exception("DB error"))): + response = client.delete("/dags/cleanup") + assert response.status_code == 500 + assert "DB error" in response.json()["detail"] + +def test_main_start_server(): + with patch('maestro.server.app.uvicorn.run') as mock_run: + with patch('argparse.ArgumentParser') as mock_argparse: + mock_argparse.return_value.parse_args.return_value = MagicMock(host="127.0.0.1", port=8080, log_level="debug") + from maestro.server.app import main + main() + mock_run.assert_called_once() + + +@pytest.mark.asyncio +async def test_lifespan(monkeypatch): + mock_orchestrator = MagicMock() + monkeypatch.setattr("maestro.server.app.Orchestrator", lambda **kwargs: mock_orchestrator) + + # Import the app module to access the orchestrator variable + from maestro.server import app as app_module + + async with lifespan(app): + # Access the orchestrator from the app module, not as a global in test scope + assert app_module.orchestrator is not None + + mock_orchestrator.executor.shutdown.assert_called_once_with(wait=True) + + + diff --git a/tests/Vecchi_test/test_status_manager.py b/tests/Vecchi_test/test_status_manager.py new file mode 100644 index 0000000..1bb29d6 --- /dev/null +++ b/tests/Vecchi_test/test_status_manager.py @@ -0,0 +1,350 @@ +import pytest +import threading +import uuid +from maestro.server.internals.status_manager import StatusManager +from maestro.server.internals.models import DagORM +from datetime import datetime, timedelta + +@pytest.fixture +def status_manager(): + """Create a fresh in-memory StatusManager for each test.""" + # Use a unique database path for each test to avoid shared state + import tempfile + import os + temp_file = tempfile.NamedTemporaryFile(delete=False) + temp_file.close() + + sm = StatusManager(temp_file.name) + yield sm + + # Clean up the temporary file + try: + os.unlink(temp_file.name) + except (FileNotFoundError, PermissionError): + pass + + +def test_set_and_get_task_status(status_manager): + dag_id = "dag_123" + task_id = "task_1" + + with status_manager as sm: + # Set status without execution_id (should create default execution) + sm.set_task_status(dag_id, task_id, "completed") + status = sm.get_task_status(dag_id, task_id) + assert status == "completed" + + # Set status with specific execution_id + execution_id = "exec_456" + # Create execution first + sm.create_dag_execution(dag_id, execution_id) + sm.set_task_status(dag_id, task_id, "running", execution_id) + status = sm.get_task_status(dag_id, task_id, execution_id) + assert status == "running" + + +def test_get_dag_status(status_manager): + dag_id = "dag_123" + execution_id = "exec_456" + + with status_manager as sm: + # Create execution before setting task status + sm.create_dag_execution(dag_id, execution_id) + sm.set_task_status(dag_id, "task_1", "completed", execution_id) + sm.set_task_status(dag_id, "task_2", "running", execution_id) + + dag_status = sm.get_dag_status(dag_id, execution_id) + assert dag_status["task_1"] == "completed" + assert dag_status["task_2"] == "running" + + +def test_reset_dag_status(status_manager): + dag_id = "dag_123" + execution_id = "exec_456" + + with status_manager as sm: + # Create execution before setting task status + sm.create_dag_execution(dag_id, execution_id) + sm.set_task_status(dag_id, "task_1", "completed", execution_id) + sm.reset_dag_status(dag_id, execution_id) + + dag_status = sm.get_dag_status(dag_id, execution_id) + assert not dag_status + + +def test_create_dag_execution(status_manager): + dag_id = "dag_123" + execution_id = "exec_456" + + with status_manager as sm: + sm.create_dag_execution(dag_id, execution_id) + execution_details = sm.get_dag_execution_details(dag_id, execution_id) + assert execution_details["execution_id"] == execution_id + + +def test_logging_and_retrieval(status_manager): + dag_id = "dag_123" + execution_id = "exec_456" + task_id = "task_1" + + with status_manager as sm: + # Create execution before logging + sm.create_dag_execution(dag_id, execution_id) + sm.log_message(dag_id, execution_id, task_id, "INFO", "Test log message.") + logs = sm.get_execution_logs(dag_id, execution_id) + + assert len(logs) == 1 + assert logs[0]["message"] == "Test log message." + + +def test_cleanup_old_executions(status_manager): + dag_id = "dag_123" + old_execution_id = "old_exec" + new_execution_id = "new_exec" + old_date = datetime.now().replace(year=2000).isoformat() + + with status_manager as sm: + # Create old execution by direct SQL insertion + sm._conn.execute(""" + INSERT OR IGNORE INTO dags (id) VALUES (?) + """, (dag_id,)) + sm._conn.execute(""" + INSERT INTO executions (id, dag_id, started_at, status, thread_id, pid) + VALUES (?, ?, ?, ?, ?, ?)""", + (old_execution_id, dag_id, old_date, "completed", str(threading.get_ident()), str(threading.get_ident())) + ) + + # Create new execution + sm.create_dag_execution(dag_id, new_execution_id) + + # Clean up executions older than a year + cleaned = sm.cleanup_old_executions(days_to_keep=365) + + assert cleaned == 1 + executions = sm.get_dag_history(dag_id) + assert len(executions) == 1 + assert executions[0]["execution_id"] == new_execution_id + + +def test_get_task_status_fallback(status_manager): + """Test the fallback logic for getting task status.""" + dag_id = "dag_123" + task_id = "task_1" + execution_id = "exec_456" + + with status_manager as sm: + # Set task status without specific execution_id (uses "default") + sm.set_task_status(dag_id, task_id, "completed") + + # Create a new execution but don't set task status for it + sm.create_dag_execution(dag_id, execution_id) + + # Should fallback to default execution when task not found for specific execution + status = sm.get_task_status(dag_id, task_id, execution_id) + assert status == "completed" + + +def test_get_running_dags(status_manager): + """Test getting all running DAGs.""" + dag_id1 = "dag_1" + dag_id2 = "dag_2" + execution_id1 = "exec_1" + execution_id2 = "exec_2" + + with status_manager as sm: + # Create running executions + sm.create_dag_execution(dag_id1, execution_id1) + sm.create_dag_execution(dag_id2, execution_id2) + + # Mark one as completed + sm.update_dag_execution_status(dag_id1, execution_id1, "completed") + + running_dags = sm.get_running_dags() + assert len(running_dags) == 1 + assert running_dags[0]["dag_id"] == dag_id2 + assert running_dags[0]["execution_id"] == execution_id2 + + +def test_get_dags_by_status(status_manager): + """Test getting DAGs by status.""" + dag_id1 = "dag_1" + dag_id2 = "dag_2" + execution_id1 = "exec_1" + execution_id2 = "exec_2" + + with status_manager as sm: + # Create executions + sm.create_dag_execution(dag_id1, execution_id1) + sm.create_dag_execution(dag_id2, execution_id2) + + # Update statuses + sm.update_dag_execution_status(dag_id1, execution_id1, "completed") + sm.update_dag_execution_status(dag_id2, execution_id2, "failed") + + completed_dags = sm.get_dags_by_status("completed") + failed_dags = sm.get_dags_by_status("failed") + + assert len(completed_dags) == 1 + assert completed_dags[0]["dag_id"] == dag_id1 + + assert len(failed_dags) == 1 + assert failed_dags[0]["dag_id"] == dag_id2 + + +def test_get_all_dags(status_manager): + """Test getting all DAGs.""" + dag_id1 = "dag_1" + dag_id2 = "dag_2" + execution_id1 = "exec_1" + execution_id2 = "exec_2" + + with status_manager as sm: + sm.create_dag_execution(dag_id1, execution_id1) + sm.create_dag_execution(dag_id2, execution_id2) + + all_dags = sm.get_all_dags() + assert len(all_dags) == 2 + + dag_ids = [dag["dag_id"] for dag in all_dags] + assert dag_id1 in dag_ids + assert dag_id2 in dag_ids + + +def test_get_dag_summary(status_manager): + """Test getting DAG summary statistics.""" + with status_manager as sm: + # Create multiple executions with different statuses + for i in range(3): + dag_id = f"dag_{i}" + execution_id = f"exec_{i}" + sm.create_dag_execution(dag_id, execution_id) + + if i == 0: + sm.update_dag_execution_status(dag_id, execution_id, "completed") + elif i == 1: + sm.update_dag_execution_status(dag_id, execution_id, "failed") + # i == 2 remains "running" + + summary = sm.get_dag_summary() + + assert summary["total_executions"] == 3 + assert summary["unique_dags"] == 3 + assert summary["status_counts"]["completed"] == 1 + assert summary["status_counts"]["failed"] == 1 + assert summary["status_counts"]["running"] == 1 + + +def test_cancel_dag_execution(status_manager): + """Test cancelling a DAG execution.""" + dag_id = "dag_123" + execution_id = "exec_456" + + with status_manager as sm: + sm.create_dag_execution(dag_id, execution_id) + + # Cancel the execution + cancelled = sm.cancel_dag_execution(dag_id, execution_id) + assert cancelled is True + + # Check the status was updated + details = sm.get_dag_execution_details(dag_id, execution_id) + assert details["status"] == "cancelled" + assert details["completed_at"] is not None + + +def test_mark_incomplete_tasks_as_failed(status_manager): + """Test marking incomplete tasks as failed.""" + dag_id = "dag_123" + execution_id = "exec_456" + + with status_manager as sm: + sm.create_dag_execution(dag_id, execution_id) + + # Set various task statuses + sm.set_task_status(dag_id, "task_1", "running", execution_id) + sm.set_task_status(dag_id, "task_2", "pending", execution_id) + sm.set_task_status(dag_id, "task_3", "completed", execution_id) + + # Mark incomplete tasks as failed + sm.mark_incomplete_tasks_as_failed(dag_id, execution_id) + + # Check results + assert sm.get_task_status(dag_id, "task_1", execution_id) == "failed" + assert sm.get_task_status(dag_id, "task_2", execution_id) == "failed" + assert sm.get_task_status(dag_id, "task_3", execution_id) == "completed" # Should remain unchanged + + +def test_initialize_tasks_for_execution(status_manager): + """Test initializing tasks for an execution.""" + dag_id = "dag_123" + execution_id = "exec_456" + task_ids = ["task_1", "task_2", "task_3"] + + with status_manager as sm: + sm.create_dag_execution(dag_id, execution_id) + sm.initialize_tasks_for_execution(dag_id, execution_id, task_ids) + + # All tasks should be pending + for task_id in task_ids: + status = sm.get_task_status(dag_id, task_id, execution_id) + assert status == "pending" + + +def test_task_status_with_timestamps(status_manager): + """Test that task status updates include proper timestamps.""" + dag_id = "dag_123" + execution_id = "exec_456" + task_id = "task_1" + + with status_manager as sm: + sm.create_dag_execution(dag_id, execution_id) + + # Set to running (should set started_at) + sm.set_task_status(dag_id, task_id, "running", execution_id) + + # Set to completed (should set completed_at) + sm.set_task_status(dag_id, task_id, "completed", execution_id) + + # Get execution details to check timestamps + details = sm.get_dag_execution_details(dag_id, execution_id) + task = next(t for t in details["tasks"] if t["task_id"] == task_id) + + assert task["started_at"] is not None + assert task["completed_at"] is not None + assert task["status"] == "completed" + + +def test_get_dag_definition_returns_saved_definition(status_manager): + class DummyDag: + dag_id = "dummy_dag" + + def to_dict(self): + return {"dag_id": self.dag_id, "tasks": {}} + + with status_manager as sm: + dag = DummyDag() + sm.save_dag_definition(dag) + + stored_definition = sm.get_dag_definition(dag.dag_id) + + assert stored_definition == dag.to_dict() + + +def test_save_dag_definition_persists_filepath(status_manager): + class DummyDag: + dag_id = "dummy_dag" + + def to_dict(self): + return {"dag_id": self.dag_id, "tasks": {}} + + dag_file_path = "/tmp/workflows/sample.yaml" + + with status_manager as sm: + dag = DummyDag() + sm.save_dag_definition(dag, dag_file_path) + + with sm.Session() as session: + stored_dag = session.query(DagORM).filter_by(id=dag.dag_id).first() + assert stored_dag is not None + assert stored_dag.dag_filepath == dag_file_path + diff --git a/tests/Work_in_progress/test_status_manager.py b/tests/Work_in_progress/test_status_manager.py new file mode 100644 index 0000000..b1c4d37 --- /dev/null +++ b/tests/Work_in_progress/test_status_manager.py @@ -0,0 +1,294 @@ +import threading +from datetime import datetime, timedelta + +import pytest +from sqlalchemy import inspect + +from maestro.server.internals.models import Base, DagORM, ExecutionORM, LogORM, TaskORM +from maestro.server.internals.status_manager import StatusManager + + +# ----------------------------- +# Fixture: StatusManager pulito +# ----------------------------- +@pytest.fixture +def status_manager(): + # Reset singleton + StatusManager._instance = None + sm = StatusManager(":memory:") # DB in-memory + yield sm + StatusManager._instance = None + + +# ----------------------------- +# Test: Singleton e inizializzazione DB +# ----------------------------- +def test_singleton_instance(status_manager): + # Dopo init, get_instance deve restituire lo stesso oggetto + instance = StatusManager.get_instance() + assert instance is status_manager + + +def test_db_tables_created(status_manager): + inspector = inspect(status_manager.engine) + tables = inspector.get_table_names() + + from maestro.server.internals.models import DagORM, ExecutionORM, LogORM, TaskORM + + expected_tables = [ + DagORM.__tablename__, + ExecutionORM.__tablename__, + TaskORM.__tablename__, + LogORM.__tablename__, + ] + + for tbl in expected_tables: + assert tbl in tables + + +# ----------------------------- +# Test: DAG execution lifecycle +# ----------------------------- +def test_create_and_update_execution(status_manager): + dag_id = "dag1" + exec_id = "exec1" + + # Create execution + status_manager.create_dag_execution(dag_id, exec_id) + latest = status_manager.get_latest_execution(dag_id) + assert latest["execution_id"] == exec_id + assert latest["status"] == "running" + + # Update status to completed + status_manager.update_dag_execution_status(dag_id, exec_id, "completed") + latest = status_manager.get_latest_execution(dag_id) + assert latest["status"] == "completed" + assert latest["completed_at"] is not None + + +# ----------------------------- +# Test: Task status and output +# ----------------------------- +def test_set_and_get_task_status_and_output(status_manager): + dag_id = "dag1" + exec_id = "exec1" + task_id = "task1" + + # Initialize task + status_manager.initialize_tasks_for_execution(dag_id, exec_id, [task_id]) + + # Set task status + status_manager.set_task_status(dag_id, task_id, "running", exec_id) + status = status_manager.get_task_status(dag_id, task_id, exec_id) + assert status == "running" + + # Set output + output_data = {"result": 42} + status_manager.set_task_output(dag_id, task_id, exec_id, output_data) + output = status_manager.get_task_output(dag_id, task_id, exec_id) + assert output == output_data + + +# ----------------------------- +# Test: DAG status retrieval +# ----------------------------- +def test_get_dag_status_and_reset(status_manager): + dag_id = "dag1" + exec_id = "exec1" + tasks = ["t1", "t2"] + + status_manager.initialize_tasks_for_execution(dag_id, exec_id, tasks) + status_manager.set_task_status(dag_id, "t1", "completed", exec_id) + status_manager.set_task_status(dag_id, "t2", "running", exec_id) + + dag_status = status_manager.get_dag_status(dag_id, exec_id) + assert dag_status == {"t1": "completed", "t2": "running"} + + # Reset DAG status + status_manager.reset_dag_status(dag_id, exec_id) + dag_status = status_manager.get_dag_status(dag_id, exec_id) + assert dag_status == {} + + +# ----------------------------- +# Test: Logging +# ----------------------------- +def test_logs(status_manager): + dag_id = "dag1" + exec_id = "exec1" + task_id = "task1" + + # Log a message + status_manager.log_message(dag_id, exec_id, task_id, "INFO", "Test log") + logs = status_manager.get_execution_logs(dag_id, exec_id) + assert len(logs) == 1 + assert logs[0]["message"] == "Test log" + + +# ----------------------------- +# Test: DAG definitions +# ----------------------------- +class DummyDag: + def __init__(self, dag_id): + self.dag_id = dag_id + + def to_dict(self): + return {"dag_id": self.dag_id, "tasks": []} + + +def test_save_and_get_dag_definition(status_manager): + dag_id = "dag1" + dag = DummyDag(dag_id) + status_manager.save_dag_definition(dag) + definition = status_manager.get_dag_definition(dag_id) + assert definition["dag_id"] == dag_id + + +# ----------------------------- +# Test: DAG deletion +# ----------------------------- +def test_delete_dag(status_manager): + dag_id = "dag_to_delete" + dag = DummyDag(dag_id) + status_manager.save_dag_definition(dag) + + # Delete DAG + deleted = status_manager.delete_dag(dag_id) + assert deleted == 1 + # Check non-existent DAG + deleted = status_manager.delete_dag("nonexistent") + assert deleted == 0 + + +# ----------------------------- +# Test: DAG id validation and uniqueness +# ----------------------------- +def test_dag_id_validation_and_uniqueness(status_manager): + valid_id = "dag_123" + invalid_id = "dag 123!" + assert status_manager.validate_dag_id(valid_id) is True + assert status_manager.validate_dag_id(invalid_id) is False + + # Uniqueness + dag = DummyDag("unique_dag") + status_manager.save_dag_definition(dag) + assert status_manager.check_dag_id_uniqueness("unique_dag") is False + assert status_manager.check_dag_id_uniqueness("new_dag") is True + + +# ----------------------------- +# Test: Cleanup old executions +# ----------------------------- +def test_cleanup_old_executions(status_manager): + dag_id = "dag1" + exec_id = "old_exec" + old_time = datetime.now() - timedelta(days=60) + + # Create old execution manually + with status_manager.Session.begin() as session: + session.add(DagORM(id=dag_id)) + session.add( + ExecutionORM( + id=exec_id, dag_id=dag_id, status="completed", started_at=old_time + ) + ) + + count = status_manager.cleanup_old_executions(days_to_keep=30) + assert count == 1 + + +# ----------------------------- +# Test: Cancel DAG execution +# ----------------------------- +def test_cancel_dag_execution(status_manager): + dag_id = "dag1" + exec_id = "exec1" + task_ids = ["t1", "t2"] + + status_manager.create_dag_execution(dag_id, exec_id) + status_manager.initialize_tasks_for_execution(dag_id, exec_id, task_ids) + status_manager.set_task_status(dag_id, "t1", "running", exec_id) + + cancelled = status_manager.cancel_dag_execution(dag_id, exec_id) + assert cancelled is True + dag_status = status_manager.get_dag_status(dag_id, exec_id) + assert all(status == "cancelled" for status in dag_status.values()) + + +# ----------------------------- +# Test: Generate unique DAG id +# ----------------------------- +def test_generate_unique_dag_id(status_manager): + dag_id = status_manager.generate_unique_dag_id() + assert isinstance(dag_id, str) + assert status_manager.validate_dag_id(dag_id) + + +# ----------------------------- +# Test: Get DAG summary +# ----------------------------- +def test_get_dag_summary(status_manager): + dag_id = "dag1" + exec_id = "exec1" + status_manager.create_dag_execution(dag_id, exec_id) + summary = status_manager.get_dag_summary() + assert "total_executions" in summary + assert "unique_dags" in summary + assert "status_counts" in summary + + +# ----------------------------- +# Test extra: fallback get_task_status +# ----------------------------- +def test_task_status_fallback(status_manager): + dag_id = "dag_fallback" + exec_id = "exec1" + task_id = "task1" + + # Inseriamo un task globale (execution_id=None) + status_manager.initialize_tasks_for_execution(dag_id, None, [task_id]) + status_manager.set_task_status(dag_id, task_id, "completed", execution_id=None) + + # Chiediamo lo status per un execution-specific task inesistente + status = status_manager.get_task_status(dag_id, task_id, execution_id=exec_id) + assert status == "completed" # deve ripiegare sul task globale + + +# ----------------------------- +# Test extra: multi-thread execution +# ----------------------------- +import threading + + +def test_multithreaded_execution(status_manager): + dag_id = "dag_thread" + exec_ids = ["exec1", "exec2"] + + def create_exec(exec_id): + status_manager.create_dag_execution(dag_id, exec_id) + + threads = [threading.Thread(target=create_exec, args=(eid,)) for eid in exec_ids] + for t in threads: + t.start() + for t in threads: + t.join() + + # Controlla che entrambe le execution siano presenti + executions = status_manager.get_all_dags() + latest_exec_ids = [d["execution_id"] for d in executions if d["execution_id"]] + for eid in exec_ids: + assert eid in latest_exec_ids + +# ----------------------------- +# Test extra: insertion_order dei task +# ----------------------------- +def test_task_insertion_order(status_manager): + dag_id = "dag_order" + exec_id = "exec_order" + tasks = ["t3", "t1", "t2"] # ordine arbitrario + + status_manager.initialize_tasks_for_execution(dag_id, exec_id, tasks) + details = status_manager.get_dag_execution_details(dag_id, exec_id) + retrieved_order = [t["task_id"] for t in details["tasks"]] + # L'ordine deve rispettare insertion_order + assert retrieved_order == tasks From 0061b061e75bcd7f7b3fa69ab7a8fe817e24f8f9 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 6 Dec 2025 21:34:26 +0100 Subject: [PATCH 33/38] Phase 1 completed - patched base.py --- src/maestro/Roadmap - Dependency policy | 116 +++++++++++++ src/maestro/server/internals/orchestrator.py | 168 +++++++++++-------- src/maestro/server/tasks/base.py | 37 +++- 3 files changed, 248 insertions(+), 73 deletions(-) create mode 100644 src/maestro/Roadmap - Dependency policy diff --git a/src/maestro/Roadmap - Dependency policy b/src/maestro/Roadmap - Dependency policy new file mode 100644 index 0000000..aaed5b6 --- /dev/null +++ b/src/maestro/Roadmap - Dependency policy @@ -0,0 +1,116 @@ +📍 Fase 1 — Modifica del modello delle task +File da modificare: + +maestro/server/tasks/base.py + +Cosa fare: + +- aggiungere la Enum DependencyPolicy + +- aggiungere il campo Pydantic dependency_policy a BaseTask + +- impostare il valore di default scelto + +- aggiornare la descrizione nel model + +---------------------------------------------------------- + +📍 Fase 2 — Caricamento DAG +File da controllare/modificare: + +maestro/server/internals/dag_loader.py + +Cosa fare: + +verificare che il loader YAML passi correttamente il campo dependency_policy + +non dovrebbe servire nessuna modifica strutturale, perché Pydantic fa già il parse automatico + +ma va aggiunto 1–2 test case di DAG per confermare che funzioni + +---------------------------------------------------------- + +📍 Fase 3 — Orchestratore: logica di scheduling +File da modificare: + +maestro/server/internals/orchestrator.py + +In particolare, la sezione in cui l’orchestratore: + +- identifica le task in stato PENDING + +- controlla lo stato delle dipendenze + +- decide se mandarle in esecuzione + +Cosa fare: + +- aggiungere la funzione di supporto evaluate_dependencies(task, upstream_tasks) + +- integrare la decisione (run / wait / skip) nel loop principale dell’orchestratore + +- marcare le task come SKIPPED quando la policy lo impone + +- aggiornare correttamente il database tramite StatusManager + +---------------------------------------------------------- + +📍 Fase 4 — StatusManager e database +File da verificare: + +maestro/server/internals/status_manager.py + +maestro/server/internals/models.py (TaskORM) + +Cosa fare: + +- verificare che lo stato SKIPPED sia già gestito (lo è, perché lo abbiamo già implementato per branching) + +- assicurare che la chiamata .update_task salvi il nuovo stato + +Probabilmente non serve alcuna modifica qui, ma va controllato per sicurezza. + +---------------------------------------------------------- + +📍 Fase 5 — Esecutori (Python, Bash, Print) +File da controllare: + +maestro/server/tasks/bash_task.py + +maestro/server/tasks/python_task.py + +maestro/server/tasks/print_task.py + +Cosa verificare: + +- nessuna modifica alle task: se una task ha stato SKIPPED l’esecutore non la esegue comunque (l’orchestratore non la invia). + +---------------------------------------------------------- + +📍 Fase 6 — Aggiornamento dei test +File: + +tests/test_dag.py + +tests/test_orchestrator.py + +(eventuali test aggiuntivi che hai) + +Cosa fare: + +- aggiungere test per i 3 casi: + +-- ALL_SUCCESS + +-- ANY_SUCCESS + +-- ALWAYS_RUN (il nostro “none”) + +---------------------------------------------------------- + +📍 Fase 7 — Documentazione +File da aggiornare: + +- README del progetto + +- documentazione interna delle DAG diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index f0d6353..380e9ba 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -1,31 +1,29 @@ import logging -import uuid -import threading -from datetime import datetime import re +import threading import time -from concurrent.futures import ThreadPoolExecutor, Future, as_completed -from typing import Any, Type, Dict, Optional, Set, List +import uuid +from concurrent.futures import Future, ThreadPoolExecutor, as_completed +from datetime import datetime +from typing import Any, Dict, List, Optional, Set, Tuple, Type +from apscheduler.schedulers.background import BackgroundScheduler from rich import get_console from rich.logging import RichHandler -from maestro.shared.dag import DAG, DAGStatus -from maestro.shared.task import TaskStatus -from maestro.server.tasks.base import BaseTask -from maestro.server.internals.task_registry import TaskRegistry from maestro.server.internals.dag_loader import DAGLoader -from maestro.server.internals.status_manager import StatusManager - from maestro.server.internals.executors.factory import ExecutorFactory - -from apscheduler.schedulers.background import BackgroundScheduler +from maestro.server.internals.status_manager import StatusManager +from maestro.server.internals.task_registry import TaskRegistry +from maestro.server.tasks.base import BaseTask +from maestro.shared.dag import DAG, DAGStatus +from maestro.shared.task import TaskStatus class DatabaseLogHandler(logging.Handler): """Thread-safe logging handler that writes task logs to StatusManager.""" - _ansi_escape = re.compile(r'\x1B\[[0-?]*[ -/]*[@-~]') + _ansi_escape = re.compile(r"\x1B\[[0-?]*[ -/]*[@-~]") def __init__(self, status_manager: StatusManager): super().__init__() @@ -79,7 +77,7 @@ def emit(self, record): task_id=task_id or "system", level=record.levelname, message=msg, - timestamp=timestamp + timestamp=timestamp, ) except Exception: @@ -93,7 +91,7 @@ def __init__( self, log_level: str = "INFO", status_manager: Optional[StatusManager] = None, - db_path: Optional[str] = "maestro.db" + db_path: Optional[str] = "maestro.db", ): self.task_registry = TaskRegistry() self.dag_loader = DAGLoader(self.task_registry) @@ -115,7 +113,6 @@ def __init__( self._execution_stop_events: Dict[str, threading.Event] = {} self.task_executor = ThreadPoolExecutor(max_workers=10) - def _setup_logging(self, log_level: str): """Setup global logging with RichHandler (console) and prepare DB handler.""" @@ -144,7 +141,6 @@ def _setup_logging(self, log_level: str): self.db_handler = DatabaseLogHandler(self.status_manager) self.db_handler.setLevel(logging.DEBUG) - def register_task_type(self, name: str, task_class: Type[BaseTask]): """Register a custom task type.""" self.task_registry.register(name, task_class) @@ -157,18 +153,20 @@ def load_dag_from_file(self, filepath: str, dag_id: Optional[str] = None) -> DAG def schedule_dag(self, dag: DAG): """Schedules a DAG to run based on its cron schedule.""" if not dag.cron_schedule: - self.logger.warning(f"DAG {dag.dag_id} has no cron schedule. Cannot schedule.") + self.logger.warning( + f"DAG {dag.dag_id} has no cron schedule. Cannot schedule." + ) return job_id = f"dag:{dag.dag_id}" self.scheduler.add_job( self.execute_scheduled_dag, - trigger='cron', + trigger="cron", id=job_id, name=dag.dag_id, args=[dag.dag_id], replace_existing=True, - **dag.cron_schedule_to_aps_kwargs() + **dag.cron_schedule_to_aps_kwargs(), ) self.logger.info(f"DAG {dag.dag_id} scheduled with cron: {dag.cron_schedule}") @@ -186,7 +184,9 @@ def execute_scheduled_dag(self, dag_id: str): self.logger.info(f"Scheduler triggered for DAG: {dag_id}") dag_filepath = self.status_manager.get_dag_filepath(dag_id) if not dag_filepath: - self.logger.error(f"No file path found for scheduled DAG {dag_id}. Cannot execute.") + self.logger.error( + f"No file path found for scheduled DAG {dag_id}. Cannot execute." + ) return try: @@ -202,7 +202,7 @@ def run_dag_in_thread( resume: bool = False, fail_fast: bool = True, status_callback=None, - dag_filepath: Optional[str] = None # Add this parameter + dag_filepath: Optional[str] = None, # Add this parameter ) -> str: """Execute DAG in a separate thread with concurrency.""" # Use provided execution_id or generate a new one @@ -225,11 +225,15 @@ def run_dag_in_thread( # Tasks should already exist for this execution, don't reinitialize else: # Create new execution - sm.create_dag_execution(dag.dag_id, execution_id, dag_filepath=dag_filepath) + sm.create_dag_execution( + dag.dag_id, execution_id, dag_filepath=dag_filepath + ) # Initialize all tasks with pending status only if not resuming if not resume: task_ids = list(dag.tasks.keys()) - sm.initialize_tasks_for_execution(dag.dag_id, execution_id, task_ids) + sm.initialize_tasks_for_execution( + dag.dag_id, execution_id, task_ids + ) # the body of the thread def execute(): @@ -240,7 +244,7 @@ def execute(): resume=resume, fail_fast=fail_fast, status_callback=status_callback, - stop_event=stop_event + stop_event=stop_event, ) # After execution, check the final status of tasks @@ -248,20 +252,30 @@ def execute(): final_statuses = sm.get_dag_status(dag.dag_id, execution_id) # Count task statuses - has_failed = any(status == "failed" for status in final_statuses.values()) - has_running = any(status == "running" for status in final_statuses.values()) + has_failed = any( + status == "failed" for status in final_statuses.values() + ) + has_running = any( + status == "running" for status in final_statuses.values() + ) # Determine overall DAG status if has_running: # Should not happen after execution completes, but handle it dag.status = DAGStatus.RUNNING - sm.update_dag_execution_status(dag.dag_id, execution_id, "running") + sm.update_dag_execution_status( + dag.dag_id, execution_id, "running" + ) elif has_failed: dag.status = DAGStatus.FAILED - sm.update_dag_execution_status(dag.dag_id, execution_id, "failed") + sm.update_dag_execution_status( + dag.dag_id, execution_id, "failed" + ) else: dag.status = DAGStatus.COMPLETED - sm.update_dag_execution_status(dag.dag_id, execution_id, "completed") + sm.update_dag_execution_status( + dag.dag_id, execution_id, "completed" + ) except Exception as e: with self.status_manager as sm: @@ -291,7 +305,7 @@ def run_dag( status_manager=None, progress_tracker=None, status_callback=None, - stop_event: Optional[threading.Event] = None + stop_event: Optional[threading.Event] = None, ): """Execute DAG with concurrent task execution.""" dag_id = dag.dag_id @@ -305,14 +319,14 @@ def run_dag( db_handler = self.db_handler task_loggers = [ - 'maestro.server.tasks.terraform_task', - 'maestro.server.tasks.extended_terraform_task', - 'maestro.server.tasks.print_task', - 'maestro.server.tasks.python_task', - 'maestro.server.tasks.bash_task', - 'maestro.core.executors.ssh', - 'maestro.core.executors.docker', - 'maestro.core.executors.local', + "maestro.server.tasks.terraform_task", + "maestro.server.tasks.extended_terraform_task", + "maestro.server.tasks.print_task", + "maestro.server.tasks.python_task", + "maestro.server.tasks.bash_task", + "maestro.core.executors.ssh", + "maestro.core.executors.docker", + "maestro.core.executors.local", ] for logger_name in task_loggers: @@ -342,7 +356,7 @@ def run_dag( status_callback=status_callback, progress_tracker=progress_tracker, stop_event=stop_event, - db_handler=db_handler + db_handler=db_handler, ) finally: @@ -353,7 +367,6 @@ def run_dag( if db_handler in logger.handlers: logger.removeHandler(db_handler) - def _run_dag_concurrent( self, dag: DAG, @@ -363,7 +376,7 @@ def _run_dag_concurrent( status_callback, progress_tracker, stop_event: Optional[threading.Event], - db_handler + db_handler, ): """Execute DAG tasks concurrently based on dependencies.""" dag_id = dag.dag_id @@ -381,7 +394,9 @@ def _run_dag_concurrent( for task_id in list(pending_tasks): task_status = sm.get_task_status(dag_id, task_id, execution_id) if task_status == "completed": - self.logger.info(f"Task {task_id} already completed (resume mode)") + self.logger.info( + f"Task {task_id} already completed (resume mode)" + ) dag.tasks[task_id].status = TaskStatus.COMPLETED completed_tasks.add(task_id) pending_tasks.remove(task_id) @@ -394,9 +409,16 @@ def _run_dag_concurrent( while pending_tasks or running_tasks: # Check for cancellation if stop_event and stop_event.is_set(): - self.logger.info(f"DAG {dag_id} execution {execution_id} was cancelled") + self.logger.info( + f"DAG {dag_id} execution {execution_id} was cancelled" + ) self._handle_cancellation( - dag, execution_id, pending_tasks, running_tasks, sm, status_callback + dag, + execution_id, + pending_tasks, + running_tasks, + sm, + status_callback, ) break @@ -432,7 +454,9 @@ def _run_dag_concurrent( # Check if dependencies failed if self._has_failed_dependencies(task, failed_tasks, skipped_tasks): - self.logger.warning(f"Skipping task {task_id} because its dependencies failed") + self.logger.warning( + f"Skipping task {task_id} because its dependencies failed" + ) task.status = TaskStatus.SKIPPED sm.set_task_status(dag_id, task_id, "skipped", execution_id) skipped_tasks.add(task_id) @@ -450,7 +474,7 @@ def _run_dag_concurrent( execution_id=execution_id, status_callback=status_callback, progress_tracker=progress_tracker, - db_handler=db_handler + db_handler=db_handler, ) running_tasks[task_id] = future pending_tasks.remove(task_id) @@ -459,14 +483,13 @@ def _run_dag_concurrent( if not ready_tasks and running_tasks: threading.Event().wait(0.1) - def _find_ready_tasks( self, dag: DAG, pending_tasks: Set[str], completed_tasks: Set[str], failed_tasks: Set[str], - skipped_tasks: Set[str] + skipped_tasks: Set[str], ) -> List[str]: """ Find tasks that are ready to run. @@ -522,12 +545,8 @@ def _find_ready_tasks( return ready_tasks - def _has_failed_dependencies( - self, - task, - failed_tasks: Set[str], - skipped_tasks: Set[str] + self, task, failed_tasks: Set[str], skipped_tasks: Set[str] ) -> bool: """ A task is blocked ONLY if one of its dependencies FAILED. @@ -537,7 +556,6 @@ def _has_failed_dependencies( """ return any(dep in failed_tasks for dep in task.dependencies) - def _evaluate_condition(self, task, dag_id: str, execution_id: str) -> bool: """Evaluate the boolean condition of a task using output_of().""" condition = getattr(task, "condition", None) @@ -567,12 +585,14 @@ def output_of(tid: str): ) return False - print("[DEBUG] condition eval:", + print( + "[DEBUG] condition eval:", task.task_id, expr, "output_of(check_condition)=", safe_evaluator_globals["output_of"]("check_condition"), - flush=True) + flush=True, + ) def _execute_task_async( self, @@ -581,9 +601,8 @@ def _execute_task_async( execution_id: str, status_callback, progress_tracker, - db_handler + db_handler, ): - """Execute a single task asynchronously WITH retry support.""" task_id = task.task_id @@ -655,7 +674,6 @@ def _execute_task_async( return # task completata con successo → si esce - except BaseException as e: # IGNORA SOLO I SEGNALI CHE DEVONO DAVVERO FERMARE TUTTO if isinstance(e, KeyboardInterrupt): @@ -680,13 +698,11 @@ def _execute_task_async( status_callback() raise Exception(f"Task {task_id} failed: {e}") from e - finally: # Pulizia del contesto di logging if db_handler: db_handler.clear_context() - def _handle_cancellation( self, dag: DAG, @@ -694,7 +710,7 @@ def _handle_cancellation( pending_tasks: Set[str], running_tasks: Dict[str, Future], sm, - status_callback + status_callback, ): """Handle DAG cancellation.""" # Cancel all running tasks @@ -733,10 +749,14 @@ def visualize_dag(self, dag: DAG): task = dag.tasks[task_id] status_color = self._get_status_color(task.status) - self.console.print(f"{i}. {task_id} ({task.status.value})", style=status_color) + self.console.print( + f"{i}. {task_id} ({task.status.value})", style=status_color + ) if task.dependencies: - self.console.print(f" Dependencies: {', '.join(task.dependencies)}", style="dim") + self.console.print( + f" Dependencies: {', '.join(task.dependencies)}", style="dim" + ) def _get_status_color(self, status: TaskStatus) -> str: """Get color for task status.""" @@ -745,7 +765,7 @@ def _get_status_color(self, status: TaskStatus) -> str: TaskStatus.RUNNING: "yellow", TaskStatus.COMPLETED: "green", TaskStatus.FAILED: "red", - TaskStatus.SKIPPED: "magenta" + TaskStatus.SKIPPED: "magenta", } return color_map.get(status, "white") @@ -757,7 +777,7 @@ def get_dag_status(self, dag: DAG) -> Dict[str, Any]: "running": 0, "completed": 0, "failed": 0, - "skipped": 0 + "skipped": 0, } for task_id, task in dag.tasks.items(): @@ -765,14 +785,14 @@ def get_dag_status(self, dag: DAG) -> Dict[str, Any]: tasks_status[task_id] = { "status": status, "dependencies": task.dependencies, - "type": task.__class__.__name__ + "type": task.__class__.__name__, } summary[status] += 1 return { "tasks": tasks_status, "summary": summary, - "total_tasks": len(dag.tasks) + "total_tasks": len(dag.tasks), } def cancel_dag_execution(self, dag_id: str, execution_id: str = None) -> bool: @@ -808,8 +828,12 @@ def stop_all_running_dags(self) -> int: if self.cancel_dag_execution(dag_id, execution_id): stopped_count += 1 - self.logger.info(f"Stopped DAG {dag_id} (execution: {execution_id})") + self.logger.info( + f"Stopped DAG {dag_id} (execution: {execution_id})" + ) else: - self.logger.warning(f"Failed to stop DAG {dag_id} (execution: {execution_id})") + self.logger.warning( + f"Failed to stop DAG {dag_id} (execution: {execution_id})" + ) return stopped_count diff --git a/src/maestro/server/tasks/base.py b/src/maestro/server/tasks/base.py index 8abca3e..3c4aa1c 100644 --- a/src/maestro/server/tasks/base.py +++ b/src/maestro/server/tasks/base.py @@ -1,7 +1,27 @@ +from enum import Enum from typing import Optional + from pydantic import Field + from maestro.shared.task import Task + +class DependencyPolicy(str, Enum): + """ + Come interpretare lo stato delle dipendenze upstream per decidere se + una task deve essere eseguita. + + Valori: + - ALL -> tutte le dipendenze devono essere COMPLETED (successo) + - ANY -> almeno una dipendenza deve essere COMPLETED + - NONE -> esegui comunque quando tutte le dipendenze sono in stato terminale + """ + + ALL = "all" + ANY = "any" + NONE = "none" + + class BaseTask(Task): """Base class for a task, inheriting from the core Task which is a Pydantic model.""" @@ -11,7 +31,22 @@ class BaseTask(Task): # 🆕 Campi che devono essere letti dallo YAML retries: int = Field(default=0, description="Number of retries on failure") - retry_delay: int = Field(default=0, description="Delay in seconds between retry attempts") + retry_delay: int = Field( + default=0, description="Delay in seconds between retry attempts" + ) + + # 🔁 Politica sulle dipendenze (opzionale; default = ANY) + dependency_policy: DependencyPolicy = Field( + default=DependencyPolicy.ANY, + description=( + "How to interpret upstream tasks status when deciding if this task should run. " + "Valid values: 'all', 'any', 'none'.\n" + "'all' -> run only if ALL upstream tasks are COMPLETED (success).\n" + "'any' -> run if AT LEAST ONE upstream task is COMPLETED.\n" + "'none' -> run whenever all upstream tasks are in a terminal state " + "(COMPLETED/FAILED/SKIPPED) — useful for cleanup/finalization tasks." + ), + ) def execute_local(self): raise NotImplementedError("Subclasses must implement this method.") From bcb9d1b25f457a3fcd04196ebbdd7d98456feb34 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 6 Dec 2025 22:02:40 +0100 Subject: [PATCH 34/38] Phase 2 completed - still test on dag_loader.py on its way --- tests/test_dag_loader.py | 56 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/tests/test_dag_loader.py b/tests/test_dag_loader.py index 5f65767..b0a8137 100644 --- a/tests/test_dag_loader.py +++ b/tests/test_dag_loader.py @@ -7,6 +7,7 @@ from maestro.server.internals.dag_loader import DAGLoader from maestro.server.internals.task_registry import TaskRegistry from maestro.server.tasks.base import BaseTask +from maestro.server.tasks.print_task import PrintTask from maestro.shared.dag import DAG @@ -261,3 +262,58 @@ def test_load_dag_from_dict_multiple_tasks(dag_loader, mock_task_registry): # Ora accediamo ai task correttamente (dag.tasks è dict) task_ids = {task.task_id for task in dag.tasks.values()} assert task_ids == {"task1", "task2", "task3"} + + +# ----------------------------- +# Test per dependency_policy nel task +# ----------------------------- +def test_dependency_policy_loaded_from_yaml(tmp_path): + # Registry + registry = TaskRegistry() + registry.register("PrintTask", PrintTask) + + # File YAML + yaml_path = tmp_path / "dag_test.yaml" + yaml_path.write_text( + """ + dag: + tasks: + - task_id: task1 + type: PrintTask + message: "Hello" + dependency_policy: any + """ + ) + + loader = DAGLoader(registry) + dag = loader.load_dag_from_file(str(yaml_path)) + + # Access diretto al dict + t1 = dag.tasks["task1"] + + assert t1.dependency_policy.value == "any" # NB: è un Enum + + +def test_dependency_policy_default(tmp_path): + registry = TaskRegistry() + registry.register("PrintTask", PrintTask) + + yaml_path = tmp_path / "dag_test2.yaml" + yaml_path.write_text( + """ + dag: + tasks: + - task_id: t2 + type: PrintTask + message: "Test" + """ + ) + + loader = DAGLoader(registry) + dag = loader.load_dag_from_file(str(yaml_path)) + + t2 = dag.tasks["t2"] + + # Default impostato in BaseTask + assert t2.dependency_policy == t2.dependency_policy.__class__.ANY + assert t2.dependency_policy.value == "any" From 23add319f8ead141dc2b63f5accf5ef4121fd653 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 6 Dec 2025 22:24:12 +0100 Subject: [PATCH 35/38] Phase 3 completed - updated orchestrator.py - Added evaluate_dependencies() - Updated _run_dag_concurrent() - Deleted _has_failed_dependencies() --- src/maestro/server/internals/orchestrator.py | 123 ++++++++++++------- 1 file changed, 81 insertions(+), 42 deletions(-) diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index 380e9ba..335c930 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -378,7 +378,14 @@ def _run_dag_concurrent( stop_event: Optional[threading.Event], db_handler, ): - """Execute DAG tasks concurrently based on dependencies.""" + """ + Execute DAG tasks concurrently based on dependencies and dependency_policy. + + - Tasks are evaluated with evaluate_dependencies() + - Tasks can be in states: run, wait, skip + - Skipped tasks are updated in StatusManager + """ + dag_id = dag.dag_id # Initialize task tracking sets @@ -443,45 +450,47 @@ def _run_dag_concurrent( self._cancel_running_tasks(running_tasks) raise Exception(f"Task {task_id} failed: {e}") - # Find tasks ready to run - ready_tasks = self._find_ready_tasks( - dag, pending_tasks, completed_tasks, failed_tasks, skipped_tasks - ) + # Build upstream status dict + upstream_statuses = {tid: "completed" for tid in completed_tasks} + upstream_statuses.update({tid: "failed" for tid in failed_tasks}) + upstream_statuses.update({tid: "skipped" for tid in skipped_tasks}) + upstream_statuses.update({tid: "pending" for tid in pending_tasks}) + upstream_statuses.update({tid: "running" for tid in running_tasks}) - # Submit ready tasks for execution - for task_id in ready_tasks: + # Loop sui task pending + for task_id in list(pending_tasks): task = dag.tasks[task_id] - - # Check if dependencies failed - if self._has_failed_dependencies(task, failed_tasks, skipped_tasks): - self.logger.warning( - f"Skipping task {task_id} because its dependencies failed" + action = evaluate_dependencies(task, upstream_statuses) + + if action == "run": + # Submit task for execution + self.logger.info(f"Submitting task {task_id} for execution") + future = self.task_executor.submit( + self._execute_task_async, + task=task, + dag_id=dag_id, + execution_id=execution_id, + status_callback=status_callback, + progress_tracker=progress_tracker, + db_handler=db_handler, ) + running_tasks[task_id] = future + pending_tasks.remove(task_id) + + elif action == "wait": + continue # lascia in pending + + elif action == "skip": task.status = TaskStatus.SKIPPED sm.set_task_status(dag_id, task_id, "skipped", execution_id) skipped_tasks.add(task_id) pending_tasks.remove(task_id) if status_callback: status_callback() - continue - - # Submit task for execution - self.logger.info(f"Submitting task {task_id} for execution") - future = self.task_executor.submit( - self._execute_task_async, - task=task, - dag_id=dag_id, - execution_id=execution_id, - status_callback=status_callback, - progress_tracker=progress_tracker, - db_handler=db_handler, - ) - running_tasks[task_id] = future - pending_tasks.remove(task_id) - # Brief sleep to prevent busy waiting - if not ready_tasks and running_tasks: - threading.Event().wait(0.1) + # Evita busy-waiting + if pending_tasks or running_tasks: + threading.Event().wait(0.05) def _find_ready_tasks( self, @@ -496,7 +505,7 @@ def _find_ready_tasks( Un task è "ready" quando: - tutte le sue dipendenze sono in stato terminale - (completed, failed o skipped) + (completed, failed o skipped) - e, se ha una condition, questa viene valutata a True. I task la cui condition risulta False vengono marcati come SKIPPED qui. @@ -545,17 +554,6 @@ def _find_ready_tasks( return ready_tasks - def _has_failed_dependencies( - self, task, failed_tasks: Set[str], skipped_tasks: Set[str] - ) -> bool: - """ - A task is blocked ONLY if one of its dependencies FAILED. - - Skipped dependencies do not block downstream tasks — this enables - branching with a final merge step, where only one branch runs. - """ - return any(dep in failed_tasks for dep in task.dependencies) - def _evaluate_condition(self, task, dag_id: str, execution_id: str) -> bool: """Evaluate the boolean condition of a task using output_of().""" condition = getattr(task, "condition", None) @@ -837,3 +835,44 @@ def stop_all_running_dags(self) -> int: ) return stopped_count + + +def evaluate_dependencies(task: BaseTask, upstream_statuses: Dict[str, str]) -> str: + """ + Decide lo stato della task in base alle dipendenze. + + Parametri: + task: il task da valutare + upstream_statuses: dict {task_id: TaskStatus} dei task upstream + + Ritorna: + "run" -> tutte le condizioni soddisfatte, task pronta a partire + "wait" -> task deve attendere + "skip" -> task deve essere skippata + """ + policy = getattr(task, "dependency_policy", "none") + dependencies = getattr(task, "dependencies", []) + + if not dependencies or policy == "none": + return "run" + + statuses = [upstream_statuses.get(dep) for dep in dependencies] + + if policy == "all": + if any(s is None or s in ("pending", "running") for s in statuses): + return "wait" + if all(s == "completed" for s in statuses): + return "run" + # Se una dipendenza ha fallito, skip task + if any(s == "failed" for s in statuses): + return "skip" + + elif policy == "any": + if any(s == "completed" for s in statuses): + return "run" + if all(s in ("pending", "running", None) for s in statuses): + return "wait" + if all(s in ("failed", "skipped") for s in statuses): + return "skip" + + return "run" From dd9d4ebb9299fbf4df387e01f5e93cfa293e12bd Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 6 Dec 2025 22:32:08 +0100 Subject: [PATCH 36/38] Phase 4 completed - Patched set_task_status() in status_manager.py and checked models.py --- src/maestro/server/internals/status_manager.py | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/src/maestro/server/internals/status_manager.py b/src/maestro/server/internals/status_manager.py index 7976e24..6c06ce5 100644 --- a/src/maestro/server/internals/status_manager.py +++ b/src/maestro/server/internals/status_manager.py @@ -169,7 +169,10 @@ def set_task_status( if not task: task = TaskORM(dag_id=dag_id, id=task_id, execution_id=execution_id) session.add(task) + task.status = status + task.thread_id = str(threading.current_thread().ident) + if status == "running": task.started_at = datetime.now() elif status in ["completed", "failed", "cancelled", "skipped"]: @@ -251,12 +254,7 @@ def get_task_status( # ---------------------------------------------------------------------- # Metodo: get_task_output - def get_task_output( - self, - dag_id: str, - task_id: str, - execution_id: str - ) -> Any: + def get_task_output(self, dag_id: str, task_id: str, execution_id: str) -> Any: """Return the decoded output of a task, or None if not found.""" with self.Session() as session: task = ( From 97fe7a57d346ebfbe6cb96caf48e18c5b57680d9 Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 6 Dec 2025 23:33:45 +0100 Subject: [PATCH 37/38] Fine tuning of orchestrator.py completed --- .../4.1.1.Conditional_branching.yaml | 5 + ...3.1.conditional_branching_triple_tree.yaml | 3 +- ....3.3.conditional_branching_hypergraph.yaml | 12 +- maestro.db | Bin 0 -> 204800 bytes src/maestro/server/internals/orchestrator.py | 118 ++++++++++++++---- 5 files changed, 104 insertions(+), 34 deletions(-) create mode 100644 maestro.db diff --git a/examples/2_New_examples/4.1.1.Conditional_branching.yaml b/examples/2_New_examples/4.1.1.Conditional_branching.yaml index 48ecbac..43f94d5 100644 --- a/examples/2_New_examples/4.1.1.Conditional_branching.yaml +++ b/examples/2_New_examples/4.1.1.Conditional_branching.yaml @@ -21,3 +21,8 @@ dag: message: "Eseguo branch B" dependencies: [check_condition] condition: "output_of('check_condition')['branch'] == 'branch_b'" + - task_id: end_message + type: "PrintTask" + params: + message: "End of DAG - success" + dependencies: [branch_a, branch_b] diff --git a/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml b/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml index db32861..6e9619c 100644 --- a/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml +++ b/examples/2_New_examples/4.3.1.conditional_branching_triple_tree.yaml @@ -30,7 +30,7 @@ dag: params: code: | import random - condition = random.choice(["1", "2", "3"]) + condition = random.choice(["1", "2", "3", "4"]) print("Condition evaluated:", condition) task_output = {"branch": condition} dependencies: ["init"] @@ -105,6 +105,7 @@ dag: - "branch1_step2" - "branch2_step2" - "branch3_step2" + dependency_policy: none # --------------------------------------------------------- # FINAL TASK diff --git a/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml b/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml index 6bef54f..5f6868e 100644 --- a/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml +++ b/examples/2_New_examples/4.3.3.conditional_branching_hypergraph.yaml @@ -54,21 +54,21 @@ dag: type: "PythonTask" params: code: | - print("Branch A - Step 1") + print("Branch A - Step 1 completed") dependencies: ["evaluate_condition"] condition: "output_of('evaluate_condition')['branch'] == 'A'" - task_id: "A2" type: "BashTask" params: - command: "echo 'Branch A - Step 2'" + command: "echo 'Branch A - Step 2 completed'" dependencies: ["A1"] # -------- BRANCH B -------- - task_id: "B1" type: "BashTask" params: - command: "echo 'Branch B - Step 1'" + command: "echo 'Branch B - Step 1 completed'" dependencies: ["evaluate_condition"] condition: "output_of('evaluate_condition')['branch'] == 'B'" @@ -76,7 +76,7 @@ dag: type: "PythonTask" params: code: | - print("Branch B - Step 2") + print("Branch B - Step 2 completed") dependencies: ["B1"] # -------- BRANCH C -------- @@ -84,14 +84,14 @@ dag: type: "PythonTask" params: code: | - print("Branch C - Step 1") + print("Branch C - Step 1 completed") dependencies: ["evaluate_condition"] condition: "output_of('evaluate_condition')['branch'] == 'C'" - task_id: "C2" type: "BashTask" params: - command: "echo 'Branch C - Step 2'" + command: "echo 'Branch C - Step 2 completed'" dependencies: ["C1"] # ========================================================= diff --git a/maestro.db b/maestro.db new file mode 100644 index 0000000000000000000000000000000000000000..b6d4ab442aa10e8bad7f0cba55736302dd3c25ae GIT binary patch literal 204800 zcmeIb33wc5b{NOLS*5&%LHN8&Lgppk~)^i)?>6D>Bn zTYZp(D2}GbjuSgGamJ24p7pWbi9J?6JDYDk*>Mslnb;fK*^M|fF^;NYKaY?s=NO8zxV$4zyF*r&4D>`%kR=oMig!a#(92m=uYA`Cb&A5gpUU=o=(i*x96VErA3Miz@uUuSqedE&QbLW>XzfQco`Z{r{q|IgLp396( zUm@c$1A~n>%b>^Rl~%KA{O!jl?VUmPTgFFH)6>)OzqW16xT!R5z~Jzfa?Q=SM4WO? z$OW0ET+esr>bj-h*40+CRIPOT0Ge#8f$@Y;09WcwUCVsH>Cp~CAO_;MeO*lcH zL!W%%;^ozIFJ3?dIAwllj<~$~!s_MK3oENv(Dxb!VZ5JpJ;?bD`-b_w?A%;xWcti> zH-Z{u$kWD;^d}D|Q`1vZ@o!BUq$^jiH_*Qk7vazqi*DA8UBq$I4S{V#`=nL)HxlcYkoSHs>9bVsZ3u$c06d z@w-b0y;!Q0(BiXM12JB)RMu-sbJNK}Dy(GB3MbC4zOc0Z${MkFSnA7+5$AFwGr;w-Xca z_RuK2J%G;pljyuZfxP(R$OJGpl!f;r6F(T6_?e0SFqxbD`;#A<{M_WXPBte0pNaor z@?TE=>f}R{>g1Wp_{7)XL(wb3K!kw^0}%!y3`7`+Fc4uN!a#(92m=uYA`A=y0}l+( z$G0JtWwY2S=hd=Ozo8$Q96k}hq3e}=qo!9>{qTdsbX>WmH>z8D{#L16E-71$YUR+8 z;kkILRse^3zSgXk^{ws6so{mVx~bH+N|oz*_@${sK+VMTFdJ`G>aAk23{f|5xmMo3 zUaE|LV3>?IZs~e6zon|>Ql(i{Z;VY0ACIfmYB_&hX*C+72)<^$)o7w8zM7hoV=6fd~T;1|kd`C=9g6kEGYI;D^ztrbX4W;C)O%M1xm8Lz>plzDx_R2NTeuf3@;ZcSrs_H zrqK-~agpJ2f+*6Iq<0wfGy1+CsmR=_fm{^^ri=3d4 znyzsqrxr9)R)DdKx~l3jR}?u)@M@Ob0RqP6Sz2JcT*M=m7e$t)x{xhBnqEI{0HjHx z%rUw~im+UPv6&)Skbs|52CmbBEV8oF(Q12u!saN66@>s$*c>YfzyQJY8qZ?+BdA4M zmo$Ye%77M5R|HZ5S^qWktofhQFn)j6>~BpGrS)jR?5+UW{wVqXEc*upEXdU zs32hwAFI5sk{qk5q%2Aj34)O+s+1xyOwrcDkv88Q8b-{)>H-n!<7T+SGb|-{;hanw zOG8yd0*B{FiY|aOLPlwsr33ulWBNbhGe!*A>Dvsk>CMF6Ls0QD3N7MZTeoF=M_t|+48 z)Y=6WT1H$VQXqId{DbbuIZ0w*;=!;S8A`A71~UVTX{sowB&{;Q%sLOSC>&a1g3gLm zL6=xP=)~LwC`v{M;#dYmryD`$9Z*6P!L5Mdy~K!kw^0}%!y3`7_R z!hn=mpO1$y3?JbV>nGxUT5u0Pomi*i;Y_WE#KihsynmbMgv)hdtbPQo&%LKCyl>uJvhKN*Ra@ zV+0y8kP^aLlQj4=oLMDdV5U#|$MBiNIv)>fs2GB!F`o%xQ9$_s$;sbEa$&xc{~shK zqH7TbA`CFQ)*Zvu>uZ@}BJv+#B}2XBXt zqVxaANrn*1NlU(g@XE5bm8fd~T;1|keZ7>FLH%fk$6V(kIsA`c0+Wg6e~LwOY}P&VY_G2s_Cd$iVQAypWuo zUym2+N=4mdplPaJqdkRFc9IIRg6jXpP@{n86=5L4K!kw^0}%!y3`7`+Fc4uN!a#(9 z2m=uYA`E=PVgS|ui_ZVwu;h&<7GWU5K!kw^0}%!y3`7`+Fc4uN!a#(92m=uY3=Bl) z|A-MH3`7`+Fc4uN!a#(92m=uYA`C-my4vq;PY-Jdy4%ru%xy_s*8wsiPZrByD&HML%?aJ z%T1+mqXD0~GXuBO8=brk;~Khajpe-!k3@hJOJzM@Q<|Fyo`ucomcF1VHz7Zyt}R^I zuBZ#IURkMX`ofl?H=6b8f__`ss+IM|0+YX>-^$y6aye*ToLj+JB}zGOriYX&*K^y- zRvAr8(`$M~(<^F8M<8FlW=^%)LVwMmd+=%WLH)L_wwl#Cx?HZRit%kbV+F2SII%Hl z{cV1)1c);5YhA7D&AK@v13TE~)7Hj8fS~kf<O)zut9TQ>jw2c|_*SxYZLzIfv)A4A}5>$>S^m0S*e6X-FgU2jx%v{iKHwl!L z18iYF>F&l13{ck_t#T6vxPwR7fVtsMRM31IYQ0ozVur8~*xareNVv4!+^kmCfTw2e z!e=p%@Qgb%4o|f{!;Qls2F>;b%~1%gUc#}rKZ;oU|ietGa~~AL))n5G#G+au?eSDT|3B@L&q`k3Nq({X_LV7A{^7U zOf~2>oFj9&+zt+qKni~$A1~1YKCz!jIAbu?0XXmP)CBCX>)M83WZA`rS+L7GtiXm{ z*fQzDM{I^L%)d+TYxiAt8GKi;*`2^d*T5!gKhw1S8Wvj9hUql{J3blO3Y=HeCRpCR zT!^ha`1ZbHL5wYj*CaecI_<;$2m4IKOZUj4s8+YOfZ2d9GrGE2C1#hw_n>YPON+!S zrG_DvkXg4e`!rE*5F|~!O>F9lMvxVPN1_N@jF*+hrq5uEsT%A7DB#->(JjBg@_i=y za);=(dKH-~4Z^z^V;Qz7WE+DW={PV&qpa&Sf_|DXKXkTQ(Tzz2@ zIxt;~4!26p&3pm0+;23igMGSF$_``0DWuSoFkfAp+p*OFp~rDp@~Zvry93bh!qe^m znH%i$8iambC7w8S>hjWsvlq`3CsCLfLF#W36mjioqPeM8HY$Xvmv78;Yyf7M*CGK{ z1jqf0>uZl+&`0(c zt9ZCPyyovT3Ai6RJ7>g+?E@pgSs)Z)5DE@^T9F_(`?CO)!HAI8N_8^`)gJ~}$ag;R z%}VBdaO^)@s_QC*)NVsyXuYL2TXo&C_-Mpa3k37*f~MbGKu1;L?YFJ7{)H+5gf+ef z@w5B&)W3^8p^GT8kNzFb4KD-k$LZge=UsPvunV9t?&e+`|93_J_xYcpA#}|+|0k+3 z_&<6@7>FMnk&A)ANep~xrEoFoo8wPgYTnz-NnOO zkV&~wfpotHaPCza;y+Fk>}i5KP4L$?DvdW=3Rq{*t;$msqFAjH6{1u@>46QnjgtnK zEe}0mecs?~>vv#kV^8WoPp^>2yy7+68N=l}_ak0%XY<{=r{)Z%x`mj^c#5B^mI>pNDj0qZzB2v_ z!BuW=zK5mUH}@ky1qo35+>aMZcxiP%PW_iGbJ$HJ;_314AN%vM>!V*CeIomr%zw&UO#fs$nYuOdiz81aKb`oU z#G}I>8T$VCuf-?e3-9-8Ge*+s+dk&|VC;yIcTWLl6#o|Pn)XWD0{vj8+Z z)t(9qwawHW>$?H_GB{>X?AEobjiv@6fY*rTR;}0|Ixqe-bsqkrnU~Qu{MSihq28)2 zKnnhiz-fpwZ%;!IWSJK@slD=`HH`=E{WRFq@{0h!D5J^1#w($}SeOPfZZcL4Q?G66dRxrqZl}6b+sS&oPX59#0-YqKKWkI{0Z|y>SsiZ8>!U z0p&!BV!ePqco+dXH?BuSK6pIjsknSi}MP{3@VPT2m_U=TS8 z#I{Ft&rc#|otkRL!@`K7;l2Beoj0K3`0lRueDL50Ldy!Ah!JYbw$X`9?8N%~)YD7z zj*O9=bpN$ABs18%3yo2Xs2ds-?kDFYhUIz8clNgT)Hs4>cIs~5(DdA-SNn*7^aCs9 zBvC*_HVJK`?%^>Lb-;@QL7laPoiVt9z%dLj`!G2(Y64dpFmM4u#DPH4GRNVa$tL&1 zSp+0Kb)|10{il9&!Pr%49GaOk5n?$AI8oqP3X|6c?oh@A?zuq#XQh8Y`5u-D28m(( zM13Ni1$>#JrB9aYDN}ZoixdbZ&C@K_mu$>El*+{T^%GO)m*#`2T5*D)siL5gw92p~ zr}HAIC|rT$ctK}Hs-R1(?kG^=m4rcsHr%iowJ>tLNK3pAw!VFSOqe*C7e&FlV;oB&nC5qbFHaF7-(ql-fx_Si;S83jWNYy@J&z>@4&P;0Z7f(8 zG#>~b$8(@WdUo8g;Y@7aq^YO62`AGZ8h2=>WgpsC$HZ3x&g%71S zIXyZGg>u$&v?TMCkHwA-A#eqnV@luKx0MPrsUE1AVKpHQDy@{V@00w zf`-oj561Y|E9)?j5m#qlxp;Z?{OoCB_S~f_>&7oe}oz z_pBLik!N!>ErLetW!**6n!qEwngHE>0*2`i3hd36t95+~!C0y_TE;J06|SKxjZzc+ zP`31DFz^&7v6N4DeE77zefNp{u-mtXpAcr5oGbxmz5LXEV!pFi_6c3jy%OpNLa+>J zvX3jJrdF!yWeAY=>)$4rAb$KOtw)~9#7>$jkRdFr*lLZZVUaK@scQ-dE{+8bRtgG9 zYm_ePEG_CH6&8t{hdo(Hzsa$z+_8Mv^5W!N7N9c~j()-0b1WKgGBVxS0c=%VIF*f& z>(f*1hnD6`l}4=u2DM~JAv`>xpL|;2$ zO%%=-_h_PEhD7b(tCkhGwrxAwOeH4J?HBXlk@QHTaET!T&%I=;TS* zYR9Qd*o@ms{VCoAJc%mRktZ#h zxpHZpIC*~6Fhak2pXRfR*tr75cc@6$vJ9mDSSh;xqJY-q~&x@r5 z-|BmkFlFJ5!A;B3NkJm6pz_X0YFo}py%7i~%kd08EZM8)@y8LMhRr1mhlN!-JXQ|I z5M+YLh_p{9o_h?jh^cgh0dW9|2W)`?!*T*Pv)S0>j%DGj2m5zwY2FhqsK^YZD!dBD z4dhT1WC~o1GzV!`U`o`bf(~cUo>t4geYhSbKyK9>xFJOUv19I0bPgn^wn9N|?i6_i z2a$9;d@R{PD@OPW1ir|p30i_h9_x;qy&)FV1hs-=HOJsVqcT3x|0tf&V*G2 zUcV4{>$!-!_#gtJErZ}?C03-p0u;IN{e{D6Lph{Tl+)f;4~O%3mq1zbe3x&6+T1lr zHmKrK`1p`MquJ>DTsMU7nEbmHi=in*u^vf*CS94}NJe~fzY*OOdzdA|)BB#q19_@d zRU0_05ZC>oq3{QIyHzMx)m#J%P+6~>sy1>c)d`ZlPR-_K=TI|(a6yLV!Hy25F}}Iq zXzZ89yC_56=fY<0^TG~PEFm$l)*zJD?LIO=ts{Le$}{^;q`Qv4yCC{KU)^S;FkaF< z?Ip9J7qe+6@3NTj&1f-0F*U2Y7kkeZvnNh?$98`2X26Dv87;d|Etrsv&MQxi zYx_)`9xVB3M_TVF(qjiQ6Q~hjR`6uW1$<(kQSqRslG{dWP1~t3ef3$EHVM9gvOF?_N!% zPQF(K8RHDiaS9pd^+?KqsJzb|WF3ljBxQp807p`0e@U4K-`yBVk?)m2%6NV8PD$gv z7fG8)+5~AmkxqkZWA0h+aa;A+kxsLZm6;>x{697P>S};lo3p7*dBO;{Q0V#V2F`1pbR&5e6a* z3>pI;pURT+>yN~i=3{Zw_}!@&6e7~7aiav4Yc!`#(2i^pA-4mkoj}x(L{q#l+r6SZ}PxccJ@RN z+xMB|-Gz)Ob{`cnpndfr$)2$7%E%No5ddX6fTH#cCPL6W*GsT3gJ933vr`uB3?|=G&?uRsJc82#je%%>ui$JyoXVcyE$(*Y^@lAD z&;@N)(my1f$R5~mIRKugy|LTF?dg&1F$^4L@_|4G@&Ki!x!4>R1pq;Xmmii>LvJM! zUv6ZsoiK zTe%8*H!w|eDA(FNd*FeGZ;` zqGT7-!yBY?4`4y#nV8N_k){wfG#;- zkGz+2=HT3H*P6f_mbm?dcVugq*vmPKC78B~z40VH*d=R;GRQ@$4)H$t5879yE|v zkh$#MxXdCh0uL+NgRJ!1k0HpM3a1rh9|p2+8wVs*ixOZ4tGjqvr`|pmJUYGCqtg=Y zGd+#8k3QNVi@AGPk~GZl-PadPUqTKK&tcr;7>O4c#=E|f?W2zbgBA)iJwX#egYdaK z%$hh7sO;EIfR|U|?T5WIvqW(ir1j(&FkoQBV2&An`=Q{`L*b>@=nPnuz5F`VoY=3D^*Qj*iv+8XIfp*Z!4&sWn+QKU(j#m?LWEP&J$G3sO}nAtM!Ts{Z7GO zdew5B;vn1u5Y!&!|CjUTR*9+s?)$G$7OGo+tuxH-!Yjij*dcOq%yV?8zZSGP2|NYYf9qInoVH_!rPF7+L!cO8y zON#RU4OQP|_c~xCJc#-KycC`PcdA)L=l}b9LWs`)AR(=^xeq3pfBc6s3Jt($&=3Fkc>^13*;&PwA3Mem@s;bpDUd|5|6K)Nku* z3)K$EmmF8?IvAx5??U*D;OP7xo&Q0L3dc(2GBQ|1=YQSf>Wj|*NFdqjJ#KeDb9W&1 z|4?`+I{zERlMMOj@wG;$gic}M==dKU|I5lYs1|66YFKh@=WEc+O-0vdl*%>|>yi6E za{mXi1+>wtAT4x8hXQy6F7F(b|5rDoVLVaDuxZ(u4D>&<2`Gk<}iFm z^|-q@QH_QY2S_0 zT>5)cf0_E0RC=VA{H^4r#NUnocI@KN-PrFXo{SfVKQ;7cldXwgo>(9MP2V}OJ-yti zzGa4XT9qF>nvtLuX>%pP$H?9kf z2T!%9mx5_RhZwylO`sGA&v;55GX`2HC?yAl{Yp>GJX#-gC z&?M2Gd?`EaDijxD)E?*%p|YMR3*MqcCOT-2wo3rvswtWF^ot!rxlEyZ4jC&7P*cWT zGXx+5n4yfEge$6G$kKOKy@aw{HX#7(MJN_9h6l&|JJNpY`Cg^{LxAu;R1;)xOzsq(r?a;&P7uu>&bfQpMnl~SPYP_a9<7%|6)0#xkwaT2-%NR+TJ!km<8Po2rm zVt8CUzQ=IE_5cNzy&R6<;wV()$}Ni0+s`1doZEvXq5B4_8-K&l(0l<3czXmHkG~|BBaUh87o)4HylIzXpC{Tk?|1(r)@jR}U`Xpp+hut2bjQIJ_+e58XdTdJGM#9_o2{ zxMZk(O32dYnhmOG)%Vdc8a#rn8K09TJ7o*DH9s|cn z4C}46W$-Zr6l1ZDf%&+~Wv6kaW<&P&S@AWPQV_Pw<`^K9uU>KH7Tc*;>=LefH!uxt z5PY@R)9nWsG_tcdkR?C+9Jw2B7?Y+csFmy$Bjz30hXkxSVBSbUYx5JnBT953!a#%p zjDc+IU&R(ub!sK^Mekk_o$)oY#oA^KDKQZxzcy(ec&P;Hz|2+QZLq9YAvqRrH z-X4E`?61cD`PkcIi=%%!`lZq4D3kr8?0=tCvkzy!mig|?%b8UA%jvh%eCqd8Ka#qd zdT8WVM?N+3(nu`%&yqKj^ND|-_<_Vn5>vy!JpAp$OGAG-^dyGD`uR-q@cjDF60;4R zoHvWDa$YSf^&5JD<)xyi7D+G-NF)ckDOrF%KO}Typ`o#l#8Ltm`vG{`yytn{{xru~ zqr;5zBj%S^LIL^1)|W$#{NI{iUJf<#4_RLhHS!MvB0t&dmhq>!qDKL&oR12 ziUKQ=(14&w79>U@RR*qu;8jKENKuh!l7p^Dqzw8X z3C2xMQ5cmL1npxlB*7MGAvnB!blJ(i)?@l)-<*?WT#x9FeP4C5ul1aM+4qS6Si-RH z!%p_KVCkEEB`5n@59yD6UvaW8enJ@beZ$YbbpPzz@Ut)73dg>AKl_F;B`^8ex8J@h zq@QK1NA$wfm_yLSJH09eAX??Zm}?f;A~`q`K6k9}YEvo8%ou7BxU`d|{eQpWe9 zA?3tU5?V+O6}UXB+|nCRv^0OKR4$j4twyy%$wiUX*dnQ^1qQYhRw3nLK_W#-V|ckp z$tq}xf8Blp6Y>e)W#v-rgG zh#%qwCLeMZ?^q9@!%pJk_5(uB;jaCFkW+YDN*+G3KD5Rf+ie39>Q$ZPL{`)(g%oM6 zKtcyZjVwS9QWC6c9K$Oz*!=m}_gY^x+u-!^#qY7cXiCUFzW8^{FRt`G;@`Hu*!PIv zX?}6J?-Bo&^~JtN{IvPSrM^e}lzEU@3UxsHerwde!Tp=rWGTkQzAzH|!?9n9eIfaU z)c=XiiGMONHvDtR$Fkp-__g77iW&L+vE|V} z8~v%#+UUvb<&lN#|1tCjLq9aMHgqKZ)ybcpEKkl%{F{kSO)L+yBX5rX>G)5K=f@w+ z{CxW7lgHEFlA;oce$I`~y~uLghuVwEP9?}X@P$-@6z6tt!DK4m{izKaq~TcVFLD)%BR8Fjnq@NYPf@!R1Il-flNJkwa67L$^>G&`))mNH z-$|cK+CQ5<)y|7CCls!AwPPqKr?|j44WqhNoZ`a5NJw#^j{Kw8nSMl`>$6U@V)2^R&mGlhJ))M)0lQoW(+O&!@)!Hs0`7I+LHl3N5@rcQxB zie_cxL;_-RDoYo2iDHV`13=5HB^oNM*L~`@I899A3d9i&9Bc+y}mRbLP$E(59o6j zUbuL5S!rydYU$UmHkum57+fRHRZ7j0QYIRD6J>Pfnzx(2&@)IjK(X4;9@HH(+CH{o zf_HUb;1#Q7O|R#nzg|miw(5FU_(TZ9#c_K${Hc9-*@SOpVDRB4e}Om?p*f}xXX8sI zOivFCCatPAg5jZ1h@OXr#PcRR%HZJ93oYn^2W$sD5v^*(x!QIhhyrwY_J$_6iRT6l z;>yJfE9;jpuU=ToufDdrvc7ii;)UFn77QZC3Vslu9W;o`b+ABe67(YR3W!Zuszk8_ z%_aju zOHW-11-4&VFz2w{+^kj%QAn)dtYiXzSX(6g0k}r~15Mj4$Zf>RPb0+6PPGq*g3uMF z>P2w4wn%>DLhc5~FdXJv8@ZVj8Los=`NSwM4j zY2Ma7cB-SfBKv&|U@a}ELwU*6ZHPt-N>$dk3AI`->uOWiay~eX(0zym#&Fswy(VQ1 zINR5D323)1yI!p}iO!272r5TeZ|E}oZoEHd{Dil2&%hfkFY-LPzDQBT)f-U0?S@`o zxJ=g7R{nbJTG0Ghnc?mE+3c>0*8EEMW`6AZ&hHAEsC#}jWK2VIe6yvu^n4v6rIdyq zJWGgB60p>LAC zOBc=tP71n}v$&zU&5I{kQ}W#y(40E*Zv`Dak#1of@&fb=lp2K6i+zeM%OKR7`bFQI zHE8Vupk^dqVlmXVw0@LExHYtkzJbQ-Lg%dR+gbv33x9$z0#D=_3FoKTq<@k^NHzE8 zzTq3d?u-D1JZ`-U}57LC&(AggF4C7#-z}vq@C{-&LY#cJAt| zsW#1CBhH*5W~2K5QT_j@{=c$SgH!hclfR(f0&|A-CzoS$bj~hsr9}1rqx%0~@~W4B zS>T7R=eE1`igO-K>0R`Eyl1s5ad(fX{a-hd`t`zH(6!HAo>Be(sQ$mB(VuN_kJ|s; zg_@JDgNBz;+&A}Ig-Uauu+pedwL|B|Vf3{yFId44yO218-$|nS|J}9@HxC|w>c@^s z*5RE&Mmbo$bGW!y{r^zaDCGY?8v9gi@{NgKnvlkSXgogl!ON&u%F?c+;yn z$cuTS+NxKSvd<97i8%pGQ9L%4*hZcYTs6T17w>Cig+Xx1%@-1fVF2b_pRHfk$350fab-K>>)5%7Iix`Tss< zoA|H^6a-2P3aG$H2?I-%C}Q7&jiuwSngBth#J~XBVW|+ zBMUAX?{YZ4HgFi1cNi>zVB~qhzak#Lf?zb<0%2(FKLPH{GvJRp00?=R6K&5uHe0ua z%UOWXbhm^7anK$Y18`9ieA007LkMs)Tp|qME)~fee6UCH5(s!+5_z9LX6_OK-t@2Vrjm=j4Ibj!pFJR;$n$bt)|yi@T)HpS7%?jczN~w z>}g{5+@&k)#xG|t!Zmc|%DFZ4W9j_r+N|GI&9ONs_Xnk7y{>Bcf(8F0yMf=-gF4M6 zuMsx8bp4eU$iP*L%%diP4q+A(wo7sGd!dh`!=tK&SwDkhWzLc z^)%+xBrS6?#Qb?8{-{?FUQNl-A8-Sg82~bP7yTOEiI)+`=CbP#>}5;Naj9hF)a8%`-AeH$TR?(LP~>oKYsX}8N0Y1Dg?}w6_#D1o2;OXU_}V0YE_zx z1QSqr8S0s{P#e~#+fTn_0(WuOz;!zi)wUtn;SGrL$~CK7{=*OoewJCNu;yJxhhH=y z8(eU4tz^VJ2gi6qC0)PBeB=caCOAJ1C|-OA$44VvT%;wuA=&HWfmIVaI6e*>I-fW+ zpkgHsi)suNI>tx-fB&`<`TrrjI$u+oo2aqM!Y0((iTwW^6P>0*p%o2ktpqxHH60<% z>K#&TY*gINLf3Wx>Qz~9~GRaP^QkLgJqhw?Spm{ z_YAV`n_H2U_U5(jyKn9{>%M>e?ZSZHr^}F;HSDj-b(fE55S}9ls;(gC)QqlfR*6|l z5r?>If862h(?qR7D6QMX+r%csxDsTA;1Qgd_BfTvM|q|f<@aj?aLlv^z~UeI|KBg9 z<9;sY$p0U_{IOefzX^VGTQZ?LP~ks@@dSFC?%aB^QLzn1=zF(v#-FF=h+?%)+#*U9 z!g%I2ueRVcG=Z{UKO}^P;5uh}8O42be=Uwfw6Z+e}8m6`<1Mi`S$c5 zq_3oYE%o7%Umv-e{1-`b=-&>_Po^fmE&l1)A5NZ({d4&5Aidg;6tYmo+u5hY)IFK7 zb-k)6A{tU8eHIs<&OzS-C_Ni^2b^xK*TtG_KcWOf7q(J!w{9x@B0vX?u%KCA5OmPG zi?O zKpN9L1DRqTbMr(ZcEr74ZoZs_Lix_lW9BZ`XW_iyUP1gK_?IB@nPss#5;v3@YM=aI z7D}<9r4U5U{w(<284bp#Wm=L%AEP12pdhU;Ro_WP7Fs%`&t<1fIAaDC_YyJ+jNATx zP6%d>g;oS!qxYeW>@?|YrWeKu+5=a}+l?^d#}#<$p_?7;jV8#nk6z6}0}4~1xwg-} z2aS^h$p}CTg5=ATPQU#T1eueuU6rc(E0fxly~UgeejU582WIco+ph4%4v6D_7Eu08YMAcK|+uV12w8J)i773+D9%11J$SY(_<+b`i~?B zusYlkAR`(ViNLHha`S>$BeR5ee~9WTwo$|n^ozvfV!@fbc|LgDK5@}wT!x4C>bZB5F~{IOzepxN=N>Gzsn->|406Ry#ZEJ3T&ixy;aRfh4rC)d-fzG2(kp; zd1K;VPcY;4vHxl8vC%^IOPSwD|4I62>h#D9$?r@2hXgs?82Z;kGC=^>Z`6!&55dKNf%t4&lgKnb(-P?#C`4~jl&1RYce ze+420A`2ymd{`E@5SFb9L}6eYbRi-HIR_DHI9$(0?$c$9KTwIEFp%GqMGDYUEN{oT z+0Z|A13_<HUUsqa;y+W>g+2U)r>03BBL$-7E*%0nV5!2O-toxn8|)%@sPp-MhK& zaUVEicF;;03o?6k6^hpEpeXIVo84f#!J$b)rAgVxuCM9VB%$Ke04F)Xo^Zz0AlM&= zO560C(@g0IJ>dp8vyP1PS7`A|a+FcY$pQ|qwqcc3Lh%0cYx04{Xa$k-S;J7z!qO%}~&{o-T5N0)6W>=vJ>5G*VU=jV$Wm?v}YC zs&s9tB~~OyH?_o7A}gq3B1-Xss;~Yd%+ZgU&_Tt--9pz*3wfWCi2#~_vN&;!zfJff z*G$j`SZBOz%|rwg=$Ur1Ed~l5=A-z3zmSUJ|M!c@&?zqtRES_y|8K`$DF6D4>iw!G~^W)<`8=D?} zBsRV@_Aka}Mn9GPv+T>6U&@?M|4RB|>WiuAkvqxXO}>`+i3BzIt&{1APfZ*e{)fYl z41GuZ>+$D?zA_{Z|MsBL0ix$d7`U$(Xg|1}ot+D+;@f|6O85G-MoI|&7RctHK=#<1*(r;8%-$2Vg04U( z$7Z#xZ*3R(f=;V6NAd+mA~~)okxD_)NQx8mBCT*zQB`|G$5NaW5F-iD@zB%*2kl|# z+J|e|=>?0I4cmXH^o<@Mp=1|cfWdtTL&+opYuA3pAZ6MQRkO1gCYR-Z?_q;H9D(xJ zrACP^G{Zo%6Eo5u!Ymmb+k4n}NVI`c=K*-*MOc8|^r2Myp{-!p=)H%H zgLO#xi;N?@$*knV+sNHA0?jF!aOvzY`8Hjp6XdT!T&=%?_T3x7!-vC6ui+`^C5{&v z7LJL#Z+M5d!dycUrq}QS!}B;;8xKExw-h}5p3k0x8t)ulGkEx+_UX+mZLa4a@f->V z?;~(HQ4G$zgU$x9bGp-aoDv%Km6^ProwCG_*;ghEi}nN%`fbxvFo2Mq=Fe()pqQP; z&8Z9_9Wp$;2T&-6o)=;3@QE8pU!Yh9S_imAVW$0v-f2qZ64L{Lj)BC(06~p{(FK{p z!Spy7J^ha6E!>WEp-R0~ES7aG4~$nUZ(lD}c9JyLb$cr?W!v(r5vQ2QIJSzavDbE$XhqPV_TcYo{lY2 zvR`&4-)aSqP4DU0kPZSF3!Zr=-fDWswziEu9b2SvBk#y9tyz`mcm zbmaf<#Vb+#f4K~+1-0`rBhSJ&_lf-fk^es|D@FDHyM>Oq%uBln0gdASH-mI8hSZwrr;4ON6Q)A#KKYl1h{|?(8WtCfcqq?Q%Z=8_v?uU}}~@ zZ!CJ_W}(rL`2kn&e+h@@U7wb)Y!<+41am#<0EHA zhLb;^{IlmZdM0?U#Gnv+RUWS}57 zE3uTo#ok>_p1|MF8^*LSPx-M|k~Fy<_luJ-1vS@}lg9ug)3Ud+6b7EZ>wLuWN|;A{ z-uZ~7}qgM*a=G*3c`0TBLPJQ^=wW$7f zMxwOmUrf>`f}Uk!xUPVwSk>fWLE<@*q6-C*-kS^Xtni{eJzsoL^sB?)U4Tfn^U1 z&RF)PN~2bS`aRe475!GTT2YItj*v(eDMg2px-3IVRFNk+O)63>C96Ek5YHs1XRYrS z!L?P<%G>!uwV^92pqCLfQss4(EqUy)oQhzzpk_z z4NVedj?py|V$((7KBh<(BuG_I8MsbEv2j*bE}$<>WMNv5eoSNXbQ! z)z~7bsRaf=U=x@ZvA*C`4;tyNtz9ORJc&mk1i&s zfF;p->X#wD{rTh+y&gxFrEqw;b0#?rXfcC`U1x}$VB)RSBqSKcUEY}f@wxrMg6wrYDumd5aXC2)V%PPUJ|!;uVsaLiYQQ75RJ9Ccx2o!mom9cTx{{nd z>3NR9C7a5v8?gE{9^$kqMkRGkVIe^XmVpErLL{wGx~N0H5M89^0$>RvJ{~`toCWdW znp59m%zEt% z+DDBigy7NVWF$3=Z}-WX^RRydYa+4q7 zE5JsiXt0;)n#^gU%IJzBra{mFFWZ8y4;DV&fjJE4PIpjkGpwrA!C$37LY1oUDyb*} zRHBE}0cel|`E~_GkU>2w=p4r%c{(`-e2?I^zrE94lQ$(a`mP4ny82eB0;vYp%s~LK zq73-bbP9BJnw3#^9smQjZ@Q>U6jPMOrQ{T>7xaMsmdUy2l90Ayelg7IxNZtTV`~6i ztI<%d>wOY@%oNA2N7xwLIsP9!mz;%N&h?mZ$UF>+1T0!}y9>8!i_fB?m-&bg%l4&3 zr2U#-?{nR94(%_@!rcJH8?BlOucp}qtsHSAKu6JngnJv|Jn8ns?Zob(TVB}-bjv1mD}m5S zvIs4(yj=M}d)TBZbfpWyl*@jqTK>9aaPeG0K5wSOgh3>g7?!sCa@k0Gs6AwY1#RTQ zfYq%pjTbn&H*jD%1e$tr_E$QlOP=tv^vO-OA& z(T*Vqr>5?NLLbt&y!>YSYc@MF+Yu~eOYK0J0&qgF9~PIZz^SS2$KJ`rPMBLtpXdKQ z+)fy25w%K^1zEc7#5G+tb~^0+FUf7Vld2I{ZK@6Qcn{;Hq#YU1SXQ zNr~XFOmE2vagpJm{hK7&HA8Hf#WhjSb*njL--hDAjVgR0JP$WQcVs9PK*34KciuSG z$N<=8U&_9r4$F`MWhtQC5_m=I?zuKvXX^;9R!hph0qn_?X0!}hFrzPbq@|!f)2Nsj z4UK+1O9vg}XRyFB+C@M>nLh>-T^m3wN$40%6ySCMBK_RWqU?}aJ3m^kwm$!39fVC0i)lcv=6A-hBuI(>MG6#W@j^Hk96SavLB4(2f zh}lHf_7P$3ZB+Yx%iGJZ0X4$ijo-q= zyR!kZ7G2v%j8H&~c_3tMaD4@vVx4$*qyB#|g7Y zz*&wcR$Fz;Aq=UxN?j+kYDM2kw%6Uey&7=s;4JO5uR#7kG7F6lt;ZgU{e#%>Zw)^? z`p#%N`l503mVBOgiryZBe)k4=6iS&#jLiSM0ECjRfln-hug z+e5!I_S-}2W6zEL=a>%v-HTVbJ)S*$!YVMlO!sN*3$93Mcu06;dBw(8T#+)-ezKj- zLZtTNOH1=XHEsJG-_GrG5`taCmQ4=ouLj-`cyNSy3^&edwkuH0Hq*{zA#e;q8Z*k- z?kQ}59SN+{PIo#sc1paA@rI3^z!1N8FSy~pkI{boP8I@n4faSQg)f|Gw;Nwrc3Tec z>cPEnM+ANmG48SH(a9I6%JIZoS%?HS*uo4E>OUm18+fKw5u(|GDyt*QEdYnX!>p01 zw-09_Qr|2Qj-wO$`n)G+_6@B#5FFn_v6w~*xNhb(N;U@8kgSlaUmd_ zvK_p5zS&kg;TRV*rad1Q97j-W-`aohxS3Xa*fs6|@C@BBAneFpB|6<|54px20G?Tn z2WN?U+*GR_cZ>@f>HzQztH_JzkybnA8g~$QmL+I3IRQ%8b)LlHun1I`Zu^ai4^5U`cRa(kkqBQ5Wx zjwCYBM#8fklAs<=I@H5{kCi>A&O?YL?Ts8T)kB5>kzm1sr#@0{k9MZM)GsK(TADc= z2m;BgwG1I1MD8Ml3I?%{y$2QTSeAO$ym;<;n0MiuP zr3OhS@Uv?Pl9`nQVgwHMgV_W!>*E9(y_I^?kOcOt}jy>)=z$ zj!NDcrPh3Q7LH`TI+AMx z@el7t(=%5Y$mWAcvYv?}p-5(!jm>Ny?=I%Nwa?a_DZEX204#O@Na;{Kc9HeEv(3(Q z{(CA2niuyO9i1$8u!y+R4DcOLtheufhqBITfC2wMG7F&m|M>8C#^C?x6=5L4!2QL* zKPnzho%o|n`vKPBqIQ=3bzDYLW8c^O-JxVMB_L|}1~miC-b_~ymaOk1g?tC%9!TrY zQuFFw0iW1sTQ^E%I|zywKB-zQJA)c%Y|Jc~Ng~VUj};?31f}2YsDH*vK;d8F4*L50 zHqS6W)OYKQC7(3|uIw}Z2;kuGtG>@h8cX*qVm{DcvV?QN!}&|VKIhThO$T7n|fZYYkfCu+lBm|0e%Zl&HLUb!zNG@jj5Zn*N8J` zpvCPyTVugKFS|~U*)<2dQd|ir;F<8OB-2;1pH>oZ)qz+^fs)Q90u##k1YTDUec(#8 z#_Tn=5_*lnQ74CvJC@oV#@F9TZU{$P-`b13{taOFFnwu*C1A(v-yxO3FoCkr_AYr= zoPB+tt-Hwo@0E=vr6T|TZhAaB%<$(U|9`WEB+?ApBp_WB?Y#PJU2Qe1b#%F0g#sn$ zB08m(P~lLxde3zC_n}ne|G%%)!B!QVjvV>_4TJ4|UQj-gMgD(eV?R(Ni~Rp+!@4>X z`Txy>1lIK30i==tzg)I;8BkmDC9R{m?2-1kzy1G4U2oI>|8i{j<@@XSj7E-^$yigt#0F|JmjnrJOfW2F9NJPO{=6AAXH+blM2z2Z_A-d!>CtE-v)W zMXLOaQRQ>is@B`_=I_K068ZI^WMkfz3A1xoXHB87@BI2ZqNdNw7DcJkMoZb>%zh~IPcvld>GV4zUrCQ8e{Qli zvOe+uCB8JFCSMu;^!UX1A0>{B-8tarv^Qj5guDoIm}z&o>m?5>Cm6u`tqk#H@N(e_k( zHhTiYbHFHB@f@&j!xNIsF<7tg<^P5MNX?3BEv9t zg)j_N`eyqC0?Jw1aAhe3P=Q5U%z-6QpnXMLvcNaTI()Od$9%)&p#L-DEkS9*BZ{I7 z1sBcoQkZWNwf5r(91{^MVZiBu0$>zL_LlG&skI;J47s$|LvoT}7uvLIceTgc4`rd# zw^269D7hMf);-oM>MO`YDLVgpg*&_`u()I!#{1YCvFCRq*lO)Z9XMn{%%%s~Bv{Ba zu6BfRINW}!{cx5xaTru!Hv|m5F)9NoJ;gdLPNqag#O0Cjl6DqTU4~++qwS+vX!Go- zrs~MYVmXl&bxI*cS}Ty8%xGjmmKhT0$1%Jji@L^l0nV(YE6N=0U1sJT2})$yrSdRv zlkLOp>Ffy?IA)Ueo&p60y6C{tzaq~;;aAVvpMY9g4|Zy4tvU19c8wrG<|#3-jF<>5 z*Pxi7TMmr_EK?3x`mH0@lUM)?F9z225@E~mHCh`hw?ELS>$SFrb-kdk8*7UKys$>0 zrrQJFnrsY@f5;z*KfF$x)a!mp85}>L**5R<_Fyr>*Fo~g9&QmSiM1#bLpfgw^ J5_5d={|BNZu;Ks! literal 0 HcmV?d00001 diff --git a/src/maestro/server/internals/orchestrator.py b/src/maestro/server/internals/orchestrator.py index 335c930..077ed8e 100644 --- a/src/maestro/server/internals/orchestrator.py +++ b/src/maestro/server/internals/orchestrator.py @@ -450,7 +450,7 @@ def _run_dag_concurrent( self._cancel_running_tasks(running_tasks) raise Exception(f"Task {task_id} failed: {e}") - # Build upstream status dict + # Build upstream status dict (simple string statuses) upstream_statuses = {tid: "completed" for tid in completed_tasks} upstream_statuses.update({tid: "failed" for tid in failed_tasks}) upstream_statuses.update({tid: "skipped" for tid in skipped_tasks}) @@ -460,7 +460,11 @@ def _run_dag_concurrent( # Loop sui task pending for task_id in list(pending_tasks): task = dag.tasks[task_id] - action = evaluate_dependencies(task, upstream_statuses) + + # Call evaluate_dependencies with StatusManager + ids so it can evaluate condition/output_of() + action = evaluate_dependencies( + task, upstream_statuses, sm, dag_id, execution_id + ) if action == "run": # Submit task for execution @@ -478,17 +482,21 @@ def _run_dag_concurrent( pending_tasks.remove(task_id) elif action == "wait": - continue # lascia in pending + # leave in pending; continue to next task + continue elif action == "skip": + # mark skipped in memory and persist task.status = TaskStatus.SKIPPED sm.set_task_status(dag_id, task_id, "skipped", execution_id) skipped_tasks.add(task_id) - pending_tasks.remove(task_id) + if task_id in pending_tasks: + pending_tasks.remove(task_id) if status_callback: status_callback() + self.logger.info(f"Task {task_id} skipped (policy/condition)") - # Evita busy-waiting + # Evita busy-waiting: aspetta un breve intervallo se ci sono task attivi if pending_tasks or running_tasks: threading.Event().wait(0.05) @@ -837,42 +845,98 @@ def stop_all_running_dags(self) -> int: return stopped_count -def evaluate_dependencies(task: BaseTask, upstream_statuses: Dict[str, str]) -> str: +def evaluate_dependencies( + task: BaseTask, + upstream_statuses: Dict[str, str], + sm, + dag_id: str, + execution_id: str, +) -> str: """ - Decide lo stato della task in base alle dipendenze. - - Parametri: - task: il task da valutare - upstream_statuses: dict {task_id: TaskStatus} dei task upstream + Decide lo stato della task in base a: + - dependency_policy (all/any) + - stato delle dipendenze + - condition dinamica basata sugli output delle task upstream Ritorna: - "run" -> tutte le condizioni soddisfatte, task pronta a partire + "run" -> task eseguibile "wait" -> task deve attendere "skip" -> task deve essere skippata """ policy = getattr(task, "dependency_policy", "none") dependencies = getattr(task, "dependencies", []) + condition = getattr(task, "condition", None) - if not dependencies or policy == "none": + # --------------------------------------------------------- + # 1) Se la task NON ha dipendenze e NON ha condition → RUN + # --------------------------------------------------------- + if not dependencies and not condition: return "run" - statuses = [upstream_statuses.get(dep) for dep in dependencies] + # --------------------------------------------------------- + # 2) Gestione dipendenze + # --------------------------------------------------------- + if dependencies and policy != "none": + statuses = [upstream_statuses.get(dep) for dep in dependencies] - if policy == "all": - if any(s is None or s in ("pending", "running") for s in statuses): + # Se qualche dipendenza non è ancora nota → WAIT + if any(s is None for s in statuses): return "wait" - if all(s == "completed" for s in statuses): - return "run" - # Se una dipendenza ha fallito, skip task - if any(s == "failed" for s in statuses): - return "skip" - elif policy == "any": - if any(s == "completed" for s in statuses): - return "run" - if all(s in ("pending", "running", None) for s in statuses): - return "wait" - if all(s in ("failed", "skipped") for s in statuses): + # ---- policy ALL ---- + if policy == "all": + # Se una è pending/running → WAIT + if any(s in ("pending", "running") for s in statuses): + return "wait" + + # Se tutte sono completed → OK, continua + if all(s == "completed" for s in statuses): + pass + else: + # altrimenti se una è failed → SKIP + if any(s == "failed" for s in statuses): + return "skip" + # se una è skipped → SKIP + if any(s == "skipped" for s in statuses): + return "skip" + + # ---- policy ANY ---- + elif policy == "any": + # Se almeno una completed → OK, continua + if any(s == "completed" for s in statuses): + pass + else: + # tutte pending/running/None → WAIT + if all(s in ("pending", "running") for s in statuses): + return "wait" + + # Se tutte failed o skipped → SKIP + if all(s in ("failed", "skipped") for s in statuses): + return "skip" + + # --------------------------------------------------------- + # 3) Gestione condition basata sugli output + # --------------------------------------------------------- + if condition: + try: + # Helper per leggere gli output delle task upstream + def output_of(tid): + return sm.get_task_output(dag_id, tid, execution_id) + + env = {"output_of": output_of} + + cond_result = eval(condition, {}, env) + + if not cond_result: + return "skip" + + except Exception as e: + print( + f"[evaluate_dependencies] Error evaluating condition for {task.task_id}: {e}" + ) return "skip" + # --------------------------------------------------------- + # 4) Se tutto è OK → RUN + # --------------------------------------------------------- return "run" From 5c0c7c712eabea128bae30e17786cdd534beaa3a Mon Sep 17 00:00:00 2001 From: "Davide Corigliano (Data Science - Bicocca)" Date: Sat, 6 Dec 2025 23:41:42 +0100 Subject: [PATCH 38/38] Fixed last example of conditional branching --- .../4.3.2.conditional_branching_matrix.yaml | 9 ++++----- maestro.db | Bin 204800 -> 245760 bytes 2 files changed, 4 insertions(+), 5 deletions(-) diff --git a/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml b/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml index b0a95a4..7aa428c 100644 --- a/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml +++ b/examples/2_New_examples/4.3.2.conditional_branching_matrix.yaml @@ -11,7 +11,6 @@ dag: name: "conditional_branching_matrix" tasks: - # --------------------------------------------------------- # 1) ROOT: condition/value evaluation (informational only). # --------------------------------------------------------- @@ -37,7 +36,7 @@ dag: nums = [1, 2, 3] print("Branch 1 - step 1, nums:", nums) dependencies: ["root"] - condition: "output_of('evaluate_condition')['branch'] == '1'" + condition: "output_of('root')['branch'] == '1'" - task_id: "b1_2" type: "BashTask" @@ -60,7 +59,7 @@ dag: params: command: "echo 'Branch 2 - step 1, listing current dir'; ls -1 | head -n 5" dependencies: ["root"] - condition: "output_of('evaluate_condition')['branch'] == '2'" + condition: "output_of('root')['branch'] == '2'" - task_id: "b2_2" type: "PythonTask" @@ -86,7 +85,7 @@ dag: words = ["alpha", "beta", "gamma"] print("Branch 3 - step 1, words:", words) dependencies: ["root"] - condition: "output_of('evaluate_condition')['branch'] == '3'" + condition: "output_of('root')['branch'] == '3'" - task_id: "b3_2" type: "BashTask" @@ -109,7 +108,7 @@ dag: params: command: "echo 'Branch 4 - step 1, checking date'; date" dependencies: ["root"] - condition: "output_of('evaluate_condition')['branch'] == '4'" + condition: "output_of('root')['branch'] == '4'" - task_id: "b4_2" type: "PythonTask" diff --git a/maestro.db b/maestro.db index b6d4ab442aa10e8bad7f0cba55736302dd3c25ae..878ed8c9ad633697b95c931ee4ca44133fccc8f3 100644 GIT binary patch delta 14541 zcmb_jdvp}_y`SH2cJ{q9fn*ax2m~Ug7&h*_XCXW`;Z1lc@s=Wn%>&3G5CTE3kFxx% zRSy;+F7m?%J)%OZ6)>W#SRSImX9ca=%h6M<=si`-ZNYPEYrXZ{-^}c6b~jru8Loe< zY5lJ>twEV^f}~S zL^~!Sf6(1HZd5c{&8RCRsaam3gu<~1ACg2-3`G<{3pI-^t<9~=9U7yY^V9@=ODKPrzc#Rp=y418n(y_%o4zGnW1zzSpX?u4 zhoAQKuoKyR?79H;eS31Ydz%8Bc?M|;~Mda7C6=U@8msIh7(VK>9=n_ zycvLsIK7&hLT#hCm$*B*d0YYeTXrYAjI9Wq3+xN5355M$0gwNP|2F?*@Qt7Gz2@5j zj`FVxbksXnIT+`aj(eET9B;>&2bhHntJ7utH;D<3YlO{0Q;ri|ipr@o6PY38 z>BVBUSA3t^9NzZSlALng6{YKix2zER(QCt%E>Gb1vz~7?(O7&YiH7Rb| zyVzt$4)3H$5u5Cg^0os}1#!#q^#7-hlWZOLZkm|ExNba^GO6D>UZ=b!F=;aF$mBh8 zhAWeo@293DZvv;@Po8}Od`Y`&nIYABK*1OH0LimYV}?{3Z{K+eUOkoCIf)q((%)N| z(Te^33FBYq3+gaF`6!r#zugZa(^CftwhnqOMJ4Vz2xQOWDI$5S6FH08kvf~Vb@sF7 z+4X`U0q_W3a{xppn@85$Bll_Cxd%*gn`G9TWKN~FrOvcR=0W@vITL?Oh)c<`&K_|G zOn&Q3S?)H8tTTy>Q(Mh&TW2R?9y1xNv&Z0mLKoit2$+Bu9R!iMNk`O9*gtC05w$0u zrwsYvZ3jV@MT8w;{5nO1Jz@Mx4-{IQdJs|%f;C_)b&fm2t>u2eeoWecAF>63-vsUm zME(EpAN05TD|{dM9v~%4zRa;9smtNCWX~J06(lFW3`LMEYk^;mk5g|^+;Q$EPGHZmyV&bkF7QU+)&TGS z!vBze5h=}&`quic@SgSl%sa*NwP(Ml%`?pXw)+mZCTQ%W=cpox1xP*tiwj0 z|J@WONTPH5+V*Z+q3?-gWH-DQ;4?jN2EJzV~y|C90O<`QB2#7v5t+g z?zX0u&b6)W-R+%gZSeX+8ouB-@Zbv)5#isCfw}YNF+AaWAcid=Jc+vwgP9I;=G6uZ za}NO}S$7;92gx^k!76}1dI8MDS04s-t^o0hQfe7?Jpu)=tOr#_(OBg5P@%?P@YZOx z9r&I5;k>d`123tP8sP;pOHd-JENYr43rXi5`0dEqfp)5#`iAm<<-f%beOr7(y$^cF zkmxD6kGRLXZg&+rA9u>kE6fDP0Y`**bP(=?!@<4OH(VEK--W;lit_!HYNOhEP#Zq@ zBAk;vT1p>)*l`-(m+b0;0AS_}ych2}4cF+OcjEP@$V<-|*qI!85~?8i$tn1>!$Aaa zT@ShjzgA3l;P@-^wxXhr{Bggq4UWYB__L)QBV;CNj^FJ&OWJ@{+RBq+_wI4B67TW?ctX{cxy7 zgea*RpKng1$h;(y;gewiiV}&)ilq67$TPYWRDd)u@%jwWT&mgmG6qPQQc)!-9FF88 zNR30A_Fqk%27?A3&}bSWt46qi)|H98&m69%TRs(}hAAAeYaHM262|j<_7e{m1JlW(2a;yJQ)LAd9(k_>{IJUR14^Y=uLb zz*lF$+hVdm`b;vRBv}(RDW7!`tCbZiCR<4vqH1)Nmk(hc!csf&&k%q$x7R~&-h3d1 zMM;q`S4x-7ijMg)V$nD+DWeU+yKUCe1~REFj2i<<6!QC|C`Uvi|C#Bo00Se+*nLr){P>co(j_UVRjhZwX*;^ zl8y`T?;NY+>(vyXXH)d-UNnUsdSrZ|D|FF^3ey6`(JD~nwgL}cQU&%WmwyH;ndD3l z{k(ILsP~>%INu->bJuC^C)_M9z`o1wV1LAp2z(IO9k?Md-1Rg6hyGpeg|0dNCH`W! z-{toGhi`{(uFvD`^WNswNDI2(v%*v6e&4;r`9?3g#Thq8Uh6wMyKOW&uSc13QvO;= z!bZ^>e>wArx_TQ&yp(xDv~F-G#PzEz>ZB0csXX|HRtc4@0ie8o*rC$b4#s3Em(+9>b zOUX@8m55b>q;bZgQ6^5@CO>$49~gzNy$6n3m7T>>sC<}Lq+IR_&`1-t(GDtpazCt& z=alRwoIo;n*ixF6xzT7;uEUiCH%6-mA5@G%%Ix_w7A#%3zI|=?;@F1O%PMB!>-Ru; zDM2Nd={&jn*ZY-h6;4DWyacr}h|281;=3M&tv^Zu@PZ^-x=XV~4%XQk5^S+`{>5Ly zDXH_r{GN>SMKL_U#!FGDetvnd$rk;8cnsEC^b2CY52fiZMZ+%V z{7sL;`jmphVZBk!Rq#+$V#;Tk9s29=#aBToR`$X%28vJvf+XyDbVYr@OZ_50$7;pNbi-KB(-UO8$7p9@ucbQKXWYm>jW;Hf>as z&`>V{FumFq0`cp%r3Qi$Q8a!)a4bPXbQTP^wgo^OZ)yra)I*%R?a1W7;x93$jMiuf;-I z&%;r`6cTYeJ1j&6+|vg};?KQw5uURbaw#yCq6qm-FcL{@f$@0s7hsq-S`n(a zw!3Xz1&^ohg{7tukWP+I5BJiJG79;0NW(0|gu!1sIKEbk$5x%ZT(-2E$ep$j?BIXjtenY$ccJKDeq`Xc>f zu%Gt8`{8J6OAnecjr6tGdCBck#z=F}H7W+H$ds3`N;dmQ43X9Rn|(al-m^xDmW>6- zL}7uKl;bA8y?rNhqTpIZro1K*64@fu@)k#uqG+m>vFRW3NrwcV^KuLC>%u!SdL?XUqvF3(6Yos+Q~U8o^L z(6(0Q^7xrVX4tugLwBLMSx|$QwwPr&nk=TPMN6k}9h#E{@ig;x*2N5vR2wWr$js47 zj1~wgL$fW#Rl99X#tggB*P->zYf(_ih&+$V$S9ahoXvr!U<^EGH zzK5;HWTQbO9YoGWm-UGvik1Scn9QLfG}BAg3^EdvydML3mm&hwU|R)za{~zIiY8EDS>gv=bp?)#-p~LFTXk8JjxwT3f>m}{PUS^IP@Rrt%G$cp8$4E8;s%hWxxJ@JSiqHW!~b(Ks{#A8dsqY(~P=4}eIxlS%)dN0G!U$@Ybw@kZ7A zQN+r!lySXhFBVuoI+A5Q)#bq?gKnZYeDMq@Ol*I%I$8KMT;y=*X&s*!2DTi24(>JI zCX+|tG^gG`%DmA)I+K($*^Bl&g=Z!QUE@BThBqcx=LJ%=ZFua$An&puN$&MHf?++0rF;Y!WVo2af6*T!M z2TwVKgofqyL0+Pg1SLylQ7hQg9xXwskW!Y@V~>`i{T{*E?~gg(o!`vd znfcA!h6diyuQx|Y^bR2;$s0Eh=pBFAJ|5-ZpV%7W;Z465W(t(`^s9Vex<7asAq6`z!okJfAo_d>RipE_n zeWtGbg^K50=*pp~c)y=E;*O)V!1qr5f}r^(9b?KPV%Q81*dcEnpibI~R}auMY(Hp= z(%;cEc^saP(}%zbqzQXEZKV2;EuK1TiwQ^Qi}k8lNm59ML~iKkNQc-Ve=2X$H_>#V zMZDrz>qu3u35%5d%5o)|uD1tdK#Wzl3m2FrCrM*^uJ&&&sIAmys*9yL>KAYwcEJ(| zr>FTlyo%pver9eqb6K)fYkX^T8q1AIjzMKmoUZ&@b_$`s5aQh!kL5kI5O4O-VwAdR z0p=g41(XE&E_j{_3q{u)?0+31FlHyD;@$ww1w!z|F}exMx@fLM#v9n1L*3MgUj}HQ zSR}?Ld$6P$7Kjy6H*tHdG@RB$b8%ge=40<}$i;VGhkQX^W3Q!^7;37Hb4zFjJ{h2S zg38HcZ_dxg)yWHO^`E?dtt)b|;RZ&GOR=Sr%4q zgG`L`*<}5x*gif{VvE0o#OOBry0TVq;}@;aD-?;YhsHa1`dLWAN!NsbV9k56v5nS<3H~boo8G zNcy9+R2-$lG?shCI&#Ep<)ZL6q1F_PU%^-Um*CN}v~S1?vchYvz$5)ogcAl~0yYgo zgls)vsj6!Rx&R_^pbh3=^)STZjUo6zu$``Yt;@kH<&aE+X{Ufv+&2g{!Jc6_Dg-x_ z!Ly-31Masp!2Rz-3l0uI4WaLG~W+%&M$RyTCv=OaR{ah_|yz6*a&QR*5 zVMP}Q_)5uRekXL96=Df%G)&{BFjZgcx1M0QIGQzMS0ro0TQRH^FGaJL@%vcj#s5rY zwb*wSR^#Y7XcqKel5DS)iH+0PD%qN4MX5H|ck8kFO~#|KY%0ES4zjTBL)a4ht{)CV xn6Zn5{r_kYoDs>k;`>poN#IpvG1i`i$FVeu)#0)jwhdpK%2ti