From 7753a28d4a8773e78b0741b4b4e069d6b74fc124 Mon Sep 17 00:00:00 2001 From: Raul Bardaji Date: Wed, 26 Aug 2026 22:06:31 +0200 Subject: [PATCH] chore: untrack generated files that were already gitignored .gitignore only stops untracked files from being added, so a file committed before its rule existed stays tracked and keeps reappearing as modified. Three were in that state. .coverage is the binary SQLite database written by `pytest --cov`. It had been tracked since v1.0.0 and is rewritten on every local test run, so anyone running the suite got a spurious modification in `git status` and could commit it by accident with `git commit -a`. The two docs/.ipynb_checkpoints/ files are Jupyter editor autosaves duplicating the tutorials beside them. None of the three are used by the API, the tests, the installer or CI. They are removed from the index only; existing clones keep their local copies and simply stop tracking them. Closes #253 --- .coverage | Bin 77824 -> 0 bytes CHANGELOG.md | 6 + .../s3_api_tutorial-checkpoint.ipynb | 1180 ----------------- ...dataset_workflow_tutorial-checkpoint.ipynb | 1030 -------------- 4 files changed, 6 insertions(+), 2210 deletions(-) delete mode 100644 .coverage delete mode 100644 docs/.ipynb_checkpoints/s3_api_tutorial-checkpoint.ipynb delete mode 100644 docs/.ipynb_checkpoints/s3_to_dataset_workflow_tutorial-checkpoint.ipynb diff --git a/.coverage b/.coverage deleted file mode 100644 index 26e045b8257c6e9ea130367bf3706e170e6f9307..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 77824 zcmeHQ349aRnIB1GNh4`yd}6=`3mdQv#$XHq1IC!k<}}8}##mU$(%2TZC8NU!V9bQ1 zZIZS$P1>Zpx!UbP(lqJjBsXc3l9DzwX__V*Lg*$0ax^3m0wjR;dru>2B+DimIo)k{ z{DbeCOaJeC|M$If=FQYpueN(QV~f+(Y4I3~i9|x7AQl;o1VK>nKL!2={xq;b26jLz zMB77cQbb|*4K!Ov7}Vb(*e1G)oo2X=o~(b_&`Djdze8i8tZ+j9lNpd1kQtB}`2WoS zUv1E&j~S!5y5D1I?%+HwiB6vTvqs}4@(Uk2qk4j3jkT2E*tW`*=2E9+c5j z11HnrwtMVOhmq^$tX>aii<-a!L|Whv$P=dqu(qHCZSnt7uq9$A^y4VM6fPHj+yc%y zoB>)xKSZYC$Jl3wx3d{2-|Fq`LYV`<@Ei6xhx+AEzx6nU#D3+xiISJ0^oC_;fO3?h z{$F_=_HAA+nE3_9&@T&cG91nVBT58Bz$Cpoy(C3JKvZ(Q?rj~=-!fag9w**6L*|={ zKp3t`(rVJPvJ~Ikh64rBV{W#4+`$bhENFs~g$u#c|J7(i&Vmod<_>3bc&y=wadS{Wpe_hp*5Y+oQBiTD zG+3S8oXgV66$)ql)x!UFAVKI9QR?(%Swn?RFa*p+!A%`MMy*ND$WUBO;V=?%OHi7y z5DMgElxj3^s*|(0y{&!*MhL+cmn)I|Z z1@A8rs9r+<6T%XKj)aOpL__e^-S4t^+KhvJM~A`j+yr&{{IsDQM=qF)kW}w!RAQIE zIR>-e36(zjuL^S+9SOO-itC%!ZRvp8q|0J=L6LbWp@>Fc20(?gsKyH57j^+B2({B<^A~rNDI&z4Hf|7{zcpHF@o*k{Cuc;L zU}tb(^;q0StBZq^(3a-ss)h>#lw0N^B+)2Yi@I`rmO_(0b*h5*3Wd>M&zRlNsCQZd zZVS80upo&z8H$%GsBodTBaRcqorn&!>$iX^H-8Q``&kHFCG>T4`80T`V80~bpZq5? zATuB{ATuB{ATuB{ATuB{ATuB{ATuB{AT#i<$AC(qRA|unU&;QBU{8UN|6~Sa24n_g z24n_g24n_g24n_g24n_g24n_g2L3r2Fep{y*uWzmGgyTxBVBk1U`}C4VUZ3}rDRVM z>`C_QKW74EuVn^g24n_g24n_g24n_g24n_g24n_g24n_g1_BHiRO1wZSpcO@l|c!U z010q`lKmWAmdt+1e$JQ}ntq(#N~as1H#iJQ`sehmdX4UB9jBwI2dD-rL;JF}S36O2 zP}8YNQvXW5IN_~?{)7zjwW3D=EcoifrOhqE)cJwtyt~2}|agHn?F? z2S~5QWp|ojEyGfmy_+*Fby~fhoWtXEo2*V7XXpcjo0y2$G-M!=0HNZlTb$LkJdEG&}q zaGhpYA?C7MQ2}qt0t@A_dmfCPNDD3Ajt&vNbv)p&8bN$mwq|WNTiY#AXX3;`BJAV9 zNIX?BEJw65 z89~lO5w&qFV9$;N+s3t6yd54fQ?RBS72h#{99Qv;L>9Tdel&pR$L@Bp!1|pQc3U|Q zI9#6&n8hQ7DRQ=Q6ac451{NVLNK+95u>*Bl>?l_2Q)5pcbagy#6Ov}3-E6nCv|Icw zfz#`E=(_bXUwcwHT#0P=Q0Mar4s7(mx01GI#iG&;1!W)pDhlVaxr_5Pyc zin6F85r9)-<3dDo$ipTUfa0o&{T<5?u?So6ADC@q06(7cDL@4v-6jzWt7rhvh>Hc0 zr-s-)b;62cmp^V{!MwH2?Bd)wT-|27+wJAdE%shmK;6~1K@Z^5WApS%-~hW$2e6Z3 zgGKQLFB)jEx0>8NmR5N23vlZ}?~Fk$b2SBsxsnk@O{03z zw<~gp{rWb&R+Fs0FX3pyu7pfoi;kflG&l{(`sa1eP@51i`l}-O35wV~75WT=#{UId zNxohJUNDcs#&W@<(Z>H%`$*n9@}|DR_( zM$TrEFO>iuPJj4bX#AgDPx3`0g)Z`XLcBp5IocdF{x_OQzFdOm!Ptqk5HbGGvXT6% zk;IQS{vRK2RXn(D9%}qQF5a>h$yb!|e`X`emy9H5qKL|9Ao)(|@Pmq5q41g?@p4 zhQ6ErF?}oDOs}IW>ACcDI+q?pGc;-VZ^P?`LxzV9KQ`QG_?DsHu+3mHR2!BU<{OF) zMuSp+TK|Fm&-#PqZ`OZXzenGvx9V5w3-meqM4eA}PWN}+Yq|$@x9axluGaPG zI&?POM%^0SVqJ+YSC>XzrcO~usaL3Hsb5gHQ#VmJPF7xrG8+ZI*{u768pXxes{gL}lmzf%m5A3fy>pOP&9-pS?mzTf!#D<=~ z^#A2j?YWw_5|aDC^5nilhfkcj@W~1IyWvF6s9rc$*lX*5de;Yke(=5f?;N;%wzYp@ zzqSXCC-)qB^fS5}wlceI-qQNNp4>i9=Y`!--XC`ThSE&)z_!WrkmbuW3deh2Dt5X* z|Mi1x&*zGdKT;ieL$m)uwyQPk*h>o!+}>}?nsXuD4VTYw-#6>{s|^>n9sc8?JI|EP zNjP>V4a*J{l+pFjq>qxjsV5AtCD7a8K;|~v@%z6xd#10C>Vn|_H3|!uuW{153JG%Vvxf@$beU%@U*k|Xaw!z^k zZTojoe|Y%V`)6(%AQCT@CaPQEDATG?w;F{G;!pZ^sWd-)+6z-jyfRTKlt&-%DEumlUkq|HljI&&<33 z@k=ijzkM$2t-oul;Ba!)_m3&mYhjC7%dl%e8`pptf-oty-8Uqyh658z0BZwG`)I z$`aVlTk=7U`QrIw;1I%0Ph1QKvlqX5{KKq;y$XW(Sgk8ZKPlg+AbeSAi(tE8(Y?fO zAB#EphIaep%QMwwaF{7mGYdhFUwAWw&(Xb)o>XL(!v4Hczk9^)tN}&Vlkl@|_Mh~5 zU(Zci0B0vI$cDAT#HVL&TRy2-ubz*9<}1|mke-)pC;>gTe{QI>}rMn6aj=C!E9p{^c)NFL^Y^Zotnpv=!IP0{EDTa;l z#d&*99X$5tBb}#`XTtvEnR$D9)So}n`QjEhrcWt?V|hiP3hTs$XZ9Foz`?OIaA-fJ z(N2fGGK^c)1U;N|L7-2eDzP1eA%U)hOV7<@~2VJd8noq97|Id-yAJq7ldDJwPkpcC_NRxo+6F+Q($7o1-_d#gg7i!R7rn>ZPC z_T+uXj~_hvTUFklzDS%k365-?w2vSz55UC)vE{oL&j08~ho8CV}P8}yXybuaop{qz%b@%gfad8Vy#T#Fgo%2;T46Hr$+MPE&@I^s>>!mYG4t=oq&&ts$=;tZ>b;QN88He9{DQo&O zf9SdH?KjR)qmtoFVR9{D^xuQ-zIVVm@W_B(odn03By~w5Xiwt3sl@5uoO^7i@21i_ zpZa9ieV<&sWIJ;0iBnY{FB|wt+21}-IlAjNXHuTN^!uV0&)qon?TR0s%AGTsMMl|L z;+Vj2@xhA?S*>R-j#4uSoKY6ipquDg9r4(chmT*X`uyTYw|x277w=aMxGueW_NFs8 zoxAb)C0k#a>lU=}`~NI9!1Zm08Pf?PPCOFCZi3K)S_J<|_zy+jpP#Ue{uDFy@S_&} z_j2)RbRHeW2)kh`5&pyP&`*PZyD1&~CRO*sv4ILL1)Is#yKg5^T69qR!l^rR3>w%R ztC>}-c=B1Jf>y&`rux=_Idcg0T%|ezcA13p6p5}UlZj6>Y85)9(y5h5D-{U}lA;&| zy8q7`dI)&;-)HRK+4tDD+1J^Z*+cAa*(ceDVIJTY>|HPu@O}0M_8PXI?Pi^98*5=V zu{G>!b{V^noda_O`D`|u!KSh_t7d#KU+^jO0rL)Xg!v2e5_6C_z&yr0!2FW=DRVpX zLuN1YErw^dGj67X;g}|-o>|ANU=}m;nOV#9h35^a=VH{RYf1 zyhuMsKSe)6-%sy@d4_+ZZ=%0VUrXxE*MT5J~X^*c+2pb;jrNa!?T9R4G$XbHT=wQhv62(_YBt=b^`_SpUi;FfXsl* zfXsl*fXsl*fXsl*z}Lfo7G`Rd1W)k(jorjVb#!p z)y9oj)z@QHSBKSx4OrFIVpUUv)%x{VRaaxRZXH%tRamWEi`5!%a>!dY#CNdmtwVK308|2V^v;`)uKgMm6c(&a3NNurC2RkfYto@ zSk0SpN>^wAy(6-VO3Co z)zqn2O__pKem+)td06GAR}ZKWCmmgWCmmgWCmmgWCmmg zWCmmgWCmmgWCp%Q2GIKdYv@+={Qu|h27r&)6YPJpZ^0UXKeI2fzh|FgpJ5+|bpXF+ zf64xw{V{tRdkd@u_zwFmb`QIY?PEQ#9$+iHCtwTP$kwsduqL2_UCfrUbJ=3H5Y`1u zWV6_@Y#N)$>RC0bU@kJ}nA6PPnUk)-aXK5@sP&0&51Q!y5@EF-9hX83pSG z42*_R(wFG-^ch$?@IL)6{WiRx;4k!_VEw@F=x6CC=||}Y=wHDag8!iJpnpW)Oz)+y zhjj!4bRX@eJAnxKPi8=7KxRN@KxRN@KxRN@KxRN@KxQC522?x=FI!jfDlC;)qSvph zcml}*Bz7UukHk(Sb|A4Gi9RHHk?2998woEG9wgjIxRBU}L>CfHBpgU|BGG|FI}%%w zup`liL@N?4NN`Blkgy`rjD!V=El8M=XhLE$5+)=Xk=TSp0}>mNs7Im>i4929B2k0H zdL*imScgOv5^IrIgT!hiRw1zxi4{mxB2j_FawL`^u@s3VNGwL89En9plp(PYiBcpM zATb|_c}SEXF&BwBNL+=)Y$RqOQH;b)B#Mxjfy8tq3XzzGL;(_0k(h!+J`#CIQm7|Ty$USY0grqgfH zchVaS7Y#o#EYZKCzgnNGdxdy0VFAJE`gN13KPerG3CgjG-z$2QPf*{WrfJ{M^4c7D zpWaqYA~{xlk9uvwhY8<@;G)TQRnL<h0tl9;e%6g`hWeIlD|K;lm+^ zKG|CdS(U4d{j;qCdiV!JTR9InOzd%6NPdZQ(5PgDK0u00T-{3Y)e;J9RJS{Tdb88j zYH`@NTRe8B!yWv_>1v5zwT@~>^nLJE@l_2j&fNuWa^~;QHtNAhb%7`v5oa4t)SS*V`DO5w;>0itM`Ch_yn!F;vZ zU7XbujOv0mlCP1@2eho5hYM6y!k#!ggN7!@S9e7<6e2u3zQP%fE~0ya#O6z&+s3t6 zyd47DZj--K935ZLjKCB*3$KpjBl|NkG!G)AQSsGD?mlS1p!36dTz0EF*woII@Y!}z zyB+v;ePnw(rSN5UMVl!MDlZ_dnzth0Rw?f$LuZ8&W;SU zzc54^m}MsUa_I&l>PU02*$O0}xQXP;MiM`~uRl{_jVuI7sLL_P@gj-EISO*P%QGYv zX8~ALoTrzn2Bc7Hu{)qKOOx0`wmKax_EwX-$I{vg-B~y1@z@=$Za~znhn}VchXllS zuFu-xw6u#MDT(c301^){L@sLMbJ5o8?y^`px7bC^T9VI_o@#5m#UX}MOAHSo#iH;? zC?|Oh6??~D1fip^Bl3wp1$`@%Ko=Q~8ul0_>ksR9>N2QDG^eQ=?HAggYS-vz=>AOf z5l0k@R5O$Z6n81DN=n_B;7hnGVKI4xgm>?%eyDS^cd$jwbMQ`mn(0tKa)k*qiM>&w zEpeNKE(8$k>b66MnS`!DsB1CNO7bftWLpT);;}$vg-k@$g-{?SNvvGMZ8*IyD;F{& z%AU0ndwIj+3j@AT)d&fxmRQY(3<*7^-~qeC#`Oa7SQE+5k1HD@kwyHBagw|xE^w5` z5&Vn!;XH10bDtT8^_{2!Tqki95@aPrdMLjwHk*G4g8cw1Ra;5EKAu#GDq5j**c~31 z)7^ze5MFS`jOS9o)VNNP-#oIWptJ0^VRy7R%`GlxXVBWb4w7&D`mMoSAta5AH?Ib> zh>~XYx?C_-gOdnWN+b@Cf)O#;iYUHS5<6ERbfJ9|!a$hk3k<5uB@Wn!f=7`Yz+NP= z&pZ^ih?O#lqnseLyEwe`xzPm4@~(J`ow%PLGB9Xr3T#d7BYAH;1tGdAp$!Z>X>ka( z3shxbd5Xkr5eCLW0110tUXN(<#3pf=I3%=x^0ZlEZVw5Ib9;-#-v5xOfw*sy*s~cD zcyO;{vyJ3eNr;~4CM+Inn|ZJiQ+zWs;ejG5yp8d8nFd#wVJ+IGxceKC-9C}?Ijcy1 zj0EO`HIE2ODRHDCh_KV;#7*AD_)1_Hq{!2HiIfV?xjhz-SD24?oBT5?HqK+Q zcc2QXF1`vW92_;!TfJR<&|d@S4e>EN2pS-3<0}>6kRpm}V7SDJ0t-4y4btzvgx}H; z_hdKLOvj5o0b_Y^4{n)+j0vVej2G&Vh`7r|&;J(?@cjQ%>^3%q`8hmc zPt%XnTj_Me^9F|@N&lR_Rj<)Kt>bhw^#IjCWoTd4_G%|;4r)3zN$Ovz7bm=x(4UY& z{ua91TGh`~<;vHU-AYPvn@g;0aIxQM$S_NjnZA3NR^n2FX^ z>zx9oyd!QZN_d$gx@ztDV7c)hVHskrJ@0EV7Co@HN3pKG@5CTrf zo$$S|(u7X|Z_Sgy(2Cd%1tW3LMAqQ0m}Mf^s2E8bLi$3JWsNytV*W^)5M_ob8^B9r z2M?z|d@tmB-2^}{8X@;P8h4R=v4A!iZwEXQ zxK(NP zpMm6T$^aWBBgmO3qBf2N?AdW(2W}w;GdwWK+BgP~<0`(xOtP*Y4dD5)yB#dBQSPPJ zrvqm3NMVXNt{erxX_A3ONDJT@%z@Z}I{i1_>QiG+pa&){-6kZ>Lc1vpmV%Si5TFey z06Zslzk)G?9vckpbU{VRnf)^@b;$rNp(YKXiJ{SCpdYtBDRwT<^8?XwMOjpl2*4?^ zaUmi(fyVz+h`R|` z-M@(W99G^hfi?5j&^hpxJhvfJ|2zE_-D%zTbve{Q%0;DXpU|2#7c_Tis?_JyKUP;J ze30Imqf-fR<2`3Ltm$SIwPD1z|9;S&V9l`z9 zYA_>VXdWzIBFx~iL35m7$gvI#Sz=F55cEohA|4K~Re`-VBWw@#BSL;as+!hIfGeTnAw1~^z2nHIH8-vRLH9n3rby`nkgn)>ODhi0IrGT3gySG9>3|{J1zeF0Z z!{!=dz_J*CCCn)f2P~c-1rwXg!9-c?WCTg}H`Br0h=*4;Eduy>W@;kfQE=cx;Ce$D zU`yx_1bvOb7KKR7LV%tcJ1atn1S$-_0eler-Bc=#i{WxJz(wZ*z&6M3aFB~`ONSjQ ze^`_iYH);OoVg2DwV(%NTkKqi4c!lJnh)mUY2Tum6Y<|M5AZj~k({qKmOyJ2AvLxV zuoh3BAiA}PhtO*0f)xpsWT4uKzU*Ve96*h6wEh*Aebih9u+j^Fzc&#m27v<5FdIN6 z6j*@)am^J8ZJq_tOJWb0DA4}fE|CVBiot+{h!+e*+%FouLSyqxz?V=D4~ZYX++kx8 zz)R?|2=Jk9;h+^7^)mojLZ>1Ma=6Qt(*an*%!B|es!eQ#fFz-O4_EOdO5 zcnB*Z|Izrrkhqp$?}qpHQOw;;HT@BN9X-Qv#PAKnZ2eLFxAjH3H+4I8#ngM$jZ~@j zr1nSJrJ6IE>ooJ!AE<9pm&02ByAoECpOJTvm8y?b|E^l1JgNM#vQqJ-VxM9;aT>s{ z^tY8q^E)Fo<``nQ5qR9Cnn&|Hag4~r1_LnIhx{?~XjVr;9Wg`(;bBv#V1R{nJeuo? zai}4h>W(thSjiIr8RJxx7*c%epnd?bVl1kIvHYEyu!Jw$1;A`iIp4d@E{hC1Y9r31++V0p@U_|nsM2G9&{PmWuH6QZ6C^X~4@}WQ!!=!i z7vs$mQGN?mUGOp3HYdPsh||4L%?Z6g^fnTgf7pc{Eb}*uKtz=TEX6oSA#!|h2-I`} zV2mAl5nz-j;lRr^9dXeccv1p4umD%n9+!{8?RaQKQTIBqPHX*E0FCj%2vNcXP?-4= zVb<73!qJH0w5kn2W4uF*YCKpQ*X}k2$|%|cDl1w6EG4!`4S|XJo|{_$NW!=zII)1T zBedi1j^HAIJ7u*Tz)NqEgxMwmFZeEm1{GGayPB!4K&yfCFrm1z=;WK?XYtaUbxrb_)PX$f_az#7lhEn*lY(;yKvO z@E%Eh6JW;J&mA`VF+6Z;*$lwa=VJ%Q)!=&CY624y9tVuh@(|00>KegFjJ@F@2@+Kn zO`8B$!X!zsei%HF7Pf3?RRiEls3L_a0sqpOKs;A(1h5#pFCr4c*5L@FferNlDZMC% zL5gVAG2bPo!VsM)qF<)Hbdup-!)pBp`Wy8nT3UBhw@b%rp40SD`>95X(LS&3&}6G$ zQunE+B>W}enuM|Bi{uV6PxT7Cv9d_{rt(_lRK@Fx-HJ)XQQ}(mL3YcP2L8f%2}|)r z&8QIiLWfHHsNBGEk{2(k`6>%rM8JWu<1Ps+lteJYjUD?XtceMO9%AgcQ^Lxp;c%gX z+9BZ$0YgAWj2*X2Sl}r_9%7W&8!u|q@mTjtl22E}*7*_nDjtt{<8yF`@t7w*)IeV^ zq66uUFP@@}$6OL#{32q~U))9Gv27CG(lQ7yv@Eg9PVyV1XLMw3G|YI+DdE+fBF6`} z^9~8Gz!CvQACGm$M{neKtRp@jqmRehCA{=al<=tIv8^NGXhdDBIC}fB(29Q9`yv011sfctuNC zG<~IKor4cj8ze0Bz7jb2tap&SSwbiUJK53hy$^k>e?h#j0*Gc#6wvb}blgxe3nnL8 zK>I@_MMB4IXy{M~CQB$JL*t6LNQ!qYapc;mL95w}@#bcvg^1N`SrXrR5zNg<{AjD$ z#>e~8u;@l0i$Y7xGUI*gPNWS{HfO}03BxaBVI|aRVeu*CMrhtn ๐Ÿš€ **Run Online Options:**\n", - "> - **Google Colab**: Dependencies installed automatically in the first cell\n", - "> - **Binder**: Pre-configured environment, ready to run immediately\n", - "> - **Local**: Requires `pip install requests jupyter`\n", - "\n", - "This notebook demonstrates how to use the NDP EP API to manage S3-compatible storage using MinIO. You will learn how to:\n", - "\n", - "1. **Authenticate** with the API using authentication tokens\n", - "2. **Create and manage buckets** for organizing your files\n", - "3. **Upload and download objects** (files) to/from buckets\n", - "4. **List and manage objects** within buckets\n", - "5. **Generate presigned URLs** for secure file sharing\n", - "6. **Handle errors** and best practices for S3 operations\n", - "\n", - "## Prerequisites\n", - "\n", - "- Python 3.7+\n", - "- `requests` library\n", - "- Access to a NDP EP API instance with S3/MinIO configured\n", - "- Valid authentication credentials\n", - "\n", - "## API Overview\n", - "\n", - "The NDP EP API provides S3-compatible storage endpoints using MinIO backend. These endpoints allow you to:\n", - "- **Bucket Management**: `/s3/buckets` - Create, list, and manage storage buckets\n", - "- **Object Operations**: `/s3/objects` - Upload, download, and manage files within buckets\n", - "- **Presigned URLs**: Generate secure, temporary URLs for file access\n", - "- **Metadata Management**: View and manage object metadata" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Install required packages\n", - "!pip install requests -q" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 1. Setup and Configuration\n", - "\n", - "First, let's import the necessary libraries and configure our API connection parameters." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import requests\n", - "import json\n", - "from typing import Dict, Any, Optional\n", - "import time\n", - "import io\n", - "import os\n", - "from datetime import datetime\n", - "\n", - "# Pretty printing for JSON responses\n", - "from pprint import pprint" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Configuration Variables\n", - "\n", - "**Important:** Replace these values with your actual API endpoint and credentials." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# API Configuration\n", - "API_BASE_URL = \"http://localhost:8000\" # Replace with your API URL\n", - "\n", - "# Authentication Token\n", - "# You can obtain this token from your authentication provider\n", - "AUTH_TOKEN = \"your_auth_token_here\" # Replace with your actual token\n", - "\n", - "# Request headers with authentication\n", - "HEADERS = {\n", - " \"Authorization\": f\"Bearer {AUTH_TOKEN}\",\n", - " \"Accept\": \"application/json\"\n", - "}\n", - "\n", - "print(f\"API Base URL: {API_BASE_URL}\")\n", - "print(f\"Token configured: {'โœ“' if AUTH_TOKEN != 'your_auth_token_here' else 'โœ— Please set your token'}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Helper Functions\n", - "\n", - "Let's create some utility functions to make our API interactions cleaner and more robust." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "def make_api_request(method: str, endpoint: str, data: Optional[Dict] = None, \n", - " params: Optional[Dict] = None, files: Optional[Dict] = None,\n", - " custom_headers: Optional[Dict] = None) -> Dict[str, Any]:\n", - " \"\"\"\n", - " Make an API request with proper error handling.\n", - " \n", - " Args:\n", - " method: HTTP method (GET, POST, PUT, PATCH, DELETE)\n", - " endpoint: API endpoint (e.g., '/s3/buckets', '/s3/objects/bucket-name')\n", - " data: Request payload for POST/PUT/PATCH requests\n", - " params: Query parameters\n", - " files: File uploads for multipart requests\n", - " custom_headers: Additional headers to merge with default headers\n", - " \n", - " Returns:\n", - " Dictionary containing the response data\n", - " \"\"\"\n", - " url = f\"{API_BASE_URL}{endpoint}\"\n", - " \n", - " # Prepare headers\n", - " headers = HEADERS.copy()\n", - " if custom_headers:\n", - " headers.update(custom_headers)\n", - " \n", - " # Remove Content-Type for file uploads to let requests set it\n", - " if files and \"Content-Type\" in headers:\n", - " del headers[\"Content-Type\"]\n", - " \n", - " try:\n", - " response = requests.request(\n", - " method=method,\n", - " url=url,\n", - " headers=headers,\n", - " json=data if not files else None,\n", - " data=data if files else None,\n", - " files=files,\n", - " params=params,\n", - " stream=(method == \"GET\" and \"download\" in endpoint.lower())\n", - " )\n", - " \n", - " print(f\"๐Ÿ”— {method} {url}\")\n", - " print(f\"๐Ÿ“Š Status: {response.status_code}\")\n", - " \n", - " if response.status_code in [200, 201, 204]:\n", - " # Handle streaming responses (file downloads)\n", - " if response.headers.get('content-type', '').startswith('application/octet-stream') or \\\n", - " 'attachment' in response.headers.get('content-disposition', ''):\n", - " print(\"โœ… Success! (File download)\")\n", - " return {\n", - " \"success\": True,\n", - " \"content\": response.content,\n", - " \"headers\": dict(response.headers),\n", - " \"status_code\": response.status_code\n", - " }\n", - " \n", - " # Handle JSON responses\n", - " try:\n", - " result = response.json()\n", - " print(\"โœ… Success!\")\n", - " return result\n", - " except ValueError:\n", - " # Handle non-JSON success responses\n", - " print(\"โœ… Success! (Non-JSON response)\")\n", - " return {\"success\": True, \"status_code\": response.status_code}\n", - " else:\n", - " print(f\"โŒ Error: {response.status_code}\")\n", - " try:\n", - " error_detail = response.json()\n", - " print(f\"Error details: {json.dumps(error_detail, indent=2)}\")\n", - " except:\n", - " print(f\"Error text: {response.text}\")\n", - " return {\"error\": True, \"status_code\": response.status_code, \"detail\": response.text}\n", - " \n", - " except requests.exceptions.RequestException as e:\n", - " print(f\"โŒ Request failed: {e}\")\n", - " return {\"error\": True, \"exception\": str(e)}\n", - "\n", - "def print_response(response: Dict[str, Any], title: str = \"Response\"):\n", - " \"\"\"\n", - " Pretty print API responses.\n", - " \"\"\"\n", - " print(f\"\\n๐Ÿ“‹ {title}:\")\n", - " print(\"โ”€\" * 50)\n", - " if \"content\" in response: # File download response\n", - " print(f\"File size: {len(response['content'])} bytes\")\n", - " print(f\"Headers: {response['headers']}\")\n", - " else:\n", - " pprint(response)\n", - " print(\"โ”€\" * 50)\n", - "\n", - "def create_sample_file(filename: str, content: str = None) -> str:\n", - " \"\"\"\n", - " Create a sample file for upload testing.\n", - " \"\"\"\n", - " if content is None:\n", - " content = f\"Sample file content created on {datetime.now()}\\n\" + \\\n", - " \"This is a test file for the S3 API tutorial.\\n\" + \\\n", - " \"Feel free to replace this with your own content.\"\n", - " \n", - " with open(filename, 'w') as f:\n", - " f.write(content)\n", - " \n", - " print(f\"๐Ÿ“„ Created sample file: {filename}\")\n", - " return filename" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 2. Authentication Test\n", - "\n", - "Let's verify that our authentication is working by checking the API status." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Test API connectivity and authentication\n", - "status_response = make_api_request(\"GET\", \"/status\")\n", - "print_response(status_response, \"API Status\")\n", - "\n", - "if \"error\" not in status_response:\n", - " print(\"๐ŸŽ‰ API is accessible and responding correctly!\")\n", - " # Check if S3 is configured\n", - " if status_response.get(\"ckan_local_enabled\") or status_response.get(\"ckan_is_active_local\"):\n", - " print(\"๐Ÿ“ฆ S3/MinIO service appears to be configured\")\n", - " else:\n", - " print(\"โš ๏ธ S3/MinIO service configuration unclear from status\")\n", - "else:\n", - " print(\"โš ๏ธ API connectivity issues. Please check your configuration.\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 3. Bucket Management\n", - "\n", - "Let's start by managing S3 buckets. Buckets are containers for your files and provide a way to organize your data." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### List Existing Buckets\n", - "\n", - "First, let's see what buckets already exist." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# List existing buckets\n", - "buckets_response = make_api_request(\"GET\", \"/s3/buckets\")\n", - "print_response(buckets_response, \"Existing Buckets\")\n", - "\n", - "if \"error\" not in buckets_response and \"buckets\" in buckets_response:\n", - " buckets = buckets_response[\"buckets\"]\n", - " print(f\"\\n๐Ÿ“ˆ Found {len(buckets)} buckets\")\n", - " if buckets:\n", - " print(\"Buckets:\")\n", - " for i, bucket in enumerate(buckets[:10], 1): # Show first 10\n", - " creation_date = bucket.get(\"creation_date\", \"Unknown\")\n", - " print(f\" {i}. {bucket['name']} (created: {creation_date})\")\n", - " if len(buckets) > 10:\n", - " print(f\" ... and {len(buckets) - 10} more\")\n", - "else:\n", - " print(\"โš ๏ธ Unable to retrieve buckets list or S3 service not configured\")\n", - " buckets = []" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Create a New Bucket\n", - "\n", - "Now let's create a new bucket for our tutorial. We'll use a timestamped name to avoid conflicts." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Generate a unique bucket name\n", - "timestamp = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n", - "bucket_name = f\"tutorial-bucket-{timestamp}\"\n", - "\n", - "# Bucket creation data\n", - "bucket_data = {\n", - " \"name\": bucket_name,\n", - " \"region\": \"us-east-1\" # Optional region\n", - "}\n", - "\n", - "print(f\"๐Ÿชฃ Creating bucket with data:\")\n", - "print_response(bucket_data, \"Bucket Creation Data\")\n", - "\n", - "# Create the bucket\n", - "bucket_response = make_api_request(\"POST\", \"/s3/buckets\", data=bucket_data)\n", - "print_response(bucket_response, \"Bucket Creation Response\")\n", - "\n", - "if \"error\" not in bucket_response:\n", - " print(f\"\\nโœ… Bucket created successfully!\")\n", - " print(f\"๐Ÿ†” Bucket Name: {bucket_name}\")\n", - " print(f\"๐ŸŒ Region: {bucket_data.get('region', 'default')}\")\n", - " \n", - " # Store bucket name for later use\n", - " tutorial_bucket = bucket_name\n", - "else:\n", - " print(\"โŒ Failed to create bucket\")\n", - " tutorial_bucket = None" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Get Bucket Information\n", - "\n", - "Let's retrieve detailed information about our newly created bucket." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if tutorial_bucket:\n", - " # Get bucket information\n", - " bucket_info_response = make_api_request(\"GET\", f\"/s3/buckets/{tutorial_bucket}\")\n", - " print_response(bucket_info_response, f\"Bucket Info: {tutorial_bucket}\")\n", - " \n", - " if \"error\" not in bucket_info_response:\n", - " print(\"\\n๐Ÿ“Š Bucket details retrieved successfully!\")\n", - " print(f\"๐Ÿ“… Creation Date: {bucket_info_response.get('creation_date', 'N/A')}\")\n", - " print(f\"๐Ÿ“› Name: {bucket_info_response.get('name', 'N/A')}\")\n", - " else:\n", - " print(\"โŒ Failed to get bucket information\")\n", - "else:\n", - " print(\"โš ๏ธ No bucket available for information retrieval\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 4. Object Management\n", - "\n", - "Now let's work with objects (files) within our bucket. We'll demonstrate uploading, listing, downloading, and managing files." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Upload Objects to Bucket\n", - "\n", - "Let's create some sample files and upload them to our bucket." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if tutorial_bucket:\n", - " print(\"๐Ÿ“ Creating sample files for upload...\")\n", - " \n", - " # Create sample files\n", - " sample_files = [\n", - " {\n", - " \"filename\": \"sample_data.txt\",\n", - " \"content\": \"\"\"Sample Data File\n", - "==================\n", - "This is a sample text file for the S3 API tutorial.\n", - "Created on: {}\n", - "File type: Plain text\n", - "Purpose: Demonstration of S3 object upload functionality\n", - "\n", - "Data Contents:\n", - "- Line 1: Temperature readings\n", - "- Line 2: Humidity measurements \n", - "- Line 3: Pressure data\n", - "\"\"\".format(datetime.now().isoformat())\n", - " },\n", - " {\n", - " \"filename\": \"config.json\",\n", - " \"content\": json.dumps({\n", - " \"tutorial\": {\n", - " \"name\": \"S3 API Tutorial\",\n", - " \"version\": \"1.0\",\n", - " \"created\": datetime.now().isoformat(),\n", - " \"bucket\": tutorial_bucket,\n", - " \"settings\": {\n", - " \"upload_enabled\": True,\n", - " \"download_enabled\": True,\n", - " \"public_access\": False\n", - " }\n", - " }\n", - " }, indent=2)\n", - " },\n", - " {\n", - " \"filename\": \"readme.md\",\n", - " \"content\": \"\"\"# S3 Tutorial Files\n", - "\n", - "This directory contains sample files uploaded during the S3 API tutorial.\n", - "\n", - "## Files:\n", - "- `sample_data.txt` - Sample data file with measurements\n", - "- `config.json` - Configuration file for the tutorial\n", - "- `readme.md` - This documentation file\n", - "\n", - "## Usage:\n", - "These files demonstrate various S3 operations including:\n", - "- File upload with different content types\n", - "- Metadata management\n", - "- File listing and organization\n", - "- Download operations\n", - "\n", - "Created: {}\n", - "\"\"\".format(datetime.now().isoformat())\n", - " }\n", - " ]\n", - " \n", - " # Create and upload files\n", - " uploaded_files = []\n", - " \n", - " for file_info in sample_files:\n", - " filename = file_info[\"filename\"]\n", - " content = file_info[\"content\"]\n", - " \n", - " # Create file\n", - " create_sample_file(filename, content)\n", - " \n", - " # Upload file\n", - " print(f\"\\n๐Ÿ“ค Uploading {filename}...\")\n", - " \n", - " with open(filename, 'rb') as f:\n", - " files = {\"file\": (filename, f, \"text/plain\")}\n", - " form_data = {\"object_key\": filename}\n", - " \n", - " upload_response = make_api_request(\n", - " \"POST\", \n", - " f\"/s3/objects/{tutorial_bucket}\",\n", - " data=form_data,\n", - " files=files\n", - " )\n", - " \n", - " print_response(upload_response, f\"Upload Response: {filename}\")\n", - " \n", - " if \"error\" not in upload_response:\n", - " uploaded_files.append(filename)\n", - " print(f\"โœ… {filename} uploaded successfully!\")\n", - " print(f\"๐Ÿ“ฆ Bucket: {upload_response.get('bucket', 'N/A')}\")\n", - " print(f\"๐Ÿ”‘ Key: {upload_response.get('key', 'N/A')}\")\n", - " print(f\"๐Ÿ“ Size: {upload_response.get('size', 'N/A')} bytes\")\n", - " else:\n", - " print(f\"โŒ Failed to upload {filename}\")\n", - " \n", - " # Clean up local file\n", - " try:\n", - " os.remove(filename)\n", - " except:\n", - " pass\n", - " \n", - " print(f\"\\n๐Ÿ“Š Upload Summary:\")\n", - " print(f\"โœ… Successfully uploaded: {len(uploaded_files)} files\")\n", - " print(f\"๐Ÿ“ Files: {', '.join(uploaded_files)}\")\n", - " \n", - "else:\n", - " print(\"โš ๏ธ No bucket available for file upload\")\n", - " uploaded_files = []" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### List Objects in Bucket\n", - "\n", - "Let's see what objects are now in our bucket." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if tutorial_bucket:\n", - " print(f\"๐Ÿ“‹ Listing objects in bucket: {tutorial_bucket}\")\n", - " \n", - " # List all objects\n", - " objects_response = make_api_request(\"GET\", f\"/s3/objects/{tutorial_bucket}\")\n", - " print_response(objects_response, \"Objects in Bucket\")\n", - " \n", - " if \"error\" not in objects_response and \"objects\" in objects_response:\n", - " objects = objects_response[\"objects\"]\n", - " print(f\"\\n๐Ÿ“ˆ Found {len(objects)} objects\")\n", - " \n", - " if objects:\n", - " print(\"\\nObjects:\")\n", - " for i, obj in enumerate(objects, 1):\n", - " size_mb = obj.get(\"size\", 0) / 1024 / 1024\n", - " print(f\" {i}. {obj['key']}\")\n", - " print(f\" ๐Ÿ“ Size: {obj.get('size', 0)} bytes ({size_mb:.2f} MB)\")\n", - " print(f\" ๐Ÿ“… Modified: {obj.get('last_modified', 'N/A')}\")\n", - " print(f\" ๐Ÿท๏ธ ETag: {obj.get('etag', 'N/A')}\")\n", - " print(f\" ๐Ÿ“„ Content-Type: {obj.get('content_type', 'N/A')}\")\n", - " print()\n", - " \n", - " # Store first object for later operations\n", - " if objects:\n", - " sample_object = objects[0][\"key\"]\n", - " print(f\"๐ŸŽฏ Selected '{sample_object}' for further operations\")\n", - " else:\n", - " sample_object = None\n", - " else:\n", - " print(\"โŒ Failed to list objects\")\n", - " sample_object = None\n", - " \n", - " # Test with prefix filter\n", - " if objects:\n", - " print(f\"\\n๐Ÿ” Testing prefix filter with 'sample'...\")\n", - " filtered_response = make_api_request(\n", - " \"GET\", \n", - " f\"/s3/objects/{tutorial_bucket}\",\n", - " params={\"prefix\": \"sample\"}\n", - " )\n", - " \n", - " if \"error\" not in filtered_response:\n", - " filtered_objects = filtered_response.get(\"objects\", [])\n", - " print(f\"๐Ÿ“‹ Found {len(filtered_objects)} objects with 'sample' prefix\")\n", - " for obj in filtered_objects:\n", - " print(f\" - {obj['key']}\")\n", - " \n", - "else:\n", - " print(\"โš ๏ธ No bucket available for object listing\")\n", - " sample_object = None" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Get Object Metadata\n", - "\n", - "Let's retrieve detailed metadata for one of our objects." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if tutorial_bucket and sample_object:\n", - " print(f\"๐Ÿ“Š Getting metadata for object: {sample_object}\")\n", - " \n", - " metadata_response = make_api_request(\n", - " \"GET\", \n", - " f\"/s3/objects/{tutorial_bucket}/{sample_object}/metadata\"\n", - " )\n", - " print_response(metadata_response, f\"Metadata: {sample_object}\")\n", - " \n", - " if \"error\" not in metadata_response:\n", - " print(\"\\nโœ… Metadata retrieved successfully!\")\n", - " print(f\"๐Ÿ”‘ Key: {metadata_response.get('key', 'N/A')}\")\n", - " print(f\"๐Ÿ“ Size: {metadata_response.get('size', 'N/A')} bytes\")\n", - " print(f\"๐Ÿ“„ Content-Type: {metadata_response.get('content_type', 'N/A')}\")\n", - " print(f\"๐Ÿ“… Last Modified: {metadata_response.get('last_modified', 'N/A')}\")\n", - " print(f\"๐Ÿท๏ธ ETag: {metadata_response.get('etag', 'N/A')}\")\n", - " \n", - " custom_metadata = metadata_response.get('metadata', {})\n", - " if custom_metadata:\n", - " print(f\"๐Ÿท๏ธ Custom Metadata: {custom_metadata}\")\n", - " else:\n", - " print(\"๐Ÿท๏ธ No custom metadata found\")\n", - " else:\n", - " print(\"โŒ Failed to retrieve metadata\")\n", - "else:\n", - " print(\"โš ๏ธ No bucket or object available for metadata retrieval\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Download Objects\n", - "\n", - "Let's download one of our objects to verify the upload worked correctly." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if tutorial_bucket and sample_object:\n", - " print(f\"๐Ÿ“ฅ Downloading object: {sample_object}\")\n", - " \n", - " download_response = make_api_request(\n", - " \"GET\", \n", - " f\"/s3/objects/{tutorial_bucket}/{sample_object}\"\n", - " )\n", - " \n", - " if \"error\" not in download_response and \"content\" in download_response:\n", - " print(\"\\nโœ… File downloaded successfully!\")\n", - " \n", - " content = download_response[\"content\"]\n", - " headers = download_response[\"headers\"]\n", - " \n", - " print(f\"๐Ÿ“ Downloaded size: {len(content)} bytes\")\n", - " print(f\"๐Ÿ“„ Content-Type: {headers.get('content-type', 'N/A')}\")\n", - " print(f\"๐Ÿ“Ž Content-Disposition: {headers.get('content-disposition', 'N/A')}\")\n", - " \n", - " # Save downloaded file\n", - " download_filename = f\"downloaded_{sample_object}\"\n", - " with open(download_filename, 'wb') as f:\n", - " f.write(content)\n", - " \n", - " print(f\"๐Ÿ’พ Saved as: {download_filename}\")\n", - " \n", - " # Show first 200 characters if it's a text file\n", - " if headers.get('content-type', '').startswith('text') or sample_object.endswith('.txt'):\n", - " try:\n", - " text_content = content.decode('utf-8')\n", - " preview = text_content[:200] + (\"...\" if len(text_content) > 200 else \"\")\n", - " print(f\"\\n๐Ÿ“„ File preview:\\n{preview}\")\n", - " except:\n", - " print(\"\\n๐Ÿ“„ (Binary content - cannot preview as text)\")\n", - " \n", - " # Clean up downloaded file\n", - " try:\n", - " os.remove(download_filename)\n", - " print(f\"๐Ÿ—‘๏ธ Cleaned up: {download_filename}\")\n", - " except:\n", - " pass\n", - " \n", - " else:\n", - " print(\"โŒ Failed to download file\")\n", - " print_response(download_response, \"Download Error\")\n", - "else:\n", - " print(\"โš ๏ธ No bucket or object available for download\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 5. Presigned URLs\n", - "\n", - "Presigned URLs allow you to provide temporary, secure access to objects without requiring authentication. This is useful for sharing files or allowing uploads from client applications." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Generate Presigned Download URL\n", - "\n", - "Let's create a presigned URL for downloading one of our objects." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if tutorial_bucket and sample_object:\n", - " print(f\"๐Ÿ”— Generating presigned download URL for: {sample_object}\")\n", - " \n", - " # Request presigned URL (valid for 1 hour)\n", - " presigned_request = {\n", - " \"expires_in\": 3600 # 1 hour in seconds\n", - " }\n", - " \n", - " presigned_response = make_api_request(\n", - " \"POST\",\n", - " f\"/s3/objects/{tutorial_bucket}/{sample_object}/presigned-download\",\n", - " data=presigned_request\n", - " )\n", - " print_response(presigned_response, \"Presigned Download URL\")\n", - " \n", - " if \"error\" not in presigned_response and \"url\" in presigned_response:\n", - " print(\"\\nโœ… Presigned download URL generated successfully!\")\n", - " print(f\"๐Ÿ”— URL: {presigned_response['url'][:100]}...\")\n", - " print(f\"โฐ Expires in: {presigned_response['expires_in']} seconds\")\n", - " print(f\"๐Ÿ“… Valid until: {datetime.fromtimestamp(time.time() + presigned_response['expires_in'])}\")\n", - " \n", - " # Test the presigned URL\n", - " print(\"\\n๐Ÿงช Testing presigned URL...\")\n", - " try:\n", - " test_response = requests.get(presigned_response['url'])\n", - " if test_response.status_code == 200:\n", - " print(\"โœ… Presigned URL works correctly!\")\n", - " print(f\"๐Ÿ“ Downloaded {len(test_response.content)} bytes\")\n", - " else:\n", - " print(f\"โŒ Presigned URL test failed: {test_response.status_code}\")\n", - " except Exception as e:\n", - " print(f\"โŒ Error testing presigned URL: {e}\")\n", - " else:\n", - " print(\"โŒ Failed to generate presigned download URL\")\n", - "else:\n", - " print(\"โš ๏ธ No bucket or object available for presigned URL generation\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Generate Presigned Upload URL\n", - "\n", - "Let's create a presigned URL for uploading a new object." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if tutorial_bucket:\n", - " print(\"๐Ÿ“ค Generating presigned upload URL for new object\")\n", - " \n", - " new_object_key = f\"presigned_upload_{timestamp}.txt\"\n", - " \n", - " # Request presigned URL (valid for 30 minutes)\n", - " presigned_request = {\n", - " \"expires_in\": 1800 # 30 minutes in seconds\n", - " }\n", - " \n", - " upload_url_response = make_api_request(\n", - " \"POST\",\n", - " f\"/s3/objects/{tutorial_bucket}/{new_object_key}/presigned-upload\",\n", - " data=presigned_request\n", - " )\n", - " print_response(upload_url_response, \"Presigned Upload URL\")\n", - " \n", - " if \"error\" not in upload_url_response and \"url\" in upload_url_response:\n", - " print(\"\\nโœ… Presigned upload URL generated successfully!\")\n", - " print(f\"๐Ÿ”— URL: {upload_url_response['url'][:100]}...\")\n", - " print(f\"โฐ Expires in: {upload_url_response['expires_in']} seconds\")\n", - " print(f\"๐ŸŽฏ Target object: {new_object_key}\")\n", - " \n", - " # Test the presigned upload URL\n", - " print(\"\\n๐Ÿงช Testing presigned upload URL...\")\n", - " test_content = f\"\"\"This file was uploaded using a presigned URL!\n", - "Upload time: {datetime.now().isoformat()}\n", - "Object key: {new_object_key}\n", - "Bucket: {tutorial_bucket}\n", - "\n", - "Presigned URLs are great for:\n", - "- Client-side uploads without exposing credentials\n", - "- Temporary access to resources\n", - "- Integration with web applications\n", - "\"\"\"\n", - " \n", - " try:\n", - " test_response = requests.put(\n", - " upload_url_response['url'],\n", - " data=test_content.encode('utf-8'),\n", - " headers={'Content-Type': 'text/plain'}\n", - " )\n", - " \n", - " if test_response.status_code == 200:\n", - " print(\"โœ… Presigned upload URL works correctly!\")\n", - " print(f\"๐Ÿ“ค Uploaded {len(test_content)} bytes\")\n", - " print(f\"๐ŸŽฏ Object uploaded to: {new_object_key}\")\n", - " else:\n", - " print(f\"โŒ Presigned upload URL test failed: {test_response.status_code}\")\n", - " print(f\"Response: {test_response.text}\")\n", - " except Exception as e:\n", - " print(f\"โŒ Error testing presigned upload URL: {e}\")\n", - " else:\n", - " print(\"โŒ Failed to generate presigned upload URL\")\n", - "else:\n", - " print(\"โš ๏ธ No bucket available for presigned upload URL generation\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 6. Error Handling and Best Practices\n", - "\n", - "Let's demonstrate proper error handling and common scenarios you might encounter." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Common Error Scenarios\n", - "\n", - "Here are examples of common errors and how to handle them:" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "print(\"๐Ÿงช Testing common error scenarios...\")\n", - "\n", - "# Example 1: Non-existent bucket\n", - "print(\"\\n1๏ธโƒฃ Testing non-existent bucket:\")\n", - "error_response_1 = make_api_request(\"GET\", \"/s3/objects/non-existent-bucket\")\n", - "print_response(error_response_1, \"Error: Non-existent Bucket\")\n", - "\n", - "# Example 2: Non-existent object\n", - "if tutorial_bucket:\n", - " print(\"\\n2๏ธโƒฃ Testing non-existent object:\")\n", - " error_response_2 = make_api_request(\n", - " \"GET\", \n", - " f\"/s3/objects/{tutorial_bucket}/non-existent-file.txt/metadata\"\n", - " )\n", - " print_response(error_response_2, \"Error: Non-existent Object\")\n", - "\n", - "# Example 3: Invalid bucket name\n", - "print(\"\\n3๏ธโƒฃ Testing invalid bucket name:\")\n", - "invalid_bucket_data = {\n", - " \"name\": \"Invalid Bucket Name With Spaces!\", # Invalid characters\n", - " \"region\": \"us-east-1\"\n", - "}\n", - "error_response_3 = make_api_request(\"POST\", \"/s3/buckets\", data=invalid_bucket_data)\n", - "print_response(error_response_3, \"Error: Invalid Bucket Name\")\n", - "\n", - "# Example 4: Duplicate bucket creation\n", - "if tutorial_bucket:\n", - " print(\"\\n4๏ธโƒฃ Testing duplicate bucket creation:\")\n", - " duplicate_bucket_data = {\n", - " \"name\": tutorial_bucket,\n", - " \"region\": \"us-east-1\"\n", - " }\n", - " error_response_4 = make_api_request(\"POST\", \"/s3/buckets\", data=duplicate_bucket_data)\n", - " print_response(error_response_4, \"Error: Duplicate Bucket\")\n", - "\n", - "print(\"\\n๐Ÿ“š Key Takeaways from Error Examples:\")\n", - "print(\" โ€ข Always check for 'error' key in responses\")\n", - "print(\" โ€ข Handle HTTP status codes appropriately\")\n", - "print(\" โ€ข Bucket names must follow S3 naming conventions\")\n", - "print(\" โ€ข Check bucket existence before object operations\")\n", - "print(\" โ€ข Implement retry logic for transient failures\")\n", - "print(\" โ€ข Validate input parameters before API calls\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 7. Advanced Use Cases\n", - "\n", - "Here are some advanced patterns for using the S3 API in production environments." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Batch Operations\n", - "\n", - "For managing multiple files efficiently:" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "def batch_upload_files(bucket_name: str, files_data: list, delay: float = 0.2) -> list:\n", - " \"\"\"\n", - " Upload multiple files with rate limiting and error handling.\n", - " \n", - " Args:\n", - " bucket_name: Target bucket name\n", - " files_data: List of file dictionaries with 'filename' and 'content'\n", - " delay: Delay between uploads to avoid rate limiting\n", - " \n", - " Returns:\n", - " List of upload results\n", - " \"\"\"\n", - " results = []\n", - " \n", - " for i, file_data in enumerate(files_data, 1):\n", - " filename = file_data[\"filename\"]\n", - " content = file_data[\"content\"]\n", - " \n", - " print(f\"\\n๐Ÿ“ฆ Uploading file {i}/{len(files_data)}: {filename}\")\n", - " \n", - " try:\n", - " # Create temporary file\n", - " temp_filename = f\"temp_{filename}\"\n", - " with open(temp_filename, 'w') as f:\n", - " f.write(content)\n", - " \n", - " # Upload file\n", - " with open(temp_filename, 'rb') as f:\n", - " files = {\"file\": (filename, f, \"text/plain\")}\n", - " form_data = {\"object_key\": filename}\n", - " \n", - " response = make_api_request(\n", - " \"POST\",\n", - " f\"/s3/objects/{bucket_name}\",\n", - " data=form_data,\n", - " files=files\n", - " )\n", - " \n", - " results.append({\n", - " \"filename\": filename,\n", - " \"success\": \"error\" not in response,\n", - " \"response\": response\n", - " })\n", - " \n", - " # Clean up\n", - " os.remove(temp_filename)\n", - " \n", - " except Exception as e:\n", - " results.append({\n", - " \"filename\": filename,\n", - " \"success\": False,\n", - " \"error\": str(e)\n", - " })\n", - " \n", - " # Rate limiting\n", - " if i < len(files_data):\n", - " time.sleep(delay)\n", - " \n", - " return results\n", - "\n", - "# Example batch upload\n", - "if tutorial_bucket:\n", - " print(\"๐Ÿš€ Demonstrating batch file upload...\")\n", - " \n", - " batch_files = [\n", - " {\n", - " \"filename\": f\"batch_file_1_{timestamp}.txt\",\n", - " \"content\": \"This is batch file 1 for testing bulk uploads.\"\n", - " },\n", - " {\n", - " \"filename\": f\"batch_file_2_{timestamp}.txt\",\n", - " \"content\": \"This is batch file 2 for testing bulk uploads.\"\n", - " },\n", - " {\n", - " \"filename\": f\"batch_file_3_{timestamp}.txt\",\n", - " \"content\": \"This is batch file 3 for testing bulk uploads.\"\n", - " }\n", - " ]\n", - " \n", - " batch_results = batch_upload_files(tutorial_bucket, batch_files)\n", - " \n", - " # Summary\n", - " successful = sum(1 for r in batch_results if r[\"success\"])\n", - " failed = len(batch_results) - successful\n", - " \n", - " print(f\"\\n๐Ÿ“Š Batch Upload Summary:\")\n", - " print(f\" โœ… Successful: {successful}\")\n", - " print(f\" โŒ Failed: {failed}\")\n", - " print(f\" ๐Ÿ“ˆ Success Rate: {(successful/len(batch_results)*100):.1f}%\")\n", - "else:\n", - " print(\"โš ๏ธ Bucket required for batch upload demo\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Cleanup Operations\n", - "\n", - "Let's clean up our tutorial resources. **Warning: This will permanently delete the files and bucket created in this tutorial.**" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# OPTIONAL: Cleanup tutorial resources\n", - "# Set to True to enable cleanup\n", - "CLEANUP_ENABLED = False\n", - "\n", - "if CLEANUP_ENABLED and tutorial_bucket:\n", - " print(\"๐Ÿงน Starting cleanup of tutorial resources...\")\n", - " \n", - " # First, list all objects in the bucket\n", - " print(f\"\\n๐Ÿ“‹ Listing objects to delete in bucket: {tutorial_bucket}\")\n", - " objects_response = make_api_request(\"GET\", f\"/s3/objects/{tutorial_bucket}\")\n", - " \n", - " if \"error\" not in objects_response and \"objects\" in objects_response:\n", - " objects_to_delete = objects_response[\"objects\"]\n", - " print(f\"Found {len(objects_to_delete)} objects to delete\")\n", - " \n", - " # Delete each object\n", - " deleted_objects = 0\n", - " for obj in objects_to_delete:\n", - " object_key = obj[\"key\"]\n", - " print(f\"\\n๐Ÿ—‘๏ธ Deleting object: {object_key}\")\n", - " \n", - " delete_response = make_api_request(\n", - " \"DELETE\",\n", - " f\"/s3/objects/{tutorial_bucket}/{object_key}\"\n", - " )\n", - " \n", - " if \"error\" not in delete_response:\n", - " print(f\"โœ… Deleted: {object_key}\")\n", - " deleted_objects += 1\n", - " else:\n", - " print(f\"โŒ Failed to delete: {object_key}\")\n", - " \n", - " print(f\"\\n๐Ÿ“Š Deleted {deleted_objects}/{len(objects_to_delete)} objects\")\n", - " \n", - " # Now delete the bucket (must be empty)\n", - " if deleted_objects == len(objects_to_delete):\n", - " print(f\"\\n๐Ÿ—‘๏ธ Deleting bucket: {tutorial_bucket}\")\n", - " bucket_delete_response = make_api_request(\n", - " \"DELETE\",\n", - " f\"/s3/buckets/{tutorial_bucket}\"\n", - " )\n", - " \n", - " if \"error\" not in bucket_delete_response:\n", - " print(f\"โœ… Bucket deleted: {tutorial_bucket}\")\n", - " else:\n", - " print(f\"โŒ Failed to delete bucket: {tutorial_bucket}\")\n", - " print_response(bucket_delete_response, \"Bucket Deletion Error\")\n", - " else:\n", - " print(\"โš ๏ธ Bucket not deleted - some objects remain\")\n", - " else:\n", - " print(\"โŒ Failed to list objects for cleanup\")\n", - " \n", - " print(\"\\n๐Ÿงน Cleanup completed!\")\n", - "else:\n", - " if tutorial_bucket:\n", - " print(\"โ„น๏ธ Cleanup disabled. Tutorial resources preserved.\")\n", - " print(\"๐Ÿ’ก To clean up, set CLEANUP_ENABLED = True and run this cell again.\")\n", - " print(f\"๐Ÿ“ฆ Created bucket: {tutorial_bucket}\")\n", - " print(\"๐Ÿ—‚๏ธ Objects uploaded during tutorial remain in the bucket\")\n", - " else:\n", - " print(\"โ„น๏ธ No tutorial resources to clean up.\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 8. Summary and Best Practices\n", - "\n", - "Congratulations! You've successfully completed the S3 API tutorial. Here's a summary of what we covered:" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "print(\"๐ŸŽ‰ S3 API Tutorial Summary\")\n", - "print(\"=\" * 50)\n", - "\n", - "print(\"\\nโœ… What we accomplished:\")\n", - "print(\" ๐Ÿ“ก Tested API connectivity and authentication\")\n", - "print(\" ๐Ÿชฃ Created and managed S3 buckets\")\n", - "print(\" ๐Ÿ“ Uploaded multiple files to buckets\")\n", - "print(\" ๐Ÿ“‹ Listed and filtered objects\")\n", - "print(\" ๐Ÿ“Š Retrieved object metadata\")\n", - "print(\" ๐Ÿ“ฅ Downloaded files from buckets\")\n", - "print(\" ๐Ÿ”— Generated presigned URLs for secure sharing\")\n", - "print(\" ๐Ÿงช Handled common error scenarios\")\n", - "print(\" ๐Ÿš€ Demonstrated batch operations\")\n", - "\n", - "print(\"\\n๐ŸŽฏ Key API Endpoints Covered:\")\n", - "print(\" โ€ข GET /s3/buckets - List buckets\")\n", - "print(\" โ€ข POST /s3/buckets - Create bucket\")\n", - "print(\" โ€ข GET /s3/buckets/{name} - Get bucket info\")\n", - "print(\" โ€ข DELETE /s3/buckets/{name} - Delete bucket\")\n", - "print(\" โ€ข GET /s3/objects/{bucket} - List objects\")\n", - "print(\" โ€ข POST /s3/objects/{bucket} - Upload object\")\n", - "print(\" โ€ข GET /s3/objects/{bucket}/{key} - Download object\")\n", - "print(\" โ€ข DELETE /s3/objects/{bucket}/{key} - Delete object\")\n", - "print(\" โ€ข GET /s3/objects/{bucket}/{key}/metadata - Get metadata\")\n", - "print(\" โ€ข POST /s3/objects/{bucket}/{key}/presigned-* - Presigned URLs\")\n", - "\n", - "print(\"\\n๐Ÿ’ก Best Practices:\")\n", - "print(\" ๐Ÿ” Always authenticate your requests properly\")\n", - "print(\" ๐Ÿงช Check for errors in every API response\")\n", - "print(\" ๐Ÿ“Š Use appropriate HTTP status codes for error handling\")\n", - "print(\" ๐Ÿชฃ Follow S3 bucket naming conventions\")\n", - "print(\" ๐Ÿ“ Be mindful of file sizes and upload limits\")\n", - "print(\" ๐Ÿ”— Use presigned URLs for client-side operations\")\n", - "print(\" โฑ๏ธ Implement rate limiting for batch operations\")\n", - "print(\" ๐Ÿงน Clean up resources when no longer needed\")\n", - "print(\" ๐Ÿ“ Include meaningful metadata with your objects\")\n", - "print(\" ๐Ÿ” Use prefix filters for efficient object listing\")\n", - "\n", - "print(\"\\n๐Ÿš€ Next Steps:\")\n", - "print(\" โ€ข Integrate S3 operations into your applications\")\n", - "print(\" โ€ข Implement proper error handling and retry logic\")\n", - "print(\" โ€ข Set up monitoring and logging for production use\")\n", - "print(\" โ€ข Consider using S3 versioning and lifecycle policies\")\n", - "print(\" โ€ข Explore advanced features like multipart uploads\")\n", - "\n", - "print(\"\\n๐Ÿ“š Additional Resources:\")\n", - "print(\" โ€ข API Documentation: Check your API instance docs\")\n", - "print(\" โ€ข S3 Compatibility: Review AWS S3 documentation\")\n", - "print(\" โ€ข MinIO Documentation: https://min.io/docs/minio/\")\n", - "\n", - "if tutorial_bucket:\n", - " print(f\"\\n๐ŸŽฏ Tutorial completed successfully!\")\n", - " print(f\"๐Ÿ“ฆ Created bucket: {tutorial_bucket}\")\n", - " print(\"๐Ÿ“ Uploaded sample files and demonstrated all major operations\")\n", - "else:\n", - " print(f\"\\nโš ๏ธ Tutorial completed with some limitations due to authentication or service configuration\")\n", - "\n", - "print(\"\\n\" + \"=\" * 50)" - ] - } - ], - "metadata": { - "kernelspec": { - "display_name": "Python 3", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.8.5" - } - }, - "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file diff --git a/docs/.ipynb_checkpoints/s3_to_dataset_workflow_tutorial-checkpoint.ipynb b/docs/.ipynb_checkpoints/s3_to_dataset_workflow_tutorial-checkpoint.ipynb deleted file mode 100644 index 8f844ee..0000000 --- a/docs/.ipynb_checkpoints/s3_to_dataset_workflow_tutorial-checkpoint.ipynb +++ /dev/null @@ -1,1030 +0,0 @@ -{ - "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "# NDP EP Tutorial: S3 Storage to Dataset Registration Workflow\n", - "\n", - "[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/sci-ndp/pop/blob/main/docs/s3_to_dataset_workflow_tutorial.ipynb)\n", - "[![Open in Binder](https://mybinder.org/badge_logo.svg)](https://mybinder.org/v2/gh/sci-ndp/pop/main?filepath=docs/s3_to_dataset_workflow_tutorial.ipynb)\n", - "\n", - "> ๐Ÿš€ **Run Online Options:**\n", - "> - **Google Colab**: Dependencies installed automatically in the first cell\n", - "> - **Binder**: Pre-configured environment, ready to run immediately\n", - "> - **Local**: Requires `pip install requests jupyter`\n", - "\n", - "This notebook demonstrates a complete workflow for scientific data management using the NDP EP API:\n", - "\n", - "1. **Upload files to S3 storage** using MINIO endpoints\n", - "2. **Generate presigned URLs** for secure access\n", - "3. **Register datasets** with S3 URLs as resources\n", - "4. **Manage the complete data lifecycle** from storage to discovery\n", - "\n", - "## Use Cases\n", - "\n", - "This workflow is perfect for:\n", - "- **Research data management**: Store large datasets in S3 and register them for discovery\n", - "- **Reproducible science**: Create permanent links to data files with rich metadata\n", - "- **Data publishing**: Combine storage with catalog registration for data sharing\n", - "- **Institutional repositories**: Manage both storage and metadata in one workflow\n", - "\n", - "## Prerequisites\n", - "\n", - "- Python 3.7+\n", - "- `requests` library\n", - "- Access to a NDP EP API instance with S3/MINIO configured\n", - "- Valid authentication credentials\n", - "- Data files to upload\n", - "\n", - "## Workflow Overview\n", - "\n", - "```\n", - "Local File โ†’ S3 Upload โ†’ Generate URL โ†’ Register Dataset โ†’ Published Data\n", - "```" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Install required packages\n", - "!pip install requests -q" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 1. Setup and Configuration\n", - "\n", - "First, let's import the necessary libraries and configure our API connection parameters." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import requests\n", - "import json\n", - "from typing import Dict, Any, Optional\n", - "import time\n", - "import io\n", - "import os\n", - "from datetime import datetime\n", - "from pprint import pprint" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Configuration Variables\n", - "\n", - "**Important:** Replace these values with your actual API endpoint and credentials." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# API Configuration\n", - "API_BASE_URL = \"http://localhost:8000\" # Replace with your API URL\n", - "\n", - "# Authentication Token\n", - "AUTH_TOKEN = \"testing_token\" # Replace with your actual token\n", - "\n", - "# Request headers with authentication\n", - "HEADERS = {\n", - " \"Authorization\": f\"Bearer {AUTH_TOKEN}\",\n", - " \"Accept\": \"application/json\"\n", - "}\n", - "\n", - "print(f\"API Base URL: {API_BASE_URL}\")\n", - "print(f\"Token configured: {'โœ“' if AUTH_TOKEN != 'your_auth_token_here' else 'โœ— Please set your token'}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Helper Functions\n", - "\n", - "Let's create utility functions for both S3 and dataset operations." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "def make_api_request(method: str, endpoint: str, data: Optional[Dict] = None, \n", - " params: Optional[Dict] = None, files: Optional[Dict] = None,\n", - " custom_headers: Optional[Dict] = None) -> Dict[str, Any]:\n", - " \"\"\"\n", - " Make an API request with proper error handling for both S3 and dataset endpoints.\n", - " \"\"\"\n", - " url = f\"{API_BASE_URL}{endpoint}\"\n", - " \n", - " # Prepare headers\n", - " headers = HEADERS.copy()\n", - " if custom_headers:\n", - " headers.update(custom_headers)\n", - " \n", - " # Remove Content-Type for file uploads to let requests set it\n", - " if files and \"Content-Type\" in headers:\n", - " del headers[\"Content-Type\"]\n", - " \n", - " try:\n", - " response = requests.request(\n", - " method=method,\n", - " url=url,\n", - " headers=headers,\n", - " json=data if not files else None,\n", - " data=data if files else None,\n", - " files=files,\n", - " params=params,\n", - " stream=(method == \"GET\" and \"download\" in endpoint.lower())\n", - " )\n", - " \n", - " print(f\"๐Ÿ”— {method} {url}\")\n", - " print(f\"๐Ÿ“Š Status: {response.status_code}\")\n", - " \n", - " if response.status_code in [200, 201, 204]:\n", - " # Handle streaming responses (file downloads)\n", - " if response.headers.get('content-type', '').startswith('application/octet-stream') or \\\n", - " 'attachment' in response.headers.get('content-disposition', ''):\n", - " print(\"โœ… Success! (File download)\")\n", - " return {\n", - " \"success\": True,\n", - " \"content\": response.content,\n", - " \"headers\": dict(response.headers),\n", - " \"status_code\": response.status_code\n", - " }\n", - " \n", - " # Handle JSON responses\n", - " try:\n", - " result = response.json()\n", - " print(\"โœ… Success!\")\n", - " return result\n", - " except ValueError:\n", - " # Handle non-JSON success responses\n", - " print(\"โœ… Success! (Non-JSON response)\")\n", - " return {\"success\": True, \"status_code\": response.status_code}\n", - " else:\n", - " print(f\"โŒ Error: {response.status_code}\")\n", - " try:\n", - " error_detail = response.json()\n", - " print(f\"Error details: {json.dumps(error_detail, indent=2)}\")\n", - " except:\n", - " print(f\"Error text: {response.text}\")\n", - " return {\"error\": True, \"status_code\": response.status_code, \"detail\": response.text}\n", - " \n", - " except requests.exceptions.RequestException as e:\n", - " print(f\"โŒ Request failed: {e}\")\n", - " return {\"error\": True, \"exception\": str(e)}\n", - "\n", - "def print_response(response: Dict[str, Any], title: str = \"Response\"):\n", - " \"\"\"Pretty print API responses.\"\"\"\n", - " print(f\"\\n๐Ÿ“‹ {title}:\")\n", - " print(\"โ”€\" * 50)\n", - " if \"content\" in response: # File download response\n", - " print(f\"File size: {len(response['content'])} bytes\")\n", - " print(f\"Headers: {response['headers']}\")\n", - " else:\n", - " pprint(response)\n", - " print(\"โ”€\" * 50)\n", - "\n", - "def create_sample_file(filename: str, content: str = None, size_kb: int = None) -> str:\n", - " \"\"\"Create a sample file for upload testing.\"\"\"\n", - " if content is None:\n", - " if size_kb:\n", - " # Create file of specific size\n", - " content = \"Sample data line for workflow tutorial.\\n\" * (size_kb * 25) # Approx 1KB per 25 lines\n", - " else:\n", - " content = f\"\"\"# Research Data File - {filename}\n", - "# Created: {datetime.now().isoformat()}\n", - "# Purpose: NDP EP S3-to-Dataset workflow tutorial\n", - "\n", - "timestamp,temperature,humidity,pressure\n", - "2024-01-01T00:00:00Z,23.5,45.2,1013.25\n", - "2024-01-01T01:00:00Z,23.2,46.1,1013.15\n", - "2024-01-01T02:00:00Z,22.8,47.3,1012.98\n", - "2024-01-01T03:00:00Z,22.4,48.0,1012.75\n", - "2024-01-01T04:00:00Z,22.1,48.9,1012.60\n", - "\n", - "# This is sample weather data for the tutorial\n", - "# In a real workflow, this would be your actual research data\n", - "\"\"\"\n", - " \n", - " with open(filename, 'w') as f:\n", - " f.write(content)\n", - " \n", - " size_bytes = os.path.getsize(filename)\n", - " print(f\"๐Ÿ“„ Created sample file: {filename} ({size_bytes} bytes)\")\n", - " return filename" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 2. API Connectivity Test\n", - "\n", - "Let's verify that both S3 and dataset endpoints are accessible." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Test API connectivity\n", - "print(\"๐Ÿงช Testing API connectivity...\")\n", - "status_response = make_api_request(\"GET\", \"/status/\")\n", - "print_response(status_response, \"API Status\")\n", - "\n", - "if \"error\" not in status_response:\n", - " print(\"๐ŸŽ‰ API is accessible and responding correctly!\")\n", - " \n", - " # Test S3 endpoints\n", - " print(\"\\n๐Ÿงช Testing S3 service...\")\n", - " buckets_response = make_api_request(\"GET\", \"/s3/buckets/\")\n", - " \n", - " if \"error\" not in buckets_response:\n", - " print(\"โœ… S3 service is available and configured!\")\n", - " print(f\"๐Ÿ“ฆ Found {len(buckets_response.get('buckets', []))} existing buckets\")\n", - " else:\n", - " print(\"โš ๏ธ S3 service may not be configured\")\nelse:\n", - " print(\"โš ๏ธ API connectivity issues. Please check your configuration.\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 3. Step 1: Prepare and Upload Data to S3\n", - "\n", - "First, we'll create sample research data and upload it to S3 storage." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Create Sample Research Data\n", - "\n", - "Let's create some sample files that represent typical research data." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Generate unique identifiers for this workflow\n", - "timestamp = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n", - "workflow_id = f\"workflow_{timestamp}\"\n", - "\n", - "print(f\"๐Ÿ”ฌ Creating sample research data for workflow: {workflow_id}\")\n", - "\n", - "# Create different types of research files\n", - "sample_files = [\n", - " {\n", - " \"filename\": f\"temperature_data_{timestamp}.csv\",\n", - " \"content\": f\"\"\"# Temperature Measurement Dataset\n", - "# Workflow ID: {workflow_id}\n", - "# Created: {datetime.now().isoformat()}\n", - "# Instrument: Weather Station Network\n", - "# Location: Research Site A\n", - "\n", - "timestamp,temperature_celsius,quality_flag,station_id\n", - "2024-01-01T00:00:00Z,23.5,GOOD,WS001\n", - "2024-01-01T01:00:00Z,23.2,GOOD,WS001\n", - "2024-01-01T02:00:00Z,22.8,GOOD,WS001\n", - "2024-01-01T03:00:00Z,22.4,SUSPECT,WS001\n", - "2024-01-01T04:00:00Z,22.1,GOOD,WS001\n", - "2024-01-01T05:00:00Z,21.9,GOOD,WS001\n", - "2024-01-01T06:00:00Z,22.3,GOOD,WS001\n", - "2024-01-01T07:00:00Z,23.1,GOOD,WS001\n", - "2024-01-01T08:00:00Z,24.2,GOOD,WS001\n", - "2024-01-01T09:00:00Z,25.8,GOOD,WS001\n", - "\"\"\",\n", - " \"description\": \"Hourly temperature measurements with quality flags\",\n", - " \"format\": \"CSV\"\n", - " },\n", - " {\n", - " \"filename\": f\"analysis_results_{timestamp}.json\",\n", - " \"content\": json.dumps({\n", - " \"workflow_id\": workflow_id,\n", - " \"analysis_type\": \"statistical_summary\",\n", - " \"created\": datetime.now().isoformat(),\n", - " \"data_sources\": [\"temperature_sensors\", \"humidity_sensors\"],\n", - " \"results\": {\n", - " \"temperature_stats\": {\n", - " \"mean\": 23.21,\n", - " \"std_dev\": 1.15,\n", - " \"min\": 21.9,\n", - " \"max\": 25.8,\n", - " \"n_observations\": 10\n", - " },\n", - " \"quality_assessment\": {\n", - " \"good_data_percentage\": 90.0,\n", - " \"suspect_data_count\": 1,\n", - " \"missing_data_count\": 0\n", - " }\n", - " },\n", - " \"methodology\": \"Standard statistical analysis with outlier detection\",\n", - " \"software_version\": \"Analysis Pipeline v2.1\"\n", - " }, indent=2),\n", - " \"description\": \"Statistical analysis results in JSON format\",\n", - " \"format\": \"JSON\"\n", - " },\n", - " {\n", - " \"filename\": f\"methodology_{timestamp}.md\",\n", - " \"content\": f\"\"\"# Research Methodology Documentation\n", - "\n", - "**Workflow ID:** {workflow_id} \n", - "**Created:** {datetime.now().isoformat()} \n", - "**Study Type:** Environmental Monitoring\n", - "\n", - "## Overview\n", - "\n", - "This document describes the methodology used for collecting and analyzing environmental data as part of the NDP EP workflow tutorial.\n", - "\n", - "## Data Collection\n", - "\n", - "### Instruments\n", - "- Weather Station Network (Model: WS-2024)\n", - "- Temperature sensors: ยฑ0.1ยฐC accuracy\n", - "- Sampling interval: 1 hour\n", - "- Quality control: Automated + manual review\n", - "\n", - "### Locations\n", - "- Research Site A: 40.7128ยฐN, 74.0060ยฐW\n", - "- Elevation: 10m above sea level\n", - "- Environment: Urban research station\n", - "\n", - "## Data Processing\n", - "\n", - "1. **Raw Data Collection**\n", - " - Automated data logger retrieval\n", - " - Timestamp validation\n", - " - Initial quality flags\n", - "\n", - "2. **Quality Control**\n", - " - Range checks: -40ยฐC to +50ยฐC\n", - " - Temporal consistency checks\n", - " - Outlier detection using 3-sigma rule\n", - "\n", - "3. **Statistical Analysis**\n", - " - Descriptive statistics\n", - " - Trend analysis\n", - " - Quality metrics\n", - "\n", - "## Data Management\n", - "\n", - "- **Storage**: S3-compatible object storage\n", - "- **Format**: CSV for data, JSON for analysis results\n", - "- **Backup**: Automated daily backups\n", - "- **Access**: Controlled via NDP EP API\n", - "\n", - "## References\n", - "\n", - "- Environmental Monitoring Standards (EMS-2024)\n", - "- Data Quality Guidelines (DQG-v3.2)\n", - "- NDP EP Documentation\n", - "\"\"\",\n", - " \"description\": \"Comprehensive methodology documentation\",\n", - " \"format\": \"Markdown\"\n", - " }\n", - "]\n", - "\n", - "# Create the files\n", - "created_files = []\n", - "for file_info in sample_files:\n", - " filename = create_sample_file(file_info[\"filename\"], file_info[\"content\"])\n", - " created_files.append({\n", - " \"filename\": filename,\n", - " \"description\": file_info[\"description\"],\n", - " \"format\": file_info[\"format\"]\n", - " })\n", - "\n", - "print(f\"\\nโœ… Created {len(created_files)} sample files for the workflow\")\n", - "for i, file_info in enumerate(created_files, 1):\n", - " print(f\" {i}. {file_info['filename']} ({file_info['format']}) - {file_info['description']}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Create S3 Bucket for Research Data\n", - "\n", - "Now let's create a dedicated bucket for our research workflow." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Create a bucket for this workflow\n", - "bucket_name = f\"research-data-{timestamp}\"\n", - "\n", - "print(f\"๐Ÿชฃ Creating S3 bucket: {bucket_name}\")\n", - "\n", - "bucket_data = {\n", - " \"name\": bucket_name,\n", - " \"region\": \"us-east-1\"\n", - "}\n", - "\n", - "bucket_response = make_api_request(\"POST\", \"/s3/buckets/\", data=bucket_data)\n", - "print_response(bucket_response, \"Bucket Creation\")\n", - "\n", - "if \"error\" not in bucket_response:\n", - " print(f\"\\nโœ… Bucket '{bucket_name}' created successfully!\")\n", - " workflow_bucket = bucket_name\n", - "else:\n", - " print(\"โŒ Failed to create bucket. Using existing bucket or manual creation required.\")\n", - " workflow_bucket = None" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Upload Research Files to S3\n", - "\n", - "Let's upload our research files to the S3 bucket." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if workflow_bucket:\n", - " print(f\"๐Ÿ“ค Uploading research files to bucket: {workflow_bucket}\")\n", - " \n", - " uploaded_files = []\n", - " \n", - " for i, file_info in enumerate(created_files, 1):\n", - " filename = file_info[\"filename\"]\n", - " print(f\"\\n๐Ÿ“ Uploading file {i}/{len(created_files)}: {filename}\")\n", - " \n", - " # Upload file\n", - " with open(filename, 'rb') as f:\n", - " files = {\"file\": (filename, f, \"text/plain\")}\n", - " form_data = {\"object_key\": filename}\n", - " \n", - " upload_response = make_api_request(\n", - " \"POST\", \n", - " f\"/s3/objects/{workflow_bucket}\",\n", - " data=form_data,\n", - " files=files\n", - " )\n", - " \n", - " if \"error\" not in upload_response:\n", - " print(f\"โœ… Uploaded: {filename}\")\n", - " print(f\" ๐Ÿ“ Size: {upload_response.get('size', 'N/A')} bytes\")\n", - " print(f\" ๐Ÿ”‘ Key: {upload_response.get('key', 'N/A')}\")\n", - " print(f\" ๐Ÿ“ฆ Bucket: {upload_response.get('bucket', 'N/A')}\")\n", - " \n", - " uploaded_files.append({\n", - " \"filename\": filename,\n", - " \"bucket\": upload_response.get('bucket'),\n", - " \"key\": upload_response.get('key'),\n", - " \"size\": upload_response.get('size'),\n", - " \"description\": file_info[\"description\"],\n", - " \"format\": file_info[\"format\"]\n", - " })\n", - " else:\n", - " print(f\"โŒ Failed to upload: {filename}\")\n", - " print_response(upload_response, \"Upload Error\")\n", - " \n", - " # Clean up local file\n", - " try:\n", - " os.remove(filename)\n", - " except:\n", - " pass\n", - " \n", - " print(f\"\\n๐Ÿ“Š Upload Summary:\")\n", - " print(f\"โœ… Successfully uploaded: {len(uploaded_files)} files\")\n", - " print(f\"๐Ÿ“ฆ Bucket: {workflow_bucket}\")\n", - " print(f\"๐Ÿ’พ Total size: {sum(f.get('size', 0) for f in uploaded_files)} bytes\")\n", - " \n", - "else:\n", - " print(\"โš ๏ธ No bucket available for file upload\")\n", - " uploaded_files = []" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 4. Step 2: Generate Presigned URLs\n", - "\n", - "Now we'll generate presigned URLs for our uploaded files. These URLs will be used as resource links in our dataset." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if uploaded_files:\n", - " print(\"๐Ÿ”— Generating presigned URLs for uploaded files...\")\n", - " \n", - " # Generate presigned URLs (valid for 7 days - maximum allowed)\n", - " presigned_request = {\n", - " \"expires_in\": 604800 # 7 days in seconds (maximum)\n", - " }\n", - " \n", - " files_with_urls = []\n", - " \n", - " for file_info in uploaded_files:\n", - " bucket = file_info[\"bucket\"]\n", - " key = file_info[\"key\"]\n", - " \n", - " print(f\"\\n๐Ÿ”— Generating URL for: {key}\")\n", - " \n", - " url_response = make_api_request(\n", - " \"POST\",\n", - " f\"/s3/objects/{bucket}/{key}/presigned-download\",\n", - " data=presigned_request\n", - " )\n", - " \n", - " if \"error\" not in url_response and \"url\" in url_response:\n", - " print(f\"โœ… Generated presigned URL\")\n", - " print(f\" โฐ Expires in: {url_response['expires_in']} seconds ({url_response['expires_in']//3600} hours)\")\n", - " \n", - " # Add URL to file info\n", - " file_info_with_url = file_info.copy()\n", - " file_info_with_url[\"presigned_url\"] = url_response[\"url\"]\n", - " file_info_with_url[\"url_expires_in\"] = url_response[\"expires_in\"]\n", - " files_with_urls.append(file_info_with_url)\n", - " \n", - " else:\n", - " print(f\"โŒ Failed to generate URL for: {key}\")\n", - " print_response(url_response, \"URL Generation Error\")\n", - " \n", - " print(f\"\\n๐Ÿ“Š URL Generation Summary:\")\n", - " print(f\"โœ… Generated URLs for: {len(files_with_urls)} files\")\n", - " print(f\"๐Ÿ”— URLs valid for: {presigned_request['expires_in']//3600} hours\")\n", - " print(f\"๐Ÿ“… URLs expire on: {datetime.fromtimestamp(time.time() + presigned_request['expires_in'])}\")\n", - " \n", - "else:\n", - " print(\"โš ๏ธ No uploaded files available for URL generation\")\n", - " files_with_urls = []" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 5. Step 3: Create Organization (if needed)\n", - "\n", - "Before registering datasets, we need to ensure we have an organization." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": "# Check existing organizations in CKAN directly\nprint(\"๐Ÿข Checking organizations in local CKAN...\")\n\n# Get organizations directly from CKAN API\ntry:\n ckan_orgs_response = requests.get(\"http://localhost:5000/api/3/action/organization_list\")\n if ckan_orgs_response.status_code == 200:\n ckan_data = ckan_orgs_response.json()\n available_orgs = ckan_data.get('result', [])\n \n print(f\"๐Ÿ“ˆ Found {len(available_orgs)} organizations in local CKAN:\")\n for i, org in enumerate(available_orgs, 1):\n print(f\" {i}. {org}\")\n \n # Use test_raul if available, otherwise use the first one\n if \"test_raul\" in available_orgs:\n organization_name = \"test_raul\"\n elif available_orgs:\n organization_name = available_orgs[0]\n else:\n organization_name = None\n \n if organization_name:\n print(f\"\\nโœ… Using organization: {organization_name}\")\n else:\n print(\"โŒ Could not retrieve organizations from CKAN\")\n organization_name = None\n \nexcept Exception as e:\n print(f\"โŒ Error connecting to CKAN: {e}\")\n print(\"โš ๏ธ Using fallback organization name\")\n organization_name = \"test_raul\" # Fallback to known working organization\n\nif not organization_name:\n print(\"\\nโš ๏ธ No organizations available. Please create one manually first.\")\n print(\" The tutorial requires an existing organization to register datasets.\")\n\nprint(f\"\\n๐ŸŽฏ Organization for dataset: {organization_name or 'None available'}\")" - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 6. Step 4: Register Dataset with S3 Resources\n", - "\n", - "Now we'll create a comprehensive dataset that includes our S3-stored files as resources." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if organization_name and files_with_urls:\n", - " print(f\"๐Ÿ“Š Creating dataset with S3 resources...\")\n", - " \n", - " # Prepare resources from our uploaded files\n", - " dataset_resources = []\n", - " for file_info in files_with_urls:\n", - " resource = {\n", - " \"url\": file_info[\"presigned_url\"],\n", - " \"name\": file_info[\"filename\"],\n", - " \"description\": file_info[\"description\"],\n", - " \"format\": file_info[\"format\"],\n", - " \"size\": file_info[\"size\"]\n", - " }\n", - " \n", - " # Add mimetype based on format\n", - " format_mimetypes = {\n", - " \"CSV\": \"text/csv\",\n", - " \"JSON\": \"application/json\",\n", - " \"Markdown\": \"text/markdown\",\n", - " \"PDF\": \"application/pdf\",\n", - " \"XML\": \"application/xml\"\n", - " }\n", - " resource[\"mimetype\"] = format_mimetypes.get(file_info[\"format\"], \"application/octet-stream\")\n", - " \n", - " dataset_resources.append(resource)\n", - " \n", - " # Create comprehensive dataset payload\n", - " dataset_payload = {\n", - " # Required fields\n", - " \"name\": f\"environmental_monitoring_{timestamp}\",\n", - " \"title\": f\"Environmental Monitoring Dataset - Workflow {workflow_id}\",\n", - " \"owner_org\": organization_name,\n", - " \n", - " # Descriptive metadata\n", - " \"notes\": f\"\"\"This dataset contains environmental monitoring data collected as part of research workflow {workflow_id}. \n", - "\nThe dataset includes:\n", - "- Temperature measurements with quality control flags\n", - "- Statistical analysis results\n", - "- Comprehensive methodology documentation\n", - "\nAll data files are stored in S3 object storage and accessible via presigned URLs. This demonstrates the complete workflow from data collection through storage to dataset registration in the NDP EP platform.\n", - "\n**Data Collection Period:** {datetime.now().strftime('%Y-%m-%d')}\n", - "**Quality Level:** Research Grade\n", - "**Access:** Open Access via presigned URLs\"\"\",\n", - " \n", - " # Categorization\n", - " \"tags\": [\n", - " \"environmental-monitoring\",\n", - " \"temperature\",\n", - " \"research-data\",\n", - " \"s3-workflow\",\n", - " \"quality-controlled\",\n", - " \"open-access\",\n", - " \"tutorial\"\n", - " ],\n", - " \"groups\": [\"environmental\", \"research\", \"monitoring\"],\n", - " \n", - " # Administrative metadata\n", - " \"license_id\": \"cc-by-4.0\",\n", - " \"version\": \"1.0\",\n", - " \"private\": False,\n", - " \n", - " # Extended metadata using extras\n", - " \"extras\": {\n", - " \"workflow_id\": workflow_id,\n", - " \"creation_method\": \"S3-to-Dataset API Workflow\",\n", - " \"storage_backend\": \"S3 Object Storage\",\n", - " \"bucket_name\": workflow_bucket,\n", - " \"data_collection_date\": datetime.now().strftime('%Y-%m-%d'),\n", - " \"quality_control\": \"Automated + Manual Review\",\n", - " \"data_format_standards\": \"CSV, JSON, Markdown\",\n", - " \"access_method\": \"Presigned URLs\",\n", - " \"url_expiration_hours\": str(presigned_request['expires_in']//3600),\n", - " \"geographical_coverage\": \"Research Site A (40.7128ยฐN, 74.0060ยฐW)\",\n", - " \"temporal_coverage\": \"2024-01-01 (sample data)\",\n", - " \"instrument_type\": \"Weather Station Network\",\n", - " \"measurement_frequency\": \"Hourly\",\n", - " \"data_processing_level\": \"Level 2 - Quality Controlled\",\n", - " \"contact_info\": \"NDP EP Tutorial\",\n", - " \"methodology_reference\": \"See included methodology documentation\",\n", - " \"software_version\": \"NDP EP API v1.0\",\n", - " \"backup_location\": f\"S3 bucket: {workflow_bucket}\",\n", - " \"checksum_algorithm\": \"MD5 (via S3 ETag)\"\n", - " },\n", - " \n", - " # S3-stored resources\n", - " \"resources\": dataset_resources\n", - " }\n", - " \n", - " print(f\"๐Ÿ“‹ Dataset will include:\")\n", - " print(f\" ๐Ÿ“Š Name: {dataset_payload['name']}\")\n", - " print(f\" ๐Ÿ“ Title: {dataset_payload['title']}\")\n", - " print(f\" ๐Ÿข Organization: {organization_name}\")\n", - " print(f\" ๐Ÿท๏ธ Tags: {len(dataset_payload['tags'])} tags\")\n", - " print(f\" ๐Ÿ“ Resources: {len(dataset_resources)} S3-stored files\")\n", - " print(f\" ๐Ÿ“ฆ Storage: S3 bucket '{workflow_bucket}'\")\n", - " print(f\" ๐Ÿ”— Access: Presigned URLs (valid {presigned_request['expires_in']//3600} hours)\")\n", - " \n", - " # Create the dataset\n", - " print(f\"\\n๐Ÿš€ Registering dataset...\")\n", - " dataset_response = make_api_request(\"POST\", \"/dataset\", data=dataset_payload)\n", - " print_response(dataset_response, \"Dataset Registration\")\n", - " \n", - " if \"error\" not in dataset_response and \"id\" in dataset_response:\n", - " dataset_id = dataset_response[\"id\"]\n", - " print(f\"\\n๐ŸŽ‰ Dataset registered successfully!\")\n", - " print(f\"๐Ÿ†” Dataset ID: {dataset_id}\")\n", - " print(f\"๐Ÿ“‹ Dataset Name: {dataset_payload['name']}\")\n", - " print(f\"๐Ÿ“ Resources: {len(dataset_resources)} files from S3\")\n", - " print(f\"๐Ÿ“ฆ S3 Bucket: {workflow_bucket}\")\n", - " print(f\"๐Ÿ”— All files accessible via dataset resources\")\n", - " \n", - " workflow_dataset_id = dataset_id\n", - " else:\n", - " print(\"โŒ Failed to register dataset\")\n", - " workflow_dataset_id = None\n", - " \n", - "else:\n", - " print(\"โš ๏ธ Cannot create dataset - missing organization or S3 files\")\n", - " workflow_dataset_id = None" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 7. Verification and Testing\n", - "\n", - "Let's verify that our complete workflow worked correctly by testing access to the data." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if workflow_dataset_id and files_with_urls:\n", - " print(\"๐Ÿงช Verifying workflow completion...\")\n", - " \n", - " # Test 1: Verify we can list objects in our S3 bucket\n", - " print(\"\\n1๏ธโƒฃ Testing S3 bucket access...\")\n", - " bucket_objects = make_api_request(\"GET\", f\"/s3/objects/{workflow_bucket}\")\n", - " \n", - " if \"error\" not in bucket_objects and \"objects\" in bucket_objects:\n", - " objects = bucket_objects[\"objects\"]\n", - " print(f\"โœ… S3 bucket contains {len(objects)} objects\")\n", - " for obj in objects:\n", - " print(f\" ๐Ÿ“„ {obj['key']} ({obj['size']} bytes)\")\n", - " else:\n", - " print(\"โŒ Could not access S3 bucket objects\")\n", - " \n", - " # Test 2: Test one of the presigned URLs\n", - " print(\"\\n2๏ธโƒฃ Testing presigned URL access...\")\n", - " if files_with_urls:\n", - " test_file = files_with_urls[0] # Test first file\n", - " test_url = test_file[\"presigned_url\"]\n", - " \n", - " print(f\"๐Ÿ”— Testing access to: {test_file['filename']}\")\n", - " \n", - " try:\n", - " # Test the presigned URL directly (without authentication)\n", - " response = requests.get(test_url)\n", - " if response.status_code == 200:\n", - " print(f\"โœ… Presigned URL works! Downloaded {len(response.content)} bytes\")\n", - " \n", - " # Show preview of content if it's text\n", - " if test_file[\"format\"] in [\"CSV\", \"Markdown\", \"JSON\"]:\n", - " content_preview = response.text[:200]\n", - " print(f\"๐Ÿ“„ Content preview:\\n{content_preview}...\")\n", - " else:\n", - " print(f\"โŒ Presigned URL failed: {response.status_code}\")\n", - " except Exception as e:\n", - " print(f\"โŒ Error testing presigned URL: {e}\")\n", - " \n", - " # Test 3: Verify dataset metadata\n", - " print(\"\\n3๏ธโƒฃ Workflow verification summary...\")\n", - " print(f\"โœ… Created S3 bucket: {workflow_bucket}\")\n", - " print(f\"โœ… Uploaded {len(uploaded_files)} research files\")\n", - " print(f\"โœ… Generated {len(files_with_urls)} presigned URLs\")\n", - " print(f\"โœ… Registered dataset: {workflow_dataset_id}\")\n", - " print(f\"โœ… Dataset includes {len(dataset_resources)} S3 resources\")\n", - " \n", - " total_size_mb = sum(f.get('size', 0) for f in uploaded_files) / (1024*1024)\n", - " print(f\"๐Ÿ“Š Total data stored: {total_size_mb:.2f} MB\")\n", - " print(f\"๐Ÿ”— All files accessible via presigned URLs for {presigned_request['expires_in']//3600} hours\")\n", - " \n", - "else:\n", - " print(\"โš ๏ธ Workflow incomplete - cannot verify\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 8. Alternative: Create Direct S3 URLs (Optional)\n", - "\n", - "For reference, here's how to create \"permanent\" URLs using direct S3 access (requires bucket to be public)." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "if workflow_bucket and uploaded_files:\n", - " print(\"๐Ÿ“š Alternative approach: Direct S3 URLs\")\n", - " print(\"(Note: This requires the bucket to be configured as public)\")\n", - " \n", - " direct_urls = []\n", - " for file_info in uploaded_files:\n", - " # Direct S3 URL format (works if bucket is public)\n", - " direct_url = f\"http://localhost:9000/{file_info['bucket']}/{file_info['key']}\"\n", - " direct_urls.append({\n", - " \"filename\": file_info[\"filename\"],\n", - " \"direct_url\": direct_url,\n", - " \"description\": file_info[\"description\"]\n", - " })\n", - " \n", - " print(f\"๐Ÿ”— {file_info['filename']}: {direct_url}\")\n", - " \n", - " print(\"\\n๐Ÿ’ก Direct URLs vs Presigned URLs:\")\n", - " print(\" โ€ข Direct URLs: Permanent but require public bucket\")\n", - " print(\" โ€ข Presigned URLs: Temporary but work with private buckets\")\n", - " print(\" โ€ข For research data, presigned URLs are usually preferred for security\")\n", - " \n", - "else:\n", - " print(\"โš ๏ธ No files available for direct URL demonstration\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 9. Workflow Summary and Best Practices\n", - "\n", - "Let's summarize what we accomplished and provide best practices for real-world usage." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "print(\"๐ŸŽ‰ S3-to-Dataset Workflow Summary\")\n", - "print(\"=\" * 50)\n", - "\n", - "if workflow_dataset_id:\n", - " print(\"\\nโœ… Workflow completed successfully!\")\n", - " print(f\"\\n๐Ÿ“Š What we accomplished:\")\n", - " print(f\" ๐Ÿ—‚๏ธ Created sample research data files\")\n", - " print(f\" ๐Ÿชฃ Created S3 bucket: {workflow_bucket}\")\n", - " print(f\" ๐Ÿ“ค Uploaded {len(uploaded_files)} files to S3 storage\")\n", - " print(f\" ๐Ÿ”— Generated {len(files_with_urls)} presigned URLs\")\n", - " print(f\" ๐Ÿข Used/created organization: {organization_name}\")\n", - " print(f\" ๐Ÿ“‹ Registered dataset: {workflow_dataset_id}\")\n", - " print(f\" ๐Ÿ”„ Linked S3 files as dataset resources\")\n", - " \n", - " print(f\"\\n๐ŸŽฏ Key workflow steps:\")\n", - " print(f\" 1. Data Preparation โ†’ Created research files\")\n", - " print(f\" 2. Storage โ†’ Uploaded to S3 bucket\")\n", - " print(f\" 3. Access Control โ†’ Generated presigned URLs\")\n", - " print(f\" 4. Registration โ†’ Created dataset with metadata\")\n", - " print(f\" 5. Integration โ†’ Linked S3 resources to dataset\")\n", - " \n", - "else:\n", - " print(\"\\nโš ๏ธ Workflow partially completed\")\n", - " print(\" Check error messages above for issues\")\n", - "\n", - "print(f\"\\n๐Ÿ’ก Best Practices for Real-World Usage:\")\n", - "print(f\"\\n๐Ÿ”’ Security:\")\n", - "print(f\" โ€ข Use presigned URLs for temporary access to private data\")\n", - "print(f\" โ€ข Set appropriate expiration times (max 7 days)\")\n", - "print(f\" โ€ข Consider implementing URL refresh mechanisms for long-term access\")\n", - "print(f\" โ€ข Use proper authentication tokens in production\")\n", - "\n", - "print(f\"\\n๐Ÿ“Š Data Management:\")\n", - "print(f\" โ€ข Include comprehensive metadata in dataset extras\")\n", - "print(f\" โ€ข Use descriptive filenames and bucket organization\")\n", - "print(f\" โ€ข Document data collection and processing methodology\")\n", - "print(f\" โ€ข Include quality control information\")\n", - "\n", - "print(f\"\\n๐Ÿ”„ Automation:\")\n", - "print(f\" โ€ข Script this workflow for regular data publishing\")\n", - "print(f\" โ€ข Implement error handling and retry logic\")\n", - "print(f\" โ€ข Use templates for consistent dataset structure\")\n", - "print(f\" โ€ข Consider batch processing for multiple files\")\n", - "\n", - "print(f\"\\n๐Ÿ”— Integration:\")\n", - "print(f\" โ€ข Link datasets to organizations for proper governance\")\n", - "print(f\" โ€ข Use tags and groups for discoverability\")\n", - "print(f\" โ€ข Include proper licensing information\")\n", - "print(f\" โ€ข Consider versioning for dataset updates\")\n", - "\n", - "print(f\"\\n๐Ÿ“š Documentation:\")\n", - "print(f\" โ€ข Always include methodology documentation\")\n", - "print(f\" โ€ข Document file formats and data structures\")\n", - "print(f\" โ€ข Provide contact information for data stewards\")\n", - "print(f\" โ€ข Include references to related publications\")\n", - "\n", - "print(f\"\\n๐Ÿš€ Next Steps:\")\n", - "print(f\" โ€ข Adapt this workflow for your specific data types\")\n", - "print(f\" โ€ข Implement automated pipelines for routine data publishing\")\n", - "print(f\" โ€ข Integrate with existing research workflows\")\n", - "print(f\" โ€ข Consider implementing data versioning and updates\")\n", - "print(f\" โ€ข Set up monitoring for URL expiration and renewal\")\n", - "\n", - "if workflow_dataset_id:\n", - " print(f\"\\n๐ŸŽฏ Your workflow results:\")\n", - " print(f\" ๐Ÿ“‹ Dataset ID: {workflow_dataset_id}\")\n", - " print(f\" ๐Ÿ“ฆ S3 Bucket: {workflow_bucket}\")\n", - " print(f\" ๐Ÿข Organization: {organization_name}\")\n", - " print(f\" ๐Ÿ“ Files: {len(uploaded_files)} research files\")\n", - " print(f\" ๐Ÿ”— Access: Presigned URLs valid for {presigned_request['expires_in']//3600} hours\")\n", - "\n", - "print(\"\\n\" + \"=\" * 50)\n", - "print(\"๐ŸŽŠ Workflow tutorial completed!\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 10. Cleanup (Optional)\n", - "\n", - "If you want to clean up the resources created during this tutorial, you can run the cleanup code below." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# OPTIONAL: Cleanup tutorial resources\n", - "# Set to True to enable cleanup\n", - "CLEANUP_ENABLED = False\n", - "\n", - "if CLEANUP_ENABLED and workflow_bucket:\n", - " print(\"๐Ÿงน Starting cleanup of tutorial resources...\")\n", - " \n", - " # Delete objects from S3 bucket\n", - " if uploaded_files:\n", - " print(f\"\\n๐Ÿ—‘๏ธ Deleting objects from bucket: {workflow_bucket}\")\n", - " \n", - " for file_info in uploaded_files:\n", - " object_key = file_info[\"key\"]\n", - " print(f\" Deleting: {object_key}\")\n", - " \n", - " delete_response = make_api_request(\n", - " \"DELETE\",\n", - " f\"/s3/objects/{workflow_bucket}/{object_key}\"\n", - " )\n", - " \n", - " if \"error\" not in delete_response:\n", - " print(f\" โœ… Deleted: {object_key}\")\n", - " else:\n", - " print(f\" โŒ Failed to delete: {object_key}\")\n", - " \n", - " # Delete the S3 bucket (must be empty)\n", - " print(f\"\\n๐Ÿ—‘๏ธ Deleting bucket: {workflow_bucket}\")\n", - " bucket_delete_response = make_api_request(\"DELETE\", f\"/s3/buckets/{workflow_bucket}\")\n", - " \n", - " if \"error\" not in bucket_delete_response:\n", - " print(f\"โœ… Deleted bucket: {workflow_bucket}\")\n", - " else:\n", - " print(f\"โŒ Failed to delete bucket: {workflow_bucket}\")\n", - " \n", - " print(\"\\n๐Ÿงน Cleanup completed!\")\n", - " print(\"โ„น๏ธ Note: Datasets and organizations are preserved for reference\")\n", - " \n", - "else:\n", - " if workflow_bucket:\n", - " print(\"โ„น๏ธ Cleanup disabled. Tutorial resources preserved.\")\n", - " print(\"๐Ÿ’ก To clean up, set CLEANUP_ENABLED = True and run this cell again.\")\n", - " print(f\"๐Ÿ“ฆ Created bucket: {workflow_bucket}\")\n", - " print(f\"๐Ÿ“‹ Created dataset: {workflow_dataset_id or 'Not created'}\")\n", - " print(f\"๐Ÿข Organization: {organization_name or 'Not created'}\")\n", - " else:\n", - " print(\"โ„น๏ธ No resources to clean up.\")" - ] - } - ], - "metadata": { - "kernelspec": { - "display_name": "Python 3", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.8.5" - } - }, - "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file